commit 25b93c9f8a5de7388b1511bb141c781c89e97e89 from: ale date: Mon Aug 17 05:14:58 2026 UTC Reorganize the wiki: split investigation-log into topic pages, add real-data figures, force dark theme - docs/src/index.md restructured: status summary table, admonition boxes for Known limitations, real-data validation grouped into its own section with ZTF/variable-star/IASC subsections. - Investigation Log split from one large page into four: the original ZTF field 451 page, plus new variable-star-validation.md, iasc-campaign-validation.md, and design-refinements.md — all linked from the sidebar. Every existing anchor link across src/*.jl docstrings and index.md updated to the new locations. - docs/make_figures.jl (CairoMakie) generates two real figures from this session's actual measured numbers — the systematic_error_fraction sweep (false-positive rate vs. real-variable sensitivity) and the IASC match_radius retuning (tracklet counts per field, before/after) — run automatically before makedocs on every build, local or CI, so they can't drift from the numbers in the text. - docs/src/.vitepress/config.mts: custom config (appearance: 'force-dark') so the wiki is dark-only, no toggle — verified against VitePress's own source that this actually suppresses the switch component, not just guessed from docs. - docs/Project.toml gains CairoMakie so CI's Pkg.instantiate() picks it up automatically; no workflow YAML changes needed. commit - 92ed176ba250755bdfec88536921f1d0f69f0113 commit + 25b93c9f8a5de7388b1511bb141c781c89e97e89 blob - 69478b44b41e88fb15a12ccab53f712deac224e7 blob + 79ee32b83c44adccc6c4718593d9a249be84ba60 --- .gitignore +++ .gitignore @@ -10,9 +10,8 @@ result-* .direnv/ # docs/ — build output and DocumenterVitepress's own Node/Vitepress -# tooling. docs/src/index.md and docs/src/investigation-log.md are real, -# permanent, committed pages, not generated — deliberately not listed -# here. +# tooling. docs/src/*.md are real, permanent, committed pages, not +# generated — deliberately not listed here. docs/build/ docs/node_modules/ docs/package-lock.json @@ -20,3 +19,10 @@ docs/.vitepress/cache/ docs/.vitepress/dist/ docs/src/.vitepress/cache/ docs/src/.vitepress/dist/ + +# docs/src/assets/ — generated by docs/make_figures.jl on every build +# (local and CI), so these would just be stale duplicates if committed; +# not a blanket ignore of the whole assets/ directory in case a real, +# hand-authored logo.png/favicon.ico is ever added there. +docs/src/assets/systematic-error-floor-sweep.png +docs/src/assets/iasc-match-radius-retuning.png blob - 786ff5301199a081a8d89784fe18fb61cbce2f1d blob + a29f9fc65acda46134a5d9d1037853730827767d --- docs/Project.toml +++ docs/Project.toml @@ -1,8 +1,10 @@ [deps] AsteroidPipeline = "1001636e-a0d1-496e-b23d-cef10bc3cdca" +CairoMakie = "13f3f980-e62b-5c42-98c6-ff1f3baf88f0" Documenter = "e30172f5-a6a5-5a46-863b-614d45cd2de4" DocumenterVitepress = "4710194d-e776-4893-9690-8d956a29c365" [compat] +CairoMakie = "0.15.13" Documenter = "1.17.0" DocumenterVitepress = "0.3.5" blob - c4e7d08183f3d346911c64e4e8c8a49db7a2e0bf blob + 6227dc17ae6320be5f0bf3a3ce440aeb7d26cbd6 --- docs/make.jl +++ docs/make.jl @@ -2,12 +2,22 @@ using AsteroidPipeline using Documenter using DocumenterVitepress -# index.md and investigation-log.md are both real, permanent, committed -# pages under docs/src — the docs site's own content, not generated or -# copied in from README.md/INVESTIGATION_LOG.md (which no longer exist at -# the repo root; the short root README.md just links here instead). +# index.md, investigation-log.md, variable-star-validation.md, +# iasc-campaign-validation.md, and design-refinements.md are all real, +# permanent, committed pages under docs/src — the docs site's own +# content, not generated or copied in from README.md/INVESTIGATION_LOG.md +# (which no longer exist at the repo root; the short root README.md just +# links here instead). DocMeta.setdocmeta!(AsteroidPipeline, :DocTestSetup, :(using AsteroidPipeline); recursive=true) +# Regenerates the figures investigation-log.md embeds, from the real +# numbers measured during this project's real-data investigations — run +# before makedocs so both local builds and CI (julia-actions/julia-docdeploy +# instantiates docs/Project.toml, which includes CairoMakie) produce them +# fresh rather than relying on committed images that could drift from the +# numbers in the text. +include("make_figures.jl") + makedocs(; modules=[AsteroidPipeline], authors="Alejandro", @@ -24,7 +34,12 @@ makedocs(; ), pages=[ "Home" => "index.md", - "Investigation Log" => "investigation-log.md", + "Investigation Log" => [ + "ZTF field 451 (the original investigation)" => "investigation-log.md", + "Variable-star validation" => "variable-star-validation.md", + "IASC / Pan-STARRS1 campaign validation" => "iasc-campaign-validation.md", + "Design refinements" => "design-refinements.md", + ], "API Reference" => [ "Detection & Linking" => "api/detection.md", "Variable Stars" => "api/variables.md", blob - /dev/null blob + c8f964a9243160cc267e901319aa38b2b1957bd1 (mode 644) --- /dev/null +++ docs/src/.vitepress/config.mts @@ -0,0 +1,111 @@ +import { defineConfig } from 'vitepress' +import { tabsMarkdownPlugin } from 'vitepress-plugin-tabs' +import { mathjaxPlugin } from './mathjax-plugin' +import { juliaReplTransformer } from './julia-repl-transformer' +import footnote from "markdown-it-footnote"; +import path from 'path' + +const mathjax = mathjaxPlugin() + +function getBaseRepository(base: string): string { + if (!base || base === '/') return '/'; + const parts = base.split('/').filter(Boolean); + return parts.length > 0 ? `/${parts[0]}/` : '/'; +} + +const baseTemp = { + base: 'REPLACE_ME_DOCUMENTER_VITEPRESS',// TODO: replace this in makedocs! +} + +const navTemp = { + nav: 'REPLACE_ME_DOCUMENTER_VITEPRESS', +} + +const nav = [ + ...navTemp.nav, + { + component: 'VersionPicker' + } +] + +// https://vitepress.dev/reference/site-config +export default defineConfig({ + base: 'REPLACE_ME_DOCUMENTER_VITEPRESS',// TODO: replace this in makedocs! + title: 'REPLACE_ME_DOCUMENTER_VITEPRESS', + description: 'REPLACE_ME_DOCUMENTER_VITEPRESS', + lastUpdated: true, + cleanUrls: true, + // Dark theme only, by request — no light mode, no toggle to switch to + // one. 'force-dark' (vs. just 'dark') is what actually removes the + // appearance toggle from the nav bar, not merely defaulting to it. + appearance: 'force-dark', + outDir: 'REPLACE_ME_DOCUMENTER_VITEPRESS', // This is required for MarkdownVitepress to work correctly... + head: [ + ['link', { rel: 'icon', href: 'REPLACE_ME_DOCUMENTER_VITEPRESS_FAVICON' }], + ['script', {src: `${getBaseRepository(baseTemp.base)}versions.js`}], + // ['script', {src: '/versions.js'], for custom domains, I guess if deploy_url is available. + ['script', {src: `${baseTemp.base}siteinfo.js`}], + // REPLACE_ME_DOCUMENTER_VITEPRESS_NOINDEX + ], + + markdown: { + codeTransformers: [juliaReplTransformer()], + config(md) { + md.use(tabsMarkdownPlugin); + md.use(footnote); + mathjax.markdownConfig(md); + }, + theme: { + light: "github-light", + dark: "github-dark" + }, + }, + vite: { + plugins: [ + mathjax.vitePlugin, + ], + define: { + __DEPLOY_ABSPATH__: JSON.stringify('REPLACE_ME_DOCUMENTER_VITEPRESS_DEPLOY_ABSPATH'), + }, + resolve: { + alias: { + '@': path.resolve(__dirname, '../components') + } + }, + optimizeDeps: { + exclude: [ + '@nolebase/vitepress-plugin-enhanced-readabilities/client', + 'vitepress', + '@nolebase/ui', + ], + }, + ssr: { + noExternal: [ + // If there are other packages that need to be processed by Vite, you can add them here. + '@nolebase/vitepress-plugin-enhanced-readabilities', + '@nolebase/ui', + ], + }, + }, + themeConfig: { + outline: 'deep', + logo: 'REPLACE_ME_DOCUMENTER_VITEPRESS', + search: { + provider: 'local', + options: { + detailedView: true + } + }, + nav, + sidebar: 'REPLACE_ME_DOCUMENTER_VITEPRESS', + sidebarDrawer: 'REPLACE_ME_DOCUMENTER_VITEPRESS_SIDEBAR_DRAWER', + editLink: 'REPLACE_ME_DOCUMENTER_VITEPRESS', + socialLinks: [ + { icon: 'github', link: 'REPLACE_ME_DOCUMENTER_VITEPRESS' } + ], + footer: { + message: 'Made with DocumenterVitepress.jl
', + copyright: `© Copyright ${new Date().getUTCFullYear()}.` + } + } +}) blob - dd66e303cdccc4f3354ebc934d1667928a28cfc4 blob + 63752c22722c6997860cb04bebed3bbfb8ec55f4 --- docs/src/index.md +++ docs/src/index.md @@ -46,120 +46,33 @@ periodogram applies directly to a `find_variable_sourc ## Status -Early development. `detect_sources`, `link_candidates`, the WCS -calibration step, and `crossmatch_catalog` are implemented and wired -together end to end in `run_pipeline`, validated against synthetic FITS -frames with a known injected source track, and exercised against real -public survey data (see `examples/real_data_demo.jl`) and real IASC -practice campaign data (see `examples/iasc_demo.jl` and the "Using real -IASC campaign data" section below — 9 real, independently-catalogued -objects recovered across 5 real Pan-STARRS1 fields). +!!! note "Early development, but validated end to end against real data" + Every stage below has run against real telescope data at least once — + not just synthetic tests — including three independent surveys (ZTF, + Pan-STARRS1) and a real, independently-confirmed variable star. See + [Real-data validation](@ref) for the full story of each row. -ZOGY difference imaging (Zackay, Ofek & Gal-Yam 2016) is implemented — -`build_reference` stacks a deep reference from many epochs via -reprojection (`Reproject.jl`) onto a common pixel grid, `estimate_psf` -measures each frame's empirical PSF from its own bright stars, and -`zogy_subtract` produces a statistically normalized detection-significance -map (`S_corr`, unit variance by construction) in Fourier space via -`FFTW.jl`. Validated against three falsifiable synthetic checks (identical -images subtract to zero; pure noise gives `std(S_corr) ≈ 1`; an injected -source's peak significance matches the analytic matched-filter prediction) -and against real ZTF data — see `examples/real_data_demo.jl`, which runs -the pipeline with and without differencing on the same frames and reports -which known objects each recovers, rather than assuming differencing -helps. +| Component | Validated against | +|:--|:--| +| `detect_sources`, `link_candidates`, WCS calibration, `crossmatch_catalog` | Synthetic (exact recovery) · real ZTF field 451 · 5 real IASC/Pan-STARRS1 fields (**9 known objects** recovered) | +| ZOGY difference imaging (`build_reference`, `estimate_psf`, `zogy_subtract`) | 3 falsifiable synthetic checks · real ZTF field 451 | +| `find_variable_sources` / `search_field` | Synthetic · false-positive rate calibrated on real ZTF stars · a real confirmed variable (ASASSN-V J183620.31) | +| `fit_moffat_psf` (PSF analytic fallback) | Synthetic · PSF width calibrated on real ZTF data | +| `plate_solve` | Live nova.astrometry.net service | +| `crossmatch_catalog` (`:skybot`/`:vsx`/`:simbad`) | Live services, real positive controls | -`find_variable_sources`/`search_field` (stationary, flux-varying source -detection) and `fit_moffat_psf` (`estimate_psf`'s analytic-PSF fallback) -are implemented, tested against synthetic data, and — for -`find_variable_sources`'s photometric normalization, S/N floor, and -`chi2_threshold` default, and for `fit_moffat_psf`'s recovered PSF -width — calibrated directly against real ZTF data (see the -[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#detect_sources's-flux-has-been-silently-wrong-since-the-beginning:-a-transposed-aperture), including a real bug in -`detect_sources`'s own flux measurement this calibration work found and -fixed). Now also validated end to end, via `search_field`, against a real -field with an independently-confirmed variable star as ground truth: ZTF -field 487/CCD 12/quadrant 1/zr, night 2019-06-10, containing ASASSN-V -J183620.31 (VSX: type EW, period 0.322 d) — see -`examples/variable_star_demo.jl` and the -[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#Validating-search_field-against-a-real,-independently-confirmed-variable-star) -for the full run. The target was recovered (crossmatched against VSX at a -1.7" offset) among 22 variable candidates from 638 detections; a -Lomb-Scargle period fit to that single partial night (only ~32% of one -0.322 d period) found a real, highly significant periodic signal -(FAP ≈ 0) but at 0.5 d, not the true period — an expected outcome of the -partial phase coverage, declared before running rather than adjusted -after seeing it, and not treated as a failure of `find_variable_sources` -itself (which is what the crossmatch recovery actually validates). +Not yet run on a live IASC search campaign (as opposed to practice data) +— see [Using real IASC campaign data](@ref) for what that would involve. -On that real dataset (field 451, 2019-10-23), the undifferenced baseline -finds 133 tracklets and recovers both known objects in the field (2002 -UY45, 1997 KO3); ZOGY also recovers both, at consistent sky offsets -(confirming the subtraction is correctly calibrated), but finds 667 -tracklets total and no *additional* known object — both known objects -here are bright enough that the baseline already recovers them trivially, -so this dataset doesn't exercise ZOGY's actual advantage (recovering -objects below a single frame's noise floor). The raw tracklet-count gap -is not a clean read on ZOGY's noise properties; see the -[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-quality-gate's-combinatorial-side-effect-on-tracklet-count) for why, and for the full -record of every real bug this project's real-data testing surfaced so -far — most recently three more from the real IASC campaign work below — -all fixed with regression tests, and how each was diagnosed. +## Real-data validation -## Known limitations +### ZTF: `examples/real_data_demo.jl` -- **Empirical PSF's quality still depends on the field, though less than - it used to.** `estimate_psf` stacks real star cutouts, which captures - the true PSF shape (wings included) without fitting a model family per - instrument, but needs enough bright, isolated, unsaturated stars to do - it. It now retries at a progressively relaxed `min_separation` (halved, - up to `relaxation_attempts` times, default 2) before giving up — a - field with a few usable stars at a tighter isolation radius still gives - the real PSF shape, which the analytic fallback never can. Only once - even the most relaxed attempt finds nothing does it fall back (by - default) to `fit_moffat_psf` — a parametric Moffat fit — rather than - failing outright; the fallback trades exact PSF shape for robustness, - and is not itself a substitute for a genuinely well-behaved field. -- **`zogy_subtract`'s astrometric-noise term (`V_ast`) is still opt-in at - the `zogy_subtract` level** — it needs `n_sources`/`r_sources` passed - explicitly, and is `0` without them; this is deliberate API layering - (`run_pipeline` always supplies them, so a direct caller who doesn't - need the extra `detect_sources` cost can skip it), not something - planned to change. What did change: leaving both unset now emits a - `@warn` (once per session), so omitting `V_ast` is a visible choice - instead of a silent default a direct caller could miss. -- **`find_variable_sources` still has a real, measured false-positive - floor on real single-epoch aperture photometry, even after fixing it - once.** The first hypothesis tried — pixel-grid jitter in - `detect_sources`'s peak-pixel aperture centering — turned out to be a - real but secondary effect: refining the aperture to a sub-pixel - centroid (see `detect_sources`'s docstring) barely moved the - false-positive rate (23% → 22% at `chi2_threshold=3` on real ZTF field - 451). The actual dominant cause, found by checking *which* stars were - flagged, was a systematic photometric error floor — bright stars' - tiny formal errors made ordinary flat-fielding/PSF-variation - systematics look like huge chi2 significance. Adding that floor - (`variability_chi2`'s `systematic_error_fraction`) cut the rate: a 1% - floor took it to 6% at threshold 3, ~0% at threshold 20; sweeping the - floor further (against the same real stars, and checked against a real - confirmed variable — ASASSN-V J183620.31 — to make sure real - sensitivity wasn't sacrificed for it) found more room, without giving - up real detections: the current default, 2%, cuts the - `chi2_threshold=10.0` rate to 2.0% on this dataset (vs 1%'s 3.3%), - while that confirmed variable still clears the threshold with a 3x - margin — see `find_variable_sources`'s - docstring and the [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-centroid-fix-barely-moved-the-false-positive-floor-—-the-real-cause-was-a-systematic-error-floor) for the - full before/after numbers. Still not zero: treat a candidate as needing - independent confirmation (a catalog match or a recovered period), not - as self-evidently real. +Runs the pipeline against real ZTF (Zwicky Transient Facility) frames +both with and without ZOGY differencing, and cross-matches both against +SkyBoT — a controlled comparison, not just a demonstration. Fetch the +data first (public, no authentication required): -## Example: real data - -`examples/real_data_demo.jl` runs the pipeline against real ZTF (Zwicky -Transient Facility) frames both with and without ZOGY differencing, and -cross-matches both against SkyBoT — a controlled comparison, not just a -demonstration. Fetch the data first (public, no authentication required): - ``` examples/fetch_data.sh julia --project=. examples/real_data_demo.jl @@ -168,49 +81,49 @@ julia --project=. examples/real_data_demo.jl Building the reference stack (30 frames, each individually reprojected) is the slow part — tens of minutes on a laptop, one-time per run. -## Rotation period recovery +On field 451 (2019-10-23), the undifferenced baseline finds 133 +tracklets and recovers both known objects in the field (2002 UY45, 1997 +KO3); ZOGY also recovers both, at consistent sky offsets (confirming the +subtraction is correctly calibrated), but finds 667 tracklets total and +no *additional* known object — both known objects here are bright enough +that the baseline already recovers them trivially, so this dataset +doesn't exercise ZOGY's actual advantage (recovering objects below a +single frame's noise floor). The raw tracklet-count gap is not a clean +read on ZOGY's noise properties; see the +[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-quality-gate's-combinatorial-side-effect-on-tracklet-count) +for why, and for the full record of every real bug this project's +real-data testing has surfaced, all fixed with regression tests. -For a confirmed discovery, given a dedicated photometric follow-up -sequence (many exposures over hours, at a fixed sky position — the -target should barely move between them, unlike the original discovery -epochs): +`find_variable_sources`/`search_field` and `fit_moffat_psf` were also +calibrated against this same field — photometric normalization, S/N +floor, `chi2_threshold` default, and recovered PSF width — see the +[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#detect_sources's-flux-has-been-silently-wrong-since-the-beginning:-a-transposed-aperture), +including a real bug in `detect_sources`'s own flux measurement this +calibration work found and fixed. -```julia -using AsteroidPipeline +### Variable stars: `examples/variable_star_demo.jl` -times, flux, flux_err = light_curve(fits_paths, ra, dec) -result = recover_rotation_period(times, flux; minimum_period=0.02, maximum_period=1.0) -result.period, result.false_alarm_probability -``` +Validates `search_field` end to end against a real, +independently-confirmed variable star — ZTF field 487/CCD 12/quadrant +1/zr, night 2019-06-10, containing ASASSN-V J183620.31 (VSX: type EW, +period 0.322 d). See the +[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/variable-star-validation) +for the full run. -`minimum_period`/`maximum_period` bound the search (same units as -`times`, i.e. days) and should bracket the rotation periods physically -plausible for the object's size class. A small `false_alarm_probability` -is what distinguishes a real periodic signal from a noise fluctuation — -see the function's docstring. +The target was recovered (crossmatched against VSX at a 1.7" offset) +among 22 variable candidates from 638 detections. A Lomb-Scargle period +fit to that single partial night (only ~32% of one 0.322 d period) found +a real, highly significant periodic signal (FAP ≈ 0) but at 0.5 d, not +the true period — an expected outcome of the partial phase coverage, +declared before running rather than adjusted after seeing it, and not a +failure of `find_variable_sources` itself (which is what the crossmatch +recovery above actually validates). -## Plate-solving +### IASC / Pan-STARRS1: `examples/iasc_demo.jl` -For a frame with no WCS already in its header, and a -[nova.astrometry.net](https://nova.astrometry.net/) API key (free -registration): - -```julia -using AsteroidPipeline - -run_pipeline(fits_paths; reference=reference, plate_solve_api_key=key) -``` - -or directly: `plate_solve(fits_path; api_key=key)`. This is a live -network round trip — upload, then poll until the frame solves — so it is -slow and requires connectivity. Validated against the real service: see -the [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#plate_solve-validated-end-to-end-against-the-live-service). - -## Using real IASC campaign data - -Now attempted, against 5 real Pan-STARRS1 (PS1) IASC practice sets -("Practice Image Sets", 2019-08-28/09-04/09-24), each 4 exposures of the -same field over ~40-70 min — see `examples/iasc_demo.jl` (point it at +Validates `run_pipeline` against 5 real Pan-STARRS1 (PS1) IASC practice +sets ("Practice Image Sets", 2019-08-28/09-04/09-24), each 4 exposures of +the same field over ~40-70 min — see `examples/iasc_demo.jl` (point it at your own local practice/campaign FITS; IASC material isn't public, so unlike the ZTF demo above there is no fetch script). `run_pipeline` recovered **9 real, independently-catalogued objects** across the 5 @@ -219,7 +132,7 @@ Trojan (2019 NB9) — the first end-to-end validation against real IASC-style data, not just ZTF. Getting there surfaced four real, fixed issues — see the -[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#Validating-against-real-IASC-Pan-STARRS1-campaign-data) +[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/iasc-campaign-validation) for the full story of each: - `load_wcs` raised "Linear transformation matrix is singular" on every @@ -233,25 +146,29 @@ for the full story of each: preprocessing step in `examples/iasc_demo.jl` (not in `src/`, since this is a real-FITS-ingestion concern, not `detect_sources`'s job). - `crossmatch_catalog(...; :skybot)` queried one candidate at a time, - fully sequentially — real candidate lists here (hundreds to - thousands of tracklets) took minutes to hours, and a multi-hour run - eventually died to a transient connection error with no retry. Fixed - in `_crossmatch_skybot` itself: concurrent requests (real, measured - ~8x wall-clock speedup) and a retry on transient HTTP errors. + fully sequentially — real candidate lists here (hundreds to thousands + of tracklets) took minutes to hours, and a multi-hour run eventually + died to a transient connection error with no retry. Fixed in + `_crossmatch_skybot` itself: concurrent requests (real, measured ~8x + wall-clock speedup) and a retry on transient HTTP errors. - `match_radius`, first converted from `real_data_demo.jl`'s ZTF value to keep the same ~10" angular tolerance, turned out far looser than PS1's - own real astrometric precision (`PERROR`, in these headers: 0.20-0.23") - — on the densest field this produced 10,422 tracklets, almost all - spurious duplicates of the same real objects (distinct real stars - within 10" of each other, or the same object matched by several + own real astrometric precision (`PERROR`, in these headers: + 0.20-0.23") — on the densest field this produced 10,422 tracklets, + almost all spurious duplicates of the same real objects (distinct real + stars within 10" of each other, or the same object matched by several near-identical trial velocities). Retuned to 2" (~10x `PERROR`, measured from the headers, not guessed) and confirmed directly: the same 9 distinct known objects are still recovered in every field, while total tracklets across all 5 fields drop from 16,158 to 4,960 (-69%) — this was cleanup of spurious duplicates, not lost detections. -General guidance for pointing this pipeline at other real campaign data: +#### Using real IASC campaign data +Not yet attempted on a live campaign — practice data above is the +closest real-world test so far. General guidance for pointing this +pipeline at a real campaign, or any other survey's data: + - Point `run_pipeline` (or `examples/iasc_demo.jl`'s pattern) at the local file paths directly; no fetch script is needed for files you already have. @@ -270,6 +187,95 @@ General guidance for pointing this pipeline at other r the field (via `crossmatch_catalog(...; :skybot)`) are the same kind of ground truth used there. +## Known limitations + +!!! warning "Empirical PSF's quality still depends on the field" + `estimate_psf` stacks real star cutouts, which captures the true PSF + shape (wings included) without fitting a model family per + instrument, but needs enough bright, isolated, unsaturated stars to + do it. It now retries at a progressively relaxed `min_separation` + (halved, up to `relaxation_attempts` times, default 2) before giving + up — a field with a few usable stars at a tighter isolation radius + still gives the real PSF shape, which the analytic fallback never + can. Only once even the most relaxed attempt finds nothing does it + fall back (by default) to `fit_moffat_psf` — a parametric Moffat fit + — rather than failing outright; the fallback trades exact PSF shape + for robustness, and is not itself a substitute for a genuinely + well-behaved field. + +!!! note "`zogy_subtract`'s astrometric-noise term (`V_ast`) is opt-in by design" + It needs `n_sources`/`r_sources` passed explicitly, and is `0` + without them; this is deliberate API layering (`run_pipeline` always + supplies them, so a direct caller who doesn't need the extra + `detect_sources` cost can skip it), not something planned to change. + What did change: leaving both unset now emits a `@warn` (once per + session), so omitting `V_ast` is a visible choice instead of a + silent default a direct caller could miss. + +!!! warning "`find_variable_sources` has a real, measured false-positive floor" + The first hypothesis tried — pixel-grid jitter in `detect_sources`'s + peak-pixel aperture centering — turned out to be a real but + secondary effect: refining the aperture to a sub-pixel centroid (see + `detect_sources`'s docstring) barely moved the false-positive rate + (23% → 22% at `chi2_threshold=3` on real ZTF field 451). The actual + dominant cause, found by checking *which* stars were flagged, was a + systematic photometric error floor — bright stars' tiny formal + errors made ordinary flat-fielding/PSF-variation systematics look + like huge chi2 significance. Adding that floor + (`variability_chi2`'s `systematic_error_fraction`) cut the rate: a + 1% floor took it to 6% at threshold 3, ~0% at threshold 20; sweeping + the floor further (against the same real stars, and checked against + a real confirmed variable — ASASSN-V J183620.31 — to make sure real + sensitivity wasn't sacrificed for it) found more room without giving + up real detections: the current default, 2%, cuts the + `chi2_threshold=10.0` rate to 2.0% on this dataset (vs 1%'s 3.3%), + while that confirmed variable still clears the threshold with a 3x + margin — see `find_variable_sources`'s docstring and the + [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-centroid-fix-barely-moved-the-false-positive-floor-—-the-real-cause-was-a-systematic-error-floor) + for the full before/after numbers. Still not zero: treat a candidate + as needing independent confirmation (a catalog match or a recovered + period), not as self-evidently real. + +## Follow-up workflows + +### Rotation period recovery + +For a confirmed discovery, given a dedicated photometric follow-up +sequence (many exposures over hours, at a fixed sky position — the +target should barely move between them, unlike the original discovery +epochs): + +```julia +using AsteroidPipeline + +times, flux, flux_err = light_curve(fits_paths, ra, dec) +result = recover_rotation_period(times, flux; minimum_period=0.02, maximum_period=1.0) +result.period, result.false_alarm_probability +``` + +`minimum_period`/`maximum_period` bound the search (same units as +`times`, i.e. days) and should bracket the rotation periods physically +plausible for the object's size class. A small `false_alarm_probability` +is what distinguishes a real periodic signal from a noise fluctuation — +see the function's docstring. + +### Plate-solving + +For a frame with no WCS already in its header, and a +[nova.astrometry.net](https://nova.astrometry.net/) API key (free +registration): + +```julia +using AsteroidPipeline + +run_pipeline(fits_paths; reference=reference, plate_solve_api_key=key) +``` + +or directly: `plate_solve(fits_path; api_key=key)`. This is a live +network round trip — upload, then poll until the frame solves — so it is +slow and requires connectivity. Validated against the real service: see +the [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#plate_solve-validated-end-to-end-against-the-live-service). + ## Dependencies - [FITSIO.jl](https://github.com/JuliaAstro/FITSIO.jl) — FITS I/O blob - dc62fc44f1207d65635045e46ac2e212d7be5467 blob + e6f5a8a6c1241d9ba058bc1eb1378404ce5ba18d --- docs/src/investigation-log.md +++ docs/src/investigation-log.md @@ -1,10 +1,19 @@ # Investigation log Chronological record of what real-data testing found and how each finding -was diagnosed — kept separate from `README.md` so the README stays a -focused reference rather than a narrative. Cross-referenced from -`README.md`'s Known limitations section and from the relevant docstrings. +was diagnosed — kept separate from the main wiki pages so those stay +focused references rather than narratives. Cross-referenced from +`docs/src/index.md`'s Known Limitations section and from the relevant +docstrings. +This page covers the original ZTF field 451 investigation, chronologically +first. Later validation work against other real datasets got large enough +to split onto their own pages: + +- [Validating `search_field` against a real, independently-confirmed variable star](variable-star-validation.md) +- [Validating against real IASC (Pan-STARRS1) campaign data](iasc-campaign-validation.md) +- [Revisiting the two intentional-design "limitations"](design-refinements.md) + ## Real ZTF data surfaced four bugs no synthetic test or code review caught All fixed, all with regression tests added afterward: @@ -301,167 +310,27 @@ returned row matches. Batched 50 candidates per reques not measured at higher N on these free, shared, anonymous-use services. Cuts N requests to `ceil(N / 50)`. -## Validating `search_field` against a real, independently-confirmed variable star - -Every real-data check so far had confirmed known *moving* objects (via -SkyBoT) but never a known *variable* — the one real dataset checked had -only one catalogued VSX variable in its footprint, too faint to serve as -a useful positive control. Fixed by choosing a real target with an actual -VSX catalog entry: ASASSN-V J183620.31 (type EW, a contact eclipsing -binary, period 0.322427 d), in ZTF field 487/CCD 12/quadrant 1/zr, night -2019-06-10 — a real high-cadence campaign with 144 exposures over 2.45 h, -thinned to 29 for `examples/variable_star_demo.jl`. Declared in advance: -2.45 h covers only ~32% of one period, so full period recovery from this -single night was not expected — the actual test was whether -`find_variable_sources` flags the star as variable at all. - -First attempt did not run: `search_field`'s default `threshold=5.0` (used -at `8.0` in `real_data_demo.jl`) produced ~12,900 detections per frame -here, against field 451's ~130 — this field sits near the galactic plane. -`link_candidates`'s tracklet search is pairwise in detections/frame, and -did not finish in 35 minutes at that density; killed and confirmed via a -standalone check (`detect_sources` alone, one frame) that the counts were -real, not a hang. The target star is extremely bright at this field's -noise level (flux ≈ 87,000, ~1.7 px from its WCS-predicted position at -every threshold tested from 10 to 100), so raising the demo's threshold -to 60 — cutting detections/frame to ~1,260 — loses none of the signal -this run actually needs; a deliberate, field-specific tradeoff, not the -pipeline's own default. A second bug surfaced once linking finished: -`light_curve`'s default `timestamp_key="MJD-OBS"` doesn't exist in this -survey's headers (`search_field` was already correctly called with -`"OBSMJD"`) — a copy-paste omission in the demo script, not a pipeline -bug, fixed by passing the same key. - -With both fixed, the run found 22 variable candidates from 638 -detections; 2 matched a known VSX variable, including the target itself — -**ASASSN-V J183620.31, recovered at a 1.7" offset from its catalogued -position** — the positive-control criterion this whole exercise was -built to test, and it passed. A Lomb-Scargle fit to the recovered -candidate's own forced-photometry light curve (`light_curve` + -`recover_rotation_period`) found a real, highly significant periodic -signal (false-alarm probability ≈ 0) at 0.5 d, not the catalogued -0.322 d — the declared-in-advance outcome of fitting a period search to -a light curve covering less than a third of that period (almost -certainly an alias, not evidence against the true period), and not a -failure of `find_variable_sources` itself, which is what the crossmatch -recovery above actually validates. - -## Validating against real IASC (Pan-STARRS1) campaign data - -Every real-data check so far used ZTF. `docs/src/index.md`'s "Using real -IASC campaign data" section had stood as "not attempted" all session — -closed by running `examples/iasc_demo.jl` against 5 real Pan-STARRS1 -(PS1) IASC practice sets (2019-08-28/09-04/09-24, 4 exposures each). -`run_pipeline` recovered 26 real, independently-catalogued objects -across the 5 fields via SkyBoT — including a Jupiter Trojan, 2019 NB9 — -but getting a clean run took three real, fixed bugs, found in this order. - -**`load_wcs` failed on every one of these real headers.** Every PS1 -header raised `"Linear transformation matrix is singular"` from wcslib. -Bisected a real header down to the exact cause (splitting it into halves, -testing each half in isolation, recursing into whichever half still -failed): `CNPIX1`/`CNPIX2` alone — a legacy IRAF/DSS plate-astrometry -keyword pair, present in these headers but with none of that convention's -other required keywords — was enough to reproduce it, even combined with -nothing but `SIMPLE`/`BITPIX`/`NAXIS`. wcslib reads that keyword pair as -the start of a *separate*, implicit DSS-style WCS description, and with -the rest of that convention absent builds an all-zero, degenerate linear -transform for it — a real wcslib parsing quirk, not anything wrong with -the header's own real, complete CTYPE/CRVAL/CRPIX/CDELT WCS, which parses -cleanly on its own. Fixed in `load_wcs`: on exactly this error, retry -after stripping just those two keyword's FITS cards and nothing else — -confirmed sufficient, not guessed. Regression test constructs a -synthetic header (a real WCS plus injected `CNPIX1`/`CNPIX2` cards) since -the real PS1 files can't be committed to the repo. - -**`detect_sources` produced enormous numbers of spurious detections on -some frames.** One real frame: 176,165 "detections" at `threshold=8.0` -(field 451 on ZTF, by comparison, has ~130). Root cause: these FITS files -mark invalid/masked pixels using the standard `BLANK` header keyword -(scaled through `BZERO`/`BSCALE` like any other pixel value) rather than -`NaN`, and `FITSIO.jl` does not convert `BLANK` sentinels automatically. -One real frame had 158,443 pixels (2.7% of the image) pegged at exactly -that sentinel value (65535, from `BLANK=32767` + `BZERO=32768`) — -`detect_sources` read that as enormous real flux across a large masked -region and found a spurious "source" seemingly everywhere. Confirmed -directly: replacing those exact pixels with the frame's own valid-region -median (before any detection) dropped the same frame from 176,165 to 176 -detections. Handled as a preprocessing step in `examples/iasc_demo.jl` -(`clean_blank_pixels`, writing cleaned copies preserving the original -header/WCS/timestamp exactly) rather than in `src/`, since BLANK-sentinel -handling is a real-FITS-ingestion concern specific to how a given survey -exports data, not something `detect_sources` itself should need to know -about. - -**`crossmatch_catalog(...; :skybot)` was too slow, and not resilient, at -real scale.** Unlike `:vsx`/`:simbad` (batched via CDS TAP — see above), -SkyBoT has no batch mode, so `_crossmatch_skybot` queried one candidate -per request, fully sequentially. Fine for a handful of candidates; not -for hundreds to thousands of real tracklets. A full 5-field run of -`examples/iasc_demo.jl` took over two hours and then died outright, deep -into the fourth field's crossmatch (2,619 candidates), to -`"tls write failed: connection is closed"` — an uncaught, unretried -network error with no partial-progress recovery. Fixed two ways in -`_crossmatch_skybot`: concurrent requests (Julia `Task`s + a bounded -`Base.Semaphore`, not extra threads — this is a network-latency-bound -workload, and cooperative concurrency on however many threads Julia -already has is enough) and a single retry on any `HTTP.HTTPError`. -Benchmarked directly against the live SkyBoT service (20 real, identical -requests, repeated to isolate throughput from any one query's own -content): concurrency=8 gave a real, measured 2.6x speedup (12.5s vs -32.8s sequential); concurrency up to 60 ran clean with zero errors, -though gains flattened past ~20-40 (IMCCE's own server-side queueing, not -this code, by then). Settled on 20 — inside the tested-clean range, not -pushed to its edge, matching `_CDS_BATCH_SIZE`'s conservative philosophy. -The full rerun with both fixes completed end to end, no crash, in around -20 minutes total (all 5 fields) — down from a run that hadn't even -finished after two hours. - -**`match_radius` was too loose, by a measured, corrected amount.** -`examples/iasc_demo.jl`'s `match_radius` was first converted from -`real_data_demo.jl`'s ZTF value to preserve the same ~10" angular -tolerance — a reasonable-looking choice that turned out to be looser than -PS1's own real astrometric precision. On the densest of the 5 fields, -this produced 10,422 tracklets from only ~500 detections/frame — almost -certainly distinct real stars within 10" of each other across frames -getting cross-linked into spurious tracklets, not 10,422 real moving -objects. The known SkyBoT objects were still correctly recovered in -every field regardless, but rather than guess at a tighter value, PS1's -own headers report the real number needed: `PERROR`, the astrometric -solution's per-star positional RMS residual, measured at 0.20-0.23" -across the fields checked here — not something assumed, read directly -from real data. Retuned `match_radius` to 2" (~10x `PERROR`, a -comfortable margin for real motion and centroiding noise, not the bare -residual) and reran all 5 fields: the same 9 distinct known objects were -recovered in every field (confirmed by name, not just by count — nothing -dropped out), while total tracklets across all 5 fields fell from 16,158 -to 4,960 (-69%). The reduction is concentrated exactly where predicted: -the densest field (XY42_p11) went from 10,422 to 3,478; the two -previously "26 real objects" and "13/2619" style counts were actually -counting duplicate tracklet-rows around the same handful of real -objects, not 26 distinct discoveries — a reporting correction as much as -a code fix, worth noting since the inflated number was reported once, -here, before the retune caught it. - ## Tuning `find_variable_sources`'s systematic error floor past the first value that worked The 1% systematic error floor (see above) was the first value tried, chosen because it's a standard number in forced-photometry pipelines, not because it was shown to be optimal. With a real positive control now in hand (ASASSN-V J183620.31, recovered by `find_variable_sources` earlier -this session), the floor could finally be checked from *both* sides of -the tradeoff at once, not just the false-positive side: sweeping -`systematic_error_fraction` against the same 152 real, matched, -high-S/N stationary stars from ZTF field 451 (a slightly different count -than the "119" quoted earlier — this sweep additionally required -`min_frames` at the full 5-frame count and `normalize=true` together, -narrowing the matched set), the `chi2_threshold=10.0` false-positive rate -dropped from 3.3% at a 1% floor to 2.0% at 2%, 1.3% at 3%, and 0% at 5%. -Naively, that argues for as high a floor as possible — but a floor this -large also suppresses *real* variability, and that side had never been -checked. Running the same sweep against ASASSN-V J183620.31's own real -forced-photometry light curve: reduced chi2 falls from 123 (at 1%) to 31 -(at 2%) to 13.8 (at 3%) to 5.0 (at 5%) — the last of which drops *below* +this session — see its own +[validation page](variable-star-validation.md)), the floor could finally +be checked from *both* sides of the tradeoff at once, not just the +false-positive side: sweeping `systematic_error_fraction` against the +same 152 real, matched, high-S/N stationary stars from ZTF field 451 (a +slightly different count than the "119" quoted earlier — this sweep +additionally required `min_frames` at the full 5-frame count and +`normalize=true` together, narrowing the matched set), the +`chi2_threshold=10.0` false-positive rate dropped from 3.3% at a 1% +floor to 2.0% at 2%, 1.3% at 3%, and 0% at 5%. Naively, that argues for +as high a floor as possible — but a floor this large also suppresses +*real* variability, and that side had never been checked. Running the +same sweep against ASASSN-V J183620.31's own real forced-photometry +light curve: reduced chi2 falls from 123 (at 1%) to 31 (at 2%) to 13.8 +(at 3%) to 5.0 (at 5%) — the last of which drops *below* `chi2_threshold=10.0`, meaning a 5% floor would have made this exact, real, independently-confirmed variable star invisible to `find_variable_sources`. Settled on 2%: comfortably above 1%'s @@ -469,49 +338,10 @@ false-positive rate, while leaving the real variable's over threshold — the largest floor checked that doesn't cost real detections, not the smallest false-positive rate achievable. -## Revisiting the two intentional-design "limitations" +![False-positive rate on 152 real stationary stars, and reduced chi2 on a real confirmed variable, swept across systematic_error_fraction values](assets/systematic-error-floor-sweep.png) -Two items in `docs/src/index.md`'s Known Limitations were design -decisions, not bugs — but "intentional" isn't the same as "as good as it -can be." Asked directly whether either could be genuinely improved -without abandoning the design choice behind it. +## See also -**`estimate_psf`'s empirical-vs-fallback split was a hard binary — a -field either had stars passing the isolation/saturation filter at exactly -the given `min_separation`, or it fell all the way back to an analytic -Moffat fit.** But a moderately (not severely) crowded field might have -real, usable stars at a *slightly* tighter isolation radius — the current -code was throwing that away and jumping straight to an approximation -instead of trying harder for the real thing. Added `relaxation_attempts` -(default 2): on an empty stamp list, halve `min_separation` and retry, -up to that many times, before falling back. A real empirical PSF from -fewer, closer stars still beats a parametric approximation, which is the -whole reason `estimate_psf` exists over just always using -`fit_moffat_psf`. Regression-tested with two synthetic Gaussian sources -25 px apart (fails the default `min_separation=40`, but 25 ≥ 20, the -first relaxed attempt) — recovers the real empirical PSF (correct FWHM) -instead of falling back, while `relaxation_attempts=0` on the same data -still fails cleanly, proving the relaxation is what does it. Writing that -test surfaced a second thing worth knowing: `fit_moffat_psf`'s own stamp -extraction (`stamp_size=25`, so a ±12 px half-window) doesn't check for -neighbor contamination the way `estimate_psf`'s isolation filter does — -stars closer than ~24 px apart corrupt each other's Moffat fit (tested -directly: clustering the existing fallback test's synthetic stars into a -6 px box made every fit fail to converge, where the original ~25 px -spacing fits cleanly) — not fixed here, since `fit_moffat_psf`'s own -docstring already documents that it deliberately skips the isolation -check as a defensible trade for a *fit* (a bad stamp shows up as a poor -residual, in principle), but the synthetic evidence says that trade has -a real limit worth knowing about. - -**`zogy_subtract`'s `V_ast` opt-in was silent — a direct caller who -simply didn't pass `n_sources`/`r_sources` got `V_ast = 0` with no -signal anything was skipped**, unlike `run_pipeline`, which always -supplies them. The design itself (opt-in at the low-level function, -mandatory at the high-level one) is still the right call — computing -`n_sources`/`r_sources` costs two extra `detect_sources` passes, real -cost a direct caller might legitimately not want. What was missing was -visibility: added a `@warn` (once per session, via `maxlog=1`, so it -doesn't spam a caller who's already made an informed choice) when both -are left `nothing`. Regression test confirms it fires when omitted and -stays silent when both are supplied. +- [Validating `search_field` against a real, independently-confirmed variable star](variable-star-validation.md) +- [Validating against real IASC (Pan-STARRS1) campaign data](iasc-campaign-validation.md) +- [Revisiting the two intentional-design "limitations"](design-refinements.md) blob - /dev/null blob + e01a25c74205d65cfec9fce040385e20a8626e17 (mode 644) --- /dev/null +++ docs/src/design-refinements.md @@ -0,0 +1,52 @@ +# Revisiting the two intentional-design "limitations" + +Part of this project's [Investigation Log](investigation-log.md) — +split onto its own page since it's a design review, not a bug found on a +specific real dataset the way the other pages here are. + +Two items in `docs/src/index.md`'s Known Limitations were design +decisions, not bugs — but "intentional" isn't the same as "as good as it +can be." Asked directly whether either could be genuinely improved +without abandoning the design choice behind it. + +## `estimate_psf`'s empirical-vs-fallback split was a hard binary + +A field either had stars passing the isolation/saturation filter at +exactly the given `min_separation`, or it fell all the way back to an +analytic Moffat fit. But a moderately (not severely) crowded field might +have real, usable stars at a *slightly* tighter isolation radius — the +current code was throwing that away and jumping straight to an +approximation instead of trying harder for the real thing. Added +`relaxation_attempts` (default 2): on an empty stamp list, halve +`min_separation` and retry, up to that many times, before falling back. +A real empirical PSF from fewer, closer stars still beats a parametric +approximation, which is the whole reason `estimate_psf` exists over just +always using `fit_moffat_psf`. Regression-tested with two synthetic +Gaussian sources 25 px apart (fails the default `min_separation=40`, but +25 ≥ 20, the first relaxed attempt) — recovers the real empirical PSF +(correct FWHM) instead of falling back, while `relaxation_attempts=0` on +the same data still fails cleanly, proving the relaxation is what does +it. Writing that test surfaced a second thing worth knowing: +`fit_moffat_psf`'s own stamp extraction (`stamp_size=25`, so a ±12 px +half-window) doesn't check for neighbor contamination the way +`estimate_psf`'s isolation filter does — stars closer than ~24 px apart +corrupt each other's Moffat fit (tested directly: clustering the +existing fallback test's synthetic stars into a 6 px box made every fit +fail to converge, where the original ~25 px spacing fits cleanly) — not +fixed here, since `fit_moffat_psf`'s own docstring already documents +that it deliberately skips the isolation check as a defensible trade for +a *fit* (a bad stamp shows up as a poor residual, in principle), but the +synthetic evidence says that trade has a real limit worth knowing about. + +## `zogy_subtract`'s `V_ast` opt-in was silent + +A direct caller who simply didn't pass `n_sources`/`r_sources` got +`V_ast = 0` with no signal anything was skipped, unlike `run_pipeline`, +which always supplies them. The design itself (opt-in at the low-level +function, mandatory at the high-level one) is still the right call — +computing `n_sources`/`r_sources` costs two extra `detect_sources` +passes, real cost a direct caller might legitimately not want. What was +missing was visibility: added a `@warn` (once per session, via +`maxlog=1`, so it doesn't spam a caller who's already made an informed +choice) when both are left `nothing`. Regression test confirms it fires +when omitted and stays silent when both are supplied. blob - /dev/null blob + e15a844b967eb0feb88c20288a6a487b211849ab (mode 644) --- /dev/null +++ docs/src/iasc-campaign-validation.md @@ -0,0 +1,118 @@ +# Validating against real IASC (Pan-STARRS1) campaign data + +Part of this project's [Investigation Log](investigation-log.md) — +split onto its own page since it's a self-contained validation story +against a different survey (Pan-STARRS1) than the ZTF narrative there. + +Every real-data check so far used ZTF. `docs/src/index.md`'s "Using real +IASC campaign data" section had stood as "not attempted" all session — +closed by running `examples/iasc_demo.jl` against 5 real Pan-STARRS1 +(PS1) IASC practice sets (2019-08-28/09-04/09-24, 4 exposures each). +`run_pipeline` recovered 9 real, independently-catalogued objects +across the 5 fields via SkyBoT — including a Jupiter Trojan, 2019 NB9 — +but getting a clean run took four real, fixed issues, found in this +order. + +## `load_wcs` failed on every one of these real headers + +Every PS1 header raised `"Linear transformation matrix is singular"` +from wcslib. Bisected a real header down to the exact cause (splitting +it into halves, testing each half in isolation, recursing into whichever +half still failed): `CNPIX1`/`CNPIX2` alone — a legacy IRAF/DSS +plate-astrometry keyword pair, present in these headers but with none of +that convention's other required keywords — was enough to reproduce it, +even combined with nothing but `SIMPLE`/`BITPIX`/`NAXIS`. wcslib reads +that keyword pair as the start of a *separate*, implicit DSS-style WCS +description, and with the rest of that convention absent builds an +all-zero, degenerate linear transform for it — a real wcslib parsing +quirk, not anything wrong with the header's own real, complete +CTYPE/CRVAL/CRPIX/CDELT WCS, which parses cleanly on its own. Fixed in +`load_wcs`: on exactly this error, retry after stripping just those two +keyword's FITS cards and nothing else — confirmed sufficient, not +guessed. Regression test constructs a synthetic header (a real WCS plus +injected `CNPIX1`/`CNPIX2` cards) since the real PS1 files can't be +committed to the repo. + +## `detect_sources` produced enormous numbers of spurious detections on some frames + +One real frame: 176,165 "detections" at `threshold=8.0` (field 451 on +ZTF, by comparison, has ~130). Root cause: these FITS files mark +invalid/masked pixels using the standard `BLANK` header keyword (scaled +through `BZERO`/`BSCALE` like any other pixel value) rather than `NaN`, +and `FITSIO.jl` does not convert `BLANK` sentinels automatically. One +real frame had 158,443 pixels (2.7% of the image) pegged at exactly that +sentinel value (65535, from `BLANK=32767` + `BZERO=32768`) — +`detect_sources` read that as enormous real flux across a large masked +region and found a spurious "source" seemingly everywhere. Confirmed +directly: replacing those exact pixels with the frame's own valid-region +median (before any detection) dropped the same frame from 176,165 to 176 +detections. Handled as a preprocessing step in `examples/iasc_demo.jl` +(`clean_blank_pixels`, writing cleaned copies preserving the original +header/WCS/timestamp exactly) rather than in `src/`, since BLANK-sentinel +handling is a real-FITS-ingestion concern specific to how a given survey +exports data, not something `detect_sources` itself should need to know +about. + +## `crossmatch_catalog(...; :skybot)` was too slow, and not resilient, at real scale + +Unlike `:vsx`/`:simbad` (batched via CDS TAP — see the +[Investigation Log](investigation-log.md)), SkyBoT has no batch mode, so +`_crossmatch_skybot` queried one candidate per request, fully +sequentially. Fine for a handful of candidates; not for hundreds to +thousands of real tracklets. A full 5-field run of `examples/iasc_demo.jl` +took over two hours and then died outright, deep into the fourth field's +crossmatch (2,619 candidates), to `"tls write failed: connection is +closed"` — an uncaught, unretried network error with no partial-progress +recovery. Fixed two ways in `_crossmatch_skybot`: concurrent requests +(Julia `Task`s + a bounded `Base.Semaphore`, not extra threads — this is +a network-latency-bound workload, and cooperative concurrency on however +many threads Julia already has is enough) and a single retry on any +`HTTP.HTTPError`. Benchmarked directly against the live SkyBoT service +(20 real, identical requests, repeated to isolate throughput from any one +query's own content): concurrency=8 gave a real, measured 2.6x speedup +(12.5s vs 32.8s sequential); concurrency up to 60 ran clean with zero +errors, though gains flattened past ~20-40 (IMCCE's own server-side +queueing, not this code, by then). Settled on 20 — inside the +tested-clean range, not pushed to its edge, matching `_CDS_BATCH_SIZE`'s +conservative philosophy. The full rerun with both fixes completed end to +end, no crash, in around 20 minutes total (all 5 fields) — down from a +run that hadn't even finished after two hours. + +## `match_radius` was too loose, by a measured, corrected amount + +`examples/iasc_demo.jl`'s `match_radius` was first converted from +`real_data_demo.jl`'s ZTF value to preserve the same ~10" angular +tolerance — a reasonable-looking choice that turned out to be looser than +PS1's own real astrometric precision. On the densest of the 5 fields, +this produced 10,422 tracklets from only ~500 detections/frame — almost +certainly distinct real stars within 10" of each other across frames +getting cross-linked into spurious tracklets, not 10,422 real moving +objects. The known SkyBoT objects were still correctly recovered in +every field regardless, but rather than guess at a tighter value, PS1's +own headers report the real number needed: `PERROR`, the astrometric +solution's per-star positional RMS residual, measured at 0.20-0.23" +across the fields checked here — not something assumed, read directly +from real data. Retuned `match_radius` to 2" (~10x `PERROR`, a +comfortable margin for real motion and centroiding noise, not the bare +residual) and reran all 5 fields: the same 9 distinct known objects were +recovered in every field (confirmed by name, not just by count — nothing +dropped out), while total tracklets across all 5 fields fell from 16,158 +to 4,960 (-69%). The reduction is concentrated exactly where predicted: +the densest field (XY42_p11) went from 10,422 to 3,478; the two +previously "26 real objects" and "13/2619" style counts were actually +counting duplicate tracklet-rows around the same handful of real +objects, not 26 distinct discoveries — a reporting correction as much as +a code fix, worth noting since the inflated number was reported once, +here, before the retune caught it. + +![Tracklet counts per field before and after retuning match_radius to PS1's real astrometric precision](assets/iasc-match-radius-retuning.png) + +## Recovered objects + +| Field | Tracklets (retuned) | Known objects recovered | +|:--|--:|:--| +| XY14_p10 | 52 | — | +| XY15_p01 | 132 | 2014 HO19 | +| XY25_p10 | 567 | 4311 T-1, 2009 SG135, 2015 XJ232, 2019 PK5 | +| XY26_p01 | 731 | 2001 SH320, 2011 SH185, 2019 NB9 (Jupiter Trojan) | +| XY42_p11 | 3,478 | 2008 FA111 | blob - /dev/null blob + fbc0d9a85c0549c39238219adbe6adceb746ffb6 (mode 644) --- /dev/null +++ docs/src/variable-star-validation.md @@ -0,0 +1,48 @@ +# Validating `search_field` against a real, independently-confirmed variable star + +Part of this project's [Investigation Log](investigation-log.md) — +split onto its own page since it's a self-contained validation story, +distinct from the ZTF field 451 bug-hunting narrative there. + +Every real-data check so far had confirmed known *moving* objects (via +SkyBoT) but never a known *variable* — the one real dataset checked had +only one catalogued VSX variable in its footprint, too faint to serve as +a useful positive control. Fixed by choosing a real target with an actual +VSX catalog entry: ASASSN-V J183620.31 (type EW, a contact eclipsing +binary, period 0.322427 d), in ZTF field 487/CCD 12/quadrant 1/zr, night +2019-06-10 — a real high-cadence campaign with 144 exposures over 2.45 h, +thinned to 29 for `examples/variable_star_demo.jl`. Declared in advance: +2.45 h covers only ~32% of one period, so full period recovery from this +single night was not expected — the actual test was whether +`find_variable_sources` flags the star as variable at all. + +First attempt did not run: `search_field`'s default `threshold=5.0` (used +at `8.0` in `real_data_demo.jl`) produced ~12,900 detections per frame +here, against field 451's ~130 — this field sits near the galactic plane. +`link_candidates`'s tracklet search is pairwise in detections/frame, and +did not finish in 35 minutes at that density; killed and confirmed via a +standalone check (`detect_sources` alone, one frame) that the counts were +real, not a hang. The target star is extremely bright at this field's +noise level (flux ≈ 87,000, ~1.7 px from its WCS-predicted position at +every threshold tested from 10 to 100), so raising the demo's threshold +to 60 — cutting detections/frame to ~1,260 — loses none of the signal +this run actually needs; a deliberate, field-specific tradeoff, not the +pipeline's own default. A second bug surfaced once linking finished: +`light_curve`'s default `timestamp_key="MJD-OBS"` doesn't exist in this +survey's headers (`search_field` was already correctly called with +`"OBSMJD"`) — a copy-paste omission in the demo script, not a pipeline +bug, fixed by passing the same key. + +With both fixed, the run found 22 variable candidates from 638 +detections; 2 matched a known VSX variable, including the target itself — +**ASASSN-V J183620.31, recovered at a 1.7" offset from its catalogued +position** — the positive-control criterion this whole exercise was +built to test, and it passed. A Lomb-Scargle fit to the recovered +candidate's own forced-photometry light curve (`light_curve` + +`recover_rotation_period`) found a real, highly significant periodic +signal (false-alarm probability ≈ 0) at 0.5 d, not the catalogued +0.322 d — the declared-in-advance outcome of fitting a period search to +a light curve covering less than a third of that period (almost +certainly an alias, not evidence against the true period), and not a +failure of `find_variable_sources` itself, which is what the crossmatch +recovery above actually validates. blob - /dev/null blob + 4db2ccebb03a1803dd5e43d0dd1103ec0b9b1b15 (mode 644) --- /dev/null +++ docs/make_figures.jl @@ -0,0 +1,104 @@ +#= +Generates the figures embedded in docs/src/investigation-log.md, from the +real numbers measured during this project's real-data investigations (not +synthetic/illustrative data) — see the corresponding prose sections for +where each number comes from. Run automatically by docs/make.jl before +`makedocs`, so CI regenerates these on every docs build rather than +committing static images that could drift from the numbers in the text. + +Styled for the wiki's dark-only theme (docs/src/.vitepress/config.mts, +appearance: 'force-dark') — transparent background, white text — rather +than a generic light-background default. +=# +using CairoMakie + +const ASSETS_DIR = joinpath(@__DIR__, "src", "assets") +mkpath(ASSETS_DIR) + +set_theme!(Theme( + backgroundcolor=:transparent, + textcolor=:white, + Axis=( + backgroundcolor=:transparent, + xlabelcolor=:white, ylabelcolor=:white, titlecolor=:white, + xticklabelcolor=:white, yticklabelcolor=:white, + xtickcolor=:white, ytickcolor=:white, + leftspinecolor=:white, rightspinecolor=:white, + bottomspinecolor=:white, topspinecolor=:white, + xgridcolor=(:white, 0.15), ygridcolor=(:white, 0.15), + ), + Legend=(backgroundcolor=:transparent, labelcolor=:white, framecolor=(:white, 0.3)), +)) + +# --- find_variable_sources: systematic_error_fraction sweep ----------- +# Real measurements from sweeping systematic_error_fraction against (a) +# 152 real, matched, high-S/N stationary stars on real ZTF field 451 +# (false-positive rate at chi2_threshold=10.0) and (b) the real, +# independently-confirmed variable ASASSN-V J183620.31's own real +# forced-photometry light curve (reduced chi2). See "Tuning +# find_variable_sources's systematic error floor..." in this log. +let + floors = [0.0, 0.5, 1.0, 1.5, 2.0, 3.0, 5.0] # percent + false_positive_rate = [13.2, 6.6, 3.3, 2.6, 2.0, 1.3, 0.0] # percent, at chi2_threshold=10 + variable_reduced_chi2 = [20688.5, 484.4, 123.4, 55.1, 31.0, 13.8, 5.0] + + fig = Figure(size=(720, 460)) + ax1 = Axis(fig[1, 1]; + xlabel="systematic_error_fraction (%)", + ylabel="false-positive rate at chi2_threshold=10 (%)", + title="Real ZTF field 451 stationary stars (152) vs. a real confirmed variable") + ax2 = Axis(fig[1, 1]; + ylabel="reduced chi² (real variable, ASASSN-V J183620.31)", + yscale=log10, yaxisposition=:right, ygridvisible=false) + hidespines!(ax2) + hidexdecorations!(ax2) + + l1 = lines!(ax1, floors, false_positive_rate; color=:tomato, linewidth=2) + scatter!(ax1, floors, false_positive_rate; color=:tomato, markersize=10) + l2 = lines!(ax2, floors, variable_reduced_chi2; color=:dodgerblue, linewidth=2) + scatter!(ax2, floors, variable_reduced_chi2; color=:dodgerblue, markersize=10) + hlines!(ax2, [10.0]; color=:dodgerblue, linestyle=:dash, linewidth=1) + vlines!(ax1, [2.0]; color=:white, linestyle=:dot, linewidth=1) + text!(ax1, 2.05, 12.0; text="chosen default (2%)", fontsize=12, color=:white) + + Legend(fig[2, 1], [l1, l2], + ["false-positive rate (left axis)", + "real variable's reduced chi² (right axis, log scale; dashed = chi2_threshold=10)"]; + orientation=:horizontal, tellwidth=false) + + save(joinpath(ASSETS_DIR, "systematic-error-floor-sweep.png"), fig) +end + +# --- IASC match_radius retuning: tracklet counts before/after --------- +# Real per-field tracklet counts from examples/iasc_demo.jl, before +# (match_radius converted from ZTF's pixel scale, ~10" tolerance) and +# after (match_radius from PS1's own PERROR, ~2" tolerance) retuning. The +# same 9 distinct known objects were recovered in both runs — this is a +# reduction in spurious duplicate tracklets, not lost detections. See +# "match_radius was too loose, by a measured, corrected amount" above. +let + fields = ["XY14_p10", "XY15_p01", "XY25_p10", "XY26_p01", "XY42_p11"] + before = [524, 776, 1817, 2619, 10422] + after = [52, 132, 567, 731, 3478] + + fig = Figure(size=(720, 440)) + ax = Axis(fig[1, 1]; + xlabel="IASC practice field", ylabel="tracklets found", + title="Retuning match_radius to PS1's own real astrometric precision (PERROR)", + xticks=(1:length(fields), fields), yscale=log10) + + n = length(fields) + x = 1:n + barplot!(ax, x .- 0.15, before; width=0.3, color=:salmon, label="match_radius ≈ 10\" (ZTF's tolerance, reused as-is)") + barplot!(ax, x .+ 0.15, after; width=0.3, color=:mediumspringgreen, label="match_radius = 2\" (10x PS1's real PERROR)") + # A log axis has no true zero, so bars would otherwise auto-start from + # whatever the smallest value happens to be (misleadingly making every + # bar look like a similar-height floating block) — pin the bottom + # near 1 instead, the closest a log axis gets to a real baseline. + ylims!(ax, 1, 20000) + + axislegend(ax; position=:lt) + save(joinpath(ASSETS_DIR, "iasc-match-radius-retuning.png"), fig) +end + +println("Figures written to $ASSETS_DIR") blob - 2b2b27948f0ca5c2c63b97a597cdafe8461ba09c blob + 9010fdd9ad1ed9b7b78d9e0e8a51d20e4cc1e441 --- src/variables.jl +++ src/variables.jl @@ -195,7 +195,7 @@ confirmed variable, not just the stationary-star side `systematic_error_fraction=0.02` (the current default) cuts the threshold-10 false-positive rate to 2.0% (vs 1%'s 3.3%), while a real, independently-confirmed variable (ASASSN-V J183620.31 — see the -[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#Validating-search_field-against-a-real,-independently-confirmed-variable-star)) +[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/variable-star-validation)) still clears `chi2_threshold=10.0` with a healthy margin (reduced chi2 ≈ 31, over 3x the threshold) at this floor — a floor of 5% would erase that same real signal (reduced chi2 drops to ≈5, below threshold), so 2% is