commit - 92ed176ba250755bdfec88536921f1d0f69f0113
commit + 25b93c9f8a5de7388b1511bb141c781c89e97e89
blob - 69478b44b41e88fb15a12ccab53f712deac224e7
blob + 79ee32b83c44adccc6c4718593d9a249be84ba60
--- .gitignore
+++ .gitignore
.direnv/
# docs/ — build output and DocumenterVitepress's own Node/Vitepress
-# tooling. docs/src/index.md and docs/src/investigation-log.md are real,
-# permanent, committed pages, not generated — deliberately not listed
-# here.
+# tooling. docs/src/*.md are real, permanent, committed pages, not
+# generated — deliberately not listed here.
docs/build/
docs/node_modules/
docs/package-lock.json
docs/.vitepress/dist/
docs/src/.vitepress/cache/
docs/src/.vitepress/dist/
+
+# docs/src/assets/ — generated by docs/make_figures.jl on every build
+# (local and CI), so these would just be stale duplicates if committed;
+# not a blanket ignore of the whole assets/ directory in case a real,
+# hand-authored logo.png/favicon.ico is ever added there.
+docs/src/assets/systematic-error-floor-sweep.png
+docs/src/assets/iasc-match-radius-retuning.png
blob - 786ff5301199a081a8d89784fe18fb61cbce2f1d
blob + a29f9fc65acda46134a5d9d1037853730827767d
--- docs/Project.toml
+++ docs/Project.toml
[deps]
AsteroidPipeline = "1001636e-a0d1-496e-b23d-cef10bc3cdca"
+CairoMakie = "13f3f980-e62b-5c42-98c6-ff1f3baf88f0"
Documenter = "e30172f5-a6a5-5a46-863b-614d45cd2de4"
DocumenterVitepress = "4710194d-e776-4893-9690-8d956a29c365"
[compat]
+CairoMakie = "0.15.13"
Documenter = "1.17.0"
DocumenterVitepress = "0.3.5"
blob - c4e7d08183f3d346911c64e4e8c8a49db7a2e0bf
blob + 6227dc17ae6320be5f0bf3a3ce440aeb7d26cbd6
--- docs/make.jl
+++ docs/make.jl
using Documenter
using DocumenterVitepress
-# index.md and investigation-log.md are both real, permanent, committed
-# pages under docs/src — the docs site's own content, not generated or
-# copied in from README.md/INVESTIGATION_LOG.md (which no longer exist at
-# the repo root; the short root README.md just links here instead).
+# index.md, investigation-log.md, variable-star-validation.md,
+# iasc-campaign-validation.md, and design-refinements.md are all real,
+# permanent, committed pages under docs/src — the docs site's own
+# content, not generated or copied in from README.md/INVESTIGATION_LOG.md
+# (which no longer exist at the repo root; the short root README.md just
+# links here instead).
DocMeta.setdocmeta!(AsteroidPipeline, :DocTestSetup, :(using AsteroidPipeline); recursive=true)
+# Regenerates the figures investigation-log.md embeds, from the real
+# numbers measured during this project's real-data investigations — run
+# before makedocs so both local builds and CI (julia-actions/julia-docdeploy
+# instantiates docs/Project.toml, which includes CairoMakie) produce them
+# fresh rather than relying on committed images that could drift from the
+# numbers in the text.
+include("make_figures.jl")
+
makedocs(;
modules=[AsteroidPipeline],
authors="Alejandro",
),
pages=[
"Home" => "index.md",
- "Investigation Log" => "investigation-log.md",
+ "Investigation Log" => [
+ "ZTF field 451 (the original investigation)" => "investigation-log.md",
+ "Variable-star validation" => "variable-star-validation.md",
+ "IASC / Pan-STARRS1 campaign validation" => "iasc-campaign-validation.md",
+ "Design refinements" => "design-refinements.md",
+ ],
"API Reference" => [
"Detection & Linking" => "api/detection.md",
"Variable Stars" => "api/variables.md",
blob - /dev/null
blob + c8f964a9243160cc267e901319aa38b2b1957bd1 (mode 644)
--- /dev/null
+++ docs/src/.vitepress/config.mts
+import { defineConfig } from 'vitepress'
+import { tabsMarkdownPlugin } from 'vitepress-plugin-tabs'
+import { mathjaxPlugin } from './mathjax-plugin'
+import { juliaReplTransformer } from './julia-repl-transformer'
+import footnote from "markdown-it-footnote";
+import path from 'path'
+
+const mathjax = mathjaxPlugin()
+
+function getBaseRepository(base: string): string {
+ if (!base || base === '/') return '/';
+ const parts = base.split('/').filter(Boolean);
+ return parts.length > 0 ? `/${parts[0]}/` : '/';
+}
+
+const baseTemp = {
+ base: 'REPLACE_ME_DOCUMENTER_VITEPRESS',// TODO: replace this in makedocs!
+}
+
+const navTemp = {
+ nav: 'REPLACE_ME_DOCUMENTER_VITEPRESS',
+}
+
+const nav = [
+ ...navTemp.nav,
+ {
+ component: 'VersionPicker'
+ }
+]
+
+// https://vitepress.dev/reference/site-config
+export default defineConfig({
+ base: 'REPLACE_ME_DOCUMENTER_VITEPRESS',// TODO: replace this in makedocs!
+ title: 'REPLACE_ME_DOCUMENTER_VITEPRESS',
+ description: 'REPLACE_ME_DOCUMENTER_VITEPRESS',
+ lastUpdated: true,
+ cleanUrls: true,
+ // Dark theme only, by request — no light mode, no toggle to switch to
+ // one. 'force-dark' (vs. just 'dark') is what actually removes the
+ // appearance toggle from the nav bar, not merely defaulting to it.
+ appearance: 'force-dark',
+ outDir: 'REPLACE_ME_DOCUMENTER_VITEPRESS', // This is required for MarkdownVitepress to work correctly...
+ head: [
+ ['link', { rel: 'icon', href: 'REPLACE_ME_DOCUMENTER_VITEPRESS_FAVICON' }],
+ ['script', {src: `${getBaseRepository(baseTemp.base)}versions.js`}],
+ // ['script', {src: '/versions.js'], for custom domains, I guess if deploy_url is available.
+ ['script', {src: `${baseTemp.base}siteinfo.js`}],
+ // REPLACE_ME_DOCUMENTER_VITEPRESS_NOINDEX
+ ],
+
+ markdown: {
+ codeTransformers: [juliaReplTransformer()],
+ config(md) {
+ md.use(tabsMarkdownPlugin);
+ md.use(footnote);
+ mathjax.markdownConfig(md);
+ },
+ theme: {
+ light: "github-light",
+ dark: "github-dark"
+ },
+ },
+ vite: {
+ plugins: [
+ mathjax.vitePlugin,
+ ],
+ define: {
+ __DEPLOY_ABSPATH__: JSON.stringify('REPLACE_ME_DOCUMENTER_VITEPRESS_DEPLOY_ABSPATH'),
+ },
+ resolve: {
+ alias: {
+ '@': path.resolve(__dirname, '../components')
+ }
+ },
+ optimizeDeps: {
+ exclude: [
+ '@nolebase/vitepress-plugin-enhanced-readabilities/client',
+ 'vitepress',
+ '@nolebase/ui',
+ ],
+ },
+ ssr: {
+ noExternal: [
+ // If there are other packages that need to be processed by Vite, you can add them here.
+ '@nolebase/vitepress-plugin-enhanced-readabilities',
+ '@nolebase/ui',
+ ],
+ },
+ },
+ themeConfig: {
+ outline: 'deep',
+ logo: 'REPLACE_ME_DOCUMENTER_VITEPRESS',
+ search: {
+ provider: 'local',
+ options: {
+ detailedView: true
+ }
+ },
+ nav,
+ sidebar: 'REPLACE_ME_DOCUMENTER_VITEPRESS',
+ sidebarDrawer: 'REPLACE_ME_DOCUMENTER_VITEPRESS_SIDEBAR_DRAWER',
+ editLink: 'REPLACE_ME_DOCUMENTER_VITEPRESS',
+ socialLinks: [
+ { icon: 'github', link: 'REPLACE_ME_DOCUMENTER_VITEPRESS' }
+ ],
+ footer: {
+ message: 'Made with <a href="https://luxdl.github.io/DocumenterVitepress.jl/dev/" target="_blank"><strong>DocumenterVitepress.jl</strong></a><br>',
+ copyright: `© Copyright ${new Date().getUTCFullYear()}.`
+ }
+ }
+})
blob - dd66e303cdccc4f3354ebc934d1667928a28cfc4
blob + 63752c22722c6997860cb04bebed3bbfb8ec55f4
--- docs/src/index.md
+++ docs/src/index.md
## Status
-Early development. `detect_sources`, `link_candidates`, the WCS
-calibration step, and `crossmatch_catalog` are implemented and wired
-together end to end in `run_pipeline`, validated against synthetic FITS
-frames with a known injected source track, and exercised against real
-public survey data (see `examples/real_data_demo.jl`) and real IASC
-practice campaign data (see `examples/iasc_demo.jl` and the "Using real
-IASC campaign data" section below — 9 real, independently-catalogued
-objects recovered across 5 real Pan-STARRS1 fields).
+!!! note "Early development, but validated end to end against real data"
+ Every stage below has run against real telescope data at least once —
+ not just synthetic tests — including three independent surveys (ZTF,
+ Pan-STARRS1) and a real, independently-confirmed variable star. See
+ [Real-data validation](@ref) for the full story of each row.
-ZOGY difference imaging (Zackay, Ofek & Gal-Yam 2016) is implemented —
-`build_reference` stacks a deep reference from many epochs via
-reprojection (`Reproject.jl`) onto a common pixel grid, `estimate_psf`
-measures each frame's empirical PSF from its own bright stars, and
-`zogy_subtract` produces a statistically normalized detection-significance
-map (`S_corr`, unit variance by construction) in Fourier space via
-`FFTW.jl`. Validated against three falsifiable synthetic checks (identical
-images subtract to zero; pure noise gives `std(S_corr) ≈ 1`; an injected
-source's peak significance matches the analytic matched-filter prediction)
-and against real ZTF data — see `examples/real_data_demo.jl`, which runs
-the pipeline with and without differencing on the same frames and reports
-which known objects each recovers, rather than assuming differencing
-helps.
+| Component | Validated against |
+|:--|:--|
+| `detect_sources`, `link_candidates`, WCS calibration, `crossmatch_catalog` | Synthetic (exact recovery) · real ZTF field 451 · 5 real IASC/Pan-STARRS1 fields (**9 known objects** recovered) |
+| ZOGY difference imaging (`build_reference`, `estimate_psf`, `zogy_subtract`) | 3 falsifiable synthetic checks · real ZTF field 451 |
+| `find_variable_sources` / `search_field` | Synthetic · false-positive rate calibrated on real ZTF stars · a real confirmed variable (ASASSN-V J183620.31) |
+| `fit_moffat_psf` (PSF analytic fallback) | Synthetic · PSF width calibrated on real ZTF data |
+| `plate_solve` | Live nova.astrometry.net service |
+| `crossmatch_catalog` (`:skybot`/`:vsx`/`:simbad`) | Live services, real positive controls |
-`find_variable_sources`/`search_field` (stationary, flux-varying source
-detection) and `fit_moffat_psf` (`estimate_psf`'s analytic-PSF fallback)
-are implemented, tested against synthetic data, and — for
-`find_variable_sources`'s photometric normalization, S/N floor, and
-`chi2_threshold` default, and for `fit_moffat_psf`'s recovered PSF
-width — calibrated directly against real ZTF data (see the
-[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#detect_sources's-flux-has-been-silently-wrong-since-the-beginning:-a-transposed-aperture), including a real bug in
-`detect_sources`'s own flux measurement this calibration work found and
-fixed). Now also validated end to end, via `search_field`, against a real
-field with an independently-confirmed variable star as ground truth: ZTF
-field 487/CCD 12/quadrant 1/zr, night 2019-06-10, containing ASASSN-V
-J183620.31 (VSX: type EW, period 0.322 d) — see
-`examples/variable_star_demo.jl` and the
-[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#Validating-search_field-against-a-real,-independently-confirmed-variable-star)
-for the full run. The target was recovered (crossmatched against VSX at a
-1.7" offset) among 22 variable candidates from 638 detections; a
-Lomb-Scargle period fit to that single partial night (only ~32% of one
-0.322 d period) found a real, highly significant periodic signal
-(FAP ≈ 0) but at 0.5 d, not the true period — an expected outcome of the
-partial phase coverage, declared before running rather than adjusted
-after seeing it, and not treated as a failure of `find_variable_sources`
-itself (which is what the crossmatch recovery actually validates).
+Not yet run on a live IASC search campaign (as opposed to practice data)
+— see [Using real IASC campaign data](@ref) for what that would involve.
-On that real dataset (field 451, 2019-10-23), the undifferenced baseline
-finds 133 tracklets and recovers both known objects in the field (2002
-UY45, 1997 KO3); ZOGY also recovers both, at consistent sky offsets
-(confirming the subtraction is correctly calibrated), but finds 667
-tracklets total and no *additional* known object — both known objects
-here are bright enough that the baseline already recovers them trivially,
-so this dataset doesn't exercise ZOGY's actual advantage (recovering
-objects below a single frame's noise floor). The raw tracklet-count gap
-is not a clean read on ZOGY's noise properties; see the
-[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-quality-gate's-combinatorial-side-effect-on-tracklet-count) for why, and for the full
-record of every real bug this project's real-data testing surfaced so
-far — most recently three more from the real IASC campaign work below —
-all fixed with regression tests, and how each was diagnosed.
+## Real-data validation
-## Known limitations
+### ZTF: `examples/real_data_demo.jl`
-- **Empirical PSF's quality still depends on the field, though less than
- it used to.** `estimate_psf` stacks real star cutouts, which captures
- the true PSF shape (wings included) without fitting a model family per
- instrument, but needs enough bright, isolated, unsaturated stars to do
- it. It now retries at a progressively relaxed `min_separation` (halved,
- up to `relaxation_attempts` times, default 2) before giving up — a
- field with a few usable stars at a tighter isolation radius still gives
- the real PSF shape, which the analytic fallback never can. Only once
- even the most relaxed attempt finds nothing does it fall back (by
- default) to `fit_moffat_psf` — a parametric Moffat fit — rather than
- failing outright; the fallback trades exact PSF shape for robustness,
- and is not itself a substitute for a genuinely well-behaved field.
-- **`zogy_subtract`'s astrometric-noise term (`V_ast`) is still opt-in at
- the `zogy_subtract` level** — it needs `n_sources`/`r_sources` passed
- explicitly, and is `0` without them; this is deliberate API layering
- (`run_pipeline` always supplies them, so a direct caller who doesn't
- need the extra `detect_sources` cost can skip it), not something
- planned to change. What did change: leaving both unset now emits a
- `@warn` (once per session), so omitting `V_ast` is a visible choice
- instead of a silent default a direct caller could miss.
-- **`find_variable_sources` still has a real, measured false-positive
- floor on real single-epoch aperture photometry, even after fixing it
- once.** The first hypothesis tried — pixel-grid jitter in
- `detect_sources`'s peak-pixel aperture centering — turned out to be a
- real but secondary effect: refining the aperture to a sub-pixel
- centroid (see `detect_sources`'s docstring) barely moved the
- false-positive rate (23% → 22% at `chi2_threshold=3` on real ZTF field
- 451). The actual dominant cause, found by checking *which* stars were
- flagged, was a systematic photometric error floor — bright stars'
- tiny formal errors made ordinary flat-fielding/PSF-variation
- systematics look like huge chi2 significance. Adding that floor
- (`variability_chi2`'s `systematic_error_fraction`) cut the rate: a 1%
- floor took it to 6% at threshold 3, ~0% at threshold 20; sweeping the
- floor further (against the same real stars, and checked against a real
- confirmed variable — ASASSN-V J183620.31 — to make sure real
- sensitivity wasn't sacrificed for it) found more room, without giving
- up real detections: the current default, 2%, cuts the
- `chi2_threshold=10.0` rate to 2.0% on this dataset (vs 1%'s 3.3%),
- while that confirmed variable still clears the threshold with a 3x
- margin — see `find_variable_sources`'s
- docstring and the [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-centroid-fix-barely-moved-the-false-positive-floor-—-the-real-cause-was-a-systematic-error-floor) for the
- full before/after numbers. Still not zero: treat a candidate as needing
- independent confirmation (a catalog match or a recovered period), not
- as self-evidently real.
+Runs the pipeline against real ZTF (Zwicky Transient Facility) frames
+both with and without ZOGY differencing, and cross-matches both against
+SkyBoT — a controlled comparison, not just a demonstration. Fetch the
+data first (public, no authentication required):
-## Example: real data
-
-`examples/real_data_demo.jl` runs the pipeline against real ZTF (Zwicky
-Transient Facility) frames both with and without ZOGY differencing, and
-cross-matches both against SkyBoT — a controlled comparison, not just a
-demonstration. Fetch the data first (public, no authentication required):
-
```
examples/fetch_data.sh
julia --project=. examples/real_data_demo.jl
Building the reference stack (30 frames, each individually reprojected)
is the slow part — tens of minutes on a laptop, one-time per run.
-## Rotation period recovery
+On field 451 (2019-10-23), the undifferenced baseline finds 133
+tracklets and recovers both known objects in the field (2002 UY45, 1997
+KO3); ZOGY also recovers both, at consistent sky offsets (confirming the
+subtraction is correctly calibrated), but finds 667 tracklets total and
+no *additional* known object — both known objects here are bright enough
+that the baseline already recovers them trivially, so this dataset
+doesn't exercise ZOGY's actual advantage (recovering objects below a
+single frame's noise floor). The raw tracklet-count gap is not a clean
+read on ZOGY's noise properties; see the
+[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-quality-gate's-combinatorial-side-effect-on-tracklet-count)
+for why, and for the full record of every real bug this project's
+real-data testing has surfaced, all fixed with regression tests.
-For a confirmed discovery, given a dedicated photometric follow-up
-sequence (many exposures over hours, at a fixed sky position — the
-target should barely move between them, unlike the original discovery
-epochs):
+`find_variable_sources`/`search_field` and `fit_moffat_psf` were also
+calibrated against this same field — photometric normalization, S/N
+floor, `chi2_threshold` default, and recovered PSF width — see the
+[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#detect_sources's-flux-has-been-silently-wrong-since-the-beginning:-a-transposed-aperture),
+including a real bug in `detect_sources`'s own flux measurement this
+calibration work found and fixed.
-```julia
-using AsteroidPipeline
+### Variable stars: `examples/variable_star_demo.jl`
-times, flux, flux_err = light_curve(fits_paths, ra, dec)
-result = recover_rotation_period(times, flux; minimum_period=0.02, maximum_period=1.0)
-result.period, result.false_alarm_probability
-```
+Validates `search_field` end to end against a real,
+independently-confirmed variable star — ZTF field 487/CCD 12/quadrant
+1/zr, night 2019-06-10, containing ASASSN-V J183620.31 (VSX: type EW,
+period 0.322 d). See the
+[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/variable-star-validation)
+for the full run.
-`minimum_period`/`maximum_period` bound the search (same units as
-`times`, i.e. days) and should bracket the rotation periods physically
-plausible for the object's size class. A small `false_alarm_probability`
-is what distinguishes a real periodic signal from a noise fluctuation —
-see the function's docstring.
+The target was recovered (crossmatched against VSX at a 1.7" offset)
+among 22 variable candidates from 638 detections. A Lomb-Scargle period
+fit to that single partial night (only ~32% of one 0.322 d period) found
+a real, highly significant periodic signal (FAP ≈ 0) but at 0.5 d, not
+the true period — an expected outcome of the partial phase coverage,
+declared before running rather than adjusted after seeing it, and not a
+failure of `find_variable_sources` itself (which is what the crossmatch
+recovery above actually validates).
-## Plate-solving
+### IASC / Pan-STARRS1: `examples/iasc_demo.jl`
-For a frame with no WCS already in its header, and a
-[nova.astrometry.net](https://nova.astrometry.net/) API key (free
-registration):
-
-```julia
-using AsteroidPipeline
-
-run_pipeline(fits_paths; reference=reference, plate_solve_api_key=key)
-```
-
-or directly: `plate_solve(fits_path; api_key=key)`. This is a live
-network round trip — upload, then poll until the frame solves — so it is
-slow and requires connectivity. Validated against the real service: see
-the [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#plate_solve-validated-end-to-end-against-the-live-service).
-
-## Using real IASC campaign data
-
-Now attempted, against 5 real Pan-STARRS1 (PS1) IASC practice sets
-("Practice Image Sets", 2019-08-28/09-04/09-24), each 4 exposures of the
-same field over ~40-70 min — see `examples/iasc_demo.jl` (point it at
+Validates `run_pipeline` against 5 real Pan-STARRS1 (PS1) IASC practice
+sets ("Practice Image Sets", 2019-08-28/09-04/09-24), each 4 exposures of
+the same field over ~40-70 min — see `examples/iasc_demo.jl` (point it at
your own local practice/campaign FITS; IASC material isn't public, so
unlike the ZTF demo above there is no fetch script). `run_pipeline`
recovered **9 real, independently-catalogued objects** across the 5
against real IASC-style data, not just ZTF.
Getting there surfaced four real, fixed issues — see the
-[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#Validating-against-real-IASC-Pan-STARRS1-campaign-data)
+[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/iasc-campaign-validation)
for the full story of each:
- `load_wcs` raised "Linear transformation matrix is singular" on every
preprocessing step in `examples/iasc_demo.jl` (not in `src/`, since
this is a real-FITS-ingestion concern, not `detect_sources`'s job).
- `crossmatch_catalog(...; :skybot)` queried one candidate at a time,
- fully sequentially — real candidate lists here (hundreds to
- thousands of tracklets) took minutes to hours, and a multi-hour run
- eventually died to a transient connection error with no retry. Fixed
- in `_crossmatch_skybot` itself: concurrent requests (real, measured
- ~8x wall-clock speedup) and a retry on transient HTTP errors.
+ fully sequentially — real candidate lists here (hundreds to thousands
+ of tracklets) took minutes to hours, and a multi-hour run eventually
+ died to a transient connection error with no retry. Fixed in
+ `_crossmatch_skybot` itself: concurrent requests (real, measured ~8x
+ wall-clock speedup) and a retry on transient HTTP errors.
- `match_radius`, first converted from `real_data_demo.jl`'s ZTF value to
keep the same ~10" angular tolerance, turned out far looser than PS1's
- own real astrometric precision (`PERROR`, in these headers: 0.20-0.23")
- — on the densest field this produced 10,422 tracklets, almost all
- spurious duplicates of the same real objects (distinct real stars
- within 10" of each other, or the same object matched by several
+ own real astrometric precision (`PERROR`, in these headers:
+ 0.20-0.23") — on the densest field this produced 10,422 tracklets,
+ almost all spurious duplicates of the same real objects (distinct real
+ stars within 10" of each other, or the same object matched by several
near-identical trial velocities). Retuned to 2" (~10x `PERROR`,
measured from the headers, not guessed) and confirmed directly: the
same 9 distinct known objects are still recovered in every field,
while total tracklets across all 5 fields drop from 16,158 to 4,960
(-69%) — this was cleanup of spurious duplicates, not lost detections.
-General guidance for pointing this pipeline at other real campaign data:
+#### Using real IASC campaign data
+Not yet attempted on a live campaign — practice data above is the
+closest real-world test so far. General guidance for pointing this
+pipeline at a real campaign, or any other survey's data:
+
- Point `run_pipeline` (or `examples/iasc_demo.jl`'s pattern) at the
local file paths directly; no fetch script is needed for files you
already have.
the field (via `crossmatch_catalog(...; :skybot)`) are the same kind of
ground truth used there.
+## Known limitations
+
+!!! warning "Empirical PSF's quality still depends on the field"
+ `estimate_psf` stacks real star cutouts, which captures the true PSF
+ shape (wings included) without fitting a model family per
+ instrument, but needs enough bright, isolated, unsaturated stars to
+ do it. It now retries at a progressively relaxed `min_separation`
+ (halved, up to `relaxation_attempts` times, default 2) before giving
+ up — a field with a few usable stars at a tighter isolation radius
+ still gives the real PSF shape, which the analytic fallback never
+ can. Only once even the most relaxed attempt finds nothing does it
+ fall back (by default) to `fit_moffat_psf` — a parametric Moffat fit
+ — rather than failing outright; the fallback trades exact PSF shape
+ for robustness, and is not itself a substitute for a genuinely
+ well-behaved field.
+
+!!! note "`zogy_subtract`'s astrometric-noise term (`V_ast`) is opt-in by design"
+ It needs `n_sources`/`r_sources` passed explicitly, and is `0`
+ without them; this is deliberate API layering (`run_pipeline` always
+ supplies them, so a direct caller who doesn't need the extra
+ `detect_sources` cost can skip it), not something planned to change.
+ What did change: leaving both unset now emits a `@warn` (once per
+ session), so omitting `V_ast` is a visible choice instead of a
+ silent default a direct caller could miss.
+
+!!! warning "`find_variable_sources` has a real, measured false-positive floor"
+ The first hypothesis tried — pixel-grid jitter in `detect_sources`'s
+ peak-pixel aperture centering — turned out to be a real but
+ secondary effect: refining the aperture to a sub-pixel centroid (see
+ `detect_sources`'s docstring) barely moved the false-positive rate
+ (23% → 22% at `chi2_threshold=3` on real ZTF field 451). The actual
+ dominant cause, found by checking *which* stars were flagged, was a
+ systematic photometric error floor — bright stars' tiny formal
+ errors made ordinary flat-fielding/PSF-variation systematics look
+ like huge chi2 significance. Adding that floor
+ (`variability_chi2`'s `systematic_error_fraction`) cut the rate: a
+ 1% floor took it to 6% at threshold 3, ~0% at threshold 20; sweeping
+ the floor further (against the same real stars, and checked against
+ a real confirmed variable — ASASSN-V J183620.31 — to make sure real
+ sensitivity wasn't sacrificed for it) found more room without giving
+ up real detections: the current default, 2%, cuts the
+ `chi2_threshold=10.0` rate to 2.0% on this dataset (vs 1%'s 3.3%),
+ while that confirmed variable still clears the threshold with a 3x
+ margin — see `find_variable_sources`'s docstring and the
+ [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-centroid-fix-barely-moved-the-false-positive-floor-—-the-real-cause-was-a-systematic-error-floor)
+ for the full before/after numbers. Still not zero: treat a candidate
+ as needing independent confirmation (a catalog match or a recovered
+ period), not as self-evidently real.
+
+## Follow-up workflows
+
+### Rotation period recovery
+
+For a confirmed discovery, given a dedicated photometric follow-up
+sequence (many exposures over hours, at a fixed sky position — the
+target should barely move between them, unlike the original discovery
+epochs):
+
+```julia
+using AsteroidPipeline
+
+times, flux, flux_err = light_curve(fits_paths, ra, dec)
+result = recover_rotation_period(times, flux; minimum_period=0.02, maximum_period=1.0)
+result.period, result.false_alarm_probability
+```
+
+`minimum_period`/`maximum_period` bound the search (same units as
+`times`, i.e. days) and should bracket the rotation periods physically
+plausible for the object's size class. A small `false_alarm_probability`
+is what distinguishes a real periodic signal from a noise fluctuation —
+see the function's docstring.
+
+### Plate-solving
+
+For a frame with no WCS already in its header, and a
+[nova.astrometry.net](https://nova.astrometry.net/) API key (free
+registration):
+
+```julia
+using AsteroidPipeline
+
+run_pipeline(fits_paths; reference=reference, plate_solve_api_key=key)
+```
+
+or directly: `plate_solve(fits_path; api_key=key)`. This is a live
+network round trip — upload, then poll until the frame solves — so it is
+slow and requires connectivity. Validated against the real service: see
+the [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#plate_solve-validated-end-to-end-against-the-live-service).
+
## Dependencies
- [FITSIO.jl](https://github.com/JuliaAstro/FITSIO.jl) — FITS I/O
blob - dc62fc44f1207d65635045e46ac2e212d7be5467
blob + e6f5a8a6c1241d9ba058bc1eb1378404ce5ba18d
--- docs/src/investigation-log.md
+++ docs/src/investigation-log.md
# Investigation log
Chronological record of what real-data testing found and how each finding
-was diagnosed — kept separate from `README.md` so the README stays a
-focused reference rather than a narrative. Cross-referenced from
-`README.md`'s Known limitations section and from the relevant docstrings.
+was diagnosed — kept separate from the main wiki pages so those stay
+focused references rather than narratives. Cross-referenced from
+`docs/src/index.md`'s Known Limitations section and from the relevant
+docstrings.
+This page covers the original ZTF field 451 investigation, chronologically
+first. Later validation work against other real datasets got large enough
+to split onto their own pages:
+
+- [Validating `search_field` against a real, independently-confirmed variable star](variable-star-validation.md)
+- [Validating against real IASC (Pan-STARRS1) campaign data](iasc-campaign-validation.md)
+- [Revisiting the two intentional-design "limitations"](design-refinements.md)
+
## Real ZTF data surfaced four bugs no synthetic test or code review caught
All fixed, all with regression tests added afterward:
not measured at higher N on these free, shared, anonymous-use services.
Cuts N requests to `ceil(N / 50)`.
-## Validating `search_field` against a real, independently-confirmed variable star
-
-Every real-data check so far had confirmed known *moving* objects (via
-SkyBoT) but never a known *variable* — the one real dataset checked had
-only one catalogued VSX variable in its footprint, too faint to serve as
-a useful positive control. Fixed by choosing a real target with an actual
-VSX catalog entry: ASASSN-V J183620.31 (type EW, a contact eclipsing
-binary, period 0.322427 d), in ZTF field 487/CCD 12/quadrant 1/zr, night
-2019-06-10 — a real high-cadence campaign with 144 exposures over 2.45 h,
-thinned to 29 for `examples/variable_star_demo.jl`. Declared in advance:
-2.45 h covers only ~32% of one period, so full period recovery from this
-single night was not expected — the actual test was whether
-`find_variable_sources` flags the star as variable at all.
-
-First attempt did not run: `search_field`'s default `threshold=5.0` (used
-at `8.0` in `real_data_demo.jl`) produced ~12,900 detections per frame
-here, against field 451's ~130 — this field sits near the galactic plane.
-`link_candidates`'s tracklet search is pairwise in detections/frame, and
-did not finish in 35 minutes at that density; killed and confirmed via a
-standalone check (`detect_sources` alone, one frame) that the counts were
-real, not a hang. The target star is extremely bright at this field's
-noise level (flux ≈ 87,000, ~1.7 px from its WCS-predicted position at
-every threshold tested from 10 to 100), so raising the demo's threshold
-to 60 — cutting detections/frame to ~1,260 — loses none of the signal
-this run actually needs; a deliberate, field-specific tradeoff, not the
-pipeline's own default. A second bug surfaced once linking finished:
-`light_curve`'s default `timestamp_key="MJD-OBS"` doesn't exist in this
-survey's headers (`search_field` was already correctly called with
-`"OBSMJD"`) — a copy-paste omission in the demo script, not a pipeline
-bug, fixed by passing the same key.
-
-With both fixed, the run found 22 variable candidates from 638
-detections; 2 matched a known VSX variable, including the target itself —
-**ASASSN-V J183620.31, recovered at a 1.7" offset from its catalogued
-position** — the positive-control criterion this whole exercise was
-built to test, and it passed. A Lomb-Scargle fit to the recovered
-candidate's own forced-photometry light curve (`light_curve` +
-`recover_rotation_period`) found a real, highly significant periodic
-signal (false-alarm probability ≈ 0) at 0.5 d, not the catalogued
-0.322 d — the declared-in-advance outcome of fitting a period search to
-a light curve covering less than a third of that period (almost
-certainly an alias, not evidence against the true period), and not a
-failure of `find_variable_sources` itself, which is what the crossmatch
-recovery above actually validates.
-
-## Validating against real IASC (Pan-STARRS1) campaign data
-
-Every real-data check so far used ZTF. `docs/src/index.md`'s "Using real
-IASC campaign data" section had stood as "not attempted" all session —
-closed by running `examples/iasc_demo.jl` against 5 real Pan-STARRS1
-(PS1) IASC practice sets (2019-08-28/09-04/09-24, 4 exposures each).
-`run_pipeline` recovered 26 real, independently-catalogued objects
-across the 5 fields via SkyBoT — including a Jupiter Trojan, 2019 NB9 —
-but getting a clean run took three real, fixed bugs, found in this order.
-
-**`load_wcs` failed on every one of these real headers.** Every PS1
-header raised `"Linear transformation matrix is singular"` from wcslib.
-Bisected a real header down to the exact cause (splitting it into halves,
-testing each half in isolation, recursing into whichever half still
-failed): `CNPIX1`/`CNPIX2` alone — a legacy IRAF/DSS plate-astrometry
-keyword pair, present in these headers but with none of that convention's
-other required keywords — was enough to reproduce it, even combined with
-nothing but `SIMPLE`/`BITPIX`/`NAXIS`. wcslib reads that keyword pair as
-the start of a *separate*, implicit DSS-style WCS description, and with
-the rest of that convention absent builds an all-zero, degenerate linear
-transform for it — a real wcslib parsing quirk, not anything wrong with
-the header's own real, complete CTYPE/CRVAL/CRPIX/CDELT WCS, which parses
-cleanly on its own. Fixed in `load_wcs`: on exactly this error, retry
-after stripping just those two keyword's FITS cards and nothing else —
-confirmed sufficient, not guessed. Regression test constructs a
-synthetic header (a real WCS plus injected `CNPIX1`/`CNPIX2` cards) since
-the real PS1 files can't be committed to the repo.
-
-**`detect_sources` produced enormous numbers of spurious detections on
-some frames.** One real frame: 176,165 "detections" at `threshold=8.0`
-(field 451 on ZTF, by comparison, has ~130). Root cause: these FITS files
-mark invalid/masked pixels using the standard `BLANK` header keyword
-(scaled through `BZERO`/`BSCALE` like any other pixel value) rather than
-`NaN`, and `FITSIO.jl` does not convert `BLANK` sentinels automatically.
-One real frame had 158,443 pixels (2.7% of the image) pegged at exactly
-that sentinel value (65535, from `BLANK=32767` + `BZERO=32768`) —
-`detect_sources` read that as enormous real flux across a large masked
-region and found a spurious "source" seemingly everywhere. Confirmed
-directly: replacing those exact pixels with the frame's own valid-region
-median (before any detection) dropped the same frame from 176,165 to 176
-detections. Handled as a preprocessing step in `examples/iasc_demo.jl`
-(`clean_blank_pixels`, writing cleaned copies preserving the original
-header/WCS/timestamp exactly) rather than in `src/`, since BLANK-sentinel
-handling is a real-FITS-ingestion concern specific to how a given survey
-exports data, not something `detect_sources` itself should need to know
-about.
-
-**`crossmatch_catalog(...; :skybot)` was too slow, and not resilient, at
-real scale.** Unlike `:vsx`/`:simbad` (batched via CDS TAP — see above),
-SkyBoT has no batch mode, so `_crossmatch_skybot` queried one candidate
-per request, fully sequentially. Fine for a handful of candidates; not
-for hundreds to thousands of real tracklets. A full 5-field run of
-`examples/iasc_demo.jl` took over two hours and then died outright, deep
-into the fourth field's crossmatch (2,619 candidates), to
-`"tls write failed: connection is closed"` — an uncaught, unretried
-network error with no partial-progress recovery. Fixed two ways in
-`_crossmatch_skybot`: concurrent requests (Julia `Task`s + a bounded
-`Base.Semaphore`, not extra threads — this is a network-latency-bound
-workload, and cooperative concurrency on however many threads Julia
-already has is enough) and a single retry on any `HTTP.HTTPError`.
-Benchmarked directly against the live SkyBoT service (20 real, identical
-requests, repeated to isolate throughput from any one query's own
-content): concurrency=8 gave a real, measured 2.6x speedup (12.5s vs
-32.8s sequential); concurrency up to 60 ran clean with zero errors,
-though gains flattened past ~20-40 (IMCCE's own server-side queueing, not
-this code, by then). Settled on 20 — inside the tested-clean range, not
-pushed to its edge, matching `_CDS_BATCH_SIZE`'s conservative philosophy.
-The full rerun with both fixes completed end to end, no crash, in around
-20 minutes total (all 5 fields) — down from a run that hadn't even
-finished after two hours.
-
-**`match_radius` was too loose, by a measured, corrected amount.**
-`examples/iasc_demo.jl`'s `match_radius` was first converted from
-`real_data_demo.jl`'s ZTF value to preserve the same ~10" angular
-tolerance — a reasonable-looking choice that turned out to be looser than
-PS1's own real astrometric precision. On the densest of the 5 fields,
-this produced 10,422 tracklets from only ~500 detections/frame — almost
-certainly distinct real stars within 10" of each other across frames
-getting cross-linked into spurious tracklets, not 10,422 real moving
-objects. The known SkyBoT objects were still correctly recovered in
-every field regardless, but rather than guess at a tighter value, PS1's
-own headers report the real number needed: `PERROR`, the astrometric
-solution's per-star positional RMS residual, measured at 0.20-0.23"
-across the fields checked here — not something assumed, read directly
-from real data. Retuned `match_radius` to 2" (~10x `PERROR`, a
-comfortable margin for real motion and centroiding noise, not the bare
-residual) and reran all 5 fields: the same 9 distinct known objects were
-recovered in every field (confirmed by name, not just by count — nothing
-dropped out), while total tracklets across all 5 fields fell from 16,158
-to 4,960 (-69%). The reduction is concentrated exactly where predicted:
-the densest field (XY42_p11) went from 10,422 to 3,478; the two
-previously "26 real objects" and "13/2619" style counts were actually
-counting duplicate tracklet-rows around the same handful of real
-objects, not 26 distinct discoveries — a reporting correction as much as
-a code fix, worth noting since the inflated number was reported once,
-here, before the retune caught it.
-
## Tuning `find_variable_sources`'s systematic error floor past the first value that worked
The 1% systematic error floor (see above) was the first value tried,
chosen because it's a standard number in forced-photometry pipelines, not
because it was shown to be optimal. With a real positive control now in
hand (ASASSN-V J183620.31, recovered by `find_variable_sources` earlier
-this session), the floor could finally be checked from *both* sides of
-the tradeoff at once, not just the false-positive side: sweeping
-`systematic_error_fraction` against the same 152 real, matched,
-high-S/N stationary stars from ZTF field 451 (a slightly different count
-than the "119" quoted earlier — this sweep additionally required
-`min_frames` at the full 5-frame count and `normalize=true` together,
-narrowing the matched set), the `chi2_threshold=10.0` false-positive rate
-dropped from 3.3% at a 1% floor to 2.0% at 2%, 1.3% at 3%, and 0% at 5%.
-Naively, that argues for as high a floor as possible — but a floor this
-large also suppresses *real* variability, and that side had never been
-checked. Running the same sweep against ASASSN-V J183620.31's own real
-forced-photometry light curve: reduced chi2 falls from 123 (at 1%) to 31
-(at 2%) to 13.8 (at 3%) to 5.0 (at 5%) — the last of which drops *below*
+this session — see its own
+[validation page](variable-star-validation.md)), the floor could finally
+be checked from *both* sides of the tradeoff at once, not just the
+false-positive side: sweeping `systematic_error_fraction` against the
+same 152 real, matched, high-S/N stationary stars from ZTF field 451 (a
+slightly different count than the "119" quoted earlier — this sweep
+additionally required `min_frames` at the full 5-frame count and
+`normalize=true` together, narrowing the matched set), the
+`chi2_threshold=10.0` false-positive rate dropped from 3.3% at a 1%
+floor to 2.0% at 2%, 1.3% at 3%, and 0% at 5%. Naively, that argues for
+as high a floor as possible — but a floor this large also suppresses
+*real* variability, and that side had never been checked. Running the
+same sweep against ASASSN-V J183620.31's own real forced-photometry
+light curve: reduced chi2 falls from 123 (at 1%) to 31 (at 2%) to 13.8
+(at 3%) to 5.0 (at 5%) — the last of which drops *below*
`chi2_threshold=10.0`, meaning a 5% floor would have made this exact,
real, independently-confirmed variable star invisible to
`find_variable_sources`. Settled on 2%: comfortably above 1%'s
over threshold — the largest floor checked that doesn't cost real
detections, not the smallest false-positive rate achievable.
-## Revisiting the two intentional-design "limitations"
+
-Two items in `docs/src/index.md`'s Known Limitations were design
-decisions, not bugs — but "intentional" isn't the same as "as good as it
-can be." Asked directly whether either could be genuinely improved
-without abandoning the design choice behind it.
+## See also
-**`estimate_psf`'s empirical-vs-fallback split was a hard binary — a
-field either had stars passing the isolation/saturation filter at exactly
-the given `min_separation`, or it fell all the way back to an analytic
-Moffat fit.** But a moderately (not severely) crowded field might have
-real, usable stars at a *slightly* tighter isolation radius — the current
-code was throwing that away and jumping straight to an approximation
-instead of trying harder for the real thing. Added `relaxation_attempts`
-(default 2): on an empty stamp list, halve `min_separation` and retry,
-up to that many times, before falling back. A real empirical PSF from
-fewer, closer stars still beats a parametric approximation, which is the
-whole reason `estimate_psf` exists over just always using
-`fit_moffat_psf`. Regression-tested with two synthetic Gaussian sources
-25 px apart (fails the default `min_separation=40`, but 25 ≥ 20, the
-first relaxed attempt) — recovers the real empirical PSF (correct FWHM)
-instead of falling back, while `relaxation_attempts=0` on the same data
-still fails cleanly, proving the relaxation is what does it. Writing that
-test surfaced a second thing worth knowing: `fit_moffat_psf`'s own stamp
-extraction (`stamp_size=25`, so a ±12 px half-window) doesn't check for
-neighbor contamination the way `estimate_psf`'s isolation filter does —
-stars closer than ~24 px apart corrupt each other's Moffat fit (tested
-directly: clustering the existing fallback test's synthetic stars into a
-6 px box made every fit fail to converge, where the original ~25 px
-spacing fits cleanly) — not fixed here, since `fit_moffat_psf`'s own
-docstring already documents that it deliberately skips the isolation
-check as a defensible trade for a *fit* (a bad stamp shows up as a poor
-residual, in principle), but the synthetic evidence says that trade has
-a real limit worth knowing about.
-
-**`zogy_subtract`'s `V_ast` opt-in was silent — a direct caller who
-simply didn't pass `n_sources`/`r_sources` got `V_ast = 0` with no
-signal anything was skipped**, unlike `run_pipeline`, which always
-supplies them. The design itself (opt-in at the low-level function,
-mandatory at the high-level one) is still the right call — computing
-`n_sources`/`r_sources` costs two extra `detect_sources` passes, real
-cost a direct caller might legitimately not want. What was missing was
-visibility: added a `@warn` (once per session, via `maxlog=1`, so it
-doesn't spam a caller who's already made an informed choice) when both
-are left `nothing`. Regression test confirms it fires when omitted and
-stays silent when both are supplied.
+- [Validating `search_field` against a real, independently-confirmed variable star](variable-star-validation.md)
+- [Validating against real IASC (Pan-STARRS1) campaign data](iasc-campaign-validation.md)
+- [Revisiting the two intentional-design "limitations"](design-refinements.md)
blob - /dev/null
blob + e01a25c74205d65cfec9fce040385e20a8626e17 (mode 644)
--- /dev/null
+++ docs/src/design-refinements.md
+# Revisiting the two intentional-design "limitations"
+
+Part of this project's [Investigation Log](investigation-log.md) —
+split onto its own page since it's a design review, not a bug found on a
+specific real dataset the way the other pages here are.
+
+Two items in `docs/src/index.md`'s Known Limitations were design
+decisions, not bugs — but "intentional" isn't the same as "as good as it
+can be." Asked directly whether either could be genuinely improved
+without abandoning the design choice behind it.
+
+## `estimate_psf`'s empirical-vs-fallback split was a hard binary
+
+A field either had stars passing the isolation/saturation filter at
+exactly the given `min_separation`, or it fell all the way back to an
+analytic Moffat fit. But a moderately (not severely) crowded field might
+have real, usable stars at a *slightly* tighter isolation radius — the
+current code was throwing that away and jumping straight to an
+approximation instead of trying harder for the real thing. Added
+`relaxation_attempts` (default 2): on an empty stamp list, halve
+`min_separation` and retry, up to that many times, before falling back.
+A real empirical PSF from fewer, closer stars still beats a parametric
+approximation, which is the whole reason `estimate_psf` exists over just
+always using `fit_moffat_psf`. Regression-tested with two synthetic
+Gaussian sources 25 px apart (fails the default `min_separation=40`, but
+25 ≥ 20, the first relaxed attempt) — recovers the real empirical PSF
+(correct FWHM) instead of falling back, while `relaxation_attempts=0` on
+the same data still fails cleanly, proving the relaxation is what does
+it. Writing that test surfaced a second thing worth knowing:
+`fit_moffat_psf`'s own stamp extraction (`stamp_size=25`, so a ±12 px
+half-window) doesn't check for neighbor contamination the way
+`estimate_psf`'s isolation filter does — stars closer than ~24 px apart
+corrupt each other's Moffat fit (tested directly: clustering the
+existing fallback test's synthetic stars into a 6 px box made every fit
+fail to converge, where the original ~25 px spacing fits cleanly) — not
+fixed here, since `fit_moffat_psf`'s own docstring already documents
+that it deliberately skips the isolation check as a defensible trade for
+a *fit* (a bad stamp shows up as a poor residual, in principle), but the
+synthetic evidence says that trade has a real limit worth knowing about.
+
+## `zogy_subtract`'s `V_ast` opt-in was silent
+
+A direct caller who simply didn't pass `n_sources`/`r_sources` got
+`V_ast = 0` with no signal anything was skipped, unlike `run_pipeline`,
+which always supplies them. The design itself (opt-in at the low-level
+function, mandatory at the high-level one) is still the right call —
+computing `n_sources`/`r_sources` costs two extra `detect_sources`
+passes, real cost a direct caller might legitimately not want. What was
+missing was visibility: added a `@warn` (once per session, via
+`maxlog=1`, so it doesn't spam a caller who's already made an informed
+choice) when both are left `nothing`. Regression test confirms it fires
+when omitted and stays silent when both are supplied.
blob - /dev/null
blob + e15a844b967eb0feb88c20288a6a487b211849ab (mode 644)
--- /dev/null
+++ docs/src/iasc-campaign-validation.md
+# Validating against real IASC (Pan-STARRS1) campaign data
+
+Part of this project's [Investigation Log](investigation-log.md) —
+split onto its own page since it's a self-contained validation story
+against a different survey (Pan-STARRS1) than the ZTF narrative there.
+
+Every real-data check so far used ZTF. `docs/src/index.md`'s "Using real
+IASC campaign data" section had stood as "not attempted" all session —
+closed by running `examples/iasc_demo.jl` against 5 real Pan-STARRS1
+(PS1) IASC practice sets (2019-08-28/09-04/09-24, 4 exposures each).
+`run_pipeline` recovered 9 real, independently-catalogued objects
+across the 5 fields via SkyBoT — including a Jupiter Trojan, 2019 NB9 —
+but getting a clean run took four real, fixed issues, found in this
+order.
+
+## `load_wcs` failed on every one of these real headers
+
+Every PS1 header raised `"Linear transformation matrix is singular"`
+from wcslib. Bisected a real header down to the exact cause (splitting
+it into halves, testing each half in isolation, recursing into whichever
+half still failed): `CNPIX1`/`CNPIX2` alone — a legacy IRAF/DSS
+plate-astrometry keyword pair, present in these headers but with none of
+that convention's other required keywords — was enough to reproduce it,
+even combined with nothing but `SIMPLE`/`BITPIX`/`NAXIS`. wcslib reads
+that keyword pair as the start of a *separate*, implicit DSS-style WCS
+description, and with the rest of that convention absent builds an
+all-zero, degenerate linear transform for it — a real wcslib parsing
+quirk, not anything wrong with the header's own real, complete
+CTYPE/CRVAL/CRPIX/CDELT WCS, which parses cleanly on its own. Fixed in
+`load_wcs`: on exactly this error, retry after stripping just those two
+keyword's FITS cards and nothing else — confirmed sufficient, not
+guessed. Regression test constructs a synthetic header (a real WCS plus
+injected `CNPIX1`/`CNPIX2` cards) since the real PS1 files can't be
+committed to the repo.
+
+## `detect_sources` produced enormous numbers of spurious detections on some frames
+
+One real frame: 176,165 "detections" at `threshold=8.0` (field 451 on
+ZTF, by comparison, has ~130). Root cause: these FITS files mark
+invalid/masked pixels using the standard `BLANK` header keyword (scaled
+through `BZERO`/`BSCALE` like any other pixel value) rather than `NaN`,
+and `FITSIO.jl` does not convert `BLANK` sentinels automatically. One
+real frame had 158,443 pixels (2.7% of the image) pegged at exactly that
+sentinel value (65535, from `BLANK=32767` + `BZERO=32768`) —
+`detect_sources` read that as enormous real flux across a large masked
+region and found a spurious "source" seemingly everywhere. Confirmed
+directly: replacing those exact pixels with the frame's own valid-region
+median (before any detection) dropped the same frame from 176,165 to 176
+detections. Handled as a preprocessing step in `examples/iasc_demo.jl`
+(`clean_blank_pixels`, writing cleaned copies preserving the original
+header/WCS/timestamp exactly) rather than in `src/`, since BLANK-sentinel
+handling is a real-FITS-ingestion concern specific to how a given survey
+exports data, not something `detect_sources` itself should need to know
+about.
+
+## `crossmatch_catalog(...; :skybot)` was too slow, and not resilient, at real scale
+
+Unlike `:vsx`/`:simbad` (batched via CDS TAP — see the
+[Investigation Log](investigation-log.md)), SkyBoT has no batch mode, so
+`_crossmatch_skybot` queried one candidate per request, fully
+sequentially. Fine for a handful of candidates; not for hundreds to
+thousands of real tracklets. A full 5-field run of `examples/iasc_demo.jl`
+took over two hours and then died outright, deep into the fourth field's
+crossmatch (2,619 candidates), to `"tls write failed: connection is
+closed"` — an uncaught, unretried network error with no partial-progress
+recovery. Fixed two ways in `_crossmatch_skybot`: concurrent requests
+(Julia `Task`s + a bounded `Base.Semaphore`, not extra threads — this is
+a network-latency-bound workload, and cooperative concurrency on however
+many threads Julia already has is enough) and a single retry on any
+`HTTP.HTTPError`. Benchmarked directly against the live SkyBoT service
+(20 real, identical requests, repeated to isolate throughput from any one
+query's own content): concurrency=8 gave a real, measured 2.6x speedup
+(12.5s vs 32.8s sequential); concurrency up to 60 ran clean with zero
+errors, though gains flattened past ~20-40 (IMCCE's own server-side
+queueing, not this code, by then). Settled on 20 — inside the
+tested-clean range, not pushed to its edge, matching `_CDS_BATCH_SIZE`'s
+conservative philosophy. The full rerun with both fixes completed end to
+end, no crash, in around 20 minutes total (all 5 fields) — down from a
+run that hadn't even finished after two hours.
+
+## `match_radius` was too loose, by a measured, corrected amount
+
+`examples/iasc_demo.jl`'s `match_radius` was first converted from
+`real_data_demo.jl`'s ZTF value to preserve the same ~10" angular
+tolerance — a reasonable-looking choice that turned out to be looser than
+PS1's own real astrometric precision. On the densest of the 5 fields,
+this produced 10,422 tracklets from only ~500 detections/frame — almost
+certainly distinct real stars within 10" of each other across frames
+getting cross-linked into spurious tracklets, not 10,422 real moving
+objects. The known SkyBoT objects were still correctly recovered in
+every field regardless, but rather than guess at a tighter value, PS1's
+own headers report the real number needed: `PERROR`, the astrometric
+solution's per-star positional RMS residual, measured at 0.20-0.23"
+across the fields checked here — not something assumed, read directly
+from real data. Retuned `match_radius` to 2" (~10x `PERROR`, a
+comfortable margin for real motion and centroiding noise, not the bare
+residual) and reran all 5 fields: the same 9 distinct known objects were
+recovered in every field (confirmed by name, not just by count — nothing
+dropped out), while total tracklets across all 5 fields fell from 16,158
+to 4,960 (-69%). The reduction is concentrated exactly where predicted:
+the densest field (XY42_p11) went from 10,422 to 3,478; the two
+previously "26 real objects" and "13/2619" style counts were actually
+counting duplicate tracklet-rows around the same handful of real
+objects, not 26 distinct discoveries — a reporting correction as much as
+a code fix, worth noting since the inflated number was reported once,
+here, before the retune caught it.
+
+
+
+## Recovered objects
+
+| Field | Tracklets (retuned) | Known objects recovered |
+|:--|--:|:--|
+| XY14_p10 | 52 | — |
+| XY15_p01 | 132 | 2014 HO19 |
+| XY25_p10 | 567 | 4311 T-1, 2009 SG135, 2015 XJ232, 2019 PK5 |
+| XY26_p01 | 731 | 2001 SH320, 2011 SH185, 2019 NB9 (Jupiter Trojan) |
+| XY42_p11 | 3,478 | 2008 FA111 |
blob - /dev/null
blob + fbc0d9a85c0549c39238219adbe6adceb746ffb6 (mode 644)
--- /dev/null
+++ docs/src/variable-star-validation.md
+# Validating `search_field` against a real, independently-confirmed variable star
+
+Part of this project's [Investigation Log](investigation-log.md) —
+split onto its own page since it's a self-contained validation story,
+distinct from the ZTF field 451 bug-hunting narrative there.
+
+Every real-data check so far had confirmed known *moving* objects (via
+SkyBoT) but never a known *variable* — the one real dataset checked had
+only one catalogued VSX variable in its footprint, too faint to serve as
+a useful positive control. Fixed by choosing a real target with an actual
+VSX catalog entry: ASASSN-V J183620.31 (type EW, a contact eclipsing
+binary, period 0.322427 d), in ZTF field 487/CCD 12/quadrant 1/zr, night
+2019-06-10 — a real high-cadence campaign with 144 exposures over 2.45 h,
+thinned to 29 for `examples/variable_star_demo.jl`. Declared in advance:
+2.45 h covers only ~32% of one period, so full period recovery from this
+single night was not expected — the actual test was whether
+`find_variable_sources` flags the star as variable at all.
+
+First attempt did not run: `search_field`'s default `threshold=5.0` (used
+at `8.0` in `real_data_demo.jl`) produced ~12,900 detections per frame
+here, against field 451's ~130 — this field sits near the galactic plane.
+`link_candidates`'s tracklet search is pairwise in detections/frame, and
+did not finish in 35 minutes at that density; killed and confirmed via a
+standalone check (`detect_sources` alone, one frame) that the counts were
+real, not a hang. The target star is extremely bright at this field's
+noise level (flux ≈ 87,000, ~1.7 px from its WCS-predicted position at
+every threshold tested from 10 to 100), so raising the demo's threshold
+to 60 — cutting detections/frame to ~1,260 — loses none of the signal
+this run actually needs; a deliberate, field-specific tradeoff, not the
+pipeline's own default. A second bug surfaced once linking finished:
+`light_curve`'s default `timestamp_key="MJD-OBS"` doesn't exist in this
+survey's headers (`search_field` was already correctly called with
+`"OBSMJD"`) — a copy-paste omission in the demo script, not a pipeline
+bug, fixed by passing the same key.
+
+With both fixed, the run found 22 variable candidates from 638
+detections; 2 matched a known VSX variable, including the target itself —
+**ASASSN-V J183620.31, recovered at a 1.7" offset from its catalogued
+position** — the positive-control criterion this whole exercise was
+built to test, and it passed. A Lomb-Scargle fit to the recovered
+candidate's own forced-photometry light curve (`light_curve` +
+`recover_rotation_period`) found a real, highly significant periodic
+signal (false-alarm probability ≈ 0) at 0.5 d, not the catalogued
+0.322 d — the declared-in-advance outcome of fitting a period search to
+a light curve covering less than a third of that period (almost
+certainly an alias, not evidence against the true period), and not a
+failure of `find_variable_sources` itself, which is what the crossmatch
+recovery above actually validates.
blob - /dev/null
blob + 4db2ccebb03a1803dd5e43d0dd1103ec0b9b1b15 (mode 644)
--- /dev/null
+++ docs/make_figures.jl
+#=
+Generates the figures embedded in docs/src/investigation-log.md, from the
+real numbers measured during this project's real-data investigations (not
+synthetic/illustrative data) — see the corresponding prose sections for
+where each number comes from. Run automatically by docs/make.jl before
+`makedocs`, so CI regenerates these on every docs build rather than
+committing static images that could drift from the numbers in the text.
+
+Styled for the wiki's dark-only theme (docs/src/.vitepress/config.mts,
+appearance: 'force-dark') — transparent background, white text — rather
+than a generic light-background default.
+=#
+using CairoMakie
+
+const ASSETS_DIR = joinpath(@__DIR__, "src", "assets")
+mkpath(ASSETS_DIR)
+
+set_theme!(Theme(
+ backgroundcolor=:transparent,
+ textcolor=:white,
+ Axis=(
+ backgroundcolor=:transparent,
+ xlabelcolor=:white, ylabelcolor=:white, titlecolor=:white,
+ xticklabelcolor=:white, yticklabelcolor=:white,
+ xtickcolor=:white, ytickcolor=:white,
+ leftspinecolor=:white, rightspinecolor=:white,
+ bottomspinecolor=:white, topspinecolor=:white,
+ xgridcolor=(:white, 0.15), ygridcolor=(:white, 0.15),
+ ),
+ Legend=(backgroundcolor=:transparent, labelcolor=:white, framecolor=(:white, 0.3)),
+))
+
+# --- find_variable_sources: systematic_error_fraction sweep -----------
+# Real measurements from sweeping systematic_error_fraction against (a)
+# 152 real, matched, high-S/N stationary stars on real ZTF field 451
+# (false-positive rate at chi2_threshold=10.0) and (b) the real,
+# independently-confirmed variable ASASSN-V J183620.31's own real
+# forced-photometry light curve (reduced chi2). See "Tuning
+# find_variable_sources's systematic error floor..." in this log.
+let
+ floors = [0.0, 0.5, 1.0, 1.5, 2.0, 3.0, 5.0] # percent
+ false_positive_rate = [13.2, 6.6, 3.3, 2.6, 2.0, 1.3, 0.0] # percent, at chi2_threshold=10
+ variable_reduced_chi2 = [20688.5, 484.4, 123.4, 55.1, 31.0, 13.8, 5.0]
+
+ fig = Figure(size=(720, 460))
+ ax1 = Axis(fig[1, 1];
+ xlabel="systematic_error_fraction (%)",
+ ylabel="false-positive rate at chi2_threshold=10 (%)",
+ title="Real ZTF field 451 stationary stars (152) vs. a real confirmed variable")
+ ax2 = Axis(fig[1, 1];
+ ylabel="reduced chi² (real variable, ASASSN-V J183620.31)",
+ yscale=log10, yaxisposition=:right, ygridvisible=false)
+ hidespines!(ax2)
+ hidexdecorations!(ax2)
+
+ l1 = lines!(ax1, floors, false_positive_rate; color=:tomato, linewidth=2)
+ scatter!(ax1, floors, false_positive_rate; color=:tomato, markersize=10)
+ l2 = lines!(ax2, floors, variable_reduced_chi2; color=:dodgerblue, linewidth=2)
+ scatter!(ax2, floors, variable_reduced_chi2; color=:dodgerblue, markersize=10)
+ hlines!(ax2, [10.0]; color=:dodgerblue, linestyle=:dash, linewidth=1)
+ vlines!(ax1, [2.0]; color=:white, linestyle=:dot, linewidth=1)
+ text!(ax1, 2.05, 12.0; text="chosen default (2%)", fontsize=12, color=:white)
+
+ Legend(fig[2, 1], [l1, l2],
+ ["false-positive rate (left axis)",
+ "real variable's reduced chi² (right axis, log scale; dashed = chi2_threshold=10)"];
+ orientation=:horizontal, tellwidth=false)
+
+ save(joinpath(ASSETS_DIR, "systematic-error-floor-sweep.png"), fig)
+end
+
+# --- IASC match_radius retuning: tracklet counts before/after ---------
+# Real per-field tracklet counts from examples/iasc_demo.jl, before
+# (match_radius converted from ZTF's pixel scale, ~10" tolerance) and
+# after (match_radius from PS1's own PERROR, ~2" tolerance) retuning. The
+# same 9 distinct known objects were recovered in both runs — this is a
+# reduction in spurious duplicate tracklets, not lost detections. See
+# "match_radius was too loose, by a measured, corrected amount" above.
+let
+ fields = ["XY14_p10", "XY15_p01", "XY25_p10", "XY26_p01", "XY42_p11"]
+ before = [524, 776, 1817, 2619, 10422]
+ after = [52, 132, 567, 731, 3478]
+
+ fig = Figure(size=(720, 440))
+ ax = Axis(fig[1, 1];
+ xlabel="IASC practice field", ylabel="tracklets found",
+ title="Retuning match_radius to PS1's own real astrometric precision (PERROR)",
+ xticks=(1:length(fields), fields), yscale=log10)
+
+ n = length(fields)
+ x = 1:n
+ barplot!(ax, x .- 0.15, before; width=0.3, color=:salmon, label="match_radius ≈ 10\" (ZTF's tolerance, reused as-is)")
+ barplot!(ax, x .+ 0.15, after; width=0.3, color=:mediumspringgreen, label="match_radius = 2\" (10x PS1's real PERROR)")
+ # A log axis has no true zero, so bars would otherwise auto-start from
+ # whatever the smallest value happens to be (misleadingly making every
+ # bar look like a similar-height floating block) — pin the bottom
+ # near 1 instead, the closest a log axis gets to a real baseline.
+ ylims!(ax, 1, 20000)
+
+ axislegend(ax; position=:lt)
+ save(joinpath(ASSETS_DIR, "iasc-match-radius-retuning.png"), fig)
+end
+
+println("Figures written to $ASSETS_DIR")
blob - 2b2b27948f0ca5c2c63b97a597cdafe8461ba09c
blob + 9010fdd9ad1ed9b7b78d9e0e8a51d20e4cc1e441
--- src/variables.jl
+++ src/variables.jl
`systematic_error_fraction=0.02` (the current default) cuts the
threshold-10 false-positive rate to 2.0% (vs 1%'s 3.3%), while a real,
independently-confirmed variable (ASASSN-V J183620.31 — see the
-[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#Validating-search_field-against-a-real,-independently-confirmed-variable-star))
+[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/variable-star-validation))
still clears `chi2_threshold=10.0` with a healthy margin (reduced chi2 ≈
31, over 3x the threshold) at this floor — a floor of 5% would erase that
same real signal (reduced chi2 drops to ≈5, below threshold), so 2% is