Commit Diff


commit - 75a8a7aa40ae3ae0fe1ccdf858f417d79305141d
commit + 25619044a16363a72873b015ccb438393734b447
blob - 4db2ccebb03a1803dd5e43d0dd1103ec0b9b1b15
blob + 144870b1a2fc826d4e828b9357d0cf455890d595
--- docs/make_figures.jl
+++ docs/make_figures.jl
@@ -31,41 +31,50 @@ set_theme!(Theme(
 ))
 
 # --- find_variable_sources: systematic_error_fraction sweep -----------
-# Real measurements from sweeping systematic_error_fraction against (a)
-# 152 real, matched, high-S/N stationary stars on real ZTF field 451
-# (false-positive rate at chi2_threshold=10.0) and (b) the real,
-# independently-confirmed variable ASASSN-V J183620.31's own real
-# forced-photometry light curve (reduced chi2). See "Tuning
+# Real measurements from sweeping systematic_error_fraction against three
+# independent real ZTF fields, each with its own real matched stationary
+# stars (false-positive rate at chi2_threshold=10.0) and its own real,
+# independently-confirmed variable star (reduced chi2): field 451 (152
+# stars; variable ASASSN-V J183620.31, from a different real field), and
+# two more found and validated later, V1012 Mon (197 stars) and ASASSN-V
+# J072906.85-090518.2 (182 stars). See "Cross-validating
 # find_variable_sources's systematic error floor..." in this log.
 let
-    floors = [0.0, 0.5, 1.0, 1.5, 2.0, 3.0, 5.0]  # percent
-    false_positive_rate = [13.2, 6.6, 3.3, 2.6, 2.0, 1.3, 0.0]  # percent, at chi2_threshold=10
-    variable_reduced_chi2 = [20688.5, 484.4, 123.4, 55.1, 31.0, 13.8, 5.0]
+    floors = [0.0, 1.0, 2.0, 3.0, 5.0]  # percent, common to all three fields
+    field451_fp = [13.2, 3.3, 2.0, 1.3, 0.0]
+    field451_chi2 = [20688.5, 123.4, 31.0, 13.8, 5.0]
+    v1012_fp = [62.24, 6.12, 2.55, 2.04, 1.02]
+    v1012_chi2 = [31908.9, 542.3, 137.4, 61.2, 22.1]
+    asassn_fp = [49.72, 2.76, 0.55, 0.55, 0.55]
+    asassn_chi2 = [3298.5, 151.2, 39.2, 17.5, 6.3]
 
-    fig = Figure(size=(720, 460))
+    fig = Figure(size=(760, 520))
     ax1 = Axis(fig[1, 1];
         xlabel="systematic_error_fraction (%)",
         ylabel="false-positive rate at chi2_threshold=10 (%)",
-        title="Real ZTF field 451 stationary stars (152) vs. a real confirmed variable")
-    ax2 = Axis(fig[1, 1];
-        ylabel="reduced chi² (real variable, ASASSN-V J183620.31)",
-        yscale=log10, yaxisposition=:right, ygridvisible=false)
-    hidespines!(ax2)
-    hidexdecorations!(ax2)
+        title="Three real, independent ZTF fields: false-positive rate vs. systematic_error_fraction")
+    ax2 = Axis(fig[2, 1];
+        xlabel="systematic_error_fraction (%)",
+        ylabel="reduced chi² (real confirmed variable)", yscale=log10,
+        title="Each field's own real confirmed variable's reduced chi²")
 
-    l1 = lines!(ax1, floors, false_positive_rate; color=:tomato, linewidth=2)
-    scatter!(ax1, floors, false_positive_rate; color=:tomato, markersize=10)
-    l2 = lines!(ax2, floors, variable_reduced_chi2; color=:dodgerblue, linewidth=2)
-    scatter!(ax2, floors, variable_reduced_chi2; color=:dodgerblue, markersize=10)
-    hlines!(ax2, [10.0]; color=:dodgerblue, linestyle=:dash, linewidth=1)
-    vlines!(ax1, [2.0]; color=:white, linestyle=:dot, linewidth=1)
-    text!(ax1, 2.05, 12.0; text="chosen default (2%)", fontsize=12, color=:white)
+    colors = (:tomato, :mediumspringgreen, :dodgerblue)
+    labels = ("field 451 (152 stars)", "V1012 Mon (197 stars)", "ASASSN-V J072906.85 (182 stars)")
 
-    Legend(fig[2, 1], [l1, l2],
-           ["false-positive rate (left axis)",
-            "real variable's reduced chi² (right axis, log scale; dashed = chi2_threshold=10)"];
-           orientation=:horizontal, tellwidth=false)
+    for (fp, chi2, color, label) in zip(
+        (field451_fp, v1012_fp, asassn_fp), (field451_chi2, v1012_chi2, asassn_chi2), colors, labels)
+        lines!(ax1, floors, fp; color, linewidth=2, label)
+        scatter!(ax1, floors, fp; color, markersize=10)
+        lines!(ax2, floors, chi2; color, linewidth=2, label)
+        scatter!(ax2, floors, chi2; color, markersize=10)
+    end
+    hlines!(ax2, [10.0]; color=:white, linestyle=:dash, linewidth=1)
+    vlines!(ax1, [3.0]; color=:white, linestyle=:dot, linewidth=1)
+    vlines!(ax2, [3.0]; color=:white, linestyle=:dot, linewidth=1)
+    text!(ax1, 3.1, 55.0; text="chosen default (3%)", fontsize=12, color=:white)
 
+    Legend(fig[3, 1], ax1; orientation=:horizontal, tellwidth=false)
+
     save(joinpath(ASSETS_DIR, "systematic-error-floor-sweep.png"), fig)
 end
 
blob - 7184045140ba32486445b6c6f75e576133e97f65
blob + 75626c03ffee8b288ce67d92dffde952d81e7cab
--- docs/src/design-refinements.md
+++ docs/src/design-refinements.md
@@ -163,5 +163,66 @@ The "tens of minutes" sequential cost is, as far as th
 could establish, a real, currently-irreducible property of reprojecting
 one frame through `wcslib` at a time — but it is no longer irreducible
 *in total*: spreading that same per-frame cost across independent
-processes is a real, measured ~2x win, not something left unexplored for
-lack of trying.
+processes is a real, measured 3.21x win, not something left unexplored
+for lack of trying.
+
+## Cross-validating `find_variable_sources`'s systematic error floor against two more real, independent variable stars
+
+The 2% `systematic_error_fraction` default (see the Investigation Log's
+own floor-tuning story) rested on a single real field: false positives
+measured against 152 real ZTF field-451 stars, sensitivity checked
+against one real confirmed variable, ASASSN-V J183620.31, from a
+*different* field entirely. One confirmed variable is a thin evidence
+base for a default that trades away real sensitivity — asked directly
+whether more real variables could be found to check it against, instead
+of trusting one field to generalize.
+
+**Finding two more real, densely-sampled confirmed variables.** Queried
+VizieR's VSX table (the same real TAP service `crossmatch_catalog(...;
+:vsx)` already uses) for bright (mag 12-16), short-period (0.2-0.6 d),
+real eclipsing binaries, then checked each candidate's actual ZTF
+exposure history via IRSA's metadata API for real same-night,
+same-quadrant density — most real candidates only had 5-7 exposures on
+their best night (ordinary ZTF cadence; the earlier 144-exposure night
+was a special high-cadence campaign, not typical). Out of 150 real
+candidates checked this way, two stood out sharing the same field and
+night: **V1012 Mon** (333 exposures) and **ASASSN-V
+J072906.85-090518.2** (332 exposures), both field 360, night
+2019-01-08 — denser real coverage than the original 144-exposure case,
+and both in a real, unusually dense stellar field (Monoceros straddles
+the galactic plane): ~30,000 raw detections/frame at the pipeline's
+default `threshold=5`, needing `threshold=100` to bring that down to a
+workable ~200-580/frame, still denser than field 451's own ~130.
+
+**Both real targets recovered cleanly.** Run through the same
+`search_field`-style matching field 451 used: V1012 Mon recovered 0.63"
+from its VSX position, ASASSN-V J072906.85 at 1.55" — both well inside
+typical match tolerances — with 197 and 182 real, full-coverage matched
+stationary stars respectively (more than field 451's 152, in each
+field individually).
+
+**Repeating the exact same floor sweep three times over, not once.**
+
+| Floor | Field 451 FP% / var χ² | V1012 Mon FP% / var χ² | ASASSN-072906 FP% / var χ² |
+|:--|--:|--:|--:|
+| 1% | 3.3% / 123.4 | 6.12% / 542.3 | 2.76% / 151.2 |
+| 2% (old default) | 2.0% / 31.0 | 2.55% / 137.4 | 0.55% / 39.2 |
+| **3% (new default)** | **1.3% / 13.8** | **2.04% / 61.2** | **0.55% / 17.5** |
+| 4% | (not measured) | 1.53% / 34.5 | 0.55% / **9.9** ← below threshold |
+| 5% | 0.0% / **5.0** ← below threshold | 1.02% / 22.1 | 0.55% / 6.3 ← below threshold |
+
+At exactly 3% — a floor already measured on field 451 in the original
+sweep, so no extrapolation needed there — **all three** real confirmed
+variables stay above `chi2_threshold=10.0` (margins 1.38x, 6.1x, 1.75x),
+while the false-positive rate is the same or better than 2% in every
+single field. At 4%, ASASSN-V J072906.85's own signal already drops
+below threshold; at 5%, two of the three do. A literal 0% false-positive
+rate is real and reachable (field 451 and, separately, ASASSN-072906
+both hit it at 5%) — but as a byproduct of losing real sensitivity, not
+as a genuine improvement, exactly the tradeoff the original single-field
+sweep already warned about, now confirmed rather than assumed on two
+more independent real datasets. `systematic_error_fraction`'s default
+moved from 2% to 3% on this evidence; 0% remains achievable only by
+accepting that cost, which is why it isn't the target.
+
+![False-positive rate and each field's own real confirmed variable's reduced chi², swept across systematic_error_fraction, for three independent real ZTF fields](assets/systematic-error-floor-sweep.png)
blob - 68c83b90c734de9573b3d4901c57f2b8eac1c48a
blob + 9b3aec538c2e9ca78762449a69e516726f0626a7
--- docs/src/index.md
+++ docs/src/index.md
@@ -63,7 +63,7 @@ not the legacy 80-column format — see its docstring 
 |:--|:--|
 | `detect_sources`, `link_candidates`, WCS calibration, `crossmatch_catalog` | Synthetic (exact recovery) · real ZTF field 451 · 5 real IASC/Pan-STARRS1 fields (**9 known objects** recovered) |
 | ZOGY difference imaging (`build_reference`, `estimate_psf`, `zogy_subtract`) | 3 falsifiable synthetic checks · real ZTF field 451 |
-| `find_variable_sources` / `search_field` | Synthetic · false-positive rate calibrated on real ZTF stars · a real confirmed variable (ASASSN-V J183620.31) |
+| `find_variable_sources` / `search_field` | Synthetic · false-positive rate calibrated on real ZTF stars across 3 independent fields · 3 real confirmed variables (ASASSN-V J183620.31, V1012 Mon, ASASSN-V J072906.85-090518.2) |
 | `fit_moffat_psf` (PSF analytic fallback) | Synthetic · PSF width calibrated on real ZTF data |
 | `plate_solve` | Live nova.astrometry.net service |
 | `crossmatch_catalog` (`:skybot`/`:vsx`/`:simbad`) | Live services, real positive controls |
@@ -236,18 +236,24 @@ pipeline at a real campaign, or any other survey's dat
     errors made ordinary flat-fielding/PSF-variation systematics look
     like huge chi2 significance. Adding that floor
     (`variability_chi2`'s `systematic_error_fraction`) cut the rate: a
-    1% floor took it to 6% at threshold 3, ~0% at threshold 20; sweeping
-    the floor further (against the same real stars, and checked against
-    a real confirmed variable — ASASSN-V J183620.31 — to make sure real
-    sensitivity wasn't sacrificed for it) found more room without giving
-    up real detections: the current default, 2%, cuts the
-    `chi2_threshold=10.0` rate to 2.0% on this dataset (vs 1%'s 3.3%),
-    while that confirmed variable still clears the threshold with a 3x
-    margin — see `find_variable_sources`'s docstring and the
-    [Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/investigation-log#The-centroid-fix-barely-moved-the-false-positive-floor-—-the-real-cause-was-a-systematic-error-floor)
-    for the full before/after numbers. Still not zero: treat a candidate
-    as needing independent confirmation (a catalog match or a recovered
-    period), not as self-evidently real.
+    1% floor took it to 6% at threshold 3, ~0% at threshold 20.
+    Sweeping the floor further, first against field 451 alone and its
+    one real confirmed variable (ASASSN-V J183620.31, from a different
+    field), settled on 2% — later repeated against two more real,
+    independent fields (V1012 Mon and ASASSN-V J072906.85-090518.2,
+    each with its own real confirmed variable and 180+ real matched
+    stars), which moved the default to **3%**: the largest floor at
+    which all three confirmed variables stay above `chi2_threshold=10.0`
+    while the false-positive rate is the same or better than at 2% in
+    every field (1.3%/2.04%/0.55% at 3% vs. 2.0%/2.55%/0.55% at 2%). A
+    literal 0% rate is reachable at a 5% floor, but only by also erasing
+    two of the three real variables' own signal below threshold — a
+    different failure mode, not a real improvement. See
+    `find_variable_sources`'s docstring and
+    [Design refinements](https://richard7987.github.io/AsteroidPipeline.jl/dev/design-refinements)
+    for the full three-field before/after numbers. Still not zero: treat
+    a candidate as needing independent confirmation (a catalog match or
+    a recovered period), not as self-evidently real.
 
 ## Follow-up workflows
 
blob - e99478d48914a35114bb56bb42f86591a35d5a0f
blob + bb8ff19673eeaa65626607d86ac0844cb141738f
--- docs/src/investigation-log.md
+++ docs/src/investigation-log.md
@@ -338,7 +338,11 @@ false-positive rate, while leaving the real variable's
 over threshold — the largest floor checked that doesn't cost real
 detections, not the smallest false-positive rate achievable.
 
-![False-positive rate on 152 real stationary stars, and reduced chi2 on a real confirmed variable, swept across systematic_error_fraction values](assets/systematic-error-floor-sweep.png)
+This 2% default rested on field 451 alone — later cross-validated
+against two more real, independent confirmed variables in two more real
+fields, which moved the default to 3%; see
+[Design refinements](design-refinements.md) for the full three-field
+sweep and figure.
 
 ## See also
 
blob - 170036bc8f8cbae618a43f1a0e3cf15070638619
blob + fabf87a8f9d9999654220eeeff28551d25f27188
--- src/pipeline.jl
+++ src/pipeline.jl
@@ -280,7 +280,7 @@ end
                  variability_position_tolerance::Real=2.0,
                  variability_min_frames::Union{Nothing,Integer}=nothing,
                  variability_chi2_threshold::Real=10.0,
-                 variability_systematic_error_fraction::Real=0.02)
+                 variability_systematic_error_fraction::Real=0.03)
         -> (movers=<table>, variables=<table>)
 
 Run [`run_pipeline`](@ref)'s asteroid-candidate search and
@@ -327,7 +327,7 @@ function search_field(fits_paths::AbstractVector{<:Abs
                        variability_chi2_threshold::Real=10.0,
                        variability_normalize::Bool=(reference === nothing),
                        variability_max_relative_error::Real=0.10,
-                       variability_systematic_error_fraction::Real=0.02)
+                       variability_systematic_error_fraction::Real=0.03)
     detections_per_frame, wcs_per_frame, timestamps, n_gated = _detect_all_frames(
         fits_paths; timestamp_key, threshold, box_size, aperture_radius,
         reference, psf_threshold, psf_min_separation, quality_max_std, plate_solve_api_key,
blob - 9010fdd9ad1ed9b7b78d9e0e8a51d20e4cc1e441
blob + f24937798e01e80ff96b499bb7300ed6fa3581fe
--- src/variables.jl
+++ src/variables.jl
@@ -1,5 +1,5 @@
 """
-    variability_chi2(flux, flux_err; systematic_error_fraction::Real=0.02) -> (chi2, dof)
+    variability_chi2(flux, flux_err; systematic_error_fraction::Real=0.03) -> (chi2, dof)
 
 Chi-squared goodness of fit of `flux` against the constant-flux
 (non-variable) null hypothesis, using the error-weighted mean as the
@@ -23,13 +23,19 @@ forced-photometry pipelines, tried first) cut the fals
 a reduced-chi2 threshold of 3 from 22% (no floor) to 6%, and at
 threshold 20 from 13% to essentially 0%. Sweeping the floor further
 against the same real, matched stationary stars — not guessed — found
-more room: 2% (the current default) cuts the threshold-10 rate to 2.0%
-(vs 1%'s 3.3%), while a real, independently-confirmed variable
-(ASASSN-V J183620.31, checked at the same floor sweep) still clears
-threshold 10 with a healthy margin (reduced chi2 ≈ 31 at 2%, only
-dropping below 10 once the floor reaches 5%) — see
-[`find_variable_sources`](@ref)'s docstring for the full before/after
-table.
+more room, and was later cross-checked against two more real,
+independently-confirmed variable stars in two more real fields, not just
+one: 3% (the current default) keeps every one of three real confirmed
+variables above `chi2_threshold=10.0` (reduced chi2 13.8/61.2/17.5 across
+the three fields) while cutting the false-positive rate on real
+stationary stars in those same three fields to 1.3%/2.04%/0.55% (vs.
+2.0%/2.55%/0.55% at the previous 2% default — better or equal in every
+field, never worse) — see [`find_variable_sources`](@ref)'s docstring
+for the full before/after table across all three. A literal 0% false
+positive rate is achievable (a 5% floor reaches it on two of the three
+fields), but only by also erasing two of the three real confirmed
+variables' own signal below `chi2_threshold` — not a real improvement,
+just a different failure mode.
 
 A large `chi2 / dof` is evidence against the null hypothesis — i.e.
 evidence of genuine flux variability beyond both statistical noise and
@@ -38,7 +44,7 @@ separately since a caller may want the raw statistic r
 threshold decision.
 """
 function variability_chi2(flux::AbstractVector{<:Real}, flux_err::AbstractVector{<:Real};
-                           systematic_error_fraction::Real=0.02)
+                           systematic_error_fraction::Real=0.03)
     length(flux) == length(flux_err) ||
         throw(ArgumentError("flux and flux_err must have the same length"))
     eff_err = sqrt.(flux_err .^ 2 .+ (systematic_error_fraction .* flux) .^ 2)
@@ -130,7 +136,7 @@ end
                            chi2_threshold::Real=10.0,
                            normalize::Bool=true,
                            max_relative_error::Real=0.10,
-                           systematic_error_fraction::Real=0.02)
+                           systematic_error_fraction::Real=0.03)
 
 Match source detections across frames by consistent *position* (zero
 assumed motion) rather than [`link_candidates`](@ref)'s linear-motion
@@ -157,7 +163,7 @@ Three corrections are applied before any variability t
   a constant star doesn't read as variable just because one exposure's
   aperture happened to enclose a different fraction of its PSF. Set
   `normalize=false` to skip this (e.g. if `flux` is already calibrated).
-- **Systematic error floor** (`systematic_error_fraction`, default 2%,
+- **Systematic error floor** (`systematic_error_fraction`, default 3%,
   forwarded to [`variability_chi2`](@ref)): the actual dominant fix for
   this function's real, measured false-positive floor — see below.
 
@@ -169,14 +175,15 @@ constant-flux null hypothesis; a group is kept only if
 chi-squared (`chi2 / dof`) exceeds `chi2_threshold`.
 
 `chi2_threshold`'s default and `systematic_error_fraction` (forwarded to
-[`variability_chi2`](@ref)) were both calibrated against the same real
-ZTF field (451, 119 matched, high-S/N stationary stars) this whole
-docstring measures against — and the calibration story is worth reading,
-because the first hypothesis tried here was wrong. `detect_sources` used
-to center its aperture on `PeakMesh`'s raw *integer*-pixel peak; the
-working theory was that per-frame pixel-grid jitter in that peak, against
-a small `aperture_radius`, was producing spurious flux swings. Fixing
-that (`detect_sources` now refines to a sub-pixel centroid — see its own
+[`variability_chi2`](@ref)) were both calibrated against real ZTF data —
+first against field 451 (119 matched, high-S/N stationary stars), later
+cross-checked against two more independent real fields — and the
+calibration story is worth reading, because the first hypothesis tried
+here was wrong. `detect_sources` used to center its aperture on
+`PeakMesh`'s raw *integer*-pixel peak; the working theory was that
+per-frame pixel-grid jitter in that peak, against a small
+`aperture_radius`, was producing spurious flux swings. Fixing that
+(`detect_sources` now refines to a sub-pixel centroid — see its own
 docstring) barely moved the false-positive rate at all (23% → 22% at a
 threshold of 3; 13% → 13% at 20) — a real, measured non-result, not
 swept under the rug. Checking *which* stars were actually being flagged
@@ -188,21 +195,31 @@ vs. 2.93% among the rest, the signature of a systemati
 underestimated statistical noise. Adding that floor
 (`systematic_error_fraction`, in `variability_chi2`) is what actually
 fixed it: a 1% floor cut it to 6% at a threshold of 3, ~0% at 20; a real,
-large improvement over the 23-8% range measured before either fix, but
-sweeping the floor further (against the same 119 real stars) found more
-room without giving up real sensitivity — checked against an actual
-confirmed variable, not just the stationary-star side of the tradeoff.
-`systematic_error_fraction=0.02` (the current default) cuts the
-threshold-10 false-positive rate to 2.0% (vs 1%'s 3.3%), while a real,
-independently-confirmed variable (ASASSN-V J183620.31 — see the
-[Investigation Log](https://richard7987.github.io/AsteroidPipeline.jl/dev/variable-star-validation))
-still clears `chi2_threshold=10.0` with a healthy margin (reduced chi2 ≈
-31, over 3x the threshold) at this floor — a floor of 5% would erase that
-same real signal (reduced chi2 drops to ≈5, below threshold), so 2% is
-not "as high as possible", it's the largest floor checked that still
-leaves real variability comfortably detectable. Still not zero: treat any
-candidate here as requiring independent confirmation (a VSX/SIMBAD match
-via [`crossmatch_catalog`](@ref), or a period recovered by
+large improvement over the 23-8% range measured before either fix.
+
+Sweeping the floor further, first against field 451 alone, found more
+room without giving up real sensitivity — checked against a real
+confirmed variable (ASASSN-V J183620.31, from a different real field),
+not just the stationary-star side of the tradeoff. That first pass
+settled on 2%. Later, two more real, independently-confirmed variables
+were found in two more real ZTF fields (V1012 Mon and ASASSN-V
+J072906.85-090518.2 — dense fields near the galactic plane, each with
+over 180 real matched stationary stars of its own), letting the same
+sweep be repeated three times over, each with its own real confirmed
+variable and its own real false-positive sample, instead of trusting one
+field's result to generalize. That repeat is what moved the default from
+2% to 3%: at 3%, all **three** confirmed variables (reduced chi2
+13.8/61.2/17.5 across field 451/V1012 Mon/ASASSN-072906) still clear
+`chi2_threshold=10.0`, while the false-positive rate in every one of the
+three fields is the same or better than at 2% (1.3%/2.04%/0.55% at 3% vs.
+2.0%/2.55%/0.55% at 2%). A literal 0% false-positive rate is reachable —
+a 5% floor gets there on two of the three fields — but only by also
+pushing two of the three confirmed variables' own reduced chi2 below
+`chi2_threshold`, i.e. by going blind to real variability rather than
+genuinely improving anything; 3% is the largest floor across all three
+real fields that doesn't cost that. Still not zero: treat any candidate
+here as requiring independent confirmation (a VSX/SIMBAD match via
+[`crossmatch_catalog`](@ref), or a period recovered by
 [`recover_rotation_period`](@ref)), not as self-evidently real.
 
 With the default `min_frames` (every frame must match at high S/N), a
@@ -245,7 +262,7 @@ function find_variable_sources(detections_per_frame, t
                                 chi2_threshold::Real=10.0,
                                 normalize::Bool=true,
                                 max_relative_error::Real=0.10,
-                                systematic_error_fraction::Real=0.02)
+                                systematic_error_fraction::Real=0.03)
     nframes = length(detections_per_frame)
     nframes == length(timestamps) ||
         throw(ArgumentError("detections_per_frame and timestamps must have the same length"))