From ed92c1d66a9833c69a51ce1860300c97b7df0023 Mon Sep 17 00:00:00 2001 From: Hans Davenport <35202271+hd152@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:17:02 -0700 Subject: [PATCH] Widen --auto's robust_pca auto-upgrade ceiling from N=10 to N=25 The original threshold was set from a benchmark on a synthetic 2000x3000x3 (RGB-shaped, P=18M) array, larger than any real calibration frame this camera produces (raw mono FITS, 2048x3056, P=6.26M) -- it overstated the real cost. Direct measurement on real mono frames of the real shape, after 2.2.5's robust_pca native-kernel fusion (which changed what dominates the per-iteration cost), gave: N=10 45s, N=15 89s, N=20 164s, N=25 258s, N=30 377s/~6.3min (fit validated against a direct N=30 measurement, not just extrapolated). Today's N=25 costs less than half of what N=10 cost when 10 was first judged tolerable for a silent --auto default. Widened ROBUST_PCA_AUTO_MAX_FRAMES to 25 on that basis. N=30's ~6.3min sits right at the old N=10 tolerance bar, so left as the next candidate rather than taken now. Updated the stale comments in cli.py and test_master_method_auto.py that referenced the old synthetic-shape benchmark and threshold history. Tests reference the constant dynamically (not hardcoded), so no test changes needed beyond the stale docstring; full suite (1647 tests) passes. Co-Authored-By: Claude Sonnet 5 --- CHANGELOG.md | 11 ++++++++ src/cli.py | 22 ++++++++-------- src/models.py | 44 +++++++++++++++++++------------- tests/test_master_method_auto.py | 4 ++- 4 files changed, 51 insertions(+), 30 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c3b00f6..4c49c7e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,17 @@ match the `VERSION` file and `v*` git tags. ## [Unreleased] +### Changed + +- **`--auto`'s robust_pca auto-upgrade now applies to calibration libraries up to 25 frames, not 10.** The + threshold was set from a 2026-09-22 benchmark on a synthetic RGB-shaped (2000x3000x3) array that + overstated real cost -- this camera's actual calibration frames are raw mono FITS (2048x3056, no x3). + Direct measurement on real frames of that real shape, after 2.2.5's robust_pca native-kernel fusion, gave + N=10 45s, N=15 89s, N=20 164s, N=25 258s, N=30 377s/~6.3min -- today's N=25 costs less than half of what + N=10 cost when 10 was first judged tolerable. Widened to 25 on that basis. `--master-method robust_pca` + still applies above this ceiling regardless. See `Config.ROBUST_PCA_AUTO_MAX_FRAMES`'s comment in + `src/models.py` for the full measurement. + ## [2.2.5] - 2026-09-22 ### Changed diff --git a/src/cli.py b/src/cli.py index 62c50b5..e80b0ee 100644 --- a/src/cli.py +++ b/src/cli.py @@ -377,17 +377,17 @@ def _build_masters(frames: dict, stats: "ProcessingStats | None" = None, safe_print("\nCreating master calibration frames...") # master_method defaults to 'median' and is opt-in only for 'robust_pca' at - # real scale -- benchmarked robust_pca_decompose directly (2000x3000x3, 20 - # frames): 2910s (~48.5 min) pre-optimization, 1264s (~21 min) after - # src/robust_pca.py's Gram-matrix-trick SVD (native gram_matrix_wide/ - # small_times_wide kernels, ~9x over a direct SVD call on this shape, - # ~2.3x end-to-end -- the non-SVD per-iteration ops don't benefit and now - # dominate more of the budget). Still too slow to silently default for a - # real-sized calibration library. Below ROBUST_PCA_AUTO_MAX_FRAMES, - # though, cost scales down enough (~5min at N=10) to be a reasonable - # --auto default -- gated per calibration type (bias/dark/flat counts - # often differ) below, not globally, so one small type doesn't drag a - # large one into robust_pca too. + # real scale -- an early benchmark on a synthetic 2000x3000x3 (RGB-shaped, + # P=18M) array at N=20 gave 2910s pre-optimization, 1264s/~21min after + # src/robust_pca.py's Gram-matrix-trick SVD; that shape is larger than any + # real calibration frame here (this camera's raw mono FITS are + # 2048x3056, P=6.26M, no x3), so it overstated the real cost. Direct + # measurement on real mono frames is what actually sets + # ROBUST_PCA_AUTO_MAX_FRAMES now -- see that constant's comment in + # src/models.py for the current real numbers by N. Below the threshold, + # cost is judged a reasonable --auto default -- gated per calibration + # type (bias/dark/flat counts often differ) below, not globally, so one + # small type doesn't drag a large one into robust_pca too. master_method = getattr(args, 'master_method', 'median') or 'median' _auto_master = (getattr(args, 'auto', False) and master_method == 'median' diff --git a/src/models.py b/src/models.py index 0572196..2f38ad0 100644 --- a/src/models.py +++ b/src/models.py @@ -116,25 +116,33 @@ class Config: # Robust-PCA (Principal Component Pursuit) master calibration frames ROBUST_PCA_MIN_FRAMES = 5 # Below this, the low-rank/sparse split is # underdetermined -- make_master falls back to median - ROBUST_PCA_AUTO_MAX_FRAMES = 10 # --auto only auto-upgrades median->robust_pca for a + ROBUST_PCA_AUTO_MAX_FRAMES = 25 # --auto only auto-upgrades median->robust_pca for a # calibration type at or below this frame count. - # Measured 1264s/~21min at N=20 -- too slow for a - # silent --auto default at that scale (see - # _build_masters's comment). Cost vs. N was assumed - # O(N^2 x pixels) (quadratic) when this threshold was - # first set, giving an estimated ~316s/~5.3min at - # N<=10 via (10/20)^2 * 1264s; tools/bench_robust_pca_ - # scale.py later measured the real exponent as N^1.42 - # (sub-quadratic -- iteration count evidently doesn't - # scale down as fast as the per-iteration Gram-matrix - # cost does), giving ~404s/~6.7min at N=10 on that - # run (cross-checked against this file's own N=20 - # anchor at 0.85x -- same ballpark, not exact, since - # it's a different machine/run than the anchor). - # Still judged tolerable for --auto at N<=10; kept at - # 10 rather than widened after seeing the real N=15 - # cost (~717s/~12min) -- opt-in via --master-method - # robust_pca above this frame count regardless + # History: originally set to 10 after an N=20 real- + # shape (2048x3056 mono) measurement of 1264s/~21min + # was judged too slow for a silent --auto default, and + # N=15's ~717s/~12min (extrapolated then, from a + # measured N^1.42 exponent) was judged not worth + # widening for. Both those costs were dominated by + # robust_pca_decompose's per-iteration plain-numpy + # elementwise arithmetic -- once that got fused into + # native kernels (robust_pca_pre_svd_input/_iterate, + # the L-update matmul routed through small_times_wide; + # see CHANGELOG "Several real perf fixes"), the SVD + # step dominates more and the real N-exponent measured + # steeper (~N^1.91, tools/bench_robust_pca_scale.py) -- + # but the *absolute* cost dropped enough that it no + # longer matters: direct real-shape measurements + # (same 2048x3056 mono frames, tested up to N=30 for + # a fit, not just extrapolated) gave N=10 45s, N=15 + # 89s, N=20 164s, N=25 258s, N=30 377s/~6.3min -- + # i.e. today's N=25 costs less than half of what N=10 + # cost when this threshold was first judged tolerable. + # Widened to 25 on that basis; N=30's ~6.3min sits + # right at the old N=10 tolerance bar, so left as the + # next candidate rather than taken now. Opt-in via + # --master-method robust_pca above this frame count + # regardless ROBUST_PCA_MAX_ITERS = 50 # IALM iterations (each is one economy SVD of an # (N, H*W*C) matrix, via src.robust_pca's # Gram-matrix-trick + native gram_matrix_wide/ diff --git a/tests/test_master_method_auto.py b/tests/test_master_method_auto.py index 2856699..af1f2e9 100644 --- a/tests/test_master_method_auto.py +++ b/tests/test_master_method_auto.py @@ -5,7 +5,9 @@ count in [Config.ROBUST_PCA_MIN_FRAMES, Config.ROBUST_PCA_AUTO_MAX_FRAMES] -- below that floor robust_pca is underdetermined anyway (make_master's own existing fallback), above the ceiling it's too slow to silently add to ---auto's runtime (see _build_masters's comment: benchmarked ~21min at N=20). +--auto's runtime (see Config.ROBUST_PCA_AUTO_MAX_FRAMES's comment in +src/models.py for the current real by-N measurements this threshold is set +from). """ from __future__ import annotations