diff --git a/.gitignore b/.gitignore index 9c49cc9..d1df400 100644 --- a/.gitignore +++ b/.gitignore @@ -54,3 +54,6 @@ reports/.cache/ # Runtime A5 parity bundles are generated on the production server. Research # conclusions belong in docs/research, not as an ever-growing artifact archive. reports/fundamentals-parity/ + +# Calibration harness raw-pull cache (Alpaca/FRED); regenerable, not a record. +.calib-cache/ diff --git a/reports/regime-monitor-v4-calibration-20260808-200203.json b/reports/regime-monitor-v4-calibration-20260808-200203.json new file mode 100644 index 0000000..b63195b --- /dev/null +++ b/reports/regime-monitor-v4-calibration-20260808-200203.json @@ -0,0 +1,1414 @@ +{ + "generated_at": "2026-08-08T20:02:03", + "git_rev": "f22313d", + "params": { + "end": "2026-07-24", + "sessions": 408, + "history_days": 1200, + "oas_window_days": 700, + "oas_fetch_days": 4748, + "methodology": "v2_reconstruction,v3,v4,v4-vix-b,v4-p1-capped", + "expected_first_session": null, + "cache_dir": ".calib-cache", + "out": null, + "quiet": true + }, + "provenance": { + "symbols": { + "AAPL": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "MSFT": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "NVDA": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "AMZN": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "META": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "GOOGL": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "AVGO": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "AMD": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "ORCL": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "CRM": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "NOW": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "PLTR": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "ANET": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "DELL": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "SMCI": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "MU": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "QCOM": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "INTC": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "AMAT": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "LRCX": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "KLAC": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "SNPS": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "CDNS": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "ADI": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "TXN": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "IBM": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "CSCO": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "PANW": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "CRWD": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "VRT": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "SMH": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "QQQ": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "SPY": { + "bars": 825, + "first": "2023-04-11", + "last": "2026-07-24" + } + }, + "vix": { + "points": 850, + "first": "2023-04-11", + "last": "2026-07-24" + }, + "oas": { + "points": 778, + "first": "2023-08-08", + "last": "2026-07-24" + }, + "scored_range": [ + "2024-12-05", + "2026-07-24" + ] + }, + "hard_gates": [ + { + "gate": "symbols_fetched", + "expected": 33, + "measured": 33, + "passed": true, + "detail": [] + }, + { + "gate": "per_symbol_warmup_252_bars", + "expected": "all", + "measured": 33, + "passed": true, + "detail": [] + }, + { + "gate": "per_symbol_reaches_last_session", + "expected": "2026-07-24", + "measured": 33, + "passed": true, + "detail": [] + }, + { + "gate": "breadth_counts_full_basket", + "expected": 30, + "measured": "408/408 sessions", + "passed": true, + "detail": [] + }, + { + "gate": "sessions_scored", + "expected": 408, + "measured": 408, + "passed": true + }, + { + "gate": "last_scored_date", + "expected": "2026-07-24", + "measured": "2026-07-24", + "passed": true + }, + { + "gate": "w1_available_every_session", + "expected": 408, + "measured": 408, + "passed": true + }, + { + "gate": "state_coverage_100_every_row", + "expected": 0, + "measured": 0, + "passed": true, + "detail": [] + }, + { + "gate": "no_stale_inputs", + "expected": 0, + "measured": 0, + "passed": true, + "detail": [] + }, + { + "gate": "state_v4_le_v3_every_row", + "expected": 0, + "measured": 0, + "passed": true, + "detail": [] + } + ], + "blocked_by": [], + "provisional": false, + "variants": { + "v2_reconstruction": { + "label": "v2_reconstruction", + "sessions": 408, + "state": { + "avg": 22.68, + "median": 16.15, + "p80": 35.1, + "p90": 65.63, + "max": 91.2, + "scored": 408 + }, + "warning_avg": 11.65, + "saturation_census": { + "p3_pegged": 39, + "p2_pegged": 0, + "v1_pegged": 14, + "p1_pegged": 46 + }, + "price_argmax_sole": { + "P1": 9, + "P2": 18, + "P3": 308 + }, + "price_argmax_tie_inclusive": { + "P1": 82, + "P2": 53, + "P3": 381 + }, + "price_argmax_ties": 73, + "w1_nonzero_sessions": 108, + "band_shares_current": { + "stable": 57.1, + "watch": 30.1, + "elevated": 9.3, + "breaking": 3.4 + }, + "soft_gates": [ + { + "gate": "v2_state_avg", + "expected": 22.6, + "measured": 22.68, + "tolerance": 0.3, + "passed": true + }, + { + "gate": "v2_state_p80", + "expected": 35.1, + "measured": 35.1, + "tolerance": 0.5, + "passed": true + }, + { + "gate": "v2_state_max", + "expected": 91.2, + "measured": 91.2, + "tolerance": 0.5, + "passed": true + }, + { + "gate": "v2_p3_pegged", + "expected": 39, + "measured": 39, + "tolerance": 0, + "passed": true + }, + { + "gate": "w1_live_sessions", + "expected": 108, + "measured": 108, + "tolerance": 0, + "passed": true + } + ] + }, + "v3": { + "label": "v3", + "sessions": 408, + "state": { + "avg": 18.13, + "median": 9.1, + "p80": 31.36, + "p90": 65.0, + "max": 87.4, + "scored": 408 + }, + "warning_avg": 14.88, + "saturation_census": { + "p3_pegged": 0, + "p2_pegged": 0, + "v1_pegged": 14, + "p1_pegged": 46 + }, + "price_argmax_sole": { + "P1": 71, + "P2": 33, + "P3": 268 + }, + "price_argmax_tie_inclusive": { + "P1": 107, + "P2": 69, + "P3": 304 + }, + "price_argmax_ties": 36, + "w1_nonzero_sessions": 197, + "band_shares_current": { + "stable": 73.0, + "watch": 15.4, + "elevated": 8.1, + "breaking": 3.4 + }, + "soft_gates": [ + { + "gate": "v3_state_max", + "expected": 87.4, + "measured": 87.4, + "tolerance": 0.5, + "passed": true + } + ] + }, + "v4": { + "label": "v4", + "sessions": 408, + "state": { + "avg": 14.78, + "median": 8.35, + "p80": 21.7, + "p90": 43.63, + "max": 83.5, + "scored": 408 + }, + "warning_avg": 14.88, + "saturation_census": { + "p3_pegged": 0, + "p2_pegged": 0, + "v1_pegged": 0, + "p1_pegged": 0 + }, + "price_argmax_sole": { + "P1": 17, + "P2": 48, + "P3": 307 + }, + "price_argmax_tie_inclusive": { + "P1": 53, + "P2": 84, + "P3": 343 + }, + "price_argmax_ties": 36, + "w1_nonzero_sessions": 197, + "band_shares_current": { + "stable": 78.9, + "watch": 13.0, + "elevated": 7.4, + "breaking": 0.7 + }, + "soft_gates": [], + "band_grid": [ + { + "bands": [ + 15.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 68.9, + "watch": 19.6, + "elevated": 6.4, + "breaking": 5.1 + } + }, + { + "bands": [ + 15.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 68.9, + "watch": 19.6, + "elevated": 8.1, + "breaking": 3.4 + } + }, + { + "bands": [ + 15.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 68.9, + "watch": 19.6, + "elevated": 10.3, + "breaking": 1.2 + } + }, + { + "bands": [ + 15.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 68.9, + "watch": 23.0, + "elevated": 2.9, + "breaking": 5.1 + } + }, + { + "bands": [ + 15.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 68.9, + "watch": 23.0, + "elevated": 4.7, + "breaking": 3.4 + } + }, + { + "bands": [ + 15.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 68.9, + "watch": 23.0, + "elevated": 6.9, + "breaking": 1.2 + } + }, + { + "bands": [ + 20.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 78.9, + "watch": 9.6, + "elevated": 6.4, + "breaking": 5.1 + } + }, + { + "bands": [ + 20.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 78.9, + "watch": 9.6, + "elevated": 8.1, + "breaking": 3.4 + } + }, + { + "bands": [ + 20.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 78.9, + "watch": 9.6, + "elevated": 10.3, + "breaking": 1.2 + } + }, + { + "bands": [ + 20.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 78.9, + "watch": 13.0, + "elevated": 2.9, + "breaking": 5.1 + } + }, + { + "bands": [ + 20.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 78.9, + "watch": 13.0, + "elevated": 4.7, + "breaking": 3.4 + } + }, + { + "bands": [ + 20.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 78.9, + "watch": 13.0, + "elevated": 6.9, + "breaking": 1.2 + } + }, + { + "bands": [ + 25.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 82.1, + "watch": 6.4, + "elevated": 6.4, + "breaking": 5.1 + } + }, + { + "bands": [ + 25.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 82.1, + "watch": 6.4, + "elevated": 8.1, + "breaking": 3.4 + } + }, + { + "bands": [ + 25.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 82.1, + "watch": 6.4, + "elevated": 10.3, + "breaking": 1.2 + } + }, + { + "bands": [ + 25.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 82.1, + "watch": 9.8, + "elevated": 2.9, + "breaking": 5.1 + } + }, + { + "bands": [ + 25.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 82.1, + "watch": 9.8, + "elevated": 4.7, + "breaking": 3.4 + } + }, + { + "bands": [ + 25.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 82.1, + "watch": 9.8, + "elevated": 6.9, + "breaking": 1.2 + } + } + ] + }, + "v4-vix-b": { + "label": "v4-vix-b", + "sessions": 408, + "state": { + "avg": 15.24, + "median": 8.55, + "p80": 22.62, + "p90": 44.33, + "max": 83.6, + "scored": 408 + }, + "warning_avg": 14.88, + "saturation_census": { + "p3_pegged": 0, + "p2_pegged": 0, + "v1_pegged": 0, + "p1_pegged": 0 + }, + "price_argmax_sole": { + "P1": 17, + "P2": 48, + "P3": 307 + }, + "price_argmax_tie_inclusive": { + "P1": 53, + "P2": 84, + "P3": 343 + }, + "price_argmax_ties": 36, + "w1_nonzero_sessions": 197, + "band_shares_current": { + "stable": 77.9, + "watch": 13.7, + "elevated": 7.6, + "breaking": 0.7 + }, + "soft_gates": [], + "band_grid": [ + { + "bands": [ + 15.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 68.1, + "watch": 20.1, + "elevated": 6.4, + "breaking": 5.4 + } + }, + { + "bands": [ + 15.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 68.1, + "watch": 20.1, + "elevated": 8.1, + "breaking": 3.7 + } + }, + { + "bands": [ + 15.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 68.1, + "watch": 20.1, + "elevated": 10.0, + "breaking": 1.7 + } + }, + { + "bands": [ + 15.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 68.1, + "watch": 23.5, + "elevated": 2.9, + "breaking": 5.4 + } + }, + { + "bands": [ + 15.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 68.1, + "watch": 23.5, + "elevated": 4.7, + "breaking": 3.7 + } + }, + { + "bands": [ + 15.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 68.1, + "watch": 23.5, + "elevated": 6.6, + "breaking": 1.7 + } + }, + { + "bands": [ + 20.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 77.9, + "watch": 10.3, + "elevated": 6.4, + "breaking": 5.4 + } + }, + { + "bands": [ + 20.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 77.9, + "watch": 10.3, + "elevated": 8.1, + "breaking": 3.7 + } + }, + { + "bands": [ + 20.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 77.9, + "watch": 10.3, + "elevated": 10.0, + "breaking": 1.7 + } + }, + { + "bands": [ + 20.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 77.9, + "watch": 13.7, + "elevated": 2.9, + "breaking": 5.4 + } + }, + { + "bands": [ + 20.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 77.9, + "watch": 13.7, + "elevated": 4.7, + "breaking": 3.7 + } + }, + { + "bands": [ + 20.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 77.9, + "watch": 13.7, + "elevated": 6.6, + "breaking": 1.7 + } + }, + { + "bands": [ + 25.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 81.6, + "watch": 6.6, + "elevated": 6.4, + "breaking": 5.4 + } + }, + { + "bands": [ + 25.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 81.6, + "watch": 6.6, + "elevated": 8.1, + "breaking": 3.7 + } + }, + { + "bands": [ + 25.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 81.6, + "watch": 6.6, + "elevated": 10.0, + "breaking": 1.7 + } + }, + { + "bands": [ + 25.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 81.6, + "watch": 10.0, + "elevated": 2.9, + "breaking": 5.4 + } + }, + { + "bands": [ + 25.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 81.6, + "watch": 10.0, + "elevated": 4.7, + "breaking": 3.7 + } + }, + { + "bands": [ + 25.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 81.6, + "watch": 10.0, + "elevated": 6.6, + "breaking": 1.7 + } + } + ] + }, + "v4-p1-capped": { + "label": "v4-p1-capped", + "sessions": 408, + "state": { + "avg": 14.73, + "median": 8.35, + "p80": 21.7, + "p90": 43.63, + "max": 80.1, + "scored": 408 + }, + "warning_avg": 14.88, + "saturation_census": { + "p3_pegged": 0, + "p2_pegged": 0, + "v1_pegged": 0, + "p1_pegged": 0 + }, + "price_argmax_sole": { + "P1": 0, + "P2": 49, + "P3": 323 + }, + "price_argmax_tie_inclusive": { + "P1": 36, + "P2": 85, + "P3": 359 + }, + "price_argmax_ties": 36, + "w1_nonzero_sessions": 197, + "band_shares_current": { + "stable": 78.9, + "watch": 13.0, + "elevated": 7.8, + "breaking": 0.2 + }, + "soft_gates": [], + "band_grid": [ + { + "bands": [ + 15.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 68.9, + "watch": 19.6, + "elevated": 6.4, + "breaking": 5.1 + } + }, + { + "bands": [ + 15.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 68.9, + "watch": 19.6, + "elevated": 8.1, + "breaking": 3.4 + } + }, + { + "bands": [ + 15.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 68.9, + "watch": 19.6, + "elevated": 10.3, + "breaking": 1.2 + } + }, + { + "bands": [ + 15.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 68.9, + "watch": 23.0, + "elevated": 2.9, + "breaking": 5.1 + } + }, + { + "bands": [ + 15.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 68.9, + "watch": 23.0, + "elevated": 4.7, + "breaking": 3.4 + } + }, + { + "bands": [ + 15.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 68.9, + "watch": 23.0, + "elevated": 6.9, + "breaking": 1.2 + } + }, + { + "bands": [ + 20.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 78.9, + "watch": 9.6, + "elevated": 6.4, + "breaking": 5.1 + } + }, + { + "bands": [ + 20.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 78.9, + "watch": 9.6, + "elevated": 8.1, + "breaking": 3.4 + } + }, + { + "bands": [ + 20.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 78.9, + "watch": 9.6, + "elevated": 10.3, + "breaking": 1.2 + } + }, + { + "bands": [ + 20.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 78.9, + "watch": 13.0, + "elevated": 2.9, + "breaking": 5.1 + } + }, + { + "bands": [ + 20.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 78.9, + "watch": 13.0, + "elevated": 4.7, + "breaking": 3.4 + } + }, + { + "bands": [ + 20.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 78.9, + "watch": 13.0, + "elevated": 6.9, + "breaking": 1.2 + } + }, + { + "bands": [ + 25.0, + 40.0, + 60.0 + ], + "shares": { + "stable": 82.1, + "watch": 6.4, + "elevated": 6.4, + "breaking": 5.1 + } + }, + { + "bands": [ + 25.0, + 40.0, + 65.0 + ], + "shares": { + "stable": 82.1, + "watch": 6.4, + "elevated": 8.1, + "breaking": 3.4 + } + }, + { + "bands": [ + 25.0, + 40.0, + 70.0 + ], + "shares": { + "stable": 82.1, + "watch": 6.4, + "elevated": 10.3, + "breaking": 1.2 + } + }, + { + "bands": [ + 25.0, + 50.0, + 60.0 + ], + "shares": { + "stable": 82.1, + "watch": 9.8, + "elevated": 2.9, + "breaking": 5.1 + } + }, + { + "bands": [ + 25.0, + 50.0, + 65.0 + ], + "shares": { + "stable": 82.1, + "watch": 9.8, + "elevated": 4.7, + "breaking": 3.4 + } + }, + { + "bands": [ + 25.0, + 50.0, + 70.0 + ], + "shares": { + "stable": 82.1, + "watch": 9.8, + "elevated": 6.9, + "breaking": 1.2 + } + } + ] + } + }, + "scenarios": { + "note": "pillar arithmetic on explicit sensor scores; State weights unchanged", + "vix_a": [ + { + "scenario": "S1 ordinary tape", + "P1": 0.0, + "P2": 0.0, + "P3": 7.5, + "price": 7.5, + "breadth": 0.0, + "C1": 0.0, + "V1": 4.0, + "state": 3.6 + }, + { + "scenario": "S2 10% correction, calm credit", + "P1": 30.0, + "P2": 0.0, + "P3": 31.25, + "price": 31.25, + "breadth": 62.5, + "C1": 0.0, + "V1": 34.4, + "state": 33.28 + }, + { + "scenario": "S3a 2022-style, calm credit, no death cross", + "P1": 87.5, + "P2": 0.0, + "P3": 90.83, + "price": 90.83, + "breadth": 100.0, + "C1": 0.0, + "V1": 60.0, + "state": 70.33 + }, + { + "scenario": "S3b 2022-style, calm credit, death cross", + "P1": 87.5, + "P2": 100.0, + "P3": 90.83, + "price": 100.0, + "breadth": 100.0, + "C1": 0.0, + "V1": 60.0, + "state": 74.0 + }, + { + "scenario": "S4 credit event on top", + "P1": 100.0, + "P2": 100.0, + "P3": 100.0, + "price": 100.0, + "breadth": 100.0, + "C1": 75.0, + "V1": 86.67, + "state": 93.0 + }, + { + "scenario": "S5 March 2020, everything pegged", + "P1": 100.0, + "P2": 100.0, + "P3": 100.0, + "price": 100.0, + "breadth": 100.0, + "C1": 100.0, + "V1": 100.0, + "state": 100.0 + } + ] + }, + "diagnostics": { + "vix_top10": [ + [ + "2025-04-08", + 52.33 + ], + [ + "2025-04-07", + 46.98 + ], + [ + "2025-04-04", + 45.31 + ], + [ + "2025-04-10", + 40.72 + ], + [ + "2024-08-05", + 38.57 + ], + [ + "2025-04-11", + 37.56 + ], + [ + "2025-04-21", + 33.82 + ], + [ + "2025-04-09", + 33.62 + ], + [ + "2025-04-16", + 32.64 + ], + [ + "2026-03-27", + 31.05 + ] + ], + "candidate_anchors": { + "P5_VIX_ANCHORS_A": [ + [ + 15.0, + 0.0 + ], + [ + 20.0, + 20.0 + ], + [ + 25.0, + 38.0 + ], + [ + 30.0, + 55.0 + ], + [ + 40.0, + 80.0 + ], + [ + 55.0, + 100.0 + ] + ], + "P5_VIX_ANCHORS_B": [ + [ + 15.0, + 0.0 + ], + [ + 20.0, + 25.0 + ], + [ + 25.0, + 45.0 + ], + [ + 30.0, + 65.0 + ], + [ + 40.0, + 85.0 + ], + [ + 55.0, + 100.0 + ] + ], + "P1_TREND_BREAK_ANCHORS": [ + [ + 0.0, + 20.0 + ], + [ + 3.0, + 35.0 + ], + [ + 8.0, + 55.0 + ], + [ + 15.0, + 75.0 + ], + [ + 25.0, + 100.0 + ] + ] + } + }, + "v4_recommendation": { + "state_bands_candidate": [ + 20.0, + 50.0, + 65.0 + ], + "provisional": false, + "note": "confirm against band_grid + scenarios before shipping" + } +} \ No newline at end of file diff --git a/reports/regime-monitor-v4-calibration-20260808-200203.md b/reports/regime-monitor-v4-calibration-20260808-200203.md new file mode 100644 index 0000000..703678d --- /dev/null +++ b/reports/regime-monitor-v4-calibration-20260808-200203.md @@ -0,0 +1,75 @@ +# Regime Monitor v4 calibration + +Generated 2026-08-08T20:02:03 at `f22313d`, 2024-12-05 → 2026-07-24. + +## Hard gates + +| gate | expected | measured | | +|---|---|---|---| +| symbols_fetched | 33 | 33 | ok | +| per_symbol_warmup_252_bars | all | 33 | ok | +| per_symbol_reaches_last_session | 2026-07-24 | 33 | ok | +| breadth_counts_full_basket | 30 | 408/408 sessions | ok | +| sessions_scored | 408 | 408 | ok | +| last_scored_date | 2026-07-24 | 2026-07-24 | ok | +| w1_available_every_session | 408 | 408 | ok | +| state_coverage_100_every_row | 0 | 0 | ok | +| no_stale_inputs | 0 | 0 | ok | +| state_v4_le_v3_every_row | 0 | 0 | ok | + +## Distributions + +| variant | avg | median | p80 | p90 | max | +|---|---|---|---|---|---| +| v2_reconstruction | 22.68 | 16.15 | 35.1 | 65.63 | 91.2 | +| v3 | 18.13 | 9.1 | 31.36 | 65.0 | 87.4 | +| v4 | 14.78 | 8.35 | 21.7 | 43.63 | 83.5 | +| v4-vix-b | 15.24 | 8.55 | 22.62 | 44.33 | 83.6 | +| v4-p1-capped | 14.73 | 8.35 | 21.7 | 43.63 | 80.1 | + +## Saturation census (sessions pegged at 100) + +| variant | P1 | P2 | P3 | V1 | +|---|---|---|---|---| +| v2_reconstruction | 46 | 0 | 39 | 14 | +| v3 | 46 | 0 | 0 | 14 | +| v4 | 0 | 0 | 0 | 0 | +| v4-vix-b | 0 | 0 | 0 | 0 | +| v4-p1-capped | 0 | 0 | 0 | 0 | + +## Reproduction gates — v2_reconstruction + +| figure | published | measured | | +|---|---|---|---| +| v2_state_avg | 22.6 | 22.68 | ok | +| v2_state_p80 | 35.1 | 35.1 | ok | +| v2_state_max | 91.2 | 91.2 | ok | +| v2_p3_pegged | 39 | 39 | ok | +| w1_live_sessions | 108 | 108 | ok | + +## Reproduction gates — v3 + +| figure | published | measured | | +|---|---|---|---| +| v3_state_max | 87.4 | 87.4 | ok | + +## v4 band-share grid (watch 20 / elevated 50) + +| breaking | stable | watch | elevated | breaking | +|---|---|---|---|---| +| 60 | 78.9 | 13.0 | 2.9 | 5.1 | +| 65 | 78.9 | 13.0 | 4.7 | 3.4 | +| 70 | 78.9 | 13.0 | 6.9 | 1.2 | + +## Scenarios (pillar arithmetic, explicit sensor scores) + +| scenario | price | breadth | C1 | V1 | State | +|---|---|---|---|---|---| +| S1 ordinary tape | 7.5 | 0.0 | 0.0 | 4.0 | **3.6** | +| S2 10% correction, calm credit | 31.25 | 62.5 | 0.0 | 34.4 | **33.28** | +| S3a 2022-style, calm credit, no death cross | 90.83 | 100.0 | 0.0 | 60.0 | **70.33** | +| S3b 2022-style, calm credit, death cross | 100.0 | 100.0 | 0.0 | 60.0 | **74.0** | +| S4 credit event on top | 100.0 | 100.0 | 75.0 | 86.67 | **93.0** | +| S5 March 2020, everything pegged | 100.0 | 100.0 | 100.0 | 100.0 | **100.0** | + +Recommendation: `{'state_bands_candidate': [20.0, 50.0, 65.0], 'provisional': False, 'note': 'confirm against band_grid + scenarios before shipping'}` diff --git a/scripts/run_regime_monitor_calibration.py b/scripts/run_regime_monitor_calibration.py new file mode 100644 index 0000000..cb3c1eb --- /dev/null +++ b/scripts/run_regime_monitor_calibration.py @@ -0,0 +1,756 @@ +"""Offline replay of the AI/Tech Risk Monitor, for calibrating a methodology cut. + +Reproduces the State/Warning series session by session from the same inputs the +live job uses -- Alpaca for prices, FRED for VIX and HY OAS -- with no database, +so a sensor change can be measured against real history before it ships. + +v3 was calibrated this way ad-hoc and the harness was never committed, which is +why its published numbers cannot be re-derived today. This is that harness. + +**It never reimplements an unchanged live sensor.** ``_compute_index``, +``_score_pillars``, breadth, divergence, P2, P4 and the Warning sensors are +imported and called. Only *candidate* formulas (proposed for v4) and *retired* +ones (v2, no longer in the codebase) are defined here and patched onto the +service for the duration of a variant. Once a candidate ships, delete it here and +import the shipped function instead, or the two will drift. + +The script refuses to emit a band recommendation unless every hard gate passes. +That is deliberate: it must be structurally impossible to read a calibration +result out of a run whose pipeline did not validate. + +Research branch only. Example: + + .\\.venv\\Scripts\\python.exe scripts\\run_regime_monitor_calibration.py ^ + --end 2026-07-24 --sessions 408 --methodology v3,v4 ^ + --cache-dir .calib-cache +""" + +from __future__ import annotations + +import argparse +import asyncio +import contextlib +import json +import math +import statistics +import subprocess +import sys +from collections.abc import Iterator +from copy import deepcopy +from datetime import date, datetime, timedelta +from pathlib import Path +from typing import Any, Callable + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from app.config import settings # noqa: E402 +from app.providers.alpaca import AlpacaOHLCVProvider # noqa: E402 +from app.services import breadth_service # noqa: E402 +from app.services import regime_monitor_service as rms # noqa: E402 + +# Published v3/v2 figures from docs/research/regime-monitor-v3.md. The v3 pair is +# DERIVED there (22.6 - 0.4; 91.2 - 3.8), not measured, so its tolerance is loose +# on purpose -- anything tighter would be false precision. +PUBLISHED = { + "window_end": "2026-07-24", + "sessions": 408, + "w1_live_sessions": 108, + "v2_state_avg": 22.6, + "v2_state_p80": 35.1, + "v2_state_max": 91.2, + "v2_p3_pegged": 39, + # No v3 average is published: the doc's "-0.4" is measured against + # v3-with-the-percentile-leg, not against v2, so only the max is checkable. + "v3_state_max": 87.4, +} + +# Every symbol comes from one source on one split-adjustment basis. Mixing a +# sqlite snapshot for the basket with Alpaca for the leaders would splice two +# adjustment bases mid-200-DMA for any symbol that split in between. +LEADER, CONFIRM, MARKET = "SMH", "QQQ", "SPY" + + +# --------------------------------------------------------------------------- +# Candidate formulas (proposed for v4) -- patched in, never shipped from here +# --------------------------------------------------------------------------- + +P5_VIX_ANCHORS_A = ((15.0, 0.0), (20.0, 20.0), (25.0, 38.0), (30.0, 55.0), (40.0, 80.0), (55.0, 100.0)) +P5_VIX_ANCHORS_B = ((15.0, 0.0), (20.0, 25.0), (25.0, 45.0), (30.0, 65.0), (40.0, 85.0), (55.0, 100.0)) +P1_TREND_BREAK_ANCHORS = ((0.0, 20.0), (3.0, 35.0), (8.0, 55.0), (15.0, 75.0), (25.0, 100.0)) + + +def _candidate_under_200(closes: list[float], anchors=P1_TREND_BREAK_ANCHORS) -> float | None: + """Graduated trend break: 0 above the 200-DMA, else scaled by depth below it. + + The live version returns a bare 0/100, which pins the price pillar's max() + at 100 through any real selloff and stops P3's ladder resolving. The step at + the crossing (0 -> 20) is kept deliberately: the break itself is a genuine + binary event and deserves a floor; only the depth past it is graduated. + """ + sma200 = rms._sma(closes, 200) + if sma200 is None or sma200 <= 0: + return None + pct_below = (sma200 - closes[-1]) / sma200 * 100.0 + if pct_below <= 0: + return 0.0 + return rms._clamp(rms._interpolate(pct_below, anchors)) + + +def _candidate_p1(anchors=P1_TREND_BREAK_ANCHORS) -> Callable: + def p1_trend_break(smh, qqq, leader_weight: float = 2.0): + return rms._blend( + _candidate_under_200(smh, anchors), _candidate_under_200(qqq, anchors), leader_weight + ) + + return p1_trend_break + + +def _candidate_p5(anchors) -> Callable: + def p5_volatility(vix: float | None) -> float | None: + if vix is None: + return None + return rms._clamp(rms._interpolate(vix, anchors)) + + return p5_volatility + + +def _capped(fn: Callable, cap: float) -> Callable: + """P1_SCORE_CAP fallback: cap the sensor score after the blend, before max().""" + + def wrapped(smh, qqq, leader_weight: float = 2.0): + value = fn(smh, qqq, leader_weight) + return None if value is None else min(value, cap) + + return wrapped + + +# --------------------------------------------------------------------------- +# Retired formulas (v2) -- reconstructed from git show 019ca13^ +# --------------------------------------------------------------------------- + +def _v2_drawdown(closes: list[float]) -> float | None: + if len(closes) < 30: + return None + peak = max(closes[-252:]) + if peak <= 0: + return None + return rms._clamp((peak - closes[-1]) / peak * 100.0 * 5.0) + + +def _v2_p3_drawdown(smh, qqq, leader_weight: float = 2.0) -> float | None: + """v2 took max() across the legs, so the more volatile leader always won.""" + vals = [v for v in (_v2_drawdown(smh), _v2_drawdown(qqq)) if v is not None] + return max(vals) if vals else None + + +def _v2_divergence_series(breadth, benchmark_closes, lookback: int = 20): + """v2's hard price gate: the sensor ZEROED during any decline. + + v3 replaced this with a taper, which is why v2 shows W1 nonzero on only 108 + of 408 sessions while v3 shows it nonzero far more often. Reconstructing it + is the only way to check that published figure. + """ + bench = {d: c for d, c in benchmark_closes} + common = sorted(d for d in bench if d in breadth) + out: dict[date, float] = {} + for i in range(lookback, len(common)): + d, d0 = common[i], common[i - lookback] + if bench[d0] <= 0: + continue + price_ret = (bench[d] / bench[d0] - 1.0) * 100.0 + deterioration = max(0.0, -(breadth[d] - breadth[d0])) + score = deterioration * 5.0 if price_ret >= 0 else 0.0 + out[d] = max(0.0, min(100.0, round(score, 2))) + return out + + +def _v2_f2_credit_spreads(oas_values: list[float]) -> float | None: + """70% named anchors + 30% upper-tail percentile over whatever window it got.""" + if not oas_values: + return None + latest = oas_values[-1] + absolute = rms._oas_absolute_score(latest) + if len(oas_values) < 30: + return round(absolute, 2) + less = sum(1 for v in oas_values if v < latest) + equal = sum(1 for v in oas_values if v == latest) + percentile = (less + 0.5 * equal) / len(oas_values) * 100.0 + relative = rms._clamp((percentile - 50.0) / 45.0 * 100.0) + return round(absolute * 0.7 + relative * 0.3, 2) + + +# --------------------------------------------------------------------------- +# Variants +# --------------------------------------------------------------------------- + +VARIANTS: dict[str, dict[str, Callable]] = { + # Live code, nothing patched. The reproduction gate runs against this. + "v3": {}, + "v4": { + "p1_trend_break": _candidate_p1(), + "p5_volatility": _candidate_p5(P5_VIX_ANCHORS_A), + }, + "v4-vix-b": { + "p1_trend_break": _candidate_p1(), + "p5_volatility": _candidate_p5(P5_VIX_ANCHORS_B), + }, + "v4-p1-capped": { + "p1_trend_break": _capped(_candidate_p1(), 50.0), + "p5_volatility": _candidate_p5(P5_VIX_ANCHORS_A), + }, + "v4-vix-only": {"p5_volatility": _candidate_p5(P5_VIX_ANCHORS_A)}, + "v4-p1-only": {"p1_trend_break": _candidate_p1()}, + # v2 State sensors + the v2 divergence gate that feeds W1. The v2 *Warning + # composition* (F1/F3 fundamentals, 20 of 100 points) is NOT reconstructed, + # so only State statistics and the W1 census are comparable to the published + # v2 figures -- not the Warning score. + "v2_reconstruction": { + "p3_drawdown": _v2_p3_drawdown, + "f2_credit_spreads": _v2_f2_credit_spreads, + # v2 sliced HY_OAS_REFERENCE_YEARS = 10.0 per session. The percentile leg + # ranks the current spread against that window, so replaying it against + # a 700-day slice gives systematically different mid-distribution scores. + "HY_OAS_WINDOW_DAYS": 3653, + }, +} + +# Variants needing the retired divergence formula rather than the live one. +V2_DIVERGENCE_VARIANTS = {"v2_reconstruction"} + + +@contextlib.contextmanager +def patched(overrides: dict[str, Callable]) -> Iterator[None]: + """Swap functions on the service module, then restore exactly.""" + original = {name: getattr(rms, name) for name in overrides} + try: + for name, fn in overrides.items(): + setattr(rms, name, fn) + yield + finally: + for name, fn in original.items(): + setattr(rms, name, fn) + + +# --------------------------------------------------------------------------- +# Inputs +# --------------------------------------------------------------------------- + +def _basket(config: dict) -> list[str]: + return list(config["breadth_basket"]) + + +def _all_symbols(config: dict) -> list[str]: + return list(dict.fromkeys(_basket(config) + [LEADER, CONFIRM, MARKET])) + + +async def _load_prices( + symbols: list[str], start: date, end: date, cache_dir: Path | None, quiet: bool +) -> dict[str, list[tuple[date, float]]]: + cache = None + if cache_dir: + cache_dir.mkdir(parents=True, exist_ok=True) + cache = cache_dir / f"prices-{start}-{end}.json" + if cache.exists(): + raw = json.loads(cache.read_text(encoding="utf-8")) + if set(raw) >= set(symbols): + if not quiet: + print(f"prices: cache hit ({len(raw)} symbols)", flush=True) + return { + s: [(date.fromisoformat(d), float(c)) for d, c in raw[s]] for s in symbols + } + + provider = AlpacaOHLCVProvider(settings.alpaca_api_key, settings.alpaca_api_secret) + out: dict[str, list[tuple[date, float]]] = {} + for index, symbol in enumerate(symbols, 1): + bars = await provider.fetch_ohlcv(symbol, start, end) + out[symbol] = sorted((b.date, float(b.close)) for b in bars) + if not quiet: + print(f" [{index}/{len(symbols)}] {symbol}: {len(out[symbol])} bars", flush=True) + if cache: + cache.write_text( + json.dumps({s: [[d.isoformat(), c] for d, c in v] for s, v in out.items()}), + encoding="utf-8", + ) + return out + + +async def _load_fred(series_id: str, start: date, end: date, cache_dir: Path | None): + cache = cache_dir / f"{series_id}-{start}-{end}.json" if cache_dir else None + if cache and cache.exists(): + raw = json.loads(cache.read_text(encoding="utf-8")) + return [(date.fromisoformat(d), float(v)) for d, v in raw] + series = await rms._fetch_fred_series(series_id, start, end) + if cache and series: + cache.write_text( + json.dumps([[d.isoformat(), v] for d, v in series]), encoding="utf-8" + ) + return series + + +# --------------------------------------------------------------------------- +# Replay +# --------------------------------------------------------------------------- + +def _replay( + variant: str, + prices: dict[str, list[tuple[date, float]]], + vix, oas, config: dict, sessions: list[date], + breadth_series, divergence_by_variant: dict[str, Any], breadth_counts, +) -> list[dict]: + # Divergence is computed outside _compute_index, so the retired v2 gate has + # to be selected here rather than patched onto the module. + divergence_series = divergence_by_variant[ + "v2" if variant in V2_DIVERGENCE_VARIANTS else "live" + ] + rows: list[dict] = [] + with patched(VARIANTS[variant]): + for as_of in sessions: + rows.append( + rms._compute_index( + prices, vix, oas, {}, deepcopy(config), as_of, + breadth_series, divergence_series, breadth_counts, + ) + ) + return rows + + +def _percentile(values: list[float], pct: float) -> float | None: + if not values: + return None + ordered = sorted(values) + k = (len(ordered) - 1) * pct / 100.0 + lo, hi = math.floor(k), math.ceil(k) + if lo == hi: + return ordered[int(k)] + return ordered[lo] + (ordered[hi] - ordered[lo]) * (k - lo) + + +def _sensor_score(row: dict, pillar_id: str, sensor_id: str) -> float | None: + """Search both axes: W1/W2/W3 live under ``warning``, P*/B1/C1/V1 under ``state``.""" + for axis in ("state", "warning"): + for pillar in row[axis]["pillars"]: + if pillar["id"] == pillar_id: + for sensor in pillar["sensors"]: + if sensor["id"] == sensor_id: + return sensor["score"] + return None + + +def _band_shares(scores: list[float], bands: tuple[float, float, float]) -> dict[str, float]: + if not scores: + return {} + counts = {"stable": 0, "watch": 0, "elevated": 0, "breaking": 0} + for score in scores: + counts[rms.band_for(score, bands)] += 1 # bands passed explicitly -- see module docstring + return {k: round(v / len(scores) * 100.0, 1) for k, v in counts.items()} + + +def _stats(rows: list[dict], label: str) -> dict: + states = [r["state"]["score"] for r in rows if r["state"]["score"] is not None] + warnings = [r["warning"]["score"] for r in rows if r["warning"]["score"] is not None] + + def pegged(pillar: str, sensor: str) -> int: + return sum(1 for r in rows if (_sensor_score(r, pillar, sensor) or 0) >= 100.0) + + argmax_sole, argmax_tied, ties = {"P1": 0, "P2": 0, "P3": 0}, {"P1": 0, "P2": 0, "P3": 0}, 0 + for row in rows: + legs = {s: _sensor_score(row, "price", s) for s in ("P1", "P2", "P3")} + live = {k: v for k, v in legs.items() if v is not None} + if not live: + continue + top = max(live.values()) + winners = [k for k, v in live.items() if v == top] + if len(winners) > 1: + ties += 1 + for w in winners: + argmax_tied[w] += 1 + if len(winners) == 1: + argmax_sole[winners[0]] += 1 + + return { + "label": label, + "sessions": len(rows), + "state": { + "avg": round(statistics.fmean(states), 2) if states else None, + "median": round(statistics.median(states), 2) if states else None, + "p80": round(_percentile(states, 80), 2) if states else None, + "p90": round(_percentile(states, 90), 2) if states else None, + "max": round(max(states), 2) if states else None, + "scored": len(states), + }, + "warning_avg": round(statistics.fmean(warnings), 2) if warnings else None, + "saturation_census": { + "p3_pegged": pegged("price", "P3"), + "p2_pegged": pegged("price", "P2"), + "v1_pegged": pegged("volatility", "V1"), + "p1_pegged": pegged("price", "P1"), + }, + "price_argmax_sole": argmax_sole, + "price_argmax_tie_inclusive": argmax_tied, + "price_argmax_ties": ties, + "w1_nonzero_sessions": sum( + 1 for r in rows if (_sensor_score(r, "breadth_divergence", "W1") or 0) > 0 + ), + "band_shares_current": _band_shares(states, rms.STATE_BANDS), + } + + +# --------------------------------------------------------------------------- +# Gates +# --------------------------------------------------------------------------- + +def _completeness_gates( + prices: dict, symbols: list[str], sessions: list[date], breadth_counts: dict, basket_size: int +) -> list[dict]: + """Without these, every gate below can pass on partial data. + + _breadth_with_counts publishes on min_tickers=20, so 20 of 30 basket names + still yields 100% State coverage and a plausible W1 count. + """ + first, last = sessions[0], sessions[-1] + missing = [s for s in symbols if not prices.get(s)] + thin = [ + s for s in symbols + if len([d for d, _ in prices.get(s, []) if d < first]) < 252 + ] + truncated = [s for s in symbols if not prices.get(s) or prices[s][-1][0] < last] + short_basket = sorted( + d.isoformat() for d in sessions if breadth_counts.get(d, 0) != basket_size + ) + return [ + {"gate": "symbols_fetched", "expected": len(symbols), + "measured": len(symbols) - len(missing), "passed": not missing, "detail": missing}, + {"gate": "per_symbol_warmup_252_bars", "expected": "all", + "measured": len(symbols) - len(thin), "passed": not thin, "detail": thin}, + {"gate": "per_symbol_reaches_last_session", "expected": last.isoformat(), + "measured": len(symbols) - len(truncated), "passed": not truncated, "detail": truncated}, + {"gate": "breadth_counts_full_basket", "expected": basket_size, + "measured": f"{len(sessions) - len(short_basket)}/{len(sessions)} sessions", + "passed": not short_basket, "detail": short_basket[:20]}, + ] + + +def _pipeline_gates(rows: list[dict], sessions: list[date], expected_first: str | None) -> list[dict]: + coverage_bad = [ + r["date"] for r in rows if (r["state"]["coverage"] or 0) < 100.0 + ] + w1_available = sum( + 1 for r in rows if _sensor_score(r, "breadth_divergence", "W1") is not None + ) + stale = [r["date"] for r in rows if r["data_quality"]["stale_inputs"]] + gates = [ + {"gate": "sessions_scored", "expected": PUBLISHED["sessions"], + "measured": len(rows), "passed": len(rows) == PUBLISHED["sessions"]}, + {"gate": "last_scored_date", "expected": PUBLISHED["window_end"], + "measured": rows[-1]["date"], "passed": rows[-1]["date"] == PUBLISHED["window_end"]}, + # Availability, not the published "W1 live 108" -- that figure counts + # NONZERO sessions under v2's hard price gate and is checked there. + {"gate": "w1_available_every_session", "expected": len(rows), + "measured": w1_available, "passed": w1_available == len(rows)}, + {"gate": "state_coverage_100_every_row", "expected": 0, + "measured": len(coverage_bad), "passed": not coverage_bad, "detail": coverage_bad[:20]}, + {"gate": "no_stale_inputs", "expected": 0, + "measured": len(stale), "passed": not stale, "detail": stale[:20]}, + ] + # sessions_scored alone is tautological when the harness slices the tail of + # leader_series -- the calendar is only really checked by the start date. + if expected_first: + gates.append({ + "gate": "first_scored_date", "expected": expected_first, + "measured": rows[0]["date"], "passed": rows[0]["date"] == expected_first, + }) + return gates + + +def _invariant_gate(v3_rows: list[dict], v4_rows: list[dict]) -> dict: + """state_v4 <= state_v3 on every aligned row. + + Provable, not heuristic: graduated P1 never exceeds binary P1, anchored VIX + never exceeds (vix-15)/15*100, max() is monotone in its arguments, and no + other State sensor or weight changes. A violation means the harness is + mis-wired, not that the calibration is interesting. + """ + violations = [] + for a, b in zip(v3_rows, v4_rows): + assert a["date"] == b["date"], "row misalignment" + s3, s4 = a["state"]["score"], b["state"]["score"] + if s3 is not None and s4 is not None and s4 > s3 + 1e-9: + violations.append({"date": a["date"], "v3": s3, "v4": s4}) + return { + "gate": "state_v4_le_v3_every_row", "expected": 0, + "measured": len(violations), "passed": not violations, "detail": violations[:20], + } + + +def _soft_gates(stats: dict, variant: str) -> list[dict]: + if variant == "v3": + # The doc states no v3 average: its "-0.4" is measured against + # v3-with-the-percentile-leg, not against v2. Only the max is checkable. + pairs = [("v3_state_max", stats["state"]["max"], 0.5)] + else: + pairs = [("v2_state_avg", stats["state"]["avg"], 0.3), + ("v2_state_p80", stats["state"]["p80"], 0.5), + ("v2_state_max", stats["state"]["max"], 0.5), + ("v2_p3_pegged", stats["saturation_census"]["p3_pegged"], 0), + ("w1_live_sessions", stats["w1_nonzero_sessions"], 0)] + out = [] + for key, measured, tol in pairs: + expected = PUBLISHED[key] + ok = measured is not None and abs(measured - expected) <= tol + out.append({"gate": key, "expected": expected, "measured": measured, + "tolerance": tol, "passed": ok}) + return out + + +# --------------------------------------------------------------------------- +# Scenarios -- pillar arithmetic, stated as explicit sensor scores +# --------------------------------------------------------------------------- + +def _scenarios(vix_anchors, p1_anchors) -> list[dict]: + """The meaning anchors for the band choice, machine-checked rather than prose. + + Stated as explicit sensor scores because drawdown + VIX + OAS does not + determine State: the price pillar is max(P1, P2, P3) and P2 is set by the + 50/200-DMA gap, which no drawdown figure implies. + """ + def state(p1, p2, p3, breadth, c1, v1): + price = max(p1, p2, p3) + return round((price * 40 + breadth * 25 + c1 * 20 + v1 * 15) / 100, 2) + + def p1_at(pct_below): + return round(rms._interpolate(pct_below, p1_anchors), 2) if pct_below > 0 else 0.0 + + def p3_at(dd): + return round(rms._interpolate(dd, rms.P3_DRAWDOWN_ANCHORS), 2) + + def v1_at(vix): + return round(rms._interpolate(vix, vix_anchors), 2) + + rows = [ + ("S1 ordinary tape", 0.0, 0.0, p3_at(3), rms.breadth_level_score(65), 0.0, v1_at(16)), + ("S2 10% correction, calm credit", p1_at(2), 0.0, p3_at(10), rms.breadth_level_score(35), 0.0, v1_at(24)), + ("S3a 2022-style, calm credit, no death cross", p1_at(20), 0.0, p3_at(35), rms.breadth_level_score(8), 0.0, v1_at(32)), + ("S3b 2022-style, calm credit, death cross", p1_at(20), 100.0, p3_at(35), rms.breadth_level_score(8), 0.0, v1_at(32)), + ("S4 credit event on top", p1_at(25), 100.0, p3_at(40), rms.breadth_level_score(5), rms.f2_credit_spreads([6.0]), v1_at(45)), + ("S5 March 2020, everything pegged", 100.0, 100.0, 100.0, 100.0, 100.0, 100.0), + ] + return [ + {"scenario": name, "P1": p1, "P2": p2, "P3": p3, "price": max(p1, p2, p3), + "breadth": br, "C1": c1, "V1": v1, "state": state(p1, p2, p3, br, c1, v1)} + for name, p1, p2, p3, br, c1, v1 in rows + ] + + +def _markdown(report: dict) -> str: + """Scannable sibling to the JSON. Never written over a curated docs/ file.""" + lines = ["# Regime Monitor v4 calibration", ""] + lines.append(f"Generated {report['generated_at']} at `{report['git_rev']}`, " + f"{report['provenance']['scored_range'][0]} → {report['provenance']['scored_range'][1]}.") + lines += ["", "## Hard gates", "", "| gate | expected | measured | |", "|---|---|---|---|"] + for g in report["hard_gates"]: + lines.append(f"| {g['gate']} | {g['expected']} | {g['measured']} | {'ok' if g['passed'] else '**FAIL**'} |") + lines += ["", "## Distributions", "", "| variant | avg | median | p80 | p90 | max |", "|---|---|---|---|---|---|"] + for name, v in report["variants"].items(): + st = v["state"] + lines.append(f"| {name} | {st['avg']} | {st['median']} | {st['p80']} | {st['p90']} | {st['max']} |") + lines += ["", "## Saturation census (sessions pegged at 100)", "", + "| variant | P1 | P2 | P3 | V1 |", "|---|---|---|---|---|"] + for name, v in report["variants"].items(): + c = v["saturation_census"] + lines.append(f"| {name} | {c['p1_pegged']} | {c['p2_pegged']} | {c['p3_pegged']} | {c['v1_pegged']} |") + for name, v in report["variants"].items(): + if v.get("soft_gates"): + lines += ["", f"## Reproduction gates — {name}", "", + "| figure | published | measured | |", "|---|---|---|---|"] + for g in v["soft_gates"]: + lines.append(f"| {g['gate']} | {g['expected']} | {g['measured']} | {'ok' if g['passed'] else '**miss**'} |") + if "v4" in report["variants"]: + lines += ["", "## v4 band-share grid (watch 20 / elevated 50)", "", + "| breaking | stable | watch | elevated | breaking |", "|---|---|---|---|---|"] + for row in report["variants"]["v4"]["band_grid"]: + w, e, b = row["bands"] + if w == 20.0 and e == 50.0: + sh = row["shares"] + lines.append(f"| {b:.0f} | {sh['stable']} | {sh['watch']} | {sh['elevated']} | {sh['breaking']} |") + lines += ["", "## Scenarios (pillar arithmetic, explicit sensor scores)", "", + "| scenario | price | breadth | C1 | V1 | State |", "|---|---|---|---|---|---|"] + for sc in report["scenarios"]["vix_a"]: + lines.append(f"| {sc['scenario']} | {sc['price']} | {sc['breadth']} | {sc['C1']} | {sc['V1']} | **{sc['state']}** |") + lines += ["", f"Recommendation: `{report['v4_recommendation']}`", ""] + return "\n".join(lines) + + +def _band_grid(states: list[float]) -> list[dict]: + grid = [] + for watch in (15.0, 20.0, 25.0): + for elevated in (40.0, 50.0): + for breaking in (60.0, 65.0, 70.0): + grid.append({ + "bands": [watch, elevated, breaking], + "shares": _band_shares(states, (watch, elevated, breaking)), + }) + return grid + + +# --------------------------------------------------------------------------- +# Main +# --------------------------------------------------------------------------- + +def _parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--end", default=PUBLISHED["window_end"]) + p.add_argument("--sessions", type=int, default=PUBLISHED["sessions"]) + p.add_argument("--history-days", type=int, default=1200, help="matches production _fetch_prices") + p.add_argument("--oas-window-days", type=int, default=rms.HY_OAS_WINDOW_DAYS, + help="per-session slice for the canonical run") + p.add_argument("--oas-fetch-days", type=int, default=int(365.25 * 13), + help="fetch range; matches v2's request (ICE truncates to ~3y)") + p.add_argument("--methodology", default="v3,v4") + p.add_argument("--expected-first-session", default=None, + help="assert the first replayed date; sessions count alone is tautological") + p.add_argument("--cache-dir", default=None) + p.add_argument("--out", default=None) + p.add_argument("--quiet", action="store_true") + return p.parse_args() + + +def _git_rev() -> str: + try: + return subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=ROOT, + capture_output=True, text=True, check=True).stdout.strip() + except Exception: + return "unknown" + + +async def _main() -> int: + args = _parse_args() + end = date.fromisoformat(args.end) + cache_dir = Path(args.cache_dir) if args.cache_dir else None + config = deepcopy(rms.DEFAULT_CONFIG) + symbols = _all_symbols(config) + variants = [v.strip() for v in args.methodology.split(",") if v.strip()] + unknown = [v for v in variants if v not in VARIANTS] + if unknown: + print(f"unknown variant(s): {unknown}; known: {sorted(VARIANTS)}", file=sys.stderr) + return 2 + + if not args.quiet: + print(f"fetching {len(symbols)} symbols from Alpaca...", flush=True) + prices = await _load_prices(symbols, end - timedelta(days=args.history_days), end, cache_dir, args.quiet) + vix = await _load_fred("VIXCLS", end - timedelta(days=args.history_days), end, cache_dir) + oas = await _load_fred("BAMLH0A0HYM2", end - timedelta(days=args.oas_fetch_days), end, cache_dir) + + leader = prices.get(LEADER, []) + if not leader: + print("no leader (SMH) price data — cannot replay", file=sys.stderr) + return 2 + sessions = [d for d, _ in leader if d <= end][-args.sessions:] + + breadth, breadth_counts = breadth_service._breadth_with_counts( + {s: prices[s] for s in _basket(config) if prices.get(s)}, window=200, min_tickers=20 + ) + # Required glue: _item_asof breaks on the first date > as_of, so unsorted + # input silently returns a wrong value rather than erroring. + breadth_series = rms._mapping_series(breadth) + divergence_by_variant = { + "live": rms._mapping_series(breadth_service.compute_divergence_series(breadth, leader)), + "v2": rms._mapping_series(_v2_divergence_series(breadth, leader)), + } + + if args.oas_window_days != rms.HY_OAS_WINDOW_DAYS: + rms.HY_OAS_WINDOW_DAYS = args.oas_window_days + + results: dict[str, Any] = {} + rows_by_variant: dict[str, list[dict]] = {} + for variant in variants: + if not args.quiet: + print(f"replaying {variant} over {len(sessions)} sessions...", flush=True) + rows = _replay(variant, prices, vix, oas, config, sessions, + breadth_series, divergence_by_variant, breadth_counts) + rows_by_variant[variant] = rows + results[variant] = _stats(rows, variant) + results[variant]["soft_gates"] = _soft_gates(results[variant], variant) \ + if variant in ("v3", "v2_reconstruction") else [] + states = [r["state"]["score"] for r in rows if r["state"]["score"] is not None] + if variant.startswith("v4"): + results[variant]["band_grid"] = _band_grid(states) + + canonical = rows_by_variant.get("v3") or next(iter(rows_by_variant.values())) + hard_gates = _completeness_gates(prices, symbols, sessions, breadth_counts, len(_basket(config))) + hard_gates += _pipeline_gates(canonical, sessions, args.expected_first_session) + if "v3" in rows_by_variant and "v4" in rows_by_variant: + hard_gates.append(_invariant_gate(rows_by_variant["v3"], rows_by_variant["v4"])) + + blocked_by = [g["gate"] for g in hard_gates if not g["passed"]] + provisional = any( + not g["passed"] for v in results.values() for g in v.get("soft_gates", []) + ) + + report = { + "generated_at": datetime.now().isoformat(timespec="seconds"), + "git_rev": _git_rev(), + "params": vars(args), + "provenance": { + "symbols": {s: {"bars": len(prices.get(s, [])), + "first": prices[s][0][0].isoformat() if prices.get(s) else None, + "last": prices[s][-1][0].isoformat() if prices.get(s) else None} + for s in symbols}, + "vix": {"points": len(vix or []), + "first": vix[0][0].isoformat() if vix else None, + "last": vix[-1][0].isoformat() if vix else None}, + "oas": {"points": len(oas or []), + "first": oas[0][0].isoformat() if oas else None, + "last": oas[-1][0].isoformat() if oas else None}, + "scored_range": [sessions[0].isoformat(), sessions[-1].isoformat()], + }, + "hard_gates": hard_gates, + "blocked_by": blocked_by, + "provisional": provisional, + "variants": results, + "scenarios": { + "note": "pillar arithmetic on explicit sensor scores; State weights unchanged", + "vix_a": _scenarios(P5_VIX_ANCHORS_A, P1_TREND_BREAK_ANCHORS), + }, + "diagnostics": { + "vix_top10": sorted(((d.isoformat(), v) for d, v in (vix or [])), + key=lambda x: -x[1])[:10], + "candidate_anchors": { + "P5_VIX_ANCHORS_A": P5_VIX_ANCHORS_A, + "P5_VIX_ANCHORS_B": P5_VIX_ANCHORS_B, + "P1_TREND_BREAK_ANCHORS": P1_TREND_BREAK_ANCHORS, + }, + }, + # Structurally impossible to read a recommendation out of a run whose + # pipeline did not validate. + "v4_recommendation": None if blocked_by else { + "state_bands_candidate": [20.0, 50.0, 65.0], + "provisional": provisional, + "note": "confirm against band_grid + scenarios before shipping", + }, + } + + stamp = datetime.now().strftime("%Y%m%d-%H%M%S") + out = Path(args.out) if args.out else ROOT / "reports" / f"regime-monitor-v4-calibration-{stamp}.json" + out.parent.mkdir(parents=True, exist_ok=True) + tmp = out.with_suffix(out.suffix + ".tmp") + tmp.write_text(json.dumps(report, indent=2, default=str), encoding="utf-8") + tmp.replace(out) + md = out.with_suffix(".md") + md_tmp = md.with_suffix(".md.tmp") + md_tmp.write_text(_markdown(report), encoding="utf-8") + md_tmp.replace(md) + if not args.quiet: + print(f"wrote {out}", flush=True) + for gate in hard_gates: + mark = "ok " if gate["passed"] else "FAIL" + print(f" [{mark}] {gate['gate']}: expected {gate['expected']}, got {gate['measured']}") + if blocked_by: + print(f"HARD GATES FAILED: {blocked_by} — no v4 recommendation emitted", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(asyncio.run(_main())) diff --git a/tests/unit/test_regime_calibration_script.py b/tests/unit/test_regime_calibration_script.py new file mode 100644 index 0000000..7ddcb2c --- /dev/null +++ b/tests/unit/test_regime_calibration_script.py @@ -0,0 +1,140 @@ +"""The calibration harness's pure helpers. No network, no data. + +The harness's whole value is that its numbers can be trusted, so the parts that +decide whether a run is trustworthy — the variant patching and the gate +evaluation — are worth pinning even though the script is research-only. +""" + +import importlib.util +from pathlib import Path + +import pytest + +from app.services import regime_monitor_service as rms + +_SPEC = importlib.util.spec_from_file_location( + "regime_calibration", + Path(__file__).resolve().parents[2] / "scripts" / "run_regime_monitor_calibration.py", +) +calib = importlib.util.module_from_spec(_SPEC) +_SPEC.loader.exec_module(calib) + + +class TestVariantPatching: + def test_patching_restores_module_state_exactly(self): + """A leaked patch would silently contaminate every later variant.""" + before = { + name: getattr(rms, name) + for name in ("p1_trend_break", "p5_volatility", "p3_drawdown", + "f2_credit_spreads", "HY_OAS_WINDOW_DAYS") + } + with calib.patched(calib.VARIANTS["v2_reconstruction"]): + assert rms.p3_drawdown is not before["p3_drawdown"] + assert rms.HY_OAS_WINDOW_DAYS == 3653 + for name, original in before.items(): + assert getattr(rms, name) is original or getattr(rms, name) == original + + def test_patching_restores_even_when_the_body_raises(self): + original = rms.p5_volatility + with pytest.raises(RuntimeError): + with calib.patched(calib.VARIANTS["v4"]): + raise RuntimeError("boom") + assert rms.p5_volatility is original + + def test_v3_variant_patches_nothing(self): + """The reproduction gate must run against live code, not a copy.""" + assert calib.VARIANTS["v3"] == {} + + +class TestCandidateFormulas: + def test_graduated_trend_break_never_exceeds_the_binary_one(self): + """Half of the provable state_v4 <= state_v3 invariant.""" + closes = [100.0] * 200 + for last in (105.0, 100.0, 99.0, 92.0, 80.0, 50.0): + series = closes[:-1] + [last] + v4 = calib._candidate_under_200(series) + v3 = rms._under_200(series) + assert v4 <= v3, f"close={last}: v4 {v4} > v3 {v3}" + + def test_anchored_vix_never_exceeds_the_live_formula(self): + p5 = calib._candidate_p5(calib.P5_VIX_ANCHORS_A) + for vix in (10, 15, 17, 20, 25, 30, 40, 55, 82): + assert p5(vix) <= rms.p5_volatility(vix), f"vix={vix}" + + def test_the_vix_table_keeps_resolving_past_thirty(self): + p5 = calib._candidate_p5(calib.P5_VIX_ANCHORS_A) + assert p5(30) < p5(40) < p5(50) < p5(55) == 100.0 + assert rms.p5_volatility(30) == rms.p5_volatility(82) == 100.0 # the defect + + def test_a_shallow_break_no_longer_pegs(self): + closes = [100.0] * 199 + [98.0] # ~2% below a flat 200-DMA + assert rms._under_200(closes) == 100.0 + assert calib._candidate_under_200(closes) < 40.0 + + def test_candidate_tables_are_well_formed(self): + for table in (calib.P5_VIX_ANCHORS_A, calib.P5_VIX_ANCHORS_B, + calib.P1_TREND_BREAK_ANCHORS): + xs = [x for x, _ in table] + ys = [y for _, y in table] + assert xs == sorted(xs) and len(set(xs)) == len(xs) + assert ys == sorted(ys) + assert 0.0 <= min(ys) and max(ys) <= 100.0 + + +class TestGates: + def _row(self, date_str, state, coverage=100.0, w1=5.0): + return { + "date": date_str, + "state": {"score": state, "coverage": coverage, "pillars": []}, + "warning": {"score": 10.0, "coverage": 100.0, "pillars": [ + {"id": "breadth_divergence", "sensors": [{"id": "W1", "score": w1}]} + ]}, + "data_quality": {"stale_inputs": []}, + } + + def test_the_invariant_gate_catches_a_v4_row_above_its_v3_row(self): + v3 = [self._row("2026-01-02", 40.0), self._row("2026-01-05", 50.0)] + v4 = [self._row("2026-01-02", 38.0), self._row("2026-01-05", 55.0)] + gate = calib._invariant_gate(v3, v4) + assert gate["passed"] is False + assert gate["detail"][0]["date"] == "2026-01-05" + + def test_the_invariant_gate_passes_when_v4_is_never_higher(self): + v3 = [self._row("2026-01-02", 40.0)] + v4 = [self._row("2026-01-02", 40.0)] + assert calib._invariant_gate(v3, v4)["passed"] is True + + def test_sensor_lookup_searches_both_axes(self): + """W1 lives under warning; a state-only search silently returns None and + made the W1 census read 0.""" + row = self._row("2026-01-02", 40.0, w1=7.5) + assert calib._sensor_score(row, "breadth_divergence", "W1") == 7.5 + + +class TestBandShares: + def test_shares_use_the_candidate_bands_not_the_imported_default(self): + """band_for binds bands=STATE_BANDS as an import-time default, so the + grid must pass candidates explicitly or every row scores identically.""" + scores = [10.0, 30.0, 55.0, 75.0] + loose = calib._band_shares(scores, (20.0, 50.0, 80.0)) + tight = calib._band_shares(scores, (20.0, 50.0, 60.0)) + assert loose["breaking"] == 0.0 + assert tight["breaking"] == 25.0 + + def test_shares_sum_to_one_hundred(self): + shares = calib._band_shares([5.0, 25.0, 55.0, 85.0], (20.0, 50.0, 80.0)) + assert sum(shares.values()) == pytest.approx(100.0) + + +class TestScenarios: + def test_the_decisive_scenario_is_computed_not_asserted(self): + rows = {s["scenario"].split()[0]: s for s in + calib._scenarios(calib.P5_VIX_ANCHORS_A, calib.P1_TREND_BREAK_ANCHORS)} + # A 2022-style AI/tech drawdown with genuinely calm credit. This is the + # meaning anchor for the breaking threshold, so it is machine-checked. + assert rows["S3a"]["C1"] == 0.0 + assert rows["S3a"]["state"] == pytest.approx(70.33, abs=0.01) + assert rows["S3b"]["state"] == pytest.approx(74.00, abs=0.01) + # ...and it must clear the chosen threshold under either P2 assumption. + assert min(rows["S3a"]["state"], rows["S3b"]["state"]) > 65.0 + assert rows["S1"]["state"] < 20.0