From 333989eeab96fc8d57e75b2d2b63b315ecdbf3d6 Mon Sep 17 00:00:00 2001 From: Dennis Thiessen Date: Thu, 13 Aug 2026 11:14:33 +0200 Subject: [PATCH] feat(risk-monitor): measure the rule that fires, and give fundamentals their own channel The Warning study measured a fitted percentile crossing that nothing consumes. What reaches Telegram is a quadrant change: fixed 50/40 dividers, hysteresis, two-session confirmation, 3-day cooldown. Those thresholds are constants, not fits, so there is no training set to protect and all 11 detected corrections are evaluable instead of the 4 that fell in a holdout. Replaying it: 1/10 corrections, 0.9 false alarms/year. Random alarms at the same firing rate match or beat that in 65% of draws. The panel now carries ablations (does the quadrant machinery earn its place?), external baselines (does the score earn its complexity?), and that null, because a bare "2 of 4" was unreadable in either direction. Nothing in the alert path was retuned on the strength of it. Fundamentals become a third channel rather than a term in either score. v3 cut them arguing 12+8 of 100 points "could not change any published conclusion" -- true only when every technical sensor reads zero; weighted they moved the bar for the 40 divider from 40 to 25. But no fusion weight is measurable either: with ~10 events and no fundamental history, any weight is a policy preference presented as a measurement. So the read is a categorical state (supportive/neutral/adverse/ unknown) with an evidence grade, derived by fixed rules from stored facts, read by confluence. The LLM extracts and explains; it does not score. Absence stays absence throughout. `unknown` is unreachable by averaging, a stale or empty observation may display but never confirm, extraction failures map to `unknown` rather than `mixed`, and the study rows are coverage-matched and marked not-measurable until enough corrections are covered -- otherwise a fortnight of observations renders as 0/10 and reads as a failed test. Observations become a real time series (migration 033); they lived in a single overwritten settings slot, so no history existed to replay. Pre-rename snapshots are adapted rather than discarded. METHODOLOGY stays v4 -- no score changed -- so no reseed; STUDY_SCHEMA moves to 3 and discards the cached report. Post-deploy: re-run Event Study from Admin -> Jobs. The panel reads "not run yet" until then. Co-Authored-By: Claude Opus 5 --- .../033_regime_fundamental_observations.py | 67 ++ app/models/__init__.py | 2 + app/models/regime_fundamental_observation.py | 44 + app/scheduler.py | 10 +- app/services/alert_service.py | 108 +++ app/services/event_study_service.py | 845 ++++++++++++++++-- app/services/regime_monitor_service.py | 457 +++++++++- docs/research/regime-monitor-v4.md | 449 +++++++++- .../src/components/regime/RegimeChart.tsx | 191 +++- frontend/src/components/ui/Disclosure.tsx | 2 +- frontend/src/lib/regime.ts | 66 +- frontend/src/lib/types.ts | 132 ++- frontend/src/pages/RegimePage.tsx | 632 +++++++++---- frontend/src/styles/globals.css | 19 + scripts/run_regime_monitor_calibration.py | 9 + tests/unit/test_event_study.py | 399 ++++++++- tests/unit/test_regime_monitor.py | 322 ++++++- tests/unit/test_regime_quadrant_alert.py | 143 +++ 18 files changed, 3494 insertions(+), 403 deletions(-) create mode 100644 alembic/versions/033_regime_fundamental_observations.py create mode 100644 app/models/regime_fundamental_observation.py diff --git a/alembic/versions/033_regime_fundamental_observations.py b/alembic/versions/033_regime_fundamental_observations.py new file mode 100644 index 0000000..1352add --- /dev/null +++ b/alembic/versions/033_regime_fundamental_observations.py @@ -0,0 +1,67 @@ +"""Point-in-time history for the sourced fundamental observation + +Revision ID: 033 +Revises: 032 +Create Date: 2026-08-12 00:00:00.000000 + +The hyperscaler capex / "good news, stock down" read lived in a single +``SystemSetting`` slot, so each refresh overwrote the last and no history +existed. The read is now a categorical channel reported alongside State and +Warning (never a term in either), and a channel with no history cannot be +replayed: a snapshot rebuild would record every historical session as if nothing +had ever been observed, and the event study could not measure the channel at all. + +Keyed on ``effective_date`` (the session the observation becomes usable on, +normally the next weekday) rather than ``fetched_at``, because that is the gate +that stops a rebuild stamping today's reading onto historical rows. + +The table starts empty. ``update_regime_monitor`` records the currently stored +observation on its next run, so a deployment does not lose the live reading — +but genuine history does not exist and cannot be invented here. Backfilling it +from the SEC capex line and earnings-date reactions is separate work; until then +every historical session reads ``unknown``, which is the honest value rather than +a guessed one. +""" +from typing import Sequence, Union + +from alembic import op +import sqlalchemy as sa + + +revision: str = "033" +down_revision: Union[str, None] = "032" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "regime_fundamental_observations", + sa.Column("id", sa.Integer(), nullable=False), + sa.Column("effective_date", sa.Date(), nullable=False), + sa.Column("f1_score", sa.Float(), nullable=True), + sa.Column("f3_score", sa.Float(), nullable=True), + sa.Column("capex_json", sa.Text(), nullable=False), + sa.Column("good_news_stock_down", sa.String(length=10), nullable=False), + sa.Column("reasoning", sa.Text(), nullable=True), + sa.Column("source", sa.String(length=30), nullable=False), + sa.Column("fetched_at", sa.DateTime(timezone=True), nullable=False), + sa.Column("created_at", sa.DateTime(timezone=True), nullable=False), + sa.PrimaryKeyConstraint("id"), + sa.UniqueConstraint( + "effective_date", name="uq_regime_fundamental_observations_effective_date" + ), + ) + op.create_index( + "ix_regime_fundamental_observations_effective_date", + "regime_fundamental_observations", + ["effective_date"], + ) + + +def downgrade() -> None: + op.drop_index( + "ix_regime_fundamental_observations_effective_date", + table_name="regime_fundamental_observations", + ) + op.drop_table("regime_fundamental_observations") diff --git a/app/models/__init__.py b/app/models/__init__.py index bd003f7..4628141 100644 --- a/app/models/__init__.py +++ b/app/models/__init__.py @@ -14,6 +14,7 @@ from app.models.settings import SystemSetting, IngestionProgress from app.models.alert import AlertLog from app.models.paper_trade import PaperTrade from app.models.regime_snapshot import RegimeSnapshot +from app.models.regime_fundamental_observation import RegimeFundamentalObservation from app.models.benchmark_price import BenchmarkPrice from app.models.signal_context_snapshot import SignalContextSnapshot from app.models.system_event import SystemEvent @@ -39,6 +40,7 @@ __all__ = [ "AlertLog", "PaperTrade", "RegimeSnapshot", + "RegimeFundamentalObservation", "BenchmarkPrice", "SignalContextSnapshot", "SystemEvent", diff --git a/app/models/regime_fundamental_observation.py b/app/models/regime_fundamental_observation.py new file mode 100644 index 0000000..602a50a --- /dev/null +++ b/app/models/regime_fundamental_observation.py @@ -0,0 +1,44 @@ +from datetime import date as date_type +from datetime import datetime + +from sqlalchemy import Date, DateTime, Float, String, Text +from sqlalchemy.orm import Mapped, mapped_column + +from app.database import Base + + +class RegimeFundamentalObservation(Base): + """Point-in-time record of the sourced hyperscaler capex / earnings read. + + One row per ``effective_date`` (unique, upserted). Before this table the + observation lived in a single ``SystemSetting`` slot, so every refresh + overwrote the previous one and no history existed at all — which made the + read impossible to replay, impossible to backtest, and meant a snapshot + rebuild could only ever score historical sessions as if nothing had been + observed. + + The read is a categorical channel reported beside State and Warning, never a + term in either, so this series is not a scoring input. It is the record that + makes the channel replayable at all -- and the only route to eventually + testing whether it improves prediction conditional on Warning, which is the + one thing that could justify combining the channels later. + + ``effective_date`` rather than ``fetched_at`` is the key: it is the session + the observation becomes usable on (normally the next weekday), and the gate + that stops a rebuild stamping today's reading onto historical rows. + """ + + __tablename__ = "regime_fundamental_observations" + + id: Mapped[int] = mapped_column(primary_key=True) + effective_date: Mapped[date_type] = mapped_column( + Date, nullable=False, unique=True, index=True + ) + f1_score: Mapped[float | None] = mapped_column(Float, nullable=True) + f3_score: Mapped[float | None] = mapped_column(Float, nullable=True) + capex_json: Mapped[str] = mapped_column(Text, nullable=False) + good_news_stock_down: Mapped[str] = mapped_column(String(10), nullable=False) + reasoning: Mapped[str | None] = mapped_column(Text, nullable=True) + source: Mapped[str] = mapped_column(String(30), nullable=False) + fetched_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False) + created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False) diff --git a/app/scheduler.py b/app/scheduler.py index a69c475..07ac40b 100644 --- a/app/scheduler.py +++ b/app/scheduler.py @@ -1351,8 +1351,11 @@ async def run_event_study_job() -> None: report = await run_event_study_and_store(db) _runtime_progress(job_name, processed=1, total=1) + shipped = report.get("shipped") or {} if report.get("available"): - metrics = report.get("metrics") or {} + # The shipped quadrant rule is the headline; the fitted-threshold + # variant lives under report["fitted"] and is not what fires. + metrics = shipped.get("metrics") or {} msg = ( f"{metrics.get('events_warned', 0)}/{metrics.get('events', 0)} warned, " f"{metrics.get('false_alarms_per_year', 0)} false alarms/year" @@ -1360,7 +1363,10 @@ async def run_event_study_job() -> None: else: msg = report.get("reason", "no data") _runtime_finish(job_name, "completed", processed=1, total=1, message=msg) - _log_event(logging.INFO, "job_complete", job=job_name, events=len(report.get("events", []))) + _log_event( + logging.INFO, "job_complete", job=job_name, + events=len(shipped.get("events") or []), + ) except Exception as exc: _runtime_finish(job_name, "error", processed=0, total=1, message=str(exc)) _log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc)) diff --git a/app/services/alert_service.py b/app/services/alert_service.py index 3c07b42..afad1db 100644 --- a/app/services/alert_service.py +++ b/app/services/alert_service.py @@ -97,6 +97,14 @@ SIGNAL_BUNDLE_MAX_CHARS = 3900 # Telegram limit is 4096; keep room for HTML par # Hysteresis (a deadband around each divider) stops a point sitting on a boundary # from flip-flopping; the cooldown caps how often a genuine change can re-alert. QUAD_TYPE = "regime_quadrant" +# The fundamental channel gets its own alerts rather than shifting a score: +# "the context changed" and "both channels are elevated" are different facts from +# "the market axes moved", and fusing them into one number would destroy exactly +# the information an operator uses to decide how much the alert is worth. +FUND_TYPE = "regime_fundamental" +CONFLUENCE_TYPE = "regime_confluence" +# States that count as fundamental risk for the confluence test. +FUND_ADVERSE = "adverse" QUAD_X_DIV = 50.0 # v3 State divider (backend response is authoritative) QUAD_Y_DIV = 40.0 # v3 Warning divider; the axes have different ranges QUAD_MARGIN = 5.0 # half-width of the hysteresis deadband around each divider @@ -859,16 +867,111 @@ async def _collect_regime_quadrant(db: AsyncSession) -> list[tuple[str, str]]: ) else: metrics = f"State {x:.0f} · Warning {y:.0f}" + # The fundamental channel is reported, never added in: this alert is about + # the two market axes, and the context is stated beside them so a reader can + # judge confluence themselves rather than being handed a fused number. + context = data.get("fundamental_context") or {} + context_line = ( + f"fundamentals: {context.get('state', 'unknown')} " + f"({context.get('evidence_quality', 'unavailable')})\n" + ) text = ( f"🧭 AI/Tech risk quadrant change\n" f"{QUAD_LABELS.get(prev, prev)} → {QUAD_LABELS.get(new_q, new_q)}\n" f"{metrics}\n" + f"{context_line}" f"coverage: state {state.get('coverage'):.0f}% / warning {warning.get('coverage'):.0f}%\n" f"Risk thermometer - not a trade signal." ) return [(_quadrant_log_key(new_q, x, y, basket_hash), text)] +async def _last_logged_key(db: AsyncSession, alert_type: str) -> str | None: + """Most recent logged key for a type, our baseline for change detection.""" + result = await db.execute( + select(AlertLog.dedup_key) + .where(AlertLog.alert_type == alert_type) + .order_by(AlertLog.created_at.desc()) + .limit(1) + ) + row = result.first() + return row[0] if row else None + + +async def _collect_regime_fundamental(db: AsyncSession) -> list[tuple[str, str, str]]: + """Fundamental-context changes and market/fundamental confluence. + + Two triggers, deliberately separate from the quadrant alert and from each + other, because they answer different questions: *what the evidence says* and + *whether both channels agree*. Neither is derived by moving a score. + + ``unknown`` never alerts. An absence of evidence is not a change in the + evidence, and alerting on it would train the reader to ignore the channel. + Both seed silently on first run, exactly as the quadrant alert does. + """ + from app.services.regime_monitor_service import get_regime_monitor + + data = await get_regime_monitor(db) + if not data.get("available"): + return [] + warning = data.get("warning") or {} + context = data.get("fundamental_context") or {} + state = str(context.get("state") or "unknown") + # `usable`, not `available`: the state is deliberately preserved past its + # staleness horizon so the card can keep showing the last thing observed, and + # an observation whose extraction failed is fresh but knows nothing. Neither + # may confirm anything — without this gate a months-old adverse read silently + # corroborates every new Warning crossing forever, which is the strongest + # claim this channel makes and the one it has least right to make. + usable = bool(context.get("usable")) + score = warning.get("score") + + quality = data.get("data_quality") or {} + if not quality.get("is_fresh") or float(warning.get("coverage") or 0) < 75: + return [] + + quadrant_cfg = data.get("quadrant_config") or {} + y_div = float(quadrant_cfg.get("warning_divider", QUAD_Y_DIV)) + warning_elevated = score is not None and float(score) >= y_div + + out: list[tuple[str, str, str]] = [] + + previous_state = await _last_logged_key(db, FUND_TYPE) + if previous_state is None: + _log_alert(db, FUND_TYPE, state) # seed + elif previous_state != state and state != "unknown" and usable: + effective = context.get("effective_date") + out.append(( + FUND_TYPE, + state, + f"📋 Fundamental context changed\n" + f"{previous_state} → {state}\n" + f"evidence: {context.get('evidence_quality', 'unavailable')}" + + (f" · effective {effective}" if effective else "") + + "\nContext channel — not a score, not a trade signal.", + )) + + confluence = "yes" if (warning_elevated and state == FUND_ADVERSE and usable) else "no" + previous_confluence = await _last_logged_key(db, CONFLUENCE_TYPE) + if previous_confluence is None: + _log_alert(db, CONFLUENCE_TYPE, confluence) # seed + elif previous_confluence != confluence and confluence == "yes": + out.append(( + CONFLUENCE_TYPE, + confluence, + f"⚠️ Confluence: market and fundamental risk both elevated\n" + f"Warning {float(score):.0f} (≥ {y_div:.0f}) with fundamentals {state}\n" + f"evidence: {context.get('evidence_quality', 'unavailable')}\n" + f"Highest attention. Still a thermometer — not a trade signal.", + )) + elif previous_confluence != confluence: + # Falling out of confluence is a state change worth recording as the new + # baseline, but not worth a message. + _log_alert(db, CONFLUENCE_TYPE, confluence) + + return out + + # --------------------------------------------------------------------------- # Dispatch # --------------------------------------------------------------------------- @@ -961,6 +1064,11 @@ async def dispatch_alerts(db: AsyncSession) -> dict: # cooldown/hysteresis handled in the collector (like score drops) for key, text in await _collect_regime_quadrant(db): outgoing.append((QUAD_TYPE, key, text)) + # Deliberately three separate messages off one toggle, not one fused + # signal: the market axes and the fundamental channel are different kinds + # of evidence, and an operator needs to know which one moved. + for alert_type, key, text in await _collect_regime_fundamental(db): + outgoing.append((alert_type, key, text)) if cfg["trade_closed"]: for key, text, pnl_usd in await _collect_closed_trades(db): diff --git a/app/services/event_study_service.py b/app/services/event_study_service.py index 5cc6dd6..e8209f5 100644 --- a/app/services/event_study_service.py +++ b/app/services/event_study_service.py @@ -1,15 +1,48 @@ -"""Compact chronological validation for the AI/Tech Risk Monitor warning score. +"""Chronological validation for the AI/Tech Risk Monitor warning score. -The study calls its outcome a 10% correction, uses the first 70% of sessions to -freeze an 80th-percentile warning threshold, and reports alarm episodes only on -the final 30%. It is still labelled exploratory while the fixed breadth basket -is reconstructed before its freeze date. +The outcome is a 10% correction in the leader, never a regime break. Two rules +are measured against it, and they answer different questions: + +* **shipped** -- the quadrant-change rule that actually reaches Telegram + (``alert_service._collect_regime_quadrant``). Its thresholds are fixed + constants chosen by scenario arithmetic, so nothing is fitted, so there is no + training set to protect and the whole sample is evaluable. This is the + headline. +* **fitted** -- the original study: an 80th-percentile Warning threshold frozen + on the first 70% of sessions and measured on the last 30%. Kept because it is + what the methodology document reports, and because a fitted threshold is a + genuinely different question -- but it is measured on the four corrections that + happen to fall in the holdout, which is too few to read as a property of the + score. + +Both are scored by the same ``evaluate_alarms`` harness, alongside ablations +(does the quadrant machinery earn its place?), external baselines (does the +score earn its complexity?), and a random-alarm null (is any of this better than +chance?). Without those rows a bare "2 of 4" is unreadable in either direction. + +The fundamental channel is compared, never fused. It appears as its own rule +(transitions into an adverse state), as a confluence gate (a market crossing kept +only when the state agrees), and as a market-only comparator over the identical +window -- because with ~10 correction events and almost no fundamental history, +any weight that combined it with the market axes would be a policy preference +presented as a measurement. + +Those three rows are **coverage-matched**: scored only on the sessions where the +channel had usable context and on the corrections whose warning horizon fell +inside it, and marked ``measurable: false`` until enough corrections are covered. +A fundamental rule scores zero whether it is wrong or merely absent, so scoring +it over the market rows' full sample would turn a fortnight of observations into +a 0/10 that reads as a failed test. + +Still labelled exploratory while the fixed breadth basket is reconstructed +before its freeze date. """ from __future__ import annotations import json import logging +import random from datetime import date, datetime, timedelta, timezone from sqlalchemy.ext.asyncio import AsyncSession @@ -17,11 +50,26 @@ from sqlalchemy.ext.asyncio import AsyncSession from app.services import breadth_service, settings_store from app.services import regime_monitor_service as rms from app.services.admin_service import update_setting +from app.services.alert_service import ( + QUAD_COOLDOWN_DAYS, + QUAD_MARGIN, + QUAD_X_DIV, + QUAD_Y_DIV, + _classify_quadrant, +) logger = logging.getLogger(__name__) KEY_REPORT = "regime_event_study" +# Report shape, independent of METHODOLOGY. A cached report from an older shape +# parses fine and reports the current methodology, so without this check the +# panel would render a report missing half its blocks. Bumping discards the cache +# the way a methodology change does -- and it is the *only* thing that does so +# here, because the fundamental-channel rework left METHODOLOGY on v4 (the scores +# did not change), so the methodology check cannot catch a stale report. +STUDY_SCHEMA = 3 + EVENT_THRESHOLD_PCT = 10.0 EVENT_COOLDOWN_DAYS = 40 DRAWDOWN_LOOKBACK = 252 @@ -33,6 +81,21 @@ TRAIN_FRACTION = 0.70 MIN_EVENTS_FOR_CONFIDENCE = 8 SENSOR_MISMATCH_TOLERANCE = 0.10 +# _collect_regime_quadrant confirms against get_regime_history(db, days=14), so a +# prior session older than that window is not available to confirm with. +QUAD_HISTORY_DAYS = 14 +# Quadrants with Warning above its divider: "1" early warning, "2" active stress. +WARNING_QUADRANTS = ("1", "2") +STRESS_QUADRANT = ("2",) + +# Draws for the random-alarm null. Seeded, because a cached report that moves +# on re-run for RNG reasons is worse than no report. +NULL_DRAWS = 2000 +NULL_SEED = 20260812 + +BASELINE_SMA_WINDOW = 50 +BASELINE_VIX_LEVEL = 20.0 + def _median(values: list[float]) -> float | None: if not values: @@ -148,41 +211,355 @@ def evaluate_alarms( } -def _warning_series( +def _score_rule( + alarm_indices: list[int], + event_indices: list[int], + dates: list[date], + horizon: int, + sessions: int, +) -> dict: + """``evaluate_alarms`` plus the annualised false-alarm rate for one rule. + + The rate is ``None`` when the rule had no eligible sessions. Dividing by a + tiny floor instead produced 5e9 alarms/year for a coverage-matched rule with + an empty window -- a number that means "undefined" while looking like a + measurement, which is the failure mode this whole panel is built to avoid. + """ + metrics = evaluate_alarms(alarm_indices, event_indices, dates, horizon) + metrics["false_alarms_per_year"] = ( + round(metrics["false_alarms"] / (sessions / 252.0), 2) if sessions > 0 else None + ) + return metrics + + +# --------------------------------------------------------------------------- +# The shipped rule +# --------------------------------------------------------------------------- + +def _axis_rows( prices: dict[str, rms.Series], - breadth_divergence: dict[date, float], + vix_series: rms.Series | None, + oas_series: rms.Series | None, + breadth_series: rms.Series | None, + divergence_series: rms.Series | None, dates: list[date], config: dict, - oas_series: rms.Series | None = None, -) -> tuple[dict[date, float], dict[date, int]]: - """Warning score per session plus how many sensors backed it. + observations: list[dict] | None = None, +) -> dict[date, dict]: + """State and Warning per session, from the function that writes snapshots. - v2 re-derived this by hand from ``WARNING_WEIGHTS`` and so would have kept - measuring the old construct after a scoring change. Since v3 dropped - fundamentals from the score, this is now exactly the live Warning score - rather than a technical-only approximation of it. + Calling ``_compute_index`` rather than re-deriving the two axes is the same + anti-drift argument that produced ``warning_sensor_scores``: the v2 study + re-derived Warning by hand and would have kept measuring the old construct + through a scoring change. State has no such shared helper, so the whole + snapshot builder is the shared definition. - The sensor count matters because the score renormalises over whatever is - available: a session backed by two sensors is not drawn from the same - distribution as one backed by three, and the frozen threshold assumes it is. + ``observations`` is the point-in-time fundamental series. It does not enter + either score -- the fundamental channel is categorical and read by confluence + -- but the per-session ``fundamental_state`` it produces is what the + confluence rule below is measured on, so it has to be the same series + production reports from. Every variant in this module reads its Warning from + these rows, so there is no second derivation to fall out of step. """ - tickers = config["tickers"] - smh_full = prices.get(tickers["leaders"][0], []) - spy_full = prices.get(tickers["market"], []) - out: dict[date, float] = {} - backing: dict[date, int] = {} + rows: dict[date, dict] = {} for session in dates: - sensors = rms.warning_sensor_scores( - breadth_divergence.get(session), - rms._closes_asof(smh_full, session), - rms._closes_asof(spy_full, session), - rms._window_asof(oas_series, session, rms.HY_OAS_WINDOW_DAYS), + snapshot = rms._compute_index( + prices, + vix_series, + oas_series, + {}, + config, + session, + breadth_series=breadth_series, + divergence_series=divergence_series, + observations=observations or [], ) - score = rms.score_warning_sensors(sensors) - if score is not None: - out[session] = round(score, 2) - backing[session] = sum(1 for value in sensors.values() if value is not None) - return out, backing + state = snapshot["state"] + warning = snapshot["warning"] + rows[session] = { + "state": state.get("score"), + "warning": warning.get("score"), + "fundamental_state": (snapshot.get("fundamental_context") or {}).get("state"), + # `usable`, not `available`: a stale observation keeps its state for + # display but stops counting as evidence, and an observation whose + # extraction failed on everything is fresh but knows nothing. Either + # one counted here would inflate the covered window with sessions the + # channel could not have contributed to. + "fundamental_usable": bool( + (snapshot.get("fundamental_context") or {}).get("usable") + ), + "state_coverage": state.get("coverage") or 0.0, + "warning_coverage": warning.get("coverage") or 0.0, + # The score renormalises over available sensors, so a session backed + # by two is not drawn from the same distribution as one backed by + # three, and a frozen threshold assumes it is. + "warning_sensors": len(warning.get("available_pillars") or []), + "inputs_fresh": bool((snapshot.get("data_quality") or {}).get("inputs_fresh")), + } + return rows + + +def _publishable(row: dict | None) -> bool: + """What ``get_regime_history`` leaves for the alert to confirm against. + + Deliberately not freshness-gated: ``_collect_regime_quadrant`` checks + ``is_fresh`` on today's live reading only, while the prior session comes from + stored history where the only filter is a published band on both axes. + """ + return ( + row is not None + and row["state"] is not None + and row["warning"] is not None + and row["state_coverage"] >= rms.MIN_COVERAGE + and row["warning_coverage"] >= rms.MIN_COVERAGE + ) + + +def _prior_publishable( + rows: dict[date, dict], dates: list[date], index: int, history_days: int +) -> dict | None: + """``valid[-2]``: the previous published session inside the 14-day window. + + The monitor writes today's snapshot before the alert step runs + (``job_catalog._DAILY_PIPELINE_STEPS``), so ``valid[-1]`` is today and this + is genuinely the prior session rather than t-2. + """ + cutoff = dates[index] - timedelta(days=history_days) + for position in range(index - 1, -1, -1): + if dates[position] < cutoff: + return None + candidate = rows.get(dates[position]) + if _publishable(candidate): + return candidate + return None + + +def replay_quadrant_changes( + rows: dict[date, dict], + dates: list[date], + state_divider: float = QUAD_X_DIV, + warning_divider: float = QUAD_Y_DIV, + margin: float = QUAD_MARGIN, + cooldown_days: int = QUAD_COOLDOWN_DAYS, + history_days: int = QUAD_HISTORY_DAYS, +) -> list[dict]: + """Every quadrant change the shipped alert would have sent, in order. + + A faithful replay of ``_collect_regime_quadrant``, including three details a + state machine written from first principles gets wrong: + + * the prior session is classified against the *current baseline*, not against + its own predecessor, so confirmation asks "did yesterday already look like + this change" rather than "did yesterday change too"; + * the baseline advances only when an alert actually fires, so a change that + fails confirmation or cooldown is re-evaluated against the old quadrant on + the next session rather than being forgotten; + * one cooldown is shared by every quadrant change, so a 3->4 alert can + swallow a 4->2 alert three days later. + + Returns the fires themselves rather than alarm indices, because which + transitions count as a *warning* is the caller's question: entering + Warning-high territory and entering both-high territory are different rules + over the same replay. + """ + fires: list[dict] = [] + baseline: str | None = None + baseline_date: date | None = None + + for index, session in enumerate(dates): + row = rows.get(session) + if not _publishable(row) or not row["inputs_fresh"]: + continue + x, y = float(row["state"]), float(row["warning"]) + + if baseline is None: # seeds silently, exactly as a fresh install does + baseline = _classify_quadrant(x, y, None, margin, state_divider, warning_divider) + baseline_date = session + continue + + new_quadrant = _classify_quadrant(x, y, baseline, margin, state_divider, warning_divider) + if new_quadrant == baseline: + continue + + prior = _prior_publishable(rows, dates, index, history_days) + if prior is None: + continue + prior_quadrant = _classify_quadrant( + float(prior["state"]), float(prior["warning"]), + baseline, margin, state_divider, warning_divider, + ) + if prior_quadrant != new_quadrant: + continue + + if baseline_date is not None and (session - baseline_date).days < cooldown_days: + continue + + fires.append({ + "index": index, + "date": session.isoformat(), + "from": baseline, + "to": new_quadrant, + "state": x, + "warning": y, + }) + baseline, baseline_date = new_quadrant, session + + return fires + + +def entry_alarms(fires: list[dict], entry: tuple[str, ...]) -> list[int]: + """Fires that *enter* the given quadrant set from outside it.""" + return [f["index"] for f in fires if f["to"] in entry and f["from"] not in entry] + + +# --------------------------------------------------------------------------- +# Ablations, baselines, null +# --------------------------------------------------------------------------- + +def _usable_adverse(rows: dict[date, dict], session: date) -> bool: + """Adverse *and* still within its staleness horizon. + + Both callers need this pair, and neither may use the state alone: the state + survives going stale so the card can show it, which would otherwise let a + months-old read confirm crossings indefinitely. + """ + row = rows.get(session) or {} + return row.get("fundamental_state") == "adverse" and bool(row.get("fundamental_usable")) + + +def adverse_episodes( + rows: dict[date, dict], dates: list[date], start_index: int +) -> list[int]: + """Sessions where the fundamental state *becomes* usably adverse. + + The market rules alarm on a rising-edge crossing; a categorical state has no + crossing, so its analogue is the transition into ``adverse``. That keeps the + row comparable with every other row in the table rather than counting every + day the state happens to sit there. + """ + alarms: list[int] = [] + was_adverse = start_index > 0 and _usable_adverse(rows, dates[start_index - 1]) + for index in range(start_index, len(dates)): + if dates[index] not in rows: + continue + adverse = _usable_adverse(rows, dates[index]) + if adverse and not was_adverse: + alarms.append(index) + was_adverse = adverse + return alarms + + +def confluence_episodes( + warning_alarms: list[int], rows: dict[date, dict], dates: list[date] +) -> list[int]: + """Warning crossings that happen while the fundamental state is usably adverse. + + Deliberately gated on the market crossing rather than on either channel + moving: it preserves the rising-edge semantics every other row uses, so the + column measures "does requiring fundamental agreement help?" instead of a + differently-shaped rule that cannot be compared with the others. + """ + return [index for index in warning_alarms if _usable_adverse(rows, dates[index])] + + +def covered_events( + event_indices: list[int], + rows: dict[date, dict], + dates: list[date], + horizon: int, +) -> list[int]: + """Corrections a fundamental rule actually had a chance to warn about. + + An alarm counts only if it fires in ``[event - horizon, event - 1]``, so a + correction is *coverable* only if the channel had usable context somewhere in + that window. Scoring these rules against every correction instead would make + one day of observation render as 0/10 -- an untested rule reported as a + failed one, which is the exact mistake the ``measurable`` flag exists to + prevent for the empty-table case. + """ + covered: list[int] = [] + for event_index in event_indices: + window = range(max(0, event_index - horizon), event_index) + if any( + bool((rows.get(dates[index]) or {}).get("fundamental_usable")) + for index in window + ): + covered.append(event_index) + return covered + + +def eligible_sessions( + rows: dict[date, dict], dates: list[date], start_index: int +) -> int: + """Sessions a fundamental rule could have fired on, for the FA/year rate. + + Annualising over the whole window instead would divide a rule's false alarms + by years in which it was structurally incapable of firing, reporting a + flattering rate that means nothing. + """ + return sum( + 1 + for session in dates[start_index:] + if bool((rows.get(session) or {}).get("fundamental_usable")) + ) + + +def below_average_series( + series: rms.Series, window: int = BASELINE_SMA_WINDOW +) -> dict[date, float]: + """100 while the close sits under its ``window``-session average, else 0.""" + out: dict[date, float] = {} + closes = [value for _, value in series] + for index, (session, close) in enumerate(series): + if index + 1 < window: + continue + average = sum(closes[index + 1 - window: index + 1]) / window + out[session] = 100.0 if close < average else 0.0 + return out + + +def _null_model( + alarm_count: int, + event_indices: list[int], + dates: list[date], + horizon: int, + start_index: int, + observed_warned: int, + draws: int = NULL_DRAWS, + seed: int = NULL_SEED, +) -> dict | None: + """Recall from alarms scattered at random over the same evaluable sessions. + + Drawn only from sessions a real rule could have fired on: over the whole + sample the null would be diluted by warm-up sessions and would understate + what chance achieves. That matters here -- with ~11 events and a 20-session + horizon, a sixth of the sample already sits inside a hit window. + + Corrections cluster, and uniform placement does not, so this is the floor + rather than the bar: an alarm process that clusters would beat it for + reasons that have nothing to do with foresight. + """ + population = range(start_index, len(dates)) + if alarm_count <= 0 or not event_indices or alarm_count > len(population): + return None + rng = random.Random(seed) + recalls: list[int] = [] + for _ in range(draws): + picks = sorted(rng.sample(population, alarm_count)) + recalls.append(evaluate_alarms(picks, event_indices, dates, horizon)["events_warned"]) + mean = sum(recalls) / len(recalls) + variance = sum((value - mean) ** 2 for value in recalls) / len(recalls) + return { + "draws": draws, + "alarms_per_draw": alarm_count, + "events": len(event_indices), + "mean_warned": round(mean, 2), + "sd_warned": round(variance ** 0.5, 2), + "observed_warned": observed_warned, + "p_at_least_observed": round( + sum(1 for value in recalls if value >= observed_warned) / len(recalls), 3 + ), + } def _reliability( @@ -192,9 +569,9 @@ def _reliability( events_detected: int, events_in_holdout: int, ) -> dict: - """How far the headline metrics can actually be trusted. + """How far the *fitted* variant's headline metrics can be trusted. - Two things repeatedly invite over-reading this report: + Two things repeatedly invite over-reading it: * The holdout carries only the corrections that fall in the last 30% of the sample. A "2/4" is one event away from "3/4", and in practice the events @@ -203,6 +580,9 @@ def _reliability( * The score renormalises over available sensors, so a training window that predates a sensor's history freezes a threshold on a different construct than the holdout is measured against. + + Neither applies to the shipped rule, whose thresholds are fixed constants -- + but the second one does not vanish, it relocates: see ``_era_split``. """ expected = len(rms.WARNING_WEIGHTS) train = [backing[d] for d in dates[:split] if d in backing] @@ -221,6 +601,126 @@ def _reliability( } +def _era_split( + alarms: list[int], + event_indices: list[int], + dates: list[date], + horizon: int, + start_index: int, + credit_from: date | None, +) -> dict | None: + """Shipped-rule metrics either side of the credit sensor's first session. + + Dropping the fitted threshold makes the whole sample evaluable, which is the + point -- but most of the extra events sit before 2023-08, where W3 does not + exist and Warning renormalises to ``(W1*45 + W2*30)/75``. The fixed 40 + divider is then applied to a different construct than it was reasoned about, + so the coverage caveat does not disappear with the split; it relocates from + the threshold to the score. Reporting the two eras separately is what keeps + the fuller sample from being a differently misleading headline. + + The pre-credit era is close to a "Warning without W3" ablation on real + sessions -- and a clean one, because the fundamental channel is not a term in + Warning at all, so the two eras differ by W3 and nothing else. That stays + true however much fundamental history accumulates. + + Alarms and events are assigned to eras by index, so an alarm days before the + boundary that matched an event days after it lands in the earlier era. With + the eras years long and the events sparse, that costs nothing. + """ + if credit_from is None: + return None + boundary = next( + (index for index, session in enumerate(dates) if session >= credit_from), None + ) + if boundary is None or boundary <= start_index or boundary >= len(dates): + return None + + def slice_metrics(low: int, high: int) -> dict: + sessions = max(0, high - low) + metrics = _score_rule( + [a for a in alarms if low <= a < high], + [e for e in event_indices if low <= e < high], + dates, horizon, sessions, + ) + metrics.pop("per_event", None) + metrics["sessions"] = sessions + return metrics + + return { + "credit_from": credit_from.isoformat(), + "pre_credit": { + "label": "W1+W2 only", + "start": dates[start_index].isoformat(), + "end": dates[boundary - 1].isoformat(), + **slice_metrics(start_index, boundary), + }, + "full_coverage": { + "label": "all three sensors", + "start": dates[boundary].isoformat(), + "end": dates[-1].isoformat(), + **slice_metrics(boundary, len(dates)), + }, + } + + +def _warning_from_rows( + rows: dict[date, dict], dates: list[date] +) -> tuple[dict[date, float], dict[date, int]]: + """Published Warning per session plus how many sensors backed it. + + Read off ``_axis_rows`` rather than recomputed. v2 re-derived Warning by hand + from ``WARNING_WEIGHTS`` and would have kept measuring the old construct + after a scoring change; a second derivation here would have done the same to + any later change to how Warning is assembled -- silently, in the fitted + variant and the ``warning_bare`` ablation, while the shipped replay moved on + without it. + """ + out: dict[date, float] = {} + backing: dict[date, int] = {} + for session in dates: + row = rows.get(session) + if row is None or row["warning"] is None: + continue + out[session] = float(row["warning"]) + backing[session] = int(row["warning_sensors"]) + return out, backing + + +def _rule_row( + rule_id: str, + label: str, + kind: str, + note: str, + alarms: list[int], + event_indices: list[int], + dates: list[date], + horizon: int, + sessions: int, + measurable: bool = True, +) -> dict: + """One comparison row. + + ``measurable=False`` marks a rule whose *input* is too thin to have been + tested, not one that failed. A fundamental rule scores 0/N whether it is + wrong or merely absent, and a 0/N sitting in this table would read as + tested-and-failed -- the same false precision the whole restructure exists to + remove. It stays false until the channel has covered + ``MIN_EVENTS_FOR_CONFIDENCE`` corrections, because a 1/1 or 0/2 over a + two-week exposure is not a result either. + """ + metrics = _score_rule(alarms, event_indices, dates, horizon, sessions) + metrics.pop("per_event", None) + return { + "id": rule_id, + "label": label, + "kind": kind, + "note": note, + "measurable": measurable, + **metrics, + } + + async def run_event_study( db: AsyncSession, threshold_pct: float = EVENT_THRESHOLD_PCT, @@ -242,55 +742,202 @@ async def run_event_study( ) divergence = breadth_service.compute_divergence_series(breadth, benchmark) oas_series = await rms._fetch_fred_series("BAMLH0A0HYM2", start, end) - warning, backing = _warning_series(prices, divergence, dates, config, oas_series) + # State needs volatility, which the Warning-only study never fetched. + vix_series = await rms._fetch_fred_series("VIXCLS", start, end) + # The point-in-time fundamental series. It is not in either score; it drives + # the categorical channel the confluence rule below is measured on. + observations = await rms.get_fundamental_observations(db) # The credit sensor cannot reach back as far as the price history does (the # upstream series is capped at ~3 years), so the earlier part of the sample # scores on W1+W2 alone via renormalisation. Report where W3 starts rather # than letting the threshold quietly straddle two sensor sets. - credit_from = oas_series[0][0].isoformat() if oas_series else None + credit_from = oas_series[0][0] if oas_series else None + all_events = detect_events(closes, dates, threshold_pct) + all_event_indices = [event["index"] for event in all_events] + + # --- one pass; every rule below reads its Warning from these rows ---- + rows = _axis_rows( + prices, + vix_series, + oas_series, + rms._mapping_series(breadth), + rms._mapping_series(divergence), + dates, + config, + observations, + ) + warning, backing = _warning_from_rows(rows, dates) + fires = replay_quadrant_changes(rows, dates) + # Nothing can alarm before the baseline seeds, so every rule is measured from + # the same session and the comparison stays like-for-like. + seeded = next( + ( + index + for index, session in enumerate(dates) + if _publishable(rows.get(session)) and rows[session]["inputs_fresh"] + ), + None, + ) + if seeded is None: + return {"available": False, "reason": "no session with publishable coverage"} + evaluable_start = seeded + 1 + evaluable_sessions = max(1, len(dates) - evaluable_start) + evaluable_events = [index for index in all_event_indices if index >= evaluable_start] + + warning_alarms = entry_alarms(fires, WARNING_QUADRANTS) + shipped_metrics = _score_rule( + warning_alarms, evaluable_events, dates, horizon, evaluable_sessions + ) + shipped_events = shipped_metrics.pop("per_event") + + # --- the fitted variant, kept for continuity ------------------------- split = max(1, min(len(dates) - 1, int(len(dates) * TRAIN_FRACTION))) train_values = [warning[d] for d in dates[:split] if d in warning] warn_threshold = _percentile(train_values, WARN_PERCENTILE) if warn_threshold is None: return {"available": False, "reason": "insufficient warning history"} - - all_events = detect_events(closes, dates, threshold_pct) - holdout_events = [event["index"] for event in all_events if event["index"] >= split] - alarms = alarm_episodes(warning, dates, warn_threshold, start_index=split) - metrics = evaluate_alarms(alarms, holdout_events, dates, horizon) + holdout_events = [index for index in all_event_indices if index >= split] + fitted_alarms = alarm_episodes(warning, dates, warn_threshold, start_index=split) holdout_sessions = max(1, len(dates) - split) - metrics["false_alarms_per_year"] = round( - metrics["false_alarms"] / (holdout_sessions / 252.0), 2 + fitted_metrics = _score_rule( + fitted_alarms, holdout_events, dates, horizon, holdout_sessions ) - + fitted_events = fitted_metrics.pop("per_event") reliability = _reliability(dates, split, backing, len(all_events), len(holdout_events)) + # --- ablations and baselines, all on fixed thresholds ---------------- + # Fitted thresholds are deliberately excluded here: a threshold fitted on the + # full sample would have lookahead the shipped rule does not, and one fitted + # on a training split could only be scored on the four holdout events. Fixed + # constants keep every row on the same events over the same sessions. + state_series = { + session: row["state"] for session, row in rows.items() if row["state"] is not None + } + vix_indicator = { + session: value + for session in dates + if (value := rms._value_asof(vix_series, session)) is not None + } + # The fundamental channel is categorical and never enters a score, so it is + # compared as its own rule and as a confluence gate rather than tuned as a + # weight. With an empty observation series both are unmeasurable, and say so. + fundamental_alarms = adverse_episodes(rows, dates, evaluable_start) + confluence_alarms = confluence_episodes(warning_alarms, rows, dates) + # Coverage-matched denominators. These rules only existed on the sessions the + # channel had usable context, so scoring them over the whole window would + # report an exposure they never had -- and one day of coverage would render + # as 0/10. + fundamental_events = covered_events(evaluable_events, rows, dates, horizon) + fundamental_sessions = eligible_sessions(rows, dates, evaluable_start) + fundamental_measurable = len(fundamental_events) >= MIN_EVENTS_FOR_CONFIDENCE + comparison = [ + _rule_row( + "fundamental_adverse", "Fundamental context turns adverse", "fundamental", + "The third channel on its own: transitions into an adverse capex / " + "earnings-reaction state, with no market input at all.", + fundamental_alarms, fundamental_events, dates, horizon, fundamental_sessions, + measurable=fundamental_measurable, + ), + _rule_row( + "confluence", "Confluence: Warning crossing while adverse", "fundamental", + "The shipped market crossing, kept only when the fundamental channel " + "agrees. Answers whether requiring agreement buys precision, at what " + "cost in recall.", + confluence_alarms, fundamental_events, dates, horizon, fundamental_sessions, + measurable=fundamental_measurable, + ), + _rule_row( + "market_over_covered", "Quadrant alert, covered window only", "fundamental", + "The shipped market rule scored on exactly the events, sessions and " + "alarms the two rows above were scored on. Without it, any difference " + "between them and the headline could be the window rather than the " + "channel.", + # Alarms are restricted to the covered window too: counting crossings + # that fired when the channel had no context would compare the market + # rule's full exposure against the channel's partial one. + [ + index + for index in warning_alarms + if index >= evaluable_start + and bool((rows.get(dates[index]) or {}).get("fundamental_usable")) + ], + fundamental_events, dates, horizon, fundamental_sessions, + measurable=fundamental_measurable, + ), + _rule_row( + "quadrant_stress_entry", "Quadrant alert, both axes high", "ablation", + "The same replay, recording only entries into the both-high quadrant. " + "State is coincident by construction, so requiring it should convert " + "leads into confirmations.", + entry_alarms(fires, STRESS_QUADRANT), + evaluable_events, dates, horizon, evaluable_sessions, + ), + _rule_row( + "warning_bare", f"Warning >= {QUAD_Y_DIV:.0f} (bare crossing)", "ablation", + "The shipped divider with none of the quadrant machinery: no State " + "condition, no hysteresis, no confirmation, no cooldown.", + alarm_episodes(warning, dates, QUAD_Y_DIV, start_index=evaluable_start), + evaluable_events, dates, horizon, evaluable_sessions, + ), + _rule_row( + "state_bare", f"State >= {QUAD_X_DIV:.0f} (bare crossing)", "ablation", + "The coincident axis alone. State measures stress that has already " + "arrived, so a competitive lead here would be surprising.", + alarm_episodes(state_series, dates, QUAD_X_DIV, start_index=evaluable_start), + evaluable_events, dates, horizon, evaluable_sessions, + ), + _rule_row( + "smh_below_50dma", f"{leader} below its {BASELINE_SMA_WINDOW}-DMA", "baseline", + "The crudest possible trend rule, and free.", + alarm_episodes( + below_average_series(benchmark, BASELINE_SMA_WINDOW), dates, + 50.0, start_index=evaluable_start, + ), + evaluable_events, dates, horizon, evaluable_sessions, + ), + _rule_row( + "vix_level", f"VIX >= {BASELINE_VIX_LEVEL:.0f}", "baseline", + "The market's own risk gauge, unweighted and unmodelled.", + alarm_episodes( + vix_indicator, dates, BASELINE_VIX_LEVEL, start_index=evaluable_start + ), + evaluable_events, dates, horizon, evaluable_sessions, + ), + ] + + null_model = _null_model( + len(warning_alarms), evaluable_events, dates, horizon, + evaluable_start, shipped_metrics["events_warned"], + # Passed rather than defaulted: a default argument binds the constant at + # import, so overriding it (in tests) would silently do nothing. + draws=NULL_DRAWS, seed=NULL_SEED, + ) + eras = _era_split( + warning_alarms, evaluable_events, dates, horizon, evaluable_start, credit_from + ) basket_asof = date.fromisoformat(config["basket_asof"]) - retrospective = dates[split] < basket_asof + retrospective = dates[evaluable_start] < basket_asof evaluation = "exploratory" if retrospective else "holdout" lead_text = ( - f"median lead {metrics['median_lead_days']:.0f} sessions" - if metrics["median_lead_days"] is not None + f"median lead {shipped_metrics['median_lead_days']:.0f} sessions" + if shipped_metrics["median_lead_days"] is not None else "no successful warning lead" ) summary = ( - f"{evaluation.capitalize()} chronological test: warning episodes preceded " - f"{metrics['events_warned']}/{metrics['events']} 10% corrections; " - f"{metrics['events_missed']} missed, {metrics['false_alarms_per_year']:.1f} " - f"false alarms/year, {lead_text}. " - f"{metrics['events']} of {reliability['events_detected']} detected corrections " - f"fall in the test period" - + ( - "; too few to read recall as a property of the score." - if reliability["underpowered"] - else "." - ) + f"{evaluation.capitalize()} replay of the shipped quadrant alert over " + f"{evaluable_sessions} sessions: it entered Warning-high territory ahead of " + f"{shipped_metrics['events_warned']} of {shipped_metrics['events']} 10% " + f"corrections, with {shipped_metrics['false_alarms_per_year']:.1f} false " + f"alarms/year and {lead_text}. Its dividers are fixed constants rather than " + f"fitted, so there is no training split and every detected correction is " + f"evaluable — compare it against the ablations and baselines below before " + f"reading the ratio as good or bad." ) - per_event = metrics.pop("per_event") report = { "available": True, + "schema": STUDY_SCHEMA, "methodology": rms.METHODOLOGY, "generated_at": datetime.now(timezone.utc).isoformat(), "evaluation": evaluation, @@ -301,24 +948,69 @@ async def run_event_study( "event_threshold_pct": threshold_pct, "event_cooldown_days": EVENT_COOLDOWN_DAYS, "horizon_days": horizon, - "train_fraction": TRAIN_FRACTION, - "warn_percentile": WARN_PERCENTILE, - "warn_threshold": round(warn_threshold, 1), - "credit_sensor_from": credit_from, + "credit_sensor_from": credit_from.isoformat() if credit_from else None, "basket_hash": rms._basket_hash(config["breadth_basket"]), "basket_asof": config["basket_asof"], }, + # The channel's actual exposure, which is what its rows are scored on. + # The series starts empty -- the observation lived in a single + # overwritten settings slot until 2026-08-12 -- and it accumulates one + # observation at a time, so for a long while these rows are unmeasurable + # rather than unsuccessful. Stating the exposure is what stops the table + # inventing a failed result out of a thin one. + "fundamental_coverage": { + "observations": len(observations), + "sessions_eligible": fundamental_sessions, + "evaluable_sessions": evaluable_sessions, + "events_covered": len(fundamental_events), + "events_evaluable": len(evaluable_events), + "minimum_events": MIN_EVENTS_FOR_CONFIDENCE, + "measurable": fundamental_measurable, + }, "sample": { "start": dates[0].isoformat(), "end": dates[-1].isoformat(), - "train_end": dates[split - 1].isoformat(), - "test_start": dates[split].isoformat(), "sessions": len(dates), - "holdout_sessions": holdout_sessions, + # Not "test_start": the shipped rule fits nothing, so this is where + # the baseline seeds and every rule becomes measurable, not where a + # holdout begins. The fitted variant's split lives under "fitted". + "evaluable_from": dates[evaluable_start].isoformat(), + "evaluable_sessions": evaluable_sessions, + "events_detected": len(all_events), + "events_evaluable": len(evaluable_events), + }, + "shipped": { + "rule": { + "state_divider": QUAD_X_DIV, + "warning_divider": QUAD_Y_DIV, + "margin": QUAD_MARGIN, + "confirm_sessions": 2, + "cooldown_days": QUAD_COOLDOWN_DAYS, + "entry": "Warning-high quadrant (early warning or active stress)", + }, + "metrics": shipped_metrics, + "events": shipped_events, + "quadrant_changes": len(fires), + "fires": fires, + "by_era": eras, + }, + "comparison": comparison, + "null_model": null_model, + "fitted": { + "params": { + "train_fraction": TRAIN_FRACTION, + "warn_percentile": WARN_PERCENTILE, + "warn_threshold": round(warn_threshold, 1), + }, + "sample": { + "train_end": dates[split - 1].isoformat(), + "test_start": dates[split].isoformat(), + "holdout_sessions": holdout_sessions, + }, + "metrics": fitted_metrics, + "events": fitted_events, }, - "metrics": metrics, "reliability": reliability, - "events": per_event, "recent_breadth": [ {"date": d.isoformat(), "breadth": breadth[d], "warning": warning.get(d)} for d in dates[-90:] @@ -328,10 +1020,13 @@ async def run_event_study( logger.info(json.dumps({ "event": "regime_event_study_complete", "evaluation": evaluation, - "events": metrics["events"], - "events_detected": reliability["events_detected"], - "warned": metrics["events_warned"], - "false_alarms_per_year": metrics["false_alarms_per_year"], + "shipped_events": shipped_metrics["events"], + "shipped_warned": shipped_metrics["events_warned"], + "shipped_false_alarms_per_year": shipped_metrics["false_alarms_per_year"], + "quadrant_changes": len(fires), + "fitted_events": fitted_metrics["events"], + "fitted_warned": fitted_metrics["events_warned"], + "null_p_at_least_observed": (null_model or {}).get("p_at_least_observed"), "underpowered": reliability["underpowered"], "sensor_coverage_mismatch": reliability["sensor_coverage_mismatch"], })) @@ -352,4 +1047,8 @@ async def get_event_study_report(db: AsyncSession) -> dict | None: report = json.loads(setting.value) except (TypeError, ValueError): return None - return report if report.get("methodology") == rms.METHODOLOGY else None + if report.get("methodology") != rms.METHODOLOGY: + return None + # A pre-replay report parses fine and carries the current methodology, so the + # shape has to be checked separately or the panel renders a headline-less v4. + return report if report.get("schema") == STUDY_SCHEMA else None diff --git a/app/services/regime_monitor_service.py b/app/services/regime_monitor_service.py index 8dae0f1..457c113 100644 --- a/app/services/regime_monitor_service.py +++ b/app/services/regime_monitor_service.py @@ -7,11 +7,17 @@ two deliberately separate outputs: * Warning: deterioration/divergence that may precede State (breadth divergence, relative strength, credit impulse). -Both scores are quantitative and daily. The sourced hyperscaler capex and -earnings-reaction observations are a qualitative *overlay* since v3 rather than -weighted sensors: at a combined 20 points they could not reach the event -study's alarm threshold even when both pegged, so refreshing them appeared to -do nothing. They are reported next to the scores instead of inside them. +* Fundamental context: a categorical channel (supportive / neutral / adverse / + unknown) with an evidence-quality grade, derived by fixed rules from the + sourced hyperscaler capex and earnings-reaction observations. + +Both scores are quantitative and daily. The fundamental channel is deliberately +**not** a term in either: the three are read together by confluence, because +adding a slow categorical judgement to a fast continuous score manufactures +precision by summing unlike things, and any fusion weight would be a policy +preference presented as a measurement until there is enough point-in-time +history to fit one. A missing observation therefore stays ``unknown`` instead of +silently redistributing its weight onto the technical sensors. Daily snapshots are the point-in-time record. The first run under a new ``METHODOLOGY`` rewrites every session inside ``REBUILD_LOOKBACK_DAYS`` once; @@ -35,6 +41,7 @@ from sqlalchemy.ext.asyncio import AsyncSession from app.config import settings from app.exceptions import ProviderError, ValidationError +from app.models.regime_fundamental_observation import RegimeFundamentalObservation from app.models.regime_snapshot import RegimeSnapshot from app.providers.alpaca import AlpacaOHLCVProvider from app.services import breadth_service, settings_store @@ -55,7 +62,11 @@ METHODOLOGY = "v4" # against the *stored* blob, so omitting the current one discards the observation # on its first write, which leaves fetched_at null and locked false -- and then # update_regime_monitor refreshes it via the LLM on every single run, forever. -CATEGORICAL_FUNDAMENTAL_METHODOLOGIES = frozenset({"v2", "v3", "v4"}) +# "v5" is listed although no v5 scoring exists: a v5 was briefly built (a weighted +# fundamental modifier on Warning) and reverted, so a development box can have +# that string sitting in its settings blob. Keeping it costs nothing; omitting it +# costs the failure above. +CATEGORICAL_FUNDAMENTAL_METHODOLOGIES = frozenset({"v2", "v3", "v4", "v5"}) # Bumped when a fix changes what historical rows *should* contain without # changing the live formula, so stored history needs one reseed. Deliberately @@ -173,6 +184,34 @@ WARNING_WEIGHTS = { "credit_impulse": 25.0, } +# The sourced fundamental read is a **separate channel**, never a term in either +# score. It is reported as a categorical state beside State and Warning, and the +# three are read together by confluence rather than added up. +# +# Two things had to be true at once and only this shape gets both. +# +# **v3's reason for removing it was wrong.** v3 argued that F1+F3, at 12+8 of 100 +# Warning points, "could not change any published conclusion" because pegged they +# produced a Warning of exactly 20.0. That holds only when every technical sensor +# reads exactly zero. Weighted, those points added +10 to +20 across the +# realistic range and moved the technical score needed to reach the 40 quadrant +# divider from 40 to 25. So the observation was not inert, and demoting it to +# decoration was not justified by that argument. +# +# **But no weight is measurable either.** A weighted modifier was built (v5, +# reverted) and its size could not be derived from anything: with ~10 correction +# events and essentially no fundamental history, any fusion weight is a policy +# preference presented as a measurement. Adding a slow categorical judgement to a +# fast continuous score also manufactures precision by summing unlike things, and +# it forces a missing observation to silently redistribute its weight onto the +# technical sensors -- the opposite of leaving it unknown. +# +# So the read gets a channel, not a coefficient. Revisit only with enough +# point-in-time history to test whether the state improves prediction +# *conditional on* Warning; a fitted model then has something to fit. +FUNDAMENTAL_STATES = ("supportive", "neutral", "adverse", "unknown") +EVIDENCE_QUALITY = ("complete", "partial", "stale", "manual", "unavailable") + # Fixed at the v2 launch. These are liquid S&P 500/Nasdaq AI, semiconductor, # infrastructure, cloud, and enterprise-software names that the platform's # normal universe sync already stores. @@ -196,7 +235,12 @@ DEFAULT_CONFIG: dict = { } CAPEX_STATES = ("raising", "holding", "cutting", "unknown") -GNSD_STATES = ("yes", "no", "mixed") +# "mixed" is a genuinely observed mixed reaction; "unknown" is nobody looked or +# the extraction failed. They were the same value until 2026-08-13, so a failed +# LLM parse silently became neutral *evidence* -- an observation of normality +# manufactured out of a parse error. Same distinction the capex map already made +# with its own "unknown", and the same one the whole channel is built on. +GNSD_STATES = ("yes", "no", "mixed", "unknown") # v2 scored raising and holding identically at 0, so in a capex boom the reading # was pinned at 0 and could not express the raising -> holding deceleration that # is the actual early warning. Display-only in v3, but it should still describe. @@ -435,6 +479,104 @@ def score_warning_sensors(sensors: dict[str, float | None]) -> float | None: return sum(s * w for s, w in live) / sum(w for _, w in live) +def _capex_signal(capex: dict[str, str] | None, names: list[str]) -> str: + """Categorical read of hyperscaler capex direction. Never an average. + + Averaging is what this must not do: it would let two ``cutting`` reads and + two ``unknown`` ones land on "neutral", presenting missing evidence as + evidence of normality. Any cut is adverse on partial evidence; only a fully + known, uniformly rising basket is supportive. + """ + states = [str((capex or {}).get(name, "unknown")).strip().lower() for name in names] + known = [state for state in states if state in ("raising", "holding", "cutting")] + if not known: + return "unknown" + if "cutting" in known: + return "adverse" + if "holding" in known: + return "neutral" + return "supportive" + + +def _reaction_signal(good_news_stock_down: str | None) -> str: + """Good earnings being sold is a late-cycle tell; not being sold is healthy. + + Anything that is not one of the three observed categories -- including the + explicit ``"unknown"`` an extraction failure now writes -- falls through to + ``unknown`` rather than to ``mixed``. A parse error is not a reading. + """ + return { + "yes": "adverse", + "no": "supportive", + "mixed": "neutral", + }.get(str(good_news_stock_down or "").strip().lower(), "unknown") + + +def combine_fundamental_signals(capex_signal: str, reaction_signal: str) -> str: + """Confluence, not arithmetic: precedence over the two categorical reads. + + ``unknown`` is deliberately unreachable by combination -- it survives only + when *nothing* was observed. A single adverse read carries, because partial + evidence of deterioration is still evidence of deterioration; supportive + requires every observed signal to agree. + """ + signals = (capex_signal, reaction_signal) + if "adverse" in signals: + return "adverse" + observed = [signal for signal in signals if signal != "unknown"] + if not observed: + return "unknown" + return "supportive" if all(signal == "supportive" for signal in observed) else "neutral" + + +def _usable_context(observed: bool, pending: bool, stale: bool, state: str) -> bool: + """Whether a fundamental reading may count as evidence. + + One definition, called by both the point-in-time record and the live + reading, because they publish the same field name to the same consumers and + a second copy would drift. Distinct from `available`, which is about timing + alone: an observation whose extraction failed on everything is effective and + fresh, and still knows nothing. + """ + return observed and not pending and not stale and state != "unknown" + + +def _evidence_quality( + capex: dict[str, str] | None, + good_news_stock_down: str | None, + names: list[str], + *, + observed: bool, + stale: bool, + source: str | None, +) -> str: + """How much to trust the state above, as one field the reader can act on. + + Ordered by what an operator most needs to know: nothing collected beats + everything else, then a reading too old to be current, then a hand override, + then completeness. + """ + if not observed: + return "unavailable" + if stale: + return "stale" + if str(source or "").strip().lower() == "manual": + return "manual" + known = sum( + 1 + for name in names + if str((capex or {}).get(name, "unknown")).strip().lower() != "unknown" + ) + # `bool(names)` matters: with an empty basket `known == len(names)` is + # vacuously true, so nothing observed would grade as complete. + complete = ( + bool(names) + and known == len(names) + and _reaction_signal(good_news_stock_down) != "unknown" + ) + return "complete" if complete else "partial" + + def _sensor(sensor_id: str, label: str, score: float | None, **details: object) -> dict: return { "id": sensor_id, @@ -572,26 +714,66 @@ def _overlay_timing( return effective, pending, age, stale -def fundamental_overlay(overrides: dict, config: dict, as_of: date) -> dict: - """Point-in-time qualitative overlay. Never feeds State or Warning since v3. +def fundamental_context(overrides: dict, config: dict, as_of: date) -> dict: + """Point-in-time fundamental channel. Never a term in State or Warning. + + Called an "overlay" until 2026-08-12, which undersold it: it is the third + channel of the model, read alongside the two scores by confluence rather than + decorating them. The categorical ``state`` is what a reader and the chart + consume; ``evidence_quality`` is how far to trust it. + + Both are derived from the stored categorical facts by fixed rules, not from + an LLM's numeric judgement. The LLM's job is extraction and explanation -- + find the capex guidance, classify it, cite it -- and the rules turn those + facts into a state, so the same observation always yields the same category. The effective-date gate stays even though nothing is scored from this: the - 400-session rebuild replays historical dates, and stamping today's LLM read - onto 2024 snapshots would be plain lookahead in the stored record. + rebuild replays historical dates, and stamping today's read onto 2024 + snapshots would be plain lookahead in the stored record. - This is the *record*. For "what do we know right now", use + This is the *record*. For "what do we know right now", use ``current_observation`` -- do not add a bypass flag here, because this runs for every replayed date during a rebuild. """ effective, pending, age, stale = _overlay_timing(overrides, config, as_of) + names = list(config["tickers"]["hyperscalers"]) + capex = None if pending else overrides.get("capex") + reaction = None if pending else overrides.get("good_news_stock_down") + observed = not pending and bool(overrides.get("fetched_at")) + + capex_signal = _capex_signal(capex, names) if observed else "unknown" + reaction_signal = _reaction_signal(reaction) if observed else "unknown" + state = combine_fundamental_signals(capex_signal, reaction_signal) return { + "state": state, + "evidence_quality": _evidence_quality( + capex, reaction, names, + observed=observed, stale=stale, source=overrides.get("source"), + ), + "capex_signal": capex_signal, + "reaction_signal": reaction_signal, + # Two different questions, and conflating them is a trap: + # + # `available` is about *timing* -- there is an effective, non-stale record + # to display. `usable` is about *content* -- it also actually says + # something. A collected observation whose extraction failed on every + # hyperscaler is available (show it, with its date) but not usable: it + # knows nothing, so it must never count as evidence. + # + # The distinction is load-bearing for the event study. Coverage is + # measured in sessions with usable context, and if repeated extraction + # failures counted, they would slowly accumulate "exposure" until the + # fundamental rows flipped to measurable 0/8 -- a failed result reported + # for a channel that never knew anything, which is the exact confusion + # coverage-matching exists to prevent. "available": not pending and not stale, + "usable": _usable_context(observed, pending, stale, state), "pending": pending, "stale": stale, "effective_date": effective.isoformat() if effective else None, "age_days": age, - "capex": None if pending else overrides.get("capex"), - "good_news_stock_down": None if pending else overrides.get("good_news_stock_down"), + "capex": capex, + "good_news_stock_down": reaction, "capex_stress": None if pending else overrides.get("f1_score"), "earnings_stress": None if pending else overrides.get("f3_score"), "reasoning": None if pending else overrides.get("reasoning"), @@ -603,7 +785,7 @@ def fundamental_overlay(overrides: dict, config: dict, as_of: date) -> dict: def current_observation(overrides: dict, config: dict, as_of: date) -> dict: """The observation as it stands now, for the live reading only. - Same shape as ``fundamental_overlay``, but the effective date is *reported* + Same shape as ``fundamental_context``, but the effective date is *reported* rather than used to blank the content. A refresh stamps ``_next_weekday(today)``, so gating the live card hid a just-collected read for one day -- three over a weekend -- and refreshing appeared to do @@ -611,14 +793,36 @@ def current_observation(overrides: dict, config: dict, as_of: date) -> dict: published number; the stored snapshot keeps the gate. """ effective, pending, age, stale = _overlay_timing(overrides, config, as_of) - # The default override carries "unknown"/"mixed" placeholders for every + # The default override carries "unknown" placeholders for every # hyperscaler. Those are the absence of an observation, not an observation # of absence, and must never be presented as collected. ``fetched_at`` is # the collection timestamp and is the only field written on every path that # produces real content (LLM refresh and manual save both stamp it). observed = bool(overrides.get("fetched_at")) + names = list(config["tickers"]["hyperscalers"]) + capex_signal = _capex_signal(overrides.get("capex"), names) if observed else "unknown" + reaction_signal = ( + _reaction_signal(overrides.get("good_news_stock_down")) if observed else "unknown" + ) + state = combine_fundamental_signals(capex_signal, reaction_signal) return { "observed": observed, + "state": state, + "evidence_quality": _evidence_quality( + overrides.get("capex"), overrides.get("good_news_stock_down"), names, + observed=observed, stale=stale, source=overrides.get("source"), + ), + "capex_signal": capex_signal, + "reaction_signal": reaction_signal, + # Same shape as the record means the same *fields*, not just the same + # ones this function happens to need: the frontend types both payloads + # identically, so an omission here is an undefined at runtime that + # TypeScript cannot catch across a trusted server boundary. + # + # Note this is stricter than the `available` directly below: a pending + # observation is the freshest thing we have and worth showing, but it is + # not yet in force, so it is not yet evidence. + "usable": _usable_context(observed, pending, stale, state), # Live availability is about usefulness, not effectiveness: a pending # observation is the freshest thing we have -- but nothing collected is # never available. @@ -656,8 +860,16 @@ def _compute_index( breadth_series: Series | None = None, divergence_series: Series | None = None, breadth_counts: dict[date, int] | None = None, + observations: list[dict] | None = None, ) -> dict: - """Compute the complete State/Warning snapshot as of one trading date.""" + """Compute the complete State/Warning snapshot as of one trading date. + + ``observations`` is the point-in-time fundamental series and is authoritative + when supplied; ``overrides`` is the single-slot fallback for callers that + predate the table (the calibration harness). Either way the reading is scored + into the same ``fundamental_context`` -- only where it is read from differs, + so the live monitor and the event study cannot report different states. + """ tickers = config["tickers"] smh = _closes_asof(prices.get(tickers["leaders"][0], []), as_of) qqq = _closes_asof(prices.get(tickers["confirm"][0], []), as_of) @@ -682,7 +894,10 @@ def _compute_index( sensors = warning_sensor_scores(divergence, smh, spy, oas_window) relative_strength = sensors["relative_strength"] credit_impulse = sensors["credit_impulse"] - overlay = fundamental_overlay(overrides, config, as_of) + observation = ( + observation_asof(observations, as_of) if observations is not None else overrides + ) or {} + context = fundamental_context(observation, config, as_of) state_pillars = [ { @@ -768,7 +983,7 @@ def _compute_index( "date": as_of.isoformat(), "state": state, "warning": warning, - "fundamental_overlay": overlay, + "fundamental_context": context, "quadrant_config": { "state_divider": QUADRANT_STATE_DIVIDER, "warning_divider": QUADRANT_WARNING_DIVIDER, @@ -790,8 +1005,8 @@ def _compute_index( "breadth_pct_above_200": round(breadth_pct, 1) if breadth_pct is not None else None, "breadth_date": breadth_item[0].isoformat() if breadth_item else None, "fundamentals_fetched_at": overrides.get("fetched_at"), - "fundamentals_effective_date": overlay.get("effective_date"), - "fundamentals_age_days": overlay.get("age_days"), + "fundamentals_effective_date": context.get("effective_date"), + "fundamentals_age_days": context.get("age_days"), }, "data_quality": { "minimum_coverage": MIN_COVERAGE, @@ -859,7 +1074,7 @@ async def get_fundamental_overrides(db: AsyncSession) -> dict: "f1_score": None, "f3_score": None, "capex": {name: "unknown" for name in names}, - "good_news_stock_down": "mixed", + "good_news_stock_down": "unknown", "locked": False, "reasoning": None, "fetched_at": None, @@ -880,9 +1095,9 @@ async def get_fundamental_overrides(db: AsyncSession) -> dict: if stored.get("methodology") not in CATEGORICAL_FUNDAMENTAL_METHODOLOGIES: return default capex = _normalise_capex_states(stored.get("capex"), names) - reaction = str(stored.get("good_news_stock_down", "mixed")).strip().lower() + reaction = str(stored.get("good_news_stock_down", "unknown")).strip().lower() if reaction not in GNSD_STATES: - reaction = "mixed" + reaction = "unknown" return { **default, **stored, @@ -922,6 +1137,100 @@ def _score_capex_states(capex: dict[str, str], names: list[str]) -> float | None return round(score, 1) if score is not None else None +async def record_fundamental_observation(db: AsyncSession, observation: dict) -> None: + """Append the observation to the point-in-time series, keyed on effective date. + + Upsert rather than insert: re-saving on the same effective date is a + correction to that day's reading, not a second observation of it. + + Silently does nothing without an effective date or a ``fetched_at``. Those + are the default placeholder blob -- the absence of an observation, which must + never enter the series as though someone had looked. + + Deliberately does **not** commit. ``update_regime_monitor`` calls this inside + a run that owns its transaction and commits once after the snapshot loop; + committing here would take that boundary away from it. The two override + writers commit for themselves. + """ + effective = _parse_date(observation.get("effective_date")) + fetched_raw = observation.get("fetched_at") + if effective is None or not fetched_raw: + return + try: + fetched = datetime.fromisoformat(str(fetched_raw)) + except ValueError: + fetched = datetime.now(timezone.utc) + if fetched.tzinfo is None: + fetched = fetched.replace(tzinfo=timezone.utc) + + existing = await db.execute( + select(RegimeFundamentalObservation).where( + RegimeFundamentalObservation.effective_date == effective + ) + ) + row = existing.scalar_one_or_none() + payload = { + "f1_score": observation.get("f1_score"), + "f3_score": observation.get("f3_score"), + "capex_json": json.dumps(observation.get("capex") or {}), + "good_news_stock_down": str(observation.get("good_news_stock_down") or "unknown")[:10], + "reasoning": observation.get("reasoning"), + "source": str(observation.get("source") or "unknown")[:30], + "fetched_at": fetched, + } + if row is None: + db.add(RegimeFundamentalObservation( + effective_date=effective, + created_at=datetime.now(timezone.utc), + **payload, + )) + else: + for key, value in payload.items(): + setattr(row, key, value) + + +async def get_fundamental_observations(db: AsyncSession) -> list[dict]: + """The whole observation series, oldest first, for point-in-time scoring.""" + result = await db.execute( + select(RegimeFundamentalObservation).order_by( + RegimeFundamentalObservation.effective_date.asc() + ) + ) + out: list[dict] = [] + for row in result.scalars().all(): + try: + capex = json.loads(row.capex_json) + except (TypeError, ValueError): + capex = {} + out.append({ + "effective_date": row.effective_date, + "f1_score": row.f1_score, + "f3_score": row.f3_score, + "capex": capex, + "good_news_stock_down": row.good_news_stock_down, + "reasoning": row.reasoning, + "source": row.source, + "fetched_at": row.fetched_at.isoformat() if row.fetched_at else None, + }) + return out + + +def observation_asof(observations: list[dict] | None, as_of: date) -> dict | None: + """Latest observation effective on or before ``as_of``. + + This *is* the effective-date gate now. The settings-blob version had to + recompute it per call because there was only ever one observation to gate; + with a series, "which reading was live that day" is just a lookup. + """ + chosen: dict | None = None + for observation in observations or []: + if observation["effective_date"] <= as_of: + chosen = observation + else: + break + return chosen + + async def set_fundamental_overrides( db: AsyncSession, capex: dict[str, str] | None = None, @@ -954,7 +1263,15 @@ async def set_fundamental_overrides( "fetched_at": now.isoformat(), "effective_date": _next_weekday(now.date()).isoformat(), }) - await update_setting(db, KEY_FUNDAMENTALS, json.dumps(current)) + # The blob (what the live card reads) and the series row (what the + # point-in-time replay reads) are the same observation. Committed together: + # `update_setting` commits internally, so using it here would leave a window + # where a failure publishes the reading to the card but not to the record, + # and the two would disagree permanently with nothing to detect it. + await settings_store.upsert_setting(db, KEY_FUNDAMENTALS, json.dumps(current)) + if observation_changed: + await record_fundamental_observation(db, current) + await db.commit() return current @@ -1058,12 +1375,59 @@ def _snapshot_revision(snapshot: dict) -> int: return 1 +def _context_from_legacy_overlay(overlay: dict) -> dict: + """Rebuild the categorical channel from a pre-rename snapshot's overlay. + + The channel was called ``fundamental_overlay`` until 2026-08-12 and stored + the same underlying facts -- the capex map, the earnings reaction, the + effective date. The rename shipped without a methodology bump (no score + changed), so those rows are still served and were never reseeded: reading + only the new key would turn every one of them into ``unknown`` and silently + discard real recorded evidence -- historical Path colours, and any exposure + the event study could legitimately count. + + Derived, not guessed. The hyperscaler list comes from the overlay's own + capex keys, which is exactly the basket that was observed at the time rather + than today's configured one. + """ + capex = overlay.get("capex") or {} + reaction = overlay.get("good_news_stock_down") + names = list(capex) + pending = bool(overlay.get("pending")) + stale = bool(overlay.get("stale")) + observed = not pending and bool(overlay.get("fetched_at")) + + capex_signal = _capex_signal(capex, names) if observed else "unknown" + reaction_signal = _reaction_signal(reaction) if observed else "unknown" + state = combine_fundamental_signals(capex_signal, reaction_signal) + return { + **overlay, + "state": state, + "evidence_quality": _evidence_quality( + capex, reaction, names, + observed=observed, stale=stale, source=overlay.get("source"), + ), + "capex_signal": capex_signal, + "reaction_signal": reaction_signal, + "usable": _usable_context(observed, pending, stale, state), + } + + def _parse_snapshot(raw: str) -> dict | None: try: parsed = json.loads(raw) except (TypeError, ValueError): return None - return parsed if parsed.get("methodology") == METHODOLOGY else None + if parsed.get("methodology") != METHODOLOGY: + return None + # Normalise here rather than at each call site: every reader of a stored + # snapshot goes through this function, so a legacy row cannot reach one of + # them un-adapted. + if "fundamental_context" not in parsed and "fundamental_overlay" in parsed: + parsed["fundamental_context"] = _context_from_legacy_overlay( + parsed["fundamental_overlay"] or {} + ) + return parsed async def _latest_snapshot_row(db: AsyncSession) -> tuple[RegimeSnapshot, dict] | None: @@ -1082,6 +1446,10 @@ async def update_regime_monitor( ) -> dict: config = await get_regime_config(db) overrides = await get_fundamental_overrides(db) + # Carries the pre-v5 single-slot observation into the series on first run, so + # a deployment does not lose the live reading. A no-op once recorded, and a + # no-op for the placeholder blob (no fetched_at). + await record_fundamental_observation(db, overrides) if _fundamentals_stale(overrides, config) and not overrides.get("locked"): try: overrides = await refresh_fundamental_overrides(db, config=config) @@ -1131,6 +1499,9 @@ async def update_regime_monitor( breadth_series = _mapping_series(breadth) divergence_series = _mapping_series(divergence) + # Loaded once, after any refresh, so a reseed scores each replayed date with + # the observation that was effective on it rather than with today's. + observations = await get_fundamental_observations(db) latest_result: dict | None = None snapshots_written = 0 for snapshot_date in dates: @@ -1144,6 +1515,7 @@ async def update_regime_monitor( breadth_series, divergence_series, breadth_counts, + observations=observations, ) written, latest_result = await _upsert_snapshot( db, @@ -1221,16 +1593,22 @@ async def get_regime_monitor(db: AsyncSession) -> dict: quality["is_fresh"] = bool(quality.get("inputs_fresh")) and snapshot_age <= 4 result["data_quality"] = quality - # The snapshot's overlay is the point-in-time record; the reader also wants - # the current observation even when it is not effective until the next - # session, because otherwise refreshing it looks like it did nothing. + # The snapshot's `fundamental_context` is the point-in-time record; the + # reader also wants the current observation even when it is not effective + # until the next session, or refreshing it looks like it did nothing. config = await get_regime_config(db) overrides = await get_fundamental_overrides(db) live = current_observation(overrides, config, date.today()) - # Deliberately reads the *snapshot's* overlay, not the live one: this is how + # Deliberately reads the *snapshot's* record, not the live one: this is how # the reader tells "shown here" from "in the stored record". - live["observed_in_snapshot"] = bool((result.get("fundamental_overlay") or {}).get("available")) - result["fundamental_context"] = live + live["observed_in_snapshot"] = bool( + (result.get("fundamental_context") or {}).get("available") + ) + # `fundamental_context` is the stored channel and stays the snapshot's; + # `fundamental_live` is what we know right now. Collapsing the two under one + # key is what made a just-collected observation look like it had been + # backdated into history. + result["fundamental_live"] = live result["available"] = True return result @@ -1248,10 +1626,17 @@ async def get_regime_history(db: AsyncSession, days: int = 800) -> list[dict]: if data is None: continue state, warning = data.get("state") or {}, data.get("warning") or {} + context = data.get("fundamental_context") or {} out.append({ "date": row.date.isoformat(), "state": state.get("score") if state.get("band") is not None else None, "warning": warning.get("score") if warning.get("band") is not None else None, + # The third channel, carried per point so the Path view can colour a + # dot by the fundamental context that was on the record that day. + # Rows written before the channel existed carry nothing, which reads + # as "unknown" -- correct, since nothing was observed then either. + "fundamental_state": context.get("state") or "unknown", + "evidence_quality": context.get("evidence_quality") or "unavailable", "state_coverage": state.get("coverage"), "warning_coverage": warning.get("coverage"), "basket_hash": (data.get("basket") or {}).get("hash"), @@ -1383,7 +1768,7 @@ async def refresh_fundamental_overrides( f1 = _score_capex_states(capex, names) reaction = str(parsed.get("good_news_stock_down", "")).strip().lower() if reaction not in GNSD_STATES: - reaction = "mixed" + reaction = "unknown" f3 = _GNSD_SCORES.get(reaction) now = datetime.now(timezone.utc) result = { @@ -1398,7 +1783,11 @@ async def refresh_fundamental_overrides( "locked": False, "source": llm.get("provider"), } - await update_setting(db, KEY_FUNDAMENTALS, json.dumps(result)) + # One transaction: see set_fundamental_overrides on why these two writes must + # not be able to land separately. + await settings_store.upsert_setting(db, KEY_FUNDAMENTALS, json.dumps(result)) + await record_fundamental_observation(db, result) + await db.commit() logger.info(json.dumps({ "event": "regime_fundamentals_refreshed", "f1": result["f1_score"], diff --git a/docs/research/regime-monitor-v4.md b/docs/research/regime-monitor-v4.md index 494bbfd..7c522d9 100644 --- a/docs/research/regime-monitor-v4.md +++ b/docs/research/regime-monitor-v4.md @@ -39,6 +39,162 @@ session, the calendar anchors, 100% coverage on every row, and a row-wise `state_v4 <= state_v3` invariant. Reading a calibration result out of a run whose pipeline did not validate is meant to be structurally impossible. +## The fundamental channel (2026-08-12) + +The monitor has **three channels**, not two scores with a decoration: + +- **State** — current observable technical stress (price, breadth, credit, volatility). +- **Warning** — observable deterioration that may precede stress (breadth + divergence, relative strength, credit impulse). +- **Fundamental context** — a categorical state (`supportive` / `neutral` / + `adverse` / `unknown`) with an `evidence_quality` grade. + +The third is **never a term in the other two**. They are read together by +confluence: + +| Warning | Fundamentals | Reading | +|---|---|---| +| Calm | Supportive/neutral | Normal | +| Elevated | Supportive/neutral | Technical warning, not fundamentally confirmed | +| Calm | Adverse | Fundamental concern; tape has not confirmed | +| Elevated | Adverse | Confluence — highest attention | + +`METHODOLOGY` stays **v4**: no score changed, so partitioning the history API and +discarding the event study cache would be churn. `STUDY_SCHEMA` moved to 3 +instead, and is now the only thing that discards a stale report. + +### Why the read is a channel and not a weight + +Two things are true at once, and only this shape honours both. + +**v3's reason for removing fundamentals from the score was wrong.** Not stale — +wrong. v3 argued that F1 (capex) and F3 (good-news-stock-down), carrying 12 + 8 +of 100 Warning points, "could not change any published conclusion" because pegged +they produced a Warning of exactly 20.0, below the alarm threshold. That +arithmetic holds only when *every* technical sensor reads exactly zero, which is +the one case that never matters. Warning is a weighted average, so the sensors +add: + +| technical Warning | without fundamentals | with them pegged | delta | +|---|---|---|---| +| 0 | 0.0 | 20.0 | +20.0 | +| 20 | 20.0 | 36.0 | +16.0 | +| 25 | 25.0 | **40.0** | +15.0 | +| 35 | 35.0 | **48.0** | +13.0 | +| 50 | 50.0 | 60.0 | +10.0 | +| 80 | 80.0 | 84.0 | +4.0 | + +Pegged fundamentals lowered the technical Warning needed to reach the 40 quadrant +divider from 40 to 25. That is a 15-point shift in where the alert fires, which +is emphatically a changed conclusion. The v3 section below is kept as written, +with this correction attached, because its reasoning is cited elsewhere in this +file and a silent overwrite would hide that the error was ever made. + +**But no weight is measurable either.** A weighted modifier was built and +reverted: 0–25 points added onto the technical Warning, sized so a maxed-out read +carried a calm tape over the 40 divider on its own. Nothing could justify the 25. +With ~10 correction events and essentially no fundamental history, any fusion +weight is a policy preference presented as a measurement — and the debate it +invites ("does the read deserve 10%, 20%, 30%?") has no evidence that can settle +it. Adding a slow categorical judgement to a fast continuous score also +manufactures precision by summing unlike things, and it forces a missing +observation to silently redistribute its weight onto the technical sensors, which +is the opposite of leaving it unknown. + +So: the read gets a channel, not a coefficient. Both facts survive — the v3 +removal was badly argued *and* no weight is defensible — because "report it +separately" is the only design that neither buries the observation nor invents a +number for it. + +### Derivation + +Deterministic, from the stored categorical facts. The LLM is an **extraction and +explanation layer**: it finds the capex guidance, classifies it, and cites it. +Fixed rules turn those facts into a state, so the same observation always yields +the same category. + +`capex_signal`: any `cutting` → adverse; else any `holding` → neutral; else all +known `raising` → supportive; nothing known → unknown. +`reaction_signal`: `yes` → adverse, `mixed` → neutral, `no` → supportive, +`unknown` → unknown. + +`mixed` and `unknown` are different reaction states and were merged until +2026-08-13. A failed LLM parse fell back to `mixed`, so an extraction error +became *neutral evidence* — an observation of normality manufactured out of a +bug. `mixed` now means an observed mixed reaction; anything unreadable, missing +or unattempted is `unknown` and contributes nothing. + +Combined by precedence, never by averaging: **any adverse read carries**; both +unknown → unknown; every observed signal supportive → supportive; otherwise +neutral. + +`unknown` is deliberately unreachable by combination. Averaging would let two +`cutting` reads and two `unknown` ones land on "neutral", presenting missing +evidence as evidence of normality — the same conflation `current_observation` +already refuses between "no observation" and "an observation of zero". Two cuts +and two unknowns read **adverse with `evidence_quality: partial`**. + +`evidence_quality` is ordered by what an operator needs first: `unavailable` +(nothing collected) → `stale` (past `fundamental_staleness_days`) → `manual` +(hand override) → `complete` / `partial`. + +### Presentation and alerts + +The Path view colours each dot by the fundamental state recorded that day; the +axes are untouched, because context is confluence information rather than a +position on either axis. The card leads with the state and evidence grade. + +Alerts stay **separate**, off one toggle: + +- quadrant change — the market axes moved (existing); +- `regime_fundamental` — the context changed, e.g. neutral → adverse; +- `regime_confluence` — Warning elevated *and* fundamentals adverse. + +`unknown` never alerts: an absence of evidence is not a change in the evidence, +and alerting on it would train the reader to ignore the channel. Both new +triggers seed silently on first run, as the quadrant alert does. + +### The observation is now a real time series + +`regime_fundamental_observations` (migration 033), one row per `effective_date`, +upserted. Before this it lived in a single `SystemSetting` slot that every +refresh overwrote, so no history existed at all — which made the read impossible +to replay, impossible to backtest, and meant a rebuild recorded every historical +session as if nothing had been observed. `update_regime_monitor` carries the +pre-existing single-slot observation into the series on its next run. + +### What this does not establish + +The table starts empty and fills one observation at a time, so the fundamental +rows are **untested, not failed**. Two things enforce that rather than one: + +- they are **coverage-matched** — scored only on sessions where the channel had + usable context and on corrections whose warning horizon fell inside it, with a + market-only comparator over the identical window so any difference between them + is the channel and not the window; +- `measurable` stays false until `MIN_EVENTS_FOR_CONFIDENCE` corrections are + covered, and the panel prints "insufficient exposure" rather than a ratio. + +Without the first, one day of coverage would render as 0/10 — recreating, one +observation later, exactly the tested-versus-unavailable confusion the flag was +added to prevent. The market rows are unchanged, and the 1/10 shipped-rule figure +remains a verdict on the technical sensors and the alert machinery alone. + +The rationale for expecting the read to matter is the operator's: hyperscaler +capex is the demand side of the entire AI trade, and good earnings being sold is +a classic late-cycle tell. Both are plausible. Neither is measured here, and this +file's convention is that published numbers are reproducible. + +**The path forward is accumulation, then a test — in that order.** Once enough +point-in-time observations exist, test whether the state improves prediction +*conditional on* Warning. If it does, a fitted and calibrated model has something +to fit; until then there is nothing to calibrate against. Backfilling would get +there faster: capex direction is derivable from the 10-Q/10-K capex line, which +the SEC fundamentals import already carries, and "good news, stock down" from +earnings dates plus next-day returns, which the Dolt earnings import already +carries. That last one is worth computing deterministically rather than asking +the LLM to judge, for the same reason the state derivation is rule-based. + ## What changed in v4 **V1 stopped saturating at VIX 30.** `(vix - 15) / 15` reached 100 at VIX 30 — @@ -81,6 +237,16 @@ a qualitative overlay reported beside the scores. Capex also stopped scoring `raising` and `holding` identically at 0: `holding` is the deceleration case and now scores 50, so a boom no longer reads the same as a stall. +> **Corrected 2026-08-12.** The claim in this paragraph is false. "Pegged +> they produced a Warning of exactly 20.0" describes only the case where every +> technical sensor reads zero; Warning is a weighted average, so in the general +> case those 20 points added +10 to +20 and moved the technical score needed to +> reach the 40 quadrant divider from 40 to 25. The observation was removed for +> being *underweighted*, on reasoning that mistook a corner case for the whole +> range. See "The fundamental channel" above for what replaced it — a separate +> categorical channel, not a restored weight. The capex `holding` rescale in the second half +> of this paragraph stands and is still live. + **The drawdown sensor stopped saturating.** v2 used `dd_pct * 5`, reaching 100 at a 20% drawdown — the 90th percentile of the observed distribution. 39 of 408 sessions sat at exactly 100 with no resolution left, and the price pillar showed @@ -126,7 +292,11 @@ upper half of the Warning axis was unreachable. - 60-session SMH/SPY relative-strength deterioration, 30%. - HY OAS 20-session widening, 25%. -Combined, RSP/SPY (former F4), and the NVDA canary (former P6) do not enter v3 or v4. +**Fundamental context** — a categorical third channel, not a term in either +score. See "The fundamental channel" above. + +Combined, RSP/SPY (former F4), and the NVDA canary (former P6) do not enter v3 +or v4. ## Calibration @@ -279,31 +449,73 @@ reseed exists to close. The history API and main chart show only snapshots match the current methodology, so a bump reseeds the series rather than splicing two formulas into one line. -The fundamental overlay keeps its effective date (normally the next session after +The fundamental channel keeps its effective date (normally the next session after collection) and is never replayed backward, so a rebuild cannot stamp today's -observation onto historical snapshots. Because the observation is stored in a -single slot, a refresh replaces the previously effective record: the snapshot -therefore reports the overlay as `pending` until the new effective date. +observation onto historical snapshots. Since the observations became a real +series (`regime_fundamental_observations`, migration 033), the effective-date +lookup *is* the gate: a replayed session gets whichever observation was live on +it, and sessions before the first one read `unknown`. -Two functions, deliberately: `fundamental_overlay` is the **record** and keeps +Two functions, deliberately: `fundamental_context` is the **record** and keeps the gate — it runs for every replayed date during a rebuild, so it must never grow a bypass flag. `current_observation` is the **live reading** behind -`fundamental_context`, and *reports* the effective date instead of blanking the +`fundamental_live`, and *reports* the effective date instead of blanking the content. Until 2026-08-07 the live reading called the gated function, so a just-collected observation stayed hidden until the next weekday — three days over a weekend — and refreshing appeared to do nothing. That was the opposite of what this section -already claimed. Showing it early cannot leak into a published number, because -nothing in the overlay is scored (see "Fundamentals left the score"). +already claimed. Showing it early cannot leak into a published score, because +nothing in the channel is scored. `current_observation` gates on `observed` (a non-null `fetched_at`, the one field every path writing real content stamps). Without it, the default override — -`unknown` for every hyperscaler and `mixed` for the reaction — was reported as a -live observation with `available: true`, so the card presented placeholders as a -collected reading. Those are the absence of an observation, not an observation of -absence. `fundamental_overlay` never had this problem: no observation means no -effective date, which means `pending`, which already blanks the content. +`unknown` for every hyperscaler and, since 2026-08-13, `unknown` for the reaction +— was reported as a live observation with `available: true`, so the card +presented placeholders as a collected reading. Those are the absence of an +observation, not an observation of absence. `fundamental_context` never had this +problem: no observation means no effective date, which means `pending`, which +already blanks the content. + +**`usable` is what may confirm; `available` is only what to display.** Three +distinct things, and collapsing any two of them is a bug: + +- `state` — the last thing observed. Survives going stale, so the card can show it. +- `available` — *timing*: there is an effective, non-stale record to display. +- `usable` — *content*: available **and** the observation actually determined + something (`state != "unknown"`). + +The confluence alert and all three coverage-matched study rules gate on `usable`. +Gating on `available` instead has two failure modes, and both were live at some +point in this design: + +1. a reading past `fundamental_staleness_days` would corroborate every Warning + crossing indefinitely — the strongest claim this channel makes, from the data + with the least right to make it; +2. an LLM run that failed to extract anything produces a perfectly fresh + observation that knows nothing. Counting it as exposure means repeated + extraction failures slowly accumulate coverage until the fundamental rows flip + to a *measurable* 0/8 — a failed result published for a channel that never saw + a thing, which is precisely what coverage-matching exists to prevent. + +**Pre-rename snapshots are adapted, not discarded.** The channel was stored as +`fundamental_overlay` until 2026-08-12. The rename shipped without a methodology +bump — no score changed — so those rows are still served and were never reseeded. +Reading only the new key would have turned every one of them into `unknown`, +silently dropping real recorded evidence: historical Path colours, and exposure +the event study can legitimately count. `_parse_snapshot` derives the channel +from a legacy overlay's own stored facts (its capex map supplies the basket, so +the derivation uses the names observed at the time rather than today's config). +Normalising there rather than at each call site means no reader can receive an +un-adapted row. Delete only after a reseed has rewritten the whole window. + +**The blob and the series row are one transaction.** They are the same +observation seen by the live card and by the point-in-time replay; committing +them separately leaves a window where a failure publishes one and not the other, +and the two then disagree permanently with nothing to detect it. Both writers use +`settings_store.upsert_setting` (which does not commit) plus a single commit; +`record_fundamental_observation` deliberately takes no commit of its own so +`update_regime_monitor` keeps its own transaction boundary. Each snapshot stores the fixed basket symbols, hash, and freeze date. Reconstructed history before that freeze date is retrospective/exploratory. @@ -321,39 +533,185 @@ today's number. The quadrant dividers rendered in Path view come from ## Warning study -The study calls the outcome a **10% correction**, not a regime break. The first -70% of sessions freezes the 80th-percentile warning threshold; alarm episodes are -measured on the final 30%. Because v3 dropped fundamentals from the score, the -study now measures exactly the live Warning score rather than a technical-only -approximation of it, and both are computed from one shared sensor definition -(`warning_sensor_scores`) so they cannot drift apart. +The study calls the outcome a **10% correction**, not a regime break. It measures +two rules against that outcome, plus enough context to tell whether either number +is any good. -A cached report is discarded when its methodology no longer matches, so the panel -reverts to "not run yet" after a bump rather than showing stale numbers. **Re-run -the Event Study job after cutting over to v4.** +A cached report is discarded when its methodology no longer matches *or* when +`STUDY_SCHEMA` moves, so the panel reverts to "not run yet" rather than showing +stale numbers or a report missing half its blocks. **Re-run the Event Study job +after a methodology cutover or a schema bump.** + +### The headline is the rule that actually fires + +Until 2026-08-12 the study measured a bare rising-edge crossing of an +80th-percentile threshold fitted on the first 70% of sessions. **Nothing consumes +that rule.** What reaches Telegram is `_collect_regime_quadrant`: a quadrant +change with State ≥ 50 and Warning ≥ 40 as fixed dividers, a ±5 hysteresis +deadband, a two-session confirmation, a 3-day cooldown, and a 75% coverage gate +on both axes. The two differ on every one of those axes, including the threshold +itself (a fitted ~32 against a shipped 40). + +`replay_quadrant_changes` replays the shipped state machine over the whole +sample. Three details are reproduced rather than cleaned up, because a state +machine written from first principles gets each of them wrong: + +- the prior session is classified against the **current baseline**, not against + its own predecessor, so confirmation asks "did yesterday already look like this + change" rather than "did yesterday change too"; +- the baseline advances only when an alert actually fires, so a change blocked by + confirmation or cooldown is re-evaluated against the old quadrant next session; +- one cooldown is shared by every quadrant change, so a 3→4 alert can swallow a + 4→2 alert three days later. + +Two consequences worth stating. The alarm is dated at the **confirmation**, not +at the first crossing, which costs one session of lead by construction. And the +rule alerts on changes in both directions, so the replay's exits are recorded but +filtered out by `entry_alarms` — only entering a Warning-high quadrant is a +warning about anything. + +The replay reuses `_compute_index` rather than re-deriving the axes. That is the +same anti-drift argument that produced `warning_sensor_scores`: the v2 study +re-derived Warning by hand and would have kept measuring the old construct +through a scoring change. State has no equivalent shared helper, so the snapshot +builder itself is the shared definition. + +**Nothing is fitted, so nothing needs protecting from a training set.** There is +no split, and every detected correction is evaluable instead of the four that +happen to land in the last 30%. The `underpowered` and "threshold frozen on a +different construct" caveats do not apply to this variant. ### Reading the result -The report carries a `reliability` block and the UI renders its warnings, because -the headline numbers invite over-reading in two specific ways. +A bare "2 of 4" is unreadable in either direction, so the report scores four more +rules through the same `evaluate_alarms` harness over the same events and +sessions, and adds a null. All use fixed thresholds — a threshold fitted on the +full sample would have lookahead the shipped rule does not, and one fitted on a +split could only be scored on the holdout events. + +| kind | rules | the question | +|---|---|---| +| ablation | Warning ≥ 40 bare, State ≥ 50 bare | does the quadrant machinery earn its place? | +| baseline | leader below its 50-DMA, VIX ≥ 20 | does the score earn its complexity? | +| null | K random alarms at the observed firing rate | is any of this better than chance? | + +The two kinds must not be read as one list. If a baseline matches the score, the +composite is not earning its complexity and that is the finding — it does not +mean the monitor is worthless, since State and Warning exist to be *read*, but it +caps how much further calibration is justified. If the bare Warning crossing +beats the shipped rule, the machinery (not the sensor) is what is costing recall. + +The null draws only from sessions a rule could actually have fired on. Over the +whole sample it would be diluted by warm-up sessions and would understate what +chance achieves — which matters, because with ~11 events and a 20-session horizon +roughly a sixth of the sample already sits inside a hit window. It is seeded, so +a re-run cannot move the report. Corrections cluster and uniform placement does +not, so it is the **floor, not the bar**: an alarm process that clustered would +beat it for reasons unrelated to foresight. + +### First result (2026-08-12): the shipped rule is not distinguishable from chance + +Replayed over 2021-07-14 → 2026-08-12. The 200-DMA warm-up means the baseline +only seeds on 2022-05-26, so 1056 of 1276 sessions are evaluable and 10 of the 11 +detected corrections fall inside them. + +| rule | kind | warned | FA/yr | median lead | +|---|---|---|---|---| +| **Quadrant alert (shipped)** | | **1/10** | **0.9** | 19d | +| Quadrant alert, both axes high | ablation | 0/10 | 0.9 | — | +| Warning ≥ 40, bare crossing | ablation | 3/10 | 4.8 | 20d | +| State ≥ 50, bare crossing | ablation | 0/10 | 0.7 | — | +| SMH below its 50-DMA | baseline | 7/10 | 6.7 | 8d | +| VIX ≥ 20 | baseline | 4/10 | 7.2 | 9.5d | +| Random alarms, same firing rate | null | 0.9 ± 0.8 | — | — | + +**P(chance ≥ 1/10) = 0.65.** Alarms scattered at random over the same sessions at +the rule's own firing rate match or beat it two times in three. Whatever the +score knows, this rule is not transmitting it. + +Three readings, in order of how much they should change: + +**The machinery costs more than it protects.** The bare Warning crossing catches +3 with a 20-session lead; wrapping it in the quadrant rule drops that to 1. The +State condition is the largest single cost — requiring both axes high catches +nothing at all, which is what a coincident axis gating a leading one predicts. +Hysteresis, the two-session confirmation and the shared cooldown between them +take the rest, and the cooldown is shared across *every* quadrant change, so +exits consume the budget that entries need. Only 5 of the 15 replayed changes are +Warning-high entries. + +**The crude baselines beat everything on recall, at a price.** SMH below its +50-DMA catches 7 of 10 — but at 6.7 false alarms a year against the shipped +rule's 0.9. That is a 7× recall improvement for 7× the noise, so it is not a +clean dominance and this table cannot settle it; the missing axis is what a false +alarm actually costs, which nothing here measures. What it does settle is that +the composite is not buying recall the 50-DMA does not already have. + +**The 0.9 false alarms/year is not the achievement it looks like.** A rule that +almost never fires has few false alarms by construction. Read the two columns +together or not at all. + +Recorded from an offline replay (live Alpaca + FRED, no database, breadth +computed from the same Alpaca closes rather than the stored universe). The job in +Admin → Jobs is the canonical path and reads breadth from the DB, so re-run it to +confirm these figures before treating them as the record. + +**This is a verdict on the market channels only.** The fundamental and confluence +rows in the same table are marked `measurable: false` and print "not measurable" +rather than a ratio: with an empty observation series they never fire, and a 0/10 +sitting in a comparison column would read as tested-and-failed. `false` here means +the input does not exist yet, not that the rule lost. + +(The figures above were also produced under a briefly-built weighted modifier and +came back bit-identical, which is what confirmed the modifier was inert over the +whole window — the numbers depend on the technical sensors alone either way.) + +**Not acted on.** Nothing in the alert path was changed on the strength of this. +The obvious candidates — dropping the State condition from the entry test, +separating the entry and exit cooldowns, or lowering the Warning divider — are +threshold changes to a live alerting rule and want their own decision. + +### The coverage gap relocates, it does not close + +Dropping the fitted threshold makes the whole sample evaluable, but most of the +extra events predate 2023-08. W3 does not exist there, so Warning renormalises to +`(W1×45 + W2×30)/75` and the fixed 40 divider is applied to a different construct +than it was reasoned about. The report therefore splits shipped-rule metrics at +the credit sensor's first session and the panel states both, because replacing +one misleading headline with a differently misleading one would be no gain. + +Convenient side effect: the pre-credit era *is* the "Warning without W3" +ablation, measured on real sessions rather than simulated ones, so that ablation +is not run separately. + +Alarms and events are assigned to eras by index, so an alarm days before the +boundary matching an event days after it lands in the earlier era. With the eras +years long and the events sparse, that costs nothing. + +### The fitted variant, kept for continuity + +The 70/30 percentile study is still computed and still reported, collapsed, with +its `reliability` block intact — it is a genuinely different question, and it is +what earlier revisions of this document report. Its caveats stand: **The holdout is thin.** The study detects 11 corrections across 5 years but the -70/30 split leaves only 4 in the test period. Recall is therefore one event away -from a materially different headline, and in practice the event that flips is -decided by where the frozen threshold happens to land rather than by whether the -score saw anything. The v3 cutover run illustrates it: v3 scored 2/4 against v2's -3/4, but "v3 without the credit sensor" scores 3/4 at a *higher* threshold -(35.5) than shipped v3 misses it at (32.3) — because the alarm rule needs a -rising edge, and a lower threshold can mean the alarm already fired outside the -20-session horizon and never reset below. Below `MIN_EVENTS_FOR_CONFIDENCE` -holdout events the report says so explicitly. +70/30 split leaves only 4 in the test period. Recall is one event away from a +materially different headline, and in practice the event that flips is decided by +where the frozen threshold happens to land rather than by whether the score saw +anything. The v3 cutover run illustrates it: v3 scored 2/4 against v2's 3/4, but +"v3 without the credit sensor" scores 3/4 at a *higher* threshold (35.5) than +shipped v3 misses it at (32.3) — because the alarm rule needs a rising edge, and a +lower threshold can mean the alarm already fired outside the 20-session horizon +and never reset below. Below `MIN_EVENTS_FOR_CONFIDENCE` holdout events the +report says so explicitly. Some events carry no information at all for comparison: in that run every variant caught 2026-03-06, every variant missed 2026-06-05, and every variant "caught" 2025-11-20 with a 1-session lead, which is coincident rather than a -warning. +warning. The headline recall does not currently discount those; a minimum-lead +rule is the obvious next change and has not been made. -**Sensor coverage can straddle the split.** The score renormalises over available +**Sensor coverage straddles the split.** The score renormalises over available sensors, so a training window predating a sensor's history freezes the threshold on a different construct than the holdout is measured against. At the v3 cutover only 39% of training sessions had all three Warning sensors versus 100% of the @@ -362,9 +720,22 @@ test period, because credit history begins 2023-07-25. Restricting the threshold to sensor-matched training sessions was tried and is *not* the fix: those sessions are a calm recent stretch, so the threshold drops from 32.3 to 22.5 and false alarms rise from 3.3 to 8.6 per year. It trades a -coverage bias for a regime-selection bias. The honest position is that the -threshold is hypersensitive to window choice at this sample size; the report -states its limits rather than pretending to a precision it does not have. +coverage bias for a regime-selection bias. The honest position is that a fitted +threshold is hypersensitive to window choice at this sample size — which is the +strongest argument for making the unfitted shipped rule the headline. + +### Considered and not done + +**An ETF credit proxy (HYG/IEF) to extend W3 back over the whole sample.** It +would trade "two sensors versus three" for "proxy sensor versus real sensor" — +still a construct straddle, but no longer flagged by the coverage split. This is +the same objection that rejected `BAA10Y` as a percentile reference. If ever +revisited, check the impulse correlation on the three years of real-OAS overlap +first and report it as a sensitivity, never as the headline. + +**A depth sweep (5%/7%/15% corrections) for more events.** `EVENT_COOLDOWN_DAYS` +is 40, so at shallower thresholds re-triggers inside a single decline merge or +drop and the denominator moves for cooldown reasons rather than market ones. ## Resolved in v4 (raised 2026-08-07, shipped 2026-08-08) diff --git a/frontend/src/components/regime/RegimeChart.tsx b/frontend/src/components/regime/RegimeChart.tsx index 60a99e6..2e2c839 100644 --- a/frontend/src/components/regime/RegimeChart.tsx +++ b/frontend/src/components/regime/RegimeChart.tsx @@ -2,7 +2,6 @@ import { useMemo, useState } from 'react'; import { useQuery } from '@tanstack/react-query'; import { CartesianGrid, - Cell, Line, LineChart, ReferenceArea, @@ -19,6 +18,8 @@ import { getRegimeHistory, getRegimeMonitor } from '../../api/regime'; import { Callout } from '../ui/Callout'; import { SkeletonCard } from '../ui/Skeleton'; import { formatDate } from '../../lib/format'; +import { FUNDAMENTAL_VISUAL, QUADRANT_WASH, REGIME_VISUAL } from '../../lib/regime'; +import type { EvidenceQuality, FundamentalState } from '../../lib/types'; // Lazy-loaded (see RegimePage) so recharts stays in the regime-tab chunk. // Time and Path are two projections of one series, so they share a card and a @@ -38,8 +39,14 @@ type RangeKey = (typeof RANGES)[number]['key']; /** Sessions drawn in Path view. The full series is unreadable as a path. */ const PATH_TRAIL = 60; -const STATE_COLOR = '#60a5fa'; -const WARNING_COLOR = '#fb923c'; +const STATE_COLOR = REGIME_VISUAL.state; +const WARNING_COLOR = REGIME_VISUAL.warning; +const FUNDAMENTAL_SYMBOL: Record = { + supportive: '▲', + neutral: '●', + adverse: '◆', + unknown: '○', +}; // Fall back to the shipped constants, not v2's shared 60/60, so a missing // quadrant_config cannot draw dividers that disagree with the alert path. @@ -50,13 +57,19 @@ interface PathPoint { x: number; y: number; date: string; + /** The third channel as recorded that day. Colours the dot; never moves it. */ + fundamental: FundamentalState; + evidence: EvidenceQuality; + /** Raw dated observations are interactive dots; the smoothed copy is line-only. */ + raw: boolean; + recency: number; } /** Centered moving average to de-noise the path; today (last) kept exact. */ function smoothTrail(points: PathPoint[], half = 2): PathPoint[] { const n = points.length; return points.map((p, i) => { - if (i === n - 1) return { ...p }; + if (i === n - 1) return { ...p, raw: false }; let sx = 0; let sy = 0; let c = 0; @@ -65,14 +78,58 @@ function smoothTrail(points: PathPoint[], half = 2): PathPoint[] { sy += points[j].y; c += 1; } - return { x: sx / c, y: sy / c, date: p.date }; + return { ...p, x: sx / c, y: sy / c, raw: false }; }); } -/** Recency gradient: 0 = oldest (muted slate), 1 = newest (bright blue). */ -function recencyColor(t: number): string { - const lerp = (a: number, b: number) => Math.round(a + (b - a) * t); - return `rgba(${lerp(71, 96)}, ${lerp(85, 165)}, ${lerp(105, 250)}, ${(0.3 + 0.7 * t).toFixed(2)})`; +function FundamentalGlyph({ + cx, + cy, + state, + size, + opacity = 1, +}: { + cx: number; + cy: number; + state: FundamentalState; + size: number; + opacity?: number; +}) { + const visual = FUNDAMENTAL_VISUAL[state] ?? FUNDAMENTAL_VISUAL.unknown; + const common = { fill: visual.color, opacity, stroke: '#11131c', strokeWidth: 1 }; + if (visual.glyph === 'up') { + return ; + } + if (visual.glyph === 'diamond') { + return ; + } + if (visual.glyph === 'ring') { + return ; + } + return ; +} + +function PathPointShape({ cx = 0, cy = 0, payload }: { cx?: number; cy?: number; payload?: PathPoint }) { + if (!payload) return ; + return ( + + ); +} + +function LatestPointShape({ cx = 0, cy = 0, payload }: { cx?: number; cy?: number; payload?: PathPoint }) { + if (!payload) return ; + return ( + + + + + ); } function SegmentedControl({ @@ -94,8 +151,8 @@ function SegmentedControl({ type="button" aria-pressed={value === option} onClick={() => onChange(option)} - className={`rounded px-2 py-1 text-[11px] font-medium tabular-nums transition-colors ${ - value === option ? 'bg-white/10 text-blue-300' : 'text-gray-500 hover:text-gray-300' + className={`min-h-9 rounded px-3 py-2 text-xs font-medium tabular-nums transition-colors ${ + value === option ? 'bg-white/10 text-blue-300' : 'text-gray-400 hover:text-gray-200' }`} > {option} @@ -107,14 +164,19 @@ function SegmentedControl({ function PathTip({ active, payload }: { active?: boolean; payload?: { payload: PathPoint }[] }) { if (!active || !payload?.length) return null; - const p = payload[0].payload; + const p = payload.find((item) => item.payload.raw)?.payload ?? payload[0].payload; + const visual = FUNDAMENTAL_VISUAL[p.fundamental] ?? FUNDAMENTAL_VISUAL.unknown; + const evidence = p.evidence === 'unavailable' ? 'Unavailable' : `${p.evidence.replace(/_/g, ' ')} evidence`; return ( -
+
{formatDate(p.date)}
State {Math.round(p.x)} · Warning{' '} {Math.round(p.y)}
+
+ Fundamentals {visual.label} · {evidence} +
); } @@ -144,7 +206,15 @@ export default function RegimeChart() { }, [history.data, view, range]); const pathPoints = useMemo( - () => series.map((p) => ({ x: p.state as number, y: p.warning as number, date: p.date })), + () => series.map((p, index, points) => ({ + x: p.state as number, + y: p.warning as number, + date: p.date, + fundamental: p.fundamental_state ?? 'unknown', + evidence: p.evidence_quality ?? 'unavailable', + raw: true, + recency: points.length <= 1 ? 1 : index / (points.length - 1), + })), [series], ); const trail = useMemo(() => (view === 'Path' ? smoothTrail(pathPoints) : []), [pathPoints, view]); @@ -159,7 +229,7 @@ export default function RegimeChart() {
- + {view === 'Time' ? 'State & Warning over time' : `State × Warning path · last ${PATH_TRAIL} sessions`} @@ -168,7 +238,7 @@ export default function RegimeChart() { r.key)} value={range} onChange={setRange} label="Time range" /> ) : ( latest && ( - + now: State {Math.round(latest.x)} · Warning{' '} {Math.round(latest.y)} @@ -182,14 +252,18 @@ export default function RegimeChart() { Not enough coverage-qualified history yet — it accumulates as the daily job runs. ) : ( <> -
+
{view === 'Time' ? ( formatDate(String(d))} minTickGap={28} tickLine={false} @@ -200,7 +274,7 @@ export default function RegimeChart() { ) : ( - - - - + {/* One neutral at four opacities: denser = more axes elevated. + Hue here would collide with the fundamental glyphs drawn + on top of it — see QUADRANT_WASH. */} + + + + @@ -237,36 +314,37 @@ export default function RegimeChart() { dataKey="x" domain={[0, 100]} ticks={[0, 20, 40, 60, 80, 100]} - tick={{ fill: '#6b7280', fontSize: 10 }} + tick={{ fill: '#9aa0b0', fontSize: 10 }} tickLine={false} axisLine={{ stroke: 'rgba(255,255,255,0.08)' }} - label={{ value: 'State →', position: 'insideBottom', offset: -12, fill: '#6b7280', fontSize: 10 }} + label={{ value: 'State →', position: 'insideBottom', offset: -12, fill: '#9aa0b0', fontSize: 10 }} /> - + } /> - - {trail.map((_, i) => ( - - ))} - + } + tooltipType="none" + isAnimationActive={false} + /> + } isAnimationActive={false} /> {latest && ( ( - - )} + shape={} /> )} @@ -275,7 +353,7 @@ export default function RegimeChart() {
{view === 'Time' ? ( -
+
State @@ -284,20 +362,43 @@ export default function RegimeChart() { Warning - dashed = each axis's elevated threshold ({xDiv} / {yDiv}) + dashed = each axis's elevated threshold ({xDiv} / {yDiv})
) : ( -
- Early warning — calm, fragility rising - Active stress — damaged and deteriorating - Healthy — calm, broadly supported - Stabilizing — damage remains, warning lower - White dot = today; trail brightens toward the present, smoothed. +
+ {/* Swatches, not coloured words: the quadrant names used the + fundamental channel's colours, so "Stabilizing" was rendered in + the adverse hue while meaning damage receding. */} + {([ + ['active_stress', 'Active stress', 'damaged and deteriorating'], + ['early_warning', 'Early warning', 'calm, fragility rising'], + ['stabilizing', 'Stabilizing', 'damage remains, warning lower'], + ['healthy', 'Healthy', 'calm, broadly supported'], + ] as const).map(([key, name, gloss]) => ( + + + ))} + Raw dated points grow toward today; the connecting line is smoothed. White ring = today. + + symbol + colour = fundamentals: + {(['supportive', 'neutral', 'adverse', 'unknown'] as const).map((state) => ( + + + {FUNDAMENTAL_VISUAL[state].label} + + ))} +
)} {crossesFreeze && ( -

+

History before {basketAsOf} is reconstructed against today's basket — retrospective, not a live record.

)} diff --git a/frontend/src/components/ui/Disclosure.tsx b/frontend/src/components/ui/Disclosure.tsx index 7d596e5..bba7ea9 100644 --- a/frontend/src/components/ui/Disclosure.tsx +++ b/frontend/src/components/ui/Disclosure.tsx @@ -9,7 +9,7 @@ interface DisclosureProps { export function Disclosure({ summary, children }: DisclosureProps) { return (
- + {summary} diff --git a/frontend/src/lib/regime.ts b/frontend/src/lib/regime.ts index b8bfc94..73e0e95 100644 --- a/frontend/src/lib/regime.ts +++ b/frontend/src/lib/regime.ts @@ -1,4 +1,68 @@ -import type { MarketRegime } from './types'; +import type { FundamentalState, MarketRegime } from './types'; + +/** One visual vocabulary for the three-channel regime monitor. Keep chart SVG + * literals and DOM text in sync rather than letting Tailwind aliases and + * hard-coded colours describe the same state differently. + * + * **Hue identifies the channel, and only the channel.** Two collisions made + * that false and both are fixed here: + * + * - `supportive` was literally `state`, so teal meant "the State score" in the + * Time view and "fundamentals supportive" in the Path view of the same card. + * - `adverse` sat 19 degrees from `warning`, which is inside deuteranope + * confusion range for two channels that appear on adjacent tooltip lines. + * + * The market pair now sits at 27/190 degrees and the fundamental pair at + * 0/158, so every *cross-channel* pair is at least 27 degrees apart. All six + * clear 4.5:1 against `--surface`. Fundamentals additionally carry a glyph, so + * colour is never the sole encoding for the categorical channel. + * + * `neutral` and `unknown` are deliberately the same hue: they are two states of + * one channel, both meaning "no directional signal", separated by lightness + * (7.1:1 vs 5.4:1) and by glyph (filled circle vs ring). Do not "fix" their + * proximity by giving `unknown` a hue — that would make an absence of evidence + * look like a reading. + * + * These deliberately do *not* reuse `--up-text`/`--down-text`: those are the + * app's directional tokens, and `--up-text` is already this chart's State + * colour, which is how the first collision happened. + */ +export const REGIME_VISUAL = { + // Market channels — continuous scores, drawn as lines and positions. + state: '#6ec9db', + warning: '#fb923c', + // Fundamental channel — categorical, drawn as glyphs. + supportive: '#34d399', + neutral: '#9aa0b0', + adverse: '#f87171', + unknown: '#848a9c', +} as const; + +/** Market quadrant severity as an opacity ramp on one neutral — never a hue. + * + * The quadrants are a State x Warning construct, so colouring them borrowed + * hues that already meant something else: "Healthy" was painted in the + * fundamental supportive colour and "Stabilizing" in the adverse one, which put + * an adverse glyph on an adverse-coloured background while meaning roughly the + * opposite (damage receding). Opacity carries how many axes are elevated, the + * position and labels carry which, and hue stays free to mean channel. + */ +export const QUADRANT_WASH = { + healthy: 0.015, + early_warning: 0.05, + stabilizing: 0.05, + active_stress: 0.085, +} as const; + +export const FUNDAMENTAL_VISUAL: Record< + FundamentalState, + { label: string; color: string; glyph: 'up' | 'circle' | 'diamond' | 'ring' } +> = { + supportive: { label: 'Supportive', color: REGIME_VISUAL.supportive, glyph: 'up' }, + neutral: { label: 'Neutral', color: REGIME_VISUAL.neutral, glyph: 'circle' }, + adverse: { label: 'Adverse', color: REGIME_VISUAL.adverse, glyph: 'diamond' }, + unknown: { label: 'Unknown', color: REGIME_VISUAL.unknown, glyph: 'ring' }, +}; export function regimeDot(label: MarketRegime['label']): string { switch (label) { diff --git a/frontend/src/lib/types.ts b/frontend/src/lib/types.ts index 284e891..f52dde3 100644 --- a/frontend/src/lib/types.ts +++ b/frontend/src/lib/types.ts @@ -509,9 +509,24 @@ export interface RegimeReading { trend?: { delta_7: number | null; delta_30: number | null }; } -/** Qualitative capex / earnings-reaction context. Not part of either score. */ -export interface RegimeFundamentalOverlay { +export type FundamentalState = 'supportive' | 'neutral' | 'adverse' | 'unknown'; +export type EvidenceQuality = 'complete' | 'partial' | 'stale' | 'manual' | 'unavailable'; + +/** The third channel: capex / earnings-reaction context, read alongside State + * and Warning by confluence. Deliberately never a term in either score — see + * the methodology doc on why no fusion weight is measurable yet. */ +export interface RegimeFundamentalContext { + /** Derived from the stored facts by fixed rules, not by an LLM's judgement. */ + state: FundamentalState; + evidence_quality: EvidenceQuality; + capex_signal: FundamentalState; + reaction_signal: FundamentalState; + /** Timing only: there is an effective, non-stale record to display. */ available: boolean; + /** Content too: it is available *and* actually determined something. A + * collected observation whose extraction failed is available but not usable, + * and only `usable` may confirm anything or count as study exposure. */ + usable: boolean; pending: boolean; stale: boolean; effective_date: string | null; @@ -524,7 +539,7 @@ export interface RegimeFundamentalOverlay { source: string | null; fetched_at: string | null; /** Whether anything was actually collected. Live reading only; the snapshot's - * point-in-time overlay omits it. */ + * point-in-time record omits it. */ observed?: boolean; observed_in_snapshot?: boolean; } @@ -533,6 +548,11 @@ export interface RegimeHistoryPoint { date: string; state: number | null; warning: number | null; + /** The fundamental channel as recorded that day — drives the Path dot colour. + * Rows written before the channel existed read as "unknown", which is correct: + * nothing was observed then either. */ + fundamental_state: FundamentalState; + evidence_quality: EvidenceQuality; state_coverage: number | null; warning_coverage: number | null; basket_hash: string | null; @@ -545,10 +565,12 @@ export interface RegimeMonitor { date?: string; state?: RegimeReading; warning?: RegimeReading; - /** Point-in-time overlay recorded in the snapshot. */ - fundamental_overlay?: RegimeFundamentalOverlay; - /** Current observation, even when it is not effective until the next session. */ - fundamental_context?: RegimeFundamentalOverlay; + /** The channel as recorded in the snapshot — point-in-time, effective-date gated. */ + fundamental_context?: RegimeFundamentalContext; + /** What we know right now, even when it is not effective until the next + * session. Separate from the above so a just-collected observation cannot + * look as though it had been backdated into the record. */ + fundamental_live?: RegimeFundamentalContext; inputs?: { vix: number | null; vix_date: string | null; @@ -596,7 +618,7 @@ export interface RegimeFundamentals { } export type CapexState = 'raising' | 'holding' | 'cutting' | 'unknown'; -export type GoodNewsReaction = 'yes' | 'no' | 'mixed'; +export type GoodNewsReaction = 'yes' | 'no' | 'mixed' | 'unknown'; export interface RegimeFundamentalsUpdate { capex?: Record; @@ -611,9 +633,22 @@ export interface RegimeConfig { } // Event study — measured lead time of early-warning indicators vs. drawdowns +export interface EventStudyMetrics { + events: number; + events_warned: number; + events_missed: number; + alarm_episodes: number; + false_alarms: number; + /** null when the rule had no eligible sessions — undefined, not zero. */ + false_alarms_per_year: number | null; + median_lead_days: number | null; +} + export interface EventStudyReport { available: boolean; reason?: string; + /** Report shape, independent of methodology. Mismatched reports are discarded. */ + schema?: number; methodology?: string; generated_at?: string; evaluation?: 'exploratory' | 'holdout'; @@ -624,14 +659,11 @@ export interface EventStudyReport { event_threshold_pct: number; event_cooldown_days: number; horizon_days: number; - train_fraction: number; - warn_percentile: number; - warn_threshold: number; basket_hash: string; basket_asof: string; credit_sensor_from?: string | null; }; - /** How far the headline metrics can be trusted. See _reliability(). */ + /** How far the *fitted* variant's metrics can be trusted. See _reliability(). */ reliability?: { events_detected: number; events_in_holdout: number; @@ -645,21 +677,75 @@ export interface EventStudyReport { sample?: { start: string; end: string; - train_end: string; - test_start: string; + /** Where the quadrant baseline seeds — not a holdout boundary. */ + evaluable_from: string; sessions: number; - holdout_sessions: number; + evaluable_sessions: number; + events_detected: number; + events_evaluable: number; }; - metrics?: { + /** The quadrant-change rule that actually reaches Telegram. The headline. */ + shipped?: { + rule: { + state_divider: number; + warning_divider: number; + margin: number; + confirm_sessions: number; + cooldown_days: number; + entry: string; + }; + metrics: EventStudyMetrics; + events: { date: string; warned: boolean; lead_days: number | null }[]; + quadrant_changes: number; + /** Debugging payload: every change the replay would have alerted on. Not rendered. */ + fires: { index: number; date: string; from: string; to: string; state: number; warning: number }[]; + /** Credit history starts partway through, so Warning is W1+W2 before it. */ + by_era?: { + credit_from: string; + pre_credit: EventStudyMetrics & { label: string; start: string; end: string; sessions: number }; + full_coverage: EventStudyMetrics & { label: string; start: string; end: string; sessions: number }; + } | null; + }; + /** The fundamental channel's actual exposure — its rows are scored on this + * window, not on the market rows' full sample. */ + fundamental_coverage?: { + observations: number; + /** Sessions with usable (observed, effective, non-stale) context. */ + sessions_eligible: number; + evaluable_sessions: number; + /** Corrections whose warning horizon had usable context. */ + events_covered: number; + events_evaluable: number; + minimum_events: number; + /** False until enough corrections are covered: the fundamental rows are + * untested, not failed, and must not render as a 0/N result. */ + measurable: boolean; + }; + /** Ablations, external baselines, and the fundamental channel — all on fixed + * (unfitted) rules, so every row is scored on the same events. */ + comparison?: (EventStudyMetrics & { + id: string; + label: string; + kind: 'ablation' | 'baseline' | 'fundamental'; + note: string; + measurable: boolean; + })[]; + null_model?: { + draws: number; + alarms_per_draw: number; events: number; - events_warned: number; - events_missed: number; - alarm_episodes: number; - false_alarms: number; - false_alarms_per_year: number; - median_lead_days: number | null; + mean_warned: number; + sd_warned: number; + observed_warned: number; + p_at_least_observed: number; + } | null; + /** The original 70/30 fitted-threshold study, kept for continuity. */ + fitted?: { + params: { train_fraction: number; warn_percentile: number; warn_threshold: number }; + sample: { train_end: string; test_start: string; holdout_sessions: number }; + metrics: EventStudyMetrics; + events: { date: string; warned: boolean; lead_days: number | null }[]; }; - events?: { date: string; warned: boolean; lead_days: number | null }[]; recent_breadth?: { date: string; breadth: number; warning: number | null }[]; } diff --git a/frontend/src/pages/RegimePage.tsx b/frontend/src/pages/RegimePage.tsx index 4fac24e..b89f53a 100644 --- a/frontend/src/pages/RegimePage.tsx +++ b/frontend/src/pages/RegimePage.tsx @@ -6,6 +6,7 @@ import { Disclosure } from '../components/ui/Disclosure'; import { Badge } from '../components/ui/Badge'; import { SkeletonCard, SkeletonTable } from '../components/ui/Skeleton'; import { useAuthStore } from '../stores/authStore'; +import { FUNDAMENTAL_VISUAL } from '../lib/regime'; import { getEventStudy, getRegimeConfig, @@ -17,11 +18,13 @@ import { } from '../api/regime'; import type { CapexState, + FundamentalState, + EventStudyMetrics, EventStudyReport, GoodNewsReaction, RegimeBand, RegimeConfig, - RegimeFundamentalOverlay, + RegimeFundamentalContext, RegimeFundamentals, RegimeFundamentalsUpdate, RegimeMonitor, @@ -39,7 +42,7 @@ const BAND_STYLES: Record{label}: n/a; + return {label}: n/a; } const color = delta === 0 ? 'text-gray-400' : delta > 0 ? 'text-red-400' : 'text-emerald-400'; const arrow = delta === 0 ? '→' : delta > 0 ? '↑' : '↓'; @@ -68,21 +71,21 @@ function ScoreGauge({ // shared set would mislabel one of them. Render none rather than wrong ones. const ticks = bands ? [bands.watch, bands.elevated, bands.breaking] : []; return ( -
+
-
{label}
+
{label}
- + {score == null ? '—' : Math.round(score)} - {score != null && / 100} + {score != null && / 100}
{style?.label ?? 'Incomplete'} - coverage {Math.round(reading?.coverage ?? 0)}% + coverage {Math.round(reading?.coverage ?? 0)}%
@@ -102,7 +105,7 @@ function ScoreGauge({ />
{/* Thresholds come from the reading: the two axes no longer share them. */} -
+
0 {ticks.map((tick) => ( @@ -113,86 +116,153 @@ function ScoreGauge({
)} -

{footnote}

+

{footnote}

); } -const CAPEX_TONE: Record = { - raising: 'text-emerald-400', - holding: 'text-amber-400', - cutting: 'text-red-400', - unknown: 'text-gray-500', +/** Mirrors `_capex_signal`: holding is the neutral case, so it takes the neutral + * colour rather than an amber that reads as a third severity and sits close to + * the Warning channel's orange. */ +const CAPEX_COLOR: Record = { + raising: FUNDAMENTAL_VISUAL.supportive.color, + holding: FUNDAMENTAL_VISUAL.neutral.color, + cutting: FUNDAMENTAL_VISUAL.adverse.color, + unknown: FUNDAMENTAL_VISUAL.unknown.color, }; -const OVERLAY_TITLE = 'Fundamental overlay · context, not scored'; +function sentenceCase(value: string): string { + const text = value.replace(/_/g, ' '); + return text.charAt(0).toUpperCase() + text.slice(1); +} -function FundamentalOverlayCard({ overlay }: { overlay: RegimeFundamentalOverlay }) { - const capex = overlay.capex ?? {}; - const reaction = overlay.good_news_stock_down; - - // Nothing collected: the stored default is "unknown" for every hyperscaler - // and "mixed" for the reaction, which are placeholders, not a reading. - if (overlay.observed === false) { - return ( -
-
{OVERLAY_TITLE}
-

- No observation collected yet. An admin can collect one under Admin · Monitor settings. It is - context only — it never enters State or Warning. -

-
- ); +function reactionReading(reaction: GoodNewsReaction | null): { label: string; color: string } { + switch (reaction) { + case 'yes': + return { label: 'Yes · good news sold', color: FUNDAMENTAL_VISUAL.adverse.color }; + case 'no': + return { label: 'No · ordinary reactions', color: FUNDAMENTAL_VISUAL.supportive.color }; + case 'mixed': + return { label: 'Mixed · no clear pattern', color: FUNDAMENTAL_VISUAL.neutral.color }; + default: + return { label: 'Unknown · not observed', color: FUNDAMENTAL_VISUAL.unknown.color }; } +} + +function FundamentalSummaryCard({ overlay }: { overlay: RegimeFundamentalContext }) { + const tone = FUNDAMENTAL_VISUAL[overlay.state] ?? FUNDAMENTAL_VISUAL.unknown; + const observed = overlay.observed ?? Boolean(overlay.fetched_at); + const status = !observed + ? 'No usable observation. This channel remains Unknown.' + : overlay.pending + ? `Collected now; enters the point-in-time record ${overlay.effective_date ?? 'next session'}.` + : overlay.stale + ? 'The last state is retained for context, but stale evidence cannot confirm alerts.' + : !overlay.usable + ? 'An observation was collected, but no signal could be determined.' + : null; return ( -
-
-
{OVERLAY_TITLE}
-
- {overlay.source && {overlay.source}} - {/* When pending, the line below is the single carrier of this date. */} - {overlay.effective_date && !overlay.pending && · effective {overlay.effective_date}} +
+
+
Fundamentals · context
+
{overlay.pending && } {overlay.stale && }
- {/* A pending observation is still shown — it is the freshest read we - have, and nothing here is scored. The date says when the stored - point-in-time record picks it up. */} - {overlay.pending && ( -

- Shown as collected. The point-in-time record picks it up{' '} - {overlay.effective_date ?? 'next session'} — observations are never backdated. -

- )} -
-
-
- Hyperscaler capex guidance - {overlay.capex_stress ?? 'n/a'} -
-
- {Object.entries(capex).map(([symbol, state]) => ( -
- {symbol} - {state} -
- ))} +
+ {tone.label} + + {sentenceCase(overlay.evidence_quality)} evidence + +
+ +
+
+
Capex
+
+ {sentenceCase(overlay.capex_signal)}
-
-
- Good news, stock down - {overlay.earnings_stress ?? 'n/a'} -
-
- {reaction === 'yes' ? 'Yes — beats sold into' : reaction === 'no' ? 'No — ordinary reactions' : 'Mixed'} +
+
Reaction
+
+ {sentenceCase(overlay.reaction_signal)}
- {overlay.reasoning &&

{overlay.reasoning}

} + + {status &&

{status}

} + {(overlay.source || overlay.effective_date) && ( +

+ {overlay.source ?? 'stored observation'} + {overlay.effective_date && ` · effective ${overlay.effective_date}`} +

+ )} +
+ ); +} + +function FundamentalEvidence({ overlay }: { overlay: RegimeFundamentalContext }) { + const observed = overlay.observed ?? Boolean(overlay.fetched_at); + if (!observed) return null; + const capex = overlay.capex ?? {}; + const reaction = reactionReading(overlay.good_news_stock_down); + + return ( + +
+
+
Hyperscaler capex guidance
+ {Object.keys(capex).length === 0 ? ( +

No company-level observation.

+ ) : ( +
+ {Object.entries(capex).map(([symbol, state]) => ( +
+ {symbol} + {sentenceCase(state)} +
+ ))} +
+ )} +
+
+
Good news, stock down
+
{reaction.label}
+

+ Derived context: capex {overlay.capex_signal} · reaction {overlay.reaction_signal}. +

+
+
+ {overlay.reasoning && ( +
+ + Source reasoning + +

{overlay.reasoning}

+
+ )} +
+ ); +} + +function ConfluenceStrip({ warning, context }: { warning: RegimeReading; context?: RegimeFundamentalContext }) { + const warningElevated = warning.band === 'elevated' || warning.band === 'breaking'; + if (!warningElevated || !context?.usable || context.state !== 'adverse') return null; + + return ( +
+
); } @@ -209,7 +279,7 @@ function PillarTable({ state, warning }: { state: RegimeReading; warning: Regime
- + @@ -221,7 +291,7 @@ function PillarTable({ state, warning }: { state: RegimeReading; warning: Regime @@ -232,8 +302,8 @@ function PillarTable({ state, warning }: { state: RegimeReading; warning: Regime
{pillar.label}
{pillar.sensors.map((sensor) => ( -
- {sensor.id} {sensor.label}:{' '} +
+ {sensor.id} {sensor.label}:{' '} {sensor.score == null ? 'n/a' : sensor.score}
))} @@ -256,7 +326,7 @@ function PillarTable({ state, warning }: { state: RegimeReading; warning: Regime function MetaChip({ label, value, title }: { label: string; value: ReactNode; title?: string }) { return ( - + {label} {value} ); @@ -288,71 +358,291 @@ function MetaStrip({ data }: { data: RegimeMonitor }) { ); } +function StatTiles({ metrics }: { metrics: EventStudyMetrics }) { + return ( +
+ {[ + ['Warned', `${metrics.events_warned}/${metrics.events}`], + ['Missed', metrics.events_missed], + ['False alarms/year', metrics.false_alarms_per_year?.toFixed(1) ?? '—'], + ['Median lead', metrics.median_lead_days == null ? '—' : `${metrics.median_lead_days}d`], + ].map(([label, value]) => ( +
+
{label}
+
{value}
+
+ ))} +
+ ); +} + +function EventTable({ events }: { events: { date: string; warned: boolean; lead_days: number | null }[] }) { + return ( +
+
Pillar / sensor Score Weight
{title} - + {reading.score ?? '—'} · {Math.round(reading.coverage)}% coverage
+ + + + + + {events.map((event) => ( + + + + + + ))} +
CorrectionWarnedLead
{event.date}{event.warned ? 'yes' : 'no'}{event.lead_days == null ? '—' : `${event.lead_days}d`}
+
+ ); +} + +/** Shipped rule against ablations, external baselines, and chance. + * + * The two kinds answer different questions and must not be read as one list: + * an ablation asks whether the quadrant machinery earns its place, a baseline + * asks whether the score earns its complexity. + */ +function ComparisonTable({ report }: { report: EventStudyReport }) { + const shipped = report.shipped; + if (!shipped || !report.comparison?.length) return null; + const rows = [ + { + id: 'shipped', + label: 'Quadrant alert (shipped)', + kind: 'shipped' as const, + note: shipped.rule.entry, + measurable: true, + ...shipped.metrics, + }, + ...report.comparison, + ]; + const KIND_LABEL: Record = { + shipped: 'shipped', + ablation: 'ablation', + baseline: 'baseline', + fundamental: 'fundamental', + }; + return ( +
+
+ + + + + + + + {rows.map((row) => ( + + + {/* A rule whose input does not exist yet scores 0/N, and printing + that would read as tested-and-failed. Say "not measurable". */} + {row.measurable === false ? ( + + ) : ( + <> + + + + + )} + + ))} + {report.null_model && ( + + + + + + + )} +
RuleWarnedFA/yrMedian lead
+ {row.label} + {KIND_LABEL[row.kind]} + + {/* Not "no observations yet": once some exist but fewer than + the minimum are covered, that is simply false. Matches the + callout below. */} + insufficient exposure — not measurable + {row.events_warned}/{row.events}{row.false_alarms_per_year?.toFixed(1) ?? '—'}{row.median_lead_days == null ? '—' : `${row.median_lead_days}d`}
+ Random alarms, same firing rate + null + + {report.null_model.mean_warned.toFixed(1)} ± {report.null_model.sd_warned.toFixed(1)} +
+
+
+ ); +} + +function StudyVerdict({ report }: { report: EventStudyReport }) { + const model = report.null_model; + if (!model) return null; + const chancePct = (model.p_at_least_observed * 100).toFixed(0); + const indistinguishable = model.p_at_least_observed >= 0.1; + // The number carries the claim, not the adjective. At ~10 corrections a p of + // 0.09 is not evidence of anything, so "beats the null" would over-state a + // result this panel is otherwise careful never to over-state. + return ( + + + {indistinguishable + ? `Not distinguishable from chance (p = ${model.p_at_least_observed.toFixed(2)}).` + : `Above the firing-rate null (p = ${model.p_at_least_observed.toFixed(2)}).`} + {' '} + Random alarms match or beat {model.observed_warned}/{model.events} warned corrections in {chancePct}% of{' '} + {model.draws} draws placing {model.alarms_per_draw} alarms over the same sessions. Corrections cluster and random + placement does not, so this is the floor, not the bar. + + ); +} + +/** The credit sensor starts partway through, so Warning is a different + * construct either side of it. The share is derived, never asserted: if one era + * carries no corrections there is no comparison to draw and the per-era ratios + * would be noise dressed up as a finding. */ +function EraDisclosure({ + eras, + divider, +}: { + eras: NonNullable['by_era']>; + divider: number | undefined; +}) { + const { pre_credit: pre, full_coverage: full } = eras; + const total = pre.sessions + full.sessions; + const share = total > 0 ? Math.round((pre.sessions / total) * 100) : 0; + // An era holding one or two corrections has a recall of 0/1 or 1/2, which is + // not a rate. Below this the eras get their false-alarm rates compared and + // nothing else. + const comparable = pre.events >= 3 && full.events >= 3; + return ( + +

+ {share}% of the evaluated sessions predate the credit sensor. W3 begins {eras.credit_from}, so + before that Warning renormalises to W1+W2 and the fixed {divider} divider is applied to a different construct + than it was reasoned about. Dropping the training split makes every correction evaluable; it does not make the + coverage gap go away, it moves it from the threshold to the score. + {comparable ? ( + <> + {' '}Two sensors:{' '} + {pre.events_warned}/{pre.events} at{' '} + {pre.false_alarms_per_year?.toFixed(1) ?? '—'} FA/yr. All three:{' '} + {full.events_warned}/{full.events} at{' '} + {full.false_alarms_per_year?.toFixed(1) ?? '—'} FA/yr. + + ) : ( + <> + {' '}The corrections do not straddle that boundary ({pre.events} before, {full.events} after), so the two + eras cannot be compared on recall — only the false-alarm rates are meaningful ({pre.false_alarms_per_year?.toFixed(1) ?? '—'}{' '} + vs {full.false_alarms_per_year?.toFixed(1) ?? '—'} per year). + + )} +

+
+ ); +} + function EventStudyBody({ report }: { report: EventStudyReport }) { - const metrics = report.metrics; + const shipped = report.shipped; + const eras = shipped?.by_era; return (
- {report.generated_at && generated {new Date(report.generated_at).toLocaleDateString()}} - {report.sample && test {report.sample.test_start} → {report.sample.end}} + {report.generated_at && generated {new Date(report.generated_at).toLocaleDateString()}} + {report.sample && {report.sample.evaluable_from} → {report.sample.end}}
+

{report.summary}

- {metrics && ( -
- {[ - ['Warned', `${metrics.events_warned}/${metrics.events}`], - ['Missed', metrics.events_missed], - ['False alarms/year', metrics.false_alarms_per_year.toFixed(1)], - ['Median lead', metrics.median_lead_days == null ? '—' : `${metrics.median_lead_days}d`], - ].map(([label, value]) => ( -
-
{label}
-
{value}
-
- ))} -
+ {shipped && } + + + {shipped && shipped.events.length > 0 && ( + + + )} - {report.events && report.events.length > 0 && ( -
- - - - - - - {report.events.map((event) => ( - - - - - - ))} -
CorrectionWarnedLead
{event.date}{event.warned ? 'yes' : 'no'}{event.lead_days == null ? '—' : `${event.lead_days}d`}
-
+ + {report.null_model && ( + +

+ The null places {report.null_model.alarms_per_draw} alarms at random over the same sessions and at the + shipped rule's firing rate. Corrections cluster while random placement does not, so this is a floor rather + than a demanding benchmark: a clustering rule could beat it without genuine foresight. +

+
)} - {report.reliability && (report.reliability.underpowered || report.reliability.sensor_coverage_mismatch) && ( - -
- {report.reliability.underpowered && ( -

- Underpowered. Only {report.reliability.events_in_holdout} of{' '} - {report.reliability.events_detected} detected corrections fall in the test period ( - {report.reliability.minimum_events}+ needed). Read the direction, not the ratio. -

- )} - {report.reliability.sensor_coverage_mismatch && ( -

- Sensor coverage differs across the split.{' '} - {report.reliability.train_full_sensor_share}% of training sessions had all{' '} - {report.reliability.sensors_expected} Warning sensors versus{' '} - {report.reliability.holdout_full_sensor_share}% of test sessions - {report.params?.credit_sensor_from && ` — credit history begins ${report.params.credit_sensor_from}`} - . The threshold was frozen on a partly different construct than it is measured against. -

+ + {report.fundamental_coverage && !report.fundamental_coverage.measurable && ( + +

+ Insufficient exposure — the fundamental rows are untested, not failed.{' '} + The channel had usable context on{' '} + + {report.fundamental_coverage.sessions_eligible} of{' '} + {report.fundamental_coverage.evaluable_sessions} + {' '} + evaluated sessions, covering{' '} + + {report.fundamental_coverage.events_covered} of{' '} + {report.fundamental_coverage.events_evaluable} + {' '} + corrections ({report.fundamental_coverage.minimum_events} needed;{' '} + {report.fundamental_coverage.observations} observation + {report.fundamental_coverage.observations === 1 ? '' : 's'} recorded). Those rows are + scored only on that window, never on the market rows' full sample — otherwise a + fortnight of data would render as a 0/10 and read as a failed test. Read the market rows + as a verdict on the technical sensors and the alert machinery only. +

+
+ )} + + {eras && } + + {report.fitted && ( + +
+

+ The original study, kept because it is what the methodology document reports: an{' '} + {report.fitted.params.warn_percentile}th-percentile Warning threshold ( + {report.fitted.params.warn_threshold}) frozen on the first{' '} + {(report.fitted.params.train_fraction * 100).toFixed(0)}% of sessions and measured on the rest. Nothing + consumes this rule — the shipped alert uses fixed dividers with hysteresis, confirmation and a cooldown. +

+ + {report.fitted.events.length > 0 && } + {report.reliability && (report.reliability.underpowered || report.reliability.sensor_coverage_mismatch) && ( + +
+ {report.reliability.underpowered && ( +

+ Underpowered. Only {report.reliability.events_in_holdout} of{' '} + {report.reliability.events_detected} detected corrections fall in the holdout ( + {report.reliability.minimum_events}+ needed). Read the direction, not the ratio. +

+ )} + {report.reliability.sensor_coverage_mismatch && ( +

+ Sensor coverage differs across the split.{' '} + {report.reliability.train_full_sensor_share}% of training sessions had all{' '} + {report.reliability.sensors_expected} Warning sensors versus{' '} + {report.reliability.holdout_full_sensor_share}% of test sessions + {report.params?.credit_sensor_from && ` — credit history begins ${report.params.credit_sensor_from}`} + . The threshold was frozen on a partly different construct than it is measured against. +

+ )} +
+
)}
- +
)}
); @@ -394,15 +684,26 @@ function FundamentalsEditor({ }) { const [capex, setCapex] = useState>(() => ({ ...data.capex })); const [reaction, setReaction] = useState(data.good_news_stock_down); - const knownCapex = Object.values(capex).filter((state) => state !== 'unknown'); - // Mirrors _CAPEX_STATE_SCORES: raising 0, holding 50, cutting 100. Holding is - // the deceleration case and used to score identically to raising. - const capexPoints = knownCapex.reduce((sum, state) => sum + (state === 'cutting' ? 100 : state === 'holding' ? 50 : 0), 0); - const derivedF1 = knownCapex.length >= 3 ? Math.round((capexPoints / knownCapex.length) * 10) / 10 : null; - const derivedF3 = reaction === 'yes' ? 100 : reaction === 'no' ? 0 : null; + const values = Object.values(capex); + const counts = { + cutting: values.filter((s) => s === 'cutting').length, + holding: values.filter((s) => s === 'holding').length, + raising: values.filter((s) => s === 'raising').length, + unknown: values.filter((s) => s === 'unknown').length, + }; + // Mirrors _capex_signal: any cut is adverse on partial evidence, any hold is + // neutral, all-known-raising is supportive, nothing known is unknown. No + // average — an average would let cuts and unknowns land on "neutral". + const capexSignal: FundamentalState = + counts.cutting > 0 ? 'adverse' + : counts.holding > 0 ? 'neutral' + : counts.raising > 0 ? 'supportive' + : 'unknown'; + const reactionSignal: FundamentalState = + reaction === 'yes' ? 'adverse' : reaction === 'no' ? 'supportive' : reaction === 'mixed' ? 'neutral' : 'unknown'; return (
-
+
Source: {data.source} {data.fetched_at && · fetched {new Date(data.fetched_at).toLocaleDateString()}} {data.effective_date && · effective {data.effective_date}} @@ -411,8 +712,8 @@ function FundamentalsEditor({ {data.reasoning &&

{data.reasoning}

}
- F1 · Capex guidance by hyperscaler - score {derivedF1 ?? 'n/a'} + Capex guidance by hyperscaler + {capexSignal}
{Object.entries(capex).map(([symbol, state]) => ( @@ -428,17 +729,24 @@ function FundamentalsEditor({ ))}
-

Raising = 0, holding = 50, cutting = 100; at least three known names required.

+

+ {counts.cutting} cutting · {counts.holding} holding · {counts.raising} raising · {counts.unknown} unknown. + Any cut reads adverse on partial evidence; supportive needs every known name raising. +

@@ -465,7 +773,7 @@ function ConfigEditor({ data, onSave, saving }: { data: RegimeConfig; onSave: (u setStaleness(Number(event.target.value))} className="w-20 rounded-md border border-white/[0.08] bg-white/[0.03] px-2 py-1 text-right num text-gray-200" /> days -

Changing the basket resets its freeze date and silently reseeds quadrant alerts.

+

Changing the basket resets its freeze date and silently reseeds quadrant alerts.

); @@ -483,13 +791,13 @@ function AdminControls() {
-
Fundamental observations
+
Fundamental observations
{fundamentals.isLoading && } {fundamentals.data && saveFundamentals.mutate(body)} onRefresh={() => refresh.mutate()} saving={saveFundamentals.isPending} refreshing={refresh.isPending} />} {refresh.isError && Refresh failed: {(refresh.error as Error).message}}
-
Fixed basket & freshness
+
Fixed basket & freshness
{config.isLoading && } {config.data && saveConfig.mutate(updates)} saving={saveConfig.isPending} />} {saveConfig.isError && Save failed: {(saveConfig.error as Error).message}} @@ -508,7 +816,19 @@ export default function RegimePage() {
+ + {data?.date && as of {data.date}} + {data?.available && ( + + )} +
+ } /> {monitor.isLoading && <>} @@ -524,7 +844,7 @@ export default function RegimePage() { )} -
+
+ Breadth divergence · SMH/SPY rollover · HY credit impulse. + {data.warning.coverage < 100 && ' Missing sensors are omitted rather than filled.'} + + } /> + {data.fundamental_live && }
+ + }> + {data.fundamental_live && } + - {data.fundamental_context && } - - + + + )} diff --git a/frontend/src/styles/globals.css b/frontend/src/styles/globals.css index 185a5c7..d48335d 100644 --- a/frontend/src/styles/globals.css +++ b/frontend/src/styles/globals.css @@ -42,6 +42,14 @@ appearance: textfield; } + /* --ink, not --up-text: a focus ring must not carry a semantic colour. The + directional token reads as "up/positive" and lands at poor contrast on the + controls that are already that colour. */ + :where(button, a, input, select, textarea, summary, [tabindex]):focus-visible { + outline: 2px solid var(--ink); + outline-offset: 3px; + } + /* Atmosphere: faint starfield + soft rim-cyan / ember glows + film grain */ #root { position: relative; @@ -83,6 +91,17 @@ } } +@media (prefers-reduced-motion: reduce) { + *, + *::before, + *::after { + animation-duration: 0.01ms !important; + animation-iteration-count: 1 !important; + scroll-behavior: auto !important; + transition-duration: 0.01ms !important; + } +} + @layer components { /* Mars horizon — fixed at the viewport bottom, atmosphere only, never data */ .app-horizon { diff --git a/scripts/run_regime_monitor_calibration.py b/scripts/run_regime_monitor_calibration.py index fb1ff8c..b47fce3 100644 --- a/scripts/run_regime_monitor_calibration.py +++ b/scripts/run_regime_monitor_calibration.py @@ -18,6 +18,15 @@ The script refuses to emit a band recommendation unless every hard gate passes. That is deliberate: it must be structurally impossible to read a calibration result out of a run whose pipeline did not validate. +**Fundamental channel note.** The sourced capex / earnings read is a separate +categorical channel and is never a term in State or Warning, so every variant and +gate below is unaffected by it. This harness passes no observation, which means +the ``fundamental_context`` on each replayed row reads ``unknown`` -- correct, and +the same thing production reports for a session nobody observed. Calibrating +anything *about* that channel needs an observation series passed through +``_compute_index(..., observations=...)``, and enough history to be worth +calibrating against. + Research branch only. Example: .\\.venv\\Scripts\\python.exe scripts\\run_regime_monitor_calibration.py ^ diff --git a/tests/unit/test_event_study.py b/tests/unit/test_event_study.py index 1db857d..0393923 100644 --- a/tests/unit/test_event_study.py +++ b/tests/unit/test_event_study.py @@ -1,17 +1,27 @@ -"""Tests for v3 correction events, warning alarm episodes, and report caveats.""" +"""Tests for correction events, alarm episodes, the shipped-rule replay, and caveats.""" from __future__ import annotations +from copy import deepcopy from datetime import date, timedelta +import pytest + from app.services.breadth_service import _breadth_from_closes, compute_divergence_series from app.services.event_study_service import ( MIN_EVENTS_FOR_CONFIDENCE, + STRESS_QUADRANT, + WARNING_QUADRANTS, + _era_split, + _null_model, _percentile, _reliability, alarm_episodes, + below_average_series, detect_events, + entry_alarms, evaluate_alarms, + replay_quadrant_changes, ) @@ -19,6 +29,33 @@ def _days(count: int, start: date = date(2021, 1, 1)) -> list[date]: return [start + timedelta(days=index) for index in range(count)] +def _row( + warning: float, + state: float = 0.0, + *, + warning_coverage: float = 100.0, + state_coverage: float = 100.0, + fresh: bool = True, +) -> dict: + return { + "state": state, + "warning": warning, + "state_coverage": state_coverage, + "warning_coverage": warning_coverage, + "inputs_fresh": fresh, + } + + +def _rows( + dates: list[date], warnings: list[float], patch: dict[int, dict] | None = None +) -> dict[date, dict]: + """One publishable row per date, with per-position replacements.""" + built = {day: _row(value) for day, value in zip(dates, warnings)} + for index, replacement in (patch or {}).items(): + built[dates[index]] = replacement + return built + + def test_detect_events_uses_rising_edge_and_cooldown(): closes = [100.0] * 300 + [85.0] * 5 + [100.0] * 50 + [85.0] * 5 events = detect_events(closes, _days(len(closes)), threshold_pct=15.0, cooldown=40) @@ -88,6 +125,366 @@ def test_evaluate_alarms_counts_episodes_not_alarm_days(): assert result["median_lead_days"] == 17.5 +# --------------------------------------------------------------------------- +# The shipped quadrant rule, replayed +# --------------------------------------------------------------------------- + +def test_replay_seeds_silently_and_needs_two_sessions(): + """A one-session spike is not an alert; the second session confirms it. + + The alarm is therefore dated at the confirmation rather than at the first + crossing, which costs one session of lead. That is what ships. + """ + dates = _days(10) + spike = _rows(dates, [30] * 5 + [70] + [30] * 4) + assert replay_quadrant_changes(spike, dates) == [] + + held = _rows(dates, [30] * 5 + [70, 70] + [30] * 3) + fires = replay_quadrant_changes(held, dates) + # The rule alerts on quadrant changes in both directions, so the return to + # calm fires too. Only the entry is a warning about anything. + assert [(f["index"], f["from"], f["to"]) for f in fires] == [ + (6, "3", "1"), + (9, "1", "3"), + ] + assert entry_alarms(fires, WARNING_QUADRANTS) == [6] + + +def test_confirmation_classifies_the_prior_session_against_the_baseline(): + """Not against its own predecessor -- the distinction changes the answer. + + Warning 42 sits inside the hysteresis deadband. Measured from the standing + "3" baseline it is still "3", so it cannot confirm a move to "1". A chain + that classified each session against the one before it would read 42 as "1" + (having just seen 70) and fire a day later, which production does not do. + """ + dates = _days(10) + rows = _rows(dates, [30, 30, 30, 30, 70, 42, 70, 30, 30, 30]) + assert replay_quadrant_changes(rows, dates) == [] + + +def test_cooldown_suppresses_and_the_baseline_only_advances_on_a_fire(): + dates = _days(10) + rows = _rows(dates, [30, 30, 30, 30, 70, 70, 30, 30, 30, 30]) + fires = replay_quadrant_changes(rows, dates) + + # Entry confirmed on day 5. The exit confirms on day 7 but lands inside the + # 3-day cooldown, so it is re-evaluated and fires on day 8 instead. + assert [(f["index"], f["from"], f["to"]) for f in fires] == [ + (5, "3", "1"), + (8, "1", "3"), + ] + assert entry_alarms(fires, WARNING_QUADRANTS) == [5] + + +def test_low_coverage_sessions_cannot_confirm(): + """The confirmation source has to be a session that published a band.""" + dates = _days(10) + warnings = [30, 30, 30, 30, 30, 70, 70, 30, 30, 30] + visible = replay_quadrant_changes(_rows(dates, warnings), dates) + assert entry_alarms(visible, WARNING_QUADRANTS) == [6] + + # Day 5 is the only session that could confirm the entry on day 6; below + # MIN_COVERAGE it never published a band, so day 4 is the prior instead. + hidden = _rows(dates, warnings, {5: _row(70, warning_coverage=70.0)}) + assert replay_quadrant_changes(hidden, dates) == [] + + +def test_stale_inputs_block_todays_alert_but_not_tomorrows_confirmation(): + """is_fresh gates the live reading only; the prior session comes from history.""" + dates = _days(10) + rows = _rows(dates, [30] * 4 + [70, 70, 70] + [30] * 3, {5: _row(70, fresh=False)}) + fires = replay_quadrant_changes(rows, dates) + assert entry_alarms(fires, WARNING_QUADRANTS) == [6] + + +def test_entry_alarms_ignore_movement_inside_the_set(): + fires = [ + {"index": 3, "from": "3", "to": "1"}, + {"index": 9, "from": "1", "to": "2"}, + {"index": 20, "from": "2", "to": "4"}, + ] + assert entry_alarms(fires, WARNING_QUADRANTS) == [3] + assert entry_alarms(fires, STRESS_QUADRANT) == [9] + + +def test_below_average_series_needs_a_full_window(): + series = list(zip(_days(6), [10.0, 10.0, 10.0, 10.0, 4.0, 20.0])) + indicator = below_average_series(series, window=3) + assert _days(6)[1] not in indicator # warm-up + assert indicator[_days(6)[4]] == 100.0 # 4 is under the 3-day mean of 8 + assert indicator[_days(6)[5]] == 0.0 + + +def test_null_model_is_seeded_and_drawn_from_evaluable_sessions_only(): + dates = _days(300) + events = [100, 180, 260] + first = _null_model(6, events, dates, horizon=20, start_index=50, observed_warned=2, draws=200) + second = _null_model(6, events, dates, horizon=20, start_index=50, observed_warned=2, draws=200) + assert first == second # a re-run must not move the report + assert 0.0 <= first["p_at_least_observed"] <= 1.0 + assert first["alarms_per_draw"] == 6 + assert first["mean_warned"] <= len(events) + + # More alarms than there are sessions to place them on is not a null. + assert _null_model(500, events, dates, 20, 50, 2, draws=10) is None + assert _null_model(6, [], dates, 20, 50, 0, draws=10) is None + + +def test_era_split_reports_the_two_sensor_eras_separately(): + """The fuller sample is mostly pre-credit, where Warning is W1+W2 only.""" + dates = _days(400) + eras = _era_split( + alarms=[80, 300], + event_indices=[90, 310], + dates=dates, + horizon=20, + start_index=10, + credit_from=dates[200], + ) + assert eras["pre_credit"]["events"] == 1 + assert eras["pre_credit"]["events_warned"] == 1 + assert eras["full_coverage"]["events"] == 1 + assert eras["full_coverage"]["events_warned"] == 1 + assert eras["credit_from"] == dates[200].isoformat() + + # No credit series at all means there is no boundary to split on. + assert _era_split([80], [90], dates, 20, 10, None) is None + + +def _business_days(count: int, end: date = date(2026, 8, 7)) -> list[date]: + out: list[date] = [] + cursor = end + while len(out) < count: + if cursor.weekday() < 5: + out.append(cursor) + cursor -= timedelta(days=1) + return list(reversed(out)) + + +def _synthetic_path(sessions: int) -> list[float]: + """A rising leader with two deep drawdowns, so corrections exist to detect.""" + closes: list[float] = [] + for index in range(sessions): + if index < 350: + closes.append(100.0 + index * 0.25) + elif index < 400: + closes.append(187.5 - (index - 350) * 0.9) + elif index < 650: + closes.append(142.5 + (index - 400) * 0.4) + elif index < 700: + closes.append(242.5 - (index - 650) * 1.1) + else: + closes.append(187.5 + (index - 700) * 0.3) + return closes + + +async def test_report_assembles_every_rule_from_synthetic_inputs(monkeypatch): + """End-to-end: the shipped replay, ablations, baselines and null all score. + + Synthetic rather than recorded because the point is the wiring -- that every + rule is measured on the same events over the same sessions and the report + carries what the panel reads. The numbers are meaningless by construction. + """ + import app.services.event_study_service as ess + + sessions = 900 + dates = _business_days(sessions) + closes = _synthetic_path(sessions) + leader = list(zip(dates, closes)) + # SPY grinds up throughout, so the leader's relative strength rolls over + # exactly when it falls. + market = list(zip(dates, [100.0 + index * 0.12 for index in range(sessions)])) + # Breadth deteriorates ~15 sessions ahead of each decline, which is the + # divergence W1 exists to catch. + breadth = {} + for index, day in enumerate(dates): + weak = 335 <= index < 400 or 635 <= index < 700 + breadth[day] = 30.0 if weak else 70.0 + vix = [(day, 32.0 if (350 <= i < 400 or 650 <= i < 700) else 15.0) for i, day in enumerate(dates)] + # Credit starts late, exactly as ICE's 3-year cap makes it in production. + oas = [(day, 4.2 if (650 <= i < 700) else 3.0) for i, day in enumerate(dates) if i >= 500] + + async def fake_config(_db): + return deepcopy(ess.rms.DEFAULT_CONFIG) + + async def fake_prices(_config, _start, _end): + return {"SMH": leader, "QQQ": leader, "SPY": market} + + async def fake_fred(series_id, _start, _end): + return {"VIXCLS": vix, "BAMLH0A0HYM2": oas}.get(series_id) + + async def fake_breadth(_db, _symbols, window=200, min_tickers=20): + return breadth, {day: 30 for day in dates} + + async def fake_observations(_db): + return [] + + monkeypatch.setattr(ess.rms, "get_regime_config", fake_config) + monkeypatch.setattr(ess.rms, "_fetch_prices", fake_prices) + monkeypatch.setattr(ess.rms, "_fetch_fred_series", fake_fred) + monkeypatch.setattr(ess.rms, "get_fundamental_observations", fake_observations) + monkeypatch.setattr(ess.breadth_service, "compute_breadth_details", fake_breadth) + monkeypatch.setattr(ess, "NULL_DRAWS", 100) + + report = await ess.run_event_study(None) + + assert report["available"] is True + assert report["schema"] == ess.STUDY_SCHEMA + + # The shipped rule is measured on the whole sample, not a 30% holdout. + shipped = report["shipped"] + assert shipped["metrics"]["events"] == report["sample"]["events_evaluable"] + assert report["sample"]["events_evaluable"] >= 2 + assert shipped["metrics"]["events"] >= report["fitted"]["metrics"]["events"] + assert len(shipped["events"]) == shipped["metrics"]["events"] + + assert {row["kind"] for row in report["comparison"]} == { + "ablation", "baseline", "fundamental", + } + # Market rows share the headline's events, or the table lies. Fundamental + # rows deliberately do not: they are coverage-matched to the sessions the + # channel actually existed on, which is a different (here empty) window. + for row in report["comparison"]: + if row["kind"] != "fundamental": + assert row["events"] == shipped["metrics"]["events"] + assert row["false_alarms_per_year"] >= 0 + else: + # No eligible sessions means the rate is undefined, not zero. A + # tiny-divisor fallback here printed 5e9 alarms/year. + assert row["false_alarms_per_year"] is None + + # The credit sensor starts mid-sample, so the era split must be populated. + eras = shipped["by_era"] + assert eras["credit_from"] == dates[500].isoformat() + assert eras["pre_credit"]["events"] + eras["full_coverage"]["events"] == shipped["metrics"]["events"] + + if report["null_model"] is not None: + assert 0.0 <= report["null_model"]["p_at_least_observed"] <= 1.0 + assert report["null_model"]["observed_warned"] == shipped["metrics"]["events_warned"] + + # With an empty observation series the fundamental rows are *untested*, not + # failed, and the report has to carry that distinction or a 0/10 in the table + # reads as a measured result. + coverage = report["fundamental_coverage"] + assert coverage["observations"] == 0 + assert coverage["sessions_eligible"] == 0 + assert coverage["events_covered"] == 0 + assert coverage["measurable"] is False + fundamental_rows = [r for r in report["comparison"] if r["kind"] == "fundamental"] + assert {r["id"] for r in fundamental_rows} == { + "fundamental_adverse", "confluence", "market_over_covered", + } + assert all(row["measurable"] is False for row in fundamental_rows) + # Coverage-matched denominators: with no exposure these rows must not claim + # to have been scored against the market rows' 10 corrections. + assert all(row["events"] == 0 for row in fundamental_rows) + # Market rows are unaffected: their inputs exist for the whole window. + assert all( + row["measurable"] is True + for row in report["comparison"] + if row["kind"] != "fundamental" + ) + + +def test_fundamental_rows_are_scored_only_on_their_own_exposure(): + """One day of coverage must not render as 0/10. + + A fundamental rule scores zero whether it is wrong or merely absent, so + scoring it against corrections it could never have seen manufactures a + failed result out of a thin one — the same mistake the `measurable` flag + prevents for an empty table, arriving one observation later. + """ + import app.services.event_study_service as ess + + dates = _days(300) + events = [50, 120, 200, 280] + # Context exists for a single stretch, covering only the 120 event's horizon. + rows = { + day: { + "fundamental_state": "adverse", + "fundamental_usable": 105 <= index <= 115, + } + for index, day in enumerate(dates) + } + + covered = ess.covered_events(events, rows, dates, horizon=20) + assert covered == [120] + assert ess.eligible_sessions(rows, dates, start_index=0) == 11 + + # A stale stretch counts for nothing, however adverse it reads. + stale = { + day: {"fundamental_state": "adverse", "fundamental_usable": False} + for day in dates + } + assert ess.covered_events(events, stale, dates, horizon=20) == [] + assert ess.eligible_sessions(stale, dates, start_index=0) == 0 + assert ess.adverse_episodes(stale, dates, 0) == [] + assert ess.confluence_episodes([120], stale, dates) == [] + + # And neither does a *fresh* observation that determined nothing. Repeated + # extraction failures would otherwise accumulate exposure until the rows + # flipped to a measurable 0/8 for a channel that never knew anything — + # the same tested-versus-unavailable confusion, arriving by a slower route. + empty = { + day: {"fundamental_state": "unknown", "fundamental_usable": False} + for day in dates + } + assert ess.covered_events(events, empty, dates, horizon=20) == [] + assert ess.eligible_sessions(empty, dates, start_index=0) == 0 + + +async def test_the_fundamental_channel_never_moves_the_warning_score(): + """The channel is compared, never fused. Warning must be identical either way. + + A weighted modifier was built and reverted: with ~10 correction events and + almost no fundamental history any fusion weight is a policy preference + presented as a measurement. + """ + import app.services.event_study_service as ess + + end = date(2026, 6, 26) + dates = _business_days(400, end) + rising = [(day, 100.0 + index * 0.2) for index, day in enumerate(dates)] + prices = {"SMH": rising, "QQQ": rising, "SPY": rising} + args = (prices, [(end, 20.0)], [(day, 4.0) for day in dates]) + config = deepcopy(ess.rms.DEFAULT_CONFIG) + names = config["tickers"]["hyperscalers"] + tail = (rising, [(day, 20.0) for day in dates], dates, config) + + def adverse(effective: date) -> list[dict]: + return [{ + "effective_date": effective, + "f1_score": 100.0, + "f3_score": 100.0, + "capex": dict.fromkeys(names, "cutting"), + "good_news_stock_down": "yes", + "fetched_at": "2026-01-01T00:00:00+00:00", + }] + + bare = ess._axis_rows(*args, *tail, None) + observed = ess._axis_rows(*args, *tail, adverse(dates[-20])) + + latest, early = dates[-1], dates[-90] + assert observed[latest]["warning"] == bare[latest]["warning"] + assert observed[latest]["fundamental_state"] == "adverse" + assert bare[latest]["fundamental_state"] == "unknown" + + # Sessions before the effective date stay unknown, so a rebuild cannot stamp + # today's reading onto history. + assert observed[early]["fundamental_state"] == "unknown" + + # The confluence rule keeps only crossings the channel agrees with, and the + # fundamental rule fires on the transition into adverse -- both rising-edge, + # so both stay comparable with the market rows. + adverse_alarms = ess.adverse_episodes(observed, dates, 0) + assert [dates[i] for i in adverse_alarms] == [dates[-20]] + assert ess.adverse_episodes(bare, dates, 0) == [] + assert ess.confluence_episodes([dates.index(early), dates.index(latest)], observed, dates) == [ + dates.index(latest) + ] + + def test_breadth_from_fixed_closes_and_tapered_divergence(): dates = _days(10) closes_by_symbol = { diff --git a/tests/unit/test_regime_monitor.py b/tests/unit/test_regime_monitor.py index 7dc3f8f..683162f 100644 --- a/tests/unit/test_regime_monitor.py +++ b/tests/unit/test_regime_monitor.py @@ -28,7 +28,7 @@ from app.services.regime_monitor_service import ( drawdown_pct, f2_credit_spreads, current_observation, - fundamental_overlay, + fundamental_context, p1_trend_break, p2_death_cross, p3_drawdown, @@ -40,6 +40,24 @@ from app.services.regime_monitor_service import ( ) +async def _no_observations(_db): + return [] + + +async def _skip_recording(_db, _observation): + return None + + +class _CommitOnlyDB: + """Enough session for writers that own their own transaction boundary.""" + + def __init__(self) -> None: + self.commits = 0 + + async def commit(self) -> None: + self.commits += 1 + + def _dated(values: list[float], end: date = date(2026, 6, 26)) -> list[tuple[date, float]]: return [ (end - timedelta(days=len(values) - 1 - index), value) @@ -215,7 +233,7 @@ def test_score_pillars_gates_band_below_75_percent_coverage(): assert result["band"] is None -def test_fundamental_overlay_never_replays_before_effective_date_and_expires(): +def test_fundamental_context_never_replays_before_effective_date_and_expires(): overrides = { "f1_score": 0.0, "f3_score": 100.0, @@ -226,19 +244,19 @@ def test_fundamental_overlay_never_replays_before_effective_date_and_expires(): } config = {**DEFAULT_CONFIG, "fundamental_staleness_days": 80} - pending = fundamental_overlay(overrides, config, date(2026, 6, 1)) + pending = fundamental_context(overrides, config, date(2026, 6, 1)) assert pending["pending"] is True assert pending["available"] is False assert pending["capex"] is None # The effective date is still reported so a pending refresh is visible. assert pending["effective_date"] == "2026-06-02" - live = fundamental_overlay(overrides, config, date(2026, 6, 2)) + live = fundamental_context(overrides, config, date(2026, 6, 2)) assert live["available"] is True assert live["good_news_stock_down"] == "yes" assert live["earnings_stress"] == 100.0 - expired = fundamental_overlay(overrides, config, date(2026, 8, 22)) + expired = fundamental_context(overrides, config, date(2026, 8, 22)) assert expired["stale"] is True assert expired["available"] is False @@ -263,7 +281,7 @@ def test_live_observation_is_visible_before_its_effective_date(): config = {**DEFAULT_CONFIG, "fundamental_staleness_days": 80} before = date(2026, 6, 1) - record = fundamental_overlay(overrides, config, before) + record = fundamental_context(overrides, config, before) now = current_observation(overrides, config, before) # Same day, same observation: the record hides it, the live reading shows it. @@ -314,35 +332,256 @@ def test_an_uncollected_observation_is_not_reported_as_collected(): assert current_observation(collected, DEFAULT_CONFIG, date(2026, 8, 7))["observed"] is True -def test_fundamentals_do_not_move_the_warning_score(): - """The v3 complaint: a maxed-out LLM read must not silently do nothing. +def test_fundamental_state_never_averages_unknown_into_neutral(): + """Missing evidence must not present as evidence of normality. - It no longer feeds Warning at all, so Warning is identical either way and - the observation is reported beside the score instead of buried in it. + This is the trap that mattered when the channel replaced the weighted + modifier: treating ``unknown`` as a middle value would let two ``cutting`` + reads and two ``unknown`` ones land on "neutral". A single adverse read + carries on partial evidence; ``unknown`` survives only when *nothing* was + observed. + """ + names = DEFAULT_CONFIG["tickers"]["hyperscalers"] + + assert rms._capex_signal(dict.fromkeys(names, "unknown"), names) == "unknown" + assert rms._capex_signal(dict.fromkeys(names, "raising"), names) == "supportive" + assert rms._capex_signal(dict.fromkeys(names, "holding"), names) == "neutral" + + half_cut = {names[0]: "cutting", names[1]: "cutting", **dict.fromkeys(names[2:], "unknown")} + assert rms._capex_signal(half_cut, names) == "adverse" + + assert rms._reaction_signal("yes") == "adverse" + assert rms._reaction_signal("no") == "supportive" + assert rms._reaction_signal("mixed") == "neutral" + assert rms._reaction_signal(None) == "unknown" + + combine = rms.combine_fundamental_signals + assert combine("unknown", "unknown") == "unknown" + assert combine("adverse", "supportive") == "adverse" # one adverse read carries + assert combine("supportive", "unknown") == "supportive" + assert combine("neutral", "unknown") == "neutral" + assert combine("supportive", "neutral") == "neutral" + # Nothing combines *into* unknown -- that would be inventing missing evidence. + assert "unknown" not in { + combine(a, b) + for a in rms.FUNDAMENTAL_STATES + for b in rms.FUNDAMENTAL_STATES + if not (a == "unknown" and b == "unknown") + } + + +def test_fundamental_context_is_a_channel_not_a_term_in_warning(): + """The read is reported beside the scores and never added into them. + + A weighted modifier was built and reverted: with ~10 correction events and + almost no fundamental history, any fusion weight is a policy preference + presented as a measurement, and adding a slow categorical judgement to a fast + continuous score manufactures precision by summing unlike things. """ end = date(2026, 6, 26) rising = [100.0 + index * 0.2 for index in range(700)] prices = {"SMH": _dated(rising, end), "QQQ": _dated(rising, end), "SPY": _dated(rising, end)} args = (prices, [(end, 20.0)], [(end - timedelta(days=i), 4.0) for i in reversed(range(100))]) tail = (copy.deepcopy(DEFAULT_CONFIG), end, [(end, 55.0)], [(end, 20.0)], {end: 25}) + names = DEFAULT_CONFIG["tickers"]["hyperscalers"] - quiet = _compute_index(*args, {"f1_score": None, "f3_score": None}, *tail) - screaming = _compute_index( - *args, - { - "f1_score": 100.0, - "f3_score": 100.0, - "capex": dict.fromkeys(DEFAULT_CONFIG["tickers"]["hyperscalers"], "cutting"), - "good_news_stock_down": "yes", + def observed(capex_state: str, reaction: str) -> dict: + return { + "capex": dict.fromkeys(names, capex_state), + "good_news_stock_down": reaction, "effective_date": "2026-06-01", - }, - *tail, - ) + "fetched_at": "2026-06-01T00:00:00+00:00", + "source": "openai", + } - assert quiet["warning"]["score"] == screaming["warning"]["score"] - assert {p["id"] for p in quiet["warning"]["pillars"]} == set(WARNING_WEIGHTS) - assert screaming["fundamental_overlay"]["available"] is True - assert screaming["fundamental_overlay"]["capex_stress"] == 100.0 + unobserved = _compute_index(*args, {"f1_score": None, "f3_score": None}, *tail) + supportive = _compute_index(*args, observed("raising", "no"), *tail) + adverse = _compute_index(*args, observed("cutting", "yes"), *tail) + + # Every Warning is identical: the channel is not a term in the score. + scores = { + snapshot["warning"]["score"] + for snapshot in (unobserved, supportive, adverse) + } + assert len(scores) == 1 + assert {p["id"] for p in unobserved["warning"]["pillars"]} == set(WARNING_WEIGHTS) + # And it never touches coverage, so a missing observation cannot suppress a + # band or silently redistribute weight onto the technical sensors. + assert len({s["warning"]["coverage"] for s in (unobserved, supportive, adverse)}) == 1 + + assert unobserved["fundamental_context"]["state"] == "unknown" + assert unobserved["fundamental_context"]["evidence_quality"] == "unavailable" + assert supportive["fundamental_context"]["state"] == "supportive" + assert adverse["fundamental_context"]["state"] == "adverse" + assert adverse["fundamental_context"]["evidence_quality"] == "complete" + + +def test_a_fresh_but_empty_observation_is_available_to_show_and_not_usable(): + """Collected-but-determined-nothing must not count as evidence. + + `available` is about timing (there is an effective, non-stale record to + display); `usable` is about content. An LLM run that failed to extract + anything produces a perfectly fresh observation that knows nothing — and if + that counted, repeated extraction failures would slowly accumulate study + exposure until the fundamental rows reported a measurable 0/8 for a channel + that had never seen a thing. + """ + config = copy.deepcopy(DEFAULT_CONFIG) + names = config["tickers"]["hyperscalers"] + as_of = date(2026, 6, 26) + base = { + "effective_date": "2026-06-01", + "fetched_at": "2026-06-01T00:00:00+00:00", + "source": "openai", + } + + empty = fundamental_context( + {**base, "capex": dict.fromkeys(names, "unknown"), "good_news_stock_down": "unknown"}, + config, as_of, + ) + assert empty["state"] == "unknown" + assert empty["available"] is True # there is a record, and it has a date + assert empty["usable"] is False # but it says nothing + + # One real signal is enough to be usable, on partial evidence. + partial = fundamental_context( + { + **base, + "capex": {names[0]: "cutting", **dict.fromkeys(names[1:], "unknown")}, + "good_news_stock_down": "unknown", + }, + config, as_of, + ) + assert partial["state"] == "adverse" + assert partial["usable"] is True + assert partial["evidence_quality"] == "partial" + + # Stale is neither available nor usable — `available` means effective *and* + # non-stale. What survives is `state`, which the card renders on its own + # (with the stale badge) so the last thing observed stays visible. + stale = fundamental_context( + { + **base, + "effective_date": "2026-01-01", + "capex": dict.fromkeys(names, "cutting"), + "good_news_stock_down": "yes", + }, + config, as_of, + ) + assert stale["state"] == "adverse" + assert stale["stale"] is True + assert stale["available"] is False + assert stale["usable"] is False + + # Nothing collected at all: neither. + absent = fundamental_context({}, config, as_of) + assert (absent["available"], absent["usable"]) == (False, False) + + +def test_the_live_reading_publishes_the_same_fields_as_the_record(): + """"Same shape" has to mean the same fields, not the same ones it needs. + + The frontend types both payloads as one interface, so a field present on the + record and missing from the live reading is an undefined at runtime that + TypeScript cannot catch across a trusted server boundary. + """ + config = copy.deepcopy(DEFAULT_CONFIG) + names = config["tickers"]["hyperscalers"] + as_of = date(2026, 6, 26) + observation = { + "effective_date": "2026-06-01", + "fetched_at": "2026-06-01T00:00:00+00:00", + "source": "openai", + "capex": dict.fromkeys(names, "cutting"), + "good_news_stock_down": "yes", + } + + record = fundamental_context(observation, config, as_of) + live = current_observation(observation, config, as_of) + assert set(record) <= set(live) + assert (live["state"], live["usable"]) == ("adverse", True) + + # A just-collected observation is shown but is not yet in force, so it is + # available to read and not yet usable as evidence. + pending = current_observation( + {**observation, "effective_date": "2026-07-01"}, config, as_of + ) + assert (pending["pending"], pending["available"], pending["usable"]) == (True, True, False) + + # And an extraction that determined nothing is never usable, however fresh. + empty = current_observation( + {**observation, "capex": dict.fromkeys(names, "unknown"), "good_news_stock_down": "unknown"}, + config, as_of, + ) + assert (empty["state"], empty["usable"]) == ("unknown", False) + + +def test_pre_rename_snapshots_keep_their_recorded_fundamental_evidence(): + """The rename shipped without a methodology bump, so those rows were never reseeded. + + Reading only the new key would turn real observations into `unknown` and + silently drop historical Path colours and legitimate study exposure. + """ + names = DEFAULT_CONFIG["tickers"]["hyperscalers"] + legacy = { + "methodology": rms.METHODOLOGY, + "date": "2026-07-01", + "state": {"score": 10.0, "band": "stable"}, + "warning": {"score": 20.0, "band": "stable"}, + "fundamental_overlay": { + "available": True, + "pending": False, + "stale": False, + "effective_date": "2026-06-20", + "capex": {names[0]: "cutting", **dict.fromkeys(names[1:], "raising")}, + "good_news_stock_down": "yes", + "source": "openai", + "fetched_at": "2026-06-19T00:00:00+00:00", + }, + } + + parsed = rms._parse_snapshot(json.dumps(legacy)) + context = parsed["fundamental_context"] + assert context["state"] == "adverse" + assert context["evidence_quality"] == "complete" + assert context["usable"] is True + assert context["effective_date"] == "2026-06-20" + + # A pending legacy overlay carried no facts, so it stays unknown rather than + # inventing an observation for a session nobody had looked at. + blank = json.loads(json.dumps(legacy)) + blank["fundamental_overlay"] = {"pending": True, "stale": False, "capex": None} + blank_context = rms._parse_snapshot(json.dumps(blank))["fundamental_context"] + assert blank_context["state"] == "unknown" + assert blank_context["evidence_quality"] == "unavailable" + assert blank_context["usable"] is False + + # A row already carrying the new key is left exactly as written. + modern = json.loads(json.dumps(legacy)) + modern["fundamental_context"] = {"state": "supportive", "usable": True} + assert rms._parse_snapshot(json.dumps(modern))["fundamental_context"]["state"] == "supportive" + + +def test_evidence_quality_ranks_what_an_operator_needs_first(): + names = DEFAULT_CONFIG["tickers"]["hyperscalers"] + config = copy.deepcopy(DEFAULT_CONFIG) + full = dict.fromkeys(names, "raising") + partial = {names[0]: "raising", **dict.fromkeys(names[1:], "unknown")} + + def quality(capex, reaction, *, observed=True, stale=False, source="openai"): + return rms._evidence_quality( + capex, reaction, names, observed=observed, stale=stale, source=source + ) + + assert quality(full, "no") == "complete" + assert quality(partial, "no") == "partial" + assert quality(full, None) == "partial" # reaction unknown + assert quality(full, "no", source="manual") == "manual" + assert quality(full, "no", stale=True) == "stale" + # Nothing collected outranks every other grade. + assert quality(full, "no", observed=False, stale=True, source="manual") == "unavailable" + assert set(rms.EVIDENCE_QUALITY) >= {quality(full, "no"), quality(partial, "no")} + assert config["tickers"]["hyperscalers"] == names def test_capex_score_separates_holding_from_raising(): @@ -375,10 +614,12 @@ async def test_legacy_numeric_fundamentals_do_not_leak_into_v4(monkeypatch): result = await rms.get_fundamental_overrides(object()) - assert result["methodology"] == "v4" + assert result["methodology"] == rms.METHODOLOGY assert result["f1_score"] is None assert result["f3_score"] is None - assert result["good_news_stock_down"] == "mixed" + # Not "mixed": an unreadable blob is an absence of an observation, and + # "mixed" is a genuinely observed mixed reaction. + assert result["good_news_stock_down"] == "unknown" @pytest.mark.asyncio @@ -442,11 +683,12 @@ async def test_unlock_does_not_redate_a_fundamental_observation(monkeypatch): async def fake_update(_db, _key, value): saved.update(json.loads(value)) + return None monkeypatch.setattr(rms, "get_fundamental_overrides", fake_get) - monkeypatch.setattr(rms, "update_setting", fake_update) + monkeypatch.setattr(rms.settings_store, "upsert_setting", fake_update) - result = await rms.set_fundamental_overrides(object(), locked=False) + result = await rms.set_fundamental_overrides(_CommitOnlyDB(), locked=False) assert result["locked"] is False assert result["fetched_at"] == stored["fetched_at"] @@ -476,14 +718,22 @@ async def test_manual_fundamentals_are_categorical_and_derived(monkeypatch): async def fake_update(_db, _key, value): saved.update(json.loads(value)) + return None monkeypatch.setattr(rms, "get_fundamental_overrides", fake_get) - monkeypatch.setattr(rms, "update_setting", fake_update) + monkeypatch.setattr(rms.settings_store, "upsert_setting", fake_update) + # A manual save now also appends to the point-in-time series. + monkeypatch.setattr(rms, "record_fundamental_observation", _skip_recording) capex = {names[0]: "cutting", **dict.fromkeys(names[1:], "holding")} + db = _CommitOnlyDB() result = await rms.set_fundamental_overrides( - object(), capex=capex, good_news_stock_down="mixed" + db, capex=capex, good_news_stock_down="mixed" ) + # The series row is a second write after update_setting's own commit, so the + # writer has to take one -- record_fundamental_observation deliberately does + # not, or it would steal update_regime_monitor's transaction boundary. + assert db.commits == 1 assert result["f1_score"] == 62.5 # one cutting (100) + three holding (50) assert result["f3_score"] is None @@ -498,7 +748,9 @@ async def test_manual_fundamentals_are_categorical_and_derived(monkeypatch): async def test_prior_snapshot_is_immutable_without_explicit_rebuild(db_session): snapshot_date = date(2026, 6, 26) first = { - "methodology": "v4", + # Must be the *current* methodology: a foreign row does not parse, so it + # reads as absent and the rewrite guard never comes into play. + "methodology": rms.METHODOLOGY, "date": snapshot_date.isoformat(), "state": {"score": 10.0, "band": "stable"}, "warning": {"score": 20.0, "band": "stable"}, @@ -571,6 +823,8 @@ async def test_routine_can_refresh_latest_trading_session_after_civil_day_rolls( monkeypatch.setattr(rms.breadth_service, "compute_breadth_details", fake_breadth) monkeypatch.setattr(rms, "_latest_snapshot_row", fake_latest) monkeypatch.setattr(rms, "_upsert_snapshot", fake_upsert) + monkeypatch.setattr(rms, "get_fundamental_observations", _no_observations) + monkeypatch.setattr(rms, "record_fundamental_observation", _skip_recording) result = await rms.update_regime_monitor(FakeDB()) @@ -636,6 +890,8 @@ async def test_a_stale_sensor_revision_reseeds_stored_history( ("_fetch_fred_series", fake_fred), ("_latest_snapshot_row", fake_latest), ("_upsert_snapshot", fake_upsert), + ("get_fundamental_observations", _no_observations), + ("record_fundamental_observation", _skip_recording), ): monkeypatch.setattr(rms, name, value) monkeypatch.setattr(rms.breadth_service, "compute_breadth_details", fake_breadth) @@ -722,7 +978,7 @@ def test_compute_index_uses_one_max_price_vote_and_has_no_combined_score(): price = next(p for p in result["state"]["pillars"] if p["id"] == "price") sensor_scores = [sensor["score"] for sensor in price["sensors"] if sensor["score"] is not None] assert price["score"] == max(sensor_scores) - assert result["methodology"] == "v4" + assert result["methodology"] == rms.METHODOLOGY assert "combined" not in result assert result["basket"]["members_available"] == 25 diff --git a/tests/unit/test_regime_quadrant_alert.py b/tests/unit/test_regime_quadrant_alert.py index 4fedd27..a7aa705 100644 --- a/tests/unit/test_regime_quadrant_alert.py +++ b/tests/unit/test_regime_quadrant_alert.py @@ -5,10 +5,16 @@ different realized ranges -- Warning never exceeded 64.9 in the 408 calibration sessions, so a shared 60 left the whole upper half of that axis unreachable. """ +import pytest + +from app.services import alert_service from app.services.alert_service import ( + CONFLUENCE_TYPE, + FUND_TYPE, QUAD_X_DIV, QUAD_Y_DIV, _classify_quadrant, + _collect_regime_fundamental, _parse_quadrant_log_key, _quadrant_log_key, ) @@ -48,3 +54,140 @@ def test_quadrant_key_carries_basket_hash_and_parses_legacy_keys(): assert _parse_quadrant_log_key(key) == ("abc123", "3", 32.4, 54.6) assert _parse_quadrant_log_key("3:32.4:54.6") == (None, "3", 32.4, 54.6) assert _parse_quadrant_log_key("3") == (None, "3", None, None) + + +# --------------------------------------------------------------------------- +# Fundamental-context and confluence alerts +# --------------------------------------------------------------------------- + +def _monitor( + warning_score: float, state: str, *, coverage: float = 100.0, usable: bool = True +) -> dict: + return { + "available": True, + "warning": {"score": warning_score, "coverage": coverage}, + "fundamental_context": { + "state": state, + "evidence_quality": "complete" if usable else "stale", + # The state survives going stale so the card can still show it, and + # a failed extraction is fresh but knows nothing; `usable` is what + # says whether it may still confirm anything. + "available": usable, + "usable": usable, + }, + "data_quality": {"is_fresh": True}, + "quadrant_config": {"warning_divider": QUAD_Y_DIV}, + } + + +class _LogSpyDB: + """Records what would be logged; returns a canned "last logged key".""" + + def __init__(self, last: dict[str, str | None]) -> None: + self.last = last + self.logged: list[tuple[str, str]] = [] + + +@pytest.fixture +def patched(monkeypatch): + def apply(data: dict, last: dict[str, str | None]): + db = _LogSpyDB(last) + + async def fake_monitor(_db): + return data + + async def fake_last(_db, alert_type): + return db.last.get(alert_type) + + def fake_log(_db, alert_type, key, value=None): + db.logged.append((alert_type, key)) + + import app.services.regime_monitor_service as rms + + monkeypatch.setattr(rms, "get_regime_monitor", fake_monitor) + monkeypatch.setattr(alert_service, "_last_logged_key", fake_last) + monkeypatch.setattr(alert_service, "_log_alert", fake_log) + return db + + return apply + + +@pytest.mark.asyncio +async def test_first_run_seeds_both_channels_without_alerting(patched): + db = patched(_monitor(60.0, "adverse"), {FUND_TYPE: None, CONFLUENCE_TYPE: None}) + assert await _collect_regime_fundamental(db) == [] + assert dict(db.logged) == {FUND_TYPE: "adverse", CONFLUENCE_TYPE: "yes"} + + +@pytest.mark.asyncio +async def test_fundamental_change_and_confluence_are_separate_messages(patched): + db = patched(_monitor(60.0, "adverse"), {FUND_TYPE: "neutral", CONFLUENCE_TYPE: "no"}) + out = await _collect_regime_fundamental(db) + + assert [alert_type for alert_type, _, _ in out] == [FUND_TYPE, CONFLUENCE_TYPE] + assert "neutral → adverse" in out[0][2] + assert "Confluence" in out[1][2] + # Neither message reports a fused score; they name which channel moved. + assert "not a score" in out[0][2] + + +@pytest.mark.asyncio +async def test_unknown_never_alerts(patched): + """Absence of evidence is not a change in the evidence.""" + db = patched(_monitor(60.0, "unknown"), {FUND_TYPE: "neutral", CONFLUENCE_TYPE: "no"}) + assert await _collect_regime_fundamental(db) == [] + + +@pytest.mark.asyncio +async def test_adverse_alone_is_not_confluence(patched): + """A calm tape with adverse fundamentals is a context change, not confluence.""" + db = patched(_monitor(10.0, "adverse"), {FUND_TYPE: "neutral", CONFLUENCE_TYPE: "no"}) + out = await _collect_regime_fundamental(db) + assert [alert_type for alert_type, _, _ in out] == [FUND_TYPE] + + +@pytest.mark.asyncio +async def test_leaving_confluence_rebaselines_quietly(patched): + db = patched(_monitor(10.0, "neutral"), {FUND_TYPE: "neutral", CONFLUENCE_TYPE: "yes"}) + assert await _collect_regime_fundamental(db) == [] + assert (CONFLUENCE_TYPE, "no") in db.logged + + +@pytest.mark.asyncio +async def test_low_coverage_or_stale_inputs_stay_quiet(patched): + thin = _monitor(60.0, "adverse", coverage=50.0) + assert await _collect_regime_fundamental( + patched(thin, {FUND_TYPE: "neutral", CONFLUENCE_TYPE: "no"}) + ) == [] + + stale = _monitor(60.0, "adverse") + stale["data_quality"]["is_fresh"] = False + assert await _collect_regime_fundamental( + patched(stale, {FUND_TYPE: "neutral", CONFLUENCE_TYPE: "no"}) + ) == [] + + +@pytest.mark.asyncio +async def test_a_stale_observation_cannot_confirm_a_new_crossing(patched): + """The state is kept for display, but it stops being evidence. + + Without this, one adverse read corroborates every Warning crossing for the + rest of time — the strongest claim the channel makes, from the data with the + least right to make it. + """ + stale = _monitor(60.0, "adverse", usable=False) + db = patched(stale, {FUND_TYPE: "adverse", CONFLUENCE_TYPE: "no"}) + assert await _collect_regime_fundamental(db) == [] + # It also rebaselines to "no", so recollecting the observation re-arms it. + assert (CONFLUENCE_TYPE, "no") not in db.logged # already "no"; nothing to log + + fresh = _monitor(60.0, "adverse", usable=True) + db2 = patched(fresh, {FUND_TYPE: "adverse", CONFLUENCE_TYPE: "no"}) + out = await _collect_regime_fundamental(db2) + assert [alert_type for alert_type, _, _ in out] == [CONFLUENCE_TYPE] + + +@pytest.mark.asyncio +async def test_a_stale_state_change_does_not_alert(patched): + db = patched(_monitor(10.0, "adverse", usable=False), {FUND_TYPE: "neutral", CONFLUENCE_TYPE: "no"}) + assert await _collect_regime_fundamental(db) == []