Files
signal-platform/tests/unit/test_event_study.py
T
dennisthiessenandClaude Opus 5 c97a067e0e
Deploy / lint (push) Successful in 8s
Deploy / test (push) Successful in 1m27s
Deploy / deploy (push) Successful in 37s
fix(risk-monitor): drop an unused import and align migration 033 with its model
ruff F401 failed the deploy: `import pytest` in test_event_study.py outlived the
pytest.approx assertion it was added for.

Compiling the migration for Postgres while checking that turned up a second
defect worth fixing while the table is still empty. It created a unique
constraint *and* a plain index on effective_date, while the model declares
`unique=True, index=True` -- one unique index. Both enforce uniqueness, but the
pairing left a redundant second index on the column and a permanent diff for
autogenerate to keep trying to reconcile. Now renders byte-for-byte what the
model declares, matching RegimeSnapshot.date.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-13 11:31:05 +02:00

511 lines
21 KiB
Python

"""Tests for correction events, alarm episodes, the shipped-rule replay, and caveats."""
from __future__ import annotations
from copy import deepcopy
from datetime import date, timedelta
from app.services.breadth_service import _breadth_from_closes, compute_divergence_series
from app.services.event_study_service import (
MIN_EVENTS_FOR_CONFIDENCE,
STRESS_QUADRANT,
WARNING_QUADRANTS,
_era_split,
_null_model,
_percentile,
_reliability,
alarm_episodes,
below_average_series,
detect_events,
entry_alarms,
evaluate_alarms,
replay_quadrant_changes,
)
def _days(count: int, start: date = date(2021, 1, 1)) -> list[date]:
return [start + timedelta(days=index) for index in range(count)]
def _row(
warning: float,
state: float = 0.0,
*,
warning_coverage: float = 100.0,
state_coverage: float = 100.0,
fresh: bool = True,
) -> dict:
return {
"state": state,
"warning": warning,
"state_coverage": state_coverage,
"warning_coverage": warning_coverage,
"inputs_fresh": fresh,
}
def _rows(
dates: list[date], warnings: list[float], patch: dict[int, dict] | None = None
) -> dict[date, dict]:
"""One publishable row per date, with per-position replacements."""
built = {day: _row(value) for day, value in zip(dates, warnings)}
for index, replacement in (patch or {}).items():
built[dates[index]] = replacement
return built
def test_detect_events_uses_rising_edge_and_cooldown():
closes = [100.0] * 300 + [85.0] * 5 + [100.0] * 50 + [85.0] * 5
events = detect_events(closes, _days(len(closes)), threshold_pct=15.0, cooldown=40)
assert [event["index"] for event in events] == [300, 355]
def test_reliability_flags_a_thin_holdout():
"""2/4 must not read like a property of the score."""
dates = _days(100)
backing = dict.fromkeys(dates, 3)
thin = _reliability(dates, 70, backing, events_detected=11, events_in_holdout=4)
assert thin["underpowered"] is True
assert thin["events_detected"] == 11
assert thin["events_in_holdout"] == 4
assert thin["minimum_events"] == MIN_EVENTS_FOR_CONFIDENCE
ample = _reliability(dates, 70, backing, events_detected=20, events_in_holdout=12)
assert ample["underpowered"] is False
def test_reliability_flags_a_sensor_coverage_split():
"""The threshold must not be frozen on a different construct than it is tested on.
Credit history starts partway through the training window, so the score
renormalises over two sensors early and three later.
"""
dates = _days(100)
matched = dict.fromkeys(dates, 3)
assert _reliability(dates, 70, matched, 20, 12)["sensor_coverage_mismatch"] is False
# Training is 40% three-sensor; the holdout is entirely three-sensor.
split_backing = {d: (3 if index >= 42 else 2) for index, d in enumerate(dates)}
mismatched = _reliability(dates, 70, split_backing, 20, 12)
assert mismatched["sensor_coverage_mismatch"] is True
assert mismatched["train_full_sensor_share"] == 40.0
assert mismatched["holdout_full_sensor_share"] == 100.0
assert mismatched["sensors_expected"] == 3
def test_percentile_is_fixed_from_supplied_values():
values = [float(value) for value in range(0, 101, 10)]
assert _percentile(values, 50) == 50.0
assert _percentile(values, 80) == 80.0
assert _percentile([], 80) is None
def test_alarm_requires_upward_crossing_and_reset():
dates = _days(10)
values = [10, 70, 80, 75, 20, 70, 80, 20, 20, 70]
indicator = dict(zip(dates, values))
assert alarm_episodes(indicator, dates, threshold=60) == [1, 5, 9]
def test_holdout_start_does_not_invent_crossing_when_already_high():
dates = _days(6)
indicator = dict(zip(dates, [10, 70, 80, 80, 20, 70]))
assert alarm_episodes(indicator, dates, threshold=60, start_index=3) == [5]
def test_evaluate_alarms_counts_episodes_not_alarm_days():
dates = _days(100)
result = evaluate_alarms([10, 50, 80], [25, 70], dates, horizon=20)
assert result["events_warned"] == 2
assert result["events_missed"] == 0
assert result["false_alarms"] == 1
assert result["median_lead_days"] == 17.5
# ---------------------------------------------------------------------------
# The shipped quadrant rule, replayed
# ---------------------------------------------------------------------------
def test_replay_seeds_silently_and_needs_two_sessions():
"""A one-session spike is not an alert; the second session confirms it.
The alarm is therefore dated at the confirmation rather than at the first
crossing, which costs one session of lead. That is what ships.
"""
dates = _days(10)
spike = _rows(dates, [30] * 5 + [70] + [30] * 4)
assert replay_quadrant_changes(spike, dates) == []
held = _rows(dates, [30] * 5 + [70, 70] + [30] * 3)
fires = replay_quadrant_changes(held, dates)
# The rule alerts on quadrant changes in both directions, so the return to
# calm fires too. Only the entry is a warning about anything.
assert [(f["index"], f["from"], f["to"]) for f in fires] == [
(6, "3", "1"),
(9, "1", "3"),
]
assert entry_alarms(fires, WARNING_QUADRANTS) == [6]
def test_confirmation_classifies_the_prior_session_against_the_baseline():
"""Not against its own predecessor -- the distinction changes the answer.
Warning 42 sits inside the hysteresis deadband. Measured from the standing
"3" baseline it is still "3", so it cannot confirm a move to "1". A chain
that classified each session against the one before it would read 42 as "1"
(having just seen 70) and fire a day later, which production does not do.
"""
dates = _days(10)
rows = _rows(dates, [30, 30, 30, 30, 70, 42, 70, 30, 30, 30])
assert replay_quadrant_changes(rows, dates) == []
def test_cooldown_suppresses_and_the_baseline_only_advances_on_a_fire():
dates = _days(10)
rows = _rows(dates, [30, 30, 30, 30, 70, 70, 30, 30, 30, 30])
fires = replay_quadrant_changes(rows, dates)
# Entry confirmed on day 5. The exit confirms on day 7 but lands inside the
# 3-day cooldown, so it is re-evaluated and fires on day 8 instead.
assert [(f["index"], f["from"], f["to"]) for f in fires] == [
(5, "3", "1"),
(8, "1", "3"),
]
assert entry_alarms(fires, WARNING_QUADRANTS) == [5]
def test_low_coverage_sessions_cannot_confirm():
"""The confirmation source has to be a session that published a band."""
dates = _days(10)
warnings = [30, 30, 30, 30, 30, 70, 70, 30, 30, 30]
visible = replay_quadrant_changes(_rows(dates, warnings), dates)
assert entry_alarms(visible, WARNING_QUADRANTS) == [6]
# Day 5 is the only session that could confirm the entry on day 6; below
# MIN_COVERAGE it never published a band, so day 4 is the prior instead.
hidden = _rows(dates, warnings, {5: _row(70, warning_coverage=70.0)})
assert replay_quadrant_changes(hidden, dates) == []
def test_stale_inputs_block_todays_alert_but_not_tomorrows_confirmation():
"""is_fresh gates the live reading only; the prior session comes from history."""
dates = _days(10)
rows = _rows(dates, [30] * 4 + [70, 70, 70] + [30] * 3, {5: _row(70, fresh=False)})
fires = replay_quadrant_changes(rows, dates)
assert entry_alarms(fires, WARNING_QUADRANTS) == [6]
def test_entry_alarms_ignore_movement_inside_the_set():
fires = [
{"index": 3, "from": "3", "to": "1"},
{"index": 9, "from": "1", "to": "2"},
{"index": 20, "from": "2", "to": "4"},
]
assert entry_alarms(fires, WARNING_QUADRANTS) == [3]
assert entry_alarms(fires, STRESS_QUADRANT) == [9]
def test_below_average_series_needs_a_full_window():
series = list(zip(_days(6), [10.0, 10.0, 10.0, 10.0, 4.0, 20.0]))
indicator = below_average_series(series, window=3)
assert _days(6)[1] not in indicator # warm-up
assert indicator[_days(6)[4]] == 100.0 # 4 is under the 3-day mean of 8
assert indicator[_days(6)[5]] == 0.0
def test_null_model_is_seeded_and_drawn_from_evaluable_sessions_only():
dates = _days(300)
events = [100, 180, 260]
first = _null_model(6, events, dates, horizon=20, start_index=50, observed_warned=2, draws=200)
second = _null_model(6, events, dates, horizon=20, start_index=50, observed_warned=2, draws=200)
assert first == second # a re-run must not move the report
assert 0.0 <= first["p_at_least_observed"] <= 1.0
assert first["alarms_per_draw"] == 6
assert first["mean_warned"] <= len(events)
# More alarms than there are sessions to place them on is not a null.
assert _null_model(500, events, dates, 20, 50, 2, draws=10) is None
assert _null_model(6, [], dates, 20, 50, 0, draws=10) is None
def test_era_split_reports_the_two_sensor_eras_separately():
"""The fuller sample is mostly pre-credit, where Warning is W1+W2 only."""
dates = _days(400)
eras = _era_split(
alarms=[80, 300],
event_indices=[90, 310],
dates=dates,
horizon=20,
start_index=10,
credit_from=dates[200],
)
assert eras["pre_credit"]["events"] == 1
assert eras["pre_credit"]["events_warned"] == 1
assert eras["full_coverage"]["events"] == 1
assert eras["full_coverage"]["events_warned"] == 1
assert eras["credit_from"] == dates[200].isoformat()
# No credit series at all means there is no boundary to split on.
assert _era_split([80], [90], dates, 20, 10, None) is None
def _business_days(count: int, end: date = date(2026, 8, 7)) -> list[date]:
out: list[date] = []
cursor = end
while len(out) < count:
if cursor.weekday() < 5:
out.append(cursor)
cursor -= timedelta(days=1)
return list(reversed(out))
def _synthetic_path(sessions: int) -> list[float]:
"""A rising leader with two deep drawdowns, so corrections exist to detect."""
closes: list[float] = []
for index in range(sessions):
if index < 350:
closes.append(100.0 + index * 0.25)
elif index < 400:
closes.append(187.5 - (index - 350) * 0.9)
elif index < 650:
closes.append(142.5 + (index - 400) * 0.4)
elif index < 700:
closes.append(242.5 - (index - 650) * 1.1)
else:
closes.append(187.5 + (index - 700) * 0.3)
return closes
async def test_report_assembles_every_rule_from_synthetic_inputs(monkeypatch):
"""End-to-end: the shipped replay, ablations, baselines and null all score.
Synthetic rather than recorded because the point is the wiring -- that every
rule is measured on the same events over the same sessions and the report
carries what the panel reads. The numbers are meaningless by construction.
"""
import app.services.event_study_service as ess
sessions = 900
dates = _business_days(sessions)
closes = _synthetic_path(sessions)
leader = list(zip(dates, closes))
# SPY grinds up throughout, so the leader's relative strength rolls over
# exactly when it falls.
market = list(zip(dates, [100.0 + index * 0.12 for index in range(sessions)]))
# Breadth deteriorates ~15 sessions ahead of each decline, which is the
# divergence W1 exists to catch.
breadth = {}
for index, day in enumerate(dates):
weak = 335 <= index < 400 or 635 <= index < 700
breadth[day] = 30.0 if weak else 70.0
vix = [(day, 32.0 if (350 <= i < 400 or 650 <= i < 700) else 15.0) for i, day in enumerate(dates)]
# Credit starts late, exactly as ICE's 3-year cap makes it in production.
oas = [(day, 4.2 if (650 <= i < 700) else 3.0) for i, day in enumerate(dates) if i >= 500]
async def fake_config(_db):
return deepcopy(ess.rms.DEFAULT_CONFIG)
async def fake_prices(_config, _start, _end):
return {"SMH": leader, "QQQ": leader, "SPY": market}
async def fake_fred(series_id, _start, _end):
return {"VIXCLS": vix, "BAMLH0A0HYM2": oas}.get(series_id)
async def fake_breadth(_db, _symbols, window=200, min_tickers=20):
return breadth, {day: 30 for day in dates}
async def fake_observations(_db):
return []
monkeypatch.setattr(ess.rms, "get_regime_config", fake_config)
monkeypatch.setattr(ess.rms, "_fetch_prices", fake_prices)
monkeypatch.setattr(ess.rms, "_fetch_fred_series", fake_fred)
monkeypatch.setattr(ess.rms, "get_fundamental_observations", fake_observations)
monkeypatch.setattr(ess.breadth_service, "compute_breadth_details", fake_breadth)
monkeypatch.setattr(ess, "NULL_DRAWS", 100)
report = await ess.run_event_study(None)
assert report["available"] is True
assert report["schema"] == ess.STUDY_SCHEMA
# The shipped rule is measured on the whole sample, not a 30% holdout.
shipped = report["shipped"]
assert shipped["metrics"]["events"] == report["sample"]["events_evaluable"]
assert report["sample"]["events_evaluable"] >= 2
assert shipped["metrics"]["events"] >= report["fitted"]["metrics"]["events"]
assert len(shipped["events"]) == shipped["metrics"]["events"]
assert {row["kind"] for row in report["comparison"]} == {
"ablation", "baseline", "fundamental",
}
# Market rows share the headline's events, or the table lies. Fundamental
# rows deliberately do not: they are coverage-matched to the sessions the
# channel actually existed on, which is a different (here empty) window.
for row in report["comparison"]:
if row["kind"] != "fundamental":
assert row["events"] == shipped["metrics"]["events"]
assert row["false_alarms_per_year"] >= 0
else:
# No eligible sessions means the rate is undefined, not zero. A
# tiny-divisor fallback here printed 5e9 alarms/year.
assert row["false_alarms_per_year"] is None
# The credit sensor starts mid-sample, so the era split must be populated.
eras = shipped["by_era"]
assert eras["credit_from"] == dates[500].isoformat()
assert eras["pre_credit"]["events"] + eras["full_coverage"]["events"] == shipped["metrics"]["events"]
if report["null_model"] is not None:
assert 0.0 <= report["null_model"]["p_at_least_observed"] <= 1.0
assert report["null_model"]["observed_warned"] == shipped["metrics"]["events_warned"]
# With an empty observation series the fundamental rows are *untested*, not
# failed, and the report has to carry that distinction or a 0/10 in the table
# reads as a measured result.
coverage = report["fundamental_coverage"]
assert coverage["observations"] == 0
assert coverage["sessions_eligible"] == 0
assert coverage["events_covered"] == 0
assert coverage["measurable"] is False
fundamental_rows = [r for r in report["comparison"] if r["kind"] == "fundamental"]
assert {r["id"] for r in fundamental_rows} == {
"fundamental_adverse", "confluence", "market_over_covered",
}
assert all(row["measurable"] is False for row in fundamental_rows)
# Coverage-matched denominators: with no exposure these rows must not claim
# to have been scored against the market rows' 10 corrections.
assert all(row["events"] == 0 for row in fundamental_rows)
# Market rows are unaffected: their inputs exist for the whole window.
assert all(
row["measurable"] is True
for row in report["comparison"]
if row["kind"] != "fundamental"
)
def test_fundamental_rows_are_scored_only_on_their_own_exposure():
"""One day of coverage must not render as 0/10.
A fundamental rule scores zero whether it is wrong or merely absent, so
scoring it against corrections it could never have seen manufactures a
failed result out of a thin one — the same mistake the `measurable` flag
prevents for an empty table, arriving one observation later.
"""
import app.services.event_study_service as ess
dates = _days(300)
events = [50, 120, 200, 280]
# Context exists for a single stretch, covering only the 120 event's horizon.
rows = {
day: {
"fundamental_state": "adverse",
"fundamental_usable": 105 <= index <= 115,
}
for index, day in enumerate(dates)
}
covered = ess.covered_events(events, rows, dates, horizon=20)
assert covered == [120]
assert ess.eligible_sessions(rows, dates, start_index=0) == 11
# A stale stretch counts for nothing, however adverse it reads.
stale = {
day: {"fundamental_state": "adverse", "fundamental_usable": False}
for day in dates
}
assert ess.covered_events(events, stale, dates, horizon=20) == []
assert ess.eligible_sessions(stale, dates, start_index=0) == 0
assert ess.adverse_episodes(stale, dates, 0) == []
assert ess.confluence_episodes([120], stale, dates) == []
# And neither does a *fresh* observation that determined nothing. Repeated
# extraction failures would otherwise accumulate exposure until the rows
# flipped to a measurable 0/8 for a channel that never knew anything —
# the same tested-versus-unavailable confusion, arriving by a slower route.
empty = {
day: {"fundamental_state": "unknown", "fundamental_usable": False}
for day in dates
}
assert ess.covered_events(events, empty, dates, horizon=20) == []
assert ess.eligible_sessions(empty, dates, start_index=0) == 0
async def test_the_fundamental_channel_never_moves_the_warning_score():
"""The channel is compared, never fused. Warning must be identical either way.
A weighted modifier was built and reverted: with ~10 correction events and
almost no fundamental history any fusion weight is a policy preference
presented as a measurement.
"""
import app.services.event_study_service as ess
end = date(2026, 6, 26)
dates = _business_days(400, end)
rising = [(day, 100.0 + index * 0.2) for index, day in enumerate(dates)]
prices = {"SMH": rising, "QQQ": rising, "SPY": rising}
args = (prices, [(end, 20.0)], [(day, 4.0) for day in dates])
config = deepcopy(ess.rms.DEFAULT_CONFIG)
names = config["tickers"]["hyperscalers"]
tail = (rising, [(day, 20.0) for day in dates], dates, config)
def adverse(effective: date) -> list[dict]:
return [{
"effective_date": effective,
"f1_score": 100.0,
"f3_score": 100.0,
"capex": dict.fromkeys(names, "cutting"),
"good_news_stock_down": "yes",
"fetched_at": "2026-01-01T00:00:00+00:00",
}]
bare = ess._axis_rows(*args, *tail, None)
observed = ess._axis_rows(*args, *tail, adverse(dates[-20]))
latest, early = dates[-1], dates[-90]
assert observed[latest]["warning"] == bare[latest]["warning"]
assert observed[latest]["fundamental_state"] == "adverse"
assert bare[latest]["fundamental_state"] == "unknown"
# Sessions before the effective date stay unknown, so a rebuild cannot stamp
# today's reading onto history.
assert observed[early]["fundamental_state"] == "unknown"
# The confluence rule keeps only crossings the channel agrees with, and the
# fundamental rule fires on the transition into adverse -- both rising-edge,
# so both stay comparable with the market rows.
adverse_alarms = ess.adverse_episodes(observed, dates, 0)
assert [dates[i] for i in adverse_alarms] == [dates[-20]]
assert ess.adverse_episodes(bare, dates, 0) == []
assert ess.confluence_episodes([dates.index(early), dates.index(latest)], observed, dates) == [
dates.index(latest)
]
def test_breadth_from_fixed_closes_and_tapered_divergence():
dates = _days(10)
closes_by_symbol = {
"A": list(zip(dates, [1.0 + index for index in range(10)])),
"B": list(zip(dates, [10.0 - index for index in range(10)])),
"C": list(zip(dates, [5.0] * 10)),
}
breadth = _breadth_from_closes(closes_by_symbol, window=3, min_tickers=2)
assert breadth[dates[2]] == round(1 / 3 * 100, 2)
falling_breadth = {dates[index]: 80.0 - index * 3 for index in range(10)}
rising_benchmark = list(zip(dates, [100.0 + index for index in range(10)]))
divergence = compute_divergence_series(falling_breadth, rising_benchmark, lookback=3)
assert divergence[dates[-1]] > 0
# v3: breadth loss with price confirming it is still deterioration, scored at
# DIVERGENCE_CONFIRMED_FLOOR of the masked case rather than discarded. v2's
# hard gate zeroed this and left Warning at 0 through every selloff.
falling_benchmark = list(zip(dates, [100.0 - index for index in range(10)]))
confirmed = compute_divergence_series(falling_breadth, falling_benchmark, lookback=3)
assert 0 < confirmed[dates[-1]] < divergence[dates[-1]]
# Flat breadth is not deterioration regardless of price direction.
flat_breadth = {day: 60.0 for day in dates}
assert compute_divergence_series(flat_breadth, falling_benchmark, lookback=3)[dates[-1]] == 0