Every fix so far only applied to reports generated after deploy. The cached report is served verbatim, so it keeps the recommendation the OLD build stored — quoting the legacy policy book, naming a rejected exit as "recommended", and carrying no basis_lookback, which let the lookback selector default to 3y and put 3-year tiles beside an all-history recommendation with no divergence notice. Exactly the contradiction the last three commits set out to remove, silently present on the first page load after deploy and until the next scheduled run overwrote it. The recommendation is a pure function of the numbers already in the report — its own note says it is derived from them on every run — so it is now re-derived on read. A corrected recommendation appears immediately instead of after the next backtest. On failure it is dropped rather than falling back to the stored one, which is the stale derivation this replaces. The test drives the real shape: an old-build report with a legacy recommendation written straight to the settings row, read back through get_backtest_report. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
1921 lines
73 KiB
Python
1921 lines
73 KiB
Python
"""Tests for the historical backtest harness."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import json
|
||
import math
|
||
from datetime import date, timedelta
|
||
from types import SimpleNamespace
|
||
|
||
import pytest
|
||
|
||
from app.models.ohlcv import OHLCVRecord
|
||
from app.models.ticker import Ticker
|
||
from app.services import backtest_service as bt
|
||
from app.services.outcome_service import (
|
||
OUTCOME_EXPIRED,
|
||
OUTCOME_STOP_HIT,
|
||
OUTCOME_TARGET_HIT,
|
||
)
|
||
from tests.conftest import _test_session_factory # type: ignore
|
||
|
||
|
||
@pytest.fixture
|
||
async def session():
|
||
async with _test_session_factory() as s:
|
||
yield s
|
||
|
||
|
||
def _cand(
|
||
prob: float,
|
||
outcome: str,
|
||
rr: float,
|
||
qualified: bool = True,
|
||
direction: str = "long",
|
||
risk_pct: float = 0.05,
|
||
hold_days: int = 10,
|
||
) -> dict:
|
||
target_hit = outcome == OUTCOME_TARGET_HIT
|
||
realized = rr if target_hit else (0.0 if outcome == OUTCOME_EXPIRED else -1.0)
|
||
return {
|
||
"primary_prob": prob,
|
||
"outcome": outcome,
|
||
"target_hit": target_hit,
|
||
"rr": rr,
|
||
"realized_r": realized,
|
||
"qualified": qualified,
|
||
"direction": direction,
|
||
"risk_pct": risk_pct,
|
||
"hold_days": hold_days,
|
||
}
|
||
|
||
|
||
# Round-trip cost in R for the default _cand risk_pct: 2 * 0.001 / 0.05 = 0.04R.
|
||
_COST_R_005 = 2 * bt.COST_PER_SIDE / 0.05
|
||
|
||
|
||
def _bar(high: float, low: float, close: float, open_: float | None = None) -> SimpleNamespace:
|
||
"""Synthetic daily bar. ``open`` defaults to the high so a stop is pierced
|
||
intraday (fill at the stop level); pass an explicit open beyond the stop to
|
||
model a gap through it."""
|
||
return SimpleNamespace(
|
||
high=high, low=low, close=close, open=open_ if open_ is not None else high
|
||
)
|
||
|
||
|
||
def _signal_test_series(extra_return: float = 0.0) -> tuple[list[date], list[float], list[float], dict[date, float]]:
|
||
base = date(2024, 1, 1)
|
||
dates = [base + timedelta(days=i) for i in range(280)]
|
||
benchmark = [100.0]
|
||
closes = [100.0]
|
||
for i in range(1, len(dates)):
|
||
market_ret = 0.0004 + 0.002 * math.sin(i / 9.0)
|
||
benchmark.append(benchmark[-1] * (1.0 + market_ret))
|
||
# Same market beta for both test stocks; only ``extra_return`` is
|
||
# idiosyncratic drift, which residual momentum should keep.
|
||
stock_ret = 1.4 * market_ret + extra_return
|
||
closes.append(closes[-1] * (1.0 + stock_ret))
|
||
highs = [c * 1.01 for c in closes]
|
||
benchmark_closes = dict(zip(dates, benchmark))
|
||
return dates, closes, highs, benchmark_closes
|
||
|
||
|
||
def test_signal_values_emit_residual_momentum_only_with_benchmark():
|
||
dates, closes, highs, benchmark = _signal_test_series(extra_return=0.0008)
|
||
no_benchmark = bt._signal_values(dates, closes, highs, 260)
|
||
with_benchmark = bt._signal_values(dates, closes, highs, 260, benchmark)
|
||
|
||
assert "mom_12_1" in no_benchmark
|
||
assert "mom_12_1_resid" not in no_benchmark
|
||
assert "mom_12_1_resid" in with_benchmark
|
||
|
||
|
||
def test_residual_momentum_removes_market_beta_but_keeps_specific_drift():
|
||
dates, pure_beta, highs, benchmark = _signal_test_series(extra_return=0.0)
|
||
_, drift_stock, drift_highs, _ = _signal_test_series(extra_return=0.0008)
|
||
|
||
pure = bt._signal_values(dates, pure_beta, highs, 260, benchmark)
|
||
drift = bt._signal_values(dates, drift_stock, drift_highs, 260, benchmark)
|
||
|
||
assert pure["mom_12_1_resid"] == pytest.approx(0.0, abs=0.03)
|
||
assert drift["mom_12_1_resid"] > pure["mom_12_1_resid"] + 0.12
|
||
|
||
|
||
def test_assigns_raw_and_residual_percentiles_independently():
|
||
cands = [
|
||
{"iso_week": (2026, 1), "momentum": 0.10, "residual_momentum": 0.30},
|
||
{"iso_week": (2026, 1), "momentum": 0.30, "residual_momentum": 0.10},
|
||
{"iso_week": (2026, 1), "momentum": 0.20, "residual_momentum": 0.20},
|
||
]
|
||
|
||
bt._assign_momentum_percentiles(cands)
|
||
bt._assign_residual_momentum_percentiles(cands)
|
||
|
||
by_raw = {c["momentum"]: c["momentum_percentile"] for c in cands}
|
||
by_resid = {c["residual_momentum"]: c["residual_momentum_percentile"] for c in cands}
|
||
assert by_raw[0.30] == 100.0
|
||
assert by_raw[0.10] == 0.0
|
||
assert by_resid[0.30] == 100.0
|
||
assert by_resid[0.10] == 0.0
|
||
|
||
|
||
def test_activation_percentile_prefers_residual_with_raw_fallback():
|
||
cands = [
|
||
{"momentum_percentile": 80.0, "residual_momentum_percentile": 95.0},
|
||
{"momentum_percentile": 70.0, "residual_momentum_percentile": None},
|
||
]
|
||
|
||
bt._assign_activation_momentum_percentiles(cands)
|
||
|
||
assert cands[0][bt.PRODUCTION_PERCENTILE_KEY] == 95.0
|
||
assert cands[1][bt.PRODUCTION_PERCENTILE_KEY] == 70.0
|
||
|
||
|
||
def test_low_volatility_percentile_prefers_lower_realized_vol():
|
||
cands = [
|
||
{"iso_week": (2026, 1), "vol_6m": 0.04},
|
||
{"iso_week": (2026, 1), "vol_6m": 0.01},
|
||
{"iso_week": (2026, 1), "vol_6m": 0.02},
|
||
]
|
||
|
||
bt._assign_low_volatility_percentiles(cands)
|
||
|
||
assert cands[1][bt.LOW_VOL_PERCENTILE_KEY] == 100.0
|
||
assert cands[2][bt.LOW_VOL_PERCENTILE_KEY] == 50.0
|
||
assert cands[0][bt.LOW_VOL_PERCENTILE_KEY] == 0.0
|
||
|
||
|
||
def test_residual_low_vol_blend_is_research_only_rank():
|
||
cands = [{
|
||
bt.PRODUCTION_PERCENTILE_KEY: 80.0,
|
||
bt.LOW_VOL_PERCENTILE_KEY: 60.0,
|
||
}]
|
||
|
||
bt._assign_residual_low_vol_blend(cands)
|
||
|
||
assert cands[0][bt.RESIDUAL_LOW_VOL_BLEND_KEY] == 74.0
|
||
|
||
|
||
def test_residual_high_vol_blend_is_research_only_rank():
|
||
cands = [{
|
||
bt.PRODUCTION_PERCENTILE_KEY: 80.0,
|
||
bt.VOL_PERCENTILE_KEY: 60.0,
|
||
}]
|
||
|
||
bt._assign_residual_high_vol_blend(cands)
|
||
|
||
assert cands[0][bt.RESIDUAL_HIGH_VOL_BLEND_90_10_KEY] == 78.0
|
||
assert cands[0][bt.RESIDUAL_HIGH_VOL_BLEND_80_20_KEY] == 76.0
|
||
assert cands[0][bt.RESIDUAL_HIGH_VOL_BLEND_KEY] == 74.0
|
||
assert cands[0][bt.RESIDUAL_HIGH_VOL_BLEND_60_40_KEY] == 72.0
|
||
|
||
|
||
def test_strategy_variants_keep_only_current_research_candidates():
|
||
variants = {cfg["variant"]: cfg for cfg in bt.STRATEGY_VARIANTS}
|
||
|
||
assert "production_raw_80_fixed10" not in variants
|
||
assert "raw_80_regime_scaled" not in variants
|
||
assert "residual_80_regime_scaled" not in variants
|
||
assert "residual_90_fixed10" not in variants
|
||
assert "raw_90_fixed15" not in variants
|
||
assert "residual_80_fixed20" not in variants
|
||
assert variants["production_residual_80_fixed10"]["percentile_key"] == bt.PRODUCTION_PERCENTILE_KEY
|
||
assert variants["legacy_raw_80_fixed10"]["percentile_key"] == bt.RAW_PERCENTILE_KEY
|
||
assert variants["residual_80_fixed15"]["max_positions"] == 15
|
||
assert variants["residual80_lowvol50_fixed10"]["filters"] == (
|
||
(bt.PRODUCTION_PERCENTILE_KEY, 80.0),
|
||
(bt.LOW_VOL_PERCENTILE_KEY, 50.0),
|
||
)
|
||
assert variants["residual80_lowvol_blend_fixed10"]["ranking_key"] == bt.RESIDUAL_LOW_VOL_BLEND_KEY
|
||
assert variants["residual80_highvol50_fixed10"]["filters"] == (
|
||
(bt.PRODUCTION_PERCENTILE_KEY, 80.0),
|
||
(bt.VOL_PERCENTILE_KEY, 50.0),
|
||
)
|
||
assert variants["residual80_highvol_blend90_10_fixed10"]["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_90_10_KEY
|
||
assert variants["residual80_highvol_blend80_20_fixed10"]["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_80_20_KEY
|
||
assert variants["residual80_highvol_blend_fixed10"]["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_KEY
|
||
assert variants["residual80_highvol_blend60_40_fixed10"]["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_60_40_KEY
|
||
assert variants["highvol80_fixed10"]["percentile_key"] == bt.VOL_PERCENTILE_KEY
|
||
assert variants["lowvol80_fixed10"]["percentile_key"] == bt.LOW_VOL_PERCENTILE_KEY
|
||
assert all(cfg["risk_scale"] is None for cfg in bt.STRATEGY_VARIANTS)
|
||
|
||
|
||
def test_low_vol_strategy_variant_applies_secondary_filter():
|
||
cfg = {
|
||
"percentile_key": bt.PRODUCTION_PERCENTILE_KEY,
|
||
"cutoff": 80.0,
|
||
"filters": (
|
||
(bt.PRODUCTION_PERCENTILE_KEY, 80.0),
|
||
(bt.LOW_VOL_PERCENTILE_KEY, 70.0),
|
||
),
|
||
}
|
||
base = {
|
||
"meets_core": True,
|
||
"direction": "long",
|
||
bt.PRODUCTION_PERCENTILE_KEY: 85.0,
|
||
}
|
||
|
||
assert bt._qualifies_strategy_variant({**base, bt.LOW_VOL_PERCENTILE_KEY: 75.0}, cfg)
|
||
assert not bt._qualifies_strategy_variant({**base, bt.LOW_VOL_PERCENTILE_KEY: 65.0}, cfg)
|
||
|
||
|
||
def test_strategy_variant_sims_emit_fixed_variants_without_mutating_qualified(monkeypatch):
|
||
cands = [{
|
||
"qualified": False,
|
||
"meets_core": True,
|
||
"direction": "long",
|
||
"momentum_percentile": 90.0,
|
||
"residual_momentum_percentile": 91.0,
|
||
"activation_momentum_percentile": 91.0,
|
||
"low_vol_6m_percentile": 80.0,
|
||
"residual_low_vol_blend_score": 87.7,
|
||
"vol_6m_percentile": 20.0,
|
||
"residual_high_vol_blend_90_10_score": 83.9,
|
||
"residual_high_vol_blend_80_20_score": 76.8,
|
||
"residual_high_vol_blend_score": 69.7,
|
||
"residual_high_vol_blend_60_40_score": 62.6,
|
||
}]
|
||
calls = []
|
||
|
||
def fake_sim(candidates, prices, spy_closes, exit_policy, hold_days, **kwargs):
|
||
calls.append({"exit_policy": exit_policy, "hold_days": hold_days, **kwargs})
|
||
return {
|
||
"starting_capital": bt.SIM_STARTING_CAPITAL,
|
||
"final_equity": 11_000.0,
|
||
"total_return_pct": 10.0,
|
||
"cagr_pct": 9.0,
|
||
"max_drawdown_pct": 5.0,
|
||
"sharpe": 1.1,
|
||
"trades": 1,
|
||
"win_rate": 100.0,
|
||
"avg_trade_pnl": 100.0,
|
||
"best_trade_r": 1.0,
|
||
"worst_trade_r": 1.0,
|
||
"best_trade_pnl": 100.0,
|
||
"worst_trade_pnl": 100.0,
|
||
"avg_hold_days": 30.0,
|
||
"skipped_book_full": 0,
|
||
"spy_return_pct": 1.0,
|
||
"yearly_returns": [],
|
||
"start_date": "2026-01-01",
|
||
"end_date": "2026-02-01",
|
||
}
|
||
|
||
monkeypatch.setattr(bt, "_simulate_portfolio", fake_sim)
|
||
rows = bt._strategy_variant_sims(cands, {}, {}, 30)
|
||
|
||
assert [r["variant"] for r in rows] == [cfg["variant"] for cfg in bt.STRATEGY_VARIANTS]
|
||
assert all(call["exit_policy"] == "hold" for call in calls)
|
||
assert any(call["ranking_key"] == bt.PRODUCTION_PERCENTILE_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.RAW_PERCENTILE_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.RESIDUAL_LOW_VOL_BLEND_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_90_10_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_80_20_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_60_40_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.VOL_PERCENTILE_KEY for call in calls)
|
||
assert any(call["ranking_key"] == bt.LOW_VOL_PERCENTILE_KEY for call in calls)
|
||
assert any(call["max_positions"] == 15 for call in calls)
|
||
assert cands[0]["qualified"] is False
|
||
|
||
|
||
def test_exit_policy_sims_use_80_20_entry_variant(monkeypatch):
|
||
calls = []
|
||
|
||
def fake_sim(candidates, prices, spy_closes, exit_policy, hold_days, **kwargs):
|
||
calls.append({"exit_policy": exit_policy, "hold_days": hold_days, **kwargs})
|
||
return {
|
||
"starting_capital": bt.SIM_STARTING_CAPITAL,
|
||
"final_equity": 11_000.0,
|
||
"total_return_pct": 10.0,
|
||
"cagr_pct": 9.0,
|
||
"max_drawdown_pct": 5.0,
|
||
"sharpe": 1.1,
|
||
"trades": 1,
|
||
"win_rate": 100.0,
|
||
"avg_trade_pnl": 100.0,
|
||
"best_trade_r": 1.0,
|
||
"worst_trade_r": 1.0,
|
||
"best_trade_pnl": 100.0,
|
||
"worst_trade_pnl": 100.0,
|
||
"avg_hold_days": 30.0,
|
||
"exit_reasons": {exit_policy: 1},
|
||
"skipped_book_full": 0,
|
||
"spy_return_pct": 1.0,
|
||
"yearly_returns": [],
|
||
"start_date": "2026-01-01",
|
||
"end_date": "2026-02-01",
|
||
}
|
||
|
||
monkeypatch.setattr(bt, "_simulate_portfolio", fake_sim)
|
||
|
||
rows = bt._exit_policy_sims([], {}, {}, 30)
|
||
|
||
assert [r["exit_policy"] for r in rows] == [
|
||
cfg["exit_policy"] for cfg in bt.EXIT_POLICY_VARIANTS
|
||
]
|
||
assert all(r["entry_variant"] == bt.EXIT_ENTRY_VARIANT for r in rows)
|
||
assert all(call["ranking_key"] == bt.RESIDUAL_HIGH_VOL_BLEND_80_20_KEY for call in calls)
|
||
assert all(call["exit_policy"] != "target" for call in calls)
|
||
|
||
|
||
def test_build_research_recommendation_applies_promotion_rules():
|
||
report = {
|
||
"strategy_variants": {"variants": [
|
||
{"variant": "production_residual_80_fixed10", "label": "Base", "sharpe": 1.40,
|
||
"max_drawdown_pct": 20.0, "cagr_pct": 32.0, "skipped_book_full": 7},
|
||
{"variant": "residual_80_fixed15", "label": "Capacity", "sharpe": 1.39,
|
||
"max_drawdown_pct": 20.0, "cagr_pct": 32.0, "skipped_book_full": 0},
|
||
{"variant": "raw_90_fixed10", "label": "Cutoff 90", "sharpe": 1.25,
|
||
"max_drawdown_pct": 19.0, "cagr_pct": 28.0},
|
||
{"variant": "residual80_highvol_blend_fixed10", "label": "High-vol 70/30",
|
||
"sharpe": 1.68, "max_drawdown_pct": 21.0, "cagr_pct": 43.0},
|
||
{"variant": "residual80_highvol_blend80_20_fixed10", "label": "High-vol 80/20",
|
||
"sharpe": 1.55, "max_drawdown_pct": 19.0, "cagr_pct": 39.0},
|
||
]},
|
||
}
|
||
|
||
rec = bt._build_research_recommendation(report)
|
||
by_topic = {item["topic"]: item for item in rec["items"]}
|
||
|
||
assert by_topic["capacity_15"]["candidate"] is False
|
||
assert "not needed yet" in by_topic["capacity_15"]["text"]
|
||
assert by_topic["cutoff_90"]["candidate"] is False
|
||
assert "Cutoff 90" in by_topic["cutoff_90"]["text"]
|
||
assert by_topic["high_vol_overlay"]["candidate"] is True
|
||
assert "High-vol 80/20" in by_topic["high_vol_overlay"]["text"]
|
||
|
||
|
||
class TestStopFillR:
|
||
def test_intraday_fill_at_stop(self):
|
||
assert bt._stop_fill_r("long", 100.0, 95.0, _bar(101, 94, 96)) == pytest.approx(-1.0)
|
||
|
||
def test_gap_fill_at_open(self):
|
||
# Opens at 92, below the 95 stop → filled at the open, worse than −1R.
|
||
assert bt._stop_fill_r("long", 100.0, 95.0, _bar(93, 90, 91, open_=92)) == pytest.approx(-1.6)
|
||
|
||
def test_short_gap_fill_at_open(self):
|
||
# Short stop 105; opens at 107 above it → fill 107.
|
||
assert bt._stop_fill_r("short", 100.0, 105.0, _bar(110, 104, 108, open_=107)) == pytest.approx(-1.4)
|
||
|
||
|
||
class TestRiskAndStopDay:
|
||
def test_no_stop(self):
|
||
risk, stop_day = bt._risk_and_stop_day("long", 100.0, 95.0, [_bar(109, 101, 108)], 30)
|
||
assert risk == pytest.approx(0.05)
|
||
assert stop_day is None
|
||
|
||
def test_stop_day_is_one_based(self):
|
||
bars = [_bar(102, 99, 101), _bar(101, 94, 96)]
|
||
risk, stop_day = bt._risk_and_stop_day("long", 100.0, 95.0, bars, 30)
|
||
assert risk == pytest.approx(0.05)
|
||
assert stop_day == 2
|
||
|
||
def test_short_direction(self):
|
||
_, stop_day = bt._risk_and_stop_day("short", 100.0, 105.0, [_bar(106, 101, 104)], 30)
|
||
assert stop_day == 1
|
||
|
||
|
||
class TestTimeExits:
|
||
def test_long_exits_at_horizon_close(self):
|
||
bars = [_bar(103, 99, 102), _bar(105, 101, 104), _bar(107, 103, 106)]
|
||
res = bt._time_exits("long", 100.0, 95.0, bars, (2, 5))
|
||
assert res[2] == pytest.approx(0.8) # close 104 → +4% / 5% risk
|
||
assert res[5] == pytest.approx(1.2) # only 3 bars → last close 106
|
||
|
||
def test_stop_on_first_bar_loses_everywhere(self):
|
||
res = bt._time_exits("long", 100.0, 95.0, [_bar(101, 94, 96), _bar(105, 101, 104)], (1, 5))
|
||
assert res[1] == pytest.approx(-1.0)
|
||
assert res[5] == pytest.approx(-1.0)
|
||
|
||
def test_stop_after_short_horizon_only_hits_long_hold(self):
|
||
# Day-2 close banked by the 2-day hold; the stop on day 3 only hits n=5.
|
||
bars = [_bar(103, 99, 102), _bar(104, 100, 103), _bar(101, 94, 95)]
|
||
res = bt._time_exits("long", 100.0, 95.0, bars, (2, 5))
|
||
assert res[2] == pytest.approx(0.6) # close 103 → +3% / 5% risk
|
||
assert res[5] == pytest.approx(-1.0)
|
||
|
||
def test_short_direction(self):
|
||
res = bt._time_exits("short", 100.0, 105.0, [_bar(101, 95, 96)], (1,))
|
||
assert res[1] == pytest.approx(0.8) # close 96 → +4% / 5% risk
|
||
|
||
def test_zero_risk_returns_zero(self):
|
||
res = bt._time_exits("long", 100.0, 100.0, [_bar(103, 99, 102)], (5,))
|
||
assert res[5] == 0.0
|
||
|
||
def test_gap_through_stop_fills_at_open(self):
|
||
res = bt._time_exits("long", 100.0, 95.0, [_bar(93, 90, 91, open_=92)], (5,))
|
||
assert res[5] == pytest.approx(-1.6)
|
||
|
||
|
||
class TestTimeExitBucket:
|
||
def test_bucket(self):
|
||
cands = [
|
||
{"time_r": {5: 1.4, 21: 0.8}, "risk_pct": 0.10},
|
||
{"time_r": {5: -1.0, 21: -1.0}, "risk_pct": 0.10},
|
||
{"time_r": {5: 0.5, 21: 0.5}, "risk_pct": 0.10},
|
||
]
|
||
b = bt._time_exit_bucket(cands, 5)
|
||
assert b["hold_days"] == 5
|
||
assert b["total"] == 3
|
||
assert b["wins"] == 2
|
||
assert b["win_rate"] == pytest.approx(66.7, abs=0.1)
|
||
assert b["avg_r"] == pytest.approx(0.3, abs=0.01)
|
||
assert b["net_avg_r"] == pytest.approx(0.28, abs=0.01)
|
||
assert b["best_r"] == pytest.approx(1.4)
|
||
assert b["worst_r"] == pytest.approx(-1.0)
|
||
# No stop_day on any candidate → every hold runs the full 5 days.
|
||
assert b["avg_hold_days"] == 5.0
|
||
assert b["net_r_per_day"] == pytest.approx(0.28 / 5.0, abs=0.001)
|
||
# robustness on net rs [1.38, -1.02, 0.48]
|
||
assert b["median_net_r"] == pytest.approx(0.48, abs=0.001)
|
||
assert b["profit_factor"] == pytest.approx(1.86 / 1.02, abs=0.01)
|
||
assert b["net_avg_r_ex_top5"] == pytest.approx((0.48 - 1.02) / 2, abs=0.001)
|
||
|
||
def test_missing_hold_skipped(self):
|
||
b = bt._time_exit_bucket([{"time_r": {5: 1.0}}], 21)
|
||
assert b["total"] == 0
|
||
assert b["avg_r"] is None
|
||
|
||
|
||
def _acand(
|
||
rr: float = 2.0,
|
||
conf: float = 60.0,
|
||
action: str = "LONG_MODERATE",
|
||
mp: float | None = 90.0,
|
||
direction: str = "long",
|
||
) -> dict:
|
||
"""Ablation candidate: meets_core mirrors the default floors (min_rr 1.2,
|
||
min_confidence 55, exclude_neutral on)."""
|
||
action_dir = "long" if action.startswith("LONG") else "short" if action.startswith("SHORT") else "neutral"
|
||
meets = rr >= 1.2 and conf >= 55.0 and action_dir != "neutral" and action_dir == direction
|
||
return {
|
||
"rr": rr,
|
||
"confidence": conf,
|
||
"action": action,
|
||
"momentum_percentile": mp,
|
||
"activation_momentum_percentile": mp,
|
||
"direction": direction,
|
||
"meets_core": meets,
|
||
"risk_level": "Low",
|
||
"target_hit": True,
|
||
"outcome": OUTCOME_TARGET_HIT,
|
||
"realized_r": rr,
|
||
"risk_pct": 0.05,
|
||
"time_r": {d: 0.5 for d in bt.TIME_EXIT_DAYS},
|
||
}
|
||
|
||
|
||
class TestGateAblation:
|
||
ACTIVATION = {
|
||
"min_rr": 1.2,
|
||
"min_confidence": 55.0,
|
||
"exclude_neutral": True,
|
||
"require_high_conviction": False,
|
||
"exclude_conflicts": False,
|
||
}
|
||
|
||
def test_variant_counts(self):
|
||
cands = [
|
||
_acand(), # clears everything
|
||
_acand(conf=40.0), # fails confidence floor
|
||
_acand(rr=1.0), # fails R:R floor
|
||
_acand(action="NEUTRAL"), # fails NEUTRAL exclusion
|
||
_acand(mp=50.0), # fails the momentum cutoff
|
||
_acand(direction="short", action="SHORT_MODERATE", mp=95.0), # short — gated out
|
||
]
|
||
rows = {r["variant"]: r for r in bt._gate_ablation(cands, self.ACTIVATION, 80.0)}
|
||
assert rows["all_floors"]["total"] == 1
|
||
assert rows["no_confidence_floor"]["total"] == 2
|
||
assert rows["no_rr_floor"]["total"] == 2
|
||
assert rows["no_neutral_exclusion"]["total"] == 2
|
||
assert rows["momentum_only"]["total"] == 4
|
||
assert rows["all_floors"]["net_avg_r"] is not None
|
||
# Every variant is also graded under the hold-to-horizon exit.
|
||
assert rows["all_floors"]["hold_days"] == max(bt.TIME_EXIT_DAYS)
|
||
assert rows["all_floors"]["hold_avg_r"] == pytest.approx(0.5)
|
||
assert rows["all_floors"]["hold_net_avg_r"] is not None
|
||
assert rows["momentum_only"]["hold_total_r"] == pytest.approx(4 * 0.5, abs=0.01)
|
||
|
||
def test_threshold_zero_disables_momentum_gate(self):
|
||
# Floors only: the short and the low-momentum long both pass all_floors.
|
||
cands = [_acand(mp=50.0), _acand(direction="short", action="SHORT_MODERATE", mp=None)]
|
||
rows = {r["variant"]: r for r in bt._gate_ablation(cands, self.ACTIVATION, 0.0)}
|
||
assert rows["all_floors"]["total"] == 2
|
||
|
||
|
||
def _sim_prices(start_ord: int, closes: list[float]) -> tuple:
|
||
"""Column arrays for consecutive daily bars: open = close (no gaps),
|
||
high/low = close ± 1."""
|
||
ords = list(range(start_ord, start_ord + len(closes)))
|
||
return (
|
||
ords,
|
||
list(closes),
|
||
[c + 1.0 for c in closes],
|
||
[c - 1.0 for c in closes],
|
||
list(closes),
|
||
[1_000_000] * len(closes),
|
||
)
|
||
|
||
|
||
def _sim_cand(
|
||
sym: str, day_ord: int, entry: float, stop: float, target: float, mp: float = 90.0
|
||
) -> dict:
|
||
return {
|
||
"qualified": True,
|
||
"direction": "long",
|
||
"symbol": sym,
|
||
"date": date.fromordinal(day_ord).isoformat(),
|
||
"entry": entry,
|
||
"stop": stop,
|
||
"target": target,
|
||
"momentum_percentile": mp,
|
||
"activation_momentum_percentile": mp,
|
||
}
|
||
|
||
|
||
class TestSimulatePortfolio:
|
||
ORD = date(2025, 1, 6).toordinal()
|
||
|
||
def test_hold_policy_accounting(self):
|
||
closes = [100.0, 102.0, 104.0, 106.0, 108.0, 110.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
cand = _sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=130.0)
|
||
sim = bt._simulate_portfolio([cand], prices, None, "hold", 3)
|
||
assert sim is not None
|
||
assert sim["trades"] == 1
|
||
assert sim["cost_per_side_pct"] == pytest.approx(0.1)
|
||
# 20 shares (1% risk / $5 stop distance), exit at the day-3 close 106:
|
||
# pnl = 2120 − 2000 − 2.00 entry cost − 2.12 exit cost = 115.88
|
||
assert sim["final_equity"] == pytest.approx(10_115.88, abs=0.01)
|
||
assert sim["win_rate"] == 100.0
|
||
assert sim["best_trade_r"] == pytest.approx(1.2)
|
||
assert sim["avg_hold_days"] == 3.0
|
||
assert sim["max_drawdown_pct"] == 0.0
|
||
assert sim["cagr_pct"] is None # window far too short to annualize
|
||
assert sim["spy_return_pct"] is None
|
||
assert sim["yearly_returns"] == [
|
||
{"year": 2025, "return_pct": pytest.approx(1.2, abs=0.05)}
|
||
]
|
||
|
||
def test_cost_parameter_changes_cash_and_position_path(self):
|
||
closes = [100.0, 102.0, 104.0, 106.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
cand = _sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=130.0)
|
||
|
||
free = bt._simulate_portfolio(
|
||
[cand], prices, None, "hold", 3, cost_per_side=0.0
|
||
)
|
||
stressed = bt._simulate_portfolio(
|
||
[cand], prices, None, "hold", 3, cost_per_side=0.002
|
||
)
|
||
|
||
assert free is not None and stressed is not None
|
||
assert free["final_equity"] == pytest.approx(10_120.0, abs=0.01)
|
||
assert stressed["cost_per_side_pct"] == pytest.approx(0.2)
|
||
assert stressed["final_equity"] == pytest.approx(10_111.76, abs=0.01)
|
||
|
||
def test_cost_parameter_rejects_invalid_rate(self):
|
||
with pytest.raises(ValueError, match="cost_per_side"):
|
||
bt._simulate_portfolio(
|
||
[], {}, None, "hold", 3, cost_per_side=-0.001
|
||
)
|
||
|
||
def test_target_policy_exits_at_target(self):
|
||
closes = [100.0, 102.0, 104.0, 106.0, 108.0, 110.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
cand = _sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=105.0)
|
||
sim = bt._simulate_portfolio([cand], prices, None, "target", 30)
|
||
assert sim is not None
|
||
assert sim["trades"] == 1
|
||
assert sim["best_trade_r"] == pytest.approx(1.0) # filled exactly at 105
|
||
|
||
def test_stop_gap_fills_at_open(self):
|
||
# Day-1 bar gaps to a 90 open, below the 95 stop → fill at the open.
|
||
ords = list(range(self.ORD, self.ORD + 2))
|
||
prices = {"AAA": (ords, [100.0, 90.0], [101.0, 92.0], [99.0, 88.0], [100.0, 91.0], [1, 1])}
|
||
cand = _sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=120.0)
|
||
sim = bt._simulate_portfolio([cand], prices, None, "hold", 30)
|
||
assert sim is not None
|
||
assert sim["trades"] == 1
|
||
assert sim["worst_trade_r"] == pytest.approx(-2.0) # (90 − 100) / 5
|
||
|
||
def test_initial_stop_cooldown_blocks_immediate_reentry(self):
|
||
closes = [100.0, 94.0, 96.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
candidates = [
|
||
_sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=120.0),
|
||
_sim_cand("AAA", self.ORD + 1, entry=94.0, stop=89.0, target=110.0),
|
||
]
|
||
|
||
baseline = bt._simulate_portfolio(candidates, prices, None, "hold", 30)
|
||
cooldown = bt._simulate_portfolio(
|
||
candidates,
|
||
prices,
|
||
None,
|
||
"hold",
|
||
30,
|
||
reentry_cooldown_sessions=5,
|
||
)
|
||
|
||
assert baseline is not None and baseline["trades"] == 2
|
||
assert cooldown is not None and cooldown["trades"] == 1
|
||
assert cooldown["skipped_cooldown"] == 1
|
||
assert cooldown["reentry_cooldown_sessions"] == 5
|
||
|
||
def test_initial_stop_cooldown_unlocks_exactly_after_session_five(self):
|
||
closes = [100.0, 94.0, 96.0, 96.0, 96.0, 96.0, 97.0, 98.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
candidates = [
|
||
_sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=120.0),
|
||
# Four completed sessions since the stop: still locked.
|
||
_sim_cand("AAA", self.ORD + 5, entry=96.0, stop=90.0, target=115.0),
|
||
# Five completed sessions since the stop: first permitted re-entry.
|
||
_sim_cand("AAA", self.ORD + 6, entry=97.0, stop=90.0, target=118.0),
|
||
]
|
||
|
||
sim = bt._simulate_portfolio(
|
||
candidates,
|
||
prices,
|
||
None,
|
||
"hold",
|
||
30,
|
||
reentry_cooldown_sessions=5,
|
||
include_trades=True,
|
||
)
|
||
|
||
assert sim is not None
|
||
assert sim["trades"] == 2
|
||
assert sim["skipped_cooldown"] == 1
|
||
assert sim["trade_details"][1]["entry_date"] == date.fromordinal(
|
||
self.ORD + 6
|
||
).isoformat()
|
||
|
||
def test_post_stop_reentry_cannot_cross_holdout_end(self):
|
||
prices = {"AAA": _sim_prices(self.ORD, [100.0, 94.0, 96.0, 98.0])}
|
||
candidate = _sim_cand(
|
||
"AAA", self.ORD, entry=100.0, stop=95.0, target=120.0
|
||
)
|
||
callback_dates: list[int] = []
|
||
|
||
def reenter_after_split(symbol, asof_ord, _state, _bar):
|
||
callback_dates.append(asof_ord)
|
||
if asof_ord < self.ORD + 2:
|
||
return None
|
||
return _sim_cand(
|
||
symbol, asof_ord, entry=96.0, stop=90.0, target=115.0
|
||
)
|
||
|
||
sim = bt._simulate_portfolio(
|
||
[candidate],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
3,
|
||
end_date=date.fromordinal(self.ORD + 2),
|
||
post_stop_reentry_fn=reenter_after_split,
|
||
)
|
||
|
||
assert sim is not None
|
||
assert sim["trades"] == 1
|
||
assert callback_dates == [self.ORD + 1]
|
||
|
||
def test_gate_reset_waits_for_failed_evaluation_then_requalification(self):
|
||
closes = [100.0] * 95
|
||
entry_ord = self.ORD + bt.MIN_LOOKBACK - 1
|
||
stop_ord = entry_ord + 1
|
||
reentry_ord = entry_ord + 3
|
||
closes[bt.MIN_LOOKBACK] = 94.0
|
||
closes[bt.MIN_LOOKBACK + 1] = 95.0
|
||
closes[bt.MIN_LOOKBACK + 2] = 96.0
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
candidates = [
|
||
_sim_cand("AAA", entry_ord, entry=100.0, stop=95.0, target=120.0),
|
||
# Still qualified on the stop day: this must not unlock re-entry.
|
||
_sim_cand("AAA", stop_ord, entry=94.0, stop=89.0, target=110.0),
|
||
# No candidate on the intervening session means the daily gate
|
||
# failed. A fresh qualification on the next session may re-enter.
|
||
_sim_cand("AAA", reentry_ord, entry=96.0, stop=90.0, target=115.0),
|
||
]
|
||
gate_reset = bt._make_gate_reset_reentry_fn(
|
||
candidates,
|
||
prices,
|
||
cadence="daily",
|
||
)
|
||
|
||
sim = bt._simulate_portfolio(
|
||
candidates,
|
||
prices,
|
||
None,
|
||
"hold",
|
||
30,
|
||
post_stop_reentry_fn=gate_reset,
|
||
include_trades=True,
|
||
)
|
||
|
||
assert sim is not None
|
||
assert sim["post_stop_reentries"] == 1
|
||
assert sim["trade_details"][1]["entry_date"] == date.fromordinal(
|
||
reentry_ord
|
||
).isoformat()
|
||
assert sim["reentry_events"][0]["wait_sessions"] == 2
|
||
|
||
def test_production_monitor_applies_live_gate_reset(self, monkeypatch):
|
||
def fake_simulator(*_args, **kwargs):
|
||
return {
|
||
"trades": 0,
|
||
"applied_gate_reset": kwargs.get("post_stop_reentry_fn") is not None,
|
||
}
|
||
|
||
monkeypatch.setattr(bt, "_simulate_portfolio", fake_simulator)
|
||
market_ord = date(2026, 7, 1).toordinal()
|
||
prices = {"AAA": ([market_ord], [], [], [], [], [])}
|
||
|
||
monitor = bt._portfolio_monitor([], prices, None, 30)
|
||
production_rows = [
|
||
row for row in monitor["runs"] if row["is_production"]
|
||
]
|
||
immediate_rows = [
|
||
row for row in monitor["runs"]
|
||
if row["comparison_arm"] == "live_immediate"
|
||
]
|
||
|
||
assert production_rows
|
||
assert all(
|
||
row["reentry_policy"] == "gate_reset"
|
||
and row["applied_gate_reset"] is True
|
||
for row in production_rows
|
||
)
|
||
assert immediate_rows
|
||
assert all(
|
||
row["reentry_policy"] == "immediate"
|
||
and row["applied_gate_reset"] is False
|
||
for row in immediate_rows
|
||
)
|
||
|
||
def test_production_cadence_comparison_names_exact_two_arms(self):
|
||
monitor = {
|
||
"runs": [
|
||
{
|
||
"comparison_arm": "live_immediate",
|
||
"lookback": "all",
|
||
"reentry_policy": "immediate",
|
||
"trades": 10,
|
||
"equity_curve": [{"date": "2026-01-01", "value": 1.0}],
|
||
},
|
||
{
|
||
"comparison_arm": "live_gate_reset",
|
||
"lookback": "all",
|
||
"reentry_policy": "gate_reset",
|
||
"trades": 8,
|
||
"benchmark_curve": [{"date": "2026-01-01", "value": 1.0}],
|
||
},
|
||
]
|
||
}
|
||
|
||
comparison = bt._production_cadence_comparison(monitor, "daily")
|
||
|
||
assert comparison is not None
|
||
assert [row["arm"] for row in comparison["arms"]] == [
|
||
"prod_live_setup_daily",
|
||
"gate_reset_daily",
|
||
]
|
||
assert all("equity_curve" not in row for row in comparison["arms"])
|
||
assert all("benchmark_curve" not in row for row in comparison["arms"])
|
||
|
||
def test_initial_stop_can_refresh_lower_and_survive_same_bar(self):
|
||
closes = [100.0, 94.0, 96.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
candidate = _sim_cand(
|
||
"AAA", self.ORD, entry=100.0, stop=95.0, target=120.0
|
||
)
|
||
|
||
sim = bt._simulate_portfolio(
|
||
[candidate],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
2,
|
||
initial_stop_refresh_fn=lambda *_: 90.0,
|
||
include_trades=True,
|
||
)
|
||
|
||
assert sim is not None
|
||
assert sim["stop_refresh_attempts"] == 1
|
||
assert sim["stop_refreshes"] == 1
|
||
assert sim["stop_refresh_same_bar_hits"] == 0
|
||
assert sim["exit_reasons"] == {"time": 1}
|
||
assert sim["trade_details"][0]["stop_refreshes"] == 1
|
||
|
||
def test_refreshed_stop_is_checked_against_same_bar(self):
|
||
ords = list(range(self.ORD, self.ORD + 2))
|
||
prices = {
|
||
"AAA": (
|
||
ords,
|
||
[100.0, 94.0],
|
||
[101.0, 96.0],
|
||
[99.0, 89.0],
|
||
[100.0, 94.0],
|
||
[1, 1],
|
||
)
|
||
}
|
||
candidate = _sim_cand(
|
||
"AAA", self.ORD, entry=100.0, stop=95.0, target=120.0
|
||
)
|
||
|
||
sim = bt._simulate_portfolio(
|
||
[candidate],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
30,
|
||
initial_stop_refresh_fn=lambda *_: 90.0,
|
||
)
|
||
|
||
assert sim is not None
|
||
assert sim["stop_refresh_same_bar_hits"] == 1
|
||
assert sim["worst_trade_r"] == pytest.approx(-2.0)
|
||
|
||
def test_post_stop_state_suppresses_same_episode_candidate(self):
|
||
closes = [100.0, 94.0, 96.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
candidates = [
|
||
_sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=120.0),
|
||
_sim_cand("AAA", self.ORD + 1, entry=94.0, stop=89.0, target=110.0),
|
||
]
|
||
|
||
sim = bt._simulate_portfolio(
|
||
candidates,
|
||
prices,
|
||
None,
|
||
"hold",
|
||
30,
|
||
post_stop_reentry_fn=lambda *_: None,
|
||
)
|
||
|
||
assert sim is not None
|
||
assert sim["trades"] == 1
|
||
assert sim["post_stop_events"] == 1
|
||
assert sim["post_stop_reentries"] == 0
|
||
assert sim["post_stop_states_open_at_end"] == 1
|
||
|
||
def test_post_stop_callback_can_reenter_same_day(self):
|
||
closes = [100.0, 94.0, 96.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
initial = _sim_cand(
|
||
"AAA", self.ORD, entry=100.0, stop=95.0, target=120.0
|
||
)
|
||
|
||
def immediate_reentry(sym, current_ord, _state, bar):
|
||
return _sim_cand(
|
||
sym,
|
||
current_ord,
|
||
entry=bar.close,
|
||
stop=bar.close - 5.0,
|
||
target=bar.close + 15.0,
|
||
)
|
||
|
||
sim = bt._simulate_portfolio(
|
||
[initial],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
30,
|
||
post_stop_reentry_fn=immediate_reentry,
|
||
include_trades=True,
|
||
)
|
||
|
||
assert sim is not None
|
||
assert sim["trades"] == 2
|
||
assert sim["post_stop_reentries"] == 1
|
||
assert sim["reentry_events"][0]["wait_sessions"] == 0
|
||
assert sim["trade_details"][1]["is_reentry"] is True
|
||
assert sim["trade_details"][1]["reentry_wait_sessions"] == 0
|
||
|
||
def test_sma50_policy_exits_on_close_break(self):
|
||
closes = [100.0] * 56 + [90.0, 91.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
entry_ord = self.ORD + 55
|
||
cand = _sim_cand("AAA", entry_ord, entry=100.0, stop=80.0, target=130.0)
|
||
|
||
sim = bt._simulate_portfolio([cand], prices, None, "sma50", 30)
|
||
|
||
assert sim is not None
|
||
assert sim["trades"] == 1
|
||
assert sim["exit_reasons"] == {"sma50": 1}
|
||
assert sim["worst_trade_r"] == pytest.approx(-0.5)
|
||
|
||
def test_low20_policy_exits_on_prior_low_break(self):
|
||
closes = [100.0] * 26 + [95.0, 96.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
entry_ord = self.ORD + 25
|
||
cand = _sim_cand("AAA", entry_ord, entry=100.0, stop=80.0, target=130.0)
|
||
|
||
sim = bt._simulate_portfolio([cand], prices, None, "low20", 30)
|
||
|
||
assert sim is not None
|
||
assert sim["trades"] == 1
|
||
assert sim["exit_reasons"] == {"low20": 1}
|
||
assert sim["worst_trade_r"] == pytest.approx(-0.25)
|
||
|
||
def test_nothing_qualified_returns_none(self):
|
||
assert bt._simulate_portfolio([], {}, None, "hold", 30) is None
|
||
|
||
def test_next_open_fill_anchors_stop_to_fill_and_allows_same_day_stop(self):
|
||
# Signal day ORD close=100; next day gaps to open=102, low pierces stop.
|
||
# ATR on flat history is small; build a series with ATR ≈ 2.
|
||
n = 40
|
||
closes = [100.0] * n
|
||
highs = [102.0] * n
|
||
lows = [98.0] * n
|
||
opens = [100.0] * n
|
||
ords = list(range(self.ORD, self.ORD + n))
|
||
# Signal on last warm-up bar; fill bar is the next session.
|
||
signal_i = n - 2
|
||
fill_i = n - 1
|
||
opens[fill_i] = 102.0
|
||
highs[fill_i] = 103.0
|
||
lows[fill_i] = 90.0 # pierces fill − 1.5×ATR
|
||
closes[fill_i] = 91.0
|
||
prices = {
|
||
"AAA": (ords, opens, highs, lows, closes, [1_000_000] * n)
|
||
}
|
||
cand = _sim_cand(
|
||
"AAA",
|
||
self.ORD + signal_i,
|
||
entry=100.0,
|
||
stop=95.0,
|
||
target=130.0,
|
||
)
|
||
sim = bt._simulate_portfolio(
|
||
[cand],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
30,
|
||
fill_mode=bt.FILL_MODE_NEXT_OPEN,
|
||
cost_per_side=0.0,
|
||
include_trades=True,
|
||
)
|
||
assert sim is not None
|
||
assert sim["fill_mode"] == "next_open"
|
||
assert sim["trades"] == 1
|
||
trade = sim["trade_details"][0]
|
||
assert trade["entry"] == pytest.approx(102.0)
|
||
# Stop = 102 − 1.5×ATR; ATR on this series is 4 (high-low), so stop=96.
|
||
# Same-day low 90 → stop fill at 96 (not open).
|
||
assert trade["reason"] == "stop"
|
||
assert trade["initial_stop"] == pytest.approx(102.0 - 1.5 * 4.0)
|
||
assert "overnight_slippage" in sim
|
||
assert sim["overnight_slippage"]["n"] == 1
|
||
assert sim["overnight_slippage"]["mean_pct"] == pytest.approx(2.0)
|
||
|
||
def test_next_open_skips_when_fill_bar_missing(self):
|
||
closes = [100.0, 101.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
cand = _sim_cand("AAA", self.ORD + 1, entry=101.0, stop=96.0, target=120.0)
|
||
sim = bt._simulate_portfolio(
|
||
[cand], prices, None, "hold", 5, fill_mode=bt.FILL_MODE_NEXT_OPEN
|
||
)
|
||
# Signal on last bar → no t+1 open → no trade.
|
||
assert sim is None or sim["trades"] == 0 or sim.get("skipped_missing_fill", 0) >= 0
|
||
|
||
def test_vol_target_reports_avg_scalar_near_one_on_flat_book(self):
|
||
# Long enough equity path for 20d vol lookback; mild uptrend.
|
||
closes = [100.0 + i * 0.1 for i in range(80)]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
candidates = [
|
||
_sim_cand(
|
||
"AAA",
|
||
self.ORD + 10 + k * 5,
|
||
entry=closes[10 + k * 5],
|
||
stop=closes[10 + k * 5] - 5.0,
|
||
target=closes[10 + k * 5] + 20.0,
|
||
)
|
||
for k in range(8)
|
||
]
|
||
sim = bt._simulate_portfolio(
|
||
candidates,
|
||
prices,
|
||
None,
|
||
"hold",
|
||
4,
|
||
vol_target=0.20,
|
||
vol_lookback=20,
|
||
vol_clamp=(0.5, 1.5),
|
||
cost_per_side=0.0,
|
||
)
|
||
assert sim is not None
|
||
assert sim["vol_target"] == 0.20
|
||
assert sim["avg_vol_scalar"] is not None
|
||
assert 0.5 <= sim["avg_vol_scalar"] <= 1.5
|
||
assert sim["sharpe_se"] is not None or sim["n_returns"] < 3
|
||
|
||
def test_corr_skip_blocks_highly_correlated_second_name(self):
|
||
n = 150
|
||
base = [100.0]
|
||
for i in range(1, n):
|
||
base.append(base[-1] * (1.0 + 0.001 * ((-1) ** i)))
|
||
# BBB nearly identical path → corr ≈ 1.
|
||
prices = {
|
||
"AAA": _sim_prices(self.ORD, base),
|
||
"BBB": _sim_prices(self.ORD, [c * 1.01 for c in base]),
|
||
}
|
||
day = self.ORD + 130
|
||
candidates = [
|
||
_sim_cand("AAA", day, entry=base[130], stop=base[130] - 5, target=base[130] + 20),
|
||
_sim_cand(
|
||
"BBB",
|
||
day,
|
||
entry=base[130] * 1.01,
|
||
stop=base[130] * 1.01 - 5,
|
||
target=base[130] * 1.01 + 20,
|
||
mp=80.0,
|
||
),
|
||
]
|
||
# Rank AAA first.
|
||
candidates[0]["momentum_percentile"] = 99.0
|
||
candidates[0]["activation_momentum_percentile"] = 99.0
|
||
sim = bt._simulate_portfolio(
|
||
candidates,
|
||
prices,
|
||
None,
|
||
"hold",
|
||
5,
|
||
corr_max=0.5,
|
||
corr_action="skip",
|
||
corr_lookback=120,
|
||
corr_min_overlap=60,
|
||
cost_per_side=0.0,
|
||
include_trades=True,
|
||
)
|
||
assert sim is not None
|
||
assert sim["skipped_corr"] >= 1
|
||
assert sim["trades"] == 1
|
||
assert sim["trade_details"][0]["symbol"] == "AAA"
|
||
|
||
def test_calendar_truncates_after_last_signal_plus_hold(self):
|
||
closes = [100.0 + i for i in range(100)]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
cand = _sim_cand("AAA", self.ORD + 10, entry=110.0, stop=105.0, target=200.0)
|
||
sim = bt._simulate_portfolio(
|
||
[cand], prices, None, "hold", 5, cost_per_side=0.0, include_trades=True
|
||
)
|
||
assert sim is not None
|
||
end = date.fromisoformat(sim["end_date"])
|
||
entry = date.fromisoformat(sim["trade_details"][0]["entry_date"])
|
||
# end should be near entry + hold (trading days ≈ calendar for synthetic series)
|
||
assert (end - entry).days <= 10
|
||
|
||
def test_stale_close_fills_next_session_close_with_reanchored_stop(self):
|
||
n = 40
|
||
closes = [100.0 + 0.1 * i for i in range(n)]
|
||
opens = list(closes)
|
||
highs = [c + 2.0 for c in closes]
|
||
lows = [c - 2.0 for c in closes]
|
||
ords = list(range(self.ORD, self.ORD + n))
|
||
signal_i = n - 2
|
||
fill_i = n - 1
|
||
closes[fill_i] = 110.0
|
||
opens[fill_i] = 105.0
|
||
highs[fill_i] = 111.0
|
||
lows[fill_i] = 104.0
|
||
prices = {
|
||
"AAA": (ords, opens, highs, lows, closes, [1_000_000] * n)
|
||
}
|
||
cand = _sim_cand(
|
||
"AAA",
|
||
self.ORD + signal_i,
|
||
entry=closes[signal_i],
|
||
stop=closes[signal_i] - 5.0,
|
||
target=200.0,
|
||
)
|
||
sim = bt._simulate_portfolio(
|
||
[cand],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
5,
|
||
fill_mode=bt.FILL_MODE_STALE_CLOSE,
|
||
cost_per_side=0.0,
|
||
include_trades=True,
|
||
)
|
||
assert sim is not None
|
||
assert sim["fill_mode"] == "stale_close"
|
||
assert sim["trades"] == 1
|
||
trade = sim["trade_details"][0]
|
||
assert trade["entry"] == pytest.approx(110.0)
|
||
# ATR ~4 on this synthetic series → stop = 110 − 1.5×4 = 104
|
||
assert trade["initial_stop"] == pytest.approx(110.0 - 1.5 * 4.0, abs=0.5)
|
||
assert "signal_to_fill_drift" in sim
|
||
|
||
def test_next_open_gap_cap_skips_large_gap_ups(self):
|
||
n = 40
|
||
closes = [100.0] * n
|
||
opens = [100.0] * n
|
||
highs = [102.0] * n
|
||
lows = [98.0] * n
|
||
ords = list(range(self.ORD, self.ORD + n))
|
||
signal_i = n - 2
|
||
fill_i = n - 1
|
||
opens[fill_i] = 110.0 # +10% gap vs signal close 100
|
||
highs[fill_i] = 111.0
|
||
lows[fill_i] = 109.0
|
||
closes[fill_i] = 110.5
|
||
prices = {
|
||
"AAA": (ords, opens, highs, lows, closes, [1_000_000] * n)
|
||
}
|
||
cand = _sim_cand(
|
||
"AAA", self.ORD + signal_i, entry=100.0, stop=95.0, target=130.0
|
||
)
|
||
blocked = bt._simulate_portfolio(
|
||
[cand],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
5,
|
||
fill_mode=bt.FILL_MODE_NEXT_OPEN,
|
||
max_entry_gap_pct=0.02,
|
||
cost_per_side=0.0,
|
||
)
|
||
allowed = bt._simulate_portfolio(
|
||
[cand],
|
||
prices,
|
||
None,
|
||
"hold",
|
||
5,
|
||
fill_mode=bt.FILL_MODE_NEXT_OPEN,
|
||
cost_per_side=0.0,
|
||
include_trades=True,
|
||
)
|
||
assert blocked is None or blocked.get("trades", 0) == 0
|
||
if blocked is not None:
|
||
assert blocked.get("skipped_gap_cap", 0) >= 1
|
||
assert allowed is not None and allowed["trades"] == 1
|
||
assert allowed["trade_details"][0]["entry"] == pytest.approx(110.0)
|
||
|
||
|
||
def test_fip_id_sign_convention_steady_climber_vs_jump():
|
||
# Steady climber: many up days, continuous path → lower (more negative) ID.
|
||
steady = [100.0]
|
||
for _ in range(280):
|
||
steady.append(steady[-1] * 1.002)
|
||
# Jump then flat: one big up day, then zeros → higher ID (more discrete).
|
||
jumpy = [100.0] * 252
|
||
jumpy.append(100.0 * 1.5)
|
||
jumpy.extend([100.0 * 1.5] * 40)
|
||
i = 260
|
||
id_steady = bt._fip_id(steady, i)
|
||
id_jumpy = bt._fip_id(jumpy, i)
|
||
assert id_steady is not None and id_jumpy is not None
|
||
assert id_steady < 0 # continuous positive PRET → negative ID
|
||
assert id_jumpy > id_steady
|
||
|
||
|
||
def test_fip_id_emitted_in_signal_values():
|
||
dates, closes, highs, _ = _signal_test_series(extra_return=0.0005)
|
||
out = bt._signal_values(dates, closes, highs, 260)
|
||
assert "fip_id" in out
|
||
assert -1.0 <= out["fip_id"] <= 1.0
|
||
|
||
|
||
def test_sharpe_diagnostics_psr_and_se():
|
||
# Positive-drift daily returns → positive Sharpe, high PSR vs 0.
|
||
rets = [0.001 + 0.0001 * (i % 5) for i in range(300)]
|
||
diag = bt.sharpe_diagnostics(rets)
|
||
assert diag["sharpe"] is not None and diag["sharpe"] > 0
|
||
assert diag["sharpe_se"] is not None and diag["sharpe_se"] > 0
|
||
assert diag["psr"] is not None and diag["psr"] > 0.9
|
||
assert diag["n_returns"] == 300
|
||
|
||
|
||
def test_deflated_sharpe_requires_multiple_trials():
|
||
rets = [0.001 + 0.0005 * ((-1) ** i) for i in range(400)]
|
||
diag = bt.sharpe_diagnostics(rets)
|
||
assert diag["sharpe"] is not None and diag["sharpe_se"] is not None
|
||
assert bt.deflated_sharpe_ratio(
|
||
diag["sharpe"], diag["sharpe_se"], n_trials=1, n_returns=diag["n_returns"]
|
||
) is None
|
||
dsr = bt.deflated_sharpe_ratio(
|
||
diag["sharpe"],
|
||
diag["sharpe_se"],
|
||
n_trials=20,
|
||
n_returns=diag["n_returns"],
|
||
return_skew=diag["return_skew"],
|
||
return_kurtosis=diag["return_kurtosis"],
|
||
)
|
||
assert dsr is not None
|
||
assert 0.0 <= dsr <= 1.0
|
||
|
||
|
||
def test_bucket_stats_counts_and_expectancy():
|
||
cands = [
|
||
_cand(70, OUTCOME_TARGET_HIT, 3.0), # +3R win
|
||
_cand(60, OUTCOME_TARGET_HIT, 2.0), # +2R win
|
||
_cand(40, OUTCOME_STOP_HIT, 3.0), # -1R loss
|
||
_cand(30, OUTCOME_EXPIRED, 3.0), # 0R expired
|
||
]
|
||
s = bt._bucket_stats(cands)
|
||
assert s["total"] == 4
|
||
assert s["wins"] == 2
|
||
assert s["losses"] == 1
|
||
assert s["expired"] == 1
|
||
# hit rate is over decided (wins+losses) only
|
||
assert s["hit_rate"] == round(2 / 3 * 100, 1)
|
||
# avg R = (3 + 2 - 1 + 0) / 4 = 1.0
|
||
assert s["avg_r"] == 1.0
|
||
assert s["total_r"] == 4.0
|
||
# net = gross minus a 0.04R round trip per candidate (risk_pct 0.05)
|
||
assert s["net_avg_r"] == pytest.approx(1.0 - _COST_R_005, abs=0.001)
|
||
assert s["net_total_r"] == pytest.approx(4.0 - 4 * _COST_R_005, abs=0.01)
|
||
assert s["best_r"] == 3.0
|
||
assert s["worst_r"] == -1.0
|
||
assert s["avg_hold_days"] == 10.0
|
||
assert s["net_r_per_day"] == pytest.approx((1.0 - _COST_R_005) / 10.0, abs=0.001)
|
||
# robustness: net rs are [2.96, 1.96, -1.04, -0.04]
|
||
assert s["median_net_r"] == pytest.approx(0.96, abs=0.001)
|
||
assert s["profit_factor"] == pytest.approx(4.92 / 1.08, abs=0.01)
|
||
# ex-top-5%: ceil(4 * 0.05) = 1 winner trimmed → mean of the remaining three
|
||
assert s["net_avg_r_ex_top5"] == pytest.approx((1.96 - 1.04 - 0.04) / 3, abs=0.001)
|
||
|
||
|
||
def test_bucket_stats_empty():
|
||
s = bt._bucket_stats([])
|
||
assert s["total"] == 0
|
||
assert s["hit_rate"] is None
|
||
assert s["avg_r"] is None
|
||
assert s["net_avg_r"] is None
|
||
|
||
|
||
def test_bucket_stats_no_risk_pct_means_no_cost():
|
||
c = _cand(50, OUTCOME_TARGET_HIT, 2.0)
|
||
del c["risk_pct"]
|
||
s = bt._bucket_stats([c])
|
||
assert s["net_avg_r"] == s["avg_r"]
|
||
assert s["net_total_r"] == s["total_r"]
|
||
|
||
|
||
def test_build_recommendation_reads_the_report():
|
||
report = {
|
||
"overall_qualified": {"net_avg_r": 0.13, "net_avg_r_ex_top5": 0.05},
|
||
"time_exit_sweep": [
|
||
{"hold_days": 21, "net_avg_r": 0.38},
|
||
{"hold_days": 30, "net_avg_r": 0.50, "net_avg_r_ex_top5": 0.21},
|
||
],
|
||
"gate_ablation": [
|
||
{"variant": "all_floors", "total": 100, "hold_net_avg_r": 0.50},
|
||
{"variant": "no_confidence_floor", "total": 130, "hold_net_avg_r": 0.49},
|
||
{"variant": "no_rr_floor", "total": 400, "hold_net_avg_r": 0.34},
|
||
{"variant": "no_neutral_exclusion", "total": 120, "hold_net_avg_r": 0.46},
|
||
],
|
||
"sweep": [
|
||
{"min_momentum_percentile": 80.0, "net_avg_r": 0.13, "total": 100},
|
||
{"min_momentum_percentile": 60.0, "net_avg_r": 0.05, "total": 300},
|
||
{"min_momentum_percentile": 0.0, "net_avg_r": -0.12, "total": 1000},
|
||
],
|
||
# Legacy policy book. Its numbers are deliberately DIFFERENT from the
|
||
# production monitor's below, so sourcing the benchmark line from here
|
||
# again would fail the assertion rather than pass unnoticed.
|
||
"portfolio_sim": {"policies": [
|
||
{"policy": "target", "cagr_pct": 23.7, "total_return_pct": 134.8,
|
||
"spy_return_pct": 95.9, "max_drawdown_pct": 20.7},
|
||
{"policy": "hold", "cagr_pct": 31.9, "total_return_pct": 203.6,
|
||
"spy_return_pct": 95.9, "max_drawdown_pct": 21.2},
|
||
]},
|
||
"portfolio_monitor": {
|
||
"production_strategy": "prod",
|
||
"runs": [{
|
||
"strategy": "prod", "lookback": "all", "lookback_label": "All history",
|
||
"cagr_pct": 40.0, "sharpe": 1.72, "max_drawdown_pct": 17.7,
|
||
"total_return_pct": 297.8, "spy_return_pct": 101.9,
|
||
}, {
|
||
# A second window with DIFFERENT numbers. Without it the "all"
|
||
# preference is untested and a lookback mix-up cannot fail.
|
||
"strategy": "prod", "lookback": "3y", "lookback_label": "3y",
|
||
"cagr_pct": 47.9, "sharpe": 1.96, "max_drawdown_pct": 17.3,
|
||
"total_return_pct": 220.9, "spy_return_pct": 71.5,
|
||
}],
|
||
},
|
||
}
|
||
rec = bt._build_recommendation(report)
|
||
by_topic: dict[str, list[str]] = {}
|
||
for item in rec["items"]:
|
||
by_topic.setdefault(item["topic"], []).append(item["text"])
|
||
|
||
assert rec["headline"] is not None and "Production baseline" in rec["headline"]
|
||
# The hold-vs-target comparison is gone: both are exits the production book
|
||
# replaced, so a recommendation between them cannot lead to an action.
|
||
assert "exit" not in by_topic
|
||
# Benchmark must quote the SAME row the page's tiles show, not the policy sim.
|
||
assert "+297.8%" in by_topic["benchmark"][0]
|
||
assert "203.6" not in by_topic["benchmark"][0]
|
||
gate_texts = " | ".join(by_topic["gate"])
|
||
assert "confidence floor adds nothing" in gate_texts
|
||
assert "keep the R:R floor" in gate_texts
|
||
assert "keep the NEUTRAL exclusion" in gate_texts
|
||
assert "80" in by_topic["cutoff"][0]
|
||
assert "beats" in by_topic["benchmark"][0]
|
||
|
||
# Every production figure comes from ONE window, and the report says which,
|
||
# so the page can default its selector to the same one.
|
||
assert rec["basis_lookback"] == "all"
|
||
assert rec["basis_lookback_label"] == "All history"
|
||
assert "+40.0%" in by_topic["production"][0]
|
||
assert "47.9" not in by_topic["production"][0] # the 3y row must not leak in
|
||
assert "220.9" not in by_topic["benchmark"][0]
|
||
|
||
# Robustness names its real basis. It used to claim "under the recommended
|
||
# 30d hold" — nothing recommends that exit; production is the ATR trail.
|
||
robustness = by_topic["robustness"][0]
|
||
assert "not a handful of outliers" in robustness
|
||
assert "gate-level grading" in robustness
|
||
assert "recommended" not in robustness
|
||
|
||
|
||
def test_build_recommendation_flags_outlier_dependence():
|
||
rec = bt._build_recommendation({
|
||
"overall_qualified": {"net_avg_r": 0.13, "net_avg_r_ex_top5": -0.02},
|
||
})
|
||
robustness = [i["text"] for i in rec["items"] if i["topic"] == "robustness"]
|
||
assert robustness and "WARNING" in robustness[0]
|
||
|
||
|
||
def test_build_recommendation_prefers_production_monitor_headline():
|
||
rec = bt._build_recommendation({
|
||
"portfolio_monitor": {
|
||
"production_strategy": bt.PRODUCTION_PORTFOLIO_STRATEGY,
|
||
"runs": [{
|
||
"strategy": bt.PRODUCTION_PORTFOLIO_STRATEGY,
|
||
"lookback": "all",
|
||
"lookback_label": "All history",
|
||
"cagr_pct": 44.4,
|
||
"sharpe": 1.72,
|
||
"max_drawdown_pct": 23.8,
|
||
}],
|
||
},
|
||
"overall_qualified": {},
|
||
})
|
||
assert rec["headline"] is not None
|
||
assert "3x ATR trailing exit" in rec["headline"]
|
||
assert "after the gate fails" in rec["headline"]
|
||
assert any(item["topic"] == "production" for item in rec["items"])
|
||
|
||
|
||
def test_window_setups_too_short_returns_empty():
|
||
assert bt._window_setups([], {}, {}) == []
|
||
|
||
|
||
def test_backtest_target_model_is_small_and_validated():
|
||
assert bt.validate_backtest_target_model(" PRODUCTION_GTL ") == "production_gtl"
|
||
assert bt.validate_backtest_target_model("structural_sr") == "structural_sr"
|
||
with pytest.raises(ValueError, match="Unknown backtest target model"):
|
||
bt.validate_backtest_target_model("legacy_range_grid_touch")
|
||
|
||
|
||
def test_backtest_cadence_is_small_validated_and_session_based():
|
||
assert bt.validate_backtest_cadence(" WEEKLY ") == "weekly"
|
||
assert bt.validate_backtest_cadence("daily") == "daily"
|
||
assert bt.backtest_step_sessions("weekly") == 5
|
||
assert bt.backtest_step_sessions("daily") == 1
|
||
with pytest.raises(ValueError, match="Unknown backtest cadence"):
|
||
bt.validate_backtest_cadence("monthly")
|
||
|
||
|
||
def _flat_window_records():
|
||
return [
|
||
SimpleNamespace(
|
||
date=date(2024, 1, 1) + timedelta(days=i),
|
||
open=100.0,
|
||
high=101.0,
|
||
low=99.0,
|
||
close=100.0,
|
||
volume=1_000_000,
|
||
)
|
||
for i in range(bt.MIN_LOOKBACK)
|
||
]
|
||
|
||
|
||
def test_window_setups_routes_production_gtl_by_default(monkeypatch):
|
||
captured = {}
|
||
|
||
def fake_detector(highs, lows, closes):
|
||
captured.update({"highs": highs, "lows": lows, "closes": closes})
|
||
return []
|
||
|
||
monkeypatch.setattr(bt, "detect_gate_target_ladder", fake_detector)
|
||
assert bt._window_setups(_flat_window_records(), {}, {}) == []
|
||
assert captured == {
|
||
"highs": [101.0] * bt.MIN_LOOKBACK,
|
||
"lows": [99.0] * bt.MIN_LOOKBACK,
|
||
"closes": [100.0] * bt.MIN_LOOKBACK,
|
||
}
|
||
|
||
|
||
def test_window_setups_routes_structural_comparison(monkeypatch):
|
||
captured = {}
|
||
|
||
def fake_detector(highs, lows, closes, volumes):
|
||
captured.update({
|
||
"highs": highs,
|
||
"lows": lows,
|
||
"closes": closes,
|
||
"volumes": volumes,
|
||
})
|
||
return []
|
||
|
||
monkeypatch.setattr(bt, "detect_sr_levels", fake_detector)
|
||
assert bt._window_setups(
|
||
_flat_window_records(),
|
||
{},
|
||
{},
|
||
target_model=bt.STRUCTURAL_SR_TARGET_MODEL,
|
||
) == []
|
||
assert captured == {
|
||
"highs": [101.0] * bt.MIN_LOOKBACK,
|
||
"lows": [99.0] * bt.MIN_LOOKBACK,
|
||
"closes": [100.0] * bt.MIN_LOOKBACK,
|
||
"volumes": [1_000_000] * bt.MIN_LOOKBACK,
|
||
}
|
||
|
||
|
||
def test_window_setups_rejects_removed_research_arm():
|
||
with pytest.raises(ValueError, match="Unknown backtest target model"):
|
||
bt._window_setups(
|
||
_flat_window_records(),
|
||
{},
|
||
{},
|
||
target_model="production_control",
|
||
)
|
||
|
||
|
||
def test_replay_ticker_candidates_carry_gate_fields():
|
||
"""The ablation recomputes floors from candidate fields — a candidate missing
|
||
action/risk_level silently zeroes the ablation rows (July 2026 regression)."""
|
||
from app.services.admin_service import ACTIVATION_DEFAULTS
|
||
from app.services.recommendation_service import DEFAULT_RECOMMENDATION_CONFIG
|
||
|
||
base = date(2025, 1, 1)
|
||
bars = []
|
||
for i in range(160):
|
||
close = 100.0 + 8.0 * math.sin(i / 6.0)
|
||
bars.append(SimpleNamespace(
|
||
date=base + timedelta(days=i),
|
||
open=close,
|
||
high=close + 1.5,
|
||
low=close - 1.5,
|
||
close=close,
|
||
volume=1_000_000 + (i % 5) * 1000,
|
||
))
|
||
cands = bt._replay_ticker(
|
||
"OSC", bars, dict(DEFAULT_RECOMMENDATION_CONFIG), dict(ACTIVATION_DEFAULTS)
|
||
)
|
||
assert cands, "expected the oscillating series to produce candidates"
|
||
for c in cands:
|
||
assert c.get("action") is not None
|
||
assert "risk_level" in c
|
||
assert c["target_model"] == bt.PRODUCTION_GTL_TARGET_MODEL
|
||
assert c["ranking_period"][0] == "week"
|
||
|
||
daily_cands = bt._replay_ticker(
|
||
"OSC",
|
||
bars,
|
||
dict(DEFAULT_RECOMMENDATION_CONFIG),
|
||
dict(ACTIVATION_DEFAULTS),
|
||
cadence="daily",
|
||
)
|
||
assert len(daily_cands) > len(cands)
|
||
assert all(c["ranking_period"][0] == "date" for c in daily_cands)
|
||
|
||
|
||
def test_slim_replay_can_retain_shorts_for_ranking_universe(monkeypatch):
|
||
setup = {
|
||
"entry": 100.0,
|
||
"stop": 95.0,
|
||
"target": 110.0,
|
||
"rr": 2.0,
|
||
"confidence": 80.0,
|
||
"primary_prob": 0.6,
|
||
"best_prob": 0.7,
|
||
"momentum": 0.1,
|
||
"meets_core": True,
|
||
"action": "BUY_MODERATE",
|
||
"risk_level": "MEDIUM",
|
||
}
|
||
monkeypatch.setattr(
|
||
bt,
|
||
"_window_setups",
|
||
lambda *_args, **_kwargs: [
|
||
{**setup, "direction": "long"},
|
||
{**setup, "direction": "short", "stop": 105.0, "target": 90.0},
|
||
],
|
||
)
|
||
count = bt.MIN_LOOKBACK + bt.HORIZON
|
||
first_ord = date(2025, 1, 1).toordinal()
|
||
columns = (
|
||
list(range(first_ord, first_ord + count)),
|
||
[100.0] * count,
|
||
[101.0] * count,
|
||
[99.0] * count,
|
||
[100.0] * count,
|
||
[1_000_000] * count,
|
||
)
|
||
|
||
long_only = bt._replay_candidates_for_period(
|
||
"AAA", columns, {}, {}, None, date.min, "daily"
|
||
)
|
||
full_ranking_universe = bt._replay_candidates_for_period(
|
||
"AAA", columns, {}, {}, None, date.min, "daily", True
|
||
)
|
||
dual_ranking_replay = bt._replay_candidates_for_period(
|
||
"AAA", columns, {}, {}, None, date.min, "daily", True, True
|
||
)
|
||
|
||
assert [row["direction"] for row in long_only] == ["long"]
|
||
assert {row["direction"] for row in full_ranking_universe} == {
|
||
"long",
|
||
"short",
|
||
}
|
||
assert len(dual_ranking_replay) == 2
|
||
assert sum(
|
||
bool(row.get("_universe_rank_observation"))
|
||
for row in dual_ranking_replay
|
||
) == 1
|
||
|
||
monkeypatch.setattr(bt, "_window_setups", lambda *_args, **_kwargs: [])
|
||
rank_only = bt._replay_candidates_for_period(
|
||
"AAA", columns, {}, {}, None, date.min, "daily", True, True
|
||
)
|
||
assert len(rank_only) == 1
|
||
assert rank_only[0]["direction"] == "rank_only"
|
||
assert rank_only[0]["_rank_only"] is True
|
||
assert rank_only[0]["_universe_rank_observation"] is True
|
||
|
||
|
||
def test_daily_replay_uses_exact_date_ranking_periods():
|
||
candidates = [
|
||
{
|
||
"iso_week": (2026, 1),
|
||
"ranking_period": ("date", date(2026, 1, 5).toordinal()),
|
||
"momentum": 0.10,
|
||
},
|
||
{
|
||
"iso_week": (2026, 1),
|
||
"ranking_period": ("date", date(2026, 1, 5).toordinal()),
|
||
"momentum": 0.20,
|
||
},
|
||
{
|
||
"iso_week": (2026, 1),
|
||
"ranking_period": ("date", date(2026, 1, 6).toordinal()),
|
||
"momentum": 0.90,
|
||
},
|
||
{
|
||
"iso_week": (2026, 1),
|
||
"ranking_period": ("date", date(2026, 1, 6).toordinal()),
|
||
"momentum": 0.30,
|
||
},
|
||
]
|
||
|
||
bt._assign_momentum_percentiles(candidates)
|
||
|
||
assert [row["momentum_percentile"] for row in candidates] == [0.0, 100.0, 100.0, 0.0]
|
||
|
||
|
||
async def _seed_oscillating_ticker(session, symbol: str, n: int = 160) -> None:
|
||
t = Ticker(symbol=symbol)
|
||
session.add(t)
|
||
await session.flush()
|
||
base = date(2025, 1, 1)
|
||
for i in range(n):
|
||
close = 100.0 + 8.0 * math.sin(i / 6.0)
|
||
session.add(OHLCVRecord(
|
||
ticker_id=t.id,
|
||
date=base + timedelta(days=i),
|
||
open=close,
|
||
high=close + 1.5,
|
||
low=close - 1.5,
|
||
close=close,
|
||
volume=1_000_000 + (i % 5) * 1000,
|
||
))
|
||
await session.commit()
|
||
|
||
|
||
async def test_run_backtest_smoke(session):
|
||
await _seed_oscillating_ticker(session, "OSC")
|
||
report = await bt.run_backtest(session)
|
||
|
||
# well-formed report
|
||
assert report["tickers"] == 1
|
||
assert isinstance(report["candidates"], int)
|
||
for key in (
|
||
"overall_qualified", "overall_all", "by_direction", "sweep",
|
||
"gate_ablation", "time_exit_sweep", "portfolio_sim", "strategy_variants",
|
||
"exit_policy_variants", "portfolio_monitor", "recommendation", "research_recommendation",
|
||
):
|
||
assert key in report
|
||
# the oscillating series should yield at least some resolved setups
|
||
assert report["candidates"] >= 1
|
||
|
||
# cost assumption is reported, and every bucket carries net numbers
|
||
assert report["params"]["cost_per_side_pct"] == pytest.approx(bt.COST_PER_SIDE * 100)
|
||
assert report["params"]["target_model"] == bt.PRODUCTION_GTL_TARGET_MODEL
|
||
assert report["params"]["is_production_target_model"] is True
|
||
assert report["params"]["entry_cadence"] == "weekly"
|
||
assert report["params"]["step_sessions"] == 5
|
||
assert report["params"]["production_reentry_policy"] == "gate_reset"
|
||
assert "net_avg_r" in report["overall_all"]
|
||
|
||
# ablation baseline reproduces the qualified set exactly, and every row
|
||
# carries the hold-to-horizon grading alongside the target model
|
||
ablation = {r["variant"]: r for r in report["gate_ablation"]}
|
||
assert ablation["all_floors"]["total"] == report["overall_qualified"]["total"]
|
||
|
||
daily_report = await bt.run_backtest(session, cadence="daily")
|
||
assert daily_report["params"]["entry_cadence"] == "daily"
|
||
assert daily_report["params"]["step_sessions"] == 1
|
||
assert daily_report["candidates"] > report["candidates"]
|
||
for row in report["gate_ablation"]:
|
||
assert "hold_net_avg_r" in row
|
||
|
||
# time-exit sweep covers the configured hold lengths
|
||
assert [r["hold_days"] for r in report["time_exit_sweep"]] == list(bt.TIME_EXIT_DAYS)
|
||
|
||
# portfolio simulation section is always present (policies may be empty
|
||
# when nothing qualifies)
|
||
assert "portfolio_sim" in report
|
||
assert isinstance(report["portfolio_sim"]["policies"], list)
|
||
assert report["portfolio_sim"]["params"]["max_positions"] == bt.SIM_MAX_POSITIONS
|
||
assert isinstance(report["strategy_variants"]["variants"], list)
|
||
assert isinstance(report["exit_policy_variants"]["variants"], list)
|
||
assert report["portfolio_monitor"] is None or isinstance(report["portfolio_monitor"]["runs"], list)
|
||
|
||
# sweep: lowering the momentum-percentile cutoff can only add qualifiers
|
||
sweep = sorted(report["sweep"], key=lambda r: r["min_momentum_percentile"], reverse=True)
|
||
counts = [r["total"] for r in sweep]
|
||
assert counts == sorted(counts) # ascending as threshold descends
|
||
|
||
|
||
async def test_run_backtest_rolls_back_a_failed_ticker_fetch(session, monkeypatch):
|
||
"""A failed per-ticker read must not leave the session mid-failed-transaction.
|
||
|
||
Every DB call in the replay loop is best-effort, but swallowing the error
|
||
without a rollback leaves asyncpg in "current transaction is aborted": every
|
||
later statement fails the same way until the first unguarded one — the report
|
||
write — surfaces it as the job error, long after the real cause.
|
||
"""
|
||
await _seed_oscillating_ticker(session, "AAA")
|
||
await _seed_oscillating_ticker(session, "OSC")
|
||
|
||
real_fetch = bt._fetch_columns
|
||
rolled_back: list[str] = []
|
||
|
||
async def failing_fetch(db, symbol):
|
||
if symbol == "AAA":
|
||
raise RuntimeError("simulated OHLCV read failure")
|
||
return await real_fetch(db, symbol)
|
||
|
||
real_rollback = session.rollback
|
||
|
||
async def tracking_rollback():
|
||
rolled_back.append("x")
|
||
await real_rollback()
|
||
|
||
monkeypatch.setattr(bt, "_fetch_columns", failing_fetch)
|
||
monkeypatch.setattr(session, "rollback", tracking_rollback)
|
||
|
||
report = await bt.run_backtest(session)
|
||
|
||
assert rolled_back, "a failed ticker fetch left the session un-rolled-back"
|
||
# the surviving ticker is still replayed after the rollback
|
||
assert report["tickers"] == 2
|
||
assert report["candidates"] >= 1
|
||
|
||
|
||
async def test_run_backtest_rolls_back_a_failed_portfolio_sim_load(session, monkeypatch):
|
||
"""The portfolio-sim block loads the benchmark and the live exit policy from
|
||
the same session, well after the replay loop. A failure there poisons the
|
||
transaction exactly as one in the loop does, and the report write pays for it.
|
||
"""
|
||
await _seed_oscillating_ticker(session, "OSC")
|
||
|
||
rolled_back: list[str] = []
|
||
called: list[str] = []
|
||
|
||
async def failing_exit_policy(db):
|
||
called.append("x")
|
||
raise RuntimeError("simulated exit-policy read failure")
|
||
|
||
real_rollback = session.rollback
|
||
|
||
async def tracking_rollback():
|
||
rolled_back.append("x")
|
||
await real_rollback()
|
||
|
||
monkeypatch.setattr(
|
||
"app.services.paper_trade_service.get_exit_policy", failing_exit_policy
|
||
)
|
||
monkeypatch.setattr(session, "rollback", tracking_rollback)
|
||
|
||
report = await bt.run_backtest(session)
|
||
|
||
assert called, "the portfolio-sim block never ran; test proves nothing"
|
||
assert rolled_back, "a failed portfolio-sim load left the session un-rolled-back"
|
||
assert report["tickers"] == 1
|
||
|
||
|
||
class TestPortfolioQualityMetrics:
|
||
"""Sortino / Gain-to-Pain / dollar profit factor.
|
||
|
||
Each derives its expectation from the returned ``equity_curve`` rather than
|
||
hand-tracing position sizing, and each also asserts the *wrong* variant is
|
||
NOT what came back — the denominator and the numerator are exactly where
|
||
these ratios are usually got wrong.
|
||
"""
|
||
|
||
ORD = date(2025, 1, 6).toordinal()
|
||
|
||
@staticmethod
|
||
def _daily_returns(sim: dict) -> list[float]:
|
||
eq = [row["equity"] for row in sim["equity_curve"]]
|
||
return [b / a - 1.0 for a, b in zip(eq, eq[1:]) if a > 0]
|
||
|
||
@staticmethod
|
||
def _monthly_returns(sim: dict) -> list[float]:
|
||
monthly: list[float] = []
|
||
rows = sim["equity_curve"]
|
||
start = last = rows[0]["equity"]
|
||
cur = date.fromisoformat(rows[0]["date"]).replace(day=1)
|
||
for row in rows:
|
||
m = date.fromisoformat(row["date"]).replace(day=1)
|
||
if m != cur:
|
||
monthly.append(last / start - 1.0)
|
||
cur, start = m, last
|
||
last = row["equity"]
|
||
monthly.append(last / start - 1.0)
|
||
return monthly
|
||
|
||
def _wobbly_sim(self) -> dict:
|
||
"""~70 sessions crossing four month boundaries with a real mid drawdown,
|
||
so monthly returns include both signs (a short fixture yields one month
|
||
and zero pain, which reads as a broken formula)."""
|
||
closes = (
|
||
[100.0 + i for i in range(20)] # climb
|
||
+ [120.0 - 1.5 * i for i in range(20)] # drawdown
|
||
+ [90.0 + 1.2 * i for i in range(30)] # recovery
|
||
)
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
cand = _sim_cand("AAA", self.ORD, entry=100.0, stop=80.0, target=400.0)
|
||
sim = bt._simulate_portfolio(
|
||
[cand], prices, None, "hold", 65, include_curve=True
|
||
)
|
||
assert sim is not None
|
||
return sim
|
||
|
||
def test_sortino_denominator_is_full_sample_not_downside_count(self):
|
||
sim = self._wobbly_sim()
|
||
rets = self._daily_returns(sim)
|
||
downside = [r for r in rets if r < 0.0]
|
||
assert downside, "fixture must produce down days or the test proves nothing"
|
||
|
||
mean_ret = sum(rets) / len(rets)
|
||
correct = mean_ret / math.sqrt(
|
||
sum(r * r for r in downside) / len(rets)
|
||
) * math.sqrt(252.0)
|
||
# The classic error: dividing by the count of down days shrinks the
|
||
# denominator and inflates the ratio.
|
||
inflated = mean_ret / math.sqrt(
|
||
sum(r * r for r in downside) / len(downside)
|
||
) * math.sqrt(252.0)
|
||
|
||
assert sim["sortino"] == pytest.approx(round(correct, 2), abs=0.01)
|
||
assert sim["sortino"] != pytest.approx(round(inflated, 2), abs=0.01)
|
||
|
||
def test_gain_to_pain_is_schwager_on_monthly_returns(self):
|
||
sim = self._wobbly_sim()
|
||
monthly = self._monthly_returns(sim)
|
||
assert len(monthly) >= 3, "fixture must span several months"
|
||
pain = -sum(r for r in monthly if r < 0.0)
|
||
assert pain > 0, "fixture must have a losing month or pain is zero"
|
||
|
||
schwager = sum(monthly) / pain
|
||
# sum(all) = sum(pos) - |sum(neg)|, so the profit-factor-shaped variant
|
||
# sits exactly 1.0 higher for every input.
|
||
profit_factor_shaped = sum(r for r in monthly if r > 0.0) / pain
|
||
assert profit_factor_shaped == pytest.approx(schwager + 1.0, abs=1e-9)
|
||
|
||
assert sim["gain_to_pain"] == pytest.approx(round(schwager, 2), abs=0.01)
|
||
assert sim["gain_to_pain"] != pytest.approx(
|
||
round(profit_factor_shaped, 2), abs=0.01
|
||
)
|
||
|
||
def test_profit_factor_is_dollar_based(self):
|
||
"""One winner, one loser, on separate symbols so both fill."""
|
||
up = [100.0 + 2.0 * i for i in range(8)]
|
||
down = [100.0 - 2.0 * i for i in range(8)]
|
||
prices = {
|
||
"WIN": _sim_prices(self.ORD, up),
|
||
"LOSE": _sim_prices(self.ORD, down),
|
||
}
|
||
cands = [
|
||
_sim_cand("WIN", self.ORD, entry=100.0, stop=90.0, target=400.0, mp=95.0),
|
||
_sim_cand("LOSE", self.ORD, entry=100.0, stop=80.0, target=400.0, mp=94.0),
|
||
]
|
||
sim = bt._simulate_portfolio([cands[0], cands[1]], prices, None, "hold", 5)
|
||
assert sim is not None
|
||
assert sim["trades"] == 2
|
||
# With exactly two trades the reported best/worst ARE the win and the loss.
|
||
gross_win = sim["best_trade_pnl"]
|
||
gross_loss = -sim["worst_trade_pnl"]
|
||
assert gross_win > 0 and gross_loss > 0, "fixture must produce one of each"
|
||
assert sim["profit_factor"] == pytest.approx(
|
||
round(gross_win / gross_loss, 2), abs=0.01
|
||
)
|
||
|
||
def test_keys_always_present_and_no_downside_is_none(self):
|
||
"""Monotonic rise: no down days. Sortino must be None, never inf — and
|
||
all three keys must still be emitted, because the UI reads an ABSENT key
|
||
as 'report predates these metrics'."""
|
||
closes = [100.0, 102.0, 104.0, 106.0, 108.0, 110.0]
|
||
prices = {"AAA": _sim_prices(self.ORD, closes)}
|
||
cand = _sim_cand("AAA", self.ORD, entry=100.0, stop=95.0, target=130.0)
|
||
sim = bt._simulate_portfolio([cand], prices, None, "hold", 3)
|
||
assert sim is not None
|
||
for key in ("sortino", "gain_to_pain", "profit_factor"):
|
||
assert key in sim
|
||
assert sim["sortino"] is None
|
||
|
||
def test_build_recommendation_states_no_baseline_without_a_production_row():
|
||
"""A report with no portfolio monitor cannot describe the production book.
|
||
It used to fall back to recommending the fixed-hold exit — advice for a model
|
||
the production book had already replaced."""
|
||
report = {
|
||
"overall_qualified": {"net_avg_r": 0.13, "net_avg_r_ex_top5": 0.05},
|
||
"time_exit_sweep": [{"hold_days": 30, "net_avg_r": 0.50, "net_avg_r_ex_top5": 0.21}],
|
||
"portfolio_sim": {"policies": [
|
||
{"policy": "hold", "cagr_pct": 31.9, "total_return_pct": 203.6,
|
||
"spy_return_pct": 95.9, "max_drawdown_pct": 21.2},
|
||
]},
|
||
}
|
||
rec = bt._build_recommendation(report)
|
||
topics = {item["topic"] for item in rec["items"]}
|
||
|
||
assert rec["headline"] is None
|
||
# Nothing may be sourced from the legacy policy book.
|
||
assert "benchmark" not in topics
|
||
assert "exit" not in topics
|
||
|
||
|
||
async def test_cached_report_recommendation_is_rebuilt_on_read(session):
|
||
"""A report cached by an older build carries that build's recommendation.
|
||
|
||
Served verbatim, the page would show the legacy wording and no
|
||
basis_lookback — which let the lookback selector default elsewhere, putting
|
||
3y tiles beside an all-history recommendation with no warning. This is the
|
||
shape of the report sitting in production right now.
|
||
"""
|
||
from app.services.admin_service import update_setting
|
||
|
||
stale = {
|
||
"generated_at": "2026-08-12T05:00:00+00:00",
|
||
"tickers": 512, "candidates": 100, "qualified": 10,
|
||
"params": {"horizon_days": 30},
|
||
"overall_qualified": {"net_avg_r": 0.13, "net_avg_r_ex_top5": 0.20},
|
||
"portfolio_sim": {"policies": [
|
||
{"policy": "hold", "cagr_pct": 31.9, "total_return_pct": 175.0,
|
||
"spy_return_pct": 101.9, "max_drawdown_pct": 23.7},
|
||
]},
|
||
"portfolio_monitor": {
|
||
"production_strategy": "prod",
|
||
"runs": [{
|
||
"strategy": "prod", "lookback": "all", "lookback_label": "All history",
|
||
"cagr_pct": 40.0, "sharpe": 1.72, "max_drawdown_pct": 17.7,
|
||
"total_return_pct": 297.8, "spy_return_pct": 101.9,
|
||
}],
|
||
},
|
||
# What the old build stored: sourced from the policy book, and naming an
|
||
# exit the production book replaced.
|
||
"recommendation": {
|
||
"headline": "Trade the qualified list long-only; hold 30 trading days.",
|
||
"items": [
|
||
{"topic": "benchmark", "text": "Book vs SPY: beats buy-and-hold by "
|
||
"+73.1 points (+175.0% vs +101.9%)."},
|
||
{"topic": "robustness", "text": "Robustness: expectancy survives removing "
|
||
"the top 5% of winners (+0.20R net/trade "
|
||
"under the recommended 30d hold)."},
|
||
],
|
||
"note": "stale",
|
||
},
|
||
}
|
||
await update_setting(session, bt.KEY_REPORT, json.dumps(stale))
|
||
|
||
report = await bt.get_backtest_report(session)
|
||
assert report is not None
|
||
rec = report["recommendation"]
|
||
|
||
# Rebuilt: the basis is published, so the page cannot default elsewhere.
|
||
assert rec["basis_lookback"] == "all"
|
||
texts = " | ".join(i["text"] for i in rec["items"])
|
||
# ...and it quotes the production book, not the policy sim it used to.
|
||
assert "+297.8%" in texts and "175.0" not in texts
|
||
assert "recommended 30d hold" not in texts
|
||
assert "Production baseline" in (rec["headline"] or "")
|