research: sector residual, earnings gap/SUE, history-depth scaffolding
Tier-1 alpha research (local only, no production deploy): Sector residual momentum: two-factor SPY+sector residual and sector demean signals, IC harness + A/B. Sector resid clears pre-registered bars narrowly (PROMOTE for human wire design only). Sector demean fails t vs market resid. Earnings: earnings_events backfill (FMP bulk paid; FMP/AV per-symbol), 2a gap diagnostic report-only, 2b SUE IC (PARK; incomplete 48/506 coverage). History-depth: pre-registered doc + runner for MacBook deep rebuild/harness. Do not ship production residual or filters from this branch.
This commit is contained in:
@@ -791,31 +791,86 @@ def _residual_momentum_12_1(
|
|||||||
with an intercept estimated over the same window, the arithmetic residuals
|
with an intercept estimated over the same window, the arithmetic residuals
|
||||||
sum to ~zero by construction, which would destroy the signal.
|
sum to ~zero by construction, which would destroy the signal.
|
||||||
"""
|
"""
|
||||||
if not benchmark_closes or i - 252 < 0:
|
return _multi_factor_residual_momentum_12_1(
|
||||||
|
dates, closes, i, [benchmark_closes] if benchmark_closes else None
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _multi_factor_residual_momentum_12_1(
|
||||||
|
dates: list[date],
|
||||||
|
closes: list[float],
|
||||||
|
i: int,
|
||||||
|
factor_closes: list[dict[date, float]] | None,
|
||||||
|
) -> float | None:
|
||||||
|
"""12-1 residual momentum vs one or more factors (OLS, no intercept).
|
||||||
|
|
||||||
|
Same formation window as raw / single-factor residual momentum:
|
||||||
|
daily returns from close[i-252] → close[i-21], require ≥100 paired obs.
|
||||||
|
Factors are stacked as columns; betas are OLS without intercept so the
|
||||||
|
cumulative residual is not forced to zero.
|
||||||
|
"""
|
||||||
|
if not factor_closes or i - 252 < 0:
|
||||||
|
return None
|
||||||
|
n_factors = len(factor_closes)
|
||||||
|
if n_factors < 1:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
stock_rets: list[float] = []
|
stock_rets: list[float] = []
|
||||||
market_rets: list[float] = []
|
factor_rets: list[list[float]] = [[] for _ in range(n_factors)]
|
||||||
# Same daily intervals as mom_12_1: close[i-252] -> close[i-21].
|
|
||||||
for k in range(i - 251, i - 20):
|
for k in range(i - 251, i - 20):
|
||||||
prev_close = closes[k - 1]
|
prev_close = closes[k - 1]
|
||||||
bench_prev = benchmark_closes.get(dates[k - 1])
|
if prev_close <= 0:
|
||||||
bench_cur = benchmark_closes.get(dates[k])
|
continue
|
||||||
if prev_close <= 0 or bench_prev is None or bench_cur is None or bench_prev <= 0:
|
f_day: list[float] = []
|
||||||
|
ok = True
|
||||||
|
for fc in factor_closes:
|
||||||
|
f_prev = fc.get(dates[k - 1])
|
||||||
|
f_cur = fc.get(dates[k])
|
||||||
|
if f_prev is None or f_cur is None or f_prev <= 0:
|
||||||
|
ok = False
|
||||||
|
break
|
||||||
|
f_day.append(f_cur / f_prev - 1.0)
|
||||||
|
if not ok:
|
||||||
continue
|
continue
|
||||||
stock_rets.append(closes[k] / prev_close - 1.0)
|
stock_rets.append(closes[k] / prev_close - 1.0)
|
||||||
market_rets.append(bench_cur / bench_prev - 1.0)
|
for j, r in enumerate(f_day):
|
||||||
|
factor_rets[j].append(r)
|
||||||
|
|
||||||
if len(stock_rets) < 100:
|
n = len(stock_rets)
|
||||||
|
if n < 100:
|
||||||
return None
|
return None
|
||||||
mean_market = sum(market_rets) / len(market_rets)
|
|
||||||
mean_stock = sum(stock_rets) / len(stock_rets)
|
if n_factors == 1:
|
||||||
var_market = sum((x - mean_market) ** 2 for x in market_rets)
|
# Fast path: identical algebra to the historical single-factor form.
|
||||||
if var_market <= 0:
|
market_rets = factor_rets[0]
|
||||||
|
mean_market = sum(market_rets) / n
|
||||||
|
mean_stock = sum(stock_rets) / n
|
||||||
|
var_market = sum((x - mean_market) ** 2 for x in market_rets)
|
||||||
|
if var_market <= 0:
|
||||||
|
return None
|
||||||
|
cov = sum(
|
||||||
|
(stock_rets[k] - mean_stock) * (market_rets[k] - mean_market)
|
||||||
|
for k in range(n)
|
||||||
|
)
|
||||||
|
beta = cov / var_market
|
||||||
|
return sum(stock_rets[k] - beta * market_rets[k] for k in range(n))
|
||||||
|
|
||||||
|
# OLS without intercept: β = (X'X)^{-1} X'y for X columns = factor returns.
|
||||||
|
# Implemented for exactly two factors (market + sector); refuse larger.
|
||||||
|
if n_factors != 2:
|
||||||
return None
|
return None
|
||||||
cov = sum((stock_rets[k] - mean_stock) * (market_rets[k] - mean_market) for k in range(len(stock_rets)))
|
f1, f2 = factor_rets[0], factor_rets[1]
|
||||||
beta = cov / var_market
|
s11 = sum(a * a for a in f1)
|
||||||
return sum(stock_rets[k] - beta * market_rets[k] for k in range(len(stock_rets)))
|
s22 = sum(a * a for a in f2)
|
||||||
|
s12 = sum(f1[k] * f2[k] for k in range(n))
|
||||||
|
sy1 = sum(stock_rets[k] * f1[k] for k in range(n))
|
||||||
|
sy2 = sum(stock_rets[k] * f2[k] for k in range(n))
|
||||||
|
det = s11 * s22 - s12 * s12
|
||||||
|
if abs(det) < 1e-18:
|
||||||
|
return None
|
||||||
|
b1 = (s22 * sy1 - s12 * sy2) / det
|
||||||
|
b2 = (s11 * sy2 - s12 * sy1) / det
|
||||||
|
return sum(stock_rets[k] - b1 * f1[k] - b2 * f2[k] for k in range(n))
|
||||||
|
|
||||||
|
|
||||||
def _realized_vol_6m(closes: list[float], i: int) -> float | None:
|
def _realized_vol_6m(closes: list[float], i: int) -> float | None:
|
||||||
@@ -840,6 +895,7 @@ def _signal_values(
|
|||||||
highs: list[float],
|
highs: list[float],
|
||||||
i: int,
|
i: int,
|
||||||
benchmark_closes: dict[date, float] | None = None,
|
benchmark_closes: dict[date, float] | None = None,
|
||||||
|
sector_etf_closes: dict[date, float] | None = None,
|
||||||
) -> dict[str, float]:
|
) -> dict[str, float]:
|
||||||
"""Point-in-time candidate signals at as-of index ``i`` (price-only).
|
"""Point-in-time candidate signals at as-of index ``i`` (price-only).
|
||||||
|
|
||||||
@@ -851,6 +907,11 @@ def _signal_values(
|
|||||||
higher = nearer the high, expect positive IC). ``vol_6m`` is 126-day realized
|
higher = nearer the high, expect positive IC). ``vol_6m`` is 126-day realized
|
||||||
volatility (expect negative IC if the low-volatility anomaly holds).
|
volatility (expect negative IC if the low-volatility anomaly holds).
|
||||||
``fip_id`` is Da/Gurun/Warachka information discreteness (expect negative IC).
|
``fip_id`` is Da/Gurun/Warachka information discreteness (expect negative IC).
|
||||||
|
|
||||||
|
When ``sector_etf_closes`` is supplied (research path), also emit
|
||||||
|
``mom_12_1_sector_resid``: two-factor residual vs SPY + sector ETF.
|
||||||
|
Cross-sectional ``mom_12_1_sector_demeaned`` is injected later from the
|
||||||
|
full weekly cross-section (cannot be computed per-ticker alone).
|
||||||
"""
|
"""
|
||||||
out: dict[str, float] = {}
|
out: dict[str, float] = {}
|
||||||
if i - 252 >= 0 and closes[i - 252] > 0:
|
if i - 252 >= 0 and closes[i - 252] > 0:
|
||||||
@@ -858,6 +919,12 @@ def _signal_values(
|
|||||||
residual = _residual_momentum_12_1(dates, closes, i, benchmark_closes)
|
residual = _residual_momentum_12_1(dates, closes, i, benchmark_closes)
|
||||||
if residual is not None:
|
if residual is not None:
|
||||||
out["mom_12_1_resid"] = residual
|
out["mom_12_1_resid"] = residual
|
||||||
|
if benchmark_closes and sector_etf_closes:
|
||||||
|
sector_resid = _multi_factor_residual_momentum_12_1(
|
||||||
|
dates, closes, i, [benchmark_closes, sector_etf_closes]
|
||||||
|
)
|
||||||
|
if sector_resid is not None:
|
||||||
|
out["mom_12_1_sector_resid"] = sector_resid
|
||||||
fip = _fip_id(closes, i)
|
fip = _fip_id(closes, i)
|
||||||
if fip is not None:
|
if fip is not None:
|
||||||
out["fip_id"] = fip
|
out["fip_id"] = fip
|
||||||
@@ -944,14 +1011,16 @@ def _accumulate_signal_series(
|
|||||||
benchmark_closes: dict[date, float] | None = None,
|
benchmark_closes: dict[date, float] | None = None,
|
||||||
*,
|
*,
|
||||||
symbol: str | None = None,
|
symbol: str | None = None,
|
||||||
|
sector_etf_closes: dict[date, float] | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""For each weekly as-of bar, emit (signal, forward-return) pairs keyed by ISO
|
"""For each weekly as-of bar, emit (signal, forward-return) pairs keyed by ISO
|
||||||
week into ``collected[name][week_key]``. Forward return is close-to-close over
|
week into ``collected[name][week_key]``. Forward return is close-to-close over
|
||||||
HORIZON trading days. Mutates ``collected`` (a dict of dict of list).
|
HORIZON trading days. Mutates ``collected`` (a dict of dict of list).
|
||||||
|
|
||||||
When ``BACKTEST_LIQUID_BREADTH`` is set, observations are dicts with PIT
|
When ``BACKTEST_LIQUID_BREADTH`` is set, observations are dicts with PIT
|
||||||
liquidity fields for the mask; otherwise plain ``(val, fwd)`` tuples so the
|
liquidity fields for the mask. When ``symbol`` is provided, observations are
|
||||||
production signal path stays unchanged.
|
also dicts (so sector demeaning can group by name); otherwise plain
|
||||||
|
``(val, fwd)`` tuples keep the production path unchanged.
|
||||||
"""
|
"""
|
||||||
n = len(records)
|
n = len(records)
|
||||||
if n < HORIZON + 21:
|
if n < HORIZON + 21:
|
||||||
@@ -961,6 +1030,7 @@ def _accumulate_signal_series(
|
|||||||
volumes = [float(getattr(r, "volume", 0) or 0) for r in records]
|
volumes = [float(getattr(r, "volume", 0) or 0) for r in records]
|
||||||
dates = [r.date for r in records]
|
dates = [r.date for r in records]
|
||||||
liquid_mode = _liquid_breadth_top_n() > 0
|
liquid_mode = _liquid_breadth_top_n() > 0
|
||||||
|
rich = liquid_mode or symbol is not None
|
||||||
for i in _weekly_asof_indices(records):
|
for i in _weekly_asof_indices(records):
|
||||||
j = i + HORIZON
|
j = i + HORIZON
|
||||||
if j >= n or closes[i] <= 0:
|
if j >= n or closes[i] <= 0:
|
||||||
@@ -969,19 +1039,79 @@ def _accumulate_signal_series(
|
|||||||
iso = records[i].date.isocalendar()
|
iso = records[i].date.isocalendar()
|
||||||
week_key = (iso[0], iso[1])
|
week_key = (iso[0], iso[1])
|
||||||
dvol = _median_dollar_vol_63(closes, volumes, i) if liquid_mode else None
|
dvol = _median_dollar_vol_63(closes, volumes, i) if liquid_mode else None
|
||||||
for name, val in _signal_values(dates, closes, highs, i, benchmark_closes).items():
|
for name, val in _signal_values(
|
||||||
if liquid_mode:
|
dates, closes, highs, i, benchmark_closes, sector_etf_closes
|
||||||
collected[name][week_key].append({
|
).items():
|
||||||
|
if rich:
|
||||||
|
row = {
|
||||||
"val": val,
|
"val": val,
|
||||||
"fwd": fwd,
|
"fwd": fwd,
|
||||||
"close": closes[i],
|
|
||||||
"median_dvol_63": dvol,
|
|
||||||
"symbol": symbol,
|
"symbol": symbol,
|
||||||
})
|
}
|
||||||
|
if liquid_mode:
|
||||||
|
row["close"] = closes[i]
|
||||||
|
row["median_dvol_63"] = dvol
|
||||||
|
collected[name][week_key].append(row)
|
||||||
else:
|
else:
|
||||||
collected[name][week_key].append((val, fwd))
|
collected[name][week_key].append((val, fwd))
|
||||||
|
|
||||||
|
|
||||||
|
def _inject_sector_demeaned_momentum(
|
||||||
|
collected: dict,
|
||||||
|
symbol_to_sector: dict[str, str],
|
||||||
|
*,
|
||||||
|
min_sector_names: int = 2,
|
||||||
|
) -> None:
|
||||||
|
"""Cross-sectional demean of ``mom_12_1`` within GICS sector per week.
|
||||||
|
|
||||||
|
``mom_12_1_sector_demeaned[i] = mom_12_1[i] − mean(mom_12_1 | sector_i)``.
|
||||||
|
Requires rich observations with a ``symbol`` field (research path). Names
|
||||||
|
without a sector label, or sectors with fewer than ``min_sector_names``
|
||||||
|
members that week, are dropped from the demeaned series.
|
||||||
|
"""
|
||||||
|
if not symbol_to_sector or "mom_12_1" not in collected:
|
||||||
|
return
|
||||||
|
from app.services.sector_map import normalise_symbol
|
||||||
|
|
||||||
|
demeaned: dict = defaultdict(list)
|
||||||
|
for week_key, recs in collected["mom_12_1"].items():
|
||||||
|
parsed: list[tuple[str, float, float, object]] = []
|
||||||
|
by_sector: dict[str, list[float]] = defaultdict(list)
|
||||||
|
for rec in recs:
|
||||||
|
pair = _obs_val_fwd(rec)
|
||||||
|
if pair is None:
|
||||||
|
continue
|
||||||
|
val, fwd = pair
|
||||||
|
if isinstance(rec, dict):
|
||||||
|
sym = rec.get("symbol")
|
||||||
|
else:
|
||||||
|
sym = None
|
||||||
|
if not sym:
|
||||||
|
continue
|
||||||
|
sector = symbol_to_sector.get(normalise_symbol(str(sym)))
|
||||||
|
if not sector:
|
||||||
|
continue
|
||||||
|
parsed.append((sector, val, fwd, rec))
|
||||||
|
by_sector[sector].append(val)
|
||||||
|
means = {
|
||||||
|
sec: sum(vs) / len(vs)
|
||||||
|
for sec, vs in by_sector.items()
|
||||||
|
if len(vs) >= min_sector_names
|
||||||
|
}
|
||||||
|
for sector, val, fwd, rec in parsed:
|
||||||
|
if sector not in means:
|
||||||
|
continue
|
||||||
|
dval = val - means[sector]
|
||||||
|
if isinstance(rec, dict):
|
||||||
|
row = dict(rec)
|
||||||
|
row["val"] = dval
|
||||||
|
demeaned[week_key].append(row)
|
||||||
|
else:
|
||||||
|
demeaned[week_key].append((dval, fwd))
|
||||||
|
if demeaned:
|
||||||
|
collected["mom_12_1_sector_demeaned"] = demeaned
|
||||||
|
|
||||||
|
|
||||||
def _rank(xs: list[float]) -> list[float]:
|
def _rank(xs: list[float]) -> list[float]:
|
||||||
"""Average (tie-corrected) ranks, 1-based."""
|
"""Average (tie-corrected) ranks, 1-based."""
|
||||||
order = sorted(range(len(xs)), key=lambda k: xs[k])
|
order = sorted(range(len(xs)), key=lambda k: xs[k])
|
||||||
@@ -1258,14 +1388,38 @@ def _signal_series(
|
|||||||
benchmark_closes: dict[date, float] | None = None,
|
benchmark_closes: dict[date, float] | None = None,
|
||||||
*,
|
*,
|
||||||
symbol: str | None = None,
|
symbol: str | None = None,
|
||||||
|
sector_etf_closes: dict[date, float] | None = None,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Per-ticker signal/forward-return series as a PLAIN (picklable) nested dict
|
"""Per-ticker signal/forward-return series as a PLAIN (picklable) nested dict
|
||||||
— no defaultdict/lambda — so it can cross a process boundary."""
|
— no defaultdict/lambda — so it can cross a process boundary."""
|
||||||
tmp: dict = defaultdict(lambda: defaultdict(list))
|
tmp: dict = defaultdict(lambda: defaultdict(list))
|
||||||
_accumulate_signal_series(records, tmp, benchmark_closes, symbol=symbol)
|
_accumulate_signal_series(
|
||||||
|
records,
|
||||||
|
tmp,
|
||||||
|
benchmark_closes,
|
||||||
|
symbol=symbol,
|
||||||
|
sector_etf_closes=sector_etf_closes,
|
||||||
|
)
|
||||||
return {name: dict(weeks) for name, weeks in tmp.items()}
|
return {name: dict(weeks) for name, weeks in tmp.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def _sector_etf_closes_for_symbol(
|
||||||
|
symbol: str,
|
||||||
|
symbol_to_sector: dict[str, str] | None,
|
||||||
|
sector_etf_closes: dict[str, dict[date, float]] | None,
|
||||||
|
) -> dict[date, float] | None:
|
||||||
|
"""Resolve the sector-ETF close series for one ticker, or None."""
|
||||||
|
if not symbol_to_sector or not sector_etf_closes:
|
||||||
|
return None
|
||||||
|
from app.services.sector_map import etf_for_symbol
|
||||||
|
|
||||||
|
etf = etf_for_symbol(symbol, symbol_to_sector)
|
||||||
|
if not etf:
|
||||||
|
return None
|
||||||
|
series = sector_etf_closes.get(etf)
|
||||||
|
return series or None
|
||||||
|
|
||||||
|
|
||||||
def _replay_and_signals(
|
def _replay_and_signals(
|
||||||
symbol: str,
|
symbol: str,
|
||||||
columns: tuple,
|
columns: tuple,
|
||||||
@@ -1275,6 +1429,8 @@ def _replay_and_signals(
|
|||||||
target_model: str = PRODUCTION_GTL_TARGET_MODEL,
|
target_model: str = PRODUCTION_GTL_TARGET_MODEL,
|
||||||
cadence: str = DEFAULT_BACKTEST_CADENCE,
|
cadence: str = DEFAULT_BACKTEST_CADENCE,
|
||||||
signal_only: bool = False,
|
signal_only: bool = False,
|
||||||
|
sector_etf_closes: dict[str, dict[date, float]] | None = None,
|
||||||
|
symbol_to_sector: dict[str, str] | None = None,
|
||||||
) -> tuple[list[dict], dict]:
|
) -> tuple[list[dict], dict]:
|
||||||
"""The CPU-bound per-ticker work, as a top-level (picklable) function so it can
|
"""The CPU-bound per-ticker work, as a top-level (picklable) function so it can
|
||||||
run in a worker process. Takes primitive column arrays (cheap to pickle),
|
run in a worker process. Takes primitive column arrays (cheap to pickle),
|
||||||
@@ -1301,9 +1457,17 @@ def _replay_and_signals(
|
|||||||
target_model,
|
target_model,
|
||||||
cadence,
|
cadence,
|
||||||
)
|
)
|
||||||
|
etf_closes = _sector_etf_closes_for_symbol(
|
||||||
|
symbol, symbol_to_sector, sector_etf_closes
|
||||||
|
)
|
||||||
return (
|
return (
|
||||||
candidates,
|
candidates,
|
||||||
_signal_series(bars, benchmark_closes, symbol=symbol),
|
_signal_series(
|
||||||
|
bars,
|
||||||
|
benchmark_closes,
|
||||||
|
symbol=symbol,
|
||||||
|
sector_etf_closes=etf_closes,
|
||||||
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -4054,6 +4218,41 @@ async def run_backtest(
|
|||||||
except Exception:
|
except Exception:
|
||||||
logger.exception("Benchmark load for residual momentum failed")
|
logger.exception("Benchmark load for residual momentum failed")
|
||||||
|
|
||||||
|
# Optional sector residualisation (research): local ticker→sector map + sector
|
||||||
|
# ETF closes stored in benchmark_prices. Absent map/series → no sector signals.
|
||||||
|
symbol_to_sector: dict[str, str] = {}
|
||||||
|
sector_etf_closes: dict[str, dict[date, float]] = {}
|
||||||
|
try:
|
||||||
|
from app.services.sector_map import (
|
||||||
|
SECTOR_ETFS,
|
||||||
|
load_ticker_sector_map,
|
||||||
|
normalise_symbol,
|
||||||
|
)
|
||||||
|
from app.services.benchmark_service import load_benchmark_closes
|
||||||
|
|
||||||
|
map_path = os.getenv("BACKTEST_SECTOR_MAP_PATH", "").strip() or None
|
||||||
|
symbol_to_sector = {
|
||||||
|
normalise_symbol(k): v
|
||||||
|
for k, v in load_ticker_sector_map(map_path).items()
|
||||||
|
}
|
||||||
|
if symbol_to_sector:
|
||||||
|
for etf in SECTOR_ETFS:
|
||||||
|
try:
|
||||||
|
series = await load_benchmark_closes(db, etf)
|
||||||
|
except Exception:
|
||||||
|
series = {}
|
||||||
|
if series:
|
||||||
|
sector_etf_closes[etf] = series
|
||||||
|
logger.info(json.dumps({
|
||||||
|
"event": "backtest_sector_context_loaded",
|
||||||
|
"sector_map_size": len(symbol_to_sector),
|
||||||
|
"sector_etfs_loaded": sorted(sector_etf_closes),
|
||||||
|
}))
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Sector residual context load failed; continuing without")
|
||||||
|
symbol_to_sector = {}
|
||||||
|
sector_etf_closes = {}
|
||||||
|
|
||||||
def _merge(result: tuple[list[dict], dict]) -> None:
|
def _merge(result: tuple[list[dict], dict]) -> None:
|
||||||
cands, series = result
|
cands, series = result
|
||||||
candidates.extend(cands)
|
candidates.extend(cands)
|
||||||
@@ -4104,6 +4303,8 @@ async def run_backtest(
|
|||||||
target_model,
|
target_model,
|
||||||
cadence,
|
cadence,
|
||||||
ticker.symbol in rank_only_symbols,
|
ticker.symbol in rank_only_symbols,
|
||||||
|
sector_etf_closes or None,
|
||||||
|
symbol_to_sector or None,
|
||||||
))
|
))
|
||||||
for result in await asyncio.gather(*futures, return_exceptions=True):
|
for result in await asyncio.gather(*futures, return_exceptions=True):
|
||||||
if isinstance(result, Exception):
|
if isinstance(result, Exception):
|
||||||
@@ -4132,6 +4333,8 @@ async def run_backtest(
|
|||||||
target_model,
|
target_model,
|
||||||
cadence,
|
cadence,
|
||||||
ticker.symbol in rank_only_symbols,
|
ticker.symbol in rank_only_symbols,
|
||||||
|
sector_etf_closes or None,
|
||||||
|
symbol_to_sector or None,
|
||||||
))
|
))
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception("Backtest replay failed for %s", ticker.symbol)
|
logger.exception("Backtest replay failed for %s", ticker.symbol)
|
||||||
@@ -4139,6 +4342,13 @@ async def run_backtest(
|
|||||||
if progress_cb is not None and total:
|
if progress_cb is not None and total:
|
||||||
progress_cb(total, total, "")
|
progress_cb(total, total, "")
|
||||||
|
|
||||||
|
# Cross-sectional sector demean needs the full weekly universe.
|
||||||
|
if symbol_to_sector:
|
||||||
|
try:
|
||||||
|
_inject_sector_demeaned_momentum(collected, symbol_to_sector)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Sector demeaned momentum injection failed")
|
||||||
|
|
||||||
# Cross-sectional momentum: rank every week's universe, then "qualified" means
|
# Cross-sectional momentum: rank every week's universe, then "qualified" means
|
||||||
# floors + top ``min_momentum_percentile`` by promoted residual 12-1 momentum
|
# floors + top ``min_momentum_percentile`` by promoted residual 12-1 momentum
|
||||||
# (raw 12-1 fallback only when benchmark data is unavailable).
|
# (raw 12-1 fallback only when benchmark data is unavailable).
|
||||||
|
|||||||
@@ -0,0 +1,145 @@
|
|||||||
|
"""Ticker → GICS sector → SPDR sector ETF mapping (research only).
|
||||||
|
|
||||||
|
Sector residual momentum residualizes 12-1 momentum against SPY and the name's
|
||||||
|
sector ETF. Labels are persisted under ``data/research/ticker_sector_map.json``
|
||||||
|
so research runs do not depend on live FMP calls.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
# Eleven SPDR sector ETFs. Auxiliary series only — never tradable book members.
|
||||||
|
SECTOR_ETFS: tuple[str, ...] = (
|
||||||
|
"XLB",
|
||||||
|
"XLC",
|
||||||
|
"XLE",
|
||||||
|
"XLF",
|
||||||
|
"XLI",
|
||||||
|
"XLK",
|
||||||
|
"XLP",
|
||||||
|
"XLRE",
|
||||||
|
"XLU",
|
||||||
|
"XLV",
|
||||||
|
"XLY",
|
||||||
|
)
|
||||||
|
|
||||||
|
# GICS sector name (and common aliases) → SPDR ETF.
|
||||||
|
# Keys are lower-case for matching.
|
||||||
|
GICS_SECTOR_TO_ETF: dict[str, str] = {
|
||||||
|
"materials": "XLB",
|
||||||
|
"basic materials": "XLB",
|
||||||
|
"communication services": "XLC",
|
||||||
|
"communications": "XLC",
|
||||||
|
"telecommunication services": "XLC",
|
||||||
|
"energy": "XLE",
|
||||||
|
"financials": "XLF",
|
||||||
|
"financial services": "XLF",
|
||||||
|
"financial": "XLF",
|
||||||
|
"industrials": "XLI",
|
||||||
|
"industrial goods": "XLI",
|
||||||
|
"information technology": "XLK",
|
||||||
|
"technology": "XLK",
|
||||||
|
"consumer staples": "XLP",
|
||||||
|
"consumer defensive": "XLP",
|
||||||
|
"real estate": "XLRE",
|
||||||
|
"utilities": "XLU",
|
||||||
|
"health care": "XLV",
|
||||||
|
"healthcare": "XLV",
|
||||||
|
"consumer discretionary": "XLY",
|
||||||
|
"consumer cyclical": "XLY",
|
||||||
|
}
|
||||||
|
|
||||||
|
DEFAULT_SECTOR_MAP_PATH = Path("data/research/ticker_sector_map.json")
|
||||||
|
|
||||||
|
|
||||||
|
def normalise_symbol(symbol: str) -> str:
|
||||||
|
"""Alpaca-style symbols: BRK.B / BRK/B → BRK-B."""
|
||||||
|
s = str(symbol or "").strip().upper()
|
||||||
|
s = s.replace(".", "-").replace("/", "-")
|
||||||
|
return s
|
||||||
|
|
||||||
|
|
||||||
|
def sector_to_etf(sector: str | None) -> str | None:
|
||||||
|
if not sector:
|
||||||
|
return None
|
||||||
|
return GICS_SECTOR_TO_ETF.get(str(sector).strip().lower())
|
||||||
|
|
||||||
|
|
||||||
|
def etf_for_symbol(symbol: str, symbol_to_sector: dict[str, str]) -> str | None:
|
||||||
|
sector = symbol_to_sector.get(normalise_symbol(symbol))
|
||||||
|
return sector_to_etf(sector)
|
||||||
|
|
||||||
|
|
||||||
|
def load_ticker_sector_map(path: Path | str | None = None) -> dict[str, str]:
|
||||||
|
"""Load ``{symbol: gics_sector}`` from JSON. Empty dict if missing."""
|
||||||
|
p = Path(path) if path is not None else DEFAULT_SECTOR_MAP_PATH
|
||||||
|
if not p.exists():
|
||||||
|
return {}
|
||||||
|
raw = json.loads(p.read_text(encoding="utf-8"))
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
return {}
|
||||||
|
out: dict[str, str] = {}
|
||||||
|
# Accept either flat map or {"map": {...}, "meta": ...}
|
||||||
|
payload = raw.get("map") if "map" in raw and isinstance(raw.get("map"), dict) else raw
|
||||||
|
if not isinstance(payload, dict):
|
||||||
|
return {}
|
||||||
|
for sym, sector in payload.items():
|
||||||
|
if sym in ("meta", "schema_version", "map"):
|
||||||
|
continue
|
||||||
|
if sector is None:
|
||||||
|
continue
|
||||||
|
ns = normalise_symbol(str(sym))
|
||||||
|
if ns:
|
||||||
|
out[ns] = str(sector).strip()
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def save_ticker_sector_map(
|
||||||
|
mapping: dict[str, str],
|
||||||
|
path: Path | str | None = None,
|
||||||
|
*,
|
||||||
|
meta: dict[str, Any] | None = None,
|
||||||
|
) -> Path:
|
||||||
|
p = Path(path) if path is not None else DEFAULT_SECTOR_MAP_PATH
|
||||||
|
p.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
# Normalise keys on write.
|
||||||
|
clean = {
|
||||||
|
normalise_symbol(k): str(v).strip()
|
||||||
|
for k, v in mapping.items()
|
||||||
|
if k and v and normalise_symbol(k)
|
||||||
|
}
|
||||||
|
payload: dict[str, Any] = {
|
||||||
|
"schema_version": 1,
|
||||||
|
"map": clean,
|
||||||
|
"meta": meta or {},
|
||||||
|
}
|
||||||
|
p.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8")
|
||||||
|
return p
|
||||||
|
|
||||||
|
|
||||||
|
def coverage_stats(
|
||||||
|
symbols: list[str], mapping: dict[str, str]
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
total = len(symbols)
|
||||||
|
mapped = [s for s in symbols if normalise_symbol(s) in mapping]
|
||||||
|
with_etf = [
|
||||||
|
s
|
||||||
|
for s in mapped
|
||||||
|
if sector_to_etf(mapping[normalise_symbol(s)]) is not None
|
||||||
|
]
|
||||||
|
missing = [s for s in symbols if normalise_symbol(s) not in mapping]
|
||||||
|
by_sector: dict[str, int] = {}
|
||||||
|
for s in mapped:
|
||||||
|
sec = mapping[normalise_symbol(s)]
|
||||||
|
by_sector[sec] = by_sector.get(sec, 0) + 1
|
||||||
|
return {
|
||||||
|
"universe": total,
|
||||||
|
"mapped": len(mapped),
|
||||||
|
"mapped_pct": round(100.0 * len(mapped) / total, 1) if total else 0.0,
|
||||||
|
"with_etf": len(with_etf),
|
||||||
|
"missing": missing,
|
||||||
|
"by_sector": dict(sorted(by_sector.items(), key=lambda kv: (-kv[1], kv[0]))),
|
||||||
|
}
|
||||||
@@ -0,0 +1,543 @@
|
|||||||
|
{
|
||||||
|
"map": {
|
||||||
|
"A": "Health Care",
|
||||||
|
"AAPL": "Information Technology",
|
||||||
|
"ABBV": "Health Care",
|
||||||
|
"ABNB": "Consumer Discretionary",
|
||||||
|
"ABT": "Health Care",
|
||||||
|
"ACGL": "Financials",
|
||||||
|
"ACN": "Information Technology",
|
||||||
|
"ADBE": "Information Technology",
|
||||||
|
"ADI": "Information Technology",
|
||||||
|
"ADM": "Consumer Staples",
|
||||||
|
"ADP": "Industrials",
|
||||||
|
"ADSK": "Information Technology",
|
||||||
|
"AEE": "Utilities",
|
||||||
|
"AEP": "Utilities",
|
||||||
|
"AES": "Utilities",
|
||||||
|
"AFL": "Financials",
|
||||||
|
"AIG": "Financials",
|
||||||
|
"AIZ": "Financials",
|
||||||
|
"AJG": "Financials",
|
||||||
|
"AKAM": "Information Technology",
|
||||||
|
"ALB": "Materials",
|
||||||
|
"ALGN": "Health Care",
|
||||||
|
"ALL": "Financials",
|
||||||
|
"ALLE": "Industrials",
|
||||||
|
"AMAT": "Information Technology",
|
||||||
|
"AMCR": "Materials",
|
||||||
|
"AMD": "Information Technology",
|
||||||
|
"AME": "Industrials",
|
||||||
|
"AMGN": "Health Care",
|
||||||
|
"AMP": "Financials",
|
||||||
|
"AMT": "Real Estate",
|
||||||
|
"AMZN": "Consumer Discretionary",
|
||||||
|
"ANET": "Information Technology",
|
||||||
|
"AON": "Financials",
|
||||||
|
"AOS": "Industrials",
|
||||||
|
"APA": "Energy",
|
||||||
|
"APD": "Materials",
|
||||||
|
"APH": "Information Technology",
|
||||||
|
"APO": "Financials",
|
||||||
|
"APP": "Information Technology",
|
||||||
|
"APTV": "Consumer Discretionary",
|
||||||
|
"ARE": "Real Estate",
|
||||||
|
"ARES": "Financials",
|
||||||
|
"ATO": "Utilities",
|
||||||
|
"AVB": "Real Estate",
|
||||||
|
"AVGO": "Information Technology",
|
||||||
|
"AVY": "Materials",
|
||||||
|
"AWK": "Utilities",
|
||||||
|
"AXON": "Industrials",
|
||||||
|
"AXP": "Financials",
|
||||||
|
"AZO": "Consumer Discretionary",
|
||||||
|
"BA": "Industrials",
|
||||||
|
"BAC": "Financials",
|
||||||
|
"BALL": "Materials",
|
||||||
|
"BAX": "Health Care",
|
||||||
|
"BBY": "Consumer Discretionary",
|
||||||
|
"BDX": "Health Care",
|
||||||
|
"BEN": "Financials",
|
||||||
|
"BF-B": "Consumer Staples",
|
||||||
|
"BG": "Consumer Staples",
|
||||||
|
"BIIB": "Health Care",
|
||||||
|
"BK": "Financial Services",
|
||||||
|
"BKNG": "Consumer Discretionary",
|
||||||
|
"BKR": "Energy",
|
||||||
|
"BLDR": "Industrials",
|
||||||
|
"BLK": "Financials",
|
||||||
|
"BMY": "Health Care",
|
||||||
|
"BR": "Industrials",
|
||||||
|
"BRK-B": "Financials",
|
||||||
|
"BRO": "Financials",
|
||||||
|
"BSX": "Health Care",
|
||||||
|
"BX": "Financials",
|
||||||
|
"BXP": "Real Estate",
|
||||||
|
"C": "Financials",
|
||||||
|
"CAG": "Consumer Defensive",
|
||||||
|
"CAH": "Health Care",
|
||||||
|
"CARR": "Industrials",
|
||||||
|
"CASY": "Consumer Staples",
|
||||||
|
"CAT": "Industrials",
|
||||||
|
"CB": "Financials",
|
||||||
|
"CBOE": "Financials",
|
||||||
|
"CBRE": "Real Estate",
|
||||||
|
"CCI": "Real Estate",
|
||||||
|
"CCL": "Consumer Discretionary",
|
||||||
|
"CDNS": "Information Technology",
|
||||||
|
"CDW": "Information Technology",
|
||||||
|
"CEG": "Utilities",
|
||||||
|
"CF": "Materials",
|
||||||
|
"CFG": "Financials",
|
||||||
|
"CHD": "Consumer Staples",
|
||||||
|
"CHRW": "Industrials",
|
||||||
|
"CHTR": "Communication Services",
|
||||||
|
"CI": "Health Care",
|
||||||
|
"CIEN": "Information Technology",
|
||||||
|
"CINF": "Financials",
|
||||||
|
"CL": "Consumer Staples",
|
||||||
|
"CLX": "Consumer Staples",
|
||||||
|
"CMCSA": "Communication Services",
|
||||||
|
"CME": "Financials",
|
||||||
|
"CMG": "Consumer Discretionary",
|
||||||
|
"CMI": "Industrials",
|
||||||
|
"CMS": "Utilities",
|
||||||
|
"CNC": "Health Care",
|
||||||
|
"CNP": "Utilities",
|
||||||
|
"COF": "Financials",
|
||||||
|
"COHR": "Information Technology",
|
||||||
|
"COIN": "Financials",
|
||||||
|
"COO": "Health Care",
|
||||||
|
"COP": "Energy",
|
||||||
|
"COR": "Health Care",
|
||||||
|
"COST": "Consumer Staples",
|
||||||
|
"CPAY": "Financials",
|
||||||
|
"CPB": "Consumer Defensive",
|
||||||
|
"CPRT": "Industrials",
|
||||||
|
"CPT": "Real Estate",
|
||||||
|
"CRH": "Materials",
|
||||||
|
"CRL": "Health Care",
|
||||||
|
"CRM": "Information Technology",
|
||||||
|
"CRWD": "Information Technology",
|
||||||
|
"CSCO": "Information Technology",
|
||||||
|
"CSGP": "Real Estate",
|
||||||
|
"CSX": "Industrials",
|
||||||
|
"CTAS": "Industrials",
|
||||||
|
"CTRA": "Energy",
|
||||||
|
"CTSH": "Information Technology",
|
||||||
|
"CTVA": "Materials",
|
||||||
|
"CVNA": "Consumer Discretionary",
|
||||||
|
"CVS": "Health Care",
|
||||||
|
"CVX": "Energy",
|
||||||
|
"D": "Utilities",
|
||||||
|
"DAL": "Industrials",
|
||||||
|
"DASH": "Consumer Discretionary",
|
||||||
|
"DD": "Materials",
|
||||||
|
"DDOG": "Information Technology",
|
||||||
|
"DE": "Industrials",
|
||||||
|
"DECK": "Consumer Discretionary",
|
||||||
|
"DELL": "Information Technology",
|
||||||
|
"DG": "Consumer Staples",
|
||||||
|
"DGX": "Health Care",
|
||||||
|
"DHI": "Consumer Discretionary",
|
||||||
|
"DHR": "Health Care",
|
||||||
|
"DIS": "Communication Services",
|
||||||
|
"DLR": "Real Estate",
|
||||||
|
"DLTR": "Consumer Staples",
|
||||||
|
"DOC": "Real Estate",
|
||||||
|
"DOV": "Industrials",
|
||||||
|
"DOW": "Materials",
|
||||||
|
"DPZ": "Consumer Discretionary",
|
||||||
|
"DRI": "Consumer Discretionary",
|
||||||
|
"DTE": "Utilities",
|
||||||
|
"DUK": "Utilities",
|
||||||
|
"DVA": "Health Care",
|
||||||
|
"DVN": "Energy",
|
||||||
|
"DXCM": "Health Care",
|
||||||
|
"EA": "Communication Services",
|
||||||
|
"EBAY": "Consumer Discretionary",
|
||||||
|
"ECL": "Materials",
|
||||||
|
"ED": "Utilities",
|
||||||
|
"EFX": "Industrials",
|
||||||
|
"EG": "Financials",
|
||||||
|
"EIX": "Utilities",
|
||||||
|
"EL": "Consumer Staples",
|
||||||
|
"ELV": "Health Care",
|
||||||
|
"EME": "Industrials",
|
||||||
|
"EMR": "Industrials",
|
||||||
|
"EOG": "Energy",
|
||||||
|
"EPAM": "Technology",
|
||||||
|
"EQIX": "Real Estate",
|
||||||
|
"EQR": "Real Estate",
|
||||||
|
"EQT": "Energy",
|
||||||
|
"ERIE": "Financials",
|
||||||
|
"ES": "Utilities",
|
||||||
|
"ESS": "Real Estate",
|
||||||
|
"ETN": "Industrials",
|
||||||
|
"ETR": "Utilities",
|
||||||
|
"EVRG": "Utilities",
|
||||||
|
"EW": "Health Care",
|
||||||
|
"EXC": "Utilities",
|
||||||
|
"EXE": "Energy",
|
||||||
|
"EXPD": "Industrials",
|
||||||
|
"EXPE": "Consumer Discretionary",
|
||||||
|
"EXR": "Real Estate",
|
||||||
|
"F": "Consumer Discretionary",
|
||||||
|
"FANG": "Energy",
|
||||||
|
"FAST": "Industrials",
|
||||||
|
"FCX": "Materials",
|
||||||
|
"FDS": "Financials",
|
||||||
|
"FDX": "Industrials",
|
||||||
|
"FE": "Utilities",
|
||||||
|
"FFIV": "Information Technology",
|
||||||
|
"FICO": "Information Technology",
|
||||||
|
"FIS": "Financials",
|
||||||
|
"FISV": "Financials",
|
||||||
|
"FITB": "Financials",
|
||||||
|
"FIX": "Industrials",
|
||||||
|
"FOX": "Communication Services",
|
||||||
|
"FOXA": "Communication Services",
|
||||||
|
"FRT": "Real Estate",
|
||||||
|
"FSLR": "Information Technology",
|
||||||
|
"FTNT": "Information Technology",
|
||||||
|
"FTV": "Industrials",
|
||||||
|
"GD": "Industrials",
|
||||||
|
"GDDY": "Information Technology",
|
||||||
|
"GE": "Industrials",
|
||||||
|
"GEHC": "Health Care",
|
||||||
|
"GEN": "Information Technology",
|
||||||
|
"GEV": "Industrials",
|
||||||
|
"GILD": "Health Care",
|
||||||
|
"GIS": "Consumer Staples",
|
||||||
|
"GL": "Financials",
|
||||||
|
"GLW": "Information Technology",
|
||||||
|
"GM": "Consumer Discretionary",
|
||||||
|
"GNRC": "Industrials",
|
||||||
|
"GOOG": "Communication Services",
|
||||||
|
"GOOGL": "Communication Services",
|
||||||
|
"GPC": "Consumer Discretionary",
|
||||||
|
"GPN": "Financials",
|
||||||
|
"GRMN": "Consumer Discretionary",
|
||||||
|
"GS": "Financials",
|
||||||
|
"GWW": "Industrials",
|
||||||
|
"HAL": "Energy",
|
||||||
|
"HAS": "Consumer Discretionary",
|
||||||
|
"HBAN": "Financials",
|
||||||
|
"HCA": "Health Care",
|
||||||
|
"HD": "Consumer Discretionary",
|
||||||
|
"HIG": "Financials",
|
||||||
|
"HII": "Industrials",
|
||||||
|
"HLT": "Consumer Discretionary",
|
||||||
|
"HON": "Industrials",
|
||||||
|
"HOOD": "Financials",
|
||||||
|
"HPE": "Information Technology",
|
||||||
|
"HPQ": "Information Technology",
|
||||||
|
"HRL": "Consumer Staples",
|
||||||
|
"HSIC": "Health Care",
|
||||||
|
"HST": "Real Estate",
|
||||||
|
"HSY": "Consumer Staples",
|
||||||
|
"HUBB": "Industrials",
|
||||||
|
"HUM": "Health Care",
|
||||||
|
"HWM": "Industrials",
|
||||||
|
"IBKR": "Financials",
|
||||||
|
"IBM": "Information Technology",
|
||||||
|
"ICE": "Financials",
|
||||||
|
"IDXX": "Health Care",
|
||||||
|
"IEX": "Industrials",
|
||||||
|
"IFF": "Materials",
|
||||||
|
"INCY": "Health Care",
|
||||||
|
"INTC": "Information Technology",
|
||||||
|
"INTU": "Information Technology",
|
||||||
|
"INVH": "Real Estate",
|
||||||
|
"IP": "Materials",
|
||||||
|
"IQV": "Health Care",
|
||||||
|
"IR": "Industrials",
|
||||||
|
"IRM": "Real Estate",
|
||||||
|
"ISRG": "Health Care",
|
||||||
|
"IT": "Information Technology",
|
||||||
|
"ITW": "Industrials",
|
||||||
|
"IVZ": "Financials",
|
||||||
|
"J": "Industrials",
|
||||||
|
"JBHT": "Industrials",
|
||||||
|
"JBL": "Information Technology",
|
||||||
|
"JCI": "Industrials",
|
||||||
|
"JKHY": "Financials",
|
||||||
|
"JNJ": "Health Care",
|
||||||
|
"JPM": "Financials",
|
||||||
|
"KDP": "Consumer Staples",
|
||||||
|
"KEY": "Financials",
|
||||||
|
"KEYS": "Information Technology",
|
||||||
|
"KHC": "Consumer Staples",
|
||||||
|
"KIM": "Real Estate",
|
||||||
|
"KKR": "Financials",
|
||||||
|
"KLAC": "Information Technology",
|
||||||
|
"KMB": "Consumer Staples",
|
||||||
|
"KMI": "Energy",
|
||||||
|
"KO": "Consumer Staples",
|
||||||
|
"KR": "Consumer Staples",
|
||||||
|
"KVUE": "Consumer Staples",
|
||||||
|
"L": "Financials",
|
||||||
|
"LDOS": "Industrials",
|
||||||
|
"LEN": "Consumer Discretionary",
|
||||||
|
"LH": "Health Care",
|
||||||
|
"LHX": "Industrials",
|
||||||
|
"LII": "Industrials",
|
||||||
|
"LIN": "Materials",
|
||||||
|
"LITE": "Information Technology",
|
||||||
|
"LLY": "Health Care",
|
||||||
|
"LMT": "Industrials",
|
||||||
|
"LNT": "Utilities",
|
||||||
|
"LOW": "Consumer Discretionary",
|
||||||
|
"LRCX": "Information Technology",
|
||||||
|
"LULU": "Consumer Discretionary",
|
||||||
|
"LUV": "Industrials",
|
||||||
|
"LVS": "Consumer Discretionary",
|
||||||
|
"LYB": "Materials",
|
||||||
|
"LYV": "Communication Services",
|
||||||
|
"MA": "Financials",
|
||||||
|
"MAA": "Real Estate",
|
||||||
|
"MAR": "Consumer Discretionary",
|
||||||
|
"MAS": "Industrials",
|
||||||
|
"MCD": "Consumer Discretionary",
|
||||||
|
"MCHP": "Information Technology",
|
||||||
|
"MCK": "Health Care",
|
||||||
|
"MCO": "Financials",
|
||||||
|
"MDLZ": "Consumer Staples",
|
||||||
|
"MDT": "Health Care",
|
||||||
|
"MET": "Financials",
|
||||||
|
"META": "Communication Services",
|
||||||
|
"MGM": "Consumer Discretionary",
|
||||||
|
"MKC": "Consumer Staples",
|
||||||
|
"MLM": "Materials",
|
||||||
|
"MMM": "Industrials",
|
||||||
|
"MNST": "Consumer Staples",
|
||||||
|
"MO": "Consumer Staples",
|
||||||
|
"MOS": "Materials",
|
||||||
|
"MPC": "Energy",
|
||||||
|
"MPWR": "Information Technology",
|
||||||
|
"MRK": "Health Care",
|
||||||
|
"MRNA": "Health Care",
|
||||||
|
"MRSH": "Financials",
|
||||||
|
"MS": "Financials",
|
||||||
|
"MSCI": "Financials",
|
||||||
|
"MSFT": "Information Technology",
|
||||||
|
"MSI": "Information Technology",
|
||||||
|
"MSTR": "Technology",
|
||||||
|
"MTB": "Financials",
|
||||||
|
"MTD": "Health Care",
|
||||||
|
"MU": "Information Technology",
|
||||||
|
"NCLH": "Consumer Discretionary",
|
||||||
|
"NDAQ": "Financials",
|
||||||
|
"NDSN": "Industrials",
|
||||||
|
"NEE": "Utilities",
|
||||||
|
"NEM": "Materials",
|
||||||
|
"NFLX": "Communication Services",
|
||||||
|
"NI": "Utilities",
|
||||||
|
"NKE": "Consumer Discretionary",
|
||||||
|
"NOC": "Industrials",
|
||||||
|
"NOW": "Information Technology",
|
||||||
|
"NRG": "Utilities",
|
||||||
|
"NSC": "Industrials",
|
||||||
|
"NTAP": "Information Technology",
|
||||||
|
"NTRS": "Financials",
|
||||||
|
"NUE": "Materials",
|
||||||
|
"NVDA": "Information Technology",
|
||||||
|
"NVR": "Consumer Discretionary",
|
||||||
|
"NWS": "Communication Services",
|
||||||
|
"NWSA": "Communication Services",
|
||||||
|
"NXPI": "Information Technology",
|
||||||
|
"O": "Real Estate",
|
||||||
|
"ODFL": "Industrials",
|
||||||
|
"OKE": "Energy",
|
||||||
|
"OMC": "Communication Services",
|
||||||
|
"ON": "Information Technology",
|
||||||
|
"ORCL": "Information Technology",
|
||||||
|
"ORLY": "Consumer Discretionary",
|
||||||
|
"OTIS": "Industrials",
|
||||||
|
"OXY": "Energy",
|
||||||
|
"PANW": "Information Technology",
|
||||||
|
"PAYX": "Industrials",
|
||||||
|
"PCAR": "Industrials",
|
||||||
|
"PCG": "Utilities",
|
||||||
|
"PEG": "Utilities",
|
||||||
|
"PEP": "Consumer Staples",
|
||||||
|
"PFE": "Health Care",
|
||||||
|
"PFG": "Financials",
|
||||||
|
"PG": "Consumer Staples",
|
||||||
|
"PGR": "Financials",
|
||||||
|
"PH": "Industrials",
|
||||||
|
"PHM": "Consumer Discretionary",
|
||||||
|
"PKG": "Materials",
|
||||||
|
"PLD": "Real Estate",
|
||||||
|
"PLTR": "Information Technology",
|
||||||
|
"PM": "Consumer Staples",
|
||||||
|
"PNC": "Financials",
|
||||||
|
"PNR": "Industrials",
|
||||||
|
"PNW": "Utilities",
|
||||||
|
"PODD": "Health Care",
|
||||||
|
"POOL": "Industrials",
|
||||||
|
"PPG": "Materials",
|
||||||
|
"PPL": "Utilities",
|
||||||
|
"PRU": "Financials",
|
||||||
|
"PSA": "Real Estate",
|
||||||
|
"PSKY": "Communication Services",
|
||||||
|
"PSX": "Energy",
|
||||||
|
"PTC": "Information Technology",
|
||||||
|
"PWR": "Industrials",
|
||||||
|
"PYPL": "Financials",
|
||||||
|
"Q": "Information Technology",
|
||||||
|
"QCOM": "Information Technology",
|
||||||
|
"RCL": "Consumer Discretionary",
|
||||||
|
"REG": "Real Estate",
|
||||||
|
"REGN": "Health Care",
|
||||||
|
"RF": "Financials",
|
||||||
|
"RJF": "Financials",
|
||||||
|
"RL": "Consumer Discretionary",
|
||||||
|
"RMD": "Health Care",
|
||||||
|
"ROK": "Industrials",
|
||||||
|
"ROL": "Industrials",
|
||||||
|
"ROP": "Information Technology",
|
||||||
|
"ROST": "Consumer Discretionary",
|
||||||
|
"RSG": "Industrials",
|
||||||
|
"RTX": "Industrials",
|
||||||
|
"RVTY": "Health Care",
|
||||||
|
"SATS": "Communication Services",
|
||||||
|
"SBAC": "Real Estate",
|
||||||
|
"SBUX": "Consumer Discretionary",
|
||||||
|
"SCHW": "Financials",
|
||||||
|
"SHW": "Materials",
|
||||||
|
"SJM": "Consumer Staples",
|
||||||
|
"SLB": "Energy",
|
||||||
|
"SMCI": "Information Technology",
|
||||||
|
"SNA": "Industrials",
|
||||||
|
"SNDK": "Information Technology",
|
||||||
|
"SNPS": "Information Technology",
|
||||||
|
"SO": "Utilities",
|
||||||
|
"SOLV": "Health Care",
|
||||||
|
"SPCX": "Industrials",
|
||||||
|
"SPG": "Real Estate",
|
||||||
|
"SPGI": "Financials",
|
||||||
|
"SRE": "Utilities",
|
||||||
|
"STE": "Health Care",
|
||||||
|
"STLD": "Materials",
|
||||||
|
"STT": "Financials",
|
||||||
|
"STX": "Information Technology",
|
||||||
|
"STZ": "Consumer Staples",
|
||||||
|
"SW": "Materials",
|
||||||
|
"SWK": "Industrials",
|
||||||
|
"SWKS": "Information Technology",
|
||||||
|
"SYF": "Financials",
|
||||||
|
"SYK": "Health Care",
|
||||||
|
"SYY": "Consumer Staples",
|
||||||
|
"T": "Communication Services",
|
||||||
|
"TAP": "Consumer Staples",
|
||||||
|
"TDG": "Industrials",
|
||||||
|
"TDY": "Information Technology",
|
||||||
|
"TECH": "Health Care",
|
||||||
|
"TEL": "Information Technology",
|
||||||
|
"TER": "Information Technology",
|
||||||
|
"TFC": "Financials",
|
||||||
|
"TGT": "Consumer Staples",
|
||||||
|
"TJX": "Consumer Discretionary",
|
||||||
|
"TKO": "Communication Services",
|
||||||
|
"TMO": "Health Care",
|
||||||
|
"TMUS": "Communication Services",
|
||||||
|
"TPL": "Energy",
|
||||||
|
"TPR": "Consumer Discretionary",
|
||||||
|
"TRGP": "Energy",
|
||||||
|
"TRMB": "Information Technology",
|
||||||
|
"TROW": "Financials",
|
||||||
|
"TRV": "Financials",
|
||||||
|
"TSCO": "Consumer Discretionary",
|
||||||
|
"TSLA": "Consumer Discretionary",
|
||||||
|
"TSN": "Consumer Staples",
|
||||||
|
"TT": "Industrials",
|
||||||
|
"TTD": "Communication Services",
|
||||||
|
"TTWO": "Communication Services",
|
||||||
|
"TXN": "Information Technology",
|
||||||
|
"TXT": "Industrials",
|
||||||
|
"TYL": "Information Technology",
|
||||||
|
"UAL": "Industrials",
|
||||||
|
"UBER": "Industrials",
|
||||||
|
"UDR": "Real Estate",
|
||||||
|
"UHS": "Health Care",
|
||||||
|
"ULTA": "Consumer Discretionary",
|
||||||
|
"UNH": "Health Care",
|
||||||
|
"UNP": "Industrials",
|
||||||
|
"UPS": "Industrials",
|
||||||
|
"URI": "Industrials",
|
||||||
|
"USB": "Financials",
|
||||||
|
"V": "Financials",
|
||||||
|
"VICI": "Real Estate",
|
||||||
|
"VLO": "Energy",
|
||||||
|
"VLTO": "Industrials",
|
||||||
|
"VMC": "Materials",
|
||||||
|
"VRSK": "Industrials",
|
||||||
|
"VRSN": "Information Technology",
|
||||||
|
"VRT": "Industrials",
|
||||||
|
"VRTX": "Health Care",
|
||||||
|
"VST": "Utilities",
|
||||||
|
"VTR": "Real Estate",
|
||||||
|
"VTRS": "Health Care",
|
||||||
|
"VZ": "Communication Services",
|
||||||
|
"WAB": "Industrials",
|
||||||
|
"WAT": "Health Care",
|
||||||
|
"WBD": "Communication Services",
|
||||||
|
"WDAY": "Information Technology",
|
||||||
|
"WDC": "Information Technology",
|
||||||
|
"WEC": "Utilities",
|
||||||
|
"WELL": "Real Estate",
|
||||||
|
"WFC": "Financials",
|
||||||
|
"WM": "Industrials",
|
||||||
|
"WMB": "Energy",
|
||||||
|
"WMT": "Consumer Staples",
|
||||||
|
"WRB": "Financials",
|
||||||
|
"WSM": "Consumer Discretionary",
|
||||||
|
"WST": "Health Care",
|
||||||
|
"WTW": "Financials",
|
||||||
|
"WY": "Real Estate",
|
||||||
|
"WYNN": "Consumer Discretionary",
|
||||||
|
"XEL": "Utilities",
|
||||||
|
"XOM": "Energy",
|
||||||
|
"XYL": "Industrials",
|
||||||
|
"XYZ": "Financials",
|
||||||
|
"YUM": "Consumer Discretionary",
|
||||||
|
"ZBH": "Health Care",
|
||||||
|
"ZBRA": "Information Technology",
|
||||||
|
"ZTS": "Health Care"
|
||||||
|
},
|
||||||
|
"meta": {
|
||||||
|
"built_at": "2026-07-19T05:35:41.184460+00:00",
|
||||||
|
"coverage": {
|
||||||
|
"by_sector": {
|
||||||
|
"Communication Services": 23,
|
||||||
|
"Consumer Defensive": 2,
|
||||||
|
"Consumer Discretionary": 47,
|
||||||
|
"Consumer Staples": 34,
|
||||||
|
"Energy": 22,
|
||||||
|
"Financial Services": 1,
|
||||||
|
"Financials": 75,
|
||||||
|
"Health Care": 58,
|
||||||
|
"Industrials": 81,
|
||||||
|
"Information Technology": 72,
|
||||||
|
"Materials": 26,
|
||||||
|
"Real Estate": 31,
|
||||||
|
"Technology": 2,
|
||||||
|
"Utilities": 31
|
||||||
|
},
|
||||||
|
"mapped": 505,
|
||||||
|
"mapped_pct": 99.8,
|
||||||
|
"universe": 506,
|
||||||
|
"with_etf": 505
|
||||||
|
},
|
||||||
|
"fmp_requests": 10,
|
||||||
|
"from_existing": 0,
|
||||||
|
"from_fmp": 9,
|
||||||
|
"from_sp500_csv": 496,
|
||||||
|
"snapshot": "C:\\Workspace\\signal-platform\\backtest_snapshots\\prod.sqlite",
|
||||||
|
"still_missing": [
|
||||||
|
"RHM"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"schema_version": 1
|
||||||
|
}
|
||||||
@@ -0,0 +1,202 @@
|
|||||||
|
# Earnings gap diagnostic + SUE / PEAD (Tier-1 alpha research)
|
||||||
|
|
||||||
|
**Status:** **PARK** (incomplete earnings coverage; SUE fails iron rule on available sample).
|
||||||
|
**Branch:** `research/earnings-gap-and-sue`
|
||||||
|
**Production impact:** none. Local research only. **No filters shipped from 2a.**
|
||||||
|
**Artifacts:** `reports/earnings-gap-sue-20260719-093129.json` (+ companion `.md`)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Pre-registration (locked before first research run)
|
||||||
|
|
||||||
|
### Data
|
||||||
|
|
||||||
|
- Historical earnings calendar for the production universe over the full snapshot
|
||||||
|
window (and deeper if the feed provides it).
|
||||||
|
- Preferred source: FMP **date-range earnings-calendar** (bulk). If unavailable on
|
||||||
|
free tier, fall back to per-symbol `/stable/earnings` with request accounting.
|
||||||
|
- Store in a real local table `earnings_events` (symbol + announce_date key).
|
||||||
|
- Point-in-time: a surprise is usable only from **announce date + 1 trading day**
|
||||||
|
onward.
|
||||||
|
|
||||||
|
### Experiment 2a — earnings-gap risk (defense, report-only)
|
||||||
|
|
||||||
|
Join simulated production-config trades (`fill_mode=close`) with earnings dates.
|
||||||
|
|
||||||
|
**Pre-registered questions:**
|
||||||
|
|
||||||
|
1. What fraction of losses worse than **−1R** occur with an earnings announcement
|
||||||
|
**between entry and exit** (inclusive of the holding window)?
|
||||||
|
2. What is the mean R of entries taken within **3 trading days BEFORE** an
|
||||||
|
announcement vs all other entries — report **both tails** of the R
|
||||||
|
distribution (rule 4: any earnings-avoid entry filter is presumed guilty of
|
||||||
|
right-tail trimming until the win distribution shows otherwise)?
|
||||||
|
|
||||||
|
**Output:** distributions and counts only.
|
||||||
|
**No filter is shipped.** If numbers argue for a filter → report and stop.
|
||||||
|
|
||||||
|
### Experiment 2b — SUE / PEAD (offense)
|
||||||
|
|
||||||
|
Signal `sue_latest`:
|
||||||
|
|
||||||
|
\[
|
||||||
|
\text{SUE} = \frac{\text{actual} - \text{estimate}}{\sigma(\text{trailing 8 surprises})}
|
||||||
|
\]
|
||||||
|
|
||||||
|
Fallback if estimate history is thin: scale surprise by price.
|
||||||
|
Carry forward from announce+1 for **63 trading days**, else NaN (name drops out
|
||||||
|
of that cross-section).
|
||||||
|
|
||||||
|
**Iron rule (IC harness):** mean weekly Spearman IC on non-overlapping weeks;
|
||||||
|
\|mean IC\| ≥ ~0.03, **positive** sign (drift), `reliable: true` (≥12 windows).
|
||||||
|
|
||||||
|
Always side-by-side with `mom_12_1` and `mom_12_1_resid` on **identical**
|
||||||
|
cross-sections.
|
||||||
|
|
||||||
|
Also report **momentum-conditional** IC (within top momentum quintile).
|
||||||
|
|
||||||
|
**If it passes iron rule:** STOP and report. Book-integration design is a
|
||||||
|
separate human-approved step — do not wire.
|
||||||
|
|
||||||
|
### Verdict labels
|
||||||
|
|
||||||
|
| label | meaning |
|
||||||
|
|---|---|
|
||||||
|
| **PROMOTE** | (2b only) iron rule cleared → human designs tilt/gate |
|
||||||
|
| **PARK** | Interesting but incomplete / weak |
|
||||||
|
| **DEAD** | No edge / diagnostic argues against action |
|
||||||
|
| **REPORT-ONLY** | (2a) always — never auto-filter |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Data provenance
|
||||||
|
|
||||||
|
| item | result |
|
||||||
|
|---|---|
|
||||||
|
| Snapshot | `backtest_snapshots/prod.sqlite` (506 names) |
|
||||||
|
| FMP bulk `earnings-calendar` | **402 Premium** — not available on free tier |
|
||||||
|
| FMP per-symbol `/stable/earnings` | used; hit daily rate limit ~225 reqs |
|
||||||
|
| Alpha Vantage `EARNINGS` | used for +24 symbols (announce = `reportedDate`) |
|
||||||
|
| Symbols with events | **48 / 506 (9.5%)** |
|
||||||
|
| Total events | 5,612 (5,018 with actual+estimate) |
|
||||||
|
| Announce range | 1985-08-31 → 2026-07-16 |
|
||||||
|
| FMP requests (first day) | 260 FMP + 25 AV (see `reports/earnings-backfill-status.json`) |
|
||||||
|
|
||||||
|
**Incomplete backfill is first-class.** 2a under-detects earnings overlaps; 2b SUE
|
||||||
|
cross-section averages **~47 names**, not ~500. Resume:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Day N (FMP free ~250/day; AV free ~25/day — prefer FMP after reset)
|
||||||
|
python scripts/backfill_earnings_events.py \
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite \
|
||||||
|
--provider fmp --force-symbol --limit 250 --sleep 0.4
|
||||||
|
|
||||||
|
# When done==506:
|
||||||
|
python scripts/run_earnings_research.py \
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite \
|
||||||
|
--workers 6 --allow-spawn
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
Generated: `2026-07-19T09:31:29`
|
||||||
|
|
||||||
|
### 2a — Earnings-gap risk (report-only)
|
||||||
|
|
||||||
|
Production book sim: Sharpe 2.09 (SE 0.497), CAGR 51.6%, max DD 21.4%, **322 trades**,
|
||||||
|
`fill_mode=close`.
|
||||||
|
|
||||||
|
#### Q1 — Losses worse than −1R with earnings in hold
|
||||||
|
|
||||||
|
| metric | value |
|
||||||
|
|---|---:|
|
||||||
|
| n losses < −1R | 28 |
|
||||||
|
| of which earnings in hold | **1** |
|
||||||
|
| fraction | **3.6%** |
|
||||||
|
| all trades with earnings in hold | 14 / 322 (4.4%) |
|
||||||
|
|
||||||
|
**Read:** On incomplete earnings labels this is a **lower bound** on earnings
|
||||||
|
overlap, not a clean “earnings rarely hurt.” Do **not** conclude earnings risk is
|
||||||
|
immaterial until coverage ≥ ~95% of the book’s names.
|
||||||
|
|
||||||
|
#### Q2 — Entry within 3 trading days before announce (both tails)
|
||||||
|
|
||||||
|
| cohort | n | mean R | win rate | p05 | p50 | p95 | max |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
| pre-earn (≤3d before) | **4** | 1.94 | 50% | −1.24 | 1.12 | 6.26 | 6.84 |
|
||||||
|
| other | 318 | 0.70 | 37% | −1.11 | −0.83 | 6.08 | **12.87** |
|
||||||
|
| all | 322 | 0.71 | 37% | −1.12 | −0.83 | 6.22 | 12.87 |
|
||||||
|
|
||||||
|
**Tail-trim presumption:** n=4 is not a sample. Point estimate does **not** show
|
||||||
|
right-tail destruction of pre-earn entries (p95 similar; max actually higher in
|
||||||
|
“other”). **No earnings-avoid filter is supported.** Re-run after full backfill.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### 2b — SUE / PEAD IC
|
||||||
|
|
||||||
|
#### Full-universe harness (mom on ~500; SUE only where labeled)
|
||||||
|
|
||||||
|
| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable |
|
||||||
|
|---|---:|---:|---:|---:|---|
|
||||||
|
| mom_12_1_sector_resid | 0.0578 | 2.34 | 35 | 497.7 | true |
|
||||||
|
| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true |
|
||||||
|
| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true |
|
||||||
|
| **sue_latest** | **0.0172** | **0.6** | 44 | **47.4** | true |
|
||||||
|
| fip_id | −0.045 | −2.91 | 35 | 497.7 | true |
|
||||||
|
|
||||||
|
#### Identical SUE subset (fair side-by-side — use this while coverage is thin)
|
||||||
|
|
||||||
|
| signal | mean_ic | ic_t_stat | weeks | avg_N |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| sue_latest | 0.0172 | 0.6 | 44 | 47.4 |
|
||||||
|
| mom_12_1 | −0.0174 | −0.42 | 35 | 47.3 |
|
||||||
|
| mom_12_1_resid | −0.0104 | −0.27 | 35 | 47.3 |
|
||||||
|
|
||||||
|
On the thin labeled subset, momentum itself is noise — so the subset is not yet
|
||||||
|
a meaningful PEAD test.
|
||||||
|
|
||||||
|
#### Momentum-conditional SUE (top mom quintile)
|
||||||
|
|
||||||
|
| metric | value |
|
||||||
|
|---|---:|
|
||||||
|
| mean IC | **−0.0065** |
|
||||||
|
| t | −0.1 |
|
||||||
|
| weeks | 35 |
|
||||||
|
|
||||||
|
Wrong sign vs “ride positive surprises inside the momentum gate.”
|
||||||
|
|
||||||
|
**Iron rule:** fail (\|IC\| 0.017 < 0.03; t 0.6). **No promote.**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Verdict
|
||||||
|
|
||||||
|
| piece | verdict |
|
||||||
|
|---|---|
|
||||||
|
| **2a earnings-gap** | **REPORT-ONLY** — no filter. Coverage too thin for risk claims; tails do not argue for an avoid-filter on n=4. |
|
||||||
|
| **2b SUE** | **PARK** (effectively not green). Mild positive IC on ~48 names; fails iron bar; mom-conditional flat/negative. Re-score after full backfill before DEAD. |
|
||||||
|
| **Production** | **no change** |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## What a human must decide next
|
||||||
|
|
||||||
|
1. Resume multi-day earnings backfill to **506/506**, then re-run
|
||||||
|
`run_earnings_research.py` (heavy — MacBook OK).
|
||||||
|
2. Do **not** ship an earnings-avoid entry filter from 2a.
|
||||||
|
3. Do **not** wire SUE until a full-coverage IC clears the iron rule (and
|
||||||
|
preferably mom-conditional > 0).
|
||||||
|
4. Do not merge into main strategy docs without review.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Implementation notes
|
||||||
|
|
||||||
|
| piece | role |
|
||||||
|
|---|---|
|
||||||
|
| `scripts/backfill_earnings_events.py` | bulk attempt → FMP/AV per-symbol; `earnings_events` + meta on snapshot |
|
||||||
|
| `scripts/run_earnings_research.py` | 2a trade join + 2b SUE IC / mom-conditional |
|
||||||
|
| Snapshot table `earnings_events` | real table (not SystemSetting JSON) |
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
# History-depth extension (Tier-1 alpha research)
|
||||||
|
|
||||||
|
**Status:** PRE-REGISTERED — run on MacBook (heavy I/O + full harness).
|
||||||
|
**Branch:** `research/history-depth-extension` (create from latest research stack).
|
||||||
|
**Production impact:** none. **Do not retune any production knob on deep history.**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Pre-registration (locked before rebuild)
|
||||||
|
|
||||||
|
### Motivation
|
||||||
|
|
||||||
|
All current conclusions rest on ~35 non-overlapping weekly windows in essentially
|
||||||
|
one post-2021 regime. Extending history toward max Alpaca daily-bar depth adds
|
||||||
|
the 2018 vol shock and full 2020 crash (where the feed allows).
|
||||||
|
|
||||||
|
### Protocol
|
||||||
|
|
||||||
|
1. **Empirical coverage first** — bars per calendar year per symbol; document
|
||||||
|
where the feed thins out. Do **not** assume a uniform start date.
|
||||||
|
2. **Rebuild the research snapshot completely** from prod source + max history
|
||||||
|
per symbol (`Adjustment.SPLIT`, ~200 req/min pacing via existing extender).
|
||||||
|
3. **Race guard (rule 6)** — refuse analysis until completion manifest is
|
||||||
|
`complete=true` and live counts match.
|
||||||
|
4. **Re-run full signal harness** (all existing signals incl. sector residual /
|
||||||
|
SUE if present) on the extended window.
|
||||||
|
5. **Report per signal:** mean IC, t, window count, and **era split**
|
||||||
|
(pre-/post-2021) — diagnostic only, **not a tuning input**.
|
||||||
|
6. **Log prominently:** survivorship bias grows with depth (today’s constituents
|
||||||
|
backfilled). Absolute Sharpe/CAGR on deep history is optimistic; payload is
|
||||||
|
**relative** signal comparisons and IC stability, not levels.
|
||||||
|
7. **Do not retune** production knobs. If a knob’s confirmation looks
|
||||||
|
overturned on deep history → report only; human decides.
|
||||||
|
|
||||||
|
### Success / interpretation (not promotion of a new signal)
|
||||||
|
|
||||||
|
| outcome | meaning |
|
||||||
|
|---|---|
|
||||||
|
| Sector residual still ≥ market residual on deep IC + stable sign | strengthens Task 1 PROMOTE case |
|
||||||
|
| Sector residual collapses pre-2021 | **PARK** Task 1 wire-in |
|
||||||
|
| SUE remains weak after full earnings + depth | **DEAD** SUE for this stack |
|
||||||
|
| Any production knob looks worse deep | report; no auto-retune |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## MacBook runbook
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 0. Repo + env
|
||||||
|
git fetch origin
|
||||||
|
git checkout research/earnings-gap-and-sue # or history-depth branch once pushed
|
||||||
|
# ensure .env has ALPACA_* (and FMP if resuming earnings)
|
||||||
|
|
||||||
|
# 1. (Optional) finish earnings backfill first — multi-day free tier
|
||||||
|
python scripts/backfill_earnings_events.py \
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite \
|
||||||
|
--provider fmp --force-symbol --limit 250 --sleep 0.35
|
||||||
|
|
||||||
|
# 2. Coverage probe (before long rebuild)
|
||||||
|
python scripts/run_history_depth_research.py --phase coverage \
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite
|
||||||
|
|
||||||
|
# 3. Full deep rebuild of research.sqlite (LONG — Alpaca per symbol)
|
||||||
|
# Clears prior completion manifest; writes complete=true only at end.
|
||||||
|
python scripts/extend_snapshot_universe.py \
|
||||||
|
--source backtest_snapshots/prod.sqlite \
|
||||||
|
--output backtest_snapshots/research.sqlite \
|
||||||
|
--force-copy \
|
||||||
|
--history-days 5000 \
|
||||||
|
--min-bars 260 \
|
||||||
|
--sleep 0.15
|
||||||
|
|
||||||
|
# 4. Also refresh SPY + sector ETFs to the same depth on BOTH snapshots
|
||||||
|
python scripts/fetch_sector_etfs_to_snapshot.py \
|
||||||
|
--snapshot backtest_snapshots/research.sqlite --history-days 5000
|
||||||
|
python scripts/fetch_sector_etfs_to_snapshot.py \
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite --history-days 5000
|
||||||
|
|
||||||
|
# 5. Harness + era split (after race guard passes)
|
||||||
|
python scripts/run_history_depth_research.py --phase harness \
|
||||||
|
--snapshot backtest_snapshots/research.sqlite \
|
||||||
|
--workers 8 --allow-spawn
|
||||||
|
|
||||||
|
# 6. Copy reports/ + docs/research/history-depth-extension.md results back
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Data provenance
|
||||||
|
|
||||||
|
*(filled at run time)*
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
*(filled at run time)*
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Verdict
|
||||||
|
|
||||||
|
**Pending MacBook run.**
|
||||||
|
|
||||||
|
## What a human must decide next
|
||||||
|
|
||||||
|
- Do not retune production from deep history without explicit review.
|
||||||
|
- Use relative IC stability to accept/reject Task 1 sector residual wire-in.
|
||||||
@@ -0,0 +1,237 @@
|
|||||||
|
# Sector-residual momentum (Tier-1 alpha research)
|
||||||
|
|
||||||
|
**Status:** **PROMOTE (to human design decision only)** — IC + A/B bars cleared; **do not ship**.
|
||||||
|
**Branch:** `research/sector-residual-momentum`
|
||||||
|
**Production impact:** none. Local research only. No scheduler / gate / prod-config changes.
|
||||||
|
**Artifacts:** `reports/sector-residual-20260719-083356.json` (+ companion `.md`)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Pre-registration (locked before first research run)
|
||||||
|
|
||||||
|
### Hypothesis
|
||||||
|
|
||||||
|
Residualizing 12–1 momentum against the sector, not only the market, reduces
|
||||||
|
factor volatility at similar return (Blitz / Huij / Martens-style) → higher
|
||||||
|
Sharpe on the production book when the residual replaces market-only residual
|
||||||
|
as the momentum leg.
|
||||||
|
|
||||||
|
### Signals (candidates)
|
||||||
|
|
||||||
|
| signal | construction |
|
||||||
|
|---|---|
|
||||||
|
| `mom_12_1_sector_resid` | Two-factor residual vs SPY + ticker’s sector ETF. Same window as `mom_12_1_resid`: ≥100 daily obs, 252-bar lookback, 21-bar skip; two-factor OLS betas **without intercept**; cumulate residual returns over the formation window. |
|
||||||
|
| `mom_12_1_sector_demeaned` | Plain `mom_12_1` minus the **cross-sectional** mean of `mom_12_1` within the same GICS sector that week (≥2 names in sector). No regression. |
|
||||||
|
|
||||||
|
### Baselines (same run, same cross-sections — iron rule)
|
||||||
|
|
||||||
|
Always report side-by-side with:
|
||||||
|
|
||||||
|
- `mom_12_1`
|
||||||
|
- `mom_12_1_resid`
|
||||||
|
|
||||||
|
Computed on the **identical** weekly non-overlapping cross-sections in this run.
|
||||||
|
Never compare against IC numbers from another report.
|
||||||
|
|
||||||
|
### Iron rule (IC harness)
|
||||||
|
|
||||||
|
Source of truth: `_signal_evaluation` in `app/services/backtest_service.py`.
|
||||||
|
|
||||||
|
- Mean weekly Spearman IC on **non-overlapping** weekly windows
|
||||||
|
- Bar: \|mean IC\| ≥ ~0.03, **consistent positive sign**, `reliable: true` (≥ 12 windows)
|
||||||
|
|
||||||
|
### Promotion to portfolio A/B (candidate → book)
|
||||||
|
|
||||||
|
A candidate promotes to A/B **only if**:
|
||||||
|
|
||||||
|
1. It clears the iron-rule bar **and**
|
||||||
|
2. Its IC **t-stat ≥** that of `mom_12_1_resid` on the same cross-sections.
|
||||||
|
|
||||||
|
### Portfolio A/B grading (if and only if IC promotion fires)
|
||||||
|
|
||||||
|
- Swap candidate in as the **momentum leg** of the production 80/20 momentum/vol
|
||||||
|
rank **and** as the gate-percentile signal.
|
||||||
|
- `fill_mode=close`, `COST_PER_SIDE = 0.001`, full config otherwise unchanged.
|
||||||
|
- Validation window = entries ≥ **2024-07-01** (call it **validation**, not
|
||||||
|
holdout — contaminated by prior experiments).
|
||||||
|
- Pre-registered promotion bar:
|
||||||
|
- validation Sharpe ≥ control − 0.5·SE
|
||||||
|
- full-period Sharpe and max-DD **not worse** than control
|
||||||
|
- Report Lo / Mertens-adjusted SEs.
|
||||||
|
|
||||||
|
### Optional sector-cap sub-experiment
|
||||||
|
|
||||||
|
Only if labels are in **and** A/B ran: max **3** positions per sector in the
|
||||||
|
10-slot book. Same A/B grading. **Tail-trim presumption of guilt** (rule 4):
|
||||||
|
report entry counts and both tails of the R distribution. Rising win rate with
|
||||||
|
falling Sharpe/CAGR = red flag → do not promote.
|
||||||
|
|
||||||
|
**This run:** sector-cap arm **not executed** (optional; A/B unconstrained book
|
||||||
|
only). Can be a human-approved follow-up.
|
||||||
|
|
||||||
|
### Verdict labels
|
||||||
|
|
||||||
|
| label | meaning |
|
||||||
|
|---|---|
|
||||||
|
| **PROMOTE** | Clears pre-registered bar; human decides next (wire design separate) |
|
||||||
|
| **PARK** | Inconclusive / weak; keep machinery, no book change |
|
||||||
|
| **DEAD** | Failed iron rule or worse than residual baseline with clear sign |
|
||||||
|
|
||||||
|
### Explicit non-goals
|
||||||
|
|
||||||
|
- No production deploy from this doc
|
||||||
|
- Do not resurrect: take-profit exits, EV gate, regime entry-blocking,
|
||||||
|
inverse-vol sizing, gap-caps, unconditional FIP filter
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Data provenance
|
||||||
|
|
||||||
|
### Snapshot race guard
|
||||||
|
|
||||||
|
| check | result |
|
||||||
|
|---|---|
|
||||||
|
| Snapshot path | `backtest_snapshots/prod.sqlite` |
|
||||||
|
| Manifest | none (expected for prod snapshot); bar-count sanity applied |
|
||||||
|
| Tickers / OHLCV | **506** / **629,263** |
|
||||||
|
| Bars min / avg / max | 14 / 1246.1 / 1261 |
|
||||||
|
| OHLCV range | 2021-06-24 → 2026-07-02 |
|
||||||
|
| Partial-build red flags | none (avg bars healthy) |
|
||||||
|
|
||||||
|
Integrity fingerprint on same run: `fip_id` mean IC **−0.045** / t **−2.91**
|
||||||
|
(35 weeks, N≈498) — matches the established prod fingerprint.
|
||||||
|
|
||||||
|
### Sector labels
|
||||||
|
|
||||||
|
| source | count |
|
||||||
|
|---|---:|
|
||||||
|
| Public S&P 500 GICS CSV | 496 newly filled |
|
||||||
|
| FMP profile requests | 10 (all missing after CSV) |
|
||||||
|
| Mapped / universe | **505 / 506 (99.8%)** |
|
||||||
|
| With mappable ETF | 505 |
|
||||||
|
| Still missing | **RHM** only |
|
||||||
|
|
||||||
|
Persist path: `data/research/ticker_sector_map.json`.
|
||||||
|
|
||||||
|
FMP aliases (`Technology`, `Consumer Defensive`, `Financial Services`) map to
|
||||||
|
SPDRs via the alias table in `app/services/sector_map.py`.
|
||||||
|
|
||||||
|
### Sector ETFs in `benchmark_prices` (auxiliary only — not tradable)
|
||||||
|
|
||||||
|
| symbol | bars | min date | max date |
|
||||||
|
|---|---:|---|---|
|
||||||
|
| SPY | 1516 | 2020-07-06 | 2026-07-17 |
|
||||||
|
| XLB…XLY (11) | 1512 each | 2020-07-10 | 2026-07-17 |
|
||||||
|
|
||||||
|
Fetched via Alpaca `Adjustment.SPLIT` into **`benchmark_prices`** (same table as
|
||||||
|
SPY) so they never enter the ticker universe or candidate replay.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
Generated: `2026-07-19T08:33:56`
|
||||||
|
|
||||||
|
### IC harness (identical cross-sections, production 506-name universe)
|
||||||
|
|
||||||
|
| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable | ic+_pct | quintile spread |
|
||||||
|
|---|---:|---:|---:|---:|---|---:|---:|
|
||||||
|
| **mom_12_1_sector_resid** | **0.0578** | **2.34** | 35 | 497.7 | true | 65.7 | 0.0245 |
|
||||||
|
| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true | 60.0 | 0.0207 |
|
||||||
|
| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true | 65.7 | 0.0206 |
|
||||||
|
| mom_12_1_sector_demeaned | 0.0340 | 1.32 | 35 | 496.7 | true | 62.9 | 0.0154 |
|
||||||
|
|
||||||
|
### IC promotion grades
|
||||||
|
|
||||||
|
| candidate | iron rule | t ≥ resid | promote_to_ab |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `mom_12_1_sector_resid` | pass (IC 0.058, +sign, reliable) | **yes** (2.34 ≥ 1.98) | **yes** |
|
||||||
|
| `mom_12_1_sector_demeaned` | pass (IC 0.034, +sign, reliable) | **no** (1.32 < 1.98) | **no** |
|
||||||
|
|
||||||
|
### Portfolio A/B — `mom_12_1_sector_resid` as residual leg
|
||||||
|
|
||||||
|
Config: production 80/20 residual/high-vol rank + gate percentile, `fill_mode=close`,
|
||||||
|
cost 10 bps/side, ATR trail / gate-reset re-entry as live. Validation split
|
||||||
|
2024-07-01.
|
||||||
|
|
||||||
|
| window | arm | Sharpe | Sharpe SE (Mertens) | CAGR % | max DD % | trades | n_days |
|
||||||
|
|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
| train | control (resid) | 1.30 | 0.685 | 29.2 | 21.4 | 176 | 525 |
|
||||||
|
| train | treatment (sector resid) | **1.57** | 0.677 | **35.5** | **19.8** | 176 | 530 |
|
||||||
|
| validation | control | **2.92** | 0.709 | **76.3** | **11.7** | 150 | 501 |
|
||||||
|
| validation | treatment | 2.57 | 0.701 | 66.3 | 14.8 | 163 | 501 |
|
||||||
|
| full | control | 2.09 | 0.497 | 51.6 | 21.4 | 322 | 1000 |
|
||||||
|
| full | treatment | 2.09 | 0.491 | 51.0 | **19.8** | 337 | 1005 |
|
||||||
|
|
||||||
|
**Pre-registered A/B checks**
|
||||||
|
|
||||||
|
| check | result |
|
||||||
|
|---|---|
|
||||||
|
| val Sharpe ≥ control − 0.5·SE | **pass** (2.57 ≥ 2.92 − 0.5×0.701 = 2.5695) — **knife-edge** |
|
||||||
|
| full Sharpe not worse | **pass** (2.09 = 2.09) |
|
||||||
|
| full max DD not worse | **pass** (19.8 < 21.4) |
|
||||||
|
|
||||||
|
Qualified long candidates: control 1086 vs treatment 1210 (sector residual
|
||||||
|
gates a slightly larger set).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Verdict
|
||||||
|
|
||||||
|
| signal | verdict | note |
|
||||||
|
|---|---|---|
|
||||||
|
| **`mom_12_1_sector_resid`** | **PROMOTE → human wire-in decision** | IC modestly beats market residual; A/B clears pre-reg bar narrowly. **Do not ship from this branch.** |
|
||||||
|
| **`mom_12_1_sector_demeaned`** | **DEAD** (for promotion) | Iron-rule IC magnitude ok, but t-stat loses to `mom_12_1_resid`. Cheap variant not competitive. |
|
||||||
|
|
||||||
|
### Read carefully (for the human)
|
||||||
|
|
||||||
|
1. **IC edge is real but small.** Sector residual IC 0.0578 / t 2.34 vs market
|
||||||
|
residual 0.0552 / t 1.98 on the **same** 35 windows — better consistency
|
||||||
|
(ic+ 65.7% vs 60%) and slightly higher mean, not a different factor class.
|
||||||
|
2. **A/B is not a clear Sharpe win.** Full-period Sharpe is flat (2.09).
|
||||||
|
Validation Sharpe is **lower** than control (2.57 vs 2.92) and only clears
|
||||||
|
the pre-registered “within 0.5 SE” cushion by ~0.001. Train improves;
|
||||||
|
validation worsens — classic regime-split noise on ~2 years.
|
||||||
|
3. **Risk side is friendly.** Full max DD improves (19.8% vs 21.4%); train DD
|
||||||
|
also better. Matches the “lower factor vol” half of the hypothesis more than
|
||||||
|
the “higher Sharpe” half on this window.
|
||||||
|
4. **Survivorship / short history.** Same caveats as all current research:
|
||||||
|
today’s constituents, ~35 independent weekly windows, one post-2021 regime
|
||||||
|
dominant. Task 3 (history depth) should re-check IC stability before any
|
||||||
|
wire-in.
|
||||||
|
5. **Not shipped.** Machinery lives on the research branch; production residual
|
||||||
|
path is untouched.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## What a human must decide next
|
||||||
|
|
||||||
|
1. **Accept or reject** replacing `mom_12_1_resid` with `mom_12_1_sector_resid`
|
||||||
|
as the production residual (gate + 80/20 mom leg), **or** keep market residual
|
||||||
|
and treat sector residual as research-only.
|
||||||
|
2. If leaning accept: require **Task 3 history-depth** confirmation (IC era split
|
||||||
|
pre/post-2021) before any production PR.
|
||||||
|
3. Optional: run **sector-cap ≤3** A/B with full tail diagnostics (not run here).
|
||||||
|
4. **Do not** merge this verdict into main strategy docs without review.
|
||||||
|
5. Wire-in design (live sector map refresh, ETF series ops, fallback when sector
|
||||||
|
missing) is a **separate** approved engineering step.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Implementation notes (research machinery)
|
||||||
|
|
||||||
|
| piece | role |
|
||||||
|
|---|---|
|
||||||
|
| `app/services/sector_map.py` | GICS→ETF map, symbol normalise, JSON load/save |
|
||||||
|
| `app/services/backtest_service.py` | multi-factor residual; `mom_12_1_sector_resid` in `_signal_values`; demean inject |
|
||||||
|
| `scripts/build_ticker_sector_map.py` | SP500 CSV + FMP gap fill |
|
||||||
|
| `scripts/fetch_sector_etfs_to_snapshot.py` | Alpaca → snapshot `benchmark_prices` |
|
||||||
|
| `scripts/run_sector_residual_research.py` | race guard, IC, optional A/B, reports |
|
||||||
|
| `data/research/ticker_sector_map.json` | persisted labels (research only) |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Artifacts
|
||||||
|
|
||||||
|
- JSON: `reports/sector-residual-20260719-083356.json`
|
||||||
|
- MD copy: `reports/sector-residual-20260719-083356.md`
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
{
|
||||||
|
"mode": "per_symbol",
|
||||||
|
"fmp_requests": 25,
|
||||||
|
"events_written_this_run": 2541,
|
||||||
|
"total_events": 5612,
|
||||||
|
"symbols_done": 48,
|
||||||
|
"symbols_universe": 506,
|
||||||
|
"announce_date_range": {
|
||||||
|
"min": "1985-08-31",
|
||||||
|
"max": "2026-07-16"
|
||||||
|
},
|
||||||
|
"events_with_actual_and_estimate": 5018,
|
||||||
|
"budget": 25,
|
||||||
|
"complete": false
|
||||||
|
}
|
||||||
@@ -0,0 +1,331 @@
|
|||||||
|
{
|
||||||
|
"generated_at": "2026-07-19T09:31:29.078611",
|
||||||
|
"data_provenance": {
|
||||||
|
"snapshot": "C:\\Workspace\\signal-platform\\backtest_snapshots\\prod.sqlite",
|
||||||
|
"n_earnings_events": 5612,
|
||||||
|
"backfill_meta": {
|
||||||
|
"done": 48,
|
||||||
|
"universe_tickers": 506
|
||||||
|
},
|
||||||
|
"announce_range": {
|
||||||
|
"min": "1985-08-31",
|
||||||
|
"max": "2026-07-16"
|
||||||
|
},
|
||||||
|
"with_actual_and_estimate": 5018
|
||||||
|
},
|
||||||
|
"experiment_2a": {
|
||||||
|
"sim_summary": {
|
||||||
|
"sharpe": 2.09,
|
||||||
|
"sharpe_se": 0.497,
|
||||||
|
"cagr_pct": 51.6,
|
||||||
|
"max_drawdown_pct": 21.4,
|
||||||
|
"trades": 322,
|
||||||
|
"total_return_pct": 424.6
|
||||||
|
},
|
||||||
|
"n_trades_parsed": 322,
|
||||||
|
"q1_losses_worse_than_minus_1r": {
|
||||||
|
"n_losses_lt_minus_1r": 28,
|
||||||
|
"n_with_earnings_in_hold": 1,
|
||||||
|
"fraction_with_earnings": 0.0357,
|
||||||
|
"all_trades_with_earnings_in_hold": 14,
|
||||||
|
"fraction_all_trades_with_earnings": 0.0435
|
||||||
|
},
|
||||||
|
"q2_entry_within_3d_before_announce": {
|
||||||
|
"pre_earn_entries": {
|
||||||
|
"n": 4,
|
||||||
|
"mean": 1.9379,
|
||||||
|
"win_rate": 0.5,
|
||||||
|
"p05": -1.2428,
|
||||||
|
"p25": -0.8833,
|
||||||
|
"p50": 1.1209,
|
||||||
|
"p75": 3.942,
|
||||||
|
"p95": 6.2623,
|
||||||
|
"min": -1.3327,
|
||||||
|
"max": 6.8424
|
||||||
|
},
|
||||||
|
"other_entries": {
|
||||||
|
"n": 318,
|
||||||
|
"mean": 0.6965,
|
||||||
|
"win_rate": 0.3711,
|
||||||
|
"p05": -1.1052,
|
||||||
|
"p25": -1.0,
|
||||||
|
"p50": -0.8259,
|
||||||
|
"p75": 2.1053,
|
||||||
|
"p95": 6.077,
|
||||||
|
"min": -3.2587,
|
||||||
|
"max": 12.8654
|
||||||
|
},
|
||||||
|
"all_entries": {
|
||||||
|
"n": 322,
|
||||||
|
"mean": 0.7119,
|
||||||
|
"win_rate": 0.3727,
|
||||||
|
"p05": -1.1209,
|
||||||
|
"p25": -1.0,
|
||||||
|
"p50": -0.8251,
|
||||||
|
"p75": 2.1595,
|
||||||
|
"p95": 6.2246,
|
||||||
|
"min": -3.2587,
|
||||||
|
"max": 12.8654
|
||||||
|
},
|
||||||
|
"tail_trim_note": "Compare p95/max and mean of pre_earn vs other. Rising win_rate with falling mean/p95 = right-tail trim red flag."
|
||||||
|
},
|
||||||
|
"note": "REPORT-ONLY \u2014 no filter shipped."
|
||||||
|
},
|
||||||
|
"experiment_2b": {
|
||||||
|
"signal_eval_side_by_side": {
|
||||||
|
"mom_12_1": {
|
||||||
|
"signal": "mom_12_1",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0531,
|
||||||
|
"ic_t_stat": 1.61,
|
||||||
|
"ic_positive_pct": 65.7,
|
||||||
|
"mean_quintile_spread": 0.0206,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
"mom_12_1_resid": {
|
||||||
|
"signal": "mom_12_1_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0552,
|
||||||
|
"ic_t_stat": 1.98,
|
||||||
|
"ic_positive_pct": 60.0,
|
||||||
|
"mean_quintile_spread": 0.0207,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
"mom_12_1_sector_resid": {
|
||||||
|
"signal": "mom_12_1_sector_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0578,
|
||||||
|
"ic_t_stat": 2.34,
|
||||||
|
"ic_positive_pct": 65.7,
|
||||||
|
"mean_quintile_spread": 0.0245,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
"mom_12_1_sector_demeaned": {
|
||||||
|
"signal": "mom_12_1_sector_demeaned",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 496.7,
|
||||||
|
"mean_ic": 0.034,
|
||||||
|
"ic_t_stat": 1.32,
|
||||||
|
"ic_positive_pct": 62.9,
|
||||||
|
"mean_quintile_spread": 0.0154,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
"sue_latest": {
|
||||||
|
"signal": "sue_latest",
|
||||||
|
"weeks": 44,
|
||||||
|
"avg_cross_section": 47.4,
|
||||||
|
"mean_ic": 0.0172,
|
||||||
|
"ic_t_stat": 0.6,
|
||||||
|
"ic_positive_pct": 47.7,
|
||||||
|
"mean_quintile_spread": 0.0064,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
"fip_id": {
|
||||||
|
"signal": "fip_id",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": -0.045,
|
||||||
|
"ic_t_stat": -2.91,
|
||||||
|
"ic_positive_pct": 25.7,
|
||||||
|
"mean_quintile_spread": -0.0168,
|
||||||
|
"reliable": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"signal_eval_identical_sue_subset": {
|
||||||
|
"mom_12_1": {
|
||||||
|
"signal": "mom_12_1",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 47.3,
|
||||||
|
"mean_ic": -0.0174,
|
||||||
|
"ic_t_stat": -0.42,
|
||||||
|
"ic_positive_pct": 45.7,
|
||||||
|
"mean_quintile_spread": 0.0077,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
"mom_12_1_resid": {
|
||||||
|
"signal": "mom_12_1_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 47.3,
|
||||||
|
"mean_ic": -0.0104,
|
||||||
|
"ic_t_stat": -0.27,
|
||||||
|
"ic_positive_pct": 51.4,
|
||||||
|
"mean_quintile_spread": 0.0075,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
"sue_latest": {
|
||||||
|
"signal": "sue_latest",
|
||||||
|
"weeks": 44,
|
||||||
|
"avg_cross_section": 47.4,
|
||||||
|
"mean_ic": 0.0172,
|
||||||
|
"ic_t_stat": 0.6,
|
||||||
|
"ic_positive_pct": 47.7,
|
||||||
|
"mean_quintile_spread": 0.0064,
|
||||||
|
"reliable": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"identical_subset_note": "Mom baselines re-scored only on (week, symbol) cells where SUE exists. Use this table when backfill is incomplete \u2014 full-universe mom N is not comparable.",
|
||||||
|
"full_signal_eval": [
|
||||||
|
{
|
||||||
|
"signal": "vol_6m",
|
||||||
|
"weeks": 39,
|
||||||
|
"avg_cross_section": 498.2,
|
||||||
|
"mean_ic": 0.0609,
|
||||||
|
"ic_t_stat": 1.48,
|
||||||
|
"ic_positive_pct": 64.1,
|
||||||
|
"mean_quintile_spread": 0.0337,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1_sector_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0578,
|
||||||
|
"ic_t_stat": 2.34,
|
||||||
|
"ic_positive_pct": 65.7,
|
||||||
|
"mean_quintile_spread": 0.0245,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0552,
|
||||||
|
"ic_t_stat": 1.98,
|
||||||
|
"ic_positive_pct": 60.0,
|
||||||
|
"mean_quintile_spread": 0.0207,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0531,
|
||||||
|
"ic_t_stat": 1.61,
|
||||||
|
"ic_positive_pct": 65.7,
|
||||||
|
"mean_quintile_spread": 0.0206,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1_sector_demeaned",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 496.7,
|
||||||
|
"mean_ic": 0.034,
|
||||||
|
"ic_t_stat": 1.32,
|
||||||
|
"ic_positive_pct": 62.9,
|
||||||
|
"mean_quintile_spread": 0.0154,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "sue_latest",
|
||||||
|
"weeks": 44,
|
||||||
|
"avg_cross_section": 47.4,
|
||||||
|
"mean_ic": 0.0172,
|
||||||
|
"ic_t_stat": 0.6,
|
||||||
|
"ic_positive_pct": 47.7,
|
||||||
|
"mean_quintile_spread": 0.0064,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "trend_200",
|
||||||
|
"weeks": 37,
|
||||||
|
"avg_cross_section": 497.9,
|
||||||
|
"mean_ic": 0.0161,
|
||||||
|
"ic_t_stat": 0.44,
|
||||||
|
"ic_positive_pct": 59.5,
|
||||||
|
"mean_quintile_spread": 0.006,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "reversal_1m",
|
||||||
|
"weeks": 43,
|
||||||
|
"avg_cross_section": 498.7,
|
||||||
|
"mean_ic": 0.0059,
|
||||||
|
"ic_t_stat": 0.22,
|
||||||
|
"ic_positive_pct": 53.5,
|
||||||
|
"mean_quintile_spread": 0.0053,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_6_1",
|
||||||
|
"weeks": 39,
|
||||||
|
"avg_cross_section": 498.2,
|
||||||
|
"mean_ic": 0.0051,
|
||||||
|
"ic_t_stat": 0.21,
|
||||||
|
"ic_positive_pct": 56.4,
|
||||||
|
"mean_quintile_spread": 0.0087,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_3_1",
|
||||||
|
"weeks": 42,
|
||||||
|
"avg_cross_section": 498.5,
|
||||||
|
"mean_ic": -0.0064,
|
||||||
|
"ic_t_stat": -0.25,
|
||||||
|
"ic_positive_pct": 50.0,
|
||||||
|
"mean_quintile_spread": 0.0046,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "high_52w",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": -0.0086,
|
||||||
|
"ic_t_stat": -0.26,
|
||||||
|
"ic_positive_pct": 54.3,
|
||||||
|
"mean_quintile_spread": -0.0088,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "fip_id",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": -0.045,
|
||||||
|
"ic_t_stat": -2.91,
|
||||||
|
"ic_positive_pct": 25.7,
|
||||||
|
"mean_quintile_spread": -0.0168,
|
||||||
|
"reliable": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"sue_grade": {
|
||||||
|
"green": false,
|
||||||
|
"checks": {
|
||||||
|
"mean_ic": 0.0172,
|
||||||
|
"sign_positive": true,
|
||||||
|
"abs_ge_0_03": false,
|
||||||
|
"reliable": true,
|
||||||
|
"ic_t_stat": 0.6,
|
||||||
|
"weeks": 44
|
||||||
|
},
|
||||||
|
"reason": "iron rule not met",
|
||||||
|
"row": {
|
||||||
|
"signal": "sue_latest",
|
||||||
|
"weeks": 44,
|
||||||
|
"avg_cross_section": 47.4,
|
||||||
|
"mean_ic": 0.0172,
|
||||||
|
"ic_t_stat": 0.6,
|
||||||
|
"ic_positive_pct": 47.7,
|
||||||
|
"mean_quintile_spread": 0.0064,
|
||||||
|
"reliable": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"momentum_conditional_sue": {
|
||||||
|
"mean_ic": -0.0065,
|
||||||
|
"ic_t_stat": -0.1,
|
||||||
|
"weeks": 35,
|
||||||
|
"note": "IC of sue_latest within top mom_12_1 quintile (non-overlapping weeks)"
|
||||||
|
},
|
||||||
|
"sue_coverage": {
|
||||||
|
"symbols_with_sue": 48,
|
||||||
|
"avg_weeks_with_sue": 47.1,
|
||||||
|
"weeks_with_min_cross_section": 256
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"verdict": "PARK",
|
||||||
|
"verdict_detail": "SUE IC=0.0172 below iron bar or unreliable; keep data, no wire.",
|
||||||
|
"human_next": "- No SUE book change.\n- Read 2a tails before considering any earnings-avoid filter.",
|
||||||
|
"report_path": "reports/earnings-gap-sue-20260719-093129.json",
|
||||||
|
"fmp_note": "Bulk earnings-calendar is paid (402 on free tier). Backfill used per-symbol /stable/earnings; see earnings-backfill-status.json."
|
||||||
|
}
|
||||||
@@ -0,0 +1,202 @@
|
|||||||
|
# Earnings gap diagnostic + SUE / PEAD (Tier-1 alpha research)
|
||||||
|
|
||||||
|
**Status:** **PARK** (incomplete earnings coverage; SUE fails iron rule on available sample).
|
||||||
|
**Branch:** `research/earnings-gap-and-sue`
|
||||||
|
**Production impact:** none. Local research only. **No filters shipped from 2a.**
|
||||||
|
**Artifacts:** `reports/earnings-gap-sue-20260719-093129.json` (+ companion `.md`)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Pre-registration (locked before first research run)
|
||||||
|
|
||||||
|
### Data
|
||||||
|
|
||||||
|
- Historical earnings calendar for the production universe over the full snapshot
|
||||||
|
window (and deeper if the feed provides it).
|
||||||
|
- Preferred source: FMP **date-range earnings-calendar** (bulk). If unavailable on
|
||||||
|
free tier, fall back to per-symbol `/stable/earnings` with request accounting.
|
||||||
|
- Store in a real local table `earnings_events` (symbol + announce_date key).
|
||||||
|
- Point-in-time: a surprise is usable only from **announce date + 1 trading day**
|
||||||
|
onward.
|
||||||
|
|
||||||
|
### Experiment 2a — earnings-gap risk (defense, report-only)
|
||||||
|
|
||||||
|
Join simulated production-config trades (`fill_mode=close`) with earnings dates.
|
||||||
|
|
||||||
|
**Pre-registered questions:**
|
||||||
|
|
||||||
|
1. What fraction of losses worse than **−1R** occur with an earnings announcement
|
||||||
|
**between entry and exit** (inclusive of the holding window)?
|
||||||
|
2. What is the mean R of entries taken within **3 trading days BEFORE** an
|
||||||
|
announcement vs all other entries — report **both tails** of the R
|
||||||
|
distribution (rule 4: any earnings-avoid entry filter is presumed guilty of
|
||||||
|
right-tail trimming until the win distribution shows otherwise)?
|
||||||
|
|
||||||
|
**Output:** distributions and counts only.
|
||||||
|
**No filter is shipped.** If numbers argue for a filter → report and stop.
|
||||||
|
|
||||||
|
### Experiment 2b — SUE / PEAD (offense)
|
||||||
|
|
||||||
|
Signal `sue_latest`:
|
||||||
|
|
||||||
|
\[
|
||||||
|
\text{SUE} = \frac{\text{actual} - \text{estimate}}{\sigma(\text{trailing 8 surprises})}
|
||||||
|
\]
|
||||||
|
|
||||||
|
Fallback if estimate history is thin: scale surprise by price.
|
||||||
|
Carry forward from announce+1 for **63 trading days**, else NaN (name drops out
|
||||||
|
of that cross-section).
|
||||||
|
|
||||||
|
**Iron rule (IC harness):** mean weekly Spearman IC on non-overlapping weeks;
|
||||||
|
\|mean IC\| ≥ ~0.03, **positive** sign (drift), `reliable: true` (≥12 windows).
|
||||||
|
|
||||||
|
Always side-by-side with `mom_12_1` and `mom_12_1_resid` on **identical**
|
||||||
|
cross-sections.
|
||||||
|
|
||||||
|
Also report **momentum-conditional** IC (within top momentum quintile).
|
||||||
|
|
||||||
|
**If it passes iron rule:** STOP and report. Book-integration design is a
|
||||||
|
separate human-approved step — do not wire.
|
||||||
|
|
||||||
|
### Verdict labels
|
||||||
|
|
||||||
|
| label | meaning |
|
||||||
|
|---|---|
|
||||||
|
| **PROMOTE** | (2b only) iron rule cleared → human designs tilt/gate |
|
||||||
|
| **PARK** | Interesting but incomplete / weak |
|
||||||
|
| **DEAD** | No edge / diagnostic argues against action |
|
||||||
|
| **REPORT-ONLY** | (2a) always — never auto-filter |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Data provenance
|
||||||
|
|
||||||
|
| item | result |
|
||||||
|
|---|---|
|
||||||
|
| Snapshot | `backtest_snapshots/prod.sqlite` (506 names) |
|
||||||
|
| FMP bulk `earnings-calendar` | **402 Premium** — not available on free tier |
|
||||||
|
| FMP per-symbol `/stable/earnings` | used; hit daily rate limit ~225 reqs |
|
||||||
|
| Alpha Vantage `EARNINGS` | used for +24 symbols (announce = `reportedDate`) |
|
||||||
|
| Symbols with events | **48 / 506 (9.5%)** |
|
||||||
|
| Total events | 5,612 (5,018 with actual+estimate) |
|
||||||
|
| Announce range | 1985-08-31 → 2026-07-16 |
|
||||||
|
| FMP requests (first day) | 260 FMP + 25 AV (see `reports/earnings-backfill-status.json`) |
|
||||||
|
|
||||||
|
**Incomplete backfill is first-class.** 2a under-detects earnings overlaps; 2b SUE
|
||||||
|
cross-section averages **~47 names**, not ~500. Resume:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Day N (FMP free ~250/day; AV free ~25/day — prefer FMP after reset)
|
||||||
|
python scripts/backfill_earnings_events.py \
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite \
|
||||||
|
--provider fmp --force-symbol --limit 250 --sleep 0.4
|
||||||
|
|
||||||
|
# When done==506:
|
||||||
|
python scripts/run_earnings_research.py \
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite \
|
||||||
|
--workers 6 --allow-spawn
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
Generated: `2026-07-19T09:31:29`
|
||||||
|
|
||||||
|
### 2a — Earnings-gap risk (report-only)
|
||||||
|
|
||||||
|
Production book sim: Sharpe 2.09 (SE 0.497), CAGR 51.6%, max DD 21.4%, **322 trades**,
|
||||||
|
`fill_mode=close`.
|
||||||
|
|
||||||
|
#### Q1 — Losses worse than −1R with earnings in hold
|
||||||
|
|
||||||
|
| metric | value |
|
||||||
|
|---|---:|
|
||||||
|
| n losses < −1R | 28 |
|
||||||
|
| of which earnings in hold | **1** |
|
||||||
|
| fraction | **3.6%** |
|
||||||
|
| all trades with earnings in hold | 14 / 322 (4.4%) |
|
||||||
|
|
||||||
|
**Read:** On incomplete earnings labels this is a **lower bound** on earnings
|
||||||
|
overlap, not a clean “earnings rarely hurt.” Do **not** conclude earnings risk is
|
||||||
|
immaterial until coverage ≥ ~95% of the book’s names.
|
||||||
|
|
||||||
|
#### Q2 — Entry within 3 trading days before announce (both tails)
|
||||||
|
|
||||||
|
| cohort | n | mean R | win rate | p05 | p50 | p95 | max |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
| pre-earn (≤3d before) | **4** | 1.94 | 50% | −1.24 | 1.12 | 6.26 | 6.84 |
|
||||||
|
| other | 318 | 0.70 | 37% | −1.11 | −0.83 | 6.08 | **12.87** |
|
||||||
|
| all | 322 | 0.71 | 37% | −1.12 | −0.83 | 6.22 | 12.87 |
|
||||||
|
|
||||||
|
**Tail-trim presumption:** n=4 is not a sample. Point estimate does **not** show
|
||||||
|
right-tail destruction of pre-earn entries (p95 similar; max actually higher in
|
||||||
|
“other”). **No earnings-avoid filter is supported.** Re-run after full backfill.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### 2b — SUE / PEAD IC
|
||||||
|
|
||||||
|
#### Full-universe harness (mom on ~500; SUE only where labeled)
|
||||||
|
|
||||||
|
| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable |
|
||||||
|
|---|---:|---:|---:|---:|---|
|
||||||
|
| mom_12_1_sector_resid | 0.0578 | 2.34 | 35 | 497.7 | true |
|
||||||
|
| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true |
|
||||||
|
| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true |
|
||||||
|
| **sue_latest** | **0.0172** | **0.6** | 44 | **47.4** | true |
|
||||||
|
| fip_id | −0.045 | −2.91 | 35 | 497.7 | true |
|
||||||
|
|
||||||
|
#### Identical SUE subset (fair side-by-side — use this while coverage is thin)
|
||||||
|
|
||||||
|
| signal | mean_ic | ic_t_stat | weeks | avg_N |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| sue_latest | 0.0172 | 0.6 | 44 | 47.4 |
|
||||||
|
| mom_12_1 | −0.0174 | −0.42 | 35 | 47.3 |
|
||||||
|
| mom_12_1_resid | −0.0104 | −0.27 | 35 | 47.3 |
|
||||||
|
|
||||||
|
On the thin labeled subset, momentum itself is noise — so the subset is not yet
|
||||||
|
a meaningful PEAD test.
|
||||||
|
|
||||||
|
#### Momentum-conditional SUE (top mom quintile)
|
||||||
|
|
||||||
|
| metric | value |
|
||||||
|
|---|---:|
|
||||||
|
| mean IC | **−0.0065** |
|
||||||
|
| t | −0.1 |
|
||||||
|
| weeks | 35 |
|
||||||
|
|
||||||
|
Wrong sign vs “ride positive surprises inside the momentum gate.”
|
||||||
|
|
||||||
|
**Iron rule:** fail (\|IC\| 0.017 < 0.03; t 0.6). **No promote.**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Verdict
|
||||||
|
|
||||||
|
| piece | verdict |
|
||||||
|
|---|---|
|
||||||
|
| **2a earnings-gap** | **REPORT-ONLY** — no filter. Coverage too thin for risk claims; tails do not argue for an avoid-filter on n=4. |
|
||||||
|
| **2b SUE** | **PARK** (effectively not green). Mild positive IC on ~48 names; fails iron bar; mom-conditional flat/negative. Re-score after full backfill before DEAD. |
|
||||||
|
| **Production** | **no change** |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## What a human must decide next
|
||||||
|
|
||||||
|
1. Resume multi-day earnings backfill to **506/506**, then re-run
|
||||||
|
`run_earnings_research.py` (heavy — MacBook OK).
|
||||||
|
2. Do **not** ship an earnings-avoid entry filter from 2a.
|
||||||
|
3. Do **not** wire SUE until a full-coverage IC clears the iron rule (and
|
||||||
|
preferably mom-conditional > 0).
|
||||||
|
4. Do not merge into main strategy docs without review.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Implementation notes
|
||||||
|
|
||||||
|
| piece | role |
|
||||||
|
|---|---|
|
||||||
|
| `scripts/backfill_earnings_events.py` | bulk attempt → FMP/AV per-symbol; `earnings_events` + meta on snapshot |
|
||||||
|
| `scripts/run_earnings_research.py` | 2a trade join + 2b SUE IC / mom-conditional |
|
||||||
|
| Snapshot table `earnings_events` | real table (not SystemSetting JSON) |
|
||||||
@@ -0,0 +1,412 @@
|
|||||||
|
{
|
||||||
|
"generated_at": "2026-07-19T08:33:56.651229",
|
||||||
|
"snapshot_guard": {
|
||||||
|
"snapshot": "C:\\Workspace\\signal-platform\\backtest_snapshots\\prod.sqlite",
|
||||||
|
"manifest": null,
|
||||||
|
"manifest_ok": null,
|
||||||
|
"note": "No completion manifest (prod.sqlite is expected without one). Bar-count sanity still applied.",
|
||||||
|
"ticker_count": 506,
|
||||||
|
"ohlcv_row_count": 629263,
|
||||||
|
"bars_min_avg_max": {
|
||||||
|
"min": 14,
|
||||||
|
"avg": 1246.1,
|
||||||
|
"max": 1261
|
||||||
|
},
|
||||||
|
"ohlcv_date_range": {
|
||||||
|
"min": "2021-06-24",
|
||||||
|
"max": "2026-07-02"
|
||||||
|
},
|
||||||
|
"benchmark_prices": [
|
||||||
|
{
|
||||||
|
"symbol": "SPY",
|
||||||
|
"n": 1516,
|
||||||
|
"min": "2020-07-06",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLB",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLC",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLE",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLF",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLI",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLK",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLP",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLRE",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLU",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLV",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"symbol": "XLY",
|
||||||
|
"n": 1512,
|
||||||
|
"min": "2020-07-10",
|
||||||
|
"max": "2026-07-17"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"missing_sector_etfs": []
|
||||||
|
},
|
||||||
|
"sector_coverage": {
|
||||||
|
"universe": 506,
|
||||||
|
"mapped": 505,
|
||||||
|
"mapped_pct": 99.8,
|
||||||
|
"with_etf": 505,
|
||||||
|
"missing": [
|
||||||
|
"RHM"
|
||||||
|
],
|
||||||
|
"by_sector": {
|
||||||
|
"Industrials": 81,
|
||||||
|
"Financials": 75,
|
||||||
|
"Information Technology": 72,
|
||||||
|
"Health Care": 58,
|
||||||
|
"Consumer Discretionary": 47,
|
||||||
|
"Consumer Staples": 34,
|
||||||
|
"Real Estate": 31,
|
||||||
|
"Utilities": 31,
|
||||||
|
"Materials": 26,
|
||||||
|
"Communication Services": 23,
|
||||||
|
"Energy": 22,
|
||||||
|
"Consumer Defensive": 2,
|
||||||
|
"Technology": 2,
|
||||||
|
"Financial Services": 1
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"sector_map_path": "C:\\Workspace\\signal-platform\\data\\research\\ticker_sector_map.json",
|
||||||
|
"signal_eval": [
|
||||||
|
{
|
||||||
|
"signal": "vol_6m",
|
||||||
|
"weeks": 39,
|
||||||
|
"avg_cross_section": 498.2,
|
||||||
|
"mean_ic": 0.0609,
|
||||||
|
"ic_t_stat": 1.48,
|
||||||
|
"ic_positive_pct": 64.1,
|
||||||
|
"mean_quintile_spread": 0.0337,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1_sector_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0578,
|
||||||
|
"ic_t_stat": 2.34,
|
||||||
|
"ic_positive_pct": 65.7,
|
||||||
|
"mean_quintile_spread": 0.0245,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0552,
|
||||||
|
"ic_t_stat": 1.98,
|
||||||
|
"ic_positive_pct": 60.0,
|
||||||
|
"mean_quintile_spread": 0.0207,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0531,
|
||||||
|
"ic_t_stat": 1.61,
|
||||||
|
"ic_positive_pct": 65.7,
|
||||||
|
"mean_quintile_spread": 0.0206,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_12_1_sector_demeaned",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 496.7,
|
||||||
|
"mean_ic": 0.034,
|
||||||
|
"ic_t_stat": 1.32,
|
||||||
|
"ic_positive_pct": 62.9,
|
||||||
|
"mean_quintile_spread": 0.0154,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "trend_200",
|
||||||
|
"weeks": 37,
|
||||||
|
"avg_cross_section": 497.9,
|
||||||
|
"mean_ic": 0.0161,
|
||||||
|
"ic_t_stat": 0.44,
|
||||||
|
"ic_positive_pct": 59.5,
|
||||||
|
"mean_quintile_spread": 0.006,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "reversal_1m",
|
||||||
|
"weeks": 43,
|
||||||
|
"avg_cross_section": 498.7,
|
||||||
|
"mean_ic": 0.0059,
|
||||||
|
"ic_t_stat": 0.22,
|
||||||
|
"ic_positive_pct": 53.5,
|
||||||
|
"mean_quintile_spread": 0.0053,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_6_1",
|
||||||
|
"weeks": 39,
|
||||||
|
"avg_cross_section": 498.2,
|
||||||
|
"mean_ic": 0.0051,
|
||||||
|
"ic_t_stat": 0.21,
|
||||||
|
"ic_positive_pct": 56.4,
|
||||||
|
"mean_quintile_spread": 0.0087,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "mom_3_1",
|
||||||
|
"weeks": 42,
|
||||||
|
"avg_cross_section": 498.5,
|
||||||
|
"mean_ic": -0.0064,
|
||||||
|
"ic_t_stat": -0.25,
|
||||||
|
"ic_positive_pct": 50.0,
|
||||||
|
"mean_quintile_spread": 0.0046,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "high_52w",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": -0.0086,
|
||||||
|
"ic_t_stat": -0.26,
|
||||||
|
"ic_positive_pct": 54.3,
|
||||||
|
"mean_quintile_spread": -0.0088,
|
||||||
|
"reliable": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"signal": "fip_id",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": -0.045,
|
||||||
|
"ic_t_stat": -2.91,
|
||||||
|
"ic_positive_pct": 25.7,
|
||||||
|
"mean_quintile_spread": -0.0168,
|
||||||
|
"reliable": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"ic_grades": {
|
||||||
|
"mom_12_1_sector_resid": {
|
||||||
|
"promote_to_ab": true,
|
||||||
|
"checks": {
|
||||||
|
"sign_ok": true,
|
||||||
|
"abs_mean_ic_ge_0_03": true,
|
||||||
|
"reliable": true,
|
||||||
|
"t_ge_resid": true,
|
||||||
|
"mean_ic": 0.0578,
|
||||||
|
"ic_t_stat": 2.34,
|
||||||
|
"resid_ic_t_stat": 1.98,
|
||||||
|
"weeks": 35
|
||||||
|
},
|
||||||
|
"reason": "clears iron rule and t \u2265 mom_12_1_resid \u2014 authorized for A/B only",
|
||||||
|
"row": {
|
||||||
|
"signal": "mom_12_1_sector_resid",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 497.7,
|
||||||
|
"mean_ic": 0.0578,
|
||||||
|
"ic_t_stat": 2.34,
|
||||||
|
"ic_positive_pct": 65.7,
|
||||||
|
"mean_quintile_spread": 0.0245,
|
||||||
|
"reliable": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"mom_12_1_sector_demeaned": {
|
||||||
|
"promote_to_ab": false,
|
||||||
|
"checks": {
|
||||||
|
"sign_ok": true,
|
||||||
|
"abs_mean_ic_ge_0_03": true,
|
||||||
|
"reliable": true,
|
||||||
|
"t_ge_resid": false,
|
||||||
|
"mean_ic": 0.034,
|
||||||
|
"ic_t_stat": 1.32,
|
||||||
|
"resid_ic_t_stat": 1.98,
|
||||||
|
"weeks": 35
|
||||||
|
},
|
||||||
|
"reason": "does not clear pre-registered IC promotion bar",
|
||||||
|
"row": {
|
||||||
|
"signal": "mom_12_1_sector_demeaned",
|
||||||
|
"weeks": 35,
|
||||||
|
"avg_cross_section": 496.7,
|
||||||
|
"mean_ic": 0.034,
|
||||||
|
"ic_t_stat": 1.32,
|
||||||
|
"ic_positive_pct": 62.9,
|
||||||
|
"mean_quintile_spread": 0.0154,
|
||||||
|
"reliable": true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"portfolio_ab": {
|
||||||
|
"signal": "mom_12_1_sector_resid",
|
||||||
|
"ranking_key": "residual_high_vol_blend_80_20_score",
|
||||||
|
"fill_mode": "close",
|
||||||
|
"validation_split": "2024-07-01",
|
||||||
|
"control": {
|
||||||
|
"label": "control_mom_12_1_resid",
|
||||||
|
"n_qualified_longs": 1086,
|
||||||
|
"windows": {
|
||||||
|
"train": {
|
||||||
|
"sharpe": 1.3,
|
||||||
|
"sharpe_se": 0.685,
|
||||||
|
"cagr_pct": 29.2,
|
||||||
|
"max_drawdown_pct": 21.4,
|
||||||
|
"total_return_pct": 70.9,
|
||||||
|
"trades": 176,
|
||||||
|
"win_rate_pct": null,
|
||||||
|
"avg_r": null,
|
||||||
|
"n_returns": 525,
|
||||||
|
"return_skew": 0.3722,
|
||||||
|
"return_kurtosis": 4.6208,
|
||||||
|
"psr": 0.971
|
||||||
|
},
|
||||||
|
"validation": {
|
||||||
|
"sharpe": 2.92,
|
||||||
|
"sharpe_se": 0.709,
|
||||||
|
"cagr_pct": 76.3,
|
||||||
|
"max_drawdown_pct": 11.7,
|
||||||
|
"total_return_pct": 210.7,
|
||||||
|
"trades": 150,
|
||||||
|
"win_rate_pct": null,
|
||||||
|
"avg_r": null,
|
||||||
|
"n_returns": 501,
|
||||||
|
"return_skew": 0.1734,
|
||||||
|
"return_kurtosis": 4.4625,
|
||||||
|
"psr": 1.0
|
||||||
|
},
|
||||||
|
"full": {
|
||||||
|
"sharpe": 2.09,
|
||||||
|
"sharpe_se": 0.497,
|
||||||
|
"cagr_pct": 51.6,
|
||||||
|
"max_drawdown_pct": 21.4,
|
||||||
|
"total_return_pct": 424.6,
|
||||||
|
"trades": 322,
|
||||||
|
"win_rate_pct": null,
|
||||||
|
"avg_r": null,
|
||||||
|
"n_returns": 1000,
|
||||||
|
"return_skew": 0.2686,
|
||||||
|
"return_kurtosis": 4.5653,
|
||||||
|
"psr": 1.0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"treatment": {
|
||||||
|
"label": "treatment_mom_12_1_sector_resid",
|
||||||
|
"n_qualified_longs": 1210,
|
||||||
|
"windows": {
|
||||||
|
"train": {
|
||||||
|
"sharpe": 1.57,
|
||||||
|
"sharpe_se": 0.677,
|
||||||
|
"cagr_pct": 35.5,
|
||||||
|
"max_drawdown_pct": 19.8,
|
||||||
|
"total_return_pct": 90.0,
|
||||||
|
"trades": 176,
|
||||||
|
"win_rate_pct": null,
|
||||||
|
"avg_r": null,
|
||||||
|
"n_returns": 530,
|
||||||
|
"return_skew": 0.466,
|
||||||
|
"return_kurtosis": 4.4413,
|
||||||
|
"psr": 0.99
|
||||||
|
},
|
||||||
|
"validation": {
|
||||||
|
"sharpe": 2.57,
|
||||||
|
"sharpe_se": 0.701,
|
||||||
|
"cagr_pct": 66.3,
|
||||||
|
"max_drawdown_pct": 14.8,
|
||||||
|
"total_return_pct": 176.4,
|
||||||
|
"trades": 163,
|
||||||
|
"win_rate_pct": null,
|
||||||
|
"avg_r": null,
|
||||||
|
"n_returns": 501,
|
||||||
|
"return_skew": 0.3003,
|
||||||
|
"return_kurtosis": 4.4305,
|
||||||
|
"psr": 0.9999
|
||||||
|
},
|
||||||
|
"full": {
|
||||||
|
"sharpe": 2.09,
|
||||||
|
"sharpe_se": 0.491,
|
||||||
|
"cagr_pct": 51.0,
|
||||||
|
"max_drawdown_pct": 19.8,
|
||||||
|
"total_return_pct": 421.3,
|
||||||
|
"trades": 337,
|
||||||
|
"win_rate_pct": null,
|
||||||
|
"avg_r": null,
|
||||||
|
"n_returns": 1005,
|
||||||
|
"return_skew": 0.4066,
|
||||||
|
"return_kurtosis": 4.3627,
|
||||||
|
"psr": 1.0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"promotion": {
|
||||||
|
"promote": true,
|
||||||
|
"checks": {
|
||||||
|
"validation_sharpe_ge_control_minus_half_se": true,
|
||||||
|
"full_sharpe_not_worse": true,
|
||||||
|
"full_maxdd_not_worse": true,
|
||||||
|
"control_validation_sharpe": 2.92,
|
||||||
|
"treatment_validation_sharpe": 2.57,
|
||||||
|
"se_used": 0.701,
|
||||||
|
"control_full_sharpe": 2.09,
|
||||||
|
"treatment_full_sharpe": 2.09,
|
||||||
|
"control_full_maxdd": 21.4,
|
||||||
|
"treatment_full_maxdd": 19.8
|
||||||
|
},
|
||||||
|
"reason": "clears pre-registered A/B bar \u2014 human decides wire-in"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"verdict": "PROMOTE",
|
||||||
|
"verdict_detail": "mom_12_1_sector_resid cleared IC + A/B bars. Human must design wire-in; do not ship from this branch.",
|
||||||
|
"human_next": "- Approve or reject production residual swap vs dual-signal design.\n- If sector-cap arm ran, review tail-trim diagnostics before any cap.",
|
||||||
|
"report_path": "reports/sector-residual-20260719-083356.json",
|
||||||
|
"pre_registration": {
|
||||||
|
"iron_ic_bar": 0.03,
|
||||||
|
"validation_split": "2024-07-01",
|
||||||
|
"fill_mode": "close",
|
||||||
|
"cost_per_side": 0.001,
|
||||||
|
"ab_rule": "val Sharpe >= control - 0.5*SE; full Sharpe & maxDD not worse"
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,237 @@
|
|||||||
|
# Sector-residual momentum (Tier-1 alpha research)
|
||||||
|
|
||||||
|
**Status:** **PROMOTE (to human design decision only)** — IC + A/B bars cleared; **do not ship**.
|
||||||
|
**Branch:** `research/sector-residual-momentum`
|
||||||
|
**Production impact:** none. Local research only. No scheduler / gate / prod-config changes.
|
||||||
|
**Artifacts:** `reports/sector-residual-20260719-083356.json` (+ companion `.md`)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Pre-registration (locked before first research run)
|
||||||
|
|
||||||
|
### Hypothesis
|
||||||
|
|
||||||
|
Residualizing 12–1 momentum against the sector, not only the market, reduces
|
||||||
|
factor volatility at similar return (Blitz / Huij / Martens-style) → higher
|
||||||
|
Sharpe on the production book when the residual replaces market-only residual
|
||||||
|
as the momentum leg.
|
||||||
|
|
||||||
|
### Signals (candidates)
|
||||||
|
|
||||||
|
| signal | construction |
|
||||||
|
|---|---|
|
||||||
|
| `mom_12_1_sector_resid` | Two-factor residual vs SPY + ticker’s sector ETF. Same window as `mom_12_1_resid`: ≥100 daily obs, 252-bar lookback, 21-bar skip; two-factor OLS betas **without intercept**; cumulate residual returns over the formation window. |
|
||||||
|
| `mom_12_1_sector_demeaned` | Plain `mom_12_1` minus the **cross-sectional** mean of `mom_12_1` within the same GICS sector that week (≥2 names in sector). No regression. |
|
||||||
|
|
||||||
|
### Baselines (same run, same cross-sections — iron rule)
|
||||||
|
|
||||||
|
Always report side-by-side with:
|
||||||
|
|
||||||
|
- `mom_12_1`
|
||||||
|
- `mom_12_1_resid`
|
||||||
|
|
||||||
|
Computed on the **identical** weekly non-overlapping cross-sections in this run.
|
||||||
|
Never compare against IC numbers from another report.
|
||||||
|
|
||||||
|
### Iron rule (IC harness)
|
||||||
|
|
||||||
|
Source of truth: `_signal_evaluation` in `app/services/backtest_service.py`.
|
||||||
|
|
||||||
|
- Mean weekly Spearman IC on **non-overlapping** weekly windows
|
||||||
|
- Bar: \|mean IC\| ≥ ~0.03, **consistent positive sign**, `reliable: true` (≥ 12 windows)
|
||||||
|
|
||||||
|
### Promotion to portfolio A/B (candidate → book)
|
||||||
|
|
||||||
|
A candidate promotes to A/B **only if**:
|
||||||
|
|
||||||
|
1. It clears the iron-rule bar **and**
|
||||||
|
2. Its IC **t-stat ≥** that of `mom_12_1_resid` on the same cross-sections.
|
||||||
|
|
||||||
|
### Portfolio A/B grading (if and only if IC promotion fires)
|
||||||
|
|
||||||
|
- Swap candidate in as the **momentum leg** of the production 80/20 momentum/vol
|
||||||
|
rank **and** as the gate-percentile signal.
|
||||||
|
- `fill_mode=close`, `COST_PER_SIDE = 0.001`, full config otherwise unchanged.
|
||||||
|
- Validation window = entries ≥ **2024-07-01** (call it **validation**, not
|
||||||
|
holdout — contaminated by prior experiments).
|
||||||
|
- Pre-registered promotion bar:
|
||||||
|
- validation Sharpe ≥ control − 0.5·SE
|
||||||
|
- full-period Sharpe and max-DD **not worse** than control
|
||||||
|
- Report Lo / Mertens-adjusted SEs.
|
||||||
|
|
||||||
|
### Optional sector-cap sub-experiment
|
||||||
|
|
||||||
|
Only if labels are in **and** A/B ran: max **3** positions per sector in the
|
||||||
|
10-slot book. Same A/B grading. **Tail-trim presumption of guilt** (rule 4):
|
||||||
|
report entry counts and both tails of the R distribution. Rising win rate with
|
||||||
|
falling Sharpe/CAGR = red flag → do not promote.
|
||||||
|
|
||||||
|
**This run:** sector-cap arm **not executed** (optional; A/B unconstrained book
|
||||||
|
only). Can be a human-approved follow-up.
|
||||||
|
|
||||||
|
### Verdict labels
|
||||||
|
|
||||||
|
| label | meaning |
|
||||||
|
|---|---|
|
||||||
|
| **PROMOTE** | Clears pre-registered bar; human decides next (wire design separate) |
|
||||||
|
| **PARK** | Inconclusive / weak; keep machinery, no book change |
|
||||||
|
| **DEAD** | Failed iron rule or worse than residual baseline with clear sign |
|
||||||
|
|
||||||
|
### Explicit non-goals
|
||||||
|
|
||||||
|
- No production deploy from this doc
|
||||||
|
- Do not resurrect: take-profit exits, EV gate, regime entry-blocking,
|
||||||
|
inverse-vol sizing, gap-caps, unconditional FIP filter
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Data provenance
|
||||||
|
|
||||||
|
### Snapshot race guard
|
||||||
|
|
||||||
|
| check | result |
|
||||||
|
|---|---|
|
||||||
|
| Snapshot path | `backtest_snapshots/prod.sqlite` |
|
||||||
|
| Manifest | none (expected for prod snapshot); bar-count sanity applied |
|
||||||
|
| Tickers / OHLCV | **506** / **629,263** |
|
||||||
|
| Bars min / avg / max | 14 / 1246.1 / 1261 |
|
||||||
|
| OHLCV range | 2021-06-24 → 2026-07-02 |
|
||||||
|
| Partial-build red flags | none (avg bars healthy) |
|
||||||
|
|
||||||
|
Integrity fingerprint on same run: `fip_id` mean IC **−0.045** / t **−2.91**
|
||||||
|
(35 weeks, N≈498) — matches the established prod fingerprint.
|
||||||
|
|
||||||
|
### Sector labels
|
||||||
|
|
||||||
|
| source | count |
|
||||||
|
|---|---:|
|
||||||
|
| Public S&P 500 GICS CSV | 496 newly filled |
|
||||||
|
| FMP profile requests | 10 (all missing after CSV) |
|
||||||
|
| Mapped / universe | **505 / 506 (99.8%)** |
|
||||||
|
| With mappable ETF | 505 |
|
||||||
|
| Still missing | **RHM** only |
|
||||||
|
|
||||||
|
Persist path: `data/research/ticker_sector_map.json`.
|
||||||
|
|
||||||
|
FMP aliases (`Technology`, `Consumer Defensive`, `Financial Services`) map to
|
||||||
|
SPDRs via the alias table in `app/services/sector_map.py`.
|
||||||
|
|
||||||
|
### Sector ETFs in `benchmark_prices` (auxiliary only — not tradable)
|
||||||
|
|
||||||
|
| symbol | bars | min date | max date |
|
||||||
|
|---|---:|---|---|
|
||||||
|
| SPY | 1516 | 2020-07-06 | 2026-07-17 |
|
||||||
|
| XLB…XLY (11) | 1512 each | 2020-07-10 | 2026-07-17 |
|
||||||
|
|
||||||
|
Fetched via Alpaca `Adjustment.SPLIT` into **`benchmark_prices`** (same table as
|
||||||
|
SPY) so they never enter the ticker universe or candidate replay.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
Generated: `2026-07-19T08:33:56`
|
||||||
|
|
||||||
|
### IC harness (identical cross-sections, production 506-name universe)
|
||||||
|
|
||||||
|
| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable | ic+_pct | quintile spread |
|
||||||
|
|---|---:|---:|---:|---:|---|---:|---:|
|
||||||
|
| **mom_12_1_sector_resid** | **0.0578** | **2.34** | 35 | 497.7 | true | 65.7 | 0.0245 |
|
||||||
|
| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true | 60.0 | 0.0207 |
|
||||||
|
| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true | 65.7 | 0.0206 |
|
||||||
|
| mom_12_1_sector_demeaned | 0.0340 | 1.32 | 35 | 496.7 | true | 62.9 | 0.0154 |
|
||||||
|
|
||||||
|
### IC promotion grades
|
||||||
|
|
||||||
|
| candidate | iron rule | t ≥ resid | promote_to_ab |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `mom_12_1_sector_resid` | pass (IC 0.058, +sign, reliable) | **yes** (2.34 ≥ 1.98) | **yes** |
|
||||||
|
| `mom_12_1_sector_demeaned` | pass (IC 0.034, +sign, reliable) | **no** (1.32 < 1.98) | **no** |
|
||||||
|
|
||||||
|
### Portfolio A/B — `mom_12_1_sector_resid` as residual leg
|
||||||
|
|
||||||
|
Config: production 80/20 residual/high-vol rank + gate percentile, `fill_mode=close`,
|
||||||
|
cost 10 bps/side, ATR trail / gate-reset re-entry as live. Validation split
|
||||||
|
2024-07-01.
|
||||||
|
|
||||||
|
| window | arm | Sharpe | Sharpe SE (Mertens) | CAGR % | max DD % | trades | n_days |
|
||||||
|
|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
| train | control (resid) | 1.30 | 0.685 | 29.2 | 21.4 | 176 | 525 |
|
||||||
|
| train | treatment (sector resid) | **1.57** | 0.677 | **35.5** | **19.8** | 176 | 530 |
|
||||||
|
| validation | control | **2.92** | 0.709 | **76.3** | **11.7** | 150 | 501 |
|
||||||
|
| validation | treatment | 2.57 | 0.701 | 66.3 | 14.8 | 163 | 501 |
|
||||||
|
| full | control | 2.09 | 0.497 | 51.6 | 21.4 | 322 | 1000 |
|
||||||
|
| full | treatment | 2.09 | 0.491 | 51.0 | **19.8** | 337 | 1005 |
|
||||||
|
|
||||||
|
**Pre-registered A/B checks**
|
||||||
|
|
||||||
|
| check | result |
|
||||||
|
|---|---|
|
||||||
|
| val Sharpe ≥ control − 0.5·SE | **pass** (2.57 ≥ 2.92 − 0.5×0.701 = 2.5695) — **knife-edge** |
|
||||||
|
| full Sharpe not worse | **pass** (2.09 = 2.09) |
|
||||||
|
| full max DD not worse | **pass** (19.8 < 21.4) |
|
||||||
|
|
||||||
|
Qualified long candidates: control 1086 vs treatment 1210 (sector residual
|
||||||
|
gates a slightly larger set).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Verdict
|
||||||
|
|
||||||
|
| signal | verdict | note |
|
||||||
|
|---|---|---|
|
||||||
|
| **`mom_12_1_sector_resid`** | **PROMOTE → human wire-in decision** | IC modestly beats market residual; A/B clears pre-reg bar narrowly. **Do not ship from this branch.** |
|
||||||
|
| **`mom_12_1_sector_demeaned`** | **DEAD** (for promotion) | Iron-rule IC magnitude ok, but t-stat loses to `mom_12_1_resid`. Cheap variant not competitive. |
|
||||||
|
|
||||||
|
### Read carefully (for the human)
|
||||||
|
|
||||||
|
1. **IC edge is real but small.** Sector residual IC 0.0578 / t 2.34 vs market
|
||||||
|
residual 0.0552 / t 1.98 on the **same** 35 windows — better consistency
|
||||||
|
(ic+ 65.7% vs 60%) and slightly higher mean, not a different factor class.
|
||||||
|
2. **A/B is not a clear Sharpe win.** Full-period Sharpe is flat (2.09).
|
||||||
|
Validation Sharpe is **lower** than control (2.57 vs 2.92) and only clears
|
||||||
|
the pre-registered “within 0.5 SE” cushion by ~0.001. Train improves;
|
||||||
|
validation worsens — classic regime-split noise on ~2 years.
|
||||||
|
3. **Risk side is friendly.** Full max DD improves (19.8% vs 21.4%); train DD
|
||||||
|
also better. Matches the “lower factor vol” half of the hypothesis more than
|
||||||
|
the “higher Sharpe” half on this window.
|
||||||
|
4. **Survivorship / short history.** Same caveats as all current research:
|
||||||
|
today’s constituents, ~35 independent weekly windows, one post-2021 regime
|
||||||
|
dominant. Task 3 (history depth) should re-check IC stability before any
|
||||||
|
wire-in.
|
||||||
|
5. **Not shipped.** Machinery lives on the research branch; production residual
|
||||||
|
path is untouched.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## What a human must decide next
|
||||||
|
|
||||||
|
1. **Accept or reject** replacing `mom_12_1_resid` with `mom_12_1_sector_resid`
|
||||||
|
as the production residual (gate + 80/20 mom leg), **or** keep market residual
|
||||||
|
and treat sector residual as research-only.
|
||||||
|
2. If leaning accept: require **Task 3 history-depth** confirmation (IC era split
|
||||||
|
pre/post-2021) before any production PR.
|
||||||
|
3. Optional: run **sector-cap ≤3** A/B with full tail diagnostics (not run here).
|
||||||
|
4. **Do not** merge this verdict into main strategy docs without review.
|
||||||
|
5. Wire-in design (live sector map refresh, ETF series ops, fallback when sector
|
||||||
|
missing) is a **separate** approved engineering step.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Implementation notes (research machinery)
|
||||||
|
|
||||||
|
| piece | role |
|
||||||
|
|---|---|
|
||||||
|
| `app/services/sector_map.py` | GICS→ETF map, symbol normalise, JSON load/save |
|
||||||
|
| `app/services/backtest_service.py` | multi-factor residual; `mom_12_1_sector_resid` in `_signal_values`; demean inject |
|
||||||
|
| `scripts/build_ticker_sector_map.py` | SP500 CSV + FMP gap fill |
|
||||||
|
| `scripts/fetch_sector_etfs_to_snapshot.py` | Alpaca → snapshot `benchmark_prices` |
|
||||||
|
| `scripts/run_sector_residual_research.py` | race guard, IC, optional A/B, reports |
|
||||||
|
| `data/research/ticker_sector_map.json` | persisted labels (research only) |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Artifacts
|
||||||
|
|
||||||
|
- JSON: `reports/sector-residual-20260719-083356.json`
|
||||||
|
- MD copy: `reports/sector-residual-20260719-083356.md`
|
||||||
@@ -0,0 +1,528 @@
|
|||||||
|
"""Backfill historical earnings into a snapshot ``earnings_events`` table.
|
||||||
|
|
||||||
|
Prefers FMP bulk date-range ``earnings-calendar`` (one request per window).
|
||||||
|
On free-tier 402/403, falls back to per-symbol ``/stable/earnings`` with
|
||||||
|
resume support and request counting (≈250 req/day free tier).
|
||||||
|
|
||||||
|
Research only — writes to the local snapshot SQLite, never production Postgres.
|
||||||
|
|
||||||
|
Example
|
||||||
|
-------
|
||||||
|
python scripts/backfill_earnings_events.py \\
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite --limit 250
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from datetime import date, datetime, timedelta, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
from sqlalchemy import create_engine, text
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
if str(ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
|
||||||
|
FMP_STABLE = "https://financialmodelingprep.com/stable"
|
||||||
|
DDL = """
|
||||||
|
CREATE TABLE IF NOT EXISTS earnings_events (
|
||||||
|
id INTEGER PRIMARY KEY,
|
||||||
|
symbol TEXT NOT NULL,
|
||||||
|
announce_date TEXT NOT NULL,
|
||||||
|
announce_time TEXT,
|
||||||
|
eps_estimate REAL,
|
||||||
|
eps_actual REAL,
|
||||||
|
revenue_estimate REAL,
|
||||||
|
revenue_actual REAL,
|
||||||
|
source TEXT NOT NULL,
|
||||||
|
fetched_at TEXT NOT NULL,
|
||||||
|
UNIQUE(symbol, announce_date)
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
# Side table tracks which symbols have been fully pulled (resume).
|
||||||
|
META_DDL = """
|
||||||
|
CREATE TABLE IF NOT EXISTS earnings_backfill_meta (
|
||||||
|
symbol TEXT PRIMARY KEY,
|
||||||
|
status TEXT NOT NULL,
|
||||||
|
n_events INTEGER NOT NULL DEFAULT 0,
|
||||||
|
updated_at TEXT NOT NULL,
|
||||||
|
note TEXT
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_args() -> argparse.Namespace:
|
||||||
|
p = argparse.ArgumentParser(description=__doc__)
|
||||||
|
p.add_argument("--snapshot", default="backtest_snapshots/prod.sqlite")
|
||||||
|
p.add_argument(
|
||||||
|
"--from-date",
|
||||||
|
default="2020-01-01",
|
||||||
|
help="Bulk calendar window start (also filters per-symbol rows).",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--to-date",
|
||||||
|
default=None,
|
||||||
|
help="Bulk calendar window end (default: today).",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--limit",
|
||||||
|
type=int,
|
||||||
|
default=250,
|
||||||
|
help="Max FMP requests this run (free-tier cushion).",
|
||||||
|
)
|
||||||
|
p.add_argument("--sleep", type=float, default=0.35)
|
||||||
|
p.add_argument(
|
||||||
|
"--force-symbol",
|
||||||
|
action="store_true",
|
||||||
|
help="Skip bulk attempt; go straight to per-symbol.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--refetch-done",
|
||||||
|
action="store_true",
|
||||||
|
help="Re-fetch symbols already marked done.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--provider",
|
||||||
|
choices=("fmp", "alpha_vantage", "auto"),
|
||||||
|
default="auto",
|
||||||
|
help="Earnings provider. auto tries FMP bulk then FMP/AV per-symbol.",
|
||||||
|
)
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def _ensure_tables(engine) -> None:
|
||||||
|
with engine.begin() as conn:
|
||||||
|
conn.execute(text(DDL))
|
||||||
|
conn.execute(text(META_DDL))
|
||||||
|
|
||||||
|
|
||||||
|
def _upsert_events(conn, rows: list[dict], source: str) -> int:
|
||||||
|
if not rows:
|
||||||
|
return 0
|
||||||
|
now = datetime.now(timezone.utc).isoformat()
|
||||||
|
written = 0
|
||||||
|
for r in rows:
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
INSERT INTO earnings_events (
|
||||||
|
symbol, announce_date, announce_time,
|
||||||
|
eps_estimate, eps_actual, revenue_estimate, revenue_actual,
|
||||||
|
source, fetched_at
|
||||||
|
) VALUES (
|
||||||
|
:symbol, :announce_date, :announce_time,
|
||||||
|
:eps_estimate, :eps_actual, :revenue_estimate, :revenue_actual,
|
||||||
|
:source, :fetched_at
|
||||||
|
)
|
||||||
|
ON CONFLICT(symbol, announce_date) DO UPDATE SET
|
||||||
|
announce_time=excluded.announce_time,
|
||||||
|
eps_estimate=excluded.eps_estimate,
|
||||||
|
eps_actual=excluded.eps_actual,
|
||||||
|
revenue_estimate=excluded.revenue_estimate,
|
||||||
|
revenue_actual=excluded.revenue_actual,
|
||||||
|
source=excluded.source,
|
||||||
|
fetched_at=excluded.fetched_at
|
||||||
|
"""
|
||||||
|
),
|
||||||
|
{
|
||||||
|
"symbol": r["symbol"],
|
||||||
|
"announce_date": r["announce_date"],
|
||||||
|
"announce_time": r.get("announce_time"),
|
||||||
|
"eps_estimate": r.get("eps_estimate"),
|
||||||
|
"eps_actual": r.get("eps_actual"),
|
||||||
|
"revenue_estimate": r.get("revenue_estimate"),
|
||||||
|
"revenue_actual": r.get("revenue_actual"),
|
||||||
|
"source": source,
|
||||||
|
"fetched_at": now,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
written += 1
|
||||||
|
return written
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_bulk_item(item: dict) -> dict | None:
|
||||||
|
sym = (item.get("symbol") or "").strip().upper()
|
||||||
|
d = item.get("date") or item.get("earningsDate")
|
||||||
|
if not sym or not d:
|
||||||
|
return None
|
||||||
|
return {
|
||||||
|
"symbol": sym.replace(".", "-"),
|
||||||
|
"announce_date": str(d)[:10],
|
||||||
|
"announce_time": item.get("time") or item.get("announceTime"),
|
||||||
|
"eps_estimate": _f(item.get("epsEstimated") or item.get("estimatedEarning")),
|
||||||
|
"eps_actual": _f(item.get("epsActual") or item.get("eps")),
|
||||||
|
"revenue_estimate": _f(item.get("revenueEstimated")),
|
||||||
|
"revenue_actual": _f(item.get("revenueActual")),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_symbol_item(item: dict, symbol: str) -> dict | None:
|
||||||
|
d = item.get("date")
|
||||||
|
if not d:
|
||||||
|
return None
|
||||||
|
return {
|
||||||
|
"symbol": symbol.replace(".", "-").upper(),
|
||||||
|
"announce_date": str(d)[:10],
|
||||||
|
"announce_time": item.get("time"),
|
||||||
|
"eps_estimate": _f(item.get("epsEstimated")),
|
||||||
|
"eps_actual": _f(item.get("epsActual")),
|
||||||
|
"revenue_estimate": _f(item.get("revenueEstimated")),
|
||||||
|
"revenue_actual": _f(item.get("revenueActual")),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _f(v) -> float | None:
|
||||||
|
if v is None or v == "":
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return float(v)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
async def _try_bulk(
|
||||||
|
client: httpx.AsyncClient,
|
||||||
|
api_key: str,
|
||||||
|
start: date,
|
||||||
|
end: date,
|
||||||
|
*,
|
||||||
|
window_days: int = 30,
|
||||||
|
) -> tuple[list[dict], int, str | None]:
|
||||||
|
"""Return (rows, requests_used, error_note)."""
|
||||||
|
rows: list[dict] = []
|
||||||
|
reqs = 0
|
||||||
|
cur = start
|
||||||
|
while cur <= end:
|
||||||
|
win_end = min(end, cur + timedelta(days=window_days - 1))
|
||||||
|
resp = await client.get(
|
||||||
|
f"{FMP_STABLE}/earnings-calendar",
|
||||||
|
params={
|
||||||
|
"from": cur.isoformat(),
|
||||||
|
"to": win_end.isoformat(),
|
||||||
|
"apikey": api_key,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
reqs += 1
|
||||||
|
if resp.status_code in (402, 403):
|
||||||
|
return [], reqs, f"bulk_unavailable status={resp.status_code}"
|
||||||
|
if resp.status_code == 429:
|
||||||
|
return rows, reqs, "rate_limited"
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
if not isinstance(data, list):
|
||||||
|
return [], reqs, f"unexpected bulk payload type={type(data)}"
|
||||||
|
for item in data:
|
||||||
|
if isinstance(item, dict):
|
||||||
|
parsed = _parse_bulk_item(item)
|
||||||
|
if parsed:
|
||||||
|
rows.append(parsed)
|
||||||
|
cur = win_end + timedelta(days=1)
|
||||||
|
return rows, reqs, None
|
||||||
|
|
||||||
|
|
||||||
|
async def _fetch_symbol(
|
||||||
|
client: httpx.AsyncClient, api_key: str, symbol: str
|
||||||
|
) -> list[dict]:
|
||||||
|
resp = await client.get(
|
||||||
|
f"{FMP_STABLE}/earnings",
|
||||||
|
params={"symbol": symbol, "apikey": api_key},
|
||||||
|
)
|
||||||
|
if resp.status_code == 429:
|
||||||
|
raise RuntimeError("rate_limited")
|
||||||
|
if resp.status_code == 402:
|
||||||
|
return []
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
if not isinstance(data, list):
|
||||||
|
return []
|
||||||
|
out: list[dict] = []
|
||||||
|
for item in data:
|
||||||
|
if isinstance(item, dict):
|
||||||
|
parsed = _parse_symbol_item(item, symbol)
|
||||||
|
if parsed:
|
||||||
|
out.append(parsed)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _fetch_symbol_alpha_vantage(
|
||||||
|
client: httpx.AsyncClient, api_key: str, symbol: str
|
||||||
|
) -> list[dict]:
|
||||||
|
"""Alpha Vantage EARNINGS — includes reportedDate (announce) + estimate/actual."""
|
||||||
|
resp = await client.get(
|
||||||
|
"https://www.alphavantage.co/query",
|
||||||
|
params={"function": "EARNINGS", "symbol": symbol, "apikey": api_key},
|
||||||
|
)
|
||||||
|
if resp.status_code == 429:
|
||||||
|
raise RuntimeError("rate_limited")
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
return []
|
||||||
|
note = str(data.get("Note") or data.get("Information") or "")
|
||||||
|
if "rate limit" in note.lower() or "Thank you for using Alpha Vantage" in note:
|
||||||
|
raise RuntimeError("rate_limited")
|
||||||
|
if data.get("Error Message"):
|
||||||
|
return []
|
||||||
|
quarterly = data.get("quarterlyEarnings") or []
|
||||||
|
out: list[dict] = []
|
||||||
|
for item in quarterly:
|
||||||
|
if not isinstance(item, dict):
|
||||||
|
continue
|
||||||
|
# Prefer announce (reportedDate); fall back to fiscal end (worse PIT).
|
||||||
|
ad = item.get("reportedDate") or item.get("fiscalDateEnding")
|
||||||
|
if not ad:
|
||||||
|
continue
|
||||||
|
out.append({
|
||||||
|
"symbol": symbol.replace(".", "-").upper(),
|
||||||
|
"announce_date": str(ad)[:10],
|
||||||
|
"announce_time": item.get("reportTime"),
|
||||||
|
"eps_estimate": _f(item.get("estimatedEPS")),
|
||||||
|
"eps_actual": _f(item.get("reportedEPS")),
|
||||||
|
"revenue_estimate": None,
|
||||||
|
"revenue_actual": None,
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _main() -> None:
|
||||||
|
args = _parse_args()
|
||||||
|
snapshot = Path(args.snapshot)
|
||||||
|
if not snapshot.exists():
|
||||||
|
raise SystemExit(f"Snapshot not found: {snapshot}")
|
||||||
|
|
||||||
|
from app.config import settings
|
||||||
|
|
||||||
|
if not settings.fmp_api_key:
|
||||||
|
raise SystemExit("FMP_API_KEY required")
|
||||||
|
|
||||||
|
start = date.fromisoformat(args.from_date)
|
||||||
|
end = date.fromisoformat(args.to_date) if args.to_date else date.today()
|
||||||
|
engine = create_engine(
|
||||||
|
f"sqlite:///{snapshot.resolve().as_posix()}",
|
||||||
|
future=True,
|
||||||
|
)
|
||||||
|
_ensure_tables(engine)
|
||||||
|
|
||||||
|
with engine.connect() as conn:
|
||||||
|
symbols = [
|
||||||
|
str(r[0]).upper().replace(".", "-")
|
||||||
|
for r in conn.execute(text("SELECT symbol FROM tickers ORDER BY symbol"))
|
||||||
|
]
|
||||||
|
done = set()
|
||||||
|
if not args.refetch_done:
|
||||||
|
done = {
|
||||||
|
str(r[0])
|
||||||
|
for r in conn.execute(
|
||||||
|
text(
|
||||||
|
"SELECT symbol FROM earnings_backfill_meta "
|
||||||
|
"WHERE status='done' AND n_events > 0"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
pending = [s for s in symbols if s not in done]
|
||||||
|
print(f"Snapshot: {snapshot}")
|
||||||
|
print(f"Universe: {len(symbols)}; pending: {len(pending)}; done: {len(done)}")
|
||||||
|
print(f"Window filter: {start} → {end}")
|
||||||
|
print(f"Provider: {args.provider}")
|
||||||
|
|
||||||
|
req_budget = int(args.limit)
|
||||||
|
reqs_used = 0
|
||||||
|
events_written = 0
|
||||||
|
mode = "per_symbol"
|
||||||
|
use_av = args.provider in ("alpha_vantage", "auto") and bool(
|
||||||
|
getattr(settings, "alpha_vantage_api_key", "")
|
||||||
|
)
|
||||||
|
use_fmp = args.provider in ("fmp", "auto") and bool(settings.fmp_api_key)
|
||||||
|
|
||||||
|
async with httpx.AsyncClient(timeout=60.0) as client:
|
||||||
|
if (
|
||||||
|
not args.force_symbol
|
||||||
|
and req_budget > 0
|
||||||
|
and use_fmp
|
||||||
|
and args.provider != "alpha_vantage"
|
||||||
|
):
|
||||||
|
print("Attempting bulk earnings-calendar…")
|
||||||
|
bulk_rows, bulk_reqs, err = await _try_bulk(
|
||||||
|
client, settings.fmp_api_key, start, end
|
||||||
|
)
|
||||||
|
reqs_used += bulk_reqs
|
||||||
|
if err:
|
||||||
|
print(f" Bulk unavailable: {err} (requests={bulk_reqs})")
|
||||||
|
else:
|
||||||
|
# Filter to universe.
|
||||||
|
uni = set(symbols)
|
||||||
|
bulk_rows = [r for r in bulk_rows if r["symbol"] in uni]
|
||||||
|
with engine.begin() as conn:
|
||||||
|
events_written += _upsert_events(conn, bulk_rows, "fmp_earnings_calendar")
|
||||||
|
for sym in symbols:
|
||||||
|
n = conn.execute(
|
||||||
|
text(
|
||||||
|
"SELECT COUNT(*) FROM earnings_events WHERE symbol=:s"
|
||||||
|
),
|
||||||
|
{"s": sym},
|
||||||
|
).scalar_one()
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
INSERT INTO earnings_backfill_meta(symbol, status, n_events, updated_at, note)
|
||||||
|
VALUES (:s, 'done', :n, :t, 'bulk')
|
||||||
|
ON CONFLICT(symbol) DO UPDATE SET
|
||||||
|
status='done', n_events=excluded.n_events,
|
||||||
|
updated_at=excluded.updated_at, note=excluded.note
|
||||||
|
"""
|
||||||
|
),
|
||||||
|
{
|
||||||
|
"s": sym,
|
||||||
|
"n": int(n),
|
||||||
|
"t": datetime.now(timezone.utc).isoformat(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
mode = "bulk"
|
||||||
|
print(f" Bulk wrote {events_written} events; requests={bulk_reqs}")
|
||||||
|
pending = []
|
||||||
|
|
||||||
|
# Per-symbol fallback / completion.
|
||||||
|
fmp_limited = False
|
||||||
|
for sym in pending:
|
||||||
|
if reqs_used >= req_budget:
|
||||||
|
print(f"Request budget exhausted ({req_budget}). Resume later.")
|
||||||
|
break
|
||||||
|
items: list[dict] = []
|
||||||
|
source = "fmp_earnings"
|
||||||
|
note = "per_symbol"
|
||||||
|
try:
|
||||||
|
if use_fmp and not fmp_limited and args.provider != "alpha_vantage":
|
||||||
|
items = await _fetch_symbol(client, settings.fmp_api_key, sym)
|
||||||
|
source = "fmp_earnings"
|
||||||
|
note = "fmp_per_symbol"
|
||||||
|
# Empty list may mean soft-limit or no data — try AV if available.
|
||||||
|
if not items and use_av:
|
||||||
|
items = await _fetch_symbol_alpha_vantage(
|
||||||
|
client, settings.alpha_vantage_api_key, sym
|
||||||
|
)
|
||||||
|
source = "alpha_vantage_earnings"
|
||||||
|
note = "av_after_fmp_empty"
|
||||||
|
reqs_used += 1 # count AV call separately below too
|
||||||
|
elif use_av:
|
||||||
|
items = await _fetch_symbol_alpha_vantage(
|
||||||
|
client, settings.alpha_vantage_api_key, sym
|
||||||
|
)
|
||||||
|
source = "alpha_vantage_earnings"
|
||||||
|
note = "av_per_symbol"
|
||||||
|
else:
|
||||||
|
raise RuntimeError("no provider available")
|
||||||
|
except Exception as exc:
|
||||||
|
msg = str(exc)
|
||||||
|
print(f" FAIL {sym}: {msg}")
|
||||||
|
reqs_used += 1
|
||||||
|
if "rate_limited" in msg and note.startswith("fmp"):
|
||||||
|
fmp_limited = True
|
||||||
|
with engine.begin() as conn:
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
INSERT INTO earnings_backfill_meta(symbol, status, n_events, updated_at, note)
|
||||||
|
VALUES (:s, 'error', 0, :t, :n)
|
||||||
|
ON CONFLICT(symbol) DO UPDATE SET
|
||||||
|
status='error', updated_at=excluded.updated_at, note=excluded.note
|
||||||
|
"""
|
||||||
|
),
|
||||||
|
{
|
||||||
|
"s": sym,
|
||||||
|
"t": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"n": msg[:200],
|
||||||
|
},
|
||||||
|
)
|
||||||
|
if args.sleep > 0:
|
||||||
|
await asyncio.sleep(args.sleep)
|
||||||
|
continue
|
||||||
|
|
||||||
|
reqs_used += 1
|
||||||
|
# Keep all rows with dates on/before end — SUE needs trailing history.
|
||||||
|
filtered = [
|
||||||
|
r for r in items if r["announce_date"] <= end.isoformat()
|
||||||
|
]
|
||||||
|
# Do NOT mark empty as done — leave pending for another provider/day.
|
||||||
|
status = "done" if filtered else "empty"
|
||||||
|
with engine.begin() as conn:
|
||||||
|
n_w = _upsert_events(conn, filtered, source) if filtered else 0
|
||||||
|
events_written += n_w
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
INSERT INTO earnings_backfill_meta(symbol, status, n_events, updated_at, note)
|
||||||
|
VALUES (:s, :st, :n, :t, :note)
|
||||||
|
ON CONFLICT(symbol) DO UPDATE SET
|
||||||
|
status=excluded.status, n_events=excluded.n_events,
|
||||||
|
updated_at=excluded.updated_at, note=excluded.note
|
||||||
|
"""
|
||||||
|
),
|
||||||
|
{
|
||||||
|
"s": sym,
|
||||||
|
"st": status,
|
||||||
|
"n": len(filtered),
|
||||||
|
"t": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"note": note,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
if reqs_used % 10 == 0 or reqs_used == 1:
|
||||||
|
print(
|
||||||
|
f" progress reqs={reqs_used}/{req_budget} last={sym} "
|
||||||
|
f"events_batch={len(filtered)} src={source}"
|
||||||
|
)
|
||||||
|
# AV free tier is ~5/min or 25/day — be polite when using it.
|
||||||
|
sleep_s = float(args.sleep)
|
||||||
|
if source.startswith("alpha_vantage"):
|
||||||
|
sleep_s = max(sleep_s, 12.0)
|
||||||
|
if sleep_s > 0:
|
||||||
|
await asyncio.sleep(sleep_s)
|
||||||
|
|
||||||
|
with engine.connect() as conn:
|
||||||
|
total_events = int(
|
||||||
|
conn.execute(text("SELECT COUNT(*) FROM earnings_events")).scalar_one()
|
||||||
|
)
|
||||||
|
done_n = int(
|
||||||
|
conn.execute(
|
||||||
|
text("SELECT COUNT(*) FROM earnings_backfill_meta WHERE status='done'")
|
||||||
|
).scalar_one()
|
||||||
|
)
|
||||||
|
d_range = conn.execute(
|
||||||
|
text("SELECT MIN(announce_date), MAX(announce_date) FROM earnings_events")
|
||||||
|
).fetchone()
|
||||||
|
with_actual = int(
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"SELECT COUNT(*) FROM earnings_events "
|
||||||
|
"WHERE eps_actual IS NOT NULL AND eps_estimate IS NOT NULL"
|
||||||
|
)
|
||||||
|
).scalar_one()
|
||||||
|
)
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
"mode": mode,
|
||||||
|
"fmp_requests": reqs_used,
|
||||||
|
"events_written_this_run": events_written,
|
||||||
|
"total_events": total_events,
|
||||||
|
"symbols_done": done_n,
|
||||||
|
"symbols_universe": len(symbols),
|
||||||
|
"announce_date_range": {"min": d_range[0], "max": d_range[1]},
|
||||||
|
"events_with_actual_and_estimate": with_actual,
|
||||||
|
"budget": req_budget,
|
||||||
|
"complete": done_n >= len(symbols),
|
||||||
|
}
|
||||||
|
print(json.dumps(summary, indent=2))
|
||||||
|
out = Path("reports") / "earnings-backfill-status.json"
|
||||||
|
out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
out.write_text(json.dumps(summary, indent=2) + "\n", encoding="utf-8")
|
||||||
|
print(f"Wrote {out}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(_main())
|
||||||
@@ -0,0 +1,229 @@
|
|||||||
|
"""Build a local ticker → GICS sector map for research residualization.
|
||||||
|
|
||||||
|
Sources (in order):
|
||||||
|
1. Public S&P 500 constituents CSV (datasets/s-and-p-500-companies) — bulk, free.
|
||||||
|
2. Existing map file (resume).
|
||||||
|
3. FMP stable ``profile`` for still-missing symbols (budget ~250 req/day).
|
||||||
|
|
||||||
|
Writes ``data/research/ticker_sector_map.json``. Never touches production Postgres.
|
||||||
|
|
||||||
|
Example
|
||||||
|
-------
|
||||||
|
python scripts/build_ticker_sector_map.py \\
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite
|
||||||
|
|
||||||
|
python scripts/build_ticker_sector_map.py --fmp-limit 50
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
from sqlalchemy import create_engine, text
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
if str(ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
|
||||||
|
from app.services.sector_map import ( # noqa: E402
|
||||||
|
DEFAULT_SECTOR_MAP_PATH,
|
||||||
|
coverage_stats,
|
||||||
|
load_ticker_sector_map,
|
||||||
|
normalise_symbol,
|
||||||
|
save_ticker_sector_map,
|
||||||
|
sector_to_etf,
|
||||||
|
)
|
||||||
|
|
||||||
|
SP500_CSV_URL = (
|
||||||
|
"https://raw.githubusercontent.com/datasets/s-and-p-500-companies/"
|
||||||
|
"master/data/constituents.csv"
|
||||||
|
)
|
||||||
|
FMP_STABLE = "https://financialmodelingprep.com/stable"
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_args() -> argparse.Namespace:
|
||||||
|
p = argparse.ArgumentParser(description=__doc__)
|
||||||
|
p.add_argument(
|
||||||
|
"--snapshot",
|
||||||
|
default="backtest_snapshots/prod.sqlite",
|
||||||
|
help="Snapshot whose tickers define the universe.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--out",
|
||||||
|
default=str(DEFAULT_SECTOR_MAP_PATH),
|
||||||
|
help="Output JSON path.",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--fmp-limit",
|
||||||
|
type=int,
|
||||||
|
default=200,
|
||||||
|
help="Max FMP profile requests this run (free-tier cushion).",
|
||||||
|
)
|
||||||
|
p.add_argument(
|
||||||
|
"--skip-fmp",
|
||||||
|
action="store_true",
|
||||||
|
help="Only use public SP500 CSV + existing map.",
|
||||||
|
)
|
||||||
|
p.add_argument("--sleep", type=float, default=0.35, help="Pause between FMP calls.")
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def _snapshot_symbols(snapshot: Path) -> list[str]:
|
||||||
|
engine = create_engine(f"sqlite:///{snapshot.resolve().as_posix()}", future=True)
|
||||||
|
try:
|
||||||
|
with engine.connect() as conn:
|
||||||
|
rows = conn.execute(text("SELECT symbol FROM tickers ORDER BY symbol")).fetchall()
|
||||||
|
finally:
|
||||||
|
engine.dispose()
|
||||||
|
return [normalise_symbol(r[0]) for r in rows if r[0]]
|
||||||
|
|
||||||
|
|
||||||
|
def _fetch_sp500_map() -> dict[str, str]:
|
||||||
|
with httpx.Client(timeout=60.0, follow_redirects=True) as client:
|
||||||
|
resp = client.get(SP500_CSV_URL)
|
||||||
|
resp.raise_for_status()
|
||||||
|
reader = csv.DictReader(io.StringIO(resp.text))
|
||||||
|
out: dict[str, str] = {}
|
||||||
|
for row in reader:
|
||||||
|
sym = normalise_symbol(row.get("Symbol") or "")
|
||||||
|
sector = (row.get("GICS Sector") or "").strip()
|
||||||
|
if sym and sector:
|
||||||
|
out[sym] = sector
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _fmp_profile_sector(client: httpx.AsyncClient, api_key: str, symbol: str) -> str | None:
|
||||||
|
resp = await client.get(
|
||||||
|
f"{FMP_STABLE}/profile",
|
||||||
|
params={"symbol": symbol, "apikey": api_key},
|
||||||
|
)
|
||||||
|
if resp.status_code == 429:
|
||||||
|
raise RuntimeError(f"FMP rate limited on {symbol}")
|
||||||
|
if resp.status_code == 402:
|
||||||
|
return None
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
if isinstance(data, list):
|
||||||
|
data = data[0] if data else {}
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
return None
|
||||||
|
sector = (data.get("sector") or data.get("industry") or "").strip()
|
||||||
|
# industry alone is not a GICS sector — only accept if we can map to an ETF
|
||||||
|
if sector and sector_to_etf(sector):
|
||||||
|
return sector
|
||||||
|
# FMP sometimes returns industry under sector when sector missing; try sector field only
|
||||||
|
sec = (data.get("sector") or "").strip()
|
||||||
|
return sec or None
|
||||||
|
|
||||||
|
|
||||||
|
async def _fill_from_fmp(
|
||||||
|
missing: list[str],
|
||||||
|
*,
|
||||||
|
api_key: str,
|
||||||
|
limit: int,
|
||||||
|
sleep_s: float,
|
||||||
|
) -> tuple[dict[str, str], int]:
|
||||||
|
filled: dict[str, str] = {}
|
||||||
|
used = 0
|
||||||
|
async with httpx.AsyncClient(timeout=30.0) as client:
|
||||||
|
for sym in missing:
|
||||||
|
if used >= limit:
|
||||||
|
break
|
||||||
|
try:
|
||||||
|
sector = await _fmp_profile_sector(client, api_key, sym)
|
||||||
|
except Exception as exc:
|
||||||
|
print(f" FMP fail {sym}: {exc}")
|
||||||
|
used += 1
|
||||||
|
await asyncio.sleep(sleep_s)
|
||||||
|
continue
|
||||||
|
used += 1
|
||||||
|
if sector:
|
||||||
|
filled[sym] = sector
|
||||||
|
print(f" FMP {sym} → {sector}")
|
||||||
|
else:
|
||||||
|
print(f" FMP {sym} → (no sector)")
|
||||||
|
if sleep_s > 0:
|
||||||
|
await asyncio.sleep(sleep_s)
|
||||||
|
return filled, used
|
||||||
|
|
||||||
|
|
||||||
|
async def _main() -> None:
|
||||||
|
args = _parse_args()
|
||||||
|
snapshot = Path(args.snapshot)
|
||||||
|
if not snapshot.exists():
|
||||||
|
raise SystemExit(f"Snapshot not found: {snapshot}")
|
||||||
|
|
||||||
|
symbols = _snapshot_symbols(snapshot)
|
||||||
|
print(f"Universe: {len(symbols)} symbols from {snapshot}")
|
||||||
|
|
||||||
|
existing = load_ticker_sector_map(args.out)
|
||||||
|
print(f"Existing map entries: {len(existing)}")
|
||||||
|
|
||||||
|
print("Fetching public S&P 500 sector CSV…")
|
||||||
|
sp500 = _fetch_sp500_map()
|
||||||
|
print(f" SP500 CSV rows: {len(sp500)}")
|
||||||
|
|
||||||
|
mapping = dict(existing)
|
||||||
|
from_sp500 = 0
|
||||||
|
for sym in symbols:
|
||||||
|
if sym in mapping:
|
||||||
|
continue
|
||||||
|
if sym in sp500:
|
||||||
|
mapping[sym] = sp500[sym]
|
||||||
|
from_sp500 += 1
|
||||||
|
print(f" Newly filled from SP500 CSV: {from_sp500}")
|
||||||
|
|
||||||
|
missing = [s for s in symbols if s not in mapping]
|
||||||
|
fmp_used = 0
|
||||||
|
from_fmp = 0
|
||||||
|
if missing and not args.skip_fmp:
|
||||||
|
from app.config import settings
|
||||||
|
|
||||||
|
if not settings.fmp_api_key:
|
||||||
|
print("WARNING: FMP key missing; leaving gaps unfilled")
|
||||||
|
else:
|
||||||
|
print(f"FMP fill for {len(missing)} missing (limit={args.fmp_limit})…")
|
||||||
|
filled, fmp_used = await _fill_from_fmp(
|
||||||
|
missing,
|
||||||
|
api_key=settings.fmp_api_key,
|
||||||
|
limit=int(args.fmp_limit),
|
||||||
|
sleep_s=float(args.sleep),
|
||||||
|
)
|
||||||
|
mapping.update(filled)
|
||||||
|
from_fmp = len(filled)
|
||||||
|
|
||||||
|
still_missing = [s for s in symbols if s not in mapping]
|
||||||
|
stats = coverage_stats(symbols, mapping)
|
||||||
|
meta = {
|
||||||
|
"built_at": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"snapshot": str(snapshot.resolve()),
|
||||||
|
"from_existing": len(existing),
|
||||||
|
"from_sp500_csv": from_sp500,
|
||||||
|
"from_fmp": from_fmp,
|
||||||
|
"fmp_requests": fmp_used,
|
||||||
|
"still_missing": still_missing,
|
||||||
|
"coverage": {
|
||||||
|
k: stats[k]
|
||||||
|
for k in ("universe", "mapped", "mapped_pct", "with_etf", "by_sector")
|
||||||
|
},
|
||||||
|
}
|
||||||
|
out_path = save_ticker_sector_map(mapping, args.out, meta=meta)
|
||||||
|
print(f"Wrote {out_path}")
|
||||||
|
print(json.dumps(meta["coverage"], indent=2))
|
||||||
|
if still_missing:
|
||||||
|
print(f"Still missing ({len(still_missing)}): {still_missing[:40]}")
|
||||||
|
if len(still_missing) > 40:
|
||||||
|
print(f" … +{len(still_missing) - 40} more")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(_main())
|
||||||
@@ -0,0 +1,183 @@
|
|||||||
|
"""Fetch the 11 SPDR sector ETFs into a snapshot's ``benchmark_prices``.
|
||||||
|
|
||||||
|
Research-only. Sector ETFs are auxiliary series (like SPY) — they must not
|
||||||
|
enter the tradable ticker universe or candidate replay. Storing them in
|
||||||
|
``benchmark_prices`` keeps that invariant.
|
||||||
|
|
||||||
|
Also refreshes SPY on the same window so residual factors share a calendar.
|
||||||
|
|
||||||
|
Example
|
||||||
|
-------
|
||||||
|
python scripts/fetch_sector_etfs_to_snapshot.py \\
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite --history-days 2200
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from datetime import date, timedelta
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from sqlalchemy import create_engine, text
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
if str(ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
|
||||||
|
from app.services.sector_map import SECTOR_ETFS # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_args() -> argparse.Namespace:
|
||||||
|
p = argparse.ArgumentParser(description=__doc__)
|
||||||
|
p.add_argument("--snapshot", default="backtest_snapshots/prod.sqlite")
|
||||||
|
p.add_argument(
|
||||||
|
"--history-days",
|
||||||
|
type=int,
|
||||||
|
default=2200,
|
||||||
|
help="Lookback calendar days (default ~6y; covers 5y snapshot + cushion).",
|
||||||
|
)
|
||||||
|
p.add_argument("--sleep", type=float, default=0.25)
|
||||||
|
p.add_argument(
|
||||||
|
"--symbols",
|
||||||
|
default=None,
|
||||||
|
help="Comma-separated override (default: SPY + 11 sector ETFs).",
|
||||||
|
)
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
async def _fetch_and_upsert(
|
||||||
|
engine,
|
||||||
|
provider,
|
||||||
|
symbol: str,
|
||||||
|
start: date,
|
||||||
|
end: date,
|
||||||
|
*,
|
||||||
|
sleep_s: float,
|
||||||
|
) -> int:
|
||||||
|
from app.exceptions import ProviderError, RateLimitError
|
||||||
|
|
||||||
|
for attempt in range(5):
|
||||||
|
try:
|
||||||
|
bars = await provider.fetch_ohlcv(symbol, start, end)
|
||||||
|
break
|
||||||
|
except RateLimitError:
|
||||||
|
wait = min(60.0, 2.0 ** attempt)
|
||||||
|
print(f" rate limited {symbol}; sleep {wait:.0f}s")
|
||||||
|
await asyncio.sleep(wait)
|
||||||
|
bars = []
|
||||||
|
except ProviderError as exc:
|
||||||
|
if attempt + 1 >= 5:
|
||||||
|
raise
|
||||||
|
await asyncio.sleep(1.0)
|
||||||
|
print(f" retry {symbol}: {exc}")
|
||||||
|
bars = []
|
||||||
|
else:
|
||||||
|
bars = []
|
||||||
|
|
||||||
|
if sleep_s > 0:
|
||||||
|
await asyncio.sleep(sleep_s)
|
||||||
|
|
||||||
|
if not bars:
|
||||||
|
print(f" {symbol}: empty")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
written = 0
|
||||||
|
with engine.begin() as conn:
|
||||||
|
for bar in bars:
|
||||||
|
d = bar.date.isoformat() if hasattr(bar.date, "isoformat") else str(bar.date)
|
||||||
|
close = float(bar.close)
|
||||||
|
existing = conn.execute(
|
||||||
|
text(
|
||||||
|
"SELECT id, close FROM benchmark_prices "
|
||||||
|
"WHERE symbol = :sym AND date = :d"
|
||||||
|
),
|
||||||
|
{"sym": symbol, "d": d},
|
||||||
|
).fetchone()
|
||||||
|
if existing is None:
|
||||||
|
# id is INTEGER PK — let sqlite autoincrement if possible
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"INSERT INTO benchmark_prices (symbol, date, close) "
|
||||||
|
"VALUES (:sym, :d, :c)"
|
||||||
|
),
|
||||||
|
{"sym": symbol, "d": d, "c": close},
|
||||||
|
)
|
||||||
|
written += 1
|
||||||
|
elif abs(float(existing[1]) - close) > 1e-9:
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"UPDATE benchmark_prices SET close = :c WHERE id = :id"
|
||||||
|
),
|
||||||
|
{"c": close, "id": int(existing[0])},
|
||||||
|
)
|
||||||
|
written += 1
|
||||||
|
print(f" {symbol}: {len(bars)} bars, {written} rows written/updated")
|
||||||
|
return written
|
||||||
|
|
||||||
|
|
||||||
|
async def _main() -> None:
|
||||||
|
args = _parse_args()
|
||||||
|
snapshot = Path(args.snapshot)
|
||||||
|
if not snapshot.exists():
|
||||||
|
raise SystemExit(f"Snapshot not found: {snapshot}")
|
||||||
|
|
||||||
|
from app.config import settings
|
||||||
|
from app.providers.alpaca import AlpacaOHLCVProvider
|
||||||
|
|
||||||
|
if not settings.alpaca_api_key or not settings.alpaca_api_secret:
|
||||||
|
raise SystemExit("ALPACA_API_KEY / ALPACA_API_SECRET required")
|
||||||
|
|
||||||
|
if args.symbols:
|
||||||
|
symbols = [s.strip().upper() for s in args.symbols.split(",") if s.strip()]
|
||||||
|
else:
|
||||||
|
symbols = ["SPY", *SECTOR_ETFS]
|
||||||
|
|
||||||
|
end = date.today()
|
||||||
|
start = end - timedelta(days=int(args.history_days))
|
||||||
|
provider = AlpacaOHLCVProvider(settings.alpaca_api_key, settings.alpaca_api_secret)
|
||||||
|
engine = create_engine(
|
||||||
|
f"sqlite:///{snapshot.resolve().as_posix()}",
|
||||||
|
future=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
print(f"Snapshot: {snapshot}")
|
||||||
|
print(f"Window: {start} → {end}")
|
||||||
|
print(f"Symbols: {symbols}")
|
||||||
|
|
||||||
|
t0 = time.monotonic()
|
||||||
|
total = 0
|
||||||
|
try:
|
||||||
|
for sym in symbols:
|
||||||
|
n = await _fetch_and_upsert(
|
||||||
|
engine, provider, sym, start, end, sleep_s=float(args.sleep)
|
||||||
|
)
|
||||||
|
total += n
|
||||||
|
finally:
|
||||||
|
engine.dispose()
|
||||||
|
|
||||||
|
# Summary counts
|
||||||
|
engine = create_engine(
|
||||||
|
f"sqlite:///{snapshot.resolve().as_posix()}",
|
||||||
|
future=True,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with engine.connect() as conn:
|
||||||
|
rows = conn.execute(
|
||||||
|
text(
|
||||||
|
"SELECT symbol, COUNT(*), MIN(date), MAX(date) "
|
||||||
|
"FROM benchmark_prices GROUP BY symbol ORDER BY symbol"
|
||||||
|
)
|
||||||
|
).fetchall()
|
||||||
|
finally:
|
||||||
|
engine.dispose()
|
||||||
|
|
||||||
|
print(f"Done in {(time.monotonic() - t0) / 60:.1f}m; rows touched={total}")
|
||||||
|
for sym, n, d0, d1 in rows:
|
||||||
|
print(f" {sym}: n={n} {d0}→{d1}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(_main())
|
||||||
@@ -0,0 +1,927 @@
|
|||||||
|
"""Earnings gap diagnostic (2a) + SUE IC (2b). Local research only.
|
||||||
|
|
||||||
|
Requires ``earnings_events`` on the snapshot (see backfill_earnings_events.py).
|
||||||
|
|
||||||
|
Example
|
||||||
|
-------
|
||||||
|
python scripts/run_earnings_research.py \\
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite --workers 6 --allow-spawn
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
from collections import defaultdict
|
||||||
|
from datetime import date, datetime, timedelta
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from sqlalchemy import create_engine, text
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
if str(ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
|
||||||
|
IRON_IC_BAR = 0.03
|
||||||
|
MIN_RELIABLE = 12
|
||||||
|
SUE_CARRY_DAYS = 63
|
||||||
|
SUE_TRAIL = 8
|
||||||
|
|
||||||
|
|
||||||
|
def _sqlite_url(path: Path) -> str:
|
||||||
|
return f"sqlite+aiosqlite:///{path.resolve().as_posix()}"
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_args() -> argparse.Namespace:
|
||||||
|
p = argparse.ArgumentParser(description=__doc__)
|
||||||
|
p.add_argument("--snapshot", default="backtest_snapshots/prod.sqlite")
|
||||||
|
p.add_argument("--workers", type=int, default=6)
|
||||||
|
p.add_argument("--allow-spawn", action="store_true")
|
||||||
|
p.add_argument("--skip-2a", action="store_true")
|
||||||
|
p.add_argument("--skip-2b", action="store_true")
|
||||||
|
p.add_argument("--quiet", action="store_true")
|
||||||
|
p.add_argument("--out", default=None)
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def _load_earnings(snapshot: Path) -> list[dict]:
|
||||||
|
engine = create_engine(
|
||||||
|
f"sqlite:///{snapshot.resolve().as_posix()}",
|
||||||
|
future=True,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with engine.connect() as conn:
|
||||||
|
# Table must exist.
|
||||||
|
tables = {
|
||||||
|
r[0]
|
||||||
|
for r in conn.execute(
|
||||||
|
text("SELECT name FROM sqlite_master WHERE type='table'")
|
||||||
|
)
|
||||||
|
}
|
||||||
|
if "earnings_events" not in tables:
|
||||||
|
raise SystemExit(
|
||||||
|
"earnings_events table missing — run scripts/backfill_earnings_events.py"
|
||||||
|
)
|
||||||
|
rows = conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
SELECT symbol, announce_date, announce_time,
|
||||||
|
eps_estimate, eps_actual, revenue_estimate, revenue_actual
|
||||||
|
FROM earnings_events
|
||||||
|
ORDER BY symbol, announce_date
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
).fetchall()
|
||||||
|
meta = {}
|
||||||
|
if "earnings_backfill_meta" in tables:
|
||||||
|
meta = {
|
||||||
|
"done": int(
|
||||||
|
conn.execute(
|
||||||
|
text(
|
||||||
|
"SELECT COUNT(*) FROM earnings_backfill_meta "
|
||||||
|
"WHERE status='done'"
|
||||||
|
)
|
||||||
|
).scalar_one()
|
||||||
|
),
|
||||||
|
"universe_tickers": int(
|
||||||
|
conn.execute(text("SELECT COUNT(*) FROM tickers")).scalar_one()
|
||||||
|
),
|
||||||
|
}
|
||||||
|
finally:
|
||||||
|
engine.dispose()
|
||||||
|
|
||||||
|
events = [
|
||||||
|
{
|
||||||
|
"symbol": str(r[0]).upper(),
|
||||||
|
"announce_date": date.fromisoformat(str(r[1])[:10]),
|
||||||
|
"announce_time": r[2],
|
||||||
|
"eps_estimate": r[3],
|
||||||
|
"eps_actual": r[4],
|
||||||
|
"revenue_estimate": r[5],
|
||||||
|
"revenue_actual": r[6],
|
||||||
|
}
|
||||||
|
for r in rows
|
||||||
|
]
|
||||||
|
return events, meta
|
||||||
|
|
||||||
|
|
||||||
|
def _percentile(xs: list[float], q: float) -> float | None:
|
||||||
|
if not xs:
|
||||||
|
return None
|
||||||
|
s = sorted(xs)
|
||||||
|
if len(s) == 1:
|
||||||
|
return s[0]
|
||||||
|
idx = q * (len(s) - 1)
|
||||||
|
lo = int(math.floor(idx))
|
||||||
|
hi = int(math.ceil(idx))
|
||||||
|
if lo == hi:
|
||||||
|
return s[lo]
|
||||||
|
w = idx - lo
|
||||||
|
return s[lo] * (1 - w) + s[hi] * w
|
||||||
|
|
||||||
|
|
||||||
|
def _r_dist(rs: list[float]) -> dict[str, Any]:
|
||||||
|
if not rs:
|
||||||
|
return {"n": 0}
|
||||||
|
return {
|
||||||
|
"n": len(rs),
|
||||||
|
"mean": round(sum(rs) / len(rs), 4),
|
||||||
|
"win_rate": round(sum(1 for r in rs if r > 0) / len(rs), 4),
|
||||||
|
"p05": round(_percentile(rs, 0.05), 4),
|
||||||
|
"p25": round(_percentile(rs, 0.25), 4),
|
||||||
|
"p50": round(_percentile(rs, 0.50), 4),
|
||||||
|
"p75": round(_percentile(rs, 0.75), 4),
|
||||||
|
"p95": round(_percentile(rs, 0.95), 4),
|
||||||
|
"min": round(min(rs), 4),
|
||||||
|
"max": round(max(rs), 4),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _trading_days_between(
|
||||||
|
entry: date, exit_: date, calendar: set[date]
|
||||||
|
) -> list[date]:
|
||||||
|
"""Inclusive trading dates in [entry, exit_] present on the union calendar."""
|
||||||
|
out = []
|
||||||
|
d = entry
|
||||||
|
while d <= exit_:
|
||||||
|
if d in calendar:
|
||||||
|
out.append(d)
|
||||||
|
d += timedelta(days=1)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _nth_trading_day_after(
|
||||||
|
start: date, n: int, ordered_calendar: list[date]
|
||||||
|
) -> date | None:
|
||||||
|
"""First calendar date strictly after ``start``, then + (n-1) more sessions.
|
||||||
|
|
||||||
|
announce+1 trading day: n=1 → first session after announce date
|
||||||
|
(if announce is a trading day, still use the *next* session for PIT).
|
||||||
|
"""
|
||||||
|
# Sessions strictly after start.
|
||||||
|
after = [d for d in ordered_calendar if d > start]
|
||||||
|
if len(after) < n:
|
||||||
|
return None
|
||||||
|
return after[n - 1]
|
||||||
|
|
||||||
|
|
||||||
|
def _build_sue_series(
|
||||||
|
events_by_symbol: dict[str, list[dict]],
|
||||||
|
prices: dict[str, tuple],
|
||||||
|
) -> dict[str, dict[date, float]]:
|
||||||
|
"""symbol → {asof_date: sue_value} for days when SUE is live (announce+1 .. +63)."""
|
||||||
|
out: dict[str, dict[date, float]] = {}
|
||||||
|
for sym, cols in prices.items():
|
||||||
|
ords = cols[0]
|
||||||
|
closes = cols[4]
|
||||||
|
dates = [date.fromordinal(int(o)) for o in ords]
|
||||||
|
if not dates:
|
||||||
|
continue
|
||||||
|
ordered = dates # already chronological
|
||||||
|
cal_set = set(ordered)
|
||||||
|
events = events_by_symbol.get(sym.upper(), [])
|
||||||
|
# Chronological surprises with actual+estimate.
|
||||||
|
surprises: list[tuple[date, float, float]] = [] # announce, surprise, close_for_scale
|
||||||
|
for ev in events:
|
||||||
|
act, est = ev.get("eps_actual"), ev.get("eps_estimate")
|
||||||
|
if act is None or est is None:
|
||||||
|
continue
|
||||||
|
ad = ev["announce_date"]
|
||||||
|
# Close on/before announce for price fallback scale.
|
||||||
|
close_px = None
|
||||||
|
for d, c in zip(reversed(dates), reversed(closes)):
|
||||||
|
if d <= ad and float(c) > 0:
|
||||||
|
close_px = float(c)
|
||||||
|
break
|
||||||
|
surprises.append((ad, float(act) - float(est), close_px or 1.0))
|
||||||
|
surprises.sort(key=lambda x: x[0])
|
||||||
|
|
||||||
|
sue_on_day: dict[date, float] = {}
|
||||||
|
for i, (ad, surprise, px) in enumerate(surprises):
|
||||||
|
trail = [surprises[j][1] for j in range(max(0, i - SUE_TRAIL), i)]
|
||||||
|
# Need history of surprises; include current only for value, stdev from prior 8.
|
||||||
|
if len(trail) >= 3:
|
||||||
|
mean_t = sum(trail) / len(trail)
|
||||||
|
var = sum((x - mean_t) ** 2 for x in trail) / (len(trail) - 1)
|
||||||
|
sd = math.sqrt(var) if var > 0 else None
|
||||||
|
else:
|
||||||
|
sd = None
|
||||||
|
if sd is not None and sd > 1e-9:
|
||||||
|
sue = surprise / sd
|
||||||
|
else:
|
||||||
|
# Fallback: scale by price (EPS surprise / price).
|
||||||
|
sue = surprise / px if px > 0 else None
|
||||||
|
if sue is None or not math.isfinite(sue):
|
||||||
|
continue
|
||||||
|
usable_from = _nth_trading_day_after(ad, 1, ordered)
|
||||||
|
if usable_from is None:
|
||||||
|
continue
|
||||||
|
# Carry for SUE_CARRY_DAYS trading sessions starting at usable_from.
|
||||||
|
try:
|
||||||
|
start_idx = ordered.index(usable_from)
|
||||||
|
except ValueError:
|
||||||
|
# usable_from not in this symbol's calendar (halted etc.)
|
||||||
|
start_idx = next(
|
||||||
|
(k for k, d in enumerate(ordered) if d >= usable_from), None
|
||||||
|
)
|
||||||
|
if start_idx is None:
|
||||||
|
continue
|
||||||
|
end_idx = min(len(ordered) - 1, start_idx + SUE_CARRY_DAYS - 1)
|
||||||
|
for k in range(start_idx, end_idx + 1):
|
||||||
|
# Later announcements overwrite earlier carry (latest SUE wins).
|
||||||
|
sue_on_day[ordered[k]] = sue
|
||||||
|
if sue_on_day:
|
||||||
|
out[sym.upper()] = sue_on_day
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_2a(
|
||||||
|
snapshot: Path,
|
||||||
|
events: list[dict],
|
||||||
|
*,
|
||||||
|
quiet: bool,
|
||||||
|
workers: int,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
from app.config import settings
|
||||||
|
from app.services import backtest_service as bt
|
||||||
|
from app.services.admin_service import get_activation_config
|
||||||
|
from app.services.recommendation_service import get_recommendation_config
|
||||||
|
from app.services.paper_trade_service import get_exit_policy
|
||||||
|
from app.services.benchmark_service import load_benchmark_closes
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from sqlalchemy import select
|
||||||
|
|
||||||
|
os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1"
|
||||||
|
settings.backtest_workers = workers
|
||||||
|
|
||||||
|
engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True)
|
||||||
|
Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with Session() as db:
|
||||||
|
config = await get_recommendation_config(db)
|
||||||
|
activation = await get_activation_config(db)
|
||||||
|
exit_config = await get_exit_policy(db)
|
||||||
|
tickers = list(
|
||||||
|
(await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars()
|
||||||
|
)
|
||||||
|
spy = await load_benchmark_closes(db, "SPY")
|
||||||
|
prices: dict[str, tuple] = {}
|
||||||
|
candidates: list[dict] = []
|
||||||
|
for idx, t in enumerate(tickers):
|
||||||
|
if not quiet and idx % 50 == 0:
|
||||||
|
print(f" 2a fetch {idx}/{len(tickers)}", end="\r", flush=True)
|
||||||
|
cols = await bt._fetch_columns(db, t.symbol)
|
||||||
|
if cols is None:
|
||||||
|
continue
|
||||||
|
prices[t.symbol] = cols
|
||||||
|
cands, _ = bt._replay_and_signals(
|
||||||
|
t.symbol,
|
||||||
|
cols,
|
||||||
|
config,
|
||||||
|
activation,
|
||||||
|
spy,
|
||||||
|
bt.PRODUCTION_GTL_TARGET_MODEL,
|
||||||
|
"weekly",
|
||||||
|
False,
|
||||||
|
)
|
||||||
|
candidates.extend(cands)
|
||||||
|
finally:
|
||||||
|
await engine.dispose()
|
||||||
|
if not quiet:
|
||||||
|
print()
|
||||||
|
|
||||||
|
# Production ranks + qualify.
|
||||||
|
bt._assign_momentum_percentiles(candidates)
|
||||||
|
bt._assign_residual_momentum_percentiles(candidates)
|
||||||
|
bt._assign_low_volatility_percentiles(candidates)
|
||||||
|
bt._assign_activation_momentum_percentiles(candidates)
|
||||||
|
bt._assign_residual_high_vol_blend(candidates)
|
||||||
|
for c in candidates:
|
||||||
|
c["qualified"] = bt._momentum_qualifies(c, 80.0)
|
||||||
|
longs = [
|
||||||
|
c for c in candidates if c.get("qualified") and c.get("direction") == "long"
|
||||||
|
]
|
||||||
|
|
||||||
|
strategy = next(s for s in bt.PORTFOLIO_MONITOR_STRATEGIES if s.get("is_production"))
|
||||||
|
entry_cfg = bt._entry_variant_config(str(strategy["entry_variant"]))
|
||||||
|
assert entry_cfg is not None
|
||||||
|
ranking_key = str(entry_cfg.get("ranking_key") or entry_cfg["percentile_key"])
|
||||||
|
exit_policy = bt.LIVE_EXIT_MODE_TO_SIM.get(
|
||||||
|
str(exit_config.get("mode", "atr_trailing")), "atr_trail3"
|
||||||
|
)
|
||||||
|
hold_days = int(exit_config.get("hold_days", 30))
|
||||||
|
trail = float(exit_config.get("atr_multiplier", bt.ATR_TRAIL_MULTIPLIER))
|
||||||
|
reentry = bt._make_gate_reset_reentry_fn(
|
||||||
|
longs, prices, cadence="weekly", ranking_key=ranking_key
|
||||||
|
)
|
||||||
|
sim = bt._simulate_portfolio(
|
||||||
|
longs,
|
||||||
|
prices,
|
||||||
|
spy,
|
||||||
|
exit_policy,
|
||||||
|
hold_days,
|
||||||
|
ranking_key=ranking_key,
|
||||||
|
max_positions=int(entry_cfg["max_positions"]),
|
||||||
|
risk_per_trade=float(entry_cfg["risk_per_trade"]),
|
||||||
|
atr_trail_multiplier=trail,
|
||||||
|
post_stop_reentry_fn=reentry,
|
||||||
|
fill_mode=bt.FILL_MODE_CLOSE,
|
||||||
|
include_trades=True,
|
||||||
|
)
|
||||||
|
if sim is None:
|
||||||
|
return {"error": "no_trades"}
|
||||||
|
|
||||||
|
details = sim.get("trade_details") or []
|
||||||
|
# Build per-symbol earnings announce dates.
|
||||||
|
earns_by_sym: dict[str, list[date]] = defaultdict(list)
|
||||||
|
for ev in events:
|
||||||
|
earns_by_sym[ev["symbol"]].append(ev["announce_date"])
|
||||||
|
for sym in earns_by_sym:
|
||||||
|
earns_by_sym[sym].sort()
|
||||||
|
|
||||||
|
# Union trading calendar from prices.
|
||||||
|
cal: set[date] = set()
|
||||||
|
for cols in prices.values():
|
||||||
|
for o in cols[0]:
|
||||||
|
cal.add(date.fromordinal(int(o)))
|
||||||
|
ordered_cal = sorted(cal)
|
||||||
|
|
||||||
|
# Map entry date → list of announce dates for symbol (for pre-entry lookback).
|
||||||
|
trades_parsed: list[dict] = []
|
||||||
|
for t in details:
|
||||||
|
sym = str(t.get("symbol") or "").upper()
|
||||||
|
# Field names from simulator.
|
||||||
|
entry_s = t.get("entry_date") or t.get("open_date") or t.get("date")
|
||||||
|
exit_s = t.get("exit_date") or t.get("close_date")
|
||||||
|
r = t.get("realized_r")
|
||||||
|
if r is None:
|
||||||
|
r = t.get("r")
|
||||||
|
if entry_s is None or exit_s is None or r is None:
|
||||||
|
continue
|
||||||
|
entry_d = date.fromisoformat(str(entry_s)[:10])
|
||||||
|
exit_d = date.fromisoformat(str(exit_s)[:10])
|
||||||
|
announces = earns_by_sym.get(sym, [])
|
||||||
|
# Earnings between entry and exit (exclusive of entry day? inclusive hold).
|
||||||
|
# "between entry and exit" — any announce with entry < announce <= exit
|
||||||
|
# (gap often overnight after entry). Also count announce on entry day.
|
||||||
|
in_hold = [
|
||||||
|
a for a in announces if entry_d <= a <= exit_d
|
||||||
|
]
|
||||||
|
# Entries within 3 trading days BEFORE an announcement:
|
||||||
|
# exists announce such that entry is in the 3 sessions immediately before announce.
|
||||||
|
pre_earn = False
|
||||||
|
for a in announces:
|
||||||
|
# trading sessions in (a-lookback, a)
|
||||||
|
sessions_before = [d for d in ordered_cal if d < a]
|
||||||
|
last3 = sessions_before[-3:] if len(sessions_before) >= 3 else sessions_before
|
||||||
|
if entry_d in last3:
|
||||||
|
pre_earn = True
|
||||||
|
break
|
||||||
|
trades_parsed.append({
|
||||||
|
"symbol": sym,
|
||||||
|
"entry": entry_d.isoformat(),
|
||||||
|
"exit": exit_d.isoformat(),
|
||||||
|
"r": float(r),
|
||||||
|
"earnings_in_hold": len(in_hold) > 0,
|
||||||
|
"n_earnings_in_hold": len(in_hold),
|
||||||
|
"entry_within_3d_before_earn": pre_earn,
|
||||||
|
})
|
||||||
|
|
||||||
|
all_r = [t["r"] for t in trades_parsed]
|
||||||
|
loss_lt_1r = [t for t in trades_parsed if t["r"] < -1.0]
|
||||||
|
loss_with_earn = [t for t in loss_lt_1r if t["earnings_in_hold"]]
|
||||||
|
pre = [t["r"] for t in trades_parsed if t["entry_within_3d_before_earn"]]
|
||||||
|
other = [t["r"] for t in trades_parsed if not t["entry_within_3d_before_earn"]]
|
||||||
|
|
||||||
|
return {
|
||||||
|
"sim_summary": {
|
||||||
|
k: sim.get(k)
|
||||||
|
for k in (
|
||||||
|
"sharpe",
|
||||||
|
"sharpe_se",
|
||||||
|
"cagr_pct",
|
||||||
|
"max_drawdown_pct",
|
||||||
|
"trades",
|
||||||
|
"total_return_pct",
|
||||||
|
)
|
||||||
|
},
|
||||||
|
"n_trades_parsed": len(trades_parsed),
|
||||||
|
"q1_losses_worse_than_minus_1r": {
|
||||||
|
"n_losses_lt_minus_1r": len(loss_lt_1r),
|
||||||
|
"n_with_earnings_in_hold": len(loss_with_earn),
|
||||||
|
"fraction_with_earnings": (
|
||||||
|
round(len(loss_with_earn) / len(loss_lt_1r), 4) if loss_lt_1r else None
|
||||||
|
),
|
||||||
|
"all_trades_with_earnings_in_hold": sum(
|
||||||
|
1 for t in trades_parsed if t["earnings_in_hold"]
|
||||||
|
),
|
||||||
|
"fraction_all_trades_with_earnings": (
|
||||||
|
round(
|
||||||
|
sum(1 for t in trades_parsed if t["earnings_in_hold"])
|
||||||
|
/ len(trades_parsed),
|
||||||
|
4,
|
||||||
|
)
|
||||||
|
if trades_parsed
|
||||||
|
else None
|
||||||
|
),
|
||||||
|
},
|
||||||
|
"q2_entry_within_3d_before_announce": {
|
||||||
|
"pre_earn_entries": _r_dist(pre),
|
||||||
|
"other_entries": _r_dist(other),
|
||||||
|
"all_entries": _r_dist(all_r),
|
||||||
|
"tail_trim_note": (
|
||||||
|
"Compare p95/max and mean of pre_earn vs other. "
|
||||||
|
"Rising win_rate with falling mean/p95 = right-tail trim red flag."
|
||||||
|
),
|
||||||
|
},
|
||||||
|
"note": "REPORT-ONLY — no filter shipped.",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_2b_ic(
|
||||||
|
snapshot: Path,
|
||||||
|
events: list[dict],
|
||||||
|
*,
|
||||||
|
quiet: bool,
|
||||||
|
workers: int,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""SUE IC via harness on identical cross-sections as momentum baselines."""
|
||||||
|
from app.config import settings
|
||||||
|
from app.services import backtest_service as bt
|
||||||
|
from app.services.benchmark_service import load_benchmark_closes
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from sqlalchemy import select
|
||||||
|
from collections import defaultdict as dd
|
||||||
|
|
||||||
|
os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1"
|
||||||
|
os.environ["BACKTEST_SIGNAL_EVAL_ONLY"] = "1"
|
||||||
|
# Load sector map if present so sector signals also appear (side-by-side optional).
|
||||||
|
if Path("data/research/ticker_sector_map.json").exists():
|
||||||
|
os.environ["BACKTEST_SECTOR_MAP_PATH"] = str(
|
||||||
|
Path("data/research/ticker_sector_map.json").resolve()
|
||||||
|
)
|
||||||
|
settings.backtest_workers = workers
|
||||||
|
|
||||||
|
engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True)
|
||||||
|
Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
|
||||||
|
# Collect base signals + attach SUE.
|
||||||
|
collected: dict = dd(lambda: dd(list))
|
||||||
|
try:
|
||||||
|
async with Session() as db:
|
||||||
|
tickers = list(
|
||||||
|
(await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars()
|
||||||
|
)
|
||||||
|
spy = await load_benchmark_closes(db, "SPY")
|
||||||
|
sector_etf: dict[str, dict] = {}
|
||||||
|
try:
|
||||||
|
from app.services.sector_map import SECTOR_ETFS, load_ticker_sector_map
|
||||||
|
|
||||||
|
symbol_to_sector = load_ticker_sector_map()
|
||||||
|
for etf in SECTOR_ETFS:
|
||||||
|
series = await load_benchmark_closes(db, etf)
|
||||||
|
if series:
|
||||||
|
sector_etf[etf] = series
|
||||||
|
except Exception:
|
||||||
|
symbol_to_sector = {}
|
||||||
|
sector_etf = {}
|
||||||
|
|
||||||
|
prices: dict[str, tuple] = {}
|
||||||
|
for idx, t in enumerate(tickers):
|
||||||
|
if not quiet and idx % 50 == 0:
|
||||||
|
print(f" 2b fetch {idx}/{len(tickers)}", end="\r", flush=True)
|
||||||
|
cols = await bt._fetch_columns(db, t.symbol)
|
||||||
|
if cols is None:
|
||||||
|
continue
|
||||||
|
prices[t.symbol] = cols
|
||||||
|
series = bt._signal_series(
|
||||||
|
[
|
||||||
|
type(
|
||||||
|
"R",
|
||||||
|
(),
|
||||||
|
{
|
||||||
|
"date": date.fromordinal(int(cols[0][i])),
|
||||||
|
"close": cols[4][i],
|
||||||
|
"high": cols[2][i],
|
||||||
|
"volume": cols[5][i] if len(cols) > 5 else 0,
|
||||||
|
},
|
||||||
|
)()
|
||||||
|
for i in range(len(cols[0]))
|
||||||
|
],
|
||||||
|
spy,
|
||||||
|
symbol=t.symbol,
|
||||||
|
sector_etf_closes=bt._sector_etf_closes_for_symbol(
|
||||||
|
t.symbol, symbol_to_sector, sector_etf
|
||||||
|
),
|
||||||
|
)
|
||||||
|
for name, weeks in series.items():
|
||||||
|
for wk, pairs in weeks.items():
|
||||||
|
collected[name][wk].extend(pairs)
|
||||||
|
finally:
|
||||||
|
await engine.dispose()
|
||||||
|
if not quiet:
|
||||||
|
print()
|
||||||
|
|
||||||
|
if symbol_to_sector:
|
||||||
|
bt._inject_sector_demeaned_momentum(collected, symbol_to_sector)
|
||||||
|
|
||||||
|
# SUE series.
|
||||||
|
events_by_sym: dict[str, list[dict]] = defaultdict(list)
|
||||||
|
for ev in events:
|
||||||
|
events_by_sym[ev["symbol"]].append(ev)
|
||||||
|
sue_map = _build_sue_series(events_by_sym, prices)
|
||||||
|
|
||||||
|
# Inject sue_latest into collected using mom_12_1 observations as the
|
||||||
|
# weekly as-of skeleton (same weeks / symbols).
|
||||||
|
sue_collected: dict = dd(list)
|
||||||
|
mom_weeks = collected.get("mom_12_1") or {}
|
||||||
|
for week_key, recs in mom_weeks.items():
|
||||||
|
for rec in recs:
|
||||||
|
pair = bt._obs_val_fwd(rec)
|
||||||
|
if pair is None:
|
||||||
|
continue
|
||||||
|
_val, fwd = pair
|
||||||
|
sym = None
|
||||||
|
if isinstance(rec, dict):
|
||||||
|
sym = rec.get("symbol")
|
||||||
|
if not sym:
|
||||||
|
continue
|
||||||
|
# Need as-of date: recover from week — use Friday of ISO week as proxy
|
||||||
|
# is weak. Better: re-derive from prices weekly indices.
|
||||||
|
# Store asof on rich recs? Current rich rows lack asof date.
|
||||||
|
# Fall back: compute SUE observations directly from prices weekly as-ofs.
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Direct weekly as-of SUE + forward return (authoritative).
|
||||||
|
for sym, cols in prices.items():
|
||||||
|
ords, _o, highs, _l, closes, _v = cols
|
||||||
|
dates = [date.fromordinal(int(o)) for o in ords]
|
||||||
|
sue_days = sue_map.get(sym.upper()) or {}
|
||||||
|
if not sue_days:
|
||||||
|
continue
|
||||||
|
n = len(dates)
|
||||||
|
# weekly as-of indices: reuse harness helper via fake records.
|
||||||
|
records = [
|
||||||
|
type("R", (), {"date": dates[i], "close": closes[i], "high": highs[i]})()
|
||||||
|
for i in range(n)
|
||||||
|
]
|
||||||
|
for i in bt._weekly_asof_indices(records):
|
||||||
|
j = i + bt.HORIZON
|
||||||
|
if j >= n or closes[i] <= 0:
|
||||||
|
continue
|
||||||
|
asof = dates[i]
|
||||||
|
sue = sue_days.get(asof)
|
||||||
|
if sue is None:
|
||||||
|
continue
|
||||||
|
fwd = float(closes[j]) / float(closes[i]) - 1.0
|
||||||
|
iso = asof.isocalendar()
|
||||||
|
week_key = (iso[0], iso[1])
|
||||||
|
# Also grab mom for conditional.
|
||||||
|
mom = None
|
||||||
|
if i >= 252 and closes[i - 252] > 0:
|
||||||
|
mom = float(closes[i - 21]) / float(closes[i - 252]) - 1.0
|
||||||
|
sue_collected[week_key].append({
|
||||||
|
"val": float(sue),
|
||||||
|
"fwd": fwd,
|
||||||
|
"symbol": sym,
|
||||||
|
"mom_12_1": mom,
|
||||||
|
})
|
||||||
|
collected["sue_latest"] = sue_collected
|
||||||
|
|
||||||
|
signal_eval = bt._signal_evaluation(collected)
|
||||||
|
|
||||||
|
# Fair side-by-side: re-evaluate mom baselines on the *same* (symbol, week)
|
||||||
|
# observations where SUE is present (incomplete backfill otherwise inflates
|
||||||
|
# mom N relative to SUE).
|
||||||
|
sue_pairs_by_week = sue_collected
|
||||||
|
restricted: dict = dd(lambda: dd(list))
|
||||||
|
for week_key, recs in sue_pairs_by_week.items():
|
||||||
|
syms = {str(r.get("symbol")).upper() for r in recs if r.get("symbol")}
|
||||||
|
for base_name in ("mom_12_1", "mom_12_1_resid"):
|
||||||
|
base_recs = (collected.get(base_name) or {}).get(week_key) or []
|
||||||
|
for rec in base_recs:
|
||||||
|
pair = bt._obs_val_fwd(rec)
|
||||||
|
if pair is None:
|
||||||
|
continue
|
||||||
|
sym = None
|
||||||
|
if isinstance(rec, dict):
|
||||||
|
sym = rec.get("symbol")
|
||||||
|
if not sym or str(sym).upper() not in syms:
|
||||||
|
continue
|
||||||
|
restricted[base_name][week_key].append(rec)
|
||||||
|
restricted["sue_latest"][week_key].extend(recs)
|
||||||
|
restricted_eval = bt._signal_evaluation(restricted)
|
||||||
|
|
||||||
|
# Momentum-conditional: IC of SUE within top mom quintile each week.
|
||||||
|
cond_ics: list[float] = []
|
||||||
|
stride = max(1, round(bt.HORIZON / 5))
|
||||||
|
usable = [wk for wk, recs in sue_collected.items() if len(recs) >= bt.MIN_CROSS_SECTION]
|
||||||
|
kept = bt._nonoverlapping_weeks(usable, stride)
|
||||||
|
for wk in kept:
|
||||||
|
recs = sue_collected[wk]
|
||||||
|
with_mom = [r for r in recs if r.get("mom_12_1") is not None]
|
||||||
|
if len(with_mom) < bt.MIN_CROSS_SECTION:
|
||||||
|
continue
|
||||||
|
ordered = sorted(with_mom, key=lambda r: float(r["mom_12_1"]))
|
||||||
|
k = max(1, len(ordered) // 5)
|
||||||
|
top = ordered[-k:]
|
||||||
|
if len(top) < 5:
|
||||||
|
continue
|
||||||
|
ic = bt._spearman(
|
||||||
|
[float(r["val"]) for r in top],
|
||||||
|
[float(r["fwd"]) for r in top],
|
||||||
|
)
|
||||||
|
if ic is not None:
|
||||||
|
cond_ics.append(ic)
|
||||||
|
if cond_ics:
|
||||||
|
mean_c = sum(cond_ics) / len(cond_ics)
|
||||||
|
if len(cond_ics) > 1:
|
||||||
|
std = math.sqrt(
|
||||||
|
sum((x - mean_c) ** 2 for x in cond_ics) / (len(cond_ics) - 1)
|
||||||
|
)
|
||||||
|
t_c = mean_c / std * math.sqrt(len(cond_ics)) if std > 0 else None
|
||||||
|
else:
|
||||||
|
t_c = None
|
||||||
|
mom_cond = {
|
||||||
|
"mean_ic": round(mean_c, 4),
|
||||||
|
"ic_t_stat": round(t_c, 2) if t_c is not None else None,
|
||||||
|
"weeks": len(cond_ics),
|
||||||
|
"note": "IC of sue_latest within top mom_12_1 quintile (non-overlapping weeks)",
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
mom_cond = {"mean_ic": None, "weeks": 0}
|
||||||
|
|
||||||
|
def _find(name: str) -> dict | None:
|
||||||
|
for row in signal_eval:
|
||||||
|
if row.get("signal") == name:
|
||||||
|
return row
|
||||||
|
return None
|
||||||
|
|
||||||
|
sue = _find("sue_latest")
|
||||||
|
grade = {
|
||||||
|
"green": False,
|
||||||
|
"reason": "sue_latest missing",
|
||||||
|
}
|
||||||
|
if sue:
|
||||||
|
mean_ic = sue.get("mean_ic")
|
||||||
|
t = sue.get("ic_t_stat")
|
||||||
|
reliable = bool(sue.get("reliable"))
|
||||||
|
sign_ok = mean_ic is not None and float(mean_ic) > 0
|
||||||
|
mag_ok = mean_ic is not None and abs(float(mean_ic)) >= IRON_IC_BAR
|
||||||
|
grade = {
|
||||||
|
"green": bool(sign_ok and mag_ok and reliable),
|
||||||
|
"checks": {
|
||||||
|
"mean_ic": mean_ic,
|
||||||
|
"sign_positive": sign_ok,
|
||||||
|
"abs_ge_0_03": mag_ok,
|
||||||
|
"reliable": reliable,
|
||||||
|
"ic_t_stat": t,
|
||||||
|
"weeks": sue.get("weeks"),
|
||||||
|
},
|
||||||
|
"reason": (
|
||||||
|
"iron rule cleared — STOP; book-integration is a separate human step"
|
||||||
|
if (sign_ok and mag_ok and reliable)
|
||||||
|
else "iron rule not met"
|
||||||
|
),
|
||||||
|
"row": sue,
|
||||||
|
}
|
||||||
|
|
||||||
|
def _find_r(name: str) -> dict | None:
|
||||||
|
for row in restricted_eval:
|
||||||
|
if row.get("signal") == name:
|
||||||
|
return row
|
||||||
|
return None
|
||||||
|
|
||||||
|
# Side-by-side baselines from same evaluation.
|
||||||
|
side = {
|
||||||
|
name: _find(name)
|
||||||
|
for name in (
|
||||||
|
"mom_12_1",
|
||||||
|
"mom_12_1_resid",
|
||||||
|
"mom_12_1_sector_resid",
|
||||||
|
"mom_12_1_sector_demeaned",
|
||||||
|
"sue_latest",
|
||||||
|
"fip_id",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
side_restricted = {
|
||||||
|
name: _find_r(name)
|
||||||
|
for name in ("mom_12_1", "mom_12_1_resid", "sue_latest")
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"signal_eval_side_by_side": side,
|
||||||
|
"signal_eval_identical_sue_subset": side_restricted,
|
||||||
|
"identical_subset_note": (
|
||||||
|
"Mom baselines re-scored only on (week, symbol) cells where SUE exists. "
|
||||||
|
"Use this table when backfill is incomplete — full-universe mom N is not comparable."
|
||||||
|
),
|
||||||
|
"full_signal_eval": signal_eval,
|
||||||
|
"sue_grade": grade,
|
||||||
|
"momentum_conditional_sue": mom_cond,
|
||||||
|
"sue_coverage": {
|
||||||
|
"symbols_with_sue": len(sue_map),
|
||||||
|
"avg_weeks_with_sue": (
|
||||||
|
round(
|
||||||
|
sum(len(v) for v in sue_collected.values())
|
||||||
|
/ max(1, len(sue_collected)),
|
||||||
|
1,
|
||||||
|
)
|
||||||
|
if sue_collected
|
||||||
|
else 0
|
||||||
|
),
|
||||||
|
"weeks_with_min_cross_section": len(usable),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _write_md(path: Path, payload: dict) -> None:
|
||||||
|
pre = path.read_text(encoding="utf-8") if path.exists() else ""
|
||||||
|
marker = "## Results"
|
||||||
|
idx = pre.find(marker)
|
||||||
|
header = pre[:idx] if idx >= 0 else pre.split("## Verdict")[0]
|
||||||
|
|
||||||
|
lines = [
|
||||||
|
header.rstrip(),
|
||||||
|
"",
|
||||||
|
"## Results",
|
||||||
|
"",
|
||||||
|
f"Generated: `{payload.get('generated_at')}`",
|
||||||
|
"",
|
||||||
|
"### Data provenance",
|
||||||
|
"",
|
||||||
|
f"```json\n{json.dumps(payload.get('data_provenance') or {}, indent=2, default=str)}\n```",
|
||||||
|
"",
|
||||||
|
"### 2a — Earnings-gap risk (report-only)",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
a = payload.get("experiment_2a")
|
||||||
|
if not a:
|
||||||
|
lines.append("_Skipped or unavailable._")
|
||||||
|
else:
|
||||||
|
lines.append(f"```json\n{json.dumps(a, indent=2, default=str)}\n```")
|
||||||
|
lines.extend(["", "### 2b — SUE / PEAD IC", ""])
|
||||||
|
b = payload.get("experiment_2b")
|
||||||
|
if not b:
|
||||||
|
lines.append("_Skipped or unavailable._")
|
||||||
|
else:
|
||||||
|
side = b.get("signal_eval_side_by_side") or {}
|
||||||
|
lines.extend([
|
||||||
|
"| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable |",
|
||||||
|
"|---|---:|---:|---:|---:|---|",
|
||||||
|
])
|
||||||
|
for name in (
|
||||||
|
"mom_12_1",
|
||||||
|
"mom_12_1_resid",
|
||||||
|
"sue_latest",
|
||||||
|
"mom_12_1_sector_resid",
|
||||||
|
"fip_id",
|
||||||
|
):
|
||||||
|
r = side.get(name) or {}
|
||||||
|
lines.append(
|
||||||
|
f"| {name} | {r.get('mean_ic', '')} | {r.get('ic_t_stat', '')} | "
|
||||||
|
f"{r.get('weeks', '')} | {r.get('avg_cross_section', '')} | "
|
||||||
|
f"{r.get('reliable', '')} |"
|
||||||
|
)
|
||||||
|
lines.extend([
|
||||||
|
"",
|
||||||
|
f"**SUE grade:** `{json.dumps(b.get('sue_grade') or {}, default=str)}`",
|
||||||
|
"",
|
||||||
|
f"**Momentum-conditional SUE:** `{json.dumps(b.get('momentum_conditional_sue') or {}, default=str)}`",
|
||||||
|
"",
|
||||||
|
])
|
||||||
|
|
||||||
|
lines.extend([
|
||||||
|
"",
|
||||||
|
"## Verdict",
|
||||||
|
"",
|
||||||
|
f"**{payload.get('verdict')}**",
|
||||||
|
"",
|
||||||
|
payload.get("verdict_detail") or "",
|
||||||
|
"",
|
||||||
|
"## What a human must decide next",
|
||||||
|
"",
|
||||||
|
payload.get("human_next") or "- Review; no auto-ship.",
|
||||||
|
"",
|
||||||
|
f"Artifacts: `{payload.get('report_path')}`",
|
||||||
|
"",
|
||||||
|
])
|
||||||
|
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
async def _main() -> None:
|
||||||
|
args = _parse_args()
|
||||||
|
snapshot = Path(args.snapshot)
|
||||||
|
if not snapshot.exists():
|
||||||
|
raise SystemExit(f"Missing snapshot {snapshot}")
|
||||||
|
if args.allow_spawn:
|
||||||
|
os.environ["BACKTEST_ALLOW_SPAWN"] = "1"
|
||||||
|
|
||||||
|
events, meta = _load_earnings(snapshot)
|
||||||
|
# Race guard lite on earnings completeness.
|
||||||
|
provenance = {
|
||||||
|
"snapshot": str(snapshot.resolve()),
|
||||||
|
"n_earnings_events": len(events),
|
||||||
|
"backfill_meta": meta,
|
||||||
|
"announce_range": {
|
||||||
|
"min": min((e["announce_date"] for e in events), default=None),
|
||||||
|
"max": max((e["announce_date"] for e in events), default=None),
|
||||||
|
},
|
||||||
|
"with_actual_and_estimate": sum(
|
||||||
|
1
|
||||||
|
for e in events
|
||||||
|
if e.get("eps_actual") is not None and e.get("eps_estimate") is not None
|
||||||
|
),
|
||||||
|
}
|
||||||
|
print(
|
||||||
|
f"Earnings events: {provenance['n_earnings_events']} "
|
||||||
|
f"(with act+est={provenance['with_actual_and_estimate']}) meta={meta}"
|
||||||
|
)
|
||||||
|
if meta and meta.get("done", 0) < 0.9 * (meta.get("universe_tickers") or 1):
|
||||||
|
print(
|
||||||
|
"WARNING: earnings backfill incomplete "
|
||||||
|
f"({meta.get('done')}/{meta.get('universe_tickers')}). "
|
||||||
|
"Results may be biased; resume backfill."
|
||||||
|
)
|
||||||
|
|
||||||
|
exp_2a = None
|
||||||
|
exp_2b = None
|
||||||
|
if not args.skip_2a:
|
||||||
|
print("Running 2a earnings-gap diagnostic…")
|
||||||
|
exp_2a = await _run_2a(
|
||||||
|
snapshot, events, quiet=args.quiet, workers=args.workers
|
||||||
|
)
|
||||||
|
print(
|
||||||
|
" 2a losses<-1R with earnings:",
|
||||||
|
(exp_2a.get("q1_losses_worse_than_minus_1r") or {}),
|
||||||
|
)
|
||||||
|
if not args.skip_2b:
|
||||||
|
print("Running 2b SUE IC harness…")
|
||||||
|
exp_2b = await _run_2b_ic(
|
||||||
|
snapshot, events, quiet=args.quiet, workers=args.workers
|
||||||
|
)
|
||||||
|
g = exp_2b.get("sue_grade") or {}
|
||||||
|
print(f" 2b SUE green={g.get('green')} {g.get('reason')}")
|
||||||
|
|
||||||
|
# Verdict
|
||||||
|
if exp_2b and (exp_2b.get("sue_grade") or {}).get("green"):
|
||||||
|
verdict = "PROMOTE (2b SUE) — STOP for human wire design"
|
||||||
|
detail = (
|
||||||
|
"SUE cleared iron rule. No book integration without human approval. "
|
||||||
|
"2a remains report-only."
|
||||||
|
)
|
||||||
|
human = (
|
||||||
|
"- Design tilt vs second gate if desired.\n"
|
||||||
|
"- Do not auto-filter from 2a without separate approval + tail review."
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
sue_ic = None
|
||||||
|
if exp_2b:
|
||||||
|
sue_ic = ((exp_2b.get("sue_grade") or {}).get("row") or {}).get("mean_ic")
|
||||||
|
if sue_ic is not None and abs(float(sue_ic)) >= 0.015:
|
||||||
|
verdict = "PARK"
|
||||||
|
detail = f"SUE IC={sue_ic} below iron bar or unreliable; keep data, no wire."
|
||||||
|
else:
|
||||||
|
verdict = "DEAD (2b) / REPORT-ONLY (2a)"
|
||||||
|
detail = (
|
||||||
|
"SUE does not clear iron rule on this window. "
|
||||||
|
"2a distributions for human risk review only — no filter."
|
||||||
|
)
|
||||||
|
human = (
|
||||||
|
"- No SUE book change.\n"
|
||||||
|
"- Read 2a tails before considering any earnings-avoid filter."
|
||||||
|
)
|
||||||
|
|
||||||
|
stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
|
||||||
|
out = Path(args.out) if args.out else Path("reports") / f"earnings-gap-sue-{stamp}.json"
|
||||||
|
payload = {
|
||||||
|
"generated_at": datetime.now().isoformat(),
|
||||||
|
"data_provenance": provenance,
|
||||||
|
"experiment_2a": exp_2a,
|
||||||
|
"experiment_2b": exp_2b,
|
||||||
|
"verdict": verdict,
|
||||||
|
"verdict_detail": detail,
|
||||||
|
"human_next": human,
|
||||||
|
"report_path": str(out.as_posix()),
|
||||||
|
"fmp_note": (
|
||||||
|
"Bulk earnings-calendar is paid (402 on free tier). "
|
||||||
|
"Backfill used per-symbol /stable/earnings; see earnings-backfill-status.json."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
out.write_text(json.dumps(payload, indent=2, default=str) + "\n", encoding="utf-8")
|
||||||
|
md = Path("docs/research/earnings-gap-and-sue.md")
|
||||||
|
_write_md(md, payload)
|
||||||
|
out.with_suffix(".md").write_text(md.read_text(encoding="utf-8"), encoding="utf-8")
|
||||||
|
print(f"Verdict: {verdict}")
|
||||||
|
print(f"Wrote {out}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(_main())
|
||||||
@@ -0,0 +1,476 @@
|
|||||||
|
"""History-depth extension research (local / MacBook).
|
||||||
|
|
||||||
|
Phases
|
||||||
|
------
|
||||||
|
coverage — bars per calendar year; no rebuild
|
||||||
|
harness — race-guard snapshot, full signal_eval, era split pre/post-2021
|
||||||
|
|
||||||
|
Does not retune production knobs. Does not modify scheduler/gates.
|
||||||
|
|
||||||
|
Example
|
||||||
|
-------
|
||||||
|
python scripts/run_history_depth_research.py --phase coverage \\
|
||||||
|
--snapshot backtest_snapshots/prod.sqlite
|
||||||
|
|
||||||
|
python scripts/run_history_depth_research.py --phase harness \\
|
||||||
|
--snapshot backtest_snapshots/research.sqlite --workers 8 --allow-spawn
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
from collections import defaultdict
|
||||||
|
from datetime import date, datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from sqlalchemy import create_engine, text
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
if str(ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(ROOT))
|
||||||
|
|
||||||
|
ERA_SPLIT = date(2021, 1, 1)
|
||||||
|
SURVIVORSHIP_BANNER = (
|
||||||
|
"SURVIVORSHIP BIAS: today's constituents backfilled historically. "
|
||||||
|
"Absolute Sharpe/CAGR levels on deep history are optimistic. "
|
||||||
|
"Use RELATIVE signal IC comparisons and era stability only — not levels."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _sqlite_url(path: Path) -> str:
|
||||||
|
return f"sqlite+aiosqlite:///{path.resolve().as_posix()}"
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_args() -> argparse.Namespace:
|
||||||
|
p = argparse.ArgumentParser(description=__doc__)
|
||||||
|
p.add_argument("--phase", choices=("coverage", "harness", "all"), default="all")
|
||||||
|
p.add_argument("--snapshot", default="backtest_snapshots/research.sqlite")
|
||||||
|
p.add_argument("--workers", type=int, default=8)
|
||||||
|
p.add_argument("--allow-spawn", action="store_true")
|
||||||
|
p.add_argument("--quiet", action="store_true")
|
||||||
|
p.add_argument("--out", default=None)
|
||||||
|
return p.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def _coverage_report(snapshot: Path) -> dict[str, Any]:
|
||||||
|
engine = create_engine(
|
||||||
|
f"sqlite:///{snapshot.resolve().as_posix()}",
|
||||||
|
future=True,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with engine.connect() as conn:
|
||||||
|
ticker_n = int(conn.execute(text("SELECT COUNT(*) FROM tickers")).scalar_one())
|
||||||
|
ohlcv_n = int(
|
||||||
|
conn.execute(text("SELECT COUNT(*) FROM ohlcv_records")).scalar_one()
|
||||||
|
)
|
||||||
|
d_range = conn.execute(
|
||||||
|
text("SELECT MIN(date), MAX(date) FROM ohlcv_records")
|
||||||
|
).fetchone()
|
||||||
|
# Bars per calendar year (global).
|
||||||
|
by_year = conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
SELECT substr(date, 1, 4) AS y, COUNT(*) AS n,
|
||||||
|
COUNT(DISTINCT ticker_id) AS tickers
|
||||||
|
FROM ohlcv_records
|
||||||
|
GROUP BY substr(date, 1, 4)
|
||||||
|
ORDER BY y
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
).fetchall()
|
||||||
|
# Per-symbol min/max date + bar count (summary percentiles).
|
||||||
|
per_sym = conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
SELECT t.symbol, COUNT(*) AS n, MIN(o.date), MAX(o.date)
|
||||||
|
FROM ohlcv_records o
|
||||||
|
JOIN tickers t ON t.id = o.ticker_id
|
||||||
|
GROUP BY t.symbol
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
).fetchall()
|
||||||
|
finally:
|
||||||
|
engine.dispose()
|
||||||
|
|
||||||
|
ns = sorted(int(r[1]) for r in per_sym)
|
||||||
|
def pct(p: float) -> int | None:
|
||||||
|
if not ns:
|
||||||
|
return None
|
||||||
|
i = int(round(p * (len(ns) - 1)))
|
||||||
|
return ns[i]
|
||||||
|
|
||||||
|
starts = sorted(str(r[2]) for r in per_sym if r[2])
|
||||||
|
start_hist: dict[str, int] = defaultdict(int)
|
||||||
|
for s in starts:
|
||||||
|
start_hist[s[:4]] += 1
|
||||||
|
|
||||||
|
return {
|
||||||
|
"snapshot": str(snapshot.resolve()),
|
||||||
|
"ticker_count": ticker_n,
|
||||||
|
"ohlcv_row_count": ohlcv_n,
|
||||||
|
"date_range": {"min": d_range[0], "max": d_range[1]},
|
||||||
|
"bars_per_year": [
|
||||||
|
{"year": y, "bars": n, "tickers_with_bars": t} for y, n, t in by_year
|
||||||
|
],
|
||||||
|
"bars_per_symbol": {
|
||||||
|
"min": ns[0] if ns else None,
|
||||||
|
"p10": pct(0.10),
|
||||||
|
"p50": pct(0.50),
|
||||||
|
"p90": pct(0.90),
|
||||||
|
"max": ns[-1] if ns else None,
|
||||||
|
},
|
||||||
|
"symbols_by_start_year": dict(sorted(start_hist.items())),
|
||||||
|
"note": (
|
||||||
|
"Where ticker counts drop in early years, the feed (or listing history) "
|
||||||
|
"thins — do not treat those years as a full 505-name cross-section."
|
||||||
|
),
|
||||||
|
"survivorship_banner": SURVIVORSHIP_BANNER,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _assert_complete(snapshot: Path) -> dict[str, Any]:
|
||||||
|
from scripts.research_snapshot_manifest import ( # type: ignore
|
||||||
|
assert_research_snapshot_complete,
|
||||||
|
load_manifest,
|
||||||
|
)
|
||||||
|
|
||||||
|
m = load_manifest(snapshot)
|
||||||
|
if m is None:
|
||||||
|
# Prod snapshot may lack manifest; still require healthy bar depth.
|
||||||
|
eng = create_engine(
|
||||||
|
f"sqlite:///{snapshot.resolve().as_posix()}",
|
||||||
|
future=True,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
with eng.connect() as conn:
|
||||||
|
avg = conn.execute(
|
||||||
|
text(
|
||||||
|
"""
|
||||||
|
SELECT AVG(c) FROM (
|
||||||
|
SELECT COUNT(*) AS c FROM ohlcv_records GROUP BY ticker_id
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
).scalar_one()
|
||||||
|
finally:
|
||||||
|
eng.dispose()
|
||||||
|
if avg is None or float(avg) < 400:
|
||||||
|
raise SystemExit(
|
||||||
|
f"No completion manifest and avg bars={avg} look short. "
|
||||||
|
"Rebuild research.sqlite via extend_snapshot_universe.py"
|
||||||
|
)
|
||||||
|
return {"manifest": None, "avg_bars": float(avg), "ok": True}
|
||||||
|
return {"manifest": assert_research_snapshot_complete(snapshot), "ok": True}
|
||||||
|
|
||||||
|
|
||||||
|
async def _harness(snapshot: Path, *, workers: int, quiet: bool) -> dict[str, Any]:
|
||||||
|
from app.config import settings
|
||||||
|
from app.services.backtest_service import run_backtest
|
||||||
|
|
||||||
|
os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1"
|
||||||
|
os.environ["BACKTEST_SIGNAL_EVAL_ONLY"] = "1"
|
||||||
|
if Path("data/research/ticker_sector_map.json").exists():
|
||||||
|
os.environ["BACKTEST_SECTOR_MAP_PATH"] = str(
|
||||||
|
Path("data/research/ticker_sector_map.json").resolve()
|
||||||
|
)
|
||||||
|
settings.backtest_workers = workers
|
||||||
|
|
||||||
|
engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True)
|
||||||
|
Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
|
||||||
|
def progress(done: int, total: int, symbol: str) -> None:
|
||||||
|
if quiet:
|
||||||
|
return
|
||||||
|
print(f" progress {done}/{total} {symbol}", end="\r", flush=True)
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with Session() as db:
|
||||||
|
report = await run_backtest(db, progress_cb=progress, cadence="weekly")
|
||||||
|
finally:
|
||||||
|
await engine.dispose()
|
||||||
|
if not quiet:
|
||||||
|
print()
|
||||||
|
|
||||||
|
signal_eval = report.get("signal_eval") or []
|
||||||
|
|
||||||
|
# Era-split IC: recompute from collected is not available post-run.
|
||||||
|
# Approximate via second pass is expensive; instead document that era split
|
||||||
|
# requires collecting weekly ICs. We re-run evaluation if the report embeds
|
||||||
|
# nothing — for v1, call internal collection is too heavy to duplicate.
|
||||||
|
# Lightweight approach: mark era_split as requiring BACKTEST with custom
|
||||||
|
# filter — implemented below by re-scoring from a dedicated collection pass.
|
||||||
|
era = await _era_split_ics(snapshot, workers=workers, quiet=quiet)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"survivorship_banner": SURVIVORSHIP_BANNER,
|
||||||
|
"signal_eval": signal_eval,
|
||||||
|
"era_split": era,
|
||||||
|
"params": report.get("params"),
|
||||||
|
"tickers": report.get("tickers"),
|
||||||
|
"generated_at_run": report.get("generated_at"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def _era_split_ics(
|
||||||
|
snapshot: Path, *, workers: int, quiet: bool
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Collect weekly signal series and evaluate pre/post ERA_SPLIT separately."""
|
||||||
|
from app.config import settings
|
||||||
|
from app.services import backtest_service as bt
|
||||||
|
from app.services.benchmark_service import load_benchmark_closes
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from sqlalchemy import select
|
||||||
|
from collections import defaultdict as dd
|
||||||
|
|
||||||
|
os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1"
|
||||||
|
settings.backtest_workers = max(1, workers)
|
||||||
|
|
||||||
|
engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True)
|
||||||
|
Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
|
||||||
|
collected: dict = dd(lambda: dd(list))
|
||||||
|
try:
|
||||||
|
async with Session() as db:
|
||||||
|
tickers = list(
|
||||||
|
(await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars()
|
||||||
|
)
|
||||||
|
spy = await load_benchmark_closes(db, "SPY")
|
||||||
|
symbol_to_sector = {}
|
||||||
|
sector_etf: dict = {}
|
||||||
|
try:
|
||||||
|
from app.services.sector_map import (
|
||||||
|
SECTOR_ETFS,
|
||||||
|
load_ticker_sector_map,
|
||||||
|
)
|
||||||
|
|
||||||
|
symbol_to_sector = load_ticker_sector_map()
|
||||||
|
for etf in SECTOR_ETFS:
|
||||||
|
series = await load_benchmark_closes(db, etf)
|
||||||
|
if series:
|
||||||
|
sector_etf[etf] = series
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
for idx, t in enumerate(tickers):
|
||||||
|
if not quiet and idx % 100 == 0:
|
||||||
|
print(f" era-collect {idx}/{len(tickers)}", end="\r", flush=True)
|
||||||
|
cols = await bt._fetch_columns(db, t.symbol)
|
||||||
|
if cols is None:
|
||||||
|
continue
|
||||||
|
records = [
|
||||||
|
type(
|
||||||
|
"R",
|
||||||
|
(),
|
||||||
|
{
|
||||||
|
"date": date.fromordinal(int(cols[0][i])),
|
||||||
|
"close": cols[4][i],
|
||||||
|
"high": cols[2][i],
|
||||||
|
"volume": cols[5][i] if len(cols) > 5 else 0,
|
||||||
|
},
|
||||||
|
)()
|
||||||
|
for i in range(len(cols[0]))
|
||||||
|
]
|
||||||
|
series = bt._signal_series(
|
||||||
|
records,
|
||||||
|
spy,
|
||||||
|
symbol=t.symbol,
|
||||||
|
sector_etf_closes=bt._sector_etf_closes_for_symbol(
|
||||||
|
t.symbol, symbol_to_sector, sector_etf
|
||||||
|
),
|
||||||
|
)
|
||||||
|
for name, weeks in series.items():
|
||||||
|
for wk, pairs in weeks.items():
|
||||||
|
collected[name][wk].extend(pairs)
|
||||||
|
if symbol_to_sector:
|
||||||
|
bt._inject_sector_demeaned_momentum(collected, symbol_to_sector)
|
||||||
|
finally:
|
||||||
|
await engine.dispose()
|
||||||
|
if not quiet:
|
||||||
|
print()
|
||||||
|
|
||||||
|
def _filter_era(coll: dict, *, pre: bool) -> dict:
|
||||||
|
out: dict = dd(lambda: dd(list))
|
||||||
|
for name, weeks in coll.items():
|
||||||
|
for wk, recs in weeks.items():
|
||||||
|
# ISO week key (year, week) — approximate era by ISO year.
|
||||||
|
year = int(wk[0]) if isinstance(wk, tuple) else int(str(wk)[:4])
|
||||||
|
if pre and year >= ERA_SPLIT.year:
|
||||||
|
continue
|
||||||
|
if not pre and year < ERA_SPLIT.year:
|
||||||
|
continue
|
||||||
|
out[name][wk].extend(recs)
|
||||||
|
return out
|
||||||
|
|
||||||
|
pre_eval = bt._signal_evaluation(_filter_era(collected, pre=True))
|
||||||
|
post_eval = bt._signal_evaluation(_filter_era(collected, pre=False))
|
||||||
|
full_eval = bt._signal_evaluation(collected)
|
||||||
|
|
||||||
|
def _index(rows: list[dict]) -> dict[str, dict]:
|
||||||
|
return {r["signal"]: r for r in rows}
|
||||||
|
|
||||||
|
return {
|
||||||
|
"era_split_date": ERA_SPLIT.isoformat(),
|
||||||
|
"note": "Diagnostic only — not a tuning input. Nested lookbacks are not OOS.",
|
||||||
|
"full": _index(full_eval),
|
||||||
|
"pre_2021": _index(pre_eval),
|
||||||
|
"post_2021": _index(post_eval),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _write_md(path: Path, payload: dict) -> None:
|
||||||
|
pre = path.read_text(encoding="utf-8") if path.exists() else ""
|
||||||
|
marker = "## Results"
|
||||||
|
idx = pre.find(marker)
|
||||||
|
header = pre[:idx] if idx >= 0 else pre.split("## Verdict")[0]
|
||||||
|
|
||||||
|
lines = [
|
||||||
|
header.rstrip(),
|
||||||
|
"",
|
||||||
|
"## Results",
|
||||||
|
"",
|
||||||
|
f"Generated: `{payload.get('generated_at')}`",
|
||||||
|
"",
|
||||||
|
f"> **{SURVIVORSHIP_BANNER}**",
|
||||||
|
"",
|
||||||
|
"### Coverage",
|
||||||
|
"",
|
||||||
|
f"```json\n{json.dumps(payload.get('coverage') or {}, indent=2, default=str)}\n```",
|
||||||
|
"",
|
||||||
|
"### Race guard",
|
||||||
|
"",
|
||||||
|
f"```json\n{json.dumps(payload.get('race_guard') or {}, indent=2, default=str)}\n```",
|
||||||
|
"",
|
||||||
|
"### Signal IC (full extended window)",
|
||||||
|
"",
|
||||||
|
]
|
||||||
|
harness = payload.get("harness") or {}
|
||||||
|
rows = harness.get("signal_eval") or []
|
||||||
|
if rows:
|
||||||
|
lines.extend([
|
||||||
|
"| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable |",
|
||||||
|
"|---|---:|---:|---:|---:|---|",
|
||||||
|
])
|
||||||
|
for r in rows:
|
||||||
|
lines.append(
|
||||||
|
f"| {r.get('signal')} | {r.get('mean_ic')} | {r.get('ic_t_stat')} | "
|
||||||
|
f"{r.get('weeks')} | {r.get('avg_cross_section')} | {r.get('reliable')} |"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
lines.append("_Harness not run this pass._")
|
||||||
|
|
||||||
|
era = (harness.get("era_split") or {})
|
||||||
|
lines.extend(["", "### Era split (diagnostic only)", ""])
|
||||||
|
if era:
|
||||||
|
for label in ("full", "pre_2021", "post_2021"):
|
||||||
|
block = era.get(label) or {}
|
||||||
|
lines.append(f"#### {label}")
|
||||||
|
lines.append("")
|
||||||
|
lines.append("| signal | mean_ic | t | weeks | N |")
|
||||||
|
lines.append("|---|---:|---:|---:|---:|")
|
||||||
|
for name in sorted(block):
|
||||||
|
r = block[name]
|
||||||
|
lines.append(
|
||||||
|
f"| {name} | {r.get('mean_ic')} | {r.get('ic_t_stat')} | "
|
||||||
|
f"{r.get('weeks')} | {r.get('avg_cross_section')} |"
|
||||||
|
)
|
||||||
|
lines.append("")
|
||||||
|
else:
|
||||||
|
lines.append("_No era split._")
|
||||||
|
|
||||||
|
lines.extend([
|
||||||
|
"",
|
||||||
|
"## Verdict",
|
||||||
|
"",
|
||||||
|
f"**{payload.get('verdict')}**",
|
||||||
|
"",
|
||||||
|
payload.get("verdict_detail") or "",
|
||||||
|
"",
|
||||||
|
"## What a human must decide next",
|
||||||
|
"",
|
||||||
|
payload.get("human_next")
|
||||||
|
or "- Do not retune production knobs from this report without review.",
|
||||||
|
"",
|
||||||
|
f"Artifacts: `{payload.get('report_path')}`",
|
||||||
|
"",
|
||||||
|
])
|
||||||
|
path.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
async def _main() -> None:
|
||||||
|
args = _parse_args()
|
||||||
|
snapshot = Path(args.snapshot)
|
||||||
|
if not snapshot.exists():
|
||||||
|
raise SystemExit(f"Missing snapshot: {snapshot}")
|
||||||
|
if args.allow_spawn:
|
||||||
|
os.environ["BACKTEST_ALLOW_SPAWN"] = "1"
|
||||||
|
|
||||||
|
coverage = None
|
||||||
|
race = None
|
||||||
|
harness = None
|
||||||
|
|
||||||
|
if args.phase in ("coverage", "all"):
|
||||||
|
print("Coverage probe…")
|
||||||
|
coverage = _coverage_report(snapshot)
|
||||||
|
print(
|
||||||
|
f" tickers={coverage['ticker_count']} ohlcv={coverage['ohlcv_row_count']} "
|
||||||
|
f"range={coverage['date_range']}"
|
||||||
|
)
|
||||||
|
for row in coverage["bars_per_year"]:
|
||||||
|
print(
|
||||||
|
f" year {row['year']}: bars={row['bars']} "
|
||||||
|
f"tickers={row['tickers_with_bars']}"
|
||||||
|
)
|
||||||
|
|
||||||
|
if args.phase in ("harness", "all"):
|
||||||
|
print("Race guard…")
|
||||||
|
race = _assert_complete(snapshot)
|
||||||
|
print(f" ok={race.get('ok')}")
|
||||||
|
print("Full harness + era split (LONG)…")
|
||||||
|
print(f" {SURVIVORSHIP_BANNER}")
|
||||||
|
harness = await _harness(
|
||||||
|
snapshot, workers=args.workers, quiet=args.quiet
|
||||||
|
)
|
||||||
|
|
||||||
|
stamp = datetime.now().strftime("%Y%m%d-%H%M%S")
|
||||||
|
out = (
|
||||||
|
Path(args.out)
|
||||||
|
if args.out
|
||||||
|
else Path("reports") / f"history-depth-{stamp}.json"
|
||||||
|
)
|
||||||
|
payload = {
|
||||||
|
"generated_at": datetime.now().isoformat(),
|
||||||
|
"survivorship_banner": SURVIVORSHIP_BANNER,
|
||||||
|
"coverage": coverage,
|
||||||
|
"race_guard": race,
|
||||||
|
"harness": harness,
|
||||||
|
"verdict": "PENDING_HUMAN" if harness else "COVERAGE_ONLY",
|
||||||
|
"verdict_detail": (
|
||||||
|
"Harness complete — human interprets relative IC / era stability. "
|
||||||
|
"No production retune from this artifact."
|
||||||
|
if harness
|
||||||
|
else "Coverage probe only; run --phase harness after deep rebuild."
|
||||||
|
),
|
||||||
|
"human_next": (
|
||||||
|
"- Compare sector residual vs market residual across eras.\n"
|
||||||
|
"- If pre-2021 IC collapses, park Task 1 wire-in.\n"
|
||||||
|
"- Do not retune production knobs on deep history levels."
|
||||||
|
),
|
||||||
|
"report_path": str(out.as_posix()),
|
||||||
|
}
|
||||||
|
out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
out.write_text(json.dumps(payload, indent=2, default=str) + "\n", encoding="utf-8")
|
||||||
|
md = Path("docs/research/history-depth-extension.md")
|
||||||
|
_write_md(md, payload)
|
||||||
|
out.with_suffix(".md").write_text(md.read_text(encoding="utf-8"), encoding="utf-8")
|
||||||
|
print(f"Wrote {out}")
|
||||||
|
print(f"Wrote {md}")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(_main())
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -100,6 +100,75 @@ def test_residual_momentum_removes_market_beta_but_keeps_specific_drift():
|
|||||||
assert drift["mom_12_1_resid"] > pure["mom_12_1_resid"] + 0.12
|
assert drift["mom_12_1_resid"] > pure["mom_12_1_resid"] + 0.12
|
||||||
|
|
||||||
|
|
||||||
|
def test_sector_residual_momentum_two_factor():
|
||||||
|
"""Pure market+sector beta stock → sector resid ~0; idiosyncratic drift kept."""
|
||||||
|
dates, pure_beta, highs, benchmark = _signal_test_series(extra_return=0.0)
|
||||||
|
# Sector ETF = leveraged market (collinear-ish but not identical).
|
||||||
|
sector = {d: benchmark[d] * 1.02 + 0.5 for d in dates}
|
||||||
|
# Stock with pure exposure to market + sector, no alpha.
|
||||||
|
closes = [100.0]
|
||||||
|
for i in range(1, len(dates)):
|
||||||
|
m_prev = benchmark[dates[i - 1]]
|
||||||
|
m_cur = benchmark[dates[i]]
|
||||||
|
s_prev = sector[dates[i - 1]]
|
||||||
|
s_cur = sector[dates[i]]
|
||||||
|
m_ret = m_cur / m_prev - 1.0
|
||||||
|
s_ret = s_cur / s_prev - 1.0
|
||||||
|
closes.append(closes[-1] * (1.0 + 0.7 * m_ret + 0.5 * s_ret))
|
||||||
|
highs_p = [c * 1.01 for c in closes]
|
||||||
|
|
||||||
|
pure = bt._signal_values(
|
||||||
|
dates, closes, highs_p, 260, benchmark, sector_etf_closes=sector
|
||||||
|
)
|
||||||
|
assert "mom_12_1_sector_resid" in pure
|
||||||
|
assert pure["mom_12_1_sector_resid"] == pytest.approx(0.0, abs=0.05)
|
||||||
|
|
||||||
|
# Add idiosyncratic drift — sector residual should keep it.
|
||||||
|
drift_closes = [100.0]
|
||||||
|
for i in range(1, len(dates)):
|
||||||
|
m_prev = benchmark[dates[i - 1]]
|
||||||
|
m_cur = benchmark[dates[i]]
|
||||||
|
s_prev = sector[dates[i - 1]]
|
||||||
|
s_cur = sector[dates[i]]
|
||||||
|
m_ret = m_cur / m_prev - 1.0
|
||||||
|
s_ret = s_cur / s_prev - 1.0
|
||||||
|
drift_closes.append(
|
||||||
|
drift_closes[-1] * (1.0 + 0.7 * m_ret + 0.5 * s_ret + 0.0008)
|
||||||
|
)
|
||||||
|
drift_highs = [c * 1.01 for c in drift_closes]
|
||||||
|
drift = bt._signal_values(
|
||||||
|
dates, drift_closes, drift_highs, 260, benchmark, sector_etf_closes=sector
|
||||||
|
)
|
||||||
|
assert drift["mom_12_1_sector_resid"] > pure["mom_12_1_sector_resid"] + 0.10
|
||||||
|
|
||||||
|
|
||||||
|
def test_inject_sector_demeaned_momentum():
|
||||||
|
collected = {
|
||||||
|
"mom_12_1": {
|
||||||
|
(2024, 1): [
|
||||||
|
{"val": 0.20, "fwd": 0.01, "symbol": "AAA"},
|
||||||
|
{"val": 0.10, "fwd": 0.02, "symbol": "BBB"},
|
||||||
|
{"val": 0.40, "fwd": -0.01, "symbol": "CCC"},
|
||||||
|
{"val": 0.00, "fwd": 0.03, "symbol": "DDD"},
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
symbol_to_sector = {
|
||||||
|
"AAA": "Information Technology",
|
||||||
|
"BBB": "Information Technology",
|
||||||
|
"CCC": "Energy",
|
||||||
|
"DDD": "Energy",
|
||||||
|
}
|
||||||
|
bt._inject_sector_demeaned_momentum(collected, symbol_to_sector)
|
||||||
|
dem = collected["mom_12_1_sector_demeaned"][(2024, 1)]
|
||||||
|
by_sym = {r["symbol"]: r["val"] for r in dem}
|
||||||
|
# IT mean = 0.15 → AAA +0.05, BBB -0.05; Energy mean = 0.20 → CCC +0.20, DDD -0.20
|
||||||
|
assert by_sym["AAA"] == pytest.approx(0.05)
|
||||||
|
assert by_sym["BBB"] == pytest.approx(-0.05)
|
||||||
|
assert by_sym["CCC"] == pytest.approx(0.20)
|
||||||
|
assert by_sym["DDD"] == pytest.approx(-0.20)
|
||||||
|
|
||||||
|
|
||||||
def test_assigns_raw_and_residual_percentiles_independently():
|
def test_assigns_raw_and_residual_percentiles_independently():
|
||||||
cands = [
|
cands = [
|
||||||
{"iso_week": (2026, 1), "momentum": 0.10, "residual_momentum": 0.30},
|
{"iso_week": (2026, 1), "momentum": 0.10, "residual_momentum": 0.30},
|
||||||
|
|||||||
Reference in New Issue
Block a user