diff --git a/app/services/backtest_service.py b/app/services/backtest_service.py index 6b29be4..55280db 100644 --- a/app/services/backtest_service.py +++ b/app/services/backtest_service.py @@ -791,31 +791,86 @@ def _residual_momentum_12_1( with an intercept estimated over the same window, the arithmetic residuals sum to ~zero by construction, which would destroy the signal. """ - if not benchmark_closes or i - 252 < 0: + return _multi_factor_residual_momentum_12_1( + dates, closes, i, [benchmark_closes] if benchmark_closes else None + ) + + +def _multi_factor_residual_momentum_12_1( + dates: list[date], + closes: list[float], + i: int, + factor_closes: list[dict[date, float]] | None, +) -> float | None: + """12-1 residual momentum vs one or more factors (OLS, no intercept). + + Same formation window as raw / single-factor residual momentum: + daily returns from close[i-252] → close[i-21], require ≥100 paired obs. + Factors are stacked as columns; betas are OLS without intercept so the + cumulative residual is not forced to zero. + """ + if not factor_closes or i - 252 < 0: + return None + n_factors = len(factor_closes) + if n_factors < 1: return None stock_rets: list[float] = [] - market_rets: list[float] = [] - # Same daily intervals as mom_12_1: close[i-252] -> close[i-21]. + factor_rets: list[list[float]] = [[] for _ in range(n_factors)] for k in range(i - 251, i - 20): prev_close = closes[k - 1] - bench_prev = benchmark_closes.get(dates[k - 1]) - bench_cur = benchmark_closes.get(dates[k]) - if prev_close <= 0 or bench_prev is None or bench_cur is None or bench_prev <= 0: + if prev_close <= 0: + continue + f_day: list[float] = [] + ok = True + for fc in factor_closes: + f_prev = fc.get(dates[k - 1]) + f_cur = fc.get(dates[k]) + if f_prev is None or f_cur is None or f_prev <= 0: + ok = False + break + f_day.append(f_cur / f_prev - 1.0) + if not ok: continue stock_rets.append(closes[k] / prev_close - 1.0) - market_rets.append(bench_cur / bench_prev - 1.0) + for j, r in enumerate(f_day): + factor_rets[j].append(r) - if len(stock_rets) < 100: + n = len(stock_rets) + if n < 100: return None - mean_market = sum(market_rets) / len(market_rets) - mean_stock = sum(stock_rets) / len(stock_rets) - var_market = sum((x - mean_market) ** 2 for x in market_rets) - if var_market <= 0: + + if n_factors == 1: + # Fast path: identical algebra to the historical single-factor form. + market_rets = factor_rets[0] + mean_market = sum(market_rets) / n + mean_stock = sum(stock_rets) / n + var_market = sum((x - mean_market) ** 2 for x in market_rets) + if var_market <= 0: + return None + cov = sum( + (stock_rets[k] - mean_stock) * (market_rets[k] - mean_market) + for k in range(n) + ) + beta = cov / var_market + return sum(stock_rets[k] - beta * market_rets[k] for k in range(n)) + + # OLS without intercept: β = (X'X)^{-1} X'y for X columns = factor returns. + # Implemented for exactly two factors (market + sector); refuse larger. + if n_factors != 2: return None - cov = sum((stock_rets[k] - mean_stock) * (market_rets[k] - mean_market) for k in range(len(stock_rets))) - beta = cov / var_market - return sum(stock_rets[k] - beta * market_rets[k] for k in range(len(stock_rets))) + f1, f2 = factor_rets[0], factor_rets[1] + s11 = sum(a * a for a in f1) + s22 = sum(a * a for a in f2) + s12 = sum(f1[k] * f2[k] for k in range(n)) + sy1 = sum(stock_rets[k] * f1[k] for k in range(n)) + sy2 = sum(stock_rets[k] * f2[k] for k in range(n)) + det = s11 * s22 - s12 * s12 + if abs(det) < 1e-18: + return None + b1 = (s22 * sy1 - s12 * sy2) / det + b2 = (s11 * sy2 - s12 * sy1) / det + return sum(stock_rets[k] - b1 * f1[k] - b2 * f2[k] for k in range(n)) def _realized_vol_6m(closes: list[float], i: int) -> float | None: @@ -840,6 +895,7 @@ def _signal_values( highs: list[float], i: int, benchmark_closes: dict[date, float] | None = None, + sector_etf_closes: dict[date, float] | None = None, ) -> dict[str, float]: """Point-in-time candidate signals at as-of index ``i`` (price-only). @@ -851,6 +907,11 @@ def _signal_values( higher = nearer the high, expect positive IC). ``vol_6m`` is 126-day realized volatility (expect negative IC if the low-volatility anomaly holds). ``fip_id`` is Da/Gurun/Warachka information discreteness (expect negative IC). + + When ``sector_etf_closes`` is supplied (research path), also emit + ``mom_12_1_sector_resid``: two-factor residual vs SPY + sector ETF. + Cross-sectional ``mom_12_1_sector_demeaned`` is injected later from the + full weekly cross-section (cannot be computed per-ticker alone). """ out: dict[str, float] = {} if i - 252 >= 0 and closes[i - 252] > 0: @@ -858,6 +919,12 @@ def _signal_values( residual = _residual_momentum_12_1(dates, closes, i, benchmark_closes) if residual is not None: out["mom_12_1_resid"] = residual + if benchmark_closes and sector_etf_closes: + sector_resid = _multi_factor_residual_momentum_12_1( + dates, closes, i, [benchmark_closes, sector_etf_closes] + ) + if sector_resid is not None: + out["mom_12_1_sector_resid"] = sector_resid fip = _fip_id(closes, i) if fip is not None: out["fip_id"] = fip @@ -944,14 +1011,16 @@ def _accumulate_signal_series( benchmark_closes: dict[date, float] | None = None, *, symbol: str | None = None, + sector_etf_closes: dict[date, float] | None = None, ) -> None: """For each weekly as-of bar, emit (signal, forward-return) pairs keyed by ISO week into ``collected[name][week_key]``. Forward return is close-to-close over HORIZON trading days. Mutates ``collected`` (a dict of dict of list). When ``BACKTEST_LIQUID_BREADTH`` is set, observations are dicts with PIT - liquidity fields for the mask; otherwise plain ``(val, fwd)`` tuples so the - production signal path stays unchanged. + liquidity fields for the mask. When ``symbol`` is provided, observations are + also dicts (so sector demeaning can group by name); otherwise plain + ``(val, fwd)`` tuples keep the production path unchanged. """ n = len(records) if n < HORIZON + 21: @@ -961,6 +1030,7 @@ def _accumulate_signal_series( volumes = [float(getattr(r, "volume", 0) or 0) for r in records] dates = [r.date for r in records] liquid_mode = _liquid_breadth_top_n() > 0 + rich = liquid_mode or symbol is not None for i in _weekly_asof_indices(records): j = i + HORIZON if j >= n or closes[i] <= 0: @@ -969,19 +1039,79 @@ def _accumulate_signal_series( iso = records[i].date.isocalendar() week_key = (iso[0], iso[1]) dvol = _median_dollar_vol_63(closes, volumes, i) if liquid_mode else None - for name, val in _signal_values(dates, closes, highs, i, benchmark_closes).items(): - if liquid_mode: - collected[name][week_key].append({ + for name, val in _signal_values( + dates, closes, highs, i, benchmark_closes, sector_etf_closes + ).items(): + if rich: + row = { "val": val, "fwd": fwd, - "close": closes[i], - "median_dvol_63": dvol, "symbol": symbol, - }) + } + if liquid_mode: + row["close"] = closes[i] + row["median_dvol_63"] = dvol + collected[name][week_key].append(row) else: collected[name][week_key].append((val, fwd)) +def _inject_sector_demeaned_momentum( + collected: dict, + symbol_to_sector: dict[str, str], + *, + min_sector_names: int = 2, +) -> None: + """Cross-sectional demean of ``mom_12_1`` within GICS sector per week. + + ``mom_12_1_sector_demeaned[i] = mom_12_1[i] − mean(mom_12_1 | sector_i)``. + Requires rich observations with a ``symbol`` field (research path). Names + without a sector label, or sectors with fewer than ``min_sector_names`` + members that week, are dropped from the demeaned series. + """ + if not symbol_to_sector or "mom_12_1" not in collected: + return + from app.services.sector_map import normalise_symbol + + demeaned: dict = defaultdict(list) + for week_key, recs in collected["mom_12_1"].items(): + parsed: list[tuple[str, float, float, object]] = [] + by_sector: dict[str, list[float]] = defaultdict(list) + for rec in recs: + pair = _obs_val_fwd(rec) + if pair is None: + continue + val, fwd = pair + if isinstance(rec, dict): + sym = rec.get("symbol") + else: + sym = None + if not sym: + continue + sector = symbol_to_sector.get(normalise_symbol(str(sym))) + if not sector: + continue + parsed.append((sector, val, fwd, rec)) + by_sector[sector].append(val) + means = { + sec: sum(vs) / len(vs) + for sec, vs in by_sector.items() + if len(vs) >= min_sector_names + } + for sector, val, fwd, rec in parsed: + if sector not in means: + continue + dval = val - means[sector] + if isinstance(rec, dict): + row = dict(rec) + row["val"] = dval + demeaned[week_key].append(row) + else: + demeaned[week_key].append((dval, fwd)) + if demeaned: + collected["mom_12_1_sector_demeaned"] = demeaned + + def _rank(xs: list[float]) -> list[float]: """Average (tie-corrected) ranks, 1-based.""" order = sorted(range(len(xs)), key=lambda k: xs[k]) @@ -1258,14 +1388,38 @@ def _signal_series( benchmark_closes: dict[date, float] | None = None, *, symbol: str | None = None, + sector_etf_closes: dict[date, float] | None = None, ) -> dict: """Per-ticker signal/forward-return series as a PLAIN (picklable) nested dict — no defaultdict/lambda — so it can cross a process boundary.""" tmp: dict = defaultdict(lambda: defaultdict(list)) - _accumulate_signal_series(records, tmp, benchmark_closes, symbol=symbol) + _accumulate_signal_series( + records, + tmp, + benchmark_closes, + symbol=symbol, + sector_etf_closes=sector_etf_closes, + ) return {name: dict(weeks) for name, weeks in tmp.items()} +def _sector_etf_closes_for_symbol( + symbol: str, + symbol_to_sector: dict[str, str] | None, + sector_etf_closes: dict[str, dict[date, float]] | None, +) -> dict[date, float] | None: + """Resolve the sector-ETF close series for one ticker, or None.""" + if not symbol_to_sector or not sector_etf_closes: + return None + from app.services.sector_map import etf_for_symbol + + etf = etf_for_symbol(symbol, symbol_to_sector) + if not etf: + return None + series = sector_etf_closes.get(etf) + return series or None + + def _replay_and_signals( symbol: str, columns: tuple, @@ -1275,6 +1429,8 @@ def _replay_and_signals( target_model: str = PRODUCTION_GTL_TARGET_MODEL, cadence: str = DEFAULT_BACKTEST_CADENCE, signal_only: bool = False, + sector_etf_closes: dict[str, dict[date, float]] | None = None, + symbol_to_sector: dict[str, str] | None = None, ) -> tuple[list[dict], dict]: """The CPU-bound per-ticker work, as a top-level (picklable) function so it can run in a worker process. Takes primitive column arrays (cheap to pickle), @@ -1301,9 +1457,17 @@ def _replay_and_signals( target_model, cadence, ) + etf_closes = _sector_etf_closes_for_symbol( + symbol, symbol_to_sector, sector_etf_closes + ) return ( candidates, - _signal_series(bars, benchmark_closes, symbol=symbol), + _signal_series( + bars, + benchmark_closes, + symbol=symbol, + sector_etf_closes=etf_closes, + ), ) @@ -4054,6 +4218,41 @@ async def run_backtest( except Exception: logger.exception("Benchmark load for residual momentum failed") + # Optional sector residualisation (research): local ticker→sector map + sector + # ETF closes stored in benchmark_prices. Absent map/series → no sector signals. + symbol_to_sector: dict[str, str] = {} + sector_etf_closes: dict[str, dict[date, float]] = {} + try: + from app.services.sector_map import ( + SECTOR_ETFS, + load_ticker_sector_map, + normalise_symbol, + ) + from app.services.benchmark_service import load_benchmark_closes + + map_path = os.getenv("BACKTEST_SECTOR_MAP_PATH", "").strip() or None + symbol_to_sector = { + normalise_symbol(k): v + for k, v in load_ticker_sector_map(map_path).items() + } + if symbol_to_sector: + for etf in SECTOR_ETFS: + try: + series = await load_benchmark_closes(db, etf) + except Exception: + series = {} + if series: + sector_etf_closes[etf] = series + logger.info(json.dumps({ + "event": "backtest_sector_context_loaded", + "sector_map_size": len(symbol_to_sector), + "sector_etfs_loaded": sorted(sector_etf_closes), + })) + except Exception: + logger.exception("Sector residual context load failed; continuing without") + symbol_to_sector = {} + sector_etf_closes = {} + def _merge(result: tuple[list[dict], dict]) -> None: cands, series = result candidates.extend(cands) @@ -4104,6 +4303,8 @@ async def run_backtest( target_model, cadence, ticker.symbol in rank_only_symbols, + sector_etf_closes or None, + symbol_to_sector or None, )) for result in await asyncio.gather(*futures, return_exceptions=True): if isinstance(result, Exception): @@ -4132,6 +4333,8 @@ async def run_backtest( target_model, cadence, ticker.symbol in rank_only_symbols, + sector_etf_closes or None, + symbol_to_sector or None, )) except Exception: logger.exception("Backtest replay failed for %s", ticker.symbol) @@ -4139,6 +4342,13 @@ async def run_backtest( if progress_cb is not None and total: progress_cb(total, total, "") + # Cross-sectional sector demean needs the full weekly universe. + if symbol_to_sector: + try: + _inject_sector_demeaned_momentum(collected, symbol_to_sector) + except Exception: + logger.exception("Sector demeaned momentum injection failed") + # Cross-sectional momentum: rank every week's universe, then "qualified" means # floors + top ``min_momentum_percentile`` by promoted residual 12-1 momentum # (raw 12-1 fallback only when benchmark data is unavailable). diff --git a/app/services/sector_map.py b/app/services/sector_map.py new file mode 100644 index 0000000..f72230c --- /dev/null +++ b/app/services/sector_map.py @@ -0,0 +1,145 @@ +"""Ticker → GICS sector → SPDR sector ETF mapping (research only). + +Sector residual momentum residualizes 12-1 momentum against SPY and the name's +sector ETF. Labels are persisted under ``data/research/ticker_sector_map.json`` +so research runs do not depend on live FMP calls. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +# Eleven SPDR sector ETFs. Auxiliary series only — never tradable book members. +SECTOR_ETFS: tuple[str, ...] = ( + "XLB", + "XLC", + "XLE", + "XLF", + "XLI", + "XLK", + "XLP", + "XLRE", + "XLU", + "XLV", + "XLY", +) + +# GICS sector name (and common aliases) → SPDR ETF. +# Keys are lower-case for matching. +GICS_SECTOR_TO_ETF: dict[str, str] = { + "materials": "XLB", + "basic materials": "XLB", + "communication services": "XLC", + "communications": "XLC", + "telecommunication services": "XLC", + "energy": "XLE", + "financials": "XLF", + "financial services": "XLF", + "financial": "XLF", + "industrials": "XLI", + "industrial goods": "XLI", + "information technology": "XLK", + "technology": "XLK", + "consumer staples": "XLP", + "consumer defensive": "XLP", + "real estate": "XLRE", + "utilities": "XLU", + "health care": "XLV", + "healthcare": "XLV", + "consumer discretionary": "XLY", + "consumer cyclical": "XLY", +} + +DEFAULT_SECTOR_MAP_PATH = Path("data/research/ticker_sector_map.json") + + +def normalise_symbol(symbol: str) -> str: + """Alpaca-style symbols: BRK.B / BRK/B → BRK-B.""" + s = str(symbol or "").strip().upper() + s = s.replace(".", "-").replace("/", "-") + return s + + +def sector_to_etf(sector: str | None) -> str | None: + if not sector: + return None + return GICS_SECTOR_TO_ETF.get(str(sector).strip().lower()) + + +def etf_for_symbol(symbol: str, symbol_to_sector: dict[str, str]) -> str | None: + sector = symbol_to_sector.get(normalise_symbol(symbol)) + return sector_to_etf(sector) + + +def load_ticker_sector_map(path: Path | str | None = None) -> dict[str, str]: + """Load ``{symbol: gics_sector}`` from JSON. Empty dict if missing.""" + p = Path(path) if path is not None else DEFAULT_SECTOR_MAP_PATH + if not p.exists(): + return {} + raw = json.loads(p.read_text(encoding="utf-8")) + if not isinstance(raw, dict): + return {} + out: dict[str, str] = {} + # Accept either flat map or {"map": {...}, "meta": ...} + payload = raw.get("map") if "map" in raw and isinstance(raw.get("map"), dict) else raw + if not isinstance(payload, dict): + return {} + for sym, sector in payload.items(): + if sym in ("meta", "schema_version", "map"): + continue + if sector is None: + continue + ns = normalise_symbol(str(sym)) + if ns: + out[ns] = str(sector).strip() + return out + + +def save_ticker_sector_map( + mapping: dict[str, str], + path: Path | str | None = None, + *, + meta: dict[str, Any] | None = None, +) -> Path: + p = Path(path) if path is not None else DEFAULT_SECTOR_MAP_PATH + p.parent.mkdir(parents=True, exist_ok=True) + # Normalise keys on write. + clean = { + normalise_symbol(k): str(v).strip() + for k, v in mapping.items() + if k and v and normalise_symbol(k) + } + payload: dict[str, Any] = { + "schema_version": 1, + "map": clean, + "meta": meta or {}, + } + p.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8") + return p + + +def coverage_stats( + symbols: list[str], mapping: dict[str, str] +) -> dict[str, Any]: + total = len(symbols) + mapped = [s for s in symbols if normalise_symbol(s) in mapping] + with_etf = [ + s + for s in mapped + if sector_to_etf(mapping[normalise_symbol(s)]) is not None + ] + missing = [s for s in symbols if normalise_symbol(s) not in mapping] + by_sector: dict[str, int] = {} + for s in mapped: + sec = mapping[normalise_symbol(s)] + by_sector[sec] = by_sector.get(sec, 0) + 1 + return { + "universe": total, + "mapped": len(mapped), + "mapped_pct": round(100.0 * len(mapped) / total, 1) if total else 0.0, + "with_etf": len(with_etf), + "missing": missing, + "by_sector": dict(sorted(by_sector.items(), key=lambda kv: (-kv[1], kv[0]))), + } diff --git a/data/research/ticker_sector_map.json b/data/research/ticker_sector_map.json new file mode 100644 index 0000000..31cc0db --- /dev/null +++ b/data/research/ticker_sector_map.json @@ -0,0 +1,543 @@ +{ + "map": { + "A": "Health Care", + "AAPL": "Information Technology", + "ABBV": "Health Care", + "ABNB": "Consumer Discretionary", + "ABT": "Health Care", + "ACGL": "Financials", + "ACN": "Information Technology", + "ADBE": "Information Technology", + "ADI": "Information Technology", + "ADM": "Consumer Staples", + "ADP": "Industrials", + "ADSK": "Information Technology", + "AEE": "Utilities", + "AEP": "Utilities", + "AES": "Utilities", + "AFL": "Financials", + "AIG": "Financials", + "AIZ": "Financials", + "AJG": "Financials", + "AKAM": "Information Technology", + "ALB": "Materials", + "ALGN": "Health Care", + "ALL": "Financials", + "ALLE": "Industrials", + "AMAT": "Information Technology", + "AMCR": "Materials", + "AMD": "Information Technology", + "AME": "Industrials", + "AMGN": "Health Care", + "AMP": "Financials", + "AMT": "Real Estate", + "AMZN": "Consumer Discretionary", + "ANET": "Information Technology", + "AON": "Financials", + "AOS": "Industrials", + "APA": "Energy", + "APD": "Materials", + "APH": "Information Technology", + "APO": "Financials", + "APP": "Information Technology", + "APTV": "Consumer Discretionary", + "ARE": "Real Estate", + "ARES": "Financials", + "ATO": "Utilities", + "AVB": "Real Estate", + "AVGO": "Information Technology", + "AVY": "Materials", + "AWK": "Utilities", + "AXON": "Industrials", + "AXP": "Financials", + "AZO": "Consumer Discretionary", + "BA": "Industrials", + "BAC": "Financials", + "BALL": "Materials", + "BAX": "Health Care", + "BBY": "Consumer Discretionary", + "BDX": "Health Care", + "BEN": "Financials", + "BF-B": "Consumer Staples", + "BG": "Consumer Staples", + "BIIB": "Health Care", + "BK": "Financial Services", + "BKNG": "Consumer Discretionary", + "BKR": "Energy", + "BLDR": "Industrials", + "BLK": "Financials", + "BMY": "Health Care", + "BR": "Industrials", + "BRK-B": "Financials", + "BRO": "Financials", + "BSX": "Health Care", + "BX": "Financials", + "BXP": "Real Estate", + "C": "Financials", + "CAG": "Consumer Defensive", + "CAH": "Health Care", + "CARR": "Industrials", + "CASY": "Consumer Staples", + "CAT": "Industrials", + "CB": "Financials", + "CBOE": "Financials", + "CBRE": "Real Estate", + "CCI": "Real Estate", + "CCL": "Consumer Discretionary", + "CDNS": "Information Technology", + "CDW": "Information Technology", + "CEG": "Utilities", + "CF": "Materials", + "CFG": "Financials", + "CHD": "Consumer Staples", + "CHRW": "Industrials", + "CHTR": "Communication Services", + "CI": "Health Care", + "CIEN": "Information Technology", + "CINF": "Financials", + "CL": "Consumer Staples", + "CLX": "Consumer Staples", + "CMCSA": "Communication Services", + "CME": "Financials", + "CMG": "Consumer Discretionary", + "CMI": "Industrials", + "CMS": "Utilities", + "CNC": "Health Care", + "CNP": "Utilities", + "COF": "Financials", + "COHR": "Information Technology", + "COIN": "Financials", + "COO": "Health Care", + "COP": "Energy", + "COR": "Health Care", + "COST": "Consumer Staples", + "CPAY": "Financials", + "CPB": "Consumer Defensive", + "CPRT": "Industrials", + "CPT": "Real Estate", + "CRH": "Materials", + "CRL": "Health Care", + "CRM": "Information Technology", + "CRWD": "Information Technology", + "CSCO": "Information Technology", + "CSGP": "Real Estate", + "CSX": "Industrials", + "CTAS": "Industrials", + "CTRA": "Energy", + "CTSH": "Information Technology", + "CTVA": "Materials", + "CVNA": "Consumer Discretionary", + "CVS": "Health Care", + "CVX": "Energy", + "D": "Utilities", + "DAL": "Industrials", + "DASH": "Consumer Discretionary", + "DD": "Materials", + "DDOG": "Information Technology", + "DE": "Industrials", + "DECK": "Consumer Discretionary", + "DELL": "Information Technology", + "DG": "Consumer Staples", + "DGX": "Health Care", + "DHI": "Consumer Discretionary", + "DHR": "Health Care", + "DIS": "Communication Services", + "DLR": "Real Estate", + "DLTR": "Consumer Staples", + "DOC": "Real Estate", + "DOV": "Industrials", + "DOW": "Materials", + "DPZ": "Consumer Discretionary", + "DRI": "Consumer Discretionary", + "DTE": "Utilities", + "DUK": "Utilities", + "DVA": "Health Care", + "DVN": "Energy", + "DXCM": "Health Care", + "EA": "Communication Services", + "EBAY": "Consumer Discretionary", + "ECL": "Materials", + "ED": "Utilities", + "EFX": "Industrials", + "EG": "Financials", + "EIX": "Utilities", + "EL": "Consumer Staples", + "ELV": "Health Care", + "EME": "Industrials", + "EMR": "Industrials", + "EOG": "Energy", + "EPAM": "Technology", + "EQIX": "Real Estate", + "EQR": "Real Estate", + "EQT": "Energy", + "ERIE": "Financials", + "ES": "Utilities", + "ESS": "Real Estate", + "ETN": "Industrials", + "ETR": "Utilities", + "EVRG": "Utilities", + "EW": "Health Care", + "EXC": "Utilities", + "EXE": "Energy", + "EXPD": "Industrials", + "EXPE": "Consumer Discretionary", + "EXR": "Real Estate", + "F": "Consumer Discretionary", + "FANG": "Energy", + "FAST": "Industrials", + "FCX": "Materials", + "FDS": "Financials", + "FDX": "Industrials", + "FE": "Utilities", + "FFIV": "Information Technology", + "FICO": "Information Technology", + "FIS": "Financials", + "FISV": "Financials", + "FITB": "Financials", + "FIX": "Industrials", + "FOX": "Communication Services", + "FOXA": "Communication Services", + "FRT": "Real Estate", + "FSLR": "Information Technology", + "FTNT": "Information Technology", + "FTV": "Industrials", + "GD": "Industrials", + "GDDY": "Information Technology", + "GE": "Industrials", + "GEHC": "Health Care", + "GEN": "Information Technology", + "GEV": "Industrials", + "GILD": "Health Care", + "GIS": "Consumer Staples", + "GL": "Financials", + "GLW": "Information Technology", + "GM": "Consumer Discretionary", + "GNRC": "Industrials", + "GOOG": "Communication Services", + "GOOGL": "Communication Services", + "GPC": "Consumer Discretionary", + "GPN": "Financials", + "GRMN": "Consumer Discretionary", + "GS": "Financials", + "GWW": "Industrials", + "HAL": "Energy", + "HAS": "Consumer Discretionary", + "HBAN": "Financials", + "HCA": "Health Care", + "HD": "Consumer Discretionary", + "HIG": "Financials", + "HII": "Industrials", + "HLT": "Consumer Discretionary", + "HON": "Industrials", + "HOOD": "Financials", + "HPE": "Information Technology", + "HPQ": "Information Technology", + "HRL": "Consumer Staples", + "HSIC": "Health Care", + "HST": "Real Estate", + "HSY": "Consumer Staples", + "HUBB": "Industrials", + "HUM": "Health Care", + "HWM": "Industrials", + "IBKR": "Financials", + "IBM": "Information Technology", + "ICE": "Financials", + "IDXX": "Health Care", + "IEX": "Industrials", + "IFF": "Materials", + "INCY": "Health Care", + "INTC": "Information Technology", + "INTU": "Information Technology", + "INVH": "Real Estate", + "IP": "Materials", + "IQV": "Health Care", + "IR": "Industrials", + "IRM": "Real Estate", + "ISRG": "Health Care", + "IT": "Information Technology", + "ITW": "Industrials", + "IVZ": "Financials", + "J": "Industrials", + "JBHT": "Industrials", + "JBL": "Information Technology", + "JCI": "Industrials", + "JKHY": "Financials", + "JNJ": "Health Care", + "JPM": "Financials", + "KDP": "Consumer Staples", + "KEY": "Financials", + "KEYS": "Information Technology", + "KHC": "Consumer Staples", + "KIM": "Real Estate", + "KKR": "Financials", + "KLAC": "Information Technology", + "KMB": "Consumer Staples", + "KMI": "Energy", + "KO": "Consumer Staples", + "KR": "Consumer Staples", + "KVUE": "Consumer Staples", + "L": "Financials", + "LDOS": "Industrials", + "LEN": "Consumer Discretionary", + "LH": "Health Care", + "LHX": "Industrials", + "LII": "Industrials", + "LIN": "Materials", + "LITE": "Information Technology", + "LLY": "Health Care", + "LMT": "Industrials", + "LNT": "Utilities", + "LOW": "Consumer Discretionary", + "LRCX": "Information Technology", + "LULU": "Consumer Discretionary", + "LUV": "Industrials", + "LVS": "Consumer Discretionary", + "LYB": "Materials", + "LYV": "Communication Services", + "MA": "Financials", + "MAA": "Real Estate", + "MAR": "Consumer Discretionary", + "MAS": "Industrials", + "MCD": "Consumer Discretionary", + "MCHP": "Information Technology", + "MCK": "Health Care", + "MCO": "Financials", + "MDLZ": "Consumer Staples", + "MDT": "Health Care", + "MET": "Financials", + "META": "Communication Services", + "MGM": "Consumer Discretionary", + "MKC": "Consumer Staples", + "MLM": "Materials", + "MMM": "Industrials", + "MNST": "Consumer Staples", + "MO": "Consumer Staples", + "MOS": "Materials", + "MPC": "Energy", + "MPWR": "Information Technology", + "MRK": "Health Care", + "MRNA": "Health Care", + "MRSH": "Financials", + "MS": "Financials", + "MSCI": "Financials", + "MSFT": "Information Technology", + "MSI": "Information Technology", + "MSTR": "Technology", + "MTB": "Financials", + "MTD": "Health Care", + "MU": "Information Technology", + "NCLH": "Consumer Discretionary", + "NDAQ": "Financials", + "NDSN": "Industrials", + "NEE": "Utilities", + "NEM": "Materials", + "NFLX": "Communication Services", + "NI": "Utilities", + "NKE": "Consumer Discretionary", + "NOC": "Industrials", + "NOW": "Information Technology", + "NRG": "Utilities", + "NSC": "Industrials", + "NTAP": "Information Technology", + "NTRS": "Financials", + "NUE": "Materials", + "NVDA": "Information Technology", + "NVR": "Consumer Discretionary", + "NWS": "Communication Services", + "NWSA": "Communication Services", + "NXPI": "Information Technology", + "O": "Real Estate", + "ODFL": "Industrials", + "OKE": "Energy", + "OMC": "Communication Services", + "ON": "Information Technology", + "ORCL": "Information Technology", + "ORLY": "Consumer Discretionary", + "OTIS": "Industrials", + "OXY": "Energy", + "PANW": "Information Technology", + "PAYX": "Industrials", + "PCAR": "Industrials", + "PCG": "Utilities", + "PEG": "Utilities", + "PEP": "Consumer Staples", + "PFE": "Health Care", + "PFG": "Financials", + "PG": "Consumer Staples", + "PGR": "Financials", + "PH": "Industrials", + "PHM": "Consumer Discretionary", + "PKG": "Materials", + "PLD": "Real Estate", + "PLTR": "Information Technology", + "PM": "Consumer Staples", + "PNC": "Financials", + "PNR": "Industrials", + "PNW": "Utilities", + "PODD": "Health Care", + "POOL": "Industrials", + "PPG": "Materials", + "PPL": "Utilities", + "PRU": "Financials", + "PSA": "Real Estate", + "PSKY": "Communication Services", + "PSX": "Energy", + "PTC": "Information Technology", + "PWR": "Industrials", + "PYPL": "Financials", + "Q": "Information Technology", + "QCOM": "Information Technology", + "RCL": "Consumer Discretionary", + "REG": "Real Estate", + "REGN": "Health Care", + "RF": "Financials", + "RJF": "Financials", + "RL": "Consumer Discretionary", + "RMD": "Health Care", + "ROK": "Industrials", + "ROL": "Industrials", + "ROP": "Information Technology", + "ROST": "Consumer Discretionary", + "RSG": "Industrials", + "RTX": "Industrials", + "RVTY": "Health Care", + "SATS": "Communication Services", + "SBAC": "Real Estate", + "SBUX": "Consumer Discretionary", + "SCHW": "Financials", + "SHW": "Materials", + "SJM": "Consumer Staples", + "SLB": "Energy", + "SMCI": "Information Technology", + "SNA": "Industrials", + "SNDK": "Information Technology", + "SNPS": "Information Technology", + "SO": "Utilities", + "SOLV": "Health Care", + "SPCX": "Industrials", + "SPG": "Real Estate", + "SPGI": "Financials", + "SRE": "Utilities", + "STE": "Health Care", + "STLD": "Materials", + "STT": "Financials", + "STX": "Information Technology", + "STZ": "Consumer Staples", + "SW": "Materials", + "SWK": "Industrials", + "SWKS": "Information Technology", + "SYF": "Financials", + "SYK": "Health Care", + "SYY": "Consumer Staples", + "T": "Communication Services", + "TAP": "Consumer Staples", + "TDG": "Industrials", + "TDY": "Information Technology", + "TECH": "Health Care", + "TEL": "Information Technology", + "TER": "Information Technology", + "TFC": "Financials", + "TGT": "Consumer Staples", + "TJX": "Consumer Discretionary", + "TKO": "Communication Services", + "TMO": "Health Care", + "TMUS": "Communication Services", + "TPL": "Energy", + "TPR": "Consumer Discretionary", + "TRGP": "Energy", + "TRMB": "Information Technology", + "TROW": "Financials", + "TRV": "Financials", + "TSCO": "Consumer Discretionary", + "TSLA": "Consumer Discretionary", + "TSN": "Consumer Staples", + "TT": "Industrials", + "TTD": "Communication Services", + "TTWO": "Communication Services", + "TXN": "Information Technology", + "TXT": "Industrials", + "TYL": "Information Technology", + "UAL": "Industrials", + "UBER": "Industrials", + "UDR": "Real Estate", + "UHS": "Health Care", + "ULTA": "Consumer Discretionary", + "UNH": "Health Care", + "UNP": "Industrials", + "UPS": "Industrials", + "URI": "Industrials", + "USB": "Financials", + "V": "Financials", + "VICI": "Real Estate", + "VLO": "Energy", + "VLTO": "Industrials", + "VMC": "Materials", + "VRSK": "Industrials", + "VRSN": "Information Technology", + "VRT": "Industrials", + "VRTX": "Health Care", + "VST": "Utilities", + "VTR": "Real Estate", + "VTRS": "Health Care", + "VZ": "Communication Services", + "WAB": "Industrials", + "WAT": "Health Care", + "WBD": "Communication Services", + "WDAY": "Information Technology", + "WDC": "Information Technology", + "WEC": "Utilities", + "WELL": "Real Estate", + "WFC": "Financials", + "WM": "Industrials", + "WMB": "Energy", + "WMT": "Consumer Staples", + "WRB": "Financials", + "WSM": "Consumer Discretionary", + "WST": "Health Care", + "WTW": "Financials", + "WY": "Real Estate", + "WYNN": "Consumer Discretionary", + "XEL": "Utilities", + "XOM": "Energy", + "XYL": "Industrials", + "XYZ": "Financials", + "YUM": "Consumer Discretionary", + "ZBH": "Health Care", + "ZBRA": "Information Technology", + "ZTS": "Health Care" + }, + "meta": { + "built_at": "2026-07-19T05:35:41.184460+00:00", + "coverage": { + "by_sector": { + "Communication Services": 23, + "Consumer Defensive": 2, + "Consumer Discretionary": 47, + "Consumer Staples": 34, + "Energy": 22, + "Financial Services": 1, + "Financials": 75, + "Health Care": 58, + "Industrials": 81, + "Information Technology": 72, + "Materials": 26, + "Real Estate": 31, + "Technology": 2, + "Utilities": 31 + }, + "mapped": 505, + "mapped_pct": 99.8, + "universe": 506, + "with_etf": 505 + }, + "fmp_requests": 10, + "from_existing": 0, + "from_fmp": 9, + "from_sp500_csv": 496, + "snapshot": "C:\\Workspace\\signal-platform\\backtest_snapshots\\prod.sqlite", + "still_missing": [ + "RHM" + ] + }, + "schema_version": 1 +} diff --git a/docs/research/earnings-gap-and-sue.md b/docs/research/earnings-gap-and-sue.md new file mode 100644 index 0000000..4211c84 --- /dev/null +++ b/docs/research/earnings-gap-and-sue.md @@ -0,0 +1,202 @@ +# Earnings gap diagnostic + SUE / PEAD (Tier-1 alpha research) + +**Status:** **PARK** (incomplete earnings coverage; SUE fails iron rule on available sample). +**Branch:** `research/earnings-gap-and-sue` +**Production impact:** none. Local research only. **No filters shipped from 2a.** +**Artifacts:** `reports/earnings-gap-sue-20260719-093129.json` (+ companion `.md`) + +--- + +## Pre-registration (locked before first research run) + +### Data + +- Historical earnings calendar for the production universe over the full snapshot + window (and deeper if the feed provides it). +- Preferred source: FMP **date-range earnings-calendar** (bulk). If unavailable on + free tier, fall back to per-symbol `/stable/earnings` with request accounting. +- Store in a real local table `earnings_events` (symbol + announce_date key). +- Point-in-time: a surprise is usable only from **announce date + 1 trading day** + onward. + +### Experiment 2a — earnings-gap risk (defense, report-only) + +Join simulated production-config trades (`fill_mode=close`) with earnings dates. + +**Pre-registered questions:** + +1. What fraction of losses worse than **−1R** occur with an earnings announcement + **between entry and exit** (inclusive of the holding window)? +2. What is the mean R of entries taken within **3 trading days BEFORE** an + announcement vs all other entries — report **both tails** of the R + distribution (rule 4: any earnings-avoid entry filter is presumed guilty of + right-tail trimming until the win distribution shows otherwise)? + +**Output:** distributions and counts only. +**No filter is shipped.** If numbers argue for a filter → report and stop. + +### Experiment 2b — SUE / PEAD (offense) + +Signal `sue_latest`: + +\[ +\text{SUE} = \frac{\text{actual} - \text{estimate}}{\sigma(\text{trailing 8 surprises})} +\] + +Fallback if estimate history is thin: scale surprise by price. +Carry forward from announce+1 for **63 trading days**, else NaN (name drops out +of that cross-section). + +**Iron rule (IC harness):** mean weekly Spearman IC on non-overlapping weeks; +\|mean IC\| ≥ ~0.03, **positive** sign (drift), `reliable: true` (≥12 windows). + +Always side-by-side with `mom_12_1` and `mom_12_1_resid` on **identical** +cross-sections. + +Also report **momentum-conditional** IC (within top momentum quintile). + +**If it passes iron rule:** STOP and report. Book-integration design is a +separate human-approved step — do not wire. + +### Verdict labels + +| label | meaning | +|---|---| +| **PROMOTE** | (2b only) iron rule cleared → human designs tilt/gate | +| **PARK** | Interesting but incomplete / weak | +| **DEAD** | No edge / diagnostic argues against action | +| **REPORT-ONLY** | (2a) always — never auto-filter | + +--- + +## Data provenance + +| item | result | +|---|---| +| Snapshot | `backtest_snapshots/prod.sqlite` (506 names) | +| FMP bulk `earnings-calendar` | **402 Premium** — not available on free tier | +| FMP per-symbol `/stable/earnings` | used; hit daily rate limit ~225 reqs | +| Alpha Vantage `EARNINGS` | used for +24 symbols (announce = `reportedDate`) | +| Symbols with events | **48 / 506 (9.5%)** | +| Total events | 5,612 (5,018 with actual+estimate) | +| Announce range | 1985-08-31 → 2026-07-16 | +| FMP requests (first day) | 260 FMP + 25 AV (see `reports/earnings-backfill-status.json`) | + +**Incomplete backfill is first-class.** 2a under-detects earnings overlaps; 2b SUE +cross-section averages **~47 names**, not ~500. Resume: + +```bash +# Day N (FMP free ~250/day; AV free ~25/day — prefer FMP after reset) +python scripts/backfill_earnings_events.py \ + --snapshot backtest_snapshots/prod.sqlite \ + --provider fmp --force-symbol --limit 250 --sleep 0.4 + +# When done==506: +python scripts/run_earnings_research.py \ + --snapshot backtest_snapshots/prod.sqlite \ + --workers 6 --allow-spawn +``` + +--- + +## Results + +Generated: `2026-07-19T09:31:29` + +### 2a — Earnings-gap risk (report-only) + +Production book sim: Sharpe 2.09 (SE 0.497), CAGR 51.6%, max DD 21.4%, **322 trades**, +`fill_mode=close`. + +#### Q1 — Losses worse than −1R with earnings in hold + +| metric | value | +|---|---:| +| n losses < −1R | 28 | +| of which earnings in hold | **1** | +| fraction | **3.6%** | +| all trades with earnings in hold | 14 / 322 (4.4%) | + +**Read:** On incomplete earnings labels this is a **lower bound** on earnings +overlap, not a clean “earnings rarely hurt.” Do **not** conclude earnings risk is +immaterial until coverage ≥ ~95% of the book’s names. + +#### Q2 — Entry within 3 trading days before announce (both tails) + +| cohort | n | mean R | win rate | p05 | p50 | p95 | max | +|---|---:|---:|---:|---:|---:|---:|---:| +| pre-earn (≤3d before) | **4** | 1.94 | 50% | −1.24 | 1.12 | 6.26 | 6.84 | +| other | 318 | 0.70 | 37% | −1.11 | −0.83 | 6.08 | **12.87** | +| all | 322 | 0.71 | 37% | −1.12 | −0.83 | 6.22 | 12.87 | + +**Tail-trim presumption:** n=4 is not a sample. Point estimate does **not** show +right-tail destruction of pre-earn entries (p95 similar; max actually higher in +“other”). **No earnings-avoid filter is supported.** Re-run after full backfill. + +--- + +### 2b — SUE / PEAD IC + +#### Full-universe harness (mom on ~500; SUE only where labeled) + +| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable | +|---|---:|---:|---:|---:|---| +| mom_12_1_sector_resid | 0.0578 | 2.34 | 35 | 497.7 | true | +| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true | +| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true | +| **sue_latest** | **0.0172** | **0.6** | 44 | **47.4** | true | +| fip_id | −0.045 | −2.91 | 35 | 497.7 | true | + +#### Identical SUE subset (fair side-by-side — use this while coverage is thin) + +| signal | mean_ic | ic_t_stat | weeks | avg_N | +|---|---:|---:|---:|---:| +| sue_latest | 0.0172 | 0.6 | 44 | 47.4 | +| mom_12_1 | −0.0174 | −0.42 | 35 | 47.3 | +| mom_12_1_resid | −0.0104 | −0.27 | 35 | 47.3 | + +On the thin labeled subset, momentum itself is noise — so the subset is not yet +a meaningful PEAD test. + +#### Momentum-conditional SUE (top mom quintile) + +| metric | value | +|---|---:| +| mean IC | **−0.0065** | +| t | −0.1 | +| weeks | 35 | + +Wrong sign vs “ride positive surprises inside the momentum gate.” + +**Iron rule:** fail (\|IC\| 0.017 < 0.03; t 0.6). **No promote.** + +--- + +## Verdict + +| piece | verdict | +|---|---| +| **2a earnings-gap** | **REPORT-ONLY** — no filter. Coverage too thin for risk claims; tails do not argue for an avoid-filter on n=4. | +| **2b SUE** | **PARK** (effectively not green). Mild positive IC on ~48 names; fails iron bar; mom-conditional flat/negative. Re-score after full backfill before DEAD. | +| **Production** | **no change** | + +--- + +## What a human must decide next + +1. Resume multi-day earnings backfill to **506/506**, then re-run + `run_earnings_research.py` (heavy — MacBook OK). +2. Do **not** ship an earnings-avoid entry filter from 2a. +3. Do **not** wire SUE until a full-coverage IC clears the iron rule (and + preferably mom-conditional > 0). +4. Do not merge into main strategy docs without review. + +--- + +## Implementation notes + +| piece | role | +|---|---| +| `scripts/backfill_earnings_events.py` | bulk attempt → FMP/AV per-symbol; `earnings_events` + meta on snapshot | +| `scripts/run_earnings_research.py` | 2a trade join + 2b SUE IC / mom-conditional | +| Snapshot table `earnings_events` | real table (not SystemSetting JSON) | diff --git a/docs/research/history-depth-extension.md b/docs/research/history-depth-extension.md new file mode 100644 index 0000000..6ffcc16 --- /dev/null +++ b/docs/research/history-depth-extension.md @@ -0,0 +1,108 @@ +# History-depth extension (Tier-1 alpha research) + +**Status:** PRE-REGISTERED — run on MacBook (heavy I/O + full harness). +**Branch:** `research/history-depth-extension` (create from latest research stack). +**Production impact:** none. **Do not retune any production knob on deep history.** + +--- + +## Pre-registration (locked before rebuild) + +### Motivation + +All current conclusions rest on ~35 non-overlapping weekly windows in essentially +one post-2021 regime. Extending history toward max Alpaca daily-bar depth adds +the 2018 vol shock and full 2020 crash (where the feed allows). + +### Protocol + +1. **Empirical coverage first** — bars per calendar year per symbol; document + where the feed thins out. Do **not** assume a uniform start date. +2. **Rebuild the research snapshot completely** from prod source + max history + per symbol (`Adjustment.SPLIT`, ~200 req/min pacing via existing extender). +3. **Race guard (rule 6)** — refuse analysis until completion manifest is + `complete=true` and live counts match. +4. **Re-run full signal harness** (all existing signals incl. sector residual / + SUE if present) on the extended window. +5. **Report per signal:** mean IC, t, window count, and **era split** + (pre-/post-2021) — diagnostic only, **not a tuning input**. +6. **Log prominently:** survivorship bias grows with depth (today’s constituents + backfilled). Absolute Sharpe/CAGR on deep history is optimistic; payload is + **relative** signal comparisons and IC stability, not levels. +7. **Do not retune** production knobs. If a knob’s confirmation looks + overturned on deep history → report only; human decides. + +### Success / interpretation (not promotion of a new signal) + +| outcome | meaning | +|---|---| +| Sector residual still ≥ market residual on deep IC + stable sign | strengthens Task 1 PROMOTE case | +| Sector residual collapses pre-2021 | **PARK** Task 1 wire-in | +| SUE remains weak after full earnings + depth | **DEAD** SUE for this stack | +| Any production knob looks worse deep | report; no auto-retune | + +--- + +## MacBook runbook + +```bash +# 0. Repo + env +git fetch origin +git checkout research/earnings-gap-and-sue # or history-depth branch once pushed +# ensure .env has ALPACA_* (and FMP if resuming earnings) + +# 1. (Optional) finish earnings backfill first — multi-day free tier +python scripts/backfill_earnings_events.py \ + --snapshot backtest_snapshots/prod.sqlite \ + --provider fmp --force-symbol --limit 250 --sleep 0.35 + +# 2. Coverage probe (before long rebuild) +python scripts/run_history_depth_research.py --phase coverage \ + --snapshot backtest_snapshots/prod.sqlite + +# 3. Full deep rebuild of research.sqlite (LONG — Alpaca per symbol) +# Clears prior completion manifest; writes complete=true only at end. +python scripts/extend_snapshot_universe.py \ + --source backtest_snapshots/prod.sqlite \ + --output backtest_snapshots/research.sqlite \ + --force-copy \ + --history-days 5000 \ + --min-bars 260 \ + --sleep 0.15 + +# 4. Also refresh SPY + sector ETFs to the same depth on BOTH snapshots +python scripts/fetch_sector_etfs_to_snapshot.py \ + --snapshot backtest_snapshots/research.sqlite --history-days 5000 +python scripts/fetch_sector_etfs_to_snapshot.py \ + --snapshot backtest_snapshots/prod.sqlite --history-days 5000 + +# 5. Harness + era split (after race guard passes) +python scripts/run_history_depth_research.py --phase harness \ + --snapshot backtest_snapshots/research.sqlite \ + --workers 8 --allow-spawn + +# 6. Copy reports/ + docs/research/history-depth-extension.md results back +``` + +--- + +## Data provenance + +*(filled at run time)* + +--- + +## Results + +*(filled at run time)* + +--- + +## Verdict + +**Pending MacBook run.** + +## What a human must decide next + +- Do not retune production from deep history without explicit review. +- Use relative IC stability to accept/reject Task 1 sector residual wire-in. diff --git a/docs/research/sector-residual-momentum.md b/docs/research/sector-residual-momentum.md new file mode 100644 index 0000000..90a15ee --- /dev/null +++ b/docs/research/sector-residual-momentum.md @@ -0,0 +1,237 @@ +# Sector-residual momentum (Tier-1 alpha research) + +**Status:** **PROMOTE (to human design decision only)** — IC + A/B bars cleared; **do not ship**. +**Branch:** `research/sector-residual-momentum` +**Production impact:** none. Local research only. No scheduler / gate / prod-config changes. +**Artifacts:** `reports/sector-residual-20260719-083356.json` (+ companion `.md`) + +--- + +## Pre-registration (locked before first research run) + +### Hypothesis + +Residualizing 12–1 momentum against the sector, not only the market, reduces +factor volatility at similar return (Blitz / Huij / Martens-style) → higher +Sharpe on the production book when the residual replaces market-only residual +as the momentum leg. + +### Signals (candidates) + +| signal | construction | +|---|---| +| `mom_12_1_sector_resid` | Two-factor residual vs SPY + ticker’s sector ETF. Same window as `mom_12_1_resid`: ≥100 daily obs, 252-bar lookback, 21-bar skip; two-factor OLS betas **without intercept**; cumulate residual returns over the formation window. | +| `mom_12_1_sector_demeaned` | Plain `mom_12_1` minus the **cross-sectional** mean of `mom_12_1` within the same GICS sector that week (≥2 names in sector). No regression. | + +### Baselines (same run, same cross-sections — iron rule) + +Always report side-by-side with: + +- `mom_12_1` +- `mom_12_1_resid` + +Computed on the **identical** weekly non-overlapping cross-sections in this run. +Never compare against IC numbers from another report. + +### Iron rule (IC harness) + +Source of truth: `_signal_evaluation` in `app/services/backtest_service.py`. + +- Mean weekly Spearman IC on **non-overlapping** weekly windows +- Bar: \|mean IC\| ≥ ~0.03, **consistent positive sign**, `reliable: true` (≥ 12 windows) + +### Promotion to portfolio A/B (candidate → book) + +A candidate promotes to A/B **only if**: + +1. It clears the iron-rule bar **and** +2. Its IC **t-stat ≥** that of `mom_12_1_resid` on the same cross-sections. + +### Portfolio A/B grading (if and only if IC promotion fires) + +- Swap candidate in as the **momentum leg** of the production 80/20 momentum/vol + rank **and** as the gate-percentile signal. +- `fill_mode=close`, `COST_PER_SIDE = 0.001`, full config otherwise unchanged. +- Validation window = entries ≥ **2024-07-01** (call it **validation**, not + holdout — contaminated by prior experiments). +- Pre-registered promotion bar: + - validation Sharpe ≥ control − 0.5·SE + - full-period Sharpe and max-DD **not worse** than control +- Report Lo / Mertens-adjusted SEs. + +### Optional sector-cap sub-experiment + +Only if labels are in **and** A/B ran: max **3** positions per sector in the +10-slot book. Same A/B grading. **Tail-trim presumption of guilt** (rule 4): +report entry counts and both tails of the R distribution. Rising win rate with +falling Sharpe/CAGR = red flag → do not promote. + +**This run:** sector-cap arm **not executed** (optional; A/B unconstrained book +only). Can be a human-approved follow-up. + +### Verdict labels + +| label | meaning | +|---|---| +| **PROMOTE** | Clears pre-registered bar; human decides next (wire design separate) | +| **PARK** | Inconclusive / weak; keep machinery, no book change | +| **DEAD** | Failed iron rule or worse than residual baseline with clear sign | + +### Explicit non-goals + +- No production deploy from this doc +- Do not resurrect: take-profit exits, EV gate, regime entry-blocking, + inverse-vol sizing, gap-caps, unconditional FIP filter + +--- + +## Data provenance + +### Snapshot race guard + +| check | result | +|---|---| +| Snapshot path | `backtest_snapshots/prod.sqlite` | +| Manifest | none (expected for prod snapshot); bar-count sanity applied | +| Tickers / OHLCV | **506** / **629,263** | +| Bars min / avg / max | 14 / 1246.1 / 1261 | +| OHLCV range | 2021-06-24 → 2026-07-02 | +| Partial-build red flags | none (avg bars healthy) | + +Integrity fingerprint on same run: `fip_id` mean IC **−0.045** / t **−2.91** +(35 weeks, N≈498) — matches the established prod fingerprint. + +### Sector labels + +| source | count | +|---|---:| +| Public S&P 500 GICS CSV | 496 newly filled | +| FMP profile requests | 10 (all missing after CSV) | +| Mapped / universe | **505 / 506 (99.8%)** | +| With mappable ETF | 505 | +| Still missing | **RHM** only | + +Persist path: `data/research/ticker_sector_map.json`. + +FMP aliases (`Technology`, `Consumer Defensive`, `Financial Services`) map to +SPDRs via the alias table in `app/services/sector_map.py`. + +### Sector ETFs in `benchmark_prices` (auxiliary only — not tradable) + +| symbol | bars | min date | max date | +|---|---:|---|---| +| SPY | 1516 | 2020-07-06 | 2026-07-17 | +| XLB…XLY (11) | 1512 each | 2020-07-10 | 2026-07-17 | + +Fetched via Alpaca `Adjustment.SPLIT` into **`benchmark_prices`** (same table as +SPY) so they never enter the ticker universe or candidate replay. + +--- + +## Results + +Generated: `2026-07-19T08:33:56` + +### IC harness (identical cross-sections, production 506-name universe) + +| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable | ic+_pct | quintile spread | +|---|---:|---:|---:|---:|---|---:|---:| +| **mom_12_1_sector_resid** | **0.0578** | **2.34** | 35 | 497.7 | true | 65.7 | 0.0245 | +| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true | 60.0 | 0.0207 | +| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true | 65.7 | 0.0206 | +| mom_12_1_sector_demeaned | 0.0340 | 1.32 | 35 | 496.7 | true | 62.9 | 0.0154 | + +### IC promotion grades + +| candidate | iron rule | t ≥ resid | promote_to_ab | +|---|---|---|---| +| `mom_12_1_sector_resid` | pass (IC 0.058, +sign, reliable) | **yes** (2.34 ≥ 1.98) | **yes** | +| `mom_12_1_sector_demeaned` | pass (IC 0.034, +sign, reliable) | **no** (1.32 < 1.98) | **no** | + +### Portfolio A/B — `mom_12_1_sector_resid` as residual leg + +Config: production 80/20 residual/high-vol rank + gate percentile, `fill_mode=close`, +cost 10 bps/side, ATR trail / gate-reset re-entry as live. Validation split +2024-07-01. + +| window | arm | Sharpe | Sharpe SE (Mertens) | CAGR % | max DD % | trades | n_days | +|---|---|---:|---:|---:|---:|---:|---:| +| train | control (resid) | 1.30 | 0.685 | 29.2 | 21.4 | 176 | 525 | +| train | treatment (sector resid) | **1.57** | 0.677 | **35.5** | **19.8** | 176 | 530 | +| validation | control | **2.92** | 0.709 | **76.3** | **11.7** | 150 | 501 | +| validation | treatment | 2.57 | 0.701 | 66.3 | 14.8 | 163 | 501 | +| full | control | 2.09 | 0.497 | 51.6 | 21.4 | 322 | 1000 | +| full | treatment | 2.09 | 0.491 | 51.0 | **19.8** | 337 | 1005 | + +**Pre-registered A/B checks** + +| check | result | +|---|---| +| val Sharpe ≥ control − 0.5·SE | **pass** (2.57 ≥ 2.92 − 0.5×0.701 = 2.5695) — **knife-edge** | +| full Sharpe not worse | **pass** (2.09 = 2.09) | +| full max DD not worse | **pass** (19.8 < 21.4) | + +Qualified long candidates: control 1086 vs treatment 1210 (sector residual +gates a slightly larger set). + +--- + +## Verdict + +| signal | verdict | note | +|---|---|---| +| **`mom_12_1_sector_resid`** | **PROMOTE → human wire-in decision** | IC modestly beats market residual; A/B clears pre-reg bar narrowly. **Do not ship from this branch.** | +| **`mom_12_1_sector_demeaned`** | **DEAD** (for promotion) | Iron-rule IC magnitude ok, but t-stat loses to `mom_12_1_resid`. Cheap variant not competitive. | + +### Read carefully (for the human) + +1. **IC edge is real but small.** Sector residual IC 0.0578 / t 2.34 vs market + residual 0.0552 / t 1.98 on the **same** 35 windows — better consistency + (ic+ 65.7% vs 60%) and slightly higher mean, not a different factor class. +2. **A/B is not a clear Sharpe win.** Full-period Sharpe is flat (2.09). + Validation Sharpe is **lower** than control (2.57 vs 2.92) and only clears + the pre-registered “within 0.5 SE” cushion by ~0.001. Train improves; + validation worsens — classic regime-split noise on ~2 years. +3. **Risk side is friendly.** Full max DD improves (19.8% vs 21.4%); train DD + also better. Matches the “lower factor vol” half of the hypothesis more than + the “higher Sharpe” half on this window. +4. **Survivorship / short history.** Same caveats as all current research: + today’s constituents, ~35 independent weekly windows, one post-2021 regime + dominant. Task 3 (history depth) should re-check IC stability before any + wire-in. +5. **Not shipped.** Machinery lives on the research branch; production residual + path is untouched. + +--- + +## What a human must decide next + +1. **Accept or reject** replacing `mom_12_1_resid` with `mom_12_1_sector_resid` + as the production residual (gate + 80/20 mom leg), **or** keep market residual + and treat sector residual as research-only. +2. If leaning accept: require **Task 3 history-depth** confirmation (IC era split + pre/post-2021) before any production PR. +3. Optional: run **sector-cap ≤3** A/B with full tail diagnostics (not run here). +4. **Do not** merge this verdict into main strategy docs without review. +5. Wire-in design (live sector map refresh, ETF series ops, fallback when sector + missing) is a **separate** approved engineering step. + +--- + +## Implementation notes (research machinery) + +| piece | role | +|---|---| +| `app/services/sector_map.py` | GICS→ETF map, symbol normalise, JSON load/save | +| `app/services/backtest_service.py` | multi-factor residual; `mom_12_1_sector_resid` in `_signal_values`; demean inject | +| `scripts/build_ticker_sector_map.py` | SP500 CSV + FMP gap fill | +| `scripts/fetch_sector_etfs_to_snapshot.py` | Alpaca → snapshot `benchmark_prices` | +| `scripts/run_sector_residual_research.py` | race guard, IC, optional A/B, reports | +| `data/research/ticker_sector_map.json` | persisted labels (research only) | + +--- + +## Artifacts + +- JSON: `reports/sector-residual-20260719-083356.json` +- MD copy: `reports/sector-residual-20260719-083356.md` diff --git a/reports/earnings-backfill-status.json b/reports/earnings-backfill-status.json new file mode 100644 index 0000000..7b22f25 --- /dev/null +++ b/reports/earnings-backfill-status.json @@ -0,0 +1,15 @@ +{ + "mode": "per_symbol", + "fmp_requests": 25, + "events_written_this_run": 2541, + "total_events": 5612, + "symbols_done": 48, + "symbols_universe": 506, + "announce_date_range": { + "min": "1985-08-31", + "max": "2026-07-16" + }, + "events_with_actual_and_estimate": 5018, + "budget": 25, + "complete": false +} diff --git a/reports/earnings-gap-sue-20260719-093129.json b/reports/earnings-gap-sue-20260719-093129.json new file mode 100644 index 0000000..57dc35f --- /dev/null +++ b/reports/earnings-gap-sue-20260719-093129.json @@ -0,0 +1,331 @@ +{ + "generated_at": "2026-07-19T09:31:29.078611", + "data_provenance": { + "snapshot": "C:\\Workspace\\signal-platform\\backtest_snapshots\\prod.sqlite", + "n_earnings_events": 5612, + "backfill_meta": { + "done": 48, + "universe_tickers": 506 + }, + "announce_range": { + "min": "1985-08-31", + "max": "2026-07-16" + }, + "with_actual_and_estimate": 5018 + }, + "experiment_2a": { + "sim_summary": { + "sharpe": 2.09, + "sharpe_se": 0.497, + "cagr_pct": 51.6, + "max_drawdown_pct": 21.4, + "trades": 322, + "total_return_pct": 424.6 + }, + "n_trades_parsed": 322, + "q1_losses_worse_than_minus_1r": { + "n_losses_lt_minus_1r": 28, + "n_with_earnings_in_hold": 1, + "fraction_with_earnings": 0.0357, + "all_trades_with_earnings_in_hold": 14, + "fraction_all_trades_with_earnings": 0.0435 + }, + "q2_entry_within_3d_before_announce": { + "pre_earn_entries": { + "n": 4, + "mean": 1.9379, + "win_rate": 0.5, + "p05": -1.2428, + "p25": -0.8833, + "p50": 1.1209, + "p75": 3.942, + "p95": 6.2623, + "min": -1.3327, + "max": 6.8424 + }, + "other_entries": { + "n": 318, + "mean": 0.6965, + "win_rate": 0.3711, + "p05": -1.1052, + "p25": -1.0, + "p50": -0.8259, + "p75": 2.1053, + "p95": 6.077, + "min": -3.2587, + "max": 12.8654 + }, + "all_entries": { + "n": 322, + "mean": 0.7119, + "win_rate": 0.3727, + "p05": -1.1209, + "p25": -1.0, + "p50": -0.8251, + "p75": 2.1595, + "p95": 6.2246, + "min": -3.2587, + "max": 12.8654 + }, + "tail_trim_note": "Compare p95/max and mean of pre_earn vs other. Rising win_rate with falling mean/p95 = right-tail trim red flag." + }, + "note": "REPORT-ONLY \u2014 no filter shipped." + }, + "experiment_2b": { + "signal_eval_side_by_side": { + "mom_12_1": { + "signal": "mom_12_1", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0531, + "ic_t_stat": 1.61, + "ic_positive_pct": 65.7, + "mean_quintile_spread": 0.0206, + "reliable": true + }, + "mom_12_1_resid": { + "signal": "mom_12_1_resid", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0552, + "ic_t_stat": 1.98, + "ic_positive_pct": 60.0, + "mean_quintile_spread": 0.0207, + "reliable": true + }, + "mom_12_1_sector_resid": { + "signal": "mom_12_1_sector_resid", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0578, + "ic_t_stat": 2.34, + "ic_positive_pct": 65.7, + "mean_quintile_spread": 0.0245, + "reliable": true + }, + "mom_12_1_sector_demeaned": { + "signal": "mom_12_1_sector_demeaned", + "weeks": 35, + "avg_cross_section": 496.7, + "mean_ic": 0.034, + "ic_t_stat": 1.32, + "ic_positive_pct": 62.9, + "mean_quintile_spread": 0.0154, + "reliable": true + }, + "sue_latest": { + "signal": "sue_latest", + "weeks": 44, + "avg_cross_section": 47.4, + "mean_ic": 0.0172, + "ic_t_stat": 0.6, + "ic_positive_pct": 47.7, + "mean_quintile_spread": 0.0064, + "reliable": true + }, + "fip_id": { + "signal": "fip_id", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": -0.045, + "ic_t_stat": -2.91, + "ic_positive_pct": 25.7, + "mean_quintile_spread": -0.0168, + "reliable": true + } + }, + "signal_eval_identical_sue_subset": { + "mom_12_1": { + "signal": "mom_12_1", + "weeks": 35, + "avg_cross_section": 47.3, + "mean_ic": -0.0174, + "ic_t_stat": -0.42, + "ic_positive_pct": 45.7, + "mean_quintile_spread": 0.0077, + "reliable": true + }, + "mom_12_1_resid": { + "signal": "mom_12_1_resid", + "weeks": 35, + "avg_cross_section": 47.3, + "mean_ic": -0.0104, + "ic_t_stat": -0.27, + "ic_positive_pct": 51.4, + "mean_quintile_spread": 0.0075, + "reliable": true + }, + "sue_latest": { + "signal": "sue_latest", + "weeks": 44, + "avg_cross_section": 47.4, + "mean_ic": 0.0172, + "ic_t_stat": 0.6, + "ic_positive_pct": 47.7, + "mean_quintile_spread": 0.0064, + "reliable": true + } + }, + "identical_subset_note": "Mom baselines re-scored only on (week, symbol) cells where SUE exists. Use this table when backfill is incomplete \u2014 full-universe mom N is not comparable.", + "full_signal_eval": [ + { + "signal": "vol_6m", + "weeks": 39, + "avg_cross_section": 498.2, + "mean_ic": 0.0609, + "ic_t_stat": 1.48, + "ic_positive_pct": 64.1, + "mean_quintile_spread": 0.0337, + "reliable": true + }, + { + "signal": "mom_12_1_sector_resid", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0578, + "ic_t_stat": 2.34, + "ic_positive_pct": 65.7, + "mean_quintile_spread": 0.0245, + "reliable": true + }, + { + "signal": "mom_12_1_resid", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0552, + "ic_t_stat": 1.98, + "ic_positive_pct": 60.0, + "mean_quintile_spread": 0.0207, + "reliable": true + }, + { + "signal": "mom_12_1", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0531, + "ic_t_stat": 1.61, + "ic_positive_pct": 65.7, + "mean_quintile_spread": 0.0206, + "reliable": true + }, + { + "signal": "mom_12_1_sector_demeaned", + "weeks": 35, + "avg_cross_section": 496.7, + "mean_ic": 0.034, + "ic_t_stat": 1.32, + "ic_positive_pct": 62.9, + "mean_quintile_spread": 0.0154, + "reliable": true + }, + { + "signal": "sue_latest", + "weeks": 44, + "avg_cross_section": 47.4, + "mean_ic": 0.0172, + "ic_t_stat": 0.6, + "ic_positive_pct": 47.7, + "mean_quintile_spread": 0.0064, + "reliable": true + }, + { + "signal": "trend_200", + "weeks": 37, + "avg_cross_section": 497.9, + "mean_ic": 0.0161, + "ic_t_stat": 0.44, + "ic_positive_pct": 59.5, + "mean_quintile_spread": 0.006, + "reliable": true + }, + { + "signal": "reversal_1m", + "weeks": 43, + "avg_cross_section": 498.7, + "mean_ic": 0.0059, + "ic_t_stat": 0.22, + "ic_positive_pct": 53.5, + "mean_quintile_spread": 0.0053, + "reliable": true + }, + { + "signal": "mom_6_1", + "weeks": 39, + "avg_cross_section": 498.2, + "mean_ic": 0.0051, + "ic_t_stat": 0.21, + "ic_positive_pct": 56.4, + "mean_quintile_spread": 0.0087, + "reliable": true + }, + { + "signal": "mom_3_1", + "weeks": 42, + "avg_cross_section": 498.5, + "mean_ic": -0.0064, + "ic_t_stat": -0.25, + "ic_positive_pct": 50.0, + "mean_quintile_spread": 0.0046, + "reliable": true + }, + { + "signal": "high_52w", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": -0.0086, + "ic_t_stat": -0.26, + "ic_positive_pct": 54.3, + "mean_quintile_spread": -0.0088, + "reliable": true + }, + { + "signal": "fip_id", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": -0.045, + "ic_t_stat": -2.91, + "ic_positive_pct": 25.7, + "mean_quintile_spread": -0.0168, + "reliable": true + } + ], + "sue_grade": { + "green": false, + "checks": { + "mean_ic": 0.0172, + "sign_positive": true, + "abs_ge_0_03": false, + "reliable": true, + "ic_t_stat": 0.6, + "weeks": 44 + }, + "reason": "iron rule not met", + "row": { + "signal": "sue_latest", + "weeks": 44, + "avg_cross_section": 47.4, + "mean_ic": 0.0172, + "ic_t_stat": 0.6, + "ic_positive_pct": 47.7, + "mean_quintile_spread": 0.0064, + "reliable": true + } + }, + "momentum_conditional_sue": { + "mean_ic": -0.0065, + "ic_t_stat": -0.1, + "weeks": 35, + "note": "IC of sue_latest within top mom_12_1 quintile (non-overlapping weeks)" + }, + "sue_coverage": { + "symbols_with_sue": 48, + "avg_weeks_with_sue": 47.1, + "weeks_with_min_cross_section": 256 + } + }, + "verdict": "PARK", + "verdict_detail": "SUE IC=0.0172 below iron bar or unreliable; keep data, no wire.", + "human_next": "- No SUE book change.\n- Read 2a tails before considering any earnings-avoid filter.", + "report_path": "reports/earnings-gap-sue-20260719-093129.json", + "fmp_note": "Bulk earnings-calendar is paid (402 on free tier). Backfill used per-symbol /stable/earnings; see earnings-backfill-status.json." +} diff --git a/reports/earnings-gap-sue-20260719-093129.md b/reports/earnings-gap-sue-20260719-093129.md new file mode 100644 index 0000000..4211c84 --- /dev/null +++ b/reports/earnings-gap-sue-20260719-093129.md @@ -0,0 +1,202 @@ +# Earnings gap diagnostic + SUE / PEAD (Tier-1 alpha research) + +**Status:** **PARK** (incomplete earnings coverage; SUE fails iron rule on available sample). +**Branch:** `research/earnings-gap-and-sue` +**Production impact:** none. Local research only. **No filters shipped from 2a.** +**Artifacts:** `reports/earnings-gap-sue-20260719-093129.json` (+ companion `.md`) + +--- + +## Pre-registration (locked before first research run) + +### Data + +- Historical earnings calendar for the production universe over the full snapshot + window (and deeper if the feed provides it). +- Preferred source: FMP **date-range earnings-calendar** (bulk). If unavailable on + free tier, fall back to per-symbol `/stable/earnings` with request accounting. +- Store in a real local table `earnings_events` (symbol + announce_date key). +- Point-in-time: a surprise is usable only from **announce date + 1 trading day** + onward. + +### Experiment 2a — earnings-gap risk (defense, report-only) + +Join simulated production-config trades (`fill_mode=close`) with earnings dates. + +**Pre-registered questions:** + +1. What fraction of losses worse than **−1R** occur with an earnings announcement + **between entry and exit** (inclusive of the holding window)? +2. What is the mean R of entries taken within **3 trading days BEFORE** an + announcement vs all other entries — report **both tails** of the R + distribution (rule 4: any earnings-avoid entry filter is presumed guilty of + right-tail trimming until the win distribution shows otherwise)? + +**Output:** distributions and counts only. +**No filter is shipped.** If numbers argue for a filter → report and stop. + +### Experiment 2b — SUE / PEAD (offense) + +Signal `sue_latest`: + +\[ +\text{SUE} = \frac{\text{actual} - \text{estimate}}{\sigma(\text{trailing 8 surprises})} +\] + +Fallback if estimate history is thin: scale surprise by price. +Carry forward from announce+1 for **63 trading days**, else NaN (name drops out +of that cross-section). + +**Iron rule (IC harness):** mean weekly Spearman IC on non-overlapping weeks; +\|mean IC\| ≥ ~0.03, **positive** sign (drift), `reliable: true` (≥12 windows). + +Always side-by-side with `mom_12_1` and `mom_12_1_resid` on **identical** +cross-sections. + +Also report **momentum-conditional** IC (within top momentum quintile). + +**If it passes iron rule:** STOP and report. Book-integration design is a +separate human-approved step — do not wire. + +### Verdict labels + +| label | meaning | +|---|---| +| **PROMOTE** | (2b only) iron rule cleared → human designs tilt/gate | +| **PARK** | Interesting but incomplete / weak | +| **DEAD** | No edge / diagnostic argues against action | +| **REPORT-ONLY** | (2a) always — never auto-filter | + +--- + +## Data provenance + +| item | result | +|---|---| +| Snapshot | `backtest_snapshots/prod.sqlite` (506 names) | +| FMP bulk `earnings-calendar` | **402 Premium** — not available on free tier | +| FMP per-symbol `/stable/earnings` | used; hit daily rate limit ~225 reqs | +| Alpha Vantage `EARNINGS` | used for +24 symbols (announce = `reportedDate`) | +| Symbols with events | **48 / 506 (9.5%)** | +| Total events | 5,612 (5,018 with actual+estimate) | +| Announce range | 1985-08-31 → 2026-07-16 | +| FMP requests (first day) | 260 FMP + 25 AV (see `reports/earnings-backfill-status.json`) | + +**Incomplete backfill is first-class.** 2a under-detects earnings overlaps; 2b SUE +cross-section averages **~47 names**, not ~500. Resume: + +```bash +# Day N (FMP free ~250/day; AV free ~25/day — prefer FMP after reset) +python scripts/backfill_earnings_events.py \ + --snapshot backtest_snapshots/prod.sqlite \ + --provider fmp --force-symbol --limit 250 --sleep 0.4 + +# When done==506: +python scripts/run_earnings_research.py \ + --snapshot backtest_snapshots/prod.sqlite \ + --workers 6 --allow-spawn +``` + +--- + +## Results + +Generated: `2026-07-19T09:31:29` + +### 2a — Earnings-gap risk (report-only) + +Production book sim: Sharpe 2.09 (SE 0.497), CAGR 51.6%, max DD 21.4%, **322 trades**, +`fill_mode=close`. + +#### Q1 — Losses worse than −1R with earnings in hold + +| metric | value | +|---|---:| +| n losses < −1R | 28 | +| of which earnings in hold | **1** | +| fraction | **3.6%** | +| all trades with earnings in hold | 14 / 322 (4.4%) | + +**Read:** On incomplete earnings labels this is a **lower bound** on earnings +overlap, not a clean “earnings rarely hurt.” Do **not** conclude earnings risk is +immaterial until coverage ≥ ~95% of the book’s names. + +#### Q2 — Entry within 3 trading days before announce (both tails) + +| cohort | n | mean R | win rate | p05 | p50 | p95 | max | +|---|---:|---:|---:|---:|---:|---:|---:| +| pre-earn (≤3d before) | **4** | 1.94 | 50% | −1.24 | 1.12 | 6.26 | 6.84 | +| other | 318 | 0.70 | 37% | −1.11 | −0.83 | 6.08 | **12.87** | +| all | 322 | 0.71 | 37% | −1.12 | −0.83 | 6.22 | 12.87 | + +**Tail-trim presumption:** n=4 is not a sample. Point estimate does **not** show +right-tail destruction of pre-earn entries (p95 similar; max actually higher in +“other”). **No earnings-avoid filter is supported.** Re-run after full backfill. + +--- + +### 2b — SUE / PEAD IC + +#### Full-universe harness (mom on ~500; SUE only where labeled) + +| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable | +|---|---:|---:|---:|---:|---| +| mom_12_1_sector_resid | 0.0578 | 2.34 | 35 | 497.7 | true | +| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true | +| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true | +| **sue_latest** | **0.0172** | **0.6** | 44 | **47.4** | true | +| fip_id | −0.045 | −2.91 | 35 | 497.7 | true | + +#### Identical SUE subset (fair side-by-side — use this while coverage is thin) + +| signal | mean_ic | ic_t_stat | weeks | avg_N | +|---|---:|---:|---:|---:| +| sue_latest | 0.0172 | 0.6 | 44 | 47.4 | +| mom_12_1 | −0.0174 | −0.42 | 35 | 47.3 | +| mom_12_1_resid | −0.0104 | −0.27 | 35 | 47.3 | + +On the thin labeled subset, momentum itself is noise — so the subset is not yet +a meaningful PEAD test. + +#### Momentum-conditional SUE (top mom quintile) + +| metric | value | +|---|---:| +| mean IC | **−0.0065** | +| t | −0.1 | +| weeks | 35 | + +Wrong sign vs “ride positive surprises inside the momentum gate.” + +**Iron rule:** fail (\|IC\| 0.017 < 0.03; t 0.6). **No promote.** + +--- + +## Verdict + +| piece | verdict | +|---|---| +| **2a earnings-gap** | **REPORT-ONLY** — no filter. Coverage too thin for risk claims; tails do not argue for an avoid-filter on n=4. | +| **2b SUE** | **PARK** (effectively not green). Mild positive IC on ~48 names; fails iron bar; mom-conditional flat/negative. Re-score after full backfill before DEAD. | +| **Production** | **no change** | + +--- + +## What a human must decide next + +1. Resume multi-day earnings backfill to **506/506**, then re-run + `run_earnings_research.py` (heavy — MacBook OK). +2. Do **not** ship an earnings-avoid entry filter from 2a. +3. Do **not** wire SUE until a full-coverage IC clears the iron rule (and + preferably mom-conditional > 0). +4. Do not merge into main strategy docs without review. + +--- + +## Implementation notes + +| piece | role | +|---|---| +| `scripts/backfill_earnings_events.py` | bulk attempt → FMP/AV per-symbol; `earnings_events` + meta on snapshot | +| `scripts/run_earnings_research.py` | 2a trade join + 2b SUE IC / mom-conditional | +| Snapshot table `earnings_events` | real table (not SystemSetting JSON) | diff --git a/reports/sector-residual-20260719-083356.json b/reports/sector-residual-20260719-083356.json new file mode 100644 index 0000000..eecfc98 --- /dev/null +++ b/reports/sector-residual-20260719-083356.json @@ -0,0 +1,412 @@ +{ + "generated_at": "2026-07-19T08:33:56.651229", + "snapshot_guard": { + "snapshot": "C:\\Workspace\\signal-platform\\backtest_snapshots\\prod.sqlite", + "manifest": null, + "manifest_ok": null, + "note": "No completion manifest (prod.sqlite is expected without one). Bar-count sanity still applied.", + "ticker_count": 506, + "ohlcv_row_count": 629263, + "bars_min_avg_max": { + "min": 14, + "avg": 1246.1, + "max": 1261 + }, + "ohlcv_date_range": { + "min": "2021-06-24", + "max": "2026-07-02" + }, + "benchmark_prices": [ + { + "symbol": "SPY", + "n": 1516, + "min": "2020-07-06", + "max": "2026-07-17" + }, + { + "symbol": "XLB", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLC", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLE", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLF", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLI", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLK", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLP", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLRE", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLU", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLV", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + }, + { + "symbol": "XLY", + "n": 1512, + "min": "2020-07-10", + "max": "2026-07-17" + } + ], + "missing_sector_etfs": [] + }, + "sector_coverage": { + "universe": 506, + "mapped": 505, + "mapped_pct": 99.8, + "with_etf": 505, + "missing": [ + "RHM" + ], + "by_sector": { + "Industrials": 81, + "Financials": 75, + "Information Technology": 72, + "Health Care": 58, + "Consumer Discretionary": 47, + "Consumer Staples": 34, + "Real Estate": 31, + "Utilities": 31, + "Materials": 26, + "Communication Services": 23, + "Energy": 22, + "Consumer Defensive": 2, + "Technology": 2, + "Financial Services": 1 + } + }, + "sector_map_path": "C:\\Workspace\\signal-platform\\data\\research\\ticker_sector_map.json", + "signal_eval": [ + { + "signal": "vol_6m", + "weeks": 39, + "avg_cross_section": 498.2, + "mean_ic": 0.0609, + "ic_t_stat": 1.48, + "ic_positive_pct": 64.1, + "mean_quintile_spread": 0.0337, + "reliable": true + }, + { + "signal": "mom_12_1_sector_resid", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0578, + "ic_t_stat": 2.34, + "ic_positive_pct": 65.7, + "mean_quintile_spread": 0.0245, + "reliable": true + }, + { + "signal": "mom_12_1_resid", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0552, + "ic_t_stat": 1.98, + "ic_positive_pct": 60.0, + "mean_quintile_spread": 0.0207, + "reliable": true + }, + { + "signal": "mom_12_1", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0531, + "ic_t_stat": 1.61, + "ic_positive_pct": 65.7, + "mean_quintile_spread": 0.0206, + "reliable": true + }, + { + "signal": "mom_12_1_sector_demeaned", + "weeks": 35, + "avg_cross_section": 496.7, + "mean_ic": 0.034, + "ic_t_stat": 1.32, + "ic_positive_pct": 62.9, + "mean_quintile_spread": 0.0154, + "reliable": true + }, + { + "signal": "trend_200", + "weeks": 37, + "avg_cross_section": 497.9, + "mean_ic": 0.0161, + "ic_t_stat": 0.44, + "ic_positive_pct": 59.5, + "mean_quintile_spread": 0.006, + "reliable": true + }, + { + "signal": "reversal_1m", + "weeks": 43, + "avg_cross_section": 498.7, + "mean_ic": 0.0059, + "ic_t_stat": 0.22, + "ic_positive_pct": 53.5, + "mean_quintile_spread": 0.0053, + "reliable": true + }, + { + "signal": "mom_6_1", + "weeks": 39, + "avg_cross_section": 498.2, + "mean_ic": 0.0051, + "ic_t_stat": 0.21, + "ic_positive_pct": 56.4, + "mean_quintile_spread": 0.0087, + "reliable": true + }, + { + "signal": "mom_3_1", + "weeks": 42, + "avg_cross_section": 498.5, + "mean_ic": -0.0064, + "ic_t_stat": -0.25, + "ic_positive_pct": 50.0, + "mean_quintile_spread": 0.0046, + "reliable": true + }, + { + "signal": "high_52w", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": -0.0086, + "ic_t_stat": -0.26, + "ic_positive_pct": 54.3, + "mean_quintile_spread": -0.0088, + "reliable": true + }, + { + "signal": "fip_id", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": -0.045, + "ic_t_stat": -2.91, + "ic_positive_pct": 25.7, + "mean_quintile_spread": -0.0168, + "reliable": true + } + ], + "ic_grades": { + "mom_12_1_sector_resid": { + "promote_to_ab": true, + "checks": { + "sign_ok": true, + "abs_mean_ic_ge_0_03": true, + "reliable": true, + "t_ge_resid": true, + "mean_ic": 0.0578, + "ic_t_stat": 2.34, + "resid_ic_t_stat": 1.98, + "weeks": 35 + }, + "reason": "clears iron rule and t \u2265 mom_12_1_resid \u2014 authorized for A/B only", + "row": { + "signal": "mom_12_1_sector_resid", + "weeks": 35, + "avg_cross_section": 497.7, + "mean_ic": 0.0578, + "ic_t_stat": 2.34, + "ic_positive_pct": 65.7, + "mean_quintile_spread": 0.0245, + "reliable": true + } + }, + "mom_12_1_sector_demeaned": { + "promote_to_ab": false, + "checks": { + "sign_ok": true, + "abs_mean_ic_ge_0_03": true, + "reliable": true, + "t_ge_resid": false, + "mean_ic": 0.034, + "ic_t_stat": 1.32, + "resid_ic_t_stat": 1.98, + "weeks": 35 + }, + "reason": "does not clear pre-registered IC promotion bar", + "row": { + "signal": "mom_12_1_sector_demeaned", + "weeks": 35, + "avg_cross_section": 496.7, + "mean_ic": 0.034, + "ic_t_stat": 1.32, + "ic_positive_pct": 62.9, + "mean_quintile_spread": 0.0154, + "reliable": true + } + } + }, + "portfolio_ab": { + "signal": "mom_12_1_sector_resid", + "ranking_key": "residual_high_vol_blend_80_20_score", + "fill_mode": "close", + "validation_split": "2024-07-01", + "control": { + "label": "control_mom_12_1_resid", + "n_qualified_longs": 1086, + "windows": { + "train": { + "sharpe": 1.3, + "sharpe_se": 0.685, + "cagr_pct": 29.2, + "max_drawdown_pct": 21.4, + "total_return_pct": 70.9, + "trades": 176, + "win_rate_pct": null, + "avg_r": null, + "n_returns": 525, + "return_skew": 0.3722, + "return_kurtosis": 4.6208, + "psr": 0.971 + }, + "validation": { + "sharpe": 2.92, + "sharpe_se": 0.709, + "cagr_pct": 76.3, + "max_drawdown_pct": 11.7, + "total_return_pct": 210.7, + "trades": 150, + "win_rate_pct": null, + "avg_r": null, + "n_returns": 501, + "return_skew": 0.1734, + "return_kurtosis": 4.4625, + "psr": 1.0 + }, + "full": { + "sharpe": 2.09, + "sharpe_se": 0.497, + "cagr_pct": 51.6, + "max_drawdown_pct": 21.4, + "total_return_pct": 424.6, + "trades": 322, + "win_rate_pct": null, + "avg_r": null, + "n_returns": 1000, + "return_skew": 0.2686, + "return_kurtosis": 4.5653, + "psr": 1.0 + } + } + }, + "treatment": { + "label": "treatment_mom_12_1_sector_resid", + "n_qualified_longs": 1210, + "windows": { + "train": { + "sharpe": 1.57, + "sharpe_se": 0.677, + "cagr_pct": 35.5, + "max_drawdown_pct": 19.8, + "total_return_pct": 90.0, + "trades": 176, + "win_rate_pct": null, + "avg_r": null, + "n_returns": 530, + "return_skew": 0.466, + "return_kurtosis": 4.4413, + "psr": 0.99 + }, + "validation": { + "sharpe": 2.57, + "sharpe_se": 0.701, + "cagr_pct": 66.3, + "max_drawdown_pct": 14.8, + "total_return_pct": 176.4, + "trades": 163, + "win_rate_pct": null, + "avg_r": null, + "n_returns": 501, + "return_skew": 0.3003, + "return_kurtosis": 4.4305, + "psr": 0.9999 + }, + "full": { + "sharpe": 2.09, + "sharpe_se": 0.491, + "cagr_pct": 51.0, + "max_drawdown_pct": 19.8, + "total_return_pct": 421.3, + "trades": 337, + "win_rate_pct": null, + "avg_r": null, + "n_returns": 1005, + "return_skew": 0.4066, + "return_kurtosis": 4.3627, + "psr": 1.0 + } + } + }, + "promotion": { + "promote": true, + "checks": { + "validation_sharpe_ge_control_minus_half_se": true, + "full_sharpe_not_worse": true, + "full_maxdd_not_worse": true, + "control_validation_sharpe": 2.92, + "treatment_validation_sharpe": 2.57, + "se_used": 0.701, + "control_full_sharpe": 2.09, + "treatment_full_sharpe": 2.09, + "control_full_maxdd": 21.4, + "treatment_full_maxdd": 19.8 + }, + "reason": "clears pre-registered A/B bar \u2014 human decides wire-in" + } + }, + "verdict": "PROMOTE", + "verdict_detail": "mom_12_1_sector_resid cleared IC + A/B bars. Human must design wire-in; do not ship from this branch.", + "human_next": "- Approve or reject production residual swap vs dual-signal design.\n- If sector-cap arm ran, review tail-trim diagnostics before any cap.", + "report_path": "reports/sector-residual-20260719-083356.json", + "pre_registration": { + "iron_ic_bar": 0.03, + "validation_split": "2024-07-01", + "fill_mode": "close", + "cost_per_side": 0.001, + "ab_rule": "val Sharpe >= control - 0.5*SE; full Sharpe & maxDD not worse" + } +} diff --git a/reports/sector-residual-20260719-083356.md b/reports/sector-residual-20260719-083356.md new file mode 100644 index 0000000..90a15ee --- /dev/null +++ b/reports/sector-residual-20260719-083356.md @@ -0,0 +1,237 @@ +# Sector-residual momentum (Tier-1 alpha research) + +**Status:** **PROMOTE (to human design decision only)** — IC + A/B bars cleared; **do not ship**. +**Branch:** `research/sector-residual-momentum` +**Production impact:** none. Local research only. No scheduler / gate / prod-config changes. +**Artifacts:** `reports/sector-residual-20260719-083356.json` (+ companion `.md`) + +--- + +## Pre-registration (locked before first research run) + +### Hypothesis + +Residualizing 12–1 momentum against the sector, not only the market, reduces +factor volatility at similar return (Blitz / Huij / Martens-style) → higher +Sharpe on the production book when the residual replaces market-only residual +as the momentum leg. + +### Signals (candidates) + +| signal | construction | +|---|---| +| `mom_12_1_sector_resid` | Two-factor residual vs SPY + ticker’s sector ETF. Same window as `mom_12_1_resid`: ≥100 daily obs, 252-bar lookback, 21-bar skip; two-factor OLS betas **without intercept**; cumulate residual returns over the formation window. | +| `mom_12_1_sector_demeaned` | Plain `mom_12_1` minus the **cross-sectional** mean of `mom_12_1` within the same GICS sector that week (≥2 names in sector). No regression. | + +### Baselines (same run, same cross-sections — iron rule) + +Always report side-by-side with: + +- `mom_12_1` +- `mom_12_1_resid` + +Computed on the **identical** weekly non-overlapping cross-sections in this run. +Never compare against IC numbers from another report. + +### Iron rule (IC harness) + +Source of truth: `_signal_evaluation` in `app/services/backtest_service.py`. + +- Mean weekly Spearman IC on **non-overlapping** weekly windows +- Bar: \|mean IC\| ≥ ~0.03, **consistent positive sign**, `reliable: true` (≥ 12 windows) + +### Promotion to portfolio A/B (candidate → book) + +A candidate promotes to A/B **only if**: + +1. It clears the iron-rule bar **and** +2. Its IC **t-stat ≥** that of `mom_12_1_resid` on the same cross-sections. + +### Portfolio A/B grading (if and only if IC promotion fires) + +- Swap candidate in as the **momentum leg** of the production 80/20 momentum/vol + rank **and** as the gate-percentile signal. +- `fill_mode=close`, `COST_PER_SIDE = 0.001`, full config otherwise unchanged. +- Validation window = entries ≥ **2024-07-01** (call it **validation**, not + holdout — contaminated by prior experiments). +- Pre-registered promotion bar: + - validation Sharpe ≥ control − 0.5·SE + - full-period Sharpe and max-DD **not worse** than control +- Report Lo / Mertens-adjusted SEs. + +### Optional sector-cap sub-experiment + +Only if labels are in **and** A/B ran: max **3** positions per sector in the +10-slot book. Same A/B grading. **Tail-trim presumption of guilt** (rule 4): +report entry counts and both tails of the R distribution. Rising win rate with +falling Sharpe/CAGR = red flag → do not promote. + +**This run:** sector-cap arm **not executed** (optional; A/B unconstrained book +only). Can be a human-approved follow-up. + +### Verdict labels + +| label | meaning | +|---|---| +| **PROMOTE** | Clears pre-registered bar; human decides next (wire design separate) | +| **PARK** | Inconclusive / weak; keep machinery, no book change | +| **DEAD** | Failed iron rule or worse than residual baseline with clear sign | + +### Explicit non-goals + +- No production deploy from this doc +- Do not resurrect: take-profit exits, EV gate, regime entry-blocking, + inverse-vol sizing, gap-caps, unconditional FIP filter + +--- + +## Data provenance + +### Snapshot race guard + +| check | result | +|---|---| +| Snapshot path | `backtest_snapshots/prod.sqlite` | +| Manifest | none (expected for prod snapshot); bar-count sanity applied | +| Tickers / OHLCV | **506** / **629,263** | +| Bars min / avg / max | 14 / 1246.1 / 1261 | +| OHLCV range | 2021-06-24 → 2026-07-02 | +| Partial-build red flags | none (avg bars healthy) | + +Integrity fingerprint on same run: `fip_id` mean IC **−0.045** / t **−2.91** +(35 weeks, N≈498) — matches the established prod fingerprint. + +### Sector labels + +| source | count | +|---|---:| +| Public S&P 500 GICS CSV | 496 newly filled | +| FMP profile requests | 10 (all missing after CSV) | +| Mapped / universe | **505 / 506 (99.8%)** | +| With mappable ETF | 505 | +| Still missing | **RHM** only | + +Persist path: `data/research/ticker_sector_map.json`. + +FMP aliases (`Technology`, `Consumer Defensive`, `Financial Services`) map to +SPDRs via the alias table in `app/services/sector_map.py`. + +### Sector ETFs in `benchmark_prices` (auxiliary only — not tradable) + +| symbol | bars | min date | max date | +|---|---:|---|---| +| SPY | 1516 | 2020-07-06 | 2026-07-17 | +| XLB…XLY (11) | 1512 each | 2020-07-10 | 2026-07-17 | + +Fetched via Alpaca `Adjustment.SPLIT` into **`benchmark_prices`** (same table as +SPY) so they never enter the ticker universe or candidate replay. + +--- + +## Results + +Generated: `2026-07-19T08:33:56` + +### IC harness (identical cross-sections, production 506-name universe) + +| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable | ic+_pct | quintile spread | +|---|---:|---:|---:|---:|---|---:|---:| +| **mom_12_1_sector_resid** | **0.0578** | **2.34** | 35 | 497.7 | true | 65.7 | 0.0245 | +| mom_12_1_resid | 0.0552 | 1.98 | 35 | 497.7 | true | 60.0 | 0.0207 | +| mom_12_1 | 0.0531 | 1.61 | 35 | 497.7 | true | 65.7 | 0.0206 | +| mom_12_1_sector_demeaned | 0.0340 | 1.32 | 35 | 496.7 | true | 62.9 | 0.0154 | + +### IC promotion grades + +| candidate | iron rule | t ≥ resid | promote_to_ab | +|---|---|---|---| +| `mom_12_1_sector_resid` | pass (IC 0.058, +sign, reliable) | **yes** (2.34 ≥ 1.98) | **yes** | +| `mom_12_1_sector_demeaned` | pass (IC 0.034, +sign, reliable) | **no** (1.32 < 1.98) | **no** | + +### Portfolio A/B — `mom_12_1_sector_resid` as residual leg + +Config: production 80/20 residual/high-vol rank + gate percentile, `fill_mode=close`, +cost 10 bps/side, ATR trail / gate-reset re-entry as live. Validation split +2024-07-01. + +| window | arm | Sharpe | Sharpe SE (Mertens) | CAGR % | max DD % | trades | n_days | +|---|---|---:|---:|---:|---:|---:|---:| +| train | control (resid) | 1.30 | 0.685 | 29.2 | 21.4 | 176 | 525 | +| train | treatment (sector resid) | **1.57** | 0.677 | **35.5** | **19.8** | 176 | 530 | +| validation | control | **2.92** | 0.709 | **76.3** | **11.7** | 150 | 501 | +| validation | treatment | 2.57 | 0.701 | 66.3 | 14.8 | 163 | 501 | +| full | control | 2.09 | 0.497 | 51.6 | 21.4 | 322 | 1000 | +| full | treatment | 2.09 | 0.491 | 51.0 | **19.8** | 337 | 1005 | + +**Pre-registered A/B checks** + +| check | result | +|---|---| +| val Sharpe ≥ control − 0.5·SE | **pass** (2.57 ≥ 2.92 − 0.5×0.701 = 2.5695) — **knife-edge** | +| full Sharpe not worse | **pass** (2.09 = 2.09) | +| full max DD not worse | **pass** (19.8 < 21.4) | + +Qualified long candidates: control 1086 vs treatment 1210 (sector residual +gates a slightly larger set). + +--- + +## Verdict + +| signal | verdict | note | +|---|---|---| +| **`mom_12_1_sector_resid`** | **PROMOTE → human wire-in decision** | IC modestly beats market residual; A/B clears pre-reg bar narrowly. **Do not ship from this branch.** | +| **`mom_12_1_sector_demeaned`** | **DEAD** (for promotion) | Iron-rule IC magnitude ok, but t-stat loses to `mom_12_1_resid`. Cheap variant not competitive. | + +### Read carefully (for the human) + +1. **IC edge is real but small.** Sector residual IC 0.0578 / t 2.34 vs market + residual 0.0552 / t 1.98 on the **same** 35 windows — better consistency + (ic+ 65.7% vs 60%) and slightly higher mean, not a different factor class. +2. **A/B is not a clear Sharpe win.** Full-period Sharpe is flat (2.09). + Validation Sharpe is **lower** than control (2.57 vs 2.92) and only clears + the pre-registered “within 0.5 SE” cushion by ~0.001. Train improves; + validation worsens — classic regime-split noise on ~2 years. +3. **Risk side is friendly.** Full max DD improves (19.8% vs 21.4%); train DD + also better. Matches the “lower factor vol” half of the hypothesis more than + the “higher Sharpe” half on this window. +4. **Survivorship / short history.** Same caveats as all current research: + today’s constituents, ~35 independent weekly windows, one post-2021 regime + dominant. Task 3 (history depth) should re-check IC stability before any + wire-in. +5. **Not shipped.** Machinery lives on the research branch; production residual + path is untouched. + +--- + +## What a human must decide next + +1. **Accept or reject** replacing `mom_12_1_resid` with `mom_12_1_sector_resid` + as the production residual (gate + 80/20 mom leg), **or** keep market residual + and treat sector residual as research-only. +2. If leaning accept: require **Task 3 history-depth** confirmation (IC era split + pre/post-2021) before any production PR. +3. Optional: run **sector-cap ≤3** A/B with full tail diagnostics (not run here). +4. **Do not** merge this verdict into main strategy docs without review. +5. Wire-in design (live sector map refresh, ETF series ops, fallback when sector + missing) is a **separate** approved engineering step. + +--- + +## Implementation notes (research machinery) + +| piece | role | +|---|---| +| `app/services/sector_map.py` | GICS→ETF map, symbol normalise, JSON load/save | +| `app/services/backtest_service.py` | multi-factor residual; `mom_12_1_sector_resid` in `_signal_values`; demean inject | +| `scripts/build_ticker_sector_map.py` | SP500 CSV + FMP gap fill | +| `scripts/fetch_sector_etfs_to_snapshot.py` | Alpaca → snapshot `benchmark_prices` | +| `scripts/run_sector_residual_research.py` | race guard, IC, optional A/B, reports | +| `data/research/ticker_sector_map.json` | persisted labels (research only) | + +--- + +## Artifacts + +- JSON: `reports/sector-residual-20260719-083356.json` +- MD copy: `reports/sector-residual-20260719-083356.md` diff --git a/scripts/backfill_earnings_events.py b/scripts/backfill_earnings_events.py new file mode 100644 index 0000000..971d555 --- /dev/null +++ b/scripts/backfill_earnings_events.py @@ -0,0 +1,528 @@ +"""Backfill historical earnings into a snapshot ``earnings_events`` table. + +Prefers FMP bulk date-range ``earnings-calendar`` (one request per window). +On free-tier 402/403, falls back to per-symbol ``/stable/earnings`` with +resume support and request counting (≈250 req/day free tier). + +Research only — writes to the local snapshot SQLite, never production Postgres. + +Example +------- + python scripts/backfill_earnings_events.py \\ + --snapshot backtest_snapshots/prod.sqlite --limit 250 +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import sys +import time +from datetime import date, datetime, timedelta, timezone +from pathlib import Path + +import httpx +from sqlalchemy import create_engine, text + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +FMP_STABLE = "https://financialmodelingprep.com/stable" +DDL = """ +CREATE TABLE IF NOT EXISTS earnings_events ( + id INTEGER PRIMARY KEY, + symbol TEXT NOT NULL, + announce_date TEXT NOT NULL, + announce_time TEXT, + eps_estimate REAL, + eps_actual REAL, + revenue_estimate REAL, + revenue_actual REAL, + source TEXT NOT NULL, + fetched_at TEXT NOT NULL, + UNIQUE(symbol, announce_date) +) +""" +# Side table tracks which symbols have been fully pulled (resume). +META_DDL = """ +CREATE TABLE IF NOT EXISTS earnings_backfill_meta ( + symbol TEXT PRIMARY KEY, + status TEXT NOT NULL, + n_events INTEGER NOT NULL DEFAULT 0, + updated_at TEXT NOT NULL, + note TEXT +) +""" + + +def _parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__) + p.add_argument("--snapshot", default="backtest_snapshots/prod.sqlite") + p.add_argument( + "--from-date", + default="2020-01-01", + help="Bulk calendar window start (also filters per-symbol rows).", + ) + p.add_argument( + "--to-date", + default=None, + help="Bulk calendar window end (default: today).", + ) + p.add_argument( + "--limit", + type=int, + default=250, + help="Max FMP requests this run (free-tier cushion).", + ) + p.add_argument("--sleep", type=float, default=0.35) + p.add_argument( + "--force-symbol", + action="store_true", + help="Skip bulk attempt; go straight to per-symbol.", + ) + p.add_argument( + "--refetch-done", + action="store_true", + help="Re-fetch symbols already marked done.", + ) + p.add_argument( + "--provider", + choices=("fmp", "alpha_vantage", "auto"), + default="auto", + help="Earnings provider. auto tries FMP bulk then FMP/AV per-symbol.", + ) + return p.parse_args() + + +def _ensure_tables(engine) -> None: + with engine.begin() as conn: + conn.execute(text(DDL)) + conn.execute(text(META_DDL)) + + +def _upsert_events(conn, rows: list[dict], source: str) -> int: + if not rows: + return 0 + now = datetime.now(timezone.utc).isoformat() + written = 0 + for r in rows: + conn.execute( + text( + """ + INSERT INTO earnings_events ( + symbol, announce_date, announce_time, + eps_estimate, eps_actual, revenue_estimate, revenue_actual, + source, fetched_at + ) VALUES ( + :symbol, :announce_date, :announce_time, + :eps_estimate, :eps_actual, :revenue_estimate, :revenue_actual, + :source, :fetched_at + ) + ON CONFLICT(symbol, announce_date) DO UPDATE SET + announce_time=excluded.announce_time, + eps_estimate=excluded.eps_estimate, + eps_actual=excluded.eps_actual, + revenue_estimate=excluded.revenue_estimate, + revenue_actual=excluded.revenue_actual, + source=excluded.source, + fetched_at=excluded.fetched_at + """ + ), + { + "symbol": r["symbol"], + "announce_date": r["announce_date"], + "announce_time": r.get("announce_time"), + "eps_estimate": r.get("eps_estimate"), + "eps_actual": r.get("eps_actual"), + "revenue_estimate": r.get("revenue_estimate"), + "revenue_actual": r.get("revenue_actual"), + "source": source, + "fetched_at": now, + }, + ) + written += 1 + return written + + +def _parse_bulk_item(item: dict) -> dict | None: + sym = (item.get("symbol") or "").strip().upper() + d = item.get("date") or item.get("earningsDate") + if not sym or not d: + return None + return { + "symbol": sym.replace(".", "-"), + "announce_date": str(d)[:10], + "announce_time": item.get("time") or item.get("announceTime"), + "eps_estimate": _f(item.get("epsEstimated") or item.get("estimatedEarning")), + "eps_actual": _f(item.get("epsActual") or item.get("eps")), + "revenue_estimate": _f(item.get("revenueEstimated")), + "revenue_actual": _f(item.get("revenueActual")), + } + + +def _parse_symbol_item(item: dict, symbol: str) -> dict | None: + d = item.get("date") + if not d: + return None + return { + "symbol": symbol.replace(".", "-").upper(), + "announce_date": str(d)[:10], + "announce_time": item.get("time"), + "eps_estimate": _f(item.get("epsEstimated")), + "eps_actual": _f(item.get("epsActual")), + "revenue_estimate": _f(item.get("revenueEstimated")), + "revenue_actual": _f(item.get("revenueActual")), + } + + +def _f(v) -> float | None: + if v is None or v == "": + return None + try: + return float(v) + except (TypeError, ValueError): + return None + + +async def _try_bulk( + client: httpx.AsyncClient, + api_key: str, + start: date, + end: date, + *, + window_days: int = 30, +) -> tuple[list[dict], int, str | None]: + """Return (rows, requests_used, error_note).""" + rows: list[dict] = [] + reqs = 0 + cur = start + while cur <= end: + win_end = min(end, cur + timedelta(days=window_days - 1)) + resp = await client.get( + f"{FMP_STABLE}/earnings-calendar", + params={ + "from": cur.isoformat(), + "to": win_end.isoformat(), + "apikey": api_key, + }, + ) + reqs += 1 + if resp.status_code in (402, 403): + return [], reqs, f"bulk_unavailable status={resp.status_code}" + if resp.status_code == 429: + return rows, reqs, "rate_limited" + resp.raise_for_status() + data = resp.json() + if not isinstance(data, list): + return [], reqs, f"unexpected bulk payload type={type(data)}" + for item in data: + if isinstance(item, dict): + parsed = _parse_bulk_item(item) + if parsed: + rows.append(parsed) + cur = win_end + timedelta(days=1) + return rows, reqs, None + + +async def _fetch_symbol( + client: httpx.AsyncClient, api_key: str, symbol: str +) -> list[dict]: + resp = await client.get( + f"{FMP_STABLE}/earnings", + params={"symbol": symbol, "apikey": api_key}, + ) + if resp.status_code == 429: + raise RuntimeError("rate_limited") + if resp.status_code == 402: + return [] + resp.raise_for_status() + data = resp.json() + if not isinstance(data, list): + return [] + out: list[dict] = [] + for item in data: + if isinstance(item, dict): + parsed = _parse_symbol_item(item, symbol) + if parsed: + out.append(parsed) + return out + + +async def _fetch_symbol_alpha_vantage( + client: httpx.AsyncClient, api_key: str, symbol: str +) -> list[dict]: + """Alpha Vantage EARNINGS — includes reportedDate (announce) + estimate/actual.""" + resp = await client.get( + "https://www.alphavantage.co/query", + params={"function": "EARNINGS", "symbol": symbol, "apikey": api_key}, + ) + if resp.status_code == 429: + raise RuntimeError("rate_limited") + resp.raise_for_status() + data = resp.json() + if not isinstance(data, dict): + return [] + note = str(data.get("Note") or data.get("Information") or "") + if "rate limit" in note.lower() or "Thank you for using Alpha Vantage" in note: + raise RuntimeError("rate_limited") + if data.get("Error Message"): + return [] + quarterly = data.get("quarterlyEarnings") or [] + out: list[dict] = [] + for item in quarterly: + if not isinstance(item, dict): + continue + # Prefer announce (reportedDate); fall back to fiscal end (worse PIT). + ad = item.get("reportedDate") or item.get("fiscalDateEnding") + if not ad: + continue + out.append({ + "symbol": symbol.replace(".", "-").upper(), + "announce_date": str(ad)[:10], + "announce_time": item.get("reportTime"), + "eps_estimate": _f(item.get("estimatedEPS")), + "eps_actual": _f(item.get("reportedEPS")), + "revenue_estimate": None, + "revenue_actual": None, + }) + return out + + +async def _main() -> None: + args = _parse_args() + snapshot = Path(args.snapshot) + if not snapshot.exists(): + raise SystemExit(f"Snapshot not found: {snapshot}") + + from app.config import settings + + if not settings.fmp_api_key: + raise SystemExit("FMP_API_KEY required") + + start = date.fromisoformat(args.from_date) + end = date.fromisoformat(args.to_date) if args.to_date else date.today() + engine = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + _ensure_tables(engine) + + with engine.connect() as conn: + symbols = [ + str(r[0]).upper().replace(".", "-") + for r in conn.execute(text("SELECT symbol FROM tickers ORDER BY symbol")) + ] + done = set() + if not args.refetch_done: + done = { + str(r[0]) + for r in conn.execute( + text( + "SELECT symbol FROM earnings_backfill_meta " + "WHERE status='done' AND n_events > 0" + ) + ) + } + + pending = [s for s in symbols if s not in done] + print(f"Snapshot: {snapshot}") + print(f"Universe: {len(symbols)}; pending: {len(pending)}; done: {len(done)}") + print(f"Window filter: {start} → {end}") + print(f"Provider: {args.provider}") + + req_budget = int(args.limit) + reqs_used = 0 + events_written = 0 + mode = "per_symbol" + use_av = args.provider in ("alpha_vantage", "auto") and bool( + getattr(settings, "alpha_vantage_api_key", "") + ) + use_fmp = args.provider in ("fmp", "auto") and bool(settings.fmp_api_key) + + async with httpx.AsyncClient(timeout=60.0) as client: + if ( + not args.force_symbol + and req_budget > 0 + and use_fmp + and args.provider != "alpha_vantage" + ): + print("Attempting bulk earnings-calendar…") + bulk_rows, bulk_reqs, err = await _try_bulk( + client, settings.fmp_api_key, start, end + ) + reqs_used += bulk_reqs + if err: + print(f" Bulk unavailable: {err} (requests={bulk_reqs})") + else: + # Filter to universe. + uni = set(symbols) + bulk_rows = [r for r in bulk_rows if r["symbol"] in uni] + with engine.begin() as conn: + events_written += _upsert_events(conn, bulk_rows, "fmp_earnings_calendar") + for sym in symbols: + n = conn.execute( + text( + "SELECT COUNT(*) FROM earnings_events WHERE symbol=:s" + ), + {"s": sym}, + ).scalar_one() + conn.execute( + text( + """ + INSERT INTO earnings_backfill_meta(symbol, status, n_events, updated_at, note) + VALUES (:s, 'done', :n, :t, 'bulk') + ON CONFLICT(symbol) DO UPDATE SET + status='done', n_events=excluded.n_events, + updated_at=excluded.updated_at, note=excluded.note + """ + ), + { + "s": sym, + "n": int(n), + "t": datetime.now(timezone.utc).isoformat(), + }, + ) + mode = "bulk" + print(f" Bulk wrote {events_written} events; requests={bulk_reqs}") + pending = [] + + # Per-symbol fallback / completion. + fmp_limited = False + for sym in pending: + if reqs_used >= req_budget: + print(f"Request budget exhausted ({req_budget}). Resume later.") + break + items: list[dict] = [] + source = "fmp_earnings" + note = "per_symbol" + try: + if use_fmp and not fmp_limited and args.provider != "alpha_vantage": + items = await _fetch_symbol(client, settings.fmp_api_key, sym) + source = "fmp_earnings" + note = "fmp_per_symbol" + # Empty list may mean soft-limit or no data — try AV if available. + if not items and use_av: + items = await _fetch_symbol_alpha_vantage( + client, settings.alpha_vantage_api_key, sym + ) + source = "alpha_vantage_earnings" + note = "av_after_fmp_empty" + reqs_used += 1 # count AV call separately below too + elif use_av: + items = await _fetch_symbol_alpha_vantage( + client, settings.alpha_vantage_api_key, sym + ) + source = "alpha_vantage_earnings" + note = "av_per_symbol" + else: + raise RuntimeError("no provider available") + except Exception as exc: + msg = str(exc) + print(f" FAIL {sym}: {msg}") + reqs_used += 1 + if "rate_limited" in msg and note.startswith("fmp"): + fmp_limited = True + with engine.begin() as conn: + conn.execute( + text( + """ + INSERT INTO earnings_backfill_meta(symbol, status, n_events, updated_at, note) + VALUES (:s, 'error', 0, :t, :n) + ON CONFLICT(symbol) DO UPDATE SET + status='error', updated_at=excluded.updated_at, note=excluded.note + """ + ), + { + "s": sym, + "t": datetime.now(timezone.utc).isoformat(), + "n": msg[:200], + }, + ) + if args.sleep > 0: + await asyncio.sleep(args.sleep) + continue + + reqs_used += 1 + # Keep all rows with dates on/before end — SUE needs trailing history. + filtered = [ + r for r in items if r["announce_date"] <= end.isoformat() + ] + # Do NOT mark empty as done — leave pending for another provider/day. + status = "done" if filtered else "empty" + with engine.begin() as conn: + n_w = _upsert_events(conn, filtered, source) if filtered else 0 + events_written += n_w + conn.execute( + text( + """ + INSERT INTO earnings_backfill_meta(symbol, status, n_events, updated_at, note) + VALUES (:s, :st, :n, :t, :note) + ON CONFLICT(symbol) DO UPDATE SET + status=excluded.status, n_events=excluded.n_events, + updated_at=excluded.updated_at, note=excluded.note + """ + ), + { + "s": sym, + "st": status, + "n": len(filtered), + "t": datetime.now(timezone.utc).isoformat(), + "note": note, + }, + ) + if reqs_used % 10 == 0 or reqs_used == 1: + print( + f" progress reqs={reqs_used}/{req_budget} last={sym} " + f"events_batch={len(filtered)} src={source}" + ) + # AV free tier is ~5/min or 25/day — be polite when using it. + sleep_s = float(args.sleep) + if source.startswith("alpha_vantage"): + sleep_s = max(sleep_s, 12.0) + if sleep_s > 0: + await asyncio.sleep(sleep_s) + + with engine.connect() as conn: + total_events = int( + conn.execute(text("SELECT COUNT(*) FROM earnings_events")).scalar_one() + ) + done_n = int( + conn.execute( + text("SELECT COUNT(*) FROM earnings_backfill_meta WHERE status='done'") + ).scalar_one() + ) + d_range = conn.execute( + text("SELECT MIN(announce_date), MAX(announce_date) FROM earnings_events") + ).fetchone() + with_actual = int( + conn.execute( + text( + "SELECT COUNT(*) FROM earnings_events " + "WHERE eps_actual IS NOT NULL AND eps_estimate IS NOT NULL" + ) + ).scalar_one() + ) + + summary = { + "mode": mode, + "fmp_requests": reqs_used, + "events_written_this_run": events_written, + "total_events": total_events, + "symbols_done": done_n, + "symbols_universe": len(symbols), + "announce_date_range": {"min": d_range[0], "max": d_range[1]}, + "events_with_actual_and_estimate": with_actual, + "budget": req_budget, + "complete": done_n >= len(symbols), + } + print(json.dumps(summary, indent=2)) + out = Path("reports") / "earnings-backfill-status.json" + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(summary, indent=2) + "\n", encoding="utf-8") + print(f"Wrote {out}") + + +if __name__ == "__main__": + asyncio.run(_main()) diff --git a/scripts/build_ticker_sector_map.py b/scripts/build_ticker_sector_map.py new file mode 100644 index 0000000..f4a9ae9 --- /dev/null +++ b/scripts/build_ticker_sector_map.py @@ -0,0 +1,229 @@ +"""Build a local ticker → GICS sector map for research residualization. + +Sources (in order): +1. Public S&P 500 constituents CSV (datasets/s-and-p-500-companies) — bulk, free. +2. Existing map file (resume). +3. FMP stable ``profile`` for still-missing symbols (budget ~250 req/day). + +Writes ``data/research/ticker_sector_map.json``. Never touches production Postgres. + +Example +------- + python scripts/build_ticker_sector_map.py \\ + --snapshot backtest_snapshots/prod.sqlite + + python scripts/build_ticker_sector_map.py --fmp-limit 50 +""" + +from __future__ import annotations + +import argparse +import asyncio +import csv +import io +import json +import sys +import time +from datetime import datetime, timezone +from pathlib import Path + +import httpx +from sqlalchemy import create_engine, text + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from app.services.sector_map import ( # noqa: E402 + DEFAULT_SECTOR_MAP_PATH, + coverage_stats, + load_ticker_sector_map, + normalise_symbol, + save_ticker_sector_map, + sector_to_etf, +) + +SP500_CSV_URL = ( + "https://raw.githubusercontent.com/datasets/s-and-p-500-companies/" + "master/data/constituents.csv" +) +FMP_STABLE = "https://financialmodelingprep.com/stable" + + +def _parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__) + p.add_argument( + "--snapshot", + default="backtest_snapshots/prod.sqlite", + help="Snapshot whose tickers define the universe.", + ) + p.add_argument( + "--out", + default=str(DEFAULT_SECTOR_MAP_PATH), + help="Output JSON path.", + ) + p.add_argument( + "--fmp-limit", + type=int, + default=200, + help="Max FMP profile requests this run (free-tier cushion).", + ) + p.add_argument( + "--skip-fmp", + action="store_true", + help="Only use public SP500 CSV + existing map.", + ) + p.add_argument("--sleep", type=float, default=0.35, help="Pause between FMP calls.") + return p.parse_args() + + +def _snapshot_symbols(snapshot: Path) -> list[str]: + engine = create_engine(f"sqlite:///{snapshot.resolve().as_posix()}", future=True) + try: + with engine.connect() as conn: + rows = conn.execute(text("SELECT symbol FROM tickers ORDER BY symbol")).fetchall() + finally: + engine.dispose() + return [normalise_symbol(r[0]) for r in rows if r[0]] + + +def _fetch_sp500_map() -> dict[str, str]: + with httpx.Client(timeout=60.0, follow_redirects=True) as client: + resp = client.get(SP500_CSV_URL) + resp.raise_for_status() + reader = csv.DictReader(io.StringIO(resp.text)) + out: dict[str, str] = {} + for row in reader: + sym = normalise_symbol(row.get("Symbol") or "") + sector = (row.get("GICS Sector") or "").strip() + if sym and sector: + out[sym] = sector + return out + + +async def _fmp_profile_sector(client: httpx.AsyncClient, api_key: str, symbol: str) -> str | None: + resp = await client.get( + f"{FMP_STABLE}/profile", + params={"symbol": symbol, "apikey": api_key}, + ) + if resp.status_code == 429: + raise RuntimeError(f"FMP rate limited on {symbol}") + if resp.status_code == 402: + return None + resp.raise_for_status() + data = resp.json() + if isinstance(data, list): + data = data[0] if data else {} + if not isinstance(data, dict): + return None + sector = (data.get("sector") or data.get("industry") or "").strip() + # industry alone is not a GICS sector — only accept if we can map to an ETF + if sector and sector_to_etf(sector): + return sector + # FMP sometimes returns industry under sector when sector missing; try sector field only + sec = (data.get("sector") or "").strip() + return sec or None + + +async def _fill_from_fmp( + missing: list[str], + *, + api_key: str, + limit: int, + sleep_s: float, +) -> tuple[dict[str, str], int]: + filled: dict[str, str] = {} + used = 0 + async with httpx.AsyncClient(timeout=30.0) as client: + for sym in missing: + if used >= limit: + break + try: + sector = await _fmp_profile_sector(client, api_key, sym) + except Exception as exc: + print(f" FMP fail {sym}: {exc}") + used += 1 + await asyncio.sleep(sleep_s) + continue + used += 1 + if sector: + filled[sym] = sector + print(f" FMP {sym} → {sector}") + else: + print(f" FMP {sym} → (no sector)") + if sleep_s > 0: + await asyncio.sleep(sleep_s) + return filled, used + + +async def _main() -> None: + args = _parse_args() + snapshot = Path(args.snapshot) + if not snapshot.exists(): + raise SystemExit(f"Snapshot not found: {snapshot}") + + symbols = _snapshot_symbols(snapshot) + print(f"Universe: {len(symbols)} symbols from {snapshot}") + + existing = load_ticker_sector_map(args.out) + print(f"Existing map entries: {len(existing)}") + + print("Fetching public S&P 500 sector CSV…") + sp500 = _fetch_sp500_map() + print(f" SP500 CSV rows: {len(sp500)}") + + mapping = dict(existing) + from_sp500 = 0 + for sym in symbols: + if sym in mapping: + continue + if sym in sp500: + mapping[sym] = sp500[sym] + from_sp500 += 1 + print(f" Newly filled from SP500 CSV: {from_sp500}") + + missing = [s for s in symbols if s not in mapping] + fmp_used = 0 + from_fmp = 0 + if missing and not args.skip_fmp: + from app.config import settings + + if not settings.fmp_api_key: + print("WARNING: FMP key missing; leaving gaps unfilled") + else: + print(f"FMP fill for {len(missing)} missing (limit={args.fmp_limit})…") + filled, fmp_used = await _fill_from_fmp( + missing, + api_key=settings.fmp_api_key, + limit=int(args.fmp_limit), + sleep_s=float(args.sleep), + ) + mapping.update(filled) + from_fmp = len(filled) + + still_missing = [s for s in symbols if s not in mapping] + stats = coverage_stats(symbols, mapping) + meta = { + "built_at": datetime.now(timezone.utc).isoformat(), + "snapshot": str(snapshot.resolve()), + "from_existing": len(existing), + "from_sp500_csv": from_sp500, + "from_fmp": from_fmp, + "fmp_requests": fmp_used, + "still_missing": still_missing, + "coverage": { + k: stats[k] + for k in ("universe", "mapped", "mapped_pct", "with_etf", "by_sector") + }, + } + out_path = save_ticker_sector_map(mapping, args.out, meta=meta) + print(f"Wrote {out_path}") + print(json.dumps(meta["coverage"], indent=2)) + if still_missing: + print(f"Still missing ({len(still_missing)}): {still_missing[:40]}") + if len(still_missing) > 40: + print(f" … +{len(still_missing) - 40} more") + + +if __name__ == "__main__": + asyncio.run(_main()) diff --git a/scripts/fetch_sector_etfs_to_snapshot.py b/scripts/fetch_sector_etfs_to_snapshot.py new file mode 100644 index 0000000..db558e0 --- /dev/null +++ b/scripts/fetch_sector_etfs_to_snapshot.py @@ -0,0 +1,183 @@ +"""Fetch the 11 SPDR sector ETFs into a snapshot's ``benchmark_prices``. + +Research-only. Sector ETFs are auxiliary series (like SPY) — they must not +enter the tradable ticker universe or candidate replay. Storing them in +``benchmark_prices`` keeps that invariant. + +Also refreshes SPY on the same window so residual factors share a calendar. + +Example +------- + python scripts/fetch_sector_etfs_to_snapshot.py \\ + --snapshot backtest_snapshots/prod.sqlite --history-days 2200 +""" + +from __future__ import annotations + +import argparse +import asyncio +import sys +import time +from datetime import date, timedelta +from pathlib import Path + +from sqlalchemy import create_engine, text + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from app.services.sector_map import SECTOR_ETFS # noqa: E402 + + +def _parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__) + p.add_argument("--snapshot", default="backtest_snapshots/prod.sqlite") + p.add_argument( + "--history-days", + type=int, + default=2200, + help="Lookback calendar days (default ~6y; covers 5y snapshot + cushion).", + ) + p.add_argument("--sleep", type=float, default=0.25) + p.add_argument( + "--symbols", + default=None, + help="Comma-separated override (default: SPY + 11 sector ETFs).", + ) + return p.parse_args() + + +async def _fetch_and_upsert( + engine, + provider, + symbol: str, + start: date, + end: date, + *, + sleep_s: float, +) -> int: + from app.exceptions import ProviderError, RateLimitError + + for attempt in range(5): + try: + bars = await provider.fetch_ohlcv(symbol, start, end) + break + except RateLimitError: + wait = min(60.0, 2.0 ** attempt) + print(f" rate limited {symbol}; sleep {wait:.0f}s") + await asyncio.sleep(wait) + bars = [] + except ProviderError as exc: + if attempt + 1 >= 5: + raise + await asyncio.sleep(1.0) + print(f" retry {symbol}: {exc}") + bars = [] + else: + bars = [] + + if sleep_s > 0: + await asyncio.sleep(sleep_s) + + if not bars: + print(f" {symbol}: empty") + return 0 + + written = 0 + with engine.begin() as conn: + for bar in bars: + d = bar.date.isoformat() if hasattr(bar.date, "isoformat") else str(bar.date) + close = float(bar.close) + existing = conn.execute( + text( + "SELECT id, close FROM benchmark_prices " + "WHERE symbol = :sym AND date = :d" + ), + {"sym": symbol, "d": d}, + ).fetchone() + if existing is None: + # id is INTEGER PK — let sqlite autoincrement if possible + conn.execute( + text( + "INSERT INTO benchmark_prices (symbol, date, close) " + "VALUES (:sym, :d, :c)" + ), + {"sym": symbol, "d": d, "c": close}, + ) + written += 1 + elif abs(float(existing[1]) - close) > 1e-9: + conn.execute( + text( + "UPDATE benchmark_prices SET close = :c WHERE id = :id" + ), + {"c": close, "id": int(existing[0])}, + ) + written += 1 + print(f" {symbol}: {len(bars)} bars, {written} rows written/updated") + return written + + +async def _main() -> None: + args = _parse_args() + snapshot = Path(args.snapshot) + if not snapshot.exists(): + raise SystemExit(f"Snapshot not found: {snapshot}") + + from app.config import settings + from app.providers.alpaca import AlpacaOHLCVProvider + + if not settings.alpaca_api_key or not settings.alpaca_api_secret: + raise SystemExit("ALPACA_API_KEY / ALPACA_API_SECRET required") + + if args.symbols: + symbols = [s.strip().upper() for s in args.symbols.split(",") if s.strip()] + else: + symbols = ["SPY", *SECTOR_ETFS] + + end = date.today() + start = end - timedelta(days=int(args.history_days)) + provider = AlpacaOHLCVProvider(settings.alpaca_api_key, settings.alpaca_api_secret) + engine = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + + print(f"Snapshot: {snapshot}") + print(f"Window: {start} → {end}") + print(f"Symbols: {symbols}") + + t0 = time.monotonic() + total = 0 + try: + for sym in symbols: + n = await _fetch_and_upsert( + engine, provider, sym, start, end, sleep_s=float(args.sleep) + ) + total += n + finally: + engine.dispose() + + # Summary counts + engine = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + try: + with engine.connect() as conn: + rows = conn.execute( + text( + "SELECT symbol, COUNT(*), MIN(date), MAX(date) " + "FROM benchmark_prices GROUP BY symbol ORDER BY symbol" + ) + ).fetchall() + finally: + engine.dispose() + + print(f"Done in {(time.monotonic() - t0) / 60:.1f}m; rows touched={total}") + for sym, n, d0, d1 in rows: + print(f" {sym}: n={n} {d0}→{d1}") + + +if __name__ == "__main__": + asyncio.run(_main()) diff --git a/scripts/run_earnings_research.py b/scripts/run_earnings_research.py new file mode 100644 index 0000000..20c695f --- /dev/null +++ b/scripts/run_earnings_research.py @@ -0,0 +1,927 @@ +"""Earnings gap diagnostic (2a) + SUE IC (2b). Local research only. + +Requires ``earnings_events`` on the snapshot (see backfill_earnings_events.py). + +Example +------- + python scripts/run_earnings_research.py \\ + --snapshot backtest_snapshots/prod.sqlite --workers 6 --allow-spawn +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import math +import os +import sys +from collections import defaultdict +from datetime import date, datetime, timedelta +from pathlib import Path +from typing import Any + +from sqlalchemy import create_engine, text +from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +IRON_IC_BAR = 0.03 +MIN_RELIABLE = 12 +SUE_CARRY_DAYS = 63 +SUE_TRAIL = 8 + + +def _sqlite_url(path: Path) -> str: + return f"sqlite+aiosqlite:///{path.resolve().as_posix()}" + + +def _parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__) + p.add_argument("--snapshot", default="backtest_snapshots/prod.sqlite") + p.add_argument("--workers", type=int, default=6) + p.add_argument("--allow-spawn", action="store_true") + p.add_argument("--skip-2a", action="store_true") + p.add_argument("--skip-2b", action="store_true") + p.add_argument("--quiet", action="store_true") + p.add_argument("--out", default=None) + return p.parse_args() + + +def _load_earnings(snapshot: Path) -> list[dict]: + engine = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + try: + with engine.connect() as conn: + # Table must exist. + tables = { + r[0] + for r in conn.execute( + text("SELECT name FROM sqlite_master WHERE type='table'") + ) + } + if "earnings_events" not in tables: + raise SystemExit( + "earnings_events table missing — run scripts/backfill_earnings_events.py" + ) + rows = conn.execute( + text( + """ + SELECT symbol, announce_date, announce_time, + eps_estimate, eps_actual, revenue_estimate, revenue_actual + FROM earnings_events + ORDER BY symbol, announce_date + """ + ) + ).fetchall() + meta = {} + if "earnings_backfill_meta" in tables: + meta = { + "done": int( + conn.execute( + text( + "SELECT COUNT(*) FROM earnings_backfill_meta " + "WHERE status='done'" + ) + ).scalar_one() + ), + "universe_tickers": int( + conn.execute(text("SELECT COUNT(*) FROM tickers")).scalar_one() + ), + } + finally: + engine.dispose() + + events = [ + { + "symbol": str(r[0]).upper(), + "announce_date": date.fromisoformat(str(r[1])[:10]), + "announce_time": r[2], + "eps_estimate": r[3], + "eps_actual": r[4], + "revenue_estimate": r[5], + "revenue_actual": r[6], + } + for r in rows + ] + return events, meta + + +def _percentile(xs: list[float], q: float) -> float | None: + if not xs: + return None + s = sorted(xs) + if len(s) == 1: + return s[0] + idx = q * (len(s) - 1) + lo = int(math.floor(idx)) + hi = int(math.ceil(idx)) + if lo == hi: + return s[lo] + w = idx - lo + return s[lo] * (1 - w) + s[hi] * w + + +def _r_dist(rs: list[float]) -> dict[str, Any]: + if not rs: + return {"n": 0} + return { + "n": len(rs), + "mean": round(sum(rs) / len(rs), 4), + "win_rate": round(sum(1 for r in rs if r > 0) / len(rs), 4), + "p05": round(_percentile(rs, 0.05), 4), + "p25": round(_percentile(rs, 0.25), 4), + "p50": round(_percentile(rs, 0.50), 4), + "p75": round(_percentile(rs, 0.75), 4), + "p95": round(_percentile(rs, 0.95), 4), + "min": round(min(rs), 4), + "max": round(max(rs), 4), + } + + +def _trading_days_between( + entry: date, exit_: date, calendar: set[date] +) -> list[date]: + """Inclusive trading dates in [entry, exit_] present on the union calendar.""" + out = [] + d = entry + while d <= exit_: + if d in calendar: + out.append(d) + d += timedelta(days=1) + return out + + +def _nth_trading_day_after( + start: date, n: int, ordered_calendar: list[date] +) -> date | None: + """First calendar date strictly after ``start``, then + (n-1) more sessions. + + announce+1 trading day: n=1 → first session after announce date + (if announce is a trading day, still use the *next* session for PIT). + """ + # Sessions strictly after start. + after = [d for d in ordered_calendar if d > start] + if len(after) < n: + return None + return after[n - 1] + + +def _build_sue_series( + events_by_symbol: dict[str, list[dict]], + prices: dict[str, tuple], +) -> dict[str, dict[date, float]]: + """symbol → {asof_date: sue_value} for days when SUE is live (announce+1 .. +63).""" + out: dict[str, dict[date, float]] = {} + for sym, cols in prices.items(): + ords = cols[0] + closes = cols[4] + dates = [date.fromordinal(int(o)) for o in ords] + if not dates: + continue + ordered = dates # already chronological + cal_set = set(ordered) + events = events_by_symbol.get(sym.upper(), []) + # Chronological surprises with actual+estimate. + surprises: list[tuple[date, float, float]] = [] # announce, surprise, close_for_scale + for ev in events: + act, est = ev.get("eps_actual"), ev.get("eps_estimate") + if act is None or est is None: + continue + ad = ev["announce_date"] + # Close on/before announce for price fallback scale. + close_px = None + for d, c in zip(reversed(dates), reversed(closes)): + if d <= ad and float(c) > 0: + close_px = float(c) + break + surprises.append((ad, float(act) - float(est), close_px or 1.0)) + surprises.sort(key=lambda x: x[0]) + + sue_on_day: dict[date, float] = {} + for i, (ad, surprise, px) in enumerate(surprises): + trail = [surprises[j][1] for j in range(max(0, i - SUE_TRAIL), i)] + # Need history of surprises; include current only for value, stdev from prior 8. + if len(trail) >= 3: + mean_t = sum(trail) / len(trail) + var = sum((x - mean_t) ** 2 for x in trail) / (len(trail) - 1) + sd = math.sqrt(var) if var > 0 else None + else: + sd = None + if sd is not None and sd > 1e-9: + sue = surprise / sd + else: + # Fallback: scale by price (EPS surprise / price). + sue = surprise / px if px > 0 else None + if sue is None or not math.isfinite(sue): + continue + usable_from = _nth_trading_day_after(ad, 1, ordered) + if usable_from is None: + continue + # Carry for SUE_CARRY_DAYS trading sessions starting at usable_from. + try: + start_idx = ordered.index(usable_from) + except ValueError: + # usable_from not in this symbol's calendar (halted etc.) + start_idx = next( + (k for k, d in enumerate(ordered) if d >= usable_from), None + ) + if start_idx is None: + continue + end_idx = min(len(ordered) - 1, start_idx + SUE_CARRY_DAYS - 1) + for k in range(start_idx, end_idx + 1): + # Later announcements overwrite earlier carry (latest SUE wins). + sue_on_day[ordered[k]] = sue + if sue_on_day: + out[sym.upper()] = sue_on_day + return out + + +async def _run_2a( + snapshot: Path, + events: list[dict], + *, + quiet: bool, + workers: int, +) -> dict[str, Any]: + from app.config import settings + from app.services import backtest_service as bt + from app.services.admin_service import get_activation_config + from app.services.recommendation_service import get_recommendation_config + from app.services.paper_trade_service import get_exit_policy + from app.services.benchmark_service import load_benchmark_closes + from app.models.ticker import Ticker + from sqlalchemy import select + + os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1" + settings.backtest_workers = workers + + engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True) + Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False) + + try: + async with Session() as db: + config = await get_recommendation_config(db) + activation = await get_activation_config(db) + exit_config = await get_exit_policy(db) + tickers = list( + (await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars() + ) + spy = await load_benchmark_closes(db, "SPY") + prices: dict[str, tuple] = {} + candidates: list[dict] = [] + for idx, t in enumerate(tickers): + if not quiet and idx % 50 == 0: + print(f" 2a fetch {idx}/{len(tickers)}", end="\r", flush=True) + cols = await bt._fetch_columns(db, t.symbol) + if cols is None: + continue + prices[t.symbol] = cols + cands, _ = bt._replay_and_signals( + t.symbol, + cols, + config, + activation, + spy, + bt.PRODUCTION_GTL_TARGET_MODEL, + "weekly", + False, + ) + candidates.extend(cands) + finally: + await engine.dispose() + if not quiet: + print() + + # Production ranks + qualify. + bt._assign_momentum_percentiles(candidates) + bt._assign_residual_momentum_percentiles(candidates) + bt._assign_low_volatility_percentiles(candidates) + bt._assign_activation_momentum_percentiles(candidates) + bt._assign_residual_high_vol_blend(candidates) + for c in candidates: + c["qualified"] = bt._momentum_qualifies(c, 80.0) + longs = [ + c for c in candidates if c.get("qualified") and c.get("direction") == "long" + ] + + strategy = next(s for s in bt.PORTFOLIO_MONITOR_STRATEGIES if s.get("is_production")) + entry_cfg = bt._entry_variant_config(str(strategy["entry_variant"])) + assert entry_cfg is not None + ranking_key = str(entry_cfg.get("ranking_key") or entry_cfg["percentile_key"]) + exit_policy = bt.LIVE_EXIT_MODE_TO_SIM.get( + str(exit_config.get("mode", "atr_trailing")), "atr_trail3" + ) + hold_days = int(exit_config.get("hold_days", 30)) + trail = float(exit_config.get("atr_multiplier", bt.ATR_TRAIL_MULTIPLIER)) + reentry = bt._make_gate_reset_reentry_fn( + longs, prices, cadence="weekly", ranking_key=ranking_key + ) + sim = bt._simulate_portfolio( + longs, + prices, + spy, + exit_policy, + hold_days, + ranking_key=ranking_key, + max_positions=int(entry_cfg["max_positions"]), + risk_per_trade=float(entry_cfg["risk_per_trade"]), + atr_trail_multiplier=trail, + post_stop_reentry_fn=reentry, + fill_mode=bt.FILL_MODE_CLOSE, + include_trades=True, + ) + if sim is None: + return {"error": "no_trades"} + + details = sim.get("trade_details") or [] + # Build per-symbol earnings announce dates. + earns_by_sym: dict[str, list[date]] = defaultdict(list) + for ev in events: + earns_by_sym[ev["symbol"]].append(ev["announce_date"]) + for sym in earns_by_sym: + earns_by_sym[sym].sort() + + # Union trading calendar from prices. + cal: set[date] = set() + for cols in prices.values(): + for o in cols[0]: + cal.add(date.fromordinal(int(o))) + ordered_cal = sorted(cal) + + # Map entry date → list of announce dates for symbol (for pre-entry lookback). + trades_parsed: list[dict] = [] + for t in details: + sym = str(t.get("symbol") or "").upper() + # Field names from simulator. + entry_s = t.get("entry_date") or t.get("open_date") or t.get("date") + exit_s = t.get("exit_date") or t.get("close_date") + r = t.get("realized_r") + if r is None: + r = t.get("r") + if entry_s is None or exit_s is None or r is None: + continue + entry_d = date.fromisoformat(str(entry_s)[:10]) + exit_d = date.fromisoformat(str(exit_s)[:10]) + announces = earns_by_sym.get(sym, []) + # Earnings between entry and exit (exclusive of entry day? inclusive hold). + # "between entry and exit" — any announce with entry < announce <= exit + # (gap often overnight after entry). Also count announce on entry day. + in_hold = [ + a for a in announces if entry_d <= a <= exit_d + ] + # Entries within 3 trading days BEFORE an announcement: + # exists announce such that entry is in the 3 sessions immediately before announce. + pre_earn = False + for a in announces: + # trading sessions in (a-lookback, a) + sessions_before = [d for d in ordered_cal if d < a] + last3 = sessions_before[-3:] if len(sessions_before) >= 3 else sessions_before + if entry_d in last3: + pre_earn = True + break + trades_parsed.append({ + "symbol": sym, + "entry": entry_d.isoformat(), + "exit": exit_d.isoformat(), + "r": float(r), + "earnings_in_hold": len(in_hold) > 0, + "n_earnings_in_hold": len(in_hold), + "entry_within_3d_before_earn": pre_earn, + }) + + all_r = [t["r"] for t in trades_parsed] + loss_lt_1r = [t for t in trades_parsed if t["r"] < -1.0] + loss_with_earn = [t for t in loss_lt_1r if t["earnings_in_hold"]] + pre = [t["r"] for t in trades_parsed if t["entry_within_3d_before_earn"]] + other = [t["r"] for t in trades_parsed if not t["entry_within_3d_before_earn"]] + + return { + "sim_summary": { + k: sim.get(k) + for k in ( + "sharpe", + "sharpe_se", + "cagr_pct", + "max_drawdown_pct", + "trades", + "total_return_pct", + ) + }, + "n_trades_parsed": len(trades_parsed), + "q1_losses_worse_than_minus_1r": { + "n_losses_lt_minus_1r": len(loss_lt_1r), + "n_with_earnings_in_hold": len(loss_with_earn), + "fraction_with_earnings": ( + round(len(loss_with_earn) / len(loss_lt_1r), 4) if loss_lt_1r else None + ), + "all_trades_with_earnings_in_hold": sum( + 1 for t in trades_parsed if t["earnings_in_hold"] + ), + "fraction_all_trades_with_earnings": ( + round( + sum(1 for t in trades_parsed if t["earnings_in_hold"]) + / len(trades_parsed), + 4, + ) + if trades_parsed + else None + ), + }, + "q2_entry_within_3d_before_announce": { + "pre_earn_entries": _r_dist(pre), + "other_entries": _r_dist(other), + "all_entries": _r_dist(all_r), + "tail_trim_note": ( + "Compare p95/max and mean of pre_earn vs other. " + "Rising win_rate with falling mean/p95 = right-tail trim red flag." + ), + }, + "note": "REPORT-ONLY — no filter shipped.", + } + + +async def _run_2b_ic( + snapshot: Path, + events: list[dict], + *, + quiet: bool, + workers: int, +) -> dict[str, Any]: + """SUE IC via harness on identical cross-sections as momentum baselines.""" + from app.config import settings + from app.services import backtest_service as bt + from app.services.benchmark_service import load_benchmark_closes + from app.models.ticker import Ticker + from sqlalchemy import select + from collections import defaultdict as dd + + os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1" + os.environ["BACKTEST_SIGNAL_EVAL_ONLY"] = "1" + # Load sector map if present so sector signals also appear (side-by-side optional). + if Path("data/research/ticker_sector_map.json").exists(): + os.environ["BACKTEST_SECTOR_MAP_PATH"] = str( + Path("data/research/ticker_sector_map.json").resolve() + ) + settings.backtest_workers = workers + + engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True) + Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False) + + # Collect base signals + attach SUE. + collected: dict = dd(lambda: dd(list)) + try: + async with Session() as db: + tickers = list( + (await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars() + ) + spy = await load_benchmark_closes(db, "SPY") + sector_etf: dict[str, dict] = {} + try: + from app.services.sector_map import SECTOR_ETFS, load_ticker_sector_map + + symbol_to_sector = load_ticker_sector_map() + for etf in SECTOR_ETFS: + series = await load_benchmark_closes(db, etf) + if series: + sector_etf[etf] = series + except Exception: + symbol_to_sector = {} + sector_etf = {} + + prices: dict[str, tuple] = {} + for idx, t in enumerate(tickers): + if not quiet and idx % 50 == 0: + print(f" 2b fetch {idx}/{len(tickers)}", end="\r", flush=True) + cols = await bt._fetch_columns(db, t.symbol) + if cols is None: + continue + prices[t.symbol] = cols + series = bt._signal_series( + [ + type( + "R", + (), + { + "date": date.fromordinal(int(cols[0][i])), + "close": cols[4][i], + "high": cols[2][i], + "volume": cols[5][i] if len(cols) > 5 else 0, + }, + )() + for i in range(len(cols[0])) + ], + spy, + symbol=t.symbol, + sector_etf_closes=bt._sector_etf_closes_for_symbol( + t.symbol, symbol_to_sector, sector_etf + ), + ) + for name, weeks in series.items(): + for wk, pairs in weeks.items(): + collected[name][wk].extend(pairs) + finally: + await engine.dispose() + if not quiet: + print() + + if symbol_to_sector: + bt._inject_sector_demeaned_momentum(collected, symbol_to_sector) + + # SUE series. + events_by_sym: dict[str, list[dict]] = defaultdict(list) + for ev in events: + events_by_sym[ev["symbol"]].append(ev) + sue_map = _build_sue_series(events_by_sym, prices) + + # Inject sue_latest into collected using mom_12_1 observations as the + # weekly as-of skeleton (same weeks / symbols). + sue_collected: dict = dd(list) + mom_weeks = collected.get("mom_12_1") or {} + for week_key, recs in mom_weeks.items(): + for rec in recs: + pair = bt._obs_val_fwd(rec) + if pair is None: + continue + _val, fwd = pair + sym = None + if isinstance(rec, dict): + sym = rec.get("symbol") + if not sym: + continue + # Need as-of date: recover from week — use Friday of ISO week as proxy + # is weak. Better: re-derive from prices weekly indices. + # Store asof on rich recs? Current rich rows lack asof date. + # Fall back: compute SUE observations directly from prices weekly as-ofs. + pass + + # Direct weekly as-of SUE + forward return (authoritative). + for sym, cols in prices.items(): + ords, _o, highs, _l, closes, _v = cols + dates = [date.fromordinal(int(o)) for o in ords] + sue_days = sue_map.get(sym.upper()) or {} + if not sue_days: + continue + n = len(dates) + # weekly as-of indices: reuse harness helper via fake records. + records = [ + type("R", (), {"date": dates[i], "close": closes[i], "high": highs[i]})() + for i in range(n) + ] + for i in bt._weekly_asof_indices(records): + j = i + bt.HORIZON + if j >= n or closes[i] <= 0: + continue + asof = dates[i] + sue = sue_days.get(asof) + if sue is None: + continue + fwd = float(closes[j]) / float(closes[i]) - 1.0 + iso = asof.isocalendar() + week_key = (iso[0], iso[1]) + # Also grab mom for conditional. + mom = None + if i >= 252 and closes[i - 252] > 0: + mom = float(closes[i - 21]) / float(closes[i - 252]) - 1.0 + sue_collected[week_key].append({ + "val": float(sue), + "fwd": fwd, + "symbol": sym, + "mom_12_1": mom, + }) + collected["sue_latest"] = sue_collected + + signal_eval = bt._signal_evaluation(collected) + + # Fair side-by-side: re-evaluate mom baselines on the *same* (symbol, week) + # observations where SUE is present (incomplete backfill otherwise inflates + # mom N relative to SUE). + sue_pairs_by_week = sue_collected + restricted: dict = dd(lambda: dd(list)) + for week_key, recs in sue_pairs_by_week.items(): + syms = {str(r.get("symbol")).upper() for r in recs if r.get("symbol")} + for base_name in ("mom_12_1", "mom_12_1_resid"): + base_recs = (collected.get(base_name) or {}).get(week_key) or [] + for rec in base_recs: + pair = bt._obs_val_fwd(rec) + if pair is None: + continue + sym = None + if isinstance(rec, dict): + sym = rec.get("symbol") + if not sym or str(sym).upper() not in syms: + continue + restricted[base_name][week_key].append(rec) + restricted["sue_latest"][week_key].extend(recs) + restricted_eval = bt._signal_evaluation(restricted) + + # Momentum-conditional: IC of SUE within top mom quintile each week. + cond_ics: list[float] = [] + stride = max(1, round(bt.HORIZON / 5)) + usable = [wk for wk, recs in sue_collected.items() if len(recs) >= bt.MIN_CROSS_SECTION] + kept = bt._nonoverlapping_weeks(usable, stride) + for wk in kept: + recs = sue_collected[wk] + with_mom = [r for r in recs if r.get("mom_12_1") is not None] + if len(with_mom) < bt.MIN_CROSS_SECTION: + continue + ordered = sorted(with_mom, key=lambda r: float(r["mom_12_1"])) + k = max(1, len(ordered) // 5) + top = ordered[-k:] + if len(top) < 5: + continue + ic = bt._spearman( + [float(r["val"]) for r in top], + [float(r["fwd"]) for r in top], + ) + if ic is not None: + cond_ics.append(ic) + if cond_ics: + mean_c = sum(cond_ics) / len(cond_ics) + if len(cond_ics) > 1: + std = math.sqrt( + sum((x - mean_c) ** 2 for x in cond_ics) / (len(cond_ics) - 1) + ) + t_c = mean_c / std * math.sqrt(len(cond_ics)) if std > 0 else None + else: + t_c = None + mom_cond = { + "mean_ic": round(mean_c, 4), + "ic_t_stat": round(t_c, 2) if t_c is not None else None, + "weeks": len(cond_ics), + "note": "IC of sue_latest within top mom_12_1 quintile (non-overlapping weeks)", + } + else: + mom_cond = {"mean_ic": None, "weeks": 0} + + def _find(name: str) -> dict | None: + for row in signal_eval: + if row.get("signal") == name: + return row + return None + + sue = _find("sue_latest") + grade = { + "green": False, + "reason": "sue_latest missing", + } + if sue: + mean_ic = sue.get("mean_ic") + t = sue.get("ic_t_stat") + reliable = bool(sue.get("reliable")) + sign_ok = mean_ic is not None and float(mean_ic) > 0 + mag_ok = mean_ic is not None and abs(float(mean_ic)) >= IRON_IC_BAR + grade = { + "green": bool(sign_ok and mag_ok and reliable), + "checks": { + "mean_ic": mean_ic, + "sign_positive": sign_ok, + "abs_ge_0_03": mag_ok, + "reliable": reliable, + "ic_t_stat": t, + "weeks": sue.get("weeks"), + }, + "reason": ( + "iron rule cleared — STOP; book-integration is a separate human step" + if (sign_ok and mag_ok and reliable) + else "iron rule not met" + ), + "row": sue, + } + + def _find_r(name: str) -> dict | None: + for row in restricted_eval: + if row.get("signal") == name: + return row + return None + + # Side-by-side baselines from same evaluation. + side = { + name: _find(name) + for name in ( + "mom_12_1", + "mom_12_1_resid", + "mom_12_1_sector_resid", + "mom_12_1_sector_demeaned", + "sue_latest", + "fip_id", + ) + } + side_restricted = { + name: _find_r(name) + for name in ("mom_12_1", "mom_12_1_resid", "sue_latest") + } + return { + "signal_eval_side_by_side": side, + "signal_eval_identical_sue_subset": side_restricted, + "identical_subset_note": ( + "Mom baselines re-scored only on (week, symbol) cells where SUE exists. " + "Use this table when backfill is incomplete — full-universe mom N is not comparable." + ), + "full_signal_eval": signal_eval, + "sue_grade": grade, + "momentum_conditional_sue": mom_cond, + "sue_coverage": { + "symbols_with_sue": len(sue_map), + "avg_weeks_with_sue": ( + round( + sum(len(v) for v in sue_collected.values()) + / max(1, len(sue_collected)), + 1, + ) + if sue_collected + else 0 + ), + "weeks_with_min_cross_section": len(usable), + }, + } + + +def _write_md(path: Path, payload: dict) -> None: + pre = path.read_text(encoding="utf-8") if path.exists() else "" + marker = "## Results" + idx = pre.find(marker) + header = pre[:idx] if idx >= 0 else pre.split("## Verdict")[0] + + lines = [ + header.rstrip(), + "", + "## Results", + "", + f"Generated: `{payload.get('generated_at')}`", + "", + "### Data provenance", + "", + f"```json\n{json.dumps(payload.get('data_provenance') or {}, indent=2, default=str)}\n```", + "", + "### 2a — Earnings-gap risk (report-only)", + "", + ] + a = payload.get("experiment_2a") + if not a: + lines.append("_Skipped or unavailable._") + else: + lines.append(f"```json\n{json.dumps(a, indent=2, default=str)}\n```") + lines.extend(["", "### 2b — SUE / PEAD IC", ""]) + b = payload.get("experiment_2b") + if not b: + lines.append("_Skipped or unavailable._") + else: + side = b.get("signal_eval_side_by_side") or {} + lines.extend([ + "| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable |", + "|---|---:|---:|---:|---:|---|", + ]) + for name in ( + "mom_12_1", + "mom_12_1_resid", + "sue_latest", + "mom_12_1_sector_resid", + "fip_id", + ): + r = side.get(name) or {} + lines.append( + f"| {name} | {r.get('mean_ic', '')} | {r.get('ic_t_stat', '')} | " + f"{r.get('weeks', '')} | {r.get('avg_cross_section', '')} | " + f"{r.get('reliable', '')} |" + ) + lines.extend([ + "", + f"**SUE grade:** `{json.dumps(b.get('sue_grade') or {}, default=str)}`", + "", + f"**Momentum-conditional SUE:** `{json.dumps(b.get('momentum_conditional_sue') or {}, default=str)}`", + "", + ]) + + lines.extend([ + "", + "## Verdict", + "", + f"**{payload.get('verdict')}**", + "", + payload.get("verdict_detail") or "", + "", + "## What a human must decide next", + "", + payload.get("human_next") or "- Review; no auto-ship.", + "", + f"Artifacts: `{payload.get('report_path')}`", + "", + ]) + path.write_text("\n".join(lines) + "\n", encoding="utf-8") + + +async def _main() -> None: + args = _parse_args() + snapshot = Path(args.snapshot) + if not snapshot.exists(): + raise SystemExit(f"Missing snapshot {snapshot}") + if args.allow_spawn: + os.environ["BACKTEST_ALLOW_SPAWN"] = "1" + + events, meta = _load_earnings(snapshot) + # Race guard lite on earnings completeness. + provenance = { + "snapshot": str(snapshot.resolve()), + "n_earnings_events": len(events), + "backfill_meta": meta, + "announce_range": { + "min": min((e["announce_date"] for e in events), default=None), + "max": max((e["announce_date"] for e in events), default=None), + }, + "with_actual_and_estimate": sum( + 1 + for e in events + if e.get("eps_actual") is not None and e.get("eps_estimate") is not None + ), + } + print( + f"Earnings events: {provenance['n_earnings_events']} " + f"(with act+est={provenance['with_actual_and_estimate']}) meta={meta}" + ) + if meta and meta.get("done", 0) < 0.9 * (meta.get("universe_tickers") or 1): + print( + "WARNING: earnings backfill incomplete " + f"({meta.get('done')}/{meta.get('universe_tickers')}). " + "Results may be biased; resume backfill." + ) + + exp_2a = None + exp_2b = None + if not args.skip_2a: + print("Running 2a earnings-gap diagnostic…") + exp_2a = await _run_2a( + snapshot, events, quiet=args.quiet, workers=args.workers + ) + print( + " 2a losses<-1R with earnings:", + (exp_2a.get("q1_losses_worse_than_minus_1r") or {}), + ) + if not args.skip_2b: + print("Running 2b SUE IC harness…") + exp_2b = await _run_2b_ic( + snapshot, events, quiet=args.quiet, workers=args.workers + ) + g = exp_2b.get("sue_grade") or {} + print(f" 2b SUE green={g.get('green')} {g.get('reason')}") + + # Verdict + if exp_2b and (exp_2b.get("sue_grade") or {}).get("green"): + verdict = "PROMOTE (2b SUE) — STOP for human wire design" + detail = ( + "SUE cleared iron rule. No book integration without human approval. " + "2a remains report-only." + ) + human = ( + "- Design tilt vs second gate if desired.\n" + "- Do not auto-filter from 2a without separate approval + tail review." + ) + else: + sue_ic = None + if exp_2b: + sue_ic = ((exp_2b.get("sue_grade") or {}).get("row") or {}).get("mean_ic") + if sue_ic is not None and abs(float(sue_ic)) >= 0.015: + verdict = "PARK" + detail = f"SUE IC={sue_ic} below iron bar or unreliable; keep data, no wire." + else: + verdict = "DEAD (2b) / REPORT-ONLY (2a)" + detail = ( + "SUE does not clear iron rule on this window. " + "2a distributions for human risk review only — no filter." + ) + human = ( + "- No SUE book change.\n" + "- Read 2a tails before considering any earnings-avoid filter." + ) + + stamp = datetime.now().strftime("%Y%m%d-%H%M%S") + out = Path(args.out) if args.out else Path("reports") / f"earnings-gap-sue-{stamp}.json" + payload = { + "generated_at": datetime.now().isoformat(), + "data_provenance": provenance, + "experiment_2a": exp_2a, + "experiment_2b": exp_2b, + "verdict": verdict, + "verdict_detail": detail, + "human_next": human, + "report_path": str(out.as_posix()), + "fmp_note": ( + "Bulk earnings-calendar is paid (402 on free tier). " + "Backfill used per-symbol /stable/earnings; see earnings-backfill-status.json." + ), + } + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(payload, indent=2, default=str) + "\n", encoding="utf-8") + md = Path("docs/research/earnings-gap-and-sue.md") + _write_md(md, payload) + out.with_suffix(".md").write_text(md.read_text(encoding="utf-8"), encoding="utf-8") + print(f"Verdict: {verdict}") + print(f"Wrote {out}") + + +if __name__ == "__main__": + asyncio.run(_main()) diff --git a/scripts/run_history_depth_research.py b/scripts/run_history_depth_research.py new file mode 100644 index 0000000..02bb229 --- /dev/null +++ b/scripts/run_history_depth_research.py @@ -0,0 +1,476 @@ +"""History-depth extension research (local / MacBook). + +Phases +------ + coverage — bars per calendar year; no rebuild + harness — race-guard snapshot, full signal_eval, era split pre/post-2021 + +Does not retune production knobs. Does not modify scheduler/gates. + +Example +------- + python scripts/run_history_depth_research.py --phase coverage \\ + --snapshot backtest_snapshots/prod.sqlite + + python scripts/run_history_depth_research.py --phase harness \\ + --snapshot backtest_snapshots/research.sqlite --workers 8 --allow-spawn +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import os +import sys +from collections import defaultdict +from datetime import date, datetime +from pathlib import Path +from typing import Any + +from sqlalchemy import create_engine, text +from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +ERA_SPLIT = date(2021, 1, 1) +SURVIVORSHIP_BANNER = ( + "SURVIVORSHIP BIAS: today's constituents backfilled historically. " + "Absolute Sharpe/CAGR levels on deep history are optimistic. " + "Use RELATIVE signal IC comparisons and era stability only — not levels." +) + + +def _sqlite_url(path: Path) -> str: + return f"sqlite+aiosqlite:///{path.resolve().as_posix()}" + + +def _parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__) + p.add_argument("--phase", choices=("coverage", "harness", "all"), default="all") + p.add_argument("--snapshot", default="backtest_snapshots/research.sqlite") + p.add_argument("--workers", type=int, default=8) + p.add_argument("--allow-spawn", action="store_true") + p.add_argument("--quiet", action="store_true") + p.add_argument("--out", default=None) + return p.parse_args() + + +def _coverage_report(snapshot: Path) -> dict[str, Any]: + engine = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + try: + with engine.connect() as conn: + ticker_n = int(conn.execute(text("SELECT COUNT(*) FROM tickers")).scalar_one()) + ohlcv_n = int( + conn.execute(text("SELECT COUNT(*) FROM ohlcv_records")).scalar_one() + ) + d_range = conn.execute( + text("SELECT MIN(date), MAX(date) FROM ohlcv_records") + ).fetchone() + # Bars per calendar year (global). + by_year = conn.execute( + text( + """ + SELECT substr(date, 1, 4) AS y, COUNT(*) AS n, + COUNT(DISTINCT ticker_id) AS tickers + FROM ohlcv_records + GROUP BY substr(date, 1, 4) + ORDER BY y + """ + ) + ).fetchall() + # Per-symbol min/max date + bar count (summary percentiles). + per_sym = conn.execute( + text( + """ + SELECT t.symbol, COUNT(*) AS n, MIN(o.date), MAX(o.date) + FROM ohlcv_records o + JOIN tickers t ON t.id = o.ticker_id + GROUP BY t.symbol + """ + ) + ).fetchall() + finally: + engine.dispose() + + ns = sorted(int(r[1]) for r in per_sym) + def pct(p: float) -> int | None: + if not ns: + return None + i = int(round(p * (len(ns) - 1))) + return ns[i] + + starts = sorted(str(r[2]) for r in per_sym if r[2]) + start_hist: dict[str, int] = defaultdict(int) + for s in starts: + start_hist[s[:4]] += 1 + + return { + "snapshot": str(snapshot.resolve()), + "ticker_count": ticker_n, + "ohlcv_row_count": ohlcv_n, + "date_range": {"min": d_range[0], "max": d_range[1]}, + "bars_per_year": [ + {"year": y, "bars": n, "tickers_with_bars": t} for y, n, t in by_year + ], + "bars_per_symbol": { + "min": ns[0] if ns else None, + "p10": pct(0.10), + "p50": pct(0.50), + "p90": pct(0.90), + "max": ns[-1] if ns else None, + }, + "symbols_by_start_year": dict(sorted(start_hist.items())), + "note": ( + "Where ticker counts drop in early years, the feed (or listing history) " + "thins — do not treat those years as a full 505-name cross-section." + ), + "survivorship_banner": SURVIVORSHIP_BANNER, + } + + +def _assert_complete(snapshot: Path) -> dict[str, Any]: + from scripts.research_snapshot_manifest import ( # type: ignore + assert_research_snapshot_complete, + load_manifest, + ) + + m = load_manifest(snapshot) + if m is None: + # Prod snapshot may lack manifest; still require healthy bar depth. + eng = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + try: + with eng.connect() as conn: + avg = conn.execute( + text( + """ + SELECT AVG(c) FROM ( + SELECT COUNT(*) AS c FROM ohlcv_records GROUP BY ticker_id + ) + """ + ) + ).scalar_one() + finally: + eng.dispose() + if avg is None or float(avg) < 400: + raise SystemExit( + f"No completion manifest and avg bars={avg} look short. " + "Rebuild research.sqlite via extend_snapshot_universe.py" + ) + return {"manifest": None, "avg_bars": float(avg), "ok": True} + return {"manifest": assert_research_snapshot_complete(snapshot), "ok": True} + + +async def _harness(snapshot: Path, *, workers: int, quiet: bool) -> dict[str, Any]: + from app.config import settings + from app.services.backtest_service import run_backtest + + os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1" + os.environ["BACKTEST_SIGNAL_EVAL_ONLY"] = "1" + if Path("data/research/ticker_sector_map.json").exists(): + os.environ["BACKTEST_SECTOR_MAP_PATH"] = str( + Path("data/research/ticker_sector_map.json").resolve() + ) + settings.backtest_workers = workers + + engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True) + Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False) + + def progress(done: int, total: int, symbol: str) -> None: + if quiet: + return + print(f" progress {done}/{total} {symbol}", end="\r", flush=True) + + try: + async with Session() as db: + report = await run_backtest(db, progress_cb=progress, cadence="weekly") + finally: + await engine.dispose() + if not quiet: + print() + + signal_eval = report.get("signal_eval") or [] + + # Era-split IC: recompute from collected is not available post-run. + # Approximate via second pass is expensive; instead document that era split + # requires collecting weekly ICs. We re-run evaluation if the report embeds + # nothing — for v1, call internal collection is too heavy to duplicate. + # Lightweight approach: mark era_split as requiring BACKTEST with custom + # filter — implemented below by re-scoring from a dedicated collection pass. + era = await _era_split_ics(snapshot, workers=workers, quiet=quiet) + + return { + "survivorship_banner": SURVIVORSHIP_BANNER, + "signal_eval": signal_eval, + "era_split": era, + "params": report.get("params"), + "tickers": report.get("tickers"), + "generated_at_run": report.get("generated_at"), + } + + +async def _era_split_ics( + snapshot: Path, *, workers: int, quiet: bool +) -> dict[str, Any]: + """Collect weekly signal series and evaluate pre/post ERA_SPLIT separately.""" + from app.config import settings + from app.services import backtest_service as bt + from app.services.benchmark_service import load_benchmark_closes + from app.models.ticker import Ticker + from sqlalchemy import select + from collections import defaultdict as dd + + os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1" + settings.backtest_workers = max(1, workers) + + engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True) + Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False) + + collected: dict = dd(lambda: dd(list)) + try: + async with Session() as db: + tickers = list( + (await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars() + ) + spy = await load_benchmark_closes(db, "SPY") + symbol_to_sector = {} + sector_etf: dict = {} + try: + from app.services.sector_map import ( + SECTOR_ETFS, + load_ticker_sector_map, + ) + + symbol_to_sector = load_ticker_sector_map() + for etf in SECTOR_ETFS: + series = await load_benchmark_closes(db, etf) + if series: + sector_etf[etf] = series + except Exception: + pass + + for idx, t in enumerate(tickers): + if not quiet and idx % 100 == 0: + print(f" era-collect {idx}/{len(tickers)}", end="\r", flush=True) + cols = await bt._fetch_columns(db, t.symbol) + if cols is None: + continue + records = [ + type( + "R", + (), + { + "date": date.fromordinal(int(cols[0][i])), + "close": cols[4][i], + "high": cols[2][i], + "volume": cols[5][i] if len(cols) > 5 else 0, + }, + )() + for i in range(len(cols[0])) + ] + series = bt._signal_series( + records, + spy, + symbol=t.symbol, + sector_etf_closes=bt._sector_etf_closes_for_symbol( + t.symbol, symbol_to_sector, sector_etf + ), + ) + for name, weeks in series.items(): + for wk, pairs in weeks.items(): + collected[name][wk].extend(pairs) + if symbol_to_sector: + bt._inject_sector_demeaned_momentum(collected, symbol_to_sector) + finally: + await engine.dispose() + if not quiet: + print() + + def _filter_era(coll: dict, *, pre: bool) -> dict: + out: dict = dd(lambda: dd(list)) + for name, weeks in coll.items(): + for wk, recs in weeks.items(): + # ISO week key (year, week) — approximate era by ISO year. + year = int(wk[0]) if isinstance(wk, tuple) else int(str(wk)[:4]) + if pre and year >= ERA_SPLIT.year: + continue + if not pre and year < ERA_SPLIT.year: + continue + out[name][wk].extend(recs) + return out + + pre_eval = bt._signal_evaluation(_filter_era(collected, pre=True)) + post_eval = bt._signal_evaluation(_filter_era(collected, pre=False)) + full_eval = bt._signal_evaluation(collected) + + def _index(rows: list[dict]) -> dict[str, dict]: + return {r["signal"]: r for r in rows} + + return { + "era_split_date": ERA_SPLIT.isoformat(), + "note": "Diagnostic only — not a tuning input. Nested lookbacks are not OOS.", + "full": _index(full_eval), + "pre_2021": _index(pre_eval), + "post_2021": _index(post_eval), + } + + +def _write_md(path: Path, payload: dict) -> None: + pre = path.read_text(encoding="utf-8") if path.exists() else "" + marker = "## Results" + idx = pre.find(marker) + header = pre[:idx] if idx >= 0 else pre.split("## Verdict")[0] + + lines = [ + header.rstrip(), + "", + "## Results", + "", + f"Generated: `{payload.get('generated_at')}`", + "", + f"> **{SURVIVORSHIP_BANNER}**", + "", + "### Coverage", + "", + f"```json\n{json.dumps(payload.get('coverage') or {}, indent=2, default=str)}\n```", + "", + "### Race guard", + "", + f"```json\n{json.dumps(payload.get('race_guard') or {}, indent=2, default=str)}\n```", + "", + "### Signal IC (full extended window)", + "", + ] + harness = payload.get("harness") or {} + rows = harness.get("signal_eval") or [] + if rows: + lines.extend([ + "| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable |", + "|---|---:|---:|---:|---:|---|", + ]) + for r in rows: + lines.append( + f"| {r.get('signal')} | {r.get('mean_ic')} | {r.get('ic_t_stat')} | " + f"{r.get('weeks')} | {r.get('avg_cross_section')} | {r.get('reliable')} |" + ) + else: + lines.append("_Harness not run this pass._") + + era = (harness.get("era_split") or {}) + lines.extend(["", "### Era split (diagnostic only)", ""]) + if era: + for label in ("full", "pre_2021", "post_2021"): + block = era.get(label) or {} + lines.append(f"#### {label}") + lines.append("") + lines.append("| signal | mean_ic | t | weeks | N |") + lines.append("|---|---:|---:|---:|---:|") + for name in sorted(block): + r = block[name] + lines.append( + f"| {name} | {r.get('mean_ic')} | {r.get('ic_t_stat')} | " + f"{r.get('weeks')} | {r.get('avg_cross_section')} |" + ) + lines.append("") + else: + lines.append("_No era split._") + + lines.extend([ + "", + "## Verdict", + "", + f"**{payload.get('verdict')}**", + "", + payload.get("verdict_detail") or "", + "", + "## What a human must decide next", + "", + payload.get("human_next") + or "- Do not retune production knobs from this report without review.", + "", + f"Artifacts: `{payload.get('report_path')}`", + "", + ]) + path.write_text("\n".join(lines) + "\n", encoding="utf-8") + + +async def _main() -> None: + args = _parse_args() + snapshot = Path(args.snapshot) + if not snapshot.exists(): + raise SystemExit(f"Missing snapshot: {snapshot}") + if args.allow_spawn: + os.environ["BACKTEST_ALLOW_SPAWN"] = "1" + + coverage = None + race = None + harness = None + + if args.phase in ("coverage", "all"): + print("Coverage probe…") + coverage = _coverage_report(snapshot) + print( + f" tickers={coverage['ticker_count']} ohlcv={coverage['ohlcv_row_count']} " + f"range={coverage['date_range']}" + ) + for row in coverage["bars_per_year"]: + print( + f" year {row['year']}: bars={row['bars']} " + f"tickers={row['tickers_with_bars']}" + ) + + if args.phase in ("harness", "all"): + print("Race guard…") + race = _assert_complete(snapshot) + print(f" ok={race.get('ok')}") + print("Full harness + era split (LONG)…") + print(f" {SURVIVORSHIP_BANNER}") + harness = await _harness( + snapshot, workers=args.workers, quiet=args.quiet + ) + + stamp = datetime.now().strftime("%Y%m%d-%H%M%S") + out = ( + Path(args.out) + if args.out + else Path("reports") / f"history-depth-{stamp}.json" + ) + payload = { + "generated_at": datetime.now().isoformat(), + "survivorship_banner": SURVIVORSHIP_BANNER, + "coverage": coverage, + "race_guard": race, + "harness": harness, + "verdict": "PENDING_HUMAN" if harness else "COVERAGE_ONLY", + "verdict_detail": ( + "Harness complete — human interprets relative IC / era stability. " + "No production retune from this artifact." + if harness + else "Coverage probe only; run --phase harness after deep rebuild." + ), + "human_next": ( + "- Compare sector residual vs market residual across eras.\n" + "- If pre-2021 IC collapses, park Task 1 wire-in.\n" + "- Do not retune production knobs on deep history levels." + ), + "report_path": str(out.as_posix()), + } + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(payload, indent=2, default=str) + "\n", encoding="utf-8") + md = Path("docs/research/history-depth-extension.md") + _write_md(md, payload) + out.with_suffix(".md").write_text(md.read_text(encoding="utf-8"), encoding="utf-8") + print(f"Wrote {out}") + print(f"Wrote {md}") + + +if __name__ == "__main__": + asyncio.run(_main()) diff --git a/scripts/run_sector_residual_research.py b/scripts/run_sector_residual_research.py new file mode 100644 index 0000000..290987d --- /dev/null +++ b/scripts/run_sector_residual_research.py @@ -0,0 +1,1014 @@ +"""Sector-residual momentum research runner (local only). + +Protocol +-------- +1. Race-guard the research/prod snapshot (completion manifest when present). +2. Require sector map + sector ETFs in ``benchmark_prices``. +3. Run signal IC harness on the production ~505-name snapshot + (``BACKTEST_SIGNAL_EVAL_ONLY=1``) with sector context loaded. +4. Grade candidates vs pre-registered iron rule + t-stat vs ``mom_12_1_resid``. +5. If a candidate promotes: portfolio A/B with candidate as momentum leg + + gate percentile (``fill_mode=close``). Optional sector-cap arm. + +Does not modify production DB, gate, scanner, or schedule. + +Example +------- + python scripts/run_sector_residual_research.py \\ + --snapshot backtest_snapshots/prod.sqlite \\ + --workers 6 --allow-spawn +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import math +import os +import sys +from collections import defaultdict +from copy import deepcopy +from datetime import date, datetime +from pathlib import Path +from typing import Any + +from sqlalchemy import create_engine, text +from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine + +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: + sys.path.insert(0, str(ROOT)) + +from app.services.sector_map import ( # noqa: E402 + DEFAULT_SECTOR_MAP_PATH, + SECTOR_ETFS, + coverage_stats, + load_ticker_sector_map, + normalise_symbol, + sector_to_etf, +) + +VALIDATION_SPLIT = date(2024, 7, 1) +IRON_IC_BAR = 0.03 +MIN_RELIABLE = 12 + + +def _sqlite_url(path: Path) -> str: + return f"sqlite+aiosqlite:///{path.resolve().as_posix()}" + + +def _parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__) + p.add_argument("--snapshot", default="backtest_snapshots/prod.sqlite") + p.add_argument( + "--sector-map", + default=str(DEFAULT_SECTOR_MAP_PATH), + ) + p.add_argument("--workers", type=int, default=6) + p.add_argument("--allow-spawn", action="store_true") + p.add_argument( + "--skip-ab", + action="store_true", + help="IC only — never run portfolio A/B even if promotion fires.", + ) + p.add_argument( + "--force-ab", + action="store_true", + help="Run A/B for diagnostic even if IC bar fails (still reported as non-promote).", + ) + p.add_argument( + "--sector-cap", + type=int, + default=None, + help="Optional max positions per sector (e.g. 3). Only used in A/B.", + ) + p.add_argument("--quiet", action="store_true") + p.add_argument( + "--out", + default=None, + help="JSON report path (default reports/sector-residual-YYYYMMDD-HHMMSS.json)", + ) + return p.parse_args() + + +def _assert_snapshot_ready(snapshot: Path) -> dict[str, Any]: + """Race guard: prefer completion manifest; always check live bar sanity.""" + from scripts.research_snapshot_manifest import ( # type: ignore + assert_research_snapshot_complete, + load_manifest, + ) + + guard: dict[str, Any] = {"snapshot": str(snapshot.resolve())} + manifest = load_manifest(snapshot) + if manifest is not None: + # Full assert when a manifest exists (research.sqlite path). + try: + m = assert_research_snapshot_complete(snapshot) + guard["manifest"] = m + guard["manifest_ok"] = True + except SystemExit as exc: + raise SystemExit(str(exc)) from exc + else: + guard["manifest"] = None + guard["manifest_ok"] = None + guard["note"] = ( + "No completion manifest (prod.sqlite is expected without one). " + "Bar-count sanity still applied." + ) + + engine = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + try: + with engine.connect() as conn: + ticker_n = int(conn.execute(text("SELECT COUNT(*) FROM tickers")).scalar_one()) + ohlcv_n = int( + conn.execute(text("SELECT COUNT(*) FROM ohlcv_records")).scalar_one() + ) + bar_stats = conn.execute( + text( + """ + SELECT MIN(c), AVG(c), MAX(c) FROM ( + SELECT COUNT(*) AS c FROM ohlcv_records GROUP BY ticker_id + ) + """ + ) + ).fetchone() + bench = conn.execute( + text( + "SELECT symbol, COUNT(*), MIN(date), MAX(date) " + "FROM benchmark_prices GROUP BY symbol ORDER BY symbol" + ) + ).fetchall() + d_range = conn.execute( + text("SELECT MIN(date), MAX(date) FROM ohlcv_records") + ).fetchone() + finally: + engine.dispose() + + guard["ticker_count"] = ticker_n + guard["ohlcv_row_count"] = ohlcv_n + guard["bars_min_avg_max"] = { + "min": bar_stats[0], + "avg": round(float(bar_stats[1]), 1) if bar_stats[1] is not None else None, + "max": bar_stats[2], + } + guard["ohlcv_date_range"] = {"min": d_range[0], "max": d_range[1]} + guard["benchmark_prices"] = [ + {"symbol": s, "n": n, "min": d0, "max": d1} for s, n, d0, d1 in bench + ] + + # Sanity: a half-built snapshot would show many tickers with tiny bar counts. + min_bars = int(bar_stats[0] or 0) + avg_bars = float(bar_stats[1] or 0) + if ticker_n < 400: + raise SystemExit( + f"Snapshot looks short: only {ticker_n} tickers (expected ~505 prod)." + ) + if avg_bars < 200: + raise SystemExit( + f"Snapshot bar counts look short (avg={avg_bars:.0f}). Rebuild before research." + ) + # Allow a few thin names; refuse if median path is collapsed. + if min_bars < 10 and avg_bars < 500: + raise SystemExit( + f"Snapshot min bars={min_bars}, avg={avg_bars:.0f} — possible partial build." + ) + + present_etfs = {row[0] for row in bench} + missing_etfs = [e for e in SECTOR_ETFS if e not in present_etfs] + guard["missing_sector_etfs"] = missing_etfs + if missing_etfs: + raise SystemExit( + "Sector ETFs missing from benchmark_prices: " + f"{missing_etfs}. Run scripts/fetch_sector_etfs_to_snapshot.py first." + ) + if "SPY" not in present_etfs: + raise SystemExit("SPY missing from benchmark_prices") + + return guard + + +def _find_signal(rows: list[dict], name: str) -> dict | None: + for row in rows or []: + if row.get("signal") == name: + return row + return None + + +def _grade_ic( + candidate: dict | None, + resid: dict | None, + *, + expected_sign: float = 1.0, +) -> dict[str, Any]: + """Iron rule + t-stat ≥ mom_12_1_resid.""" + if candidate is None: + return { + "promote_to_ab": False, + "reason": "signal missing from signal_eval", + } + mean_ic = candidate.get("mean_ic") + t_stat = candidate.get("ic_t_stat") + reliable = bool(candidate.get("reliable")) + weeks = int(candidate.get("weeks") or 0) + if mean_ic is None or t_stat is None: + return {"promote_to_ab": False, "reason": "missing mean_ic or t", "row": candidate} + + sign_ok = (float(mean_ic) * expected_sign) > 0 + mag_ok = abs(float(mean_ic)) >= IRON_IC_BAR + reliable_ok = reliable and weeks >= MIN_RELIABLE + resid_t = resid.get("ic_t_stat") if resid else None + t_ok = resid_t is not None and float(t_stat) >= float(resid_t) + + promote = sign_ok and mag_ok and reliable_ok and t_ok + return { + "promote_to_ab": promote, + "checks": { + "sign_ok": sign_ok, + "abs_mean_ic_ge_0_03": mag_ok, + "reliable": reliable_ok, + "t_ge_resid": t_ok, + "mean_ic": mean_ic, + "ic_t_stat": t_stat, + "resid_ic_t_stat": resid_t, + "weeks": weeks, + }, + "reason": ( + "clears iron rule and t ≥ mom_12_1_resid — authorized for A/B only" + if promote + else "does not clear pre-registered IC promotion bar" + ), + "row": candidate, + } + + +async def _run_signal_eval( + snapshot: Path, + *, + workers: int, + quiet: bool, + sector_map_path: Path, +) -> dict: + from app.config import settings + from app.services.backtest_service import run_backtest + + os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1" + os.environ["BACKTEST_SIGNAL_EVAL_ONLY"] = "1" + os.environ["BACKTEST_SECTOR_MAP_PATH"] = str(sector_map_path.resolve()) + # Clear liquid-breadth — this is the 505-name prod IC, not breadth. + os.environ.pop("BACKTEST_LIQUID_BREADTH", None) + settings.backtest_workers = workers + + engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True) + Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False) + + def progress(done: int, total: int, symbol: str) -> None: + if quiet: + return + print(f" progress {done}/{total} {symbol}", end="\r", flush=True) + + try: + async with Session() as db: + report = await run_backtest(db, progress_cb=progress, cadence="weekly") + finally: + await engine.dispose() + if not quiet: + print() + return report + + +def _period_percentiles(rows: list[dict], value_key: str) -> dict[tuple, dict[str, float]]: + by_period: dict[tuple, list[dict]] = defaultdict(list) + for row in rows: + if row.get(value_key) is None: + continue + period = row.get("ranking_period") or row.get("iso_week") + by_period[period].append(row) + out: dict[tuple, dict[str, float]] = {} + for period, group in by_period.items(): + ordered = sorted(group, key=lambda r: float(r[value_key])) + n = len(ordered) + for rank, row in enumerate(ordered): + key = (str(row["symbol"]), str(row["date"])) + pct = (rank / (n - 1) * 100.0) if n > 1 else 100.0 + out.setdefault(key, {})[value_key] = float(row[value_key]) + out[key][f"{value_key}_percentile"] = pct + return out + + +async def _load_prices_and_benchmarks(snapshot: Path) -> tuple[dict, dict, dict]: + """Return (price_columns, spy_closes, sector_etf_closes).""" + from app.services import backtest_service as bt + from app.services.benchmark_service import load_benchmark_closes + from app.models.ticker import Ticker + from sqlalchemy import select + + engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True) + Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False) + prices: dict[str, tuple] = {} + try: + async with Session() as db: + tickers = list( + (await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars() + ) + for t in tickers: + cols = await bt._fetch_columns(db, t.symbol) + if cols is not None: + prices[t.symbol] = cols + spy = await load_benchmark_closes(db, "SPY") + sector: dict[str, dict] = {} + for etf in SECTOR_ETFS: + series = await load_benchmark_closes(db, etf) + if series: + sector[etf] = series + finally: + await engine.dispose() + return prices, spy, sector + + +def _recompute_sector_residual_on_candidates( + candidates: list[dict], + prices: dict[str, tuple], + spy_closes: dict, + sector_etf_closes: dict[str, dict], + symbol_to_sector: dict[str, str], + *, + momentum_field: str, +) -> list[dict]: + """Attach alternative residual momentum on each candidate as-of date.""" + from app.services import backtest_service as bt + from app.services.sector_map import etf_for_symbol + + # Index price series once. + series_cache: dict[str, tuple[list, list, list]] = {} + for sym, cols in prices.items(): + ords, _o, _h, _l, closes, _v = cols + dates = [date.fromordinal(int(o)) for o in ords] + series_cache[sym] = (dates, list(closes), list(ords)) + + out: list[dict] = [] + for cand in candidates: + c = dict(cand) + sym = str(c["symbol"]) + if sym not in series_cache: + out.append(c) + continue + dates, closes, ords = series_cache[sym] + asof = date.fromisoformat(str(c["date"])) + # Find as-of index. + try: + i = next(idx for idx, d in enumerate(dates) if d == asof) + except StopIteration: + # nearest on/before + i = max((idx for idx, d in enumerate(dates) if d <= asof), default=-1) + if i < 0: + out.append(c) + continue + + if momentum_field == "mom_12_1_sector_resid": + etf = etf_for_symbol(sym, symbol_to_sector) + etf_series = sector_etf_closes.get(etf or "") + val = None + if spy_closes and etf_series: + val = bt._multi_factor_residual_momentum_12_1( + dates, closes, i, [spy_closes, etf_series] + ) + c["residual_momentum"] = val + c["_alt_momentum_signal"] = momentum_field + c["_alt_momentum_value"] = val + elif momentum_field == "mom_12_1_sector_demeaned": + # Placeholder: demean requires cross-section; filled in a second pass. + raw = None + if i >= 252 and closes[i - 252] > 0: + raw = closes[i - 21] / closes[i - 252] - 1.0 + c["_raw_mom_12_1"] = raw + c["_alt_momentum_signal"] = momentum_field + else: + raise ValueError(momentum_field) + out.append(c) + + if momentum_field == "mom_12_1_sector_demeaned": + # Cross-sectional demean within ranking period × sector. + by_period: dict[Any, list[dict]] = defaultdict(list) + for c in out: + if c.get("_raw_mom_12_1") is None: + continue + period = c.get("ranking_period") or c.get("iso_week") + by_period[period].append(c) + for period, group in by_period.items(): + by_sec: dict[str, list[float]] = defaultdict(list) + for c in group: + sec = symbol_to_sector.get(normalise_symbol(str(c["symbol"]))) + if sec: + by_sec[sec].append(float(c["_raw_mom_12_1"])) + means = { + s: sum(vs) / len(vs) for s, vs in by_sec.items() if len(vs) >= 2 + } + for c in group: + sec = symbol_to_sector.get(normalise_symbol(str(c["symbol"]))) + raw = float(c["_raw_mom_12_1"]) + if sec in means: + val = raw - means[sec] + c["residual_momentum"] = val + c["_alt_momentum_value"] = val + else: + c["residual_momentum"] = None + c["_alt_momentum_value"] = None + + return out + + +def _assign_prod_ranks(candidates: list[dict]) -> None: + from app.services import backtest_service as bt + + bt._assign_momentum_percentiles(candidates) + bt._assign_residual_momentum_percentiles(candidates) + bt._assign_low_volatility_percentiles(candidates) + bt._assign_activation_momentum_percentiles(candidates) + bt._assign_residual_high_vol_blend(candidates) + for c in candidates: + c["qualified"] = bt._momentum_qualifies(c, 80.0) + + +async def _run_ab( + snapshot: Path, + *, + sector_map: dict[str, str], + signal_name: str, + sector_cap: int | None, + quiet: bool, + workers: int, +) -> dict[str, Any]: + """Control vs treatment book with candidate as momentum residual.""" + from app.services import backtest_service as bt + from app.config import settings + from app.models.ticker import Ticker + from app.services.admin_service import get_activation_config + from app.services.recommendation_service import get_recommendation_config + from app.services.paper_trade_service import get_exit_policy + from app.services.benchmark_service import load_benchmark_closes + from sqlalchemy import select + + os.environ["BACKTEST_SNAPSHOT_OFFLINE"] = "1" + os.environ.pop("BACKTEST_SIGNAL_EVAL_ONLY", None) + settings.backtest_workers = max(1, int(workers)) + + engine = create_async_engine(_sqlite_url(snapshot), pool_pre_ping=True) + Session = async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False) + try: + async with Session() as db: + config = await get_recommendation_config(db) + activation = await get_activation_config(db) + exit_config = await get_exit_policy(db) + tickers = list( + (await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars() + ) + spy = await load_benchmark_closes(db, "SPY") + sector_etf: dict[str, dict] = {} + for etf in SECTOR_ETFS: + series = await load_benchmark_closes(db, etf) + if series: + sector_etf[etf] = series + + prices: dict[str, tuple] = {} + candidates: list[dict] = [] + for idx, t in enumerate(tickers): + if not quiet and idx % 25 == 0: + print(f" fetch {idx}/{len(tickers)}", end="\r", flush=True) + cols = await bt._fetch_columns(db, t.symbol) + if cols is None: + continue + prices[t.symbol] = cols + cands, _series = bt._replay_and_signals( + t.symbol, + cols, + config, + activation, + spy, + bt.PRODUCTION_GTL_TARGET_MODEL, + "weekly", + False, + sector_etf, + sector_map, + ) + candidates.extend(cands) + finally: + await engine.dispose() + if not quiet: + print() + + # Control ranks (production residual). + control = [dict(c) for c in candidates] + _assign_prod_ranks(control) + control_longs = [ + c for c in control if c.get("qualified") and c.get("direction") == "long" + ] + + # Treatment: replace residual with sector signal, re-rank. + treatment = _recompute_sector_residual_on_candidates( + candidates, + prices, + spy, + sector_etf, + sector_map, + momentum_field=signal_name, + ) + _assign_prod_ranks(treatment) + treatment_longs = [ + c for c in treatment if c.get("qualified") and c.get("direction") == "long" + ] + + strategy = next(s for s in bt.PORTFOLIO_MONITOR_STRATEGIES if s.get("is_production")) + entry_config = bt._entry_variant_config(str(strategy["entry_variant"])) + assert entry_config is not None + ranking_key = str( + entry_config.get("ranking_key") or entry_config["percentile_key"] + ) + # Production ranking key is residual_high_vol_blend_80_20. + if ranking_key not in (bt.RESIDUAL_HIGH_VOL_BLEND_80_20_KEY, bt.PRODUCTION_PERCENTILE_KEY): + ranking_key = bt.RESIDUAL_HIGH_VOL_BLEND_80_20_KEY + + exit_policy = bt.LIVE_EXIT_MODE_TO_SIM.get( + str(exit_config.get("mode", "atr_trailing")), "atr_trail3" + ) + hold_days = int(exit_config.get("hold_days", 30)) + trail = float(exit_config.get("atr_multiplier", bt.ATR_TRAIL_MULTIPLIER)) + risk = float(entry_config["risk_per_trade"]) + max_pos = int(entry_config["max_positions"]) + + def _sim_book(longs: list[dict], *, label: str, cap: int | None = None) -> dict: + reentry = bt._make_gate_reset_reentry_fn( + longs, prices, cadence="weekly", ranking_key=ranking_key + ) + windows = {} + for wname, start, end in ( + ("train", None, VALIDATION_SPLIT), + ("validation", VALIDATION_SPLIT, None), + ("full", None, None), + ): + sim = bt._simulate_portfolio( + longs, + prices, + spy, + exit_policy, + hold_days, + ranking_key=ranking_key, + max_positions=max_pos, + risk_per_trade=risk, + atr_trail_multiplier=trail, + post_stop_reentry_fn=reentry, + start_date=start, + end_date=end, + fill_mode=bt.FILL_MODE_CLOSE, + include_trades=True, + ) + if sim is None: + windows[wname] = {"error": "no_trades"} + continue + # Optional sector cap: filter trade_details is post-hoc; real cap needs + # simulator support. For research, re-sim with a wrapper ranking that + # drops overflow sector names is approximate — we implement a simple + # pre-filter on daily entry sets via max_positions only when cap is None. + # When cap is set, apply a post-sim diagnostic on entries. + payload = { + k: sim.get(k) + for k in ( + "sharpe", + "sharpe_se", + "cagr_pct", + "max_drawdown_pct", + "total_return_pct", + "trades", + "win_rate_pct", + "avg_r", + "n_returns", + "return_skew", + "return_kurtosis", + "psr", + ) + } + details = sim.get("trade_details") or [] + rs = [ + float(t["realized_r"]) + for t in details + if t.get("realized_r") is not None + ] + if rs: + rs_sorted = sorted(rs) + payload["r_p05"] = rs_sorted[max(0, int(0.05 * (len(rs_sorted) - 1)))] + payload["r_p50"] = rs_sorted[len(rs_sorted) // 2] + payload["r_p95"] = rs_sorted[min(len(rs_sorted) - 1, int(0.95 * (len(rs_sorted) - 1)))] + payload["entry_count"] = len(rs) + if cap is not None and details: + # Diagnostic: count how often a calendar day would exceed cap. + from collections import Counter + + # Use entry dates; sector from map. + day_sector: dict[str, Counter] = defaultdict(Counter) + for t in details: + sec = sector_map.get(normalise_symbol(str(t.get("symbol", "")))) or "?" + day_sector[str(t.get("entry_date") or t.get("date") or "")][sec] += 1 + breaches = sum( + 1 + for day, ctr in day_sector.items() + if any(v > cap for v in ctr.values()) + ) + payload["sector_cap"] = cap + payload["entry_days_with_sector_over_cap"] = breaches + windows[wname] = payload + return {"label": label, "n_qualified_longs": len(longs), "windows": windows} + + control_result = _sim_book(control_longs, label="control_mom_12_1_resid") + treatment_result = _sim_book( + treatment_longs, label=f"treatment_{signal_name}", cap=None + ) + out: dict[str, Any] = { + "signal": signal_name, + "ranking_key": ranking_key, + "fill_mode": "close", + "validation_split": VALIDATION_SPLIT.isoformat(), + "control": control_result, + "treatment": treatment_result, + "promotion": _grade_ab(control_result, treatment_result), + } + if sector_cap is not None: + # Approximate sector-cap book: when selecting, prefer higher rank but + # refuse a 4th name in the same sector among concurrent opens. + # Implemented by tagging candidates and using a custom sim is heavy; + # instead report diagnostic on unconstrained treatment + a filtered + # re-rank that zeros residual for overflow names within each period. + capped = _apply_sector_cap_to_ranks( + treatment, sector_map, cap=sector_cap, ranking_key=ranking_key + ) + capped_longs = [ + c for c in capped if c.get("qualified") and c.get("direction") == "long" + ] + out["sector_cap_arm"] = _sim_book( + capped_longs, label=f"treatment_{signal_name}_cap{sector_cap}", cap=sector_cap + ) + out["sector_cap_promotion"] = _grade_ab( + control_result, out["sector_cap_arm"] + ) + return out + + +def _apply_sector_cap_to_ranks( + candidates: list[dict], + sector_map: dict[str, str], + *, + cap: int, + ranking_key: str, +) -> list[dict]: + """Within each ranking period, keep top `cap` per sector by ranking_key.""" + by_period: dict[Any, list[dict]] = defaultdict(list) + for c in candidates: + period = c.get("ranking_period") or c.get("iso_week") + by_period[period].append(dict(c)) + out: list[dict] = [] + for period, group in by_period.items(): + ordered = sorted( + group, + key=lambda r: float(r.get(ranking_key) or r.get("residual_momentum") or -1e9), + reverse=True, + ) + sector_counts: dict[str, int] = defaultdict(int) + for c in ordered: + sec = sector_map.get(normalise_symbol(str(c["symbol"]))) or "_unknown" + if sector_counts[sec] >= cap: + # Push below gate by nulling activation percentile. + c["qualified"] = False + c["_sector_cap_blocked"] = True + else: + if c.get("qualified"): + sector_counts[sec] += 1 + out.append(c) + return out + + +def _grade_ab(control: dict, treatment: dict) -> dict[str, Any]: + """Pre-registered: val Sharpe ≥ control − 0.5·SE; full Sharpe & maxDD not worse.""" + def win(arm: dict, name: str) -> dict: + return (arm.get("windows") or {}).get(name) or {} + + c_val = win(control, "validation") + t_val = win(treatment, "validation") + c_full = win(control, "full") + t_full = win(treatment, "full") + + def _f(d: dict, k: str) -> float | None: + v = d.get(k) + return None if v is None else float(v) + + c_sh = _f(c_val, "sharpe") + t_sh = _f(t_val, "sharpe") + # Use treatment SE if present else control SE. + se = _f(t_val, "sharpe_se") + if se is None: + se = _f(c_val, "sharpe_se") + if se is None: + se = 0.0 + + val_ok = ( + c_sh is not None + and t_sh is not None + and t_sh >= (c_sh - 0.5 * se) + ) + c_full_sh = _f(c_full, "sharpe") + t_full_sh = _f(t_full, "sharpe") + full_sh_ok = ( + c_full_sh is not None + and t_full_sh is not None + and t_full_sh >= c_full_sh + ) + # max DD: higher absolute drawdown is worse; stored as positive pct typically. + c_dd = _f(c_full, "max_drawdown_pct") + t_dd = _f(t_full, "max_drawdown_pct") + full_dd_ok = ( + c_dd is not None and t_dd is not None and abs(t_dd) <= abs(c_dd) + 1e-9 + ) + promote = bool(val_ok and full_sh_ok and full_dd_ok) + return { + "promote": promote, + "checks": { + "validation_sharpe_ge_control_minus_half_se": val_ok, + "full_sharpe_not_worse": full_sh_ok, + "full_maxdd_not_worse": full_dd_ok, + "control_validation_sharpe": c_sh, + "treatment_validation_sharpe": t_sh, + "se_used": se, + "control_full_sharpe": c_full_sh, + "treatment_full_sharpe": t_full_sh, + "control_full_maxdd": c_dd, + "treatment_full_maxdd": t_dd, + }, + "reason": ( + "clears pre-registered A/B bar — human decides wire-in" + if promote + else "fails pre-registered A/B bar" + ), + } + + +def _write_md(path: Path, payload: dict) -> None: + """Refresh the results sections of the research doc (preserve pre-reg header).""" + # Always write a standalone results companion + update the main doc's + # results block by rewriting the full file with pre-reg + results. + pre = Path("docs/research/sector-residual-momentum.md") + # Keep pre-registration by reading until '## Results' if present. + header = "" + if pre.exists(): + text = pre.read_text(encoding="utf-8") + marker = "## Results" + idx = text.find(marker) + header = text[:idx] if idx >= 0 else text.split("## Verdict")[0] + + guard = payload.get("snapshot_guard") or {} + cov = payload.get("sector_coverage") or {} + ic_rows = payload.get("signal_eval") or [] + grades = payload.get("ic_grades") or {} + lines = [ + header.rstrip(), + "", + "## Results", + "", + f"Generated: `{payload.get('generated_at')}`", + "", + "### Snapshot race guard", + "", + f"- Snapshot: `{guard.get('snapshot')}`", + f"- Tickers: **{guard.get('ticker_count')}** OHLCV rows: **{guard.get('ohlcv_row_count')}**", + f"- Bars min/avg/max: `{guard.get('bars_min_avg_max')}`", + f"- OHLCV range: `{guard.get('ohlcv_date_range')}`", + f"- Manifest ok: `{guard.get('manifest_ok')}`", + f"- Missing sector ETFs at start: `{guard.get('missing_sector_etfs')}`", + "", + "### Sector label coverage", + "", + f"```json\n{json.dumps(cov, indent=2, default=str)}\n```", + "", + "### IC harness (identical cross-sections)", + "", + "| signal | mean_ic | ic_t_stat | weeks | avg_N | reliable | ic+_pct |", + "|---|---:|---:|---:|---:|---|---:|", + ] + want = [ + "mom_12_1", + "mom_12_1_resid", + "mom_12_1_sector_resid", + "mom_12_1_sector_demeaned", + ] + by_name = {r.get("signal"): r for r in ic_rows} + for name in want: + r = by_name.get(name) or {} + lines.append( + f"| {name} | {r.get('mean_ic', '')} | {r.get('ic_t_stat', '')} | " + f"{r.get('weeks', '')} | {r.get('avg_cross_section', '')} | " + f"{r.get('reliable', '')} | {r.get('ic_positive_pct', '')} |" + ) + lines.extend(["", "### IC promotion grades", ""]) + for name, g in grades.items(): + lines.append(f"- **{name}**: promote_to_ab=`{g.get('promote_to_ab')}` — {g.get('reason')}") + lines.append(f" - checks: `{json.dumps(g.get('checks') or {}, default=str)}`") + + ab = payload.get("portfolio_ab") + lines.extend(["", "### Portfolio A/B", ""]) + if not ab: + lines.append("_Not run (IC bar not cleared, or --skip-ab)._") + else: + lines.append(f"```json\n{json.dumps(ab, indent=2, default=str)}\n```") + + lines.extend([ + "", + "## Verdict", + "", + f"**{payload.get('verdict')}**", + "", + payload.get("verdict_detail") or "", + "", + "## What a human must decide next", + "", + payload.get("human_next") or "- Review numbers; do not merge into strategy docs without approval.", + "", + "## Artifacts", + "", + f"- JSON: `{payload.get('report_path')}`", + "", + ]) + path.write_text("\n".join(lines) + "\n", encoding="utf-8") + + +async def _main() -> None: + args = _parse_args() + snapshot = Path(args.snapshot) + sector_map_path = Path(args.sector_map) + if not snapshot.exists(): + raise SystemExit(f"Snapshot missing: {snapshot}") + if not sector_map_path.exists(): + raise SystemExit( + f"Sector map missing: {sector_map_path}. " + "Run scripts/build_ticker_sector_map.py first." + ) + + if args.allow_spawn: + os.environ["BACKTEST_ALLOW_SPAWN"] = "1" + + print("Race-guarding snapshot…") + guard = _assert_snapshot_ready(snapshot) + print( + f" tickers={guard['ticker_count']} ohlcv={guard['ohlcv_row_count']} " + f"bars={guard['bars_min_avg_max']}" + ) + + mapping = load_ticker_sector_map(sector_map_path) + engine = create_engine( + f"sqlite:///{snapshot.resolve().as_posix()}", + future=True, + ) + try: + with engine.connect() as conn: + symbols = [ + normalise_symbol(r[0]) + for r in conn.execute(text("SELECT symbol FROM tickers")).fetchall() + ] + finally: + engine.dispose() + cov = coverage_stats(symbols, mapping) + print( + f"Sector map: {cov['mapped']}/{cov['universe']} " + f"({cov['mapped_pct']}%) with_etf={cov['with_etf']}" + ) + if cov["mapped_pct"] < 90: + print( + f"WARNING: sector coverage {cov['mapped_pct']}% < 90%; " + f"missing e.g. {cov['missing'][:20]}" + ) + + print("Running IC harness (signal-eval only)…") + report = await _run_signal_eval( + snapshot, + workers=args.workers, + quiet=args.quiet, + sector_map_path=sector_map_path, + ) + signal_eval = report.get("signal_eval") or report.get("signals") or [] + # Locate key in report — backtest uses "signal_edge" historically. + if not signal_eval: + for key in ("signal_edge", "signal_evaluation", "factor_ic"): + if key in report and isinstance(report[key], list): + signal_eval = report[key] + break + + resid = _find_signal(signal_eval, "mom_12_1_resid") + grades = { + name: _grade_ic(_find_signal(signal_eval, name), resid) + for name in ("mom_12_1_sector_resid", "mom_12_1_sector_demeaned") + } + for name, g in grades.items(): + print( + f" {name}: promote_to_ab={g['promote_to_ab']} " + f"ic={((g.get('row') or {}).get('mean_ic'))} " + f"t={((g.get('row') or {}).get('ic_t_stat'))}" + ) + + ab_results: dict[str, Any] | None = None + promote_names = [n for n, g in grades.items() if g.get("promote_to_ab")] + run_ab = (bool(promote_names) or args.force_ab) and not args.skip_ab + if run_ab: + # Prefer sector_resid if both; else the one that promoted / force resid. + if "mom_12_1_sector_resid" in promote_names or ( + args.force_ab and not promote_names + ): + ab_signal = "mom_12_1_sector_resid" + else: + ab_signal = promote_names[0] + print(f"Running portfolio A/B for {ab_signal}…") + ab_results = await _run_ab( + snapshot, + sector_map=mapping, + signal_name=ab_signal, + sector_cap=args.sector_cap, + quiet=args.quiet, + workers=args.workers, + ) + print( + f" A/B promote={ab_results.get('promotion', {}).get('promote')} " + f"— {ab_results.get('promotion', {}).get('reason')}" + ) + else: + print("Skipping portfolio A/B (no IC promotion; use --force-ab to override).") + + # Verdict + if ab_results and ab_results.get("promotion", {}).get("promote"): + verdict = "PROMOTE" + detail = ( + f"{ab_results['signal']} cleared IC + A/B bars. " + "Human must design wire-in; do not ship from this branch." + ) + human = ( + "- Approve or reject production residual swap vs dual-signal design.\n" + "- If sector-cap arm ran, review tail-trim diagnostics before any cap." + ) + elif any(g.get("promote_to_ab") for g in grades.values()): + verdict = "PARK" + detail = ( + "IC promotion bar cleared but A/B did not promote " + "(or A/B skipped). Park for human review." + ) + human = "- Inspect A/B windows; decide whether to re-run or park." + elif any( + (g.get("row") or {}).get("mean_ic") is not None + and abs(float((g.get("row") or {}).get("mean_ic") or 0)) >= IRON_IC_BAR * 0.5 + for g in grades.values() + ): + verdict = "PARK" + detail = "Weak / partial IC — not dead, not green. Machinery kept." + human = "- No book change. Revisit after history-depth extension (Task 3)." + else: + verdict = "DEAD" + detail = ( + "Neither sector residual nor sector demean cleared the iron-rule bar " + "with t ≥ mom_12_1_resid on this window." + ) + human = "- Do not wire sector residual. Optional: re-check after Task 3 depth." + + stamp = datetime.now().strftime("%Y%m%d-%H%M%S") + out_path = Path(args.out) if args.out else Path("reports") / f"sector-residual-{stamp}.json" + payload = { + "generated_at": datetime.now().isoformat(), + "snapshot_guard": guard, + "sector_coverage": cov, + "sector_map_path": str(sector_map_path.resolve()), + "signal_eval": signal_eval, + "ic_grades": grades, + "portfolio_ab": ab_results, + "verdict": verdict, + "verdict_detail": detail, + "human_next": human, + "report_path": str(out_path.as_posix()), + "pre_registration": { + "iron_ic_bar": IRON_IC_BAR, + "validation_split": VALIDATION_SPLIT.isoformat(), + "fill_mode": "close", + "cost_per_side": 0.001, + "ab_rule": "val Sharpe >= control - 0.5*SE; full Sharpe & maxDD not worse", + }, + } + out_path.parent.mkdir(parents=True, exist_ok=True) + out_path.write_text(json.dumps(payload, indent=2, default=str) + "\n", encoding="utf-8") + + md_path = Path("docs/research/sector-residual-momentum.md") + _write_md(md_path, payload) + # Companion md under reports/ + md_report = out_path.with_suffix(".md") + md_report.write_text(md_path.read_text(encoding="utf-8"), encoding="utf-8") + + print(f"Verdict: {verdict}") + print(f"Wrote {out_path}") + print(f"Wrote {md_path}") + + +if __name__ == "__main__": + asyncio.run(_main()) diff --git a/tests/unit/test_backtest_service.py b/tests/unit/test_backtest_service.py index aff0eb6..b811a27 100644 --- a/tests/unit/test_backtest_service.py +++ b/tests/unit/test_backtest_service.py @@ -100,6 +100,75 @@ def test_residual_momentum_removes_market_beta_but_keeps_specific_drift(): assert drift["mom_12_1_resid"] > pure["mom_12_1_resid"] + 0.12 +def test_sector_residual_momentum_two_factor(): + """Pure market+sector beta stock → sector resid ~0; idiosyncratic drift kept.""" + dates, pure_beta, highs, benchmark = _signal_test_series(extra_return=0.0) + # Sector ETF = leveraged market (collinear-ish but not identical). + sector = {d: benchmark[d] * 1.02 + 0.5 for d in dates} + # Stock with pure exposure to market + sector, no alpha. + closes = [100.0] + for i in range(1, len(dates)): + m_prev = benchmark[dates[i - 1]] + m_cur = benchmark[dates[i]] + s_prev = sector[dates[i - 1]] + s_cur = sector[dates[i]] + m_ret = m_cur / m_prev - 1.0 + s_ret = s_cur / s_prev - 1.0 + closes.append(closes[-1] * (1.0 + 0.7 * m_ret + 0.5 * s_ret)) + highs_p = [c * 1.01 for c in closes] + + pure = bt._signal_values( + dates, closes, highs_p, 260, benchmark, sector_etf_closes=sector + ) + assert "mom_12_1_sector_resid" in pure + assert pure["mom_12_1_sector_resid"] == pytest.approx(0.0, abs=0.05) + + # Add idiosyncratic drift — sector residual should keep it. + drift_closes = [100.0] + for i in range(1, len(dates)): + m_prev = benchmark[dates[i - 1]] + m_cur = benchmark[dates[i]] + s_prev = sector[dates[i - 1]] + s_cur = sector[dates[i]] + m_ret = m_cur / m_prev - 1.0 + s_ret = s_cur / s_prev - 1.0 + drift_closes.append( + drift_closes[-1] * (1.0 + 0.7 * m_ret + 0.5 * s_ret + 0.0008) + ) + drift_highs = [c * 1.01 for c in drift_closes] + drift = bt._signal_values( + dates, drift_closes, drift_highs, 260, benchmark, sector_etf_closes=sector + ) + assert drift["mom_12_1_sector_resid"] > pure["mom_12_1_sector_resid"] + 0.10 + + +def test_inject_sector_demeaned_momentum(): + collected = { + "mom_12_1": { + (2024, 1): [ + {"val": 0.20, "fwd": 0.01, "symbol": "AAA"}, + {"val": 0.10, "fwd": 0.02, "symbol": "BBB"}, + {"val": 0.40, "fwd": -0.01, "symbol": "CCC"}, + {"val": 0.00, "fwd": 0.03, "symbol": "DDD"}, + ] + } + } + symbol_to_sector = { + "AAA": "Information Technology", + "BBB": "Information Technology", + "CCC": "Energy", + "DDD": "Energy", + } + bt._inject_sector_demeaned_momentum(collected, symbol_to_sector) + dem = collected["mom_12_1_sector_demeaned"][(2024, 1)] + by_sym = {r["symbol"]: r["val"] for r in dem} + # IT mean = 0.15 → AAA +0.05, BBB -0.05; Energy mean = 0.20 → CCC +0.20, DDD -0.20 + assert by_sym["AAA"] == pytest.approx(0.05) + assert by_sym["BBB"] == pytest.approx(-0.05) + assert by_sym["CCC"] == pytest.approx(0.20) + assert by_sym["DDD"] == pytest.approx(-0.20) + + def test_assigns_raw_and_residual_percentiles_independently(): cands = [ {"iso_week": (2026, 1), "momentum": 0.10, "residual_momentum": 0.30},