Retiring a symbol meant delete_ticker or bootstrap_universe(prune_missing), both of which cascade through OHLCV, setups and scores. That destroys exactly the history four research documents already apologise for: today's tracked universe projected backward is survivorship-biased, and hard-deleting every delisted name is what causes it. Keeping the rows preserves the option to fix that — it does not fix it, which needs the replay to model a delisting as an exit event. tickers gains delisted_on / delisted_reason (migration 032). NULL means actively traded. The filter is opt-in via ticker_service.active_only rather than folded into a shared getter: the registry and admin views deliberately keep delisted rows so the delisting is visible, and a silent default would undo that. Applied to the live path only — scanner, momentum ranking, scoring, breadth, fundamentals candidates, SEC universe, earnings import, ingestion loops. run_backtest keeps them on purpose. Detection runs off OHLCV staleness, not off the SEC fundamentals import: that importer stalls for days on unrelated Company-Facts gaps and would take detection down with it. On a stale symbol the scheduler asks SEC for a Form 25/25-NSE/15 and retires it only on a hit, so a halt or a rename (SATS->ECHO) keeps the existing warning. The probe waits 3 stale days so a market-data outage cannot turn into one SEC request per symbol per run. Safe to automate because it is reversible: clear_delisted un-retires a false positive, where a delete had already taken the history. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
220 lines
8.3 KiB
Python
220 lines
8.3 KiB
Python
"""Cross-sectional residual 12-1 momentum ranking for the universe.
|
|
|
|
The activation gate selects the top ``min_momentum_percentile`` of the universe
|
|
by residual 12-1 month momentum: the stock's 12-1 return after subtracting its
|
|
estimated benchmark beta contribution over the same formation window. The daily
|
|
scan ranks every ticker and stores each setup's percentile (see
|
|
``rr_scanner_service``), so the live list, the Track Record's qualified stats,
|
|
and outcome evaluation all gate on the same value.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
from datetime import date
|
|
|
|
from sqlalchemy import select
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.models.ticker import Ticker
|
|
from app.services import ticker_service
|
|
from app.services.price_service import query_ohlcv
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# 12-1 momentum: ~12 months of daily history (252 bars) with the last ~1 month
|
|
# (21 bars) skipped. Matches the backtest's _signal_values / _window_setups.
|
|
_MOM_LOOKBACK = 252
|
|
_MOM_SKIP = 21
|
|
|
|
# Promoted production ordering: strategy_rank blends the momentum and realized-
|
|
# volatility percentiles. Single source of truth — the backtest's production
|
|
# ranking key imports these so live and simulated ordering cannot drift.
|
|
STRATEGY_RANK_MOMENTUM_WEIGHT = 0.8
|
|
STRATEGY_RANK_VOL_WEIGHT = 1.0 - STRATEGY_RANK_MOMENTUM_WEIGHT
|
|
|
|
|
|
def blend_strategy_rank(
|
|
momentum_percentile: float | None,
|
|
volatility_percentile: float | None,
|
|
*,
|
|
momentum_weight: float = STRATEGY_RANK_MOMENTUM_WEIGHT,
|
|
) -> float | None:
|
|
"""80/20 production rank with mom-only fallback when vol is missing.
|
|
|
|
Live and backtest must share this policy: missing vol must not send a
|
|
residual-qualified name to the bottom of the book (that was the old
|
|
backtest behaviour when either leg was None).
|
|
"""
|
|
if momentum_percentile is not None and volatility_percentile is not None:
|
|
vol_weight = 1.0 - momentum_weight
|
|
return round(
|
|
float(momentum_percentile) * momentum_weight
|
|
+ float(volatility_percentile) * vol_weight,
|
|
2,
|
|
)
|
|
return float(momentum_percentile) if momentum_percentile is not None else None
|
|
|
|
|
|
def compute_12_1_momentum(closes: list[float]) -> float | None:
|
|
"""Return over the window ending ~1 month ago, starting ~12 months ago.
|
|
None when there isn't a full year of history."""
|
|
if len(closes) >= _MOM_LOOKBACK + 1 and closes[-(_MOM_LOOKBACK + 1)] > 0:
|
|
return closes[-(_MOM_SKIP + 1)] / closes[-(_MOM_LOOKBACK + 1)] - 1.0
|
|
return None
|
|
|
|
|
|
def compute_residual_12_1_momentum(
|
|
dates: list[date],
|
|
closes: list[float],
|
|
benchmark_closes: dict[date, float],
|
|
) -> float | None:
|
|
"""12-1 momentum after removing linear benchmark exposure.
|
|
|
|
Estimate beta from daily stock/benchmark returns over the standard 12-1
|
|
formation window, then sum stock return minus beta * benchmark return. No
|
|
intercept is subtracted: fitting an intercept over the same window would make
|
|
residuals sum to roughly zero and destroy the ranking signal.
|
|
"""
|
|
i = len(closes) - 1
|
|
if not benchmark_closes or len(dates) != len(closes) or i - _MOM_LOOKBACK < 0:
|
|
return None
|
|
|
|
stock_rets: list[float] = []
|
|
market_rets: list[float] = []
|
|
for k in range(i - _MOM_LOOKBACK + 1, i - _MOM_SKIP + 1):
|
|
prev_close = closes[k - 1]
|
|
bench_prev = benchmark_closes.get(dates[k - 1])
|
|
bench_cur = benchmark_closes.get(dates[k])
|
|
if prev_close <= 0 or bench_prev is None or bench_cur is None or bench_prev <= 0:
|
|
continue
|
|
stock_rets.append(closes[k] / prev_close - 1.0)
|
|
market_rets.append(bench_cur / bench_prev - 1.0)
|
|
|
|
if len(stock_rets) < 100:
|
|
return None
|
|
mean_market = sum(market_rets) / len(market_rets)
|
|
mean_stock = sum(stock_rets) / len(stock_rets)
|
|
var_market = sum((x - mean_market) ** 2 for x in market_rets)
|
|
if var_market <= 0:
|
|
return None
|
|
cov = sum(
|
|
(stock_rets[k] - mean_stock) * (market_rets[k] - mean_market)
|
|
for k in range(len(stock_rets))
|
|
)
|
|
beta = cov / var_market
|
|
return sum(stock_rets[k] - beta * market_rets[k] for k in range(len(stock_rets)))
|
|
|
|
|
|
async def _load_activation_benchmark(db: AsyncSession) -> dict[date, float]:
|
|
"""Load SPY closes for residual momentum; refresh once if the table is empty."""
|
|
try:
|
|
from app.services.benchmark_service import load_benchmark_closes, refresh_benchmark_prices
|
|
|
|
closes = await load_benchmark_closes(db)
|
|
if closes:
|
|
return closes
|
|
await refresh_benchmark_prices(db)
|
|
return await load_benchmark_closes(db)
|
|
except Exception:
|
|
logger.exception("Residual momentum benchmark load failed; falling back to raw momentum")
|
|
return {}
|
|
|
|
|
|
async def compute_momentum_percentiles(db: AsyncSession) -> dict[str, float]:
|
|
"""Momentum leg only — thin view of ``compute_activation_ranks``.
|
|
|
|
Prefer ``compute_activation_ranks`` in new code (includes vol + strategy_rank).
|
|
Kept so tests/helpers that only need the residual/raw percentile map stay simple.
|
|
"""
|
|
ranks = await compute_activation_ranks(db)
|
|
return {
|
|
sym: float(row["momentum_percentile"])
|
|
for sym, row in ranks.items()
|
|
if row.get("momentum_percentile") is not None
|
|
}
|
|
|
|
|
|
def compute_realized_vol_6m(closes: list[float]) -> float | None:
|
|
"""126-trading-day realized daily volatility. Higher = more volatile."""
|
|
if len(closes) < 127:
|
|
return None
|
|
rets = [
|
|
closes[k] / closes[k - 1] - 1.0
|
|
for k in range(len(closes) - 126, len(closes))
|
|
if closes[k - 1] > 0
|
|
]
|
|
if len(rets) < 2:
|
|
return None
|
|
mean = sum(rets) / len(rets)
|
|
var = sum((x - mean) ** 2 for x in rets) / (len(rets) - 1)
|
|
return var ** 0.5
|
|
|
|
|
|
def _percentiles(values: dict[str, float]) -> dict[str, float]:
|
|
ranked = sorted(values, key=lambda s: values[s])
|
|
n = len(ranked)
|
|
return {
|
|
sym: round((rank / (n - 1) * 100.0) if n > 1 else 100.0, 2)
|
|
for rank, sym in enumerate(ranked)
|
|
}
|
|
|
|
|
|
async def compute_activation_ranks(db: AsyncSession) -> dict[str, dict[str, float | None]]:
|
|
"""Compute production activation ranks for the live scanner.
|
|
|
|
``momentum_percentile`` remains the residual/raw 12-1 gate. ``strategy_rank``
|
|
is the promoted production ordering score: 80% activation momentum rank plus
|
|
20% 6-month realized-volatility percentile. Live ranks are universe-wide
|
|
before scanning; the research backtest ranked each weekly setup-candidate
|
|
cross-section, so this is the deliberate production approximation.
|
|
"""
|
|
result = await db.execute(
|
|
ticker_service.active_only(select(Ticker).order_by(Ticker.symbol))
|
|
)
|
|
tickers = list(result.scalars().all())
|
|
|
|
benchmark_closes = await _load_activation_benchmark(db)
|
|
using_residual = len(benchmark_closes) >= _MOM_LOOKBACK
|
|
|
|
momentum_values: dict[str, float] = {}
|
|
vol_values: dict[str, float] = {}
|
|
for ticker in tickers:
|
|
try:
|
|
records = await query_ohlcv(db, ticker.symbol)
|
|
except Exception:
|
|
logger.exception("Activation rank fetch failed for %s", ticker.symbol)
|
|
continue
|
|
closes = [float(r.close) for r in records]
|
|
momentum = (
|
|
compute_residual_12_1_momentum([r.date for r in records], closes, benchmark_closes)
|
|
if using_residual
|
|
else compute_12_1_momentum(closes)
|
|
)
|
|
if momentum is not None:
|
|
momentum_values[ticker.symbol] = momentum
|
|
vol = compute_realized_vol_6m(closes)
|
|
if vol is not None:
|
|
vol_values[ticker.symbol] = vol
|
|
|
|
momentum_percentiles = _percentiles(momentum_values)
|
|
vol_percentiles = _percentiles(vol_values)
|
|
symbols = set(momentum_percentiles) | set(vol_percentiles)
|
|
ranks: dict[str, dict[str, float | None]] = {}
|
|
for sym in symbols:
|
|
momentum_pct = momentum_percentiles.get(sym)
|
|
vol_pct = vol_percentiles.get(sym)
|
|
ranks[sym] = {
|
|
"momentum_percentile": momentum_pct,
|
|
"volatility_percentile": vol_pct,
|
|
"strategy_rank": blend_strategy_rank(momentum_pct, vol_pct),
|
|
}
|
|
|
|
logger.info(json.dumps({
|
|
"event": "activation_ranked",
|
|
"signal": "residual_12_1_plus_vol_80_20" if using_residual else "raw_12_1_plus_vol_80_20",
|
|
"tickers": len(ranks),
|
|
}))
|
|
return ranks
|