fix: isolate production universe in capacity research

This commit is contained in:
2026-08-05 21:00:19 +02:00
parent 23fe39fd78
commit 6fc82ae857
6 changed files with 349 additions and 97613 deletions
+156 -3
View File
@@ -1,6 +1,7 @@
from __future__ import annotations
import asyncio
import pickle
import sqlite3
from datetime import date, timedelta
@@ -13,12 +14,18 @@ from scripts.portfolio_capacity_research import (
bootstrap_median_interval,
build_cells,
build_cohort_manifest,
iqr,
summarize_simulation,
validate_cohort_manifest,
)
from scripts.run_portfolio_construction_matrix import (
CACHE_VERSION,
_assert_clean_worktree,
_build_candidate_cache,
_checkpoint_state,
_construction_candidate_view,
_construction_universe_errors,
_json_hash,
_load_snapshot,
_markdown,
_operational_summary,
@@ -126,11 +133,18 @@ def test_load_snapshot_accepts_pre_sec_ticker_schema(tmp_path, monkeypatch):
volume BIGINT NOT NULL,
created_at DATETIME NOT NULL
);
CREATE TABLE research_rank_only (
symbol VARCHAR(10) PRIMARY KEY
);
INSERT INTO tickers VALUES
(1, 'LEGACY', 'Legacy Co', '2024-01-01 00:00:00');
(1, 'LEGACY', 'Legacy Co', '2024-01-01 00:00:00'),
(2, 'RANK', 'Rank Only Co', '2024-01-01 00:00:00');
INSERT INTO ohlcv_records VALUES
(1, 1, '2024-01-02', 100, 102, 99, 101, 1000000,
'2024-01-02 00:00:00'),
(2, 2, '2024-01-02', 50, 51, 49, 50, 500000,
'2024-01-02 00:00:00');
INSERT INTO research_rank_only VALUES ('RANK');
'''
)
@@ -167,7 +181,8 @@ def test_load_snapshot_accepts_pre_sec_ticker_schema(tmp_path, monkeypatch):
loaded = asyncio.run(_load_snapshot(snapshot, quiet=True))
assert loaded['symbols'] == ['LEGACY']
assert loaded['symbols'] == ['LEGACY', 'RANK']
assert loaded['construction_symbols'] == {'LEGACY'}
assert loaded['prices']['LEGACY'] == (
[date(2024, 1, 2).toordinal()],
[100.0],
@@ -176,6 +191,11 @@ def test_load_snapshot_accepts_pre_sec_ticker_schema(tmp_path, monkeypatch):
[101.0],
[1_000_000],
)
assert loaded['prices']['RANK'][4] == [50.0]
assert loaded['construction_universe_manifest'][
'construction_ticker_rows'
] == 1
assert loaded['construction_universe_manifest']['rank_only_ticker_rows'] == 1
with sqlite3.connect(snapshot) as connection:
columns = {
row[1] for row in connection.execute('PRAGMA table_info(tickers)')
@@ -183,6 +203,111 @@ def test_load_snapshot_accepts_pre_sec_ticker_schema(tmp_path, monkeypatch):
assert {'cik', 'sic', 'sic_description'}.isdisjoint(columns)
def test_construction_view_filters_rank_only_rows_without_rebuilding_cache():
manifest = {
'ranking_ticker_rows': 506,
'ranking_symbols_with_prices': 506,
'construction_ticker_rows': 505,
'construction_symbols_with_prices': 505,
'rank_only_ticker_rows': 1,
'rank_only_symbols_with_prices': 1,
'rank_only_unknown_symbols': 0,
}
cached = {
'key': {'version': 'existing-broad-cache'},
'qualified_candidates': [
{'symbol': 'PROD', 'date': '2025-01-02'},
{'symbol': 'RANK', 'date': '2025-01-02'},
],
'qualified_long_count': 2,
'daily_rank_map': {
('RANK', '2025-01-02'): {'strategy_rank': 99.0},
},
}
view = _construction_candidate_view(
cached,
{
'construction_symbols': {'PROD'},
'construction_universe_manifest': manifest,
},
)
assert [row['symbol'] for row in view['qualified_candidates']] == ['PROD']
assert view['raw_full_universe_qualified_long_count'] == 2
assert view['filtered_rank_only_qualified_long_count'] == 1
assert view['qualified_long_count'] == 1
assert ('RANK', '2025-01-02') in view['daily_rank_map']
assert len(cached['qualified_candidates']) == 2
def test_existing_broad_candidate_cache_key_remains_reusable(tmp_path, monkeypatch):
snapshot = tmp_path / 'research.sqlite'
snapshot.write_bytes(b'snapshot-placeholder')
cache_path = tmp_path / 'broad-cache.pkl'
snapshot_data = {
'recommendation_config': {'rr': 3.0},
'activation': {'min_momentum_percentile': 80.0},
'runtime_config': {'ranking_key': 'test'},
'universe_manifest': {
'ticker_rows': 4655,
'symbols_with_prices': 4654,
'symbols_sha256': 'symbols',
},
}
key = {
'version': CACHE_VERSION,
'snapshot': str(snapshot.resolve()),
'snapshot_sha256': 'snapshot-hash',
'cadence': 'daily',
'outcome_horizon_sessions': 0,
'recommendation_config_hash': _json_hash(
snapshot_data['recommendation_config']
),
'activation_hash': _json_hash(snapshot_data['activation']),
'runtime_config': snapshot_data['runtime_config'],
'universe_manifest': snapshot_data['universe_manifest'],
}
cached = {'key': key, 'qualified_candidates': [{'symbol': 'PROD'}]}
cache_path.write_bytes(pickle.dumps(cached))
monkeypatch.setattr(
bt,
'_replay_candidates_for_period',
lambda *_args: pytest.fail('existing cache should avoid replay'),
)
loaded = _build_candidate_cache(
snapshot_data,
snapshot=snapshot,
snapshot_sha256='snapshot-hash',
cache_path=cache_path,
workers=1,
quiet=True,
)
assert loaded == cached
def test_construction_universe_guard_rejects_leaked_broad_book():
valid = {
'ranking_ticker_rows': 4655,
'construction_ticker_rows': 506,
'construction_symbols_with_prices': 506,
'rank_only_ticker_rows': 4149,
'rank_only_unknown_symbols': 0,
}
assert _construction_universe_errors(valid) == []
leaked = {
**valid,
'construction_ticker_rows': 4655,
'construction_symbols_with_prices': 4654,
'rank_only_ticker_rows': 0,
}
errors = _construction_universe_errors(leaked)
assert any('450-600' in error for error in errors)
def test_unbounded_count_and_effective_risk_floor():
start = date(2025, 1, 6)
ords = [start.toordinal() + offset for offset in range(4)]
@@ -504,6 +629,10 @@ def test_simple_cluster_bootstrap_is_deterministic_and_not_a_gate():
assert first['p05'] <= first['point'] <= first['p95']
def test_iqr_materializes_generator_before_both_quantiles():
assert iqr(value for value in (0.0, 1.0, 2.0, 3.0)) == pytest.approx(1.5)
def test_aggregate_reports_paired_years_and_separate_warm_iqrs():
cells: list[dict] = []
for cost in (0.1, 0.2):
@@ -579,16 +708,40 @@ def test_aggregate_reports_paired_years_and_separate_warm_iqrs():
)
assert set(cash_warm['headline']) == {'ev_net_r', 'calmar'}
assert 'D' not in cash_warm
assert cash_warm['headline']['ev_net_r']['median_iqr_ratio'] == pytest.approx(
1.0
)
assert cash_warm['headline']['calmar']['median_iqr_ratio'] == pytest.approx(
1.0
)
assert cash_warm['headline']['ev_net_r']['bootstrap_90']['n'] == 7
markdown = _markdown({
'generated_at': '2026-08-05T00:00:00Z',
'analysis': report,
'operational_summary': _operational_summary(cells),
'validation': {
'construction_universe_manifest': {
'construction_symbols_with_prices': 506,
'rank_only_symbols_with_prices': 4148,
'ranking_symbols_with_prices': 4654,
},
'candidate_rank_coverage': {
'construction_qualified_longs': 5000,
'filtered_rank_only_qualified_longs': 137000,
},
},
})
assert 'ΔGain-to-Pain' in markdown
assert '0.10% per fill' in markdown
assert '0.20% per fill' in markdown
assert 'Tradable setup symbols with prices: 506.' in markdown
assert 'Rank-only qualified rows removed: 137000.' in markdown
assert 'formal promotion gate' in markdown
def test_synthetic_worker_matrix_covers_four_arms_protocols_and_costs():
def test_synthetic_worker_matrix_covers_four_arms_protocols_and_costs(monkeypatch):
monkeypatch.setenv('BACKTEST_SNAPSHOT_OFFLINE', '0')
monkeypatch.setenv('BACKTEST_ALLOW_SPAWN', '0')
start = date(2025, 1, 6)
sessions = _business_days(start, date(2025, 1, 17))
ords = [session.toordinal() for session in sessions]