Compare commits
297
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6f1ee450f1 | ||
|
|
aa6cd5cac4 | ||
|
|
24482c62fe | ||
|
|
6fc82ae857 | ||
|
|
23fe39fd78 | ||
|
|
477aa4b2da | ||
|
|
e58d2bb2cf | ||
|
|
1ace6688dd | ||
|
|
07d864cf64 | ||
|
|
2435abacaf | ||
|
|
d29d603158 | ||
|
|
70157ccfc2 | ||
|
|
f49b422095 | ||
|
|
59ac108c90 | ||
|
|
e0f3d43efb | ||
|
|
4c0c0579f5 | ||
|
|
d1caac86b5 | ||
|
|
d431ee283d | ||
|
|
3a6900d45a | ||
|
|
7d703ea524 | ||
|
|
7bcdf77ef9 | ||
|
|
c8c660e63d | ||
|
|
f58f8b0818 | ||
|
|
862d1d536b | ||
|
|
5b4fdab85c | ||
|
|
a7aefa6fe7 | ||
|
|
d2a27d4a78 | ||
|
|
83c0555e52 | ||
|
|
019ca1342a | ||
|
|
49bf3b140e | ||
|
|
b0537ebe9a | ||
|
|
3df36a9bfb | ||
|
|
98793925ef | ||
|
|
2e083d3bb9 | ||
|
|
4135946cb3 | ||
|
|
3d42ca7241 | ||
|
|
77fa8b8c65 | ||
|
|
0e556d8a43 | ||
|
|
fae621475b | ||
|
|
e54f03cba6 | ||
|
|
921f3d06fb | ||
|
|
259001e419 | ||
|
|
ce8c60d957 | ||
|
|
361cfd7883 | ||
|
|
dba7ea739b | ||
|
|
34d6dda1ab | ||
|
|
7f944d718f | ||
|
|
eae4d34c06 | ||
|
|
ddc88b130b | ||
|
|
00c28ccb76 | ||
|
|
8be7635506 | ||
|
|
b68aff80a6 | ||
|
|
88fc90cc6f | ||
|
|
71d21a45b6 | ||
|
|
480baf762f | ||
|
|
103b182598 | ||
|
|
bd23f41a1d | ||
|
|
9172c1a699 | ||
|
|
fb6d39c68b | ||
|
|
b3dcf356a6 | ||
|
|
459a925e36 | ||
|
|
979b4047dc | ||
|
|
2038b84b72 | ||
|
|
256880899b | ||
|
|
a549942afe | ||
|
|
96d3b40560 | ||
|
|
8dcdcac2a6 | ||
|
|
f7ce85a33e | ||
|
|
4754dbc17b | ||
|
|
7413de9301 | ||
|
|
5939be7b7f | ||
|
|
cc67aebe61 | ||
|
|
b1397fa82e | ||
|
|
d923f41f85 | ||
|
|
18fca28b7c | ||
|
|
5d275c8df7 | ||
|
|
a4e33d7a39 | ||
|
|
54ae8ba153 | ||
|
|
e5a62ca648 | ||
|
|
1b821e1a1e | ||
|
|
35e44f681b | ||
|
|
fc192c9f74 | ||
|
|
8b45bdb361 | ||
|
|
febd741671 | ||
|
|
949cbbe7c0 | ||
|
|
e09ec4ab1d | ||
|
|
aaca82d9d7 | ||
|
|
1e072346a9 | ||
|
|
565484de87 | ||
|
|
05ba138d35 | ||
|
|
807cc4bdfa | ||
|
|
6a10c8ff09 | ||
|
|
247a92a89f | ||
|
|
ba2df8b9fd | ||
|
|
29715ef3d1 | ||
|
|
c7c60a64f2 | ||
|
|
1fa3d70dec | ||
|
|
86dc24ae8a | ||
|
|
bb8aa655a1 | ||
|
|
1c38a94dd0 | ||
|
|
a4d5ed7a93 | ||
|
|
9171e366ee | ||
|
|
9717d8176b | ||
|
|
01007fb6dd | ||
|
|
003f20de19 | ||
|
|
f3d1312a69 | ||
|
|
a9841d92b7 | ||
|
|
64761f38ba | ||
|
|
f6e0ca734f | ||
|
|
06cf054f60 | ||
|
|
32bf9c9297 | ||
|
|
fa25b6ee68 | ||
|
|
8f285acb00 | ||
|
|
2311999e57 | ||
|
|
7d60e54f5a | ||
|
|
ceaaadc49f | ||
|
|
d34c7a21b7 | ||
|
|
30286111a8 | ||
|
|
b6892d13fd | ||
|
|
c2c7244d1a | ||
|
|
9704e0d85a | ||
|
|
c2d29184dd | ||
|
|
dc08a805a8 | ||
|
|
d9c4cd35eb | ||
|
|
19d674ed62 | ||
|
|
a71dd4adb7 | ||
|
|
736451e26f | ||
|
|
5a61b164f6 | ||
|
|
99860dbd13 | ||
|
|
3eb6192a1e | ||
|
|
723d47338e | ||
|
|
529343ce82 | ||
|
|
cad4b49e7c | ||
|
|
b0e33e1606 | ||
|
|
e07da0f8f0 | ||
|
|
72f10917a8 | ||
|
|
5a531fd603 | ||
|
|
c9c6967c9c | ||
|
|
d13c54e3c7 | ||
|
|
d858475ddb | ||
|
|
5155d00d9e | ||
|
|
1a6f82bf6d | ||
|
|
5385f46064 | ||
|
|
cf294a7d2b | ||
|
|
bbc7383d3a | ||
|
|
9800114fc4 | ||
|
|
27bc8a6631 | ||
|
|
0cd9ee7689 | ||
|
|
13f57b2525 | ||
|
|
65a462271c | ||
|
|
bc50ba9136 | ||
|
|
1e9f2dc4fb | ||
|
|
f714782fa4 | ||
|
|
ced066dc01 | ||
|
|
81c6f5fe2f | ||
|
|
1d5b1489be | ||
|
|
fd21067a40 | ||
|
|
644e81d1a3 | ||
|
|
f59c0c3484 | ||
|
|
ed82d0a665 | ||
|
|
751d103f4e | ||
|
|
ac0bcf9012 | ||
|
|
cad1be96da | ||
|
|
f309bd5691 | ||
|
|
d5b0ebf895 | ||
|
|
8db535b889 | ||
|
|
cd7dc7973c | ||
|
|
bddaeb9110 | ||
|
|
9f06304100 | ||
|
|
bee5a5ce89 | ||
|
|
9d362bd568 | ||
|
|
8e09f239c8 | ||
|
|
1d84a40c04 | ||
|
|
39acf22808 | ||
|
|
7162272cc0 | ||
|
|
30fd5a029b | ||
|
|
623dc08875 | ||
|
|
3f86aec0be | ||
|
|
3999c5efc1 | ||
|
|
0873176f64 | ||
|
|
4730d19694 | ||
|
|
dc1570877c | ||
|
|
3ffd13ce6e | ||
|
|
b6df3b7ce7 | ||
|
|
8161c352a0 | ||
|
|
995a0380c3 | ||
|
|
f0e6a8fc19 | ||
|
|
b8ddf14e7d | ||
|
|
01e6f7e2c3 | ||
|
|
2e9afeec0f | ||
|
|
e6dc74df6f | ||
|
|
3280ed71f7 | ||
|
|
e31d1704a2 | ||
|
|
04ef7f44a2 | ||
|
|
1daf762bda | ||
|
|
f8e1107851 | ||
|
|
87e01e260b | ||
|
|
f7c2e35e29 | ||
|
|
37836f8824 | ||
|
|
b891122936 | ||
|
|
8df76d3288 | ||
|
|
93403b4d3a | ||
|
|
ce0df6a126 | ||
|
|
bd72fa75f9 | ||
|
|
178abd4b2d | ||
|
|
0ace776367 | ||
|
|
cb64f7bf65 | ||
|
|
4ca48ac590 | ||
|
|
681f0f95da | ||
|
|
19b81c169d | ||
|
|
57ac1d2cdd | ||
|
|
ea11efe3d1 | ||
|
|
906d1db7d1 | ||
|
|
9789e3d762 | ||
|
|
85b3ef618f | ||
|
|
fa26ec3ec4 | ||
|
|
88527f39b6 | ||
|
|
cb215e2595 | ||
|
|
727b147c81 | ||
|
|
292b9934b1 | ||
|
|
9450831ef3 | ||
|
|
25364f8e99 | ||
|
|
fdc49d0e28 | ||
|
|
8f411435ee | ||
|
|
924c474624 | ||
|
|
6464fdd4af | ||
|
|
c3dd80e203 | ||
|
|
79010a5dae | ||
|
|
06fdd92faa | ||
|
|
b0b691da9c | ||
|
|
72054d06b0 | ||
|
|
c7a198ba8e | ||
|
|
6e33897c1a | ||
|
|
16121042cd | ||
|
|
9f56a25f2d | ||
|
|
03bdcbd745 | ||
|
|
d806ede07e | ||
|
|
fc0945367a | ||
|
|
f9e12fc78c | ||
|
|
02cf9d1cba | ||
|
|
20f6981712 | ||
|
|
744ea4ddc4 | ||
|
|
ae1aeb3c84 | ||
|
|
65d2dae62a | ||
|
|
294d935030 | ||
|
|
61156684ff | ||
|
|
65335cf1f3 | ||
|
|
23d1db1f30 | ||
|
|
edc1a9757b | ||
|
|
ce6035ee3c | ||
|
|
2a4bdd16a8 | ||
|
|
02b28f5ea6 | ||
|
|
ca42e1b28d | ||
|
|
1155c9ed1b | ||
|
|
41470c32ec | ||
|
|
cfca59a2d4 | ||
|
|
5f2d108227 | ||
|
|
66ef0564c1 | ||
|
|
14327ab25a | ||
|
|
eaad935a2a | ||
|
|
d4ccea2d69 | ||
|
|
8c36cfcef1 | ||
|
|
ac51e23949 | ||
|
|
2b0068ae08 | ||
|
|
7fd34d6de8 | ||
|
|
8d5863bac4 | ||
|
|
be4d6a05ca | ||
|
|
aadec7d403 | ||
|
|
849489a4b5 | ||
|
|
80b4113280 | ||
|
|
13374087db | ||
|
|
1e82dfad7f | ||
|
|
29a61cb2ca | ||
|
|
243e369e9a | ||
|
|
0f43e755f4 | ||
|
|
942a22ce65 | ||
|
|
8750aac6d9 | ||
|
|
29b1a9a28c | ||
|
|
84ce7c5c26 | ||
|
|
da0bb3367e | ||
|
|
a9f4686157 | ||
|
|
94ed3207d7 | ||
|
|
5442b62495 | ||
|
|
f61e11adea | ||
|
|
1566b84379 | ||
|
|
ab9ce18809 | ||
|
|
c5f6b07a3e | ||
|
|
c63951ca02 | ||
|
|
6511a1020b | ||
|
|
20a1c143f3 | ||
|
|
6c2e45377c | ||
|
|
7e9a6cd7ec | ||
|
|
8bcbbfcfd0 | ||
|
|
0627787bfc | ||
|
|
30effa89b7 | ||
|
|
4a96f85cd9 | ||
|
|
146dadf06f |
+30
-1
@@ -27,8 +27,37 @@ FINNHUB_API_KEY=
|
|||||||
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
||||||
ALPHA_VANTAGE_API_KEY=
|
ALPHA_VANTAGE_API_KEY=
|
||||||
|
|
||||||
|
# Dolt bulk data — local clone of post-no-preference/earnings (workstream A).
|
||||||
|
# DOLT_BINARY: path to the dolt CLI (set the full path in dev if it's not on PATH,
|
||||||
|
# e.g. Windows: C:\Program Files\Dolt\bin\dolt.exe). DOLT_DATA_DIR holds the
|
||||||
|
# clones; in PRODUCTION it MUST be outside the deploy tree (deploy is
|
||||||
|
# rsync --delete) — e.g. /var/lib/signal-platform/dolt. The earnings clone lives
|
||||||
|
# at <DOLT_DATA_DIR>/<DOLT_EARNINGS_SUBDIR>. Production setup is automated by
|
||||||
|
# deploy/provision_fundamentals.sh; see docs/fundamentals-deployment.md.
|
||||||
|
DOLT_BINARY=dolt
|
||||||
|
DOLT_DATA_DIR=dolt-data
|
||||||
|
DOLT_EARNINGS_SUBDIR=earnings
|
||||||
|
# Free-space floor checked before a pull (clone is ~1.7 GB and grows). 5 GB is a
|
||||||
|
# safe production default; lower only on a space-constrained dev box.
|
||||||
|
DOLT_MIN_FREE_DISK_GB=5.0
|
||||||
|
# Hard timeout (s) on each dolt subprocess so a hung pull/sql can't pin the
|
||||||
|
# import connection + advisory lock.
|
||||||
|
DOLT_COMMAND_TIMEOUT_SECONDS=600.0
|
||||||
|
|
||||||
|
# SEC EDGAR (fundamentals, workstream A). SEC fair-access REQUIRES an identifying
|
||||||
|
# User-Agent with a REAL contact email — set it, or requests get 403'd. Stay well
|
||||||
|
# under 10 req/s (spacing below).
|
||||||
|
SEC_USER_AGENT=signal-platform/1.0 (contact: you@example.com)
|
||||||
|
SEC_REQUEST_SPACING_SECONDS=0.2
|
||||||
|
SEC_MAX_RETRIES=4
|
||||||
|
SEC_REQUEST_TIMEOUT_SECONDS=30.0
|
||||||
|
|
||||||
|
# A5 read-only parity report archive. In production keep this outside the
|
||||||
|
# rsync deployment tree, e.g. /var/lib/signal-platform/reports/fundamentals-parity.
|
||||||
|
FUNDAMENTALS_PARITY_REPORT_DIR=reports/fundamentals-parity
|
||||||
|
|
||||||
# Regime Monitor — FRED (VIX + HY credit spreads). Free key: https://fred.stlouisfed.org/docs/api/api_key.html
|
# Regime Monitor — FRED (VIX + HY credit spreads). Free key: https://fred.stlouisfed.org/docs/api/api_key.html
|
||||||
# Optional: without it the VIX (P5) and credit-spread (F2) signals show as n/a.
|
# Optional: without it the volatility (V1) and credit (C1) pillars show as n/a.
|
||||||
FRED_API_KEY=
|
FRED_API_KEY=
|
||||||
|
|
||||||
# Scheduled Jobs
|
# Scheduled Jobs
|
||||||
|
|||||||
+54
-29
@@ -22,6 +22,12 @@ on:
|
|||||||
type: boolean
|
type: boolean
|
||||||
default: false
|
default: false
|
||||||
|
|
||||||
|
# Serialize deploys so two quick pushes to main can't rsync/restart on top of
|
||||||
|
# each other. Don't cancel an in-flight deploy mid-restart.
|
||||||
|
concurrency:
|
||||||
|
group: deploy-main
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
lint:
|
lint:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
@@ -29,7 +35,8 @@ jobs:
|
|||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
- uses: actions/setup-python@v5
|
- uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: "3.11"
|
python-version: "3.12"
|
||||||
|
cache: "pip"
|
||||||
- run: pip install ruff
|
- run: pip install ruff
|
||||||
- run: ruff check app/
|
- run: ruff check app/
|
||||||
|
|
||||||
@@ -52,17 +59,21 @@ jobs:
|
|||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
- uses: actions/setup-python@v5
|
- uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: "3.11"
|
python-version: "3.12"
|
||||||
|
cache: "pip"
|
||||||
- uses: actions/setup-node@v4
|
- uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
node-version: "20"
|
node-version: "20"
|
||||||
|
cache: "npm"
|
||||||
|
cache-dependency-path: frontend/package-lock.json
|
||||||
- run: pip install -e ".[dev]"
|
- run: pip install -e ".[dev]"
|
||||||
|
# The Postgres service exists only to validate the migrations against real
|
||||||
|
# Postgres (what prod runs). The test suite itself uses an in-memory SQLite
|
||||||
|
# engine (tests/conftest.py), so pytest doesn't touch this service.
|
||||||
- run: alembic upgrade head
|
- run: alembic upgrade head
|
||||||
env:
|
env:
|
||||||
DATABASE_URL: postgresql+asyncpg://test_user:test_pass@postgres:5432/test_db
|
DATABASE_URL: postgresql+asyncpg://test_user:test_pass@postgres:5432/test_db
|
||||||
- run: pytest --tb=short
|
- run: pytest --tb=short
|
||||||
env:
|
|
||||||
DATABASE_URL: postgresql+asyncpg://test_user:test_pass@postgres:5432/test_db
|
|
||||||
- run: |
|
- run: |
|
||||||
cd frontend
|
cd frontend
|
||||||
npm ci
|
npm ci
|
||||||
@@ -76,37 +87,43 @@ jobs:
|
|||||||
deploy:
|
deploy:
|
||||||
needs: test
|
needs: test
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
|
env:
|
||||||
|
DEPLOY_HOST: ${{ vars.DEPLOY_HOST }}
|
||||||
|
DEPLOY_USER: ${{ vars.DEPLOY_USER }}
|
||||||
|
DEPLOY_PATH: ${{ vars.DEPLOY_PATH }}
|
||||||
|
SSH_PRIVATE_KEY: ${{ secrets.SSH_PRIVATE_KEY }}
|
||||||
|
SSH_KNOWN_HOSTS: ${{ vars.SSH_KNOWN_HOSTS }}
|
||||||
|
SSH_PORT: ${{ vars.SSH_PORT || '22' }}
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v4
|
- uses: actions/checkout@v4
|
||||||
- uses: actions/setup-node@v4
|
- uses: actions/setup-node@v4
|
||||||
with:
|
with:
|
||||||
node-version: "20"
|
node-version: "20"
|
||||||
|
cache: "npm"
|
||||||
|
cache-dependency-path: frontend/package-lock.json
|
||||||
|
|
||||||
- name: Build frontend
|
- name: Build frontend
|
||||||
run: |
|
run: |
|
||||||
cd frontend
|
cd frontend
|
||||||
npm ci
|
npm ci
|
||||||
npm run build
|
npm run build
|
||||||
- name: Deploy to server
|
|
||||||
env:
|
|
||||||
DEPLOY_HOST: ${{ vars.DEPLOY_HOST }}
|
|
||||||
DEPLOY_USER: ${{ vars.DEPLOY_USER }}
|
|
||||||
DEPLOY_PATH: ${{ vars.DEPLOY_PATH }}
|
|
||||||
SSH_PRIVATE_KEY: ${{ secrets.SSH_PRIVATE_KEY }}
|
|
||||||
SSH_KNOWN_HOSTS: ${{ vars.SSH_KNOWN_HOSTS }}
|
|
||||||
SSH_PORT: ${{ vars.SSH_PORT || '22' }}
|
|
||||||
run: |
|
|
||||||
# Install tools missing from runner image
|
|
||||||
sudo apt-get update -qq && sudo apt-get install -y -qq rsync openssh-client > /dev/null 2>&1 || true
|
|
||||||
|
|
||||||
# Write SSH credentials
|
- name: Install deploy tools
|
||||||
|
run: sudo apt-get update -qq && sudo apt-get install -y -qq rsync openssh-client > /dev/null 2>&1 || true
|
||||||
|
|
||||||
|
- name: Set up SSH
|
||||||
|
run: |
|
||||||
mkdir -p ~/.ssh
|
mkdir -p ~/.ssh
|
||||||
|
chmod 700 ~/.ssh
|
||||||
echo "$SSH_PRIVATE_KEY" > ~/.ssh/deploy_key
|
echo "$SSH_PRIVATE_KEY" > ~/.ssh/deploy_key
|
||||||
chmod 600 ~/.ssh/deploy_key
|
chmod 600 ~/.ssh/deploy_key
|
||||||
echo "$SSH_KNOWN_HOSTS" >> ~/.ssh/known_hosts
|
echo "$SSH_KNOWN_HOSTS" >> ~/.ssh/known_hosts
|
||||||
|
# known_hosts is supplied, so verify the host key instead of blindly
|
||||||
|
# trusting it (StrictHostKeyChecking=no would defeat the fingerprint).
|
||||||
|
echo "SSH_OPTS=-i $HOME/.ssh/deploy_key -o StrictHostKeyChecking=yes -p ${SSH_PORT}" >> "$GITHUB_ENV"
|
||||||
|
|
||||||
SSH_OPTS="-i ~/.ssh/deploy_key -o StrictHostKeyChecking=no -p $SSH_PORT"
|
- name: Sync files to server
|
||||||
|
run: |
|
||||||
# Sync application files
|
|
||||||
rsync -avz --delete \
|
rsync -avz --delete \
|
||||||
--exclude '.git/' \
|
--exclude '.git/' \
|
||||||
--exclude '.gitea/' \
|
--exclude '.gitea/' \
|
||||||
@@ -118,10 +135,11 @@ jobs:
|
|||||||
--exclude '*.pyc' \
|
--exclude '*.pyc' \
|
||||||
--exclude 'frontend/node_modules/' \
|
--exclude 'frontend/node_modules/' \
|
||||||
-e "ssh $SSH_OPTS" \
|
-e "ssh $SSH_OPTS" \
|
||||||
./ ${DEPLOY_USER}@${DEPLOY_HOST}:${DEPLOY_PATH}/
|
./ "${DEPLOY_USER}@${DEPLOY_HOST}:${DEPLOY_PATH}/"
|
||||||
|
|
||||||
# Install deps & restart on server
|
- name: Install deps & run migrations
|
||||||
ssh $SSH_OPTS ${DEPLOY_USER}@${DEPLOY_HOST} << REMOTE_SCRIPT
|
run: |
|
||||||
|
ssh $SSH_OPTS "${DEPLOY_USER}@${DEPLOY_HOST}" << REMOTE_SCRIPT
|
||||||
set -e
|
set -e
|
||||||
cd ${DEPLOY_PATH}
|
cd ${DEPLOY_PATH}
|
||||||
|
|
||||||
@@ -141,12 +159,19 @@ jobs:
|
|||||||
else
|
else
|
||||||
alembic upgrade head
|
alembic upgrade head
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Restart service
|
|
||||||
sudo systemctl restart signalplatform.service
|
|
||||||
echo "✓ signalplatform deployed"
|
|
||||||
|
|
||||||
REMOTE_SCRIPT
|
REMOTE_SCRIPT
|
||||||
|
|
||||||
# Cleanup
|
- name: Restart service & health check
|
||||||
rm -f ~/.ssh/deploy_key
|
run: |
|
||||||
|
ssh $SSH_OPTS "${DEPLOY_USER}@${DEPLOY_HOST}" << REMOTE_SCRIPT
|
||||||
|
set -e
|
||||||
|
sudo systemctl restart signalplatform.service
|
||||||
|
sleep 3
|
||||||
|
curl -fsS http://127.0.0.1:8998/api/v1/health > /dev/null \
|
||||||
|
|| { echo "✗ health check failed after restart"; exit 1; }
|
||||||
|
echo "✓ signalplatform deployed"
|
||||||
|
REMOTE_SCRIPT
|
||||||
|
|
||||||
|
- name: Clean up SSH key
|
||||||
|
if: always()
|
||||||
|
run: rm -f ~/.ssh/deploy_key
|
||||||
|
|||||||
+21
@@ -17,9 +17,13 @@ build/
|
|||||||
# IDE
|
# IDE
|
||||||
.vscode/
|
.vscode/
|
||||||
.idea/
|
.idea/
|
||||||
|
.claude/settings.local.json
|
||||||
*.swp
|
*.swp
|
||||||
*.swo
|
*.swo
|
||||||
|
|
||||||
|
# Local AI tool metadata
|
||||||
|
mcps/
|
||||||
|
|
||||||
# OS
|
# OS
|
||||||
.DS_Store
|
.DS_Store
|
||||||
Thumbs.db
|
Thumbs.db
|
||||||
@@ -27,9 +31,26 @@ Thumbs.db
|
|||||||
# Frontend
|
# Frontend
|
||||||
frontend/node_modules/
|
frontend/node_modules/
|
||||||
frontend/dist/
|
frontend/dist/
|
||||||
|
frontend/tsconfig.tsbuildinfo
|
||||||
|
|
||||||
# Alembic
|
# Alembic
|
||||||
alembic/versions/__pycache__/
|
alembic/versions/__pycache__/
|
||||||
|
|
||||||
# Generated SSL bundle
|
# Generated SSL bundle
|
||||||
combined-ca-bundle.pem
|
combined-ca-bundle.pem
|
||||||
|
|
||||||
|
# Dolt local dev clones. Production keeps clones in DOLT_DATA_DIR OUTSIDE the
|
||||||
|
# repo tree (deploy is rsync --delete of the tree); this dir is dev-only.
|
||||||
|
dolt-data/
|
||||||
|
|
||||||
|
# Local research artifacts
|
||||||
|
# Backtest reports in reports/ are tracked: they are the evidence behind the
|
||||||
|
# production baseline in the README. The snapshot DBs they run against are not.
|
||||||
|
backtest_snapshots/
|
||||||
|
# Rebuildable pickle caches are local accelerators, not decision evidence.
|
||||||
|
reports/*.pkl
|
||||||
|
reports/*.pk1
|
||||||
|
reports/.cache/
|
||||||
|
# Runtime A5 parity bundles are generated on the production server. Research
|
||||||
|
# conclusions belong in docs/research, not as an ever-growing artifact archive.
|
||||||
|
reports/fundamentals-parity/
|
||||||
|
|||||||
@@ -0,0 +1,24 @@
|
|||||||
|
Third-party data attribution
|
||||||
|
============================
|
||||||
|
|
||||||
|
Earnings calendar and EPS history
|
||||||
|
---------------------------------
|
||||||
|
This application ingests the earnings calendar and EPS surprise history from the
|
||||||
|
public DoltHub repository:
|
||||||
|
|
||||||
|
post-no-preference/earnings
|
||||||
|
https://www.dolthub.com/repositories/post-no-preference/earnings
|
||||||
|
|
||||||
|
Licensed under Creative Commons Attribution-ShareAlike 4.0 International
|
||||||
|
(CC BY-SA 4.0): https://creativecommons.org/licenses/by-sa/4.0/
|
||||||
|
|
||||||
|
Use in this project: private, internal ingestion only. The data is normalized
|
||||||
|
into PostgreSQL (`earnings_events`) — the announcement calendar is aligned to the
|
||||||
|
EPS history via a minimum-cost monotonic pairing, symbols are normalized, and the
|
||||||
|
session field is mapped to bmo/amc/unknown. No public API, bulk export, or
|
||||||
|
redistribution of the data is provided. This attribution and the upstream license
|
||||||
|
are preserved per the CC BY-SA 4.0 terms. Re-review licensing before any public
|
||||||
|
or commercial access.
|
||||||
|
|
||||||
|
The post-no-preference/stocks repository (workstream B) is not used at this time
|
||||||
|
and would be reviewed separately.
|
||||||
@@ -1,8 +1,299 @@
|
|||||||
# Signal Dashboard
|
# Signal Dashboard
|
||||||
|
|
||||||
Investing-signal platform for NASDAQ stocks. Surfaces the best trading opportunities through weighted multi-dimensional scoring — technical indicators, support/resistance quality, sentiment, fundamentals, and momentum — with asymmetric risk:reward scanning.
|
Investing-signal platform for US equities. It runs one strategy, and it is a boring one:
|
||||||
|
|
||||||
**Philosophy:** Don't predict price. Find the path of least resistance, key S/R zones, and asymmetric R:R setups.
|
> **A long-only cross-sectional momentum book.** Buy the top quintile by beta-adjusted 12-1 month momentum, tilt toward higher volatility, hold at most 10 names, cut at 1.5× ATR, then trail at 3× ATR for up to 30 trading days. After an initial-stop exit, re-enter only after the gate has failed and subsequently qualified again.
|
||||||
|
|
||||||
|
**Philosophy:** don't predict price — rank it. The edge is *relative* strength across the universe, and the discipline is in the exit: cut losers fast, let winners run until the trail catches them.
|
||||||
|
|
||||||
|
**What is NOT the edge — read this before trusting a number on screen.** The composite score, the 5 dimensions, sentiment, fundamentals, and Structural S/R are **display context**, not validated predictors. The Gate Target Ladder is screening machinery that preserves the production setup population; it is not a claim about true market structure. In particular:
|
||||||
|
|
||||||
|
- **The headline "target" is not an exit.** It comes from the internal **Gate Target Ladder** and exists only to compute the R:R and reach-probability used by the activation gate. Human-facing chart S/R is a separate model. The live exit reads neither. Across 472 trades in the current daily gate-reset replay, the exit reasons were **229 initial stop, 147 trailing stop, 96 max hold — and 0 targets.** Honoring the target as a take-profit was tested and *halves CAGR* ([research](docs/research/sr-levels-and-exits.md)).
|
||||||
|
- **The composite score does not select trades.** Residual momentum does.
|
||||||
|
|
||||||
|
Full experiment log — everything tested, kept, and rejected: **[docs/research/](docs/research/README.md)**.
|
||||||
|
|
||||||
|
## The strategy, end to end
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
flowchart TD
|
||||||
|
U["Universe — ~500 tickers<br/>daily OHLCV"] --> M["Residual 12-1 momentum<br/><i>12-month return, skip last month,<br/>beta-adjusted vs SPY</i>"]
|
||||||
|
M --> R["Rank cross-sectionally<br/>into percentiles"]
|
||||||
|
R --> G1{"Top 20%?<br/>percentile ≥ 80"}
|
||||||
|
G1 -->|no| SKIP["Not traded<br/><i>(still scored — the control group)</i>"]
|
||||||
|
G1 -->|yes| S["Build the setup<br/>entry = last close<br/><b>stop = entry − 1.5 × ATR</b><br/>primary = Gate Target Ladder proposal"]
|
||||||
|
|
||||||
|
S --> G2{"Activation gate"}
|
||||||
|
G2 --> G2a["R:R ≥ 2.0 <i>(to the primary gate target)</i>"]
|
||||||
|
G2 --> G2b["touch odds ≥ 20%"]
|
||||||
|
G2 --> G2c["action not NEUTRAL<br/>and matches direction"]
|
||||||
|
G2a & G2b & G2c --> Q{"qualified?"}
|
||||||
|
Q -->|no| SKIP
|
||||||
|
Q -->|yes| RANK["Rank by production score<br/>80% momentum %ile<br/>+ 20% volatility %ile"]
|
||||||
|
|
||||||
|
RANK --> BOOK{"Room in the book?<br/>max 10 positions"}
|
||||||
|
BOOK -->|no| WAIT["Wait for a slot"]
|
||||||
|
BOOK -->|yes| OPEN["OPEN — size at 1% account risk"]
|
||||||
|
|
||||||
|
OPEN --> EXIT{"Exit — whichever comes first"}
|
||||||
|
EXIT --> E1["Initial stop hit<br/>entry − 1.5 × ATR → −1R<br/><b>49% of trades</b>"]
|
||||||
|
EXIT --> E2["Trailing stop hit<br/>highest close − 3 × ATR<br/><i>only binds once price is ~1R up</i><br/><b>31% of trades</b>"]
|
||||||
|
EXIT --> E3["Max hold reached<br/>30 trading days<br/><b>20% of trades</b>"]
|
||||||
|
EXIT -.->|"NEVER"| E4["Gate Target Ladder target<br/><b>0% of trades</b>"]
|
||||||
|
|
||||||
|
E1 --> LOCK["Re-entry locked"]
|
||||||
|
LOCK --> GF{"Later daily scan<br/>fails the gate?"}
|
||||||
|
GF -->|no| LOCK
|
||||||
|
GF -->|yes| GQ{"A subsequent daily scan<br/>qualifies again?"}
|
||||||
|
GQ -->|no| GQ
|
||||||
|
GQ -->|yes| RANK
|
||||||
|
|
||||||
|
style M fill:#1e3a5f,color:#fff
|
||||||
|
style OPEN fill:#1e4d2b,color:#fff
|
||||||
|
style E4 fill:#2a2a2a,color:#888
|
||||||
|
style E1 fill:#4a1f1f,color:#fff
|
||||||
|
style E2 fill:#1e4d2b,color:#fff
|
||||||
|
style LOCK fill:#4a351f,color:#fff
|
||||||
|
```
|
||||||
|
|
||||||
|
**How to read the exit box.** The initial stop is tight (1.5× ATR) and the trail is wide (3× ATR), so the trail sits *below* the initial stop at entry and only takes over once price has advanced roughly 1R. Cut fast when wrong; give room once right. That asymmetry is what produces the right-tailed return profile the strategy depends on — most trades lose a little (win rate 36.2%), a few win big (best trade +12.0R), and *that is why there is no take-profit*.
|
||||||
|
|
||||||
|
**What happens after an initial stop.** The stop always closes the trade and realizes its costs. The ticker is then locked until a successful daily full-universe scan first observes it outside the production gate and a later scan observes a fresh qualification. A continuously qualified ticker therefore cannot generate an immediate duplicate entry. Other exit reasons do not start this reset. See the [daily post-stop re-entry study](docs/research/post-stop-reentry.md).
|
||||||
|
|
||||||
|
**Live timing matters.** The **only** full-universe R:R scan runs near the US close (~15:30 ET), then Telegram alerts fire immediately so manual fills can still hit MOC. Outcome eval runs later (~16:45 ET) after a fresh OHLCV fetch of the final bar. Morning jobs refresh data/sentiment/regime without scanning. Stops closed by earlier same-day intraday outcome evals can get a **same-day** fail observation at the near-close scan — closer to the promoted research `gate_reset` arm than the old morning-scan `strict_gate_reset` analogue. Stops after the bell still need a later day. Same-day fail+qualify cannot unlock: `trade_policy` requires the failure to fall on an earlier America/New_York trading date.
|
||||||
|
|
||||||
|
## How It Works
|
||||||
|
|
||||||
|
Scheduled pipelines turn raw prices into a ranked, gated list of tradeable setups. Everything downstream of OHLCV is recomputed from stored data, so each refresh is cheap and idempotent. Job timing is cron-based and configurable in **Admin → Jobs** (default timezone **America/New_York** so the near-close scan tracks the cash close through DST).
|
||||||
|
|
||||||
|
### Price-level architecture: two different jobs
|
||||||
|
|
||||||
|
The platform deliberately has two price-level components. Calling both of them
|
||||||
|
"S/R" hid an important distinction, so the internal screening component is now
|
||||||
|
named the **Gate Target Ladder (GTL)**.
|
||||||
|
|
||||||
|
| Component | Purpose | Lifetime | Consumed by |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **Structural S/R** | A small set of meaningful support/resistance zones for humans | Persisted as `SRLevel` | Charts and alerts |
|
||||||
|
| **Gate Target Ladder** | A broad set of price proposals that preserves the validated setup screen | Built transiently per scan; never persisted as S/R | Target table, headline R:R and activation gate |
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
flowchart TD
|
||||||
|
O["Ticker OHLCV history"] --> SR["Structural S/R detector<br/>volume peaks + prominent pivots + round numbers<br/>rejection and recency strength"]
|
||||||
|
SR --> DB[("Persisted SRLevel rows")]
|
||||||
|
DB --> UI["Charts and alerts"]
|
||||||
|
|
||||||
|
O --> GTL["Gate Target Ladder<br/>20 price-range centers + 5-bar pivots<br/>no volume calculation"]
|
||||||
|
GTL --> TRAFFIC["Score historical price traffic<br/>merge nearby proposals and tag side"]
|
||||||
|
TRAFFIC --> HAS{"Any directional proposal<br/>with R:R ≥ 1.5?"}
|
||||||
|
HAS -->|no| NONE["No setup for that direction"]
|
||||||
|
HAS -->|yes| TARGETS["Build up to 5 target candidates<br/>estimate reach-probability"]
|
||||||
|
TARGETS --> PRIMARY["Headline target<br/>most likely candidate clearing<br/>R:R ≥ 1.5 and probability ≥ 20%"]
|
||||||
|
PRIMARY --> GATE{"Live activation gate<br/>headline R:R ≥ 2.0<br/>probability ≥ 20%<br/>momentum and direction pass?"}
|
||||||
|
GATE -->|no| OBS["Keep as unqualified observation"]
|
||||||
|
GATE -->|yes| BOOK["Eligible for production ranking/book"]
|
||||||
|
BOOK --> EXIT["Exit only by ATR stop/trail<br/>or max hold — never by target"]
|
||||||
|
```
|
||||||
|
|
||||||
|
The Gate Target Ladder works step by step:
|
||||||
|
|
||||||
|
1. Build 20 evenly spaced centers over the ticker's observed high/low range and add unfiltered five-bar swing pivots. Volume is not used.
|
||||||
|
2. Count how often historical bars pass through each proposal, convert that traffic to strength, merge proposals within 0.5%, and label them above/below spot.
|
||||||
|
3. For each direction, require at least one proposal with scanner R:R ≥ 1.5 against the 1.5× ATR initial stop.
|
||||||
|
4. Collapse nearby proposals into target zones, discard unsuitable ATR distances, and retain up to five candidates spanning near to far.
|
||||||
|
5. Estimate each candidate's probability of reaching the target before the stop. The headline target is the most likely candidate with R:R ≥ 1.5 and probability ≥ 20%; if none clears both, the most likely candidate remains headline so a distant lottery target cannot game the gate.
|
||||||
|
6. Apply the separate live activation floor to that headline target: production requires R:R ≥ 2.0 and probability ≥ 20%, plus the momentum/direction rules.
|
||||||
|
7. If traded, ignore the target for exits. The initial ATR stop, 3× ATR trail and maximum hold remain authoritative.
|
||||||
|
|
||||||
|
The ladder is intentionally broad and mechanical. It is not presented as
|
||||||
|
market structure, and its transient negative level IDs must never be stored as
|
||||||
|
chart S/R. The full-period parity run reproduced all 202,765 backtest setup
|
||||||
|
candidates, all 1,086 qualified setups, and the production book exactly
|
||||||
|
(Sharpe 2.03, CAGR 50.0%, max drawdown 21.4%, 321 trades). See the
|
||||||
|
[S/R and Gate Target Ladder research](docs/research/sr-levels-and-exits.md#explicit-gate-target-ladder).
|
||||||
|
|
||||||
|
**Ticker-chart diagnostic.** The optional **GTL traffic** toggle draws a
|
||||||
|
right-edge horizontal profile aligned to the price axis. It borrows the visual
|
||||||
|
grammar of a volume profile, but not its meaning: bar width is relative
|
||||||
|
historical OHLCV-bar crossings at each GTL proposal, not traded volume at that
|
||||||
|
price. Hover a bar to inspect its price, crossing count, strength and source.
|
||||||
|
The violet profile is deliberately distinct from the Structural S/R lines and
|
||||||
|
is off by default; it is a research aid, not another trade overlay.
|
||||||
|
|
||||||
|
Below the chart, the **Production rank** strip makes the current 80/20 ordering
|
||||||
|
snapshot explicit: a blue residual-momentum contribution and amber realized-
|
||||||
|
volatility contribution add to the stored strategy rank, while separate
|
||||||
|
percentile rails show each input. Only momentum carries the live activation-
|
||||||
|
gate marker. These are cross-sectional scan percentiles, not historical chart
|
||||||
|
indicators.
|
||||||
|
|
||||||
|
### Pipelines (America/New_York)
|
||||||
|
|
||||||
|
**Morning** (~02:00 ET) — data and display only, **no** qualifying R:R scan:
|
||||||
|
|
||||||
|
1. **OHLCV** — latest daily bars (Alpaca); new tickers backfill ~5 years.
|
||||||
|
2. **Sentiment** — stale names that matter (top-pick feeders, watchlist, open paper, discovery net). Display context only; the activation gate is price-only.
|
||||||
|
3. **Market Regime** + **Regime Monitor** — breadth/trend and the v3 risk thermometer; feed no trades.
|
||||||
|
4. **Telegram alerts** — change-driven (regime-quadrant etc.); quiet days stay quiet. Setup alerts still fire on the near-close pipeline after the scan.
|
||||||
|
|
||||||
|
**Near-close** (~15:30 ET Mon–Fri) — the only full-universe qualifying observation:
|
||||||
|
|
||||||
|
1. **OHLCV fetch** — refresh the in-progress day-t bar (same path as intraday).
|
||||||
|
2. **R:R Scan** — Structural S/R, scores, Gate Target Ladder setups, residual 12‑1 + 80/20 rank. Advances post-stop gate-reset transitions; failed scans never count.
|
||||||
|
3. **Telegram alerts** — chained immediately so manual MOC fills can still hit ~15:50/15:55.
|
||||||
|
|
||||||
|
**After close** (~16:45 ET Mon–Fri):
|
||||||
|
|
||||||
|
1. **OHLCV fetch** — final bar (not the partial near-close bar).
|
||||||
|
2. **Outcome Eval** — resolve setups and auto-close paper trades (default 3× ATR trail, 30-day max hold).
|
||||||
|
|
||||||
|
A failing step is logged; the pipeline continues with the next. Near-close duration is logged; warn if > 10 minutes.
|
||||||
|
|
||||||
|
### Intraday — light refresh
|
||||||
|
|
||||||
|
Hourly mid-session (Mon–Fri ~10:00–15:00 ET): only **OHLCV → Outcome Eval**, to keep prices current and close paper trades intraday. No scan/sentiment — the dashboard recomputes live R:R from the latest price.
|
||||||
|
|
||||||
|
### Other jobs
|
||||||
|
|
||||||
|
Fundamentals (weekly, early Monday ET) · Backtest (weekly) · Ticker-universe sync (daily). Alerts auto-fire only via the near-close pipeline (still manually triggerable). Deep history backfill and event study are manual-only (Admin → Jobs).
|
||||||
|
|
||||||
|
### From score to "top pick"
|
||||||
|
|
||||||
|
1. **Composite score** — technical, S/R-quality, sentiment, fundamental and momentum sub-scores (0–100) combine into a weighted composite (weights configurable; missing dimensions re-normalize). **Display and ranking only — it does not select trades.**
|
||||||
|
2. **Setups** — the scanner builds long/short setups with a 1.5× ATR stop, generates up to five candidates from the transient Gate Target Ladder, and makes the most likely worthwhile candidate the headline target. It then adds confidence and conflict context plus a per-target reach-probability.
|
||||||
|
3. **Activation gate** — a setup *qualifies* only if it ranks in the top residual-momentum percentile of the universe (**the actual selection**, long-only), its headline target clears the live R:R floor, **and** that target carries at least a 20% reach-probability. The confidence floor was ablated to zero effect and defaults off.
|
||||||
|
4. **Top pick** — qualified setups are ordered by the production rank: 80% residual momentum percentile + 20% 6-month realized-volatility percentile. The #1 is highlighted on the Dashboard and labelled on the ticker page.
|
||||||
|
|
||||||
|
**What the R:R and reach-probability in step 3 actually are.** They are *gate inputs*, computed from a Gate Target Ladder proposal the trade will never exit at — they exist to filter setups, not to forecast the trade you're about to take. A setup with "R:R 2.4:1, 34% reach probability" is not a claim that you'll make 2.4R with 34% probability; it's a claim that this setup cleared the screen. What actually happens to a trade is in the exit box of the diagram above, and on the "what usually happens" panel in the UI. Conflating the two is the single easiest way to misread this app.
|
||||||
|
|
||||||
|
## Strategy Status — What's Validated and What Isn't
|
||||||
|
|
||||||
|
**Read this before touching scoring, gating, or setup logic.** The platform measures itself — a weekly-replay backtest plus a factor rank-IC harness (`app/services/backtest_service.py`) — and the verdicts below come from those reports (latest run July 2026, ~5 years of OHLCV), not from opinion.
|
||||||
|
|
||||||
|
> **The full experiment log lives in [docs/research/](docs/research/README.md)** — every strategy we've tested, the result, and the decision. Check it before proposing an idea; most of the obvious ones have already been run and rejected.
|
||||||
|
|
||||||
|
| Component | Verdict | Evidence |
|
||||||
|
|---|---|---|
|
||||||
|
| **Residual 12-1 cross-sectional momentum** (the activation gate, long-only) | **Production gate — in-sample edge** | Promoted July 2026 after the portfolio variant beat raw 80 on CAGR, Sharpe and drawdown. Raw 12-1 remains a fallback only when benchmark data is unavailable |
|
||||||
|
| **3× ATR trailing exit** (+ 1.5× ATR initial stop, 30-day max hold) | **Production exit — best Sharpe of every exit tested** | Beat hold / SMA50 / 20-day-low / technical-40 and both take-profit variants (July 2026) |
|
||||||
|
| **Post-stop gate reset** | **Production re-entry policy** | The initial stop always closes; the ticker must later fail the daily gate and subsequently qualify again. At the production capacity of 10: Sharpe 1.67 → 1.77, CAGR 45.2% → 48.3%, DD 24.3% → 21.6% versus immediate re-entry. [Full study](docs/research/post-stop-reentry.md) |
|
||||||
|
| **Structural S/R** | **Human-facing context only — not a gate and not an exit** | Clean, capped zones are persisted for charts and alerts. The scanner deliberately does not read them. |
|
||||||
|
| **Gate Target Ladder** | **Gate input only — not market structure and not an exit** | Volume-free range grid + pivots preserves the useful legacy screening behavior exactly: 1,086/1,086 qualified setups retained and identical Sharpe 2.03 / CAGR 50.0% / DD 21.4% / 321 trades. The exit never reads its target. [Full write-up](docs/research/sr-levels-and-exits.md#explicit-gate-target-ladder) |
|
||||||
|
| Composite score + 5 dimensions | **Display/ranking only** | Sub-scores are hand-built heuristics; none has a measured IC. Note: the "momentum" *dimension* is 5/20-day ROC — NOT the validated 12-1 factor (that lives in `momentum_service`) |
|
||||||
|
| LLM sentiment | Display + a bounded composite adjustment (± weight × 100 pts around neutral 50) | Deliberately kept out of the setup engine; no point-in-time history to validate against yet |
|
||||||
|
| Fundamentals | Feeds composite + confidence only | Latest values only, no history — same limitation |
|
||||||
|
| Short setups | **Excluded while the momentum gate is active** | Backtest showed shorts fight the trend and drag expectancy |
|
||||||
|
| Expected-value gate (removed June 2026) | Degenerate — do not resurrect | Structurally favored distant lottery targets; selected *worse*-than-random setups. Orphaned settings dropped in migration 020 |
|
||||||
|
| Gate target as a take-profit (tested July 2026) | **Rejected** | Sharpe 2.04 → 1.47, CAGR halved. Win rate *rose* — it truncates the right tail where the edge lives |
|
||||||
|
| "Clear-air" gate relaxation (tested July 2026) | **Rejected — failed out-of-sample** | Strictly better in-sample (Sharpe 2.07 / CAGR 62.3% / DD 20.1%), then lost on a real train/test split (Sharpe 2.78 → 2.45). A cautionary tale: nested lookbacks are not OOS |
|
||||||
|
|
||||||
|
Caveats on the momentum result: in-sample, roughly one market regime, costs/slippage approximated at 0.1% per side, and residual momentum still needs SPY benchmark history to compute. The **out-of-sample proof is the forward paper-trade record**: Signals → Track Record compares live qualified expectancy against the backtest.
|
||||||
|
|
||||||
|
### Daily post-stop re-entry decision (2026-07-17)
|
||||||
|
|
||||||
|
The production policy is **normal gate reset**, evaluated with daily setup opportunities and live-like full-universe ranking. An initial stop always closes. Re-entry unlocks only after a later successful daily scan observes the ticker failing the gate and a subsequent scan observes it qualifying again. The study replayed 1,011,248 point-in-time candidate observations across 505 tickers from 2022-06-24 through 2026-07-02, with the production GTL gate, 80/20 rank, exit, fees, sizing, and 10-position capacity.
|
||||||
|
|
||||||
|
| Re-entry policy | Total return | CAGR | Max DD | Sharpe | Trades |
|
||||||
|
|---|---:|---:|---:|---:|---:|
|
||||||
|
| Immediate | 348.4% | 45.2% | -24.3% | 1.67 | 489 |
|
||||||
|
| **Gate reset (selected study arm)** | **388.1%** | **48.3%** | **-21.6%** | **1.77** | **472** |
|
||||||
|
| Strict gate reset (live timing analogue) | 342.7% | 44.8% | -23.4% | 1.68 | 471 |
|
||||||
|
| Fixed five-session cooldown | 250.8% | 36.6% | -22.2% | 1.47 | 473 |
|
||||||
|
|
||||||
|
In the disjoint 2025+ book, gate reset also beat immediate re-entry (Sharpe 1.66 vs 1.55; CAGR 41.8% vs 39.3%) and the fixed five-session rule (Sharpe 1.43; CAGR 32.7%). Its lead over both survived costs of 0.2% and 0.3% per side. The result is capacity-specific: cooldown 5 won at capacity 5, while immediate had slightly higher return and Sharpe at capacity 15. Production uses capacity 10, so that is the portfolio for which this decision is valid.
|
||||||
|
|
||||||
|
Those promotion numbers belong to the selected normal-reset study arm. Under the **pre-cutover** morning-scan scheduler (scan always before any outcome eval), live first-observation timing matched the stricter `strict_gate_reset` analogue (full-period Sharpe 1.68 / CAGR 44.8% / DD 23.4%). After the **near-close cutover** (2026-07), stops closed by earlier same-day intraday evals can receive a same-day fail observation at ~15:30 ET — moving live behavior **toward** the promoted `gate_reset` arm. Requalification still requires a later America/New_York trading date than the failure (`trade_policy` distinct-day guard). Full definitions and all nine policy arms: [docs/research/post-stop-reentry.md](docs/research/post-stop-reentry.md); execution evidence: [docs/research/execution-recovery.md](docs/research/execution-recovery.md).
|
||||||
|
|
||||||
|
`gate_reset` and a simple `next_session` block happened to produce the same executed live-universe portfolio in this sample. Their rules are still different: this establishes that same-day re-entry was harmful here, but does not isolate a separate historical return premium from the reset condition. Gate reset was promoted because it represents a genuinely new signal episode and did not sacrifice results in the production book. Full definitions, all nine policy arms, cost/capacity sensitivity, and legacy-rank results are in [docs/research/post-stop-reentry.md](docs/research/post-stop-reentry.md); source report: [`reports/daily_reentry_matrix.json`](reports/daily_reentry_matrix.json).
|
||||||
|
|
||||||
|
### Historical weekly production baseline (pre gate-reset)
|
||||||
|
|
||||||
|
Use this as the historical ranking/exit regression guardrail, not as a return promise or the current re-entry-policy result. This run predates the post-stop gate reset and uses weekly entry replay, so its portfolio headline is not directly comparable with the daily matrix above. Backtest run: local production SQLite snapshot, 506 tickers, weekly cadence, 30-trading-day horizon, 2022-06-28 → 2026-07-02, 0.1% per-side costs, price-only SPY benchmark. Numbers below are the 2026-07-11 run (`reports/backtest-20260711-prod-baseline.json`) — measured *after* the primary-target probability floor shipped, which pruned lottery-target setups (1,428 → 1,089 qualified) and lifted Sharpe on all three promotion contenders.
|
||||||
|
|
||||||
|
| Item | Historical weekly baseline |
|
||||||
|
|---|---|
|
||||||
|
| Strategy version | `residual_highvol_80_20_atr_trail3_v1` |
|
||||||
|
| Production gate | Long-only, residual 12-1 momentum percentile >= 80, headline gate-target R:R >= 2.0 (live `activation_min_rr`; code default 2.0), primary-target reach-probability >= 20%, NEUTRAL excluded, confidence floor off (0) |
|
||||||
|
| Production rank | 80% residual momentum percentile + 20% 6-month realized-volatility percentile |
|
||||||
|
| Exit | Initial ATR stop plus 3x ATR trailing stop, max 30 trading days |
|
||||||
|
| Portfolio CAGR | +50.4% |
|
||||||
|
| Portfolio total return | +413.8% vs SPY +95.7% |
|
||||||
|
| Max drawdown | -21.4% |
|
||||||
|
| Sharpe | 2.04 daily, annualized |
|
||||||
|
| Trades | 320 |
|
||||||
|
| Win rate | 37.5% |
|
||||||
|
| Average hold | 15.3 trading days |
|
||||||
|
| Best / worst trade | +12.9R / -3.3R |
|
||||||
|
| **How trades actually ended** | **initial stop 144 (45%) · trailing stop 98 (31%) · max hold 78 (24%) · target 0 (0%)** |
|
||||||
|
|
||||||
|
That last row is the strategy in one line: a 37.5% win rate is *fine* because the +12.9R tail pays for every −1R stop. It is also why no take-profit exists — and why the S/R "target" shown in the UI is a screening artifact, not a plan.
|
||||||
|
|
||||||
|
Promotion evidence from the same snapshot:
|
||||||
|
|
||||||
|
| Candidate | CAGR | Max DD | Sharpe | Trades | Read |
|
||||||
|
|---|---:|---:|---:|---:|---|
|
||||||
|
| Legacy residual 80 + 30d hold | +49.6% | -15.8% | 2.02 | 300 | Previous production baseline. Still the shallowest drawdown of the three |
|
||||||
|
| Residual/high-vol 80/20 + 30d hold | +51.9% | -22.2% | 2.00 | 303 | The vol tilt buys CAGR and pays for it in drawdown |
|
||||||
|
| Residual/high-vol 80/20 + 3x ATR trail | +50.4% | -21.4% | 2.04 | 320 | Promoted: best Sharpe. The ATR trail recovers part of the drawdown the vol tilt costs |
|
||||||
|
| Pure high-vol 80 + 30d hold | +31.6% | -34.8% | 1.12 | 476 | Rejected: standalone volatility was too volatile |
|
||||||
|
| Low-vol 80 + 30d hold | +2.7% | -19.5% | 0.29 | 240 | Rejected: no useful edge |
|
||||||
|
|
||||||
|
Read the top three honestly: the production book wins on Sharpe, not on every axis. The 80/20 vol tilt buys ~2pp of CAGR over the legacy residual-only book but costs ~6pp of drawdown, and the ATR trail hands part of that drawdown back. If drawdown ever matters more than risk-adjusted return here, legacy residual 80 + hold is the row to revisit.
|
||||||
|
|
||||||
|
The conclusion is not "trade high volatility alone." Keep residual momentum as the entry gate, use realized volatility only as a small ranking tilt, and add the ATR trail as defensive exit discipline.
|
||||||
|
|
||||||
|
Live-ranking note: the backtest ranks residual momentum and volatility inside each weekly setup-candidate cross-section. The live scanner computes the same 80/20 formula across the current ticker universe before scanning so every generated setup carries a stable ticker-level rank. That is the production approximation; reconcile it later only if candidate-only post-scan ranking proves materially different.
|
||||||
|
|
||||||
|
Parity guard (July 2026): the portfolio monitor's **Production** row replays the *runtime* configuration — the live activation gate (`qualified` flag) and the Admin exit policy (mode / ATR multiplier / hold days) — so tuning the strategy in Admin is reflected in the next backtest run instead of silently diverging. Constants defined on both sides (exit defaults, trail width, the 80/20 ordering weights, the promoted cutoff) are pinned by `tests/unit/test_prod_strategy_parity.py`, and the ordering weights are single-sourced from `momentum_service`.
|
||||||
|
|
||||||
|
### Tuned and confirmed — do not retest without new data (July 2026)
|
||||||
|
|
||||||
|
A systematic single-variable sweep (offline prod snapshot, production gate/rank/exit, 2022-06 → 2026-07 plus disjoint 2022–23 / 2024–26 folds) confirmed **every** production setting. Retesting these against the same ~4-year snapshot is wasted compute and invites overfitting; revisit only with meaningfully new data (longer history or broader universe).
|
||||||
|
|
||||||
|
| Knob tested | Verdict | Evidence |
|
||||||
|
|---|---|---|
|
||||||
|
| ATR trail multiple {1.5–4.0} | **Keep 3.0** | Return+Sharpe peak; ≤2.0 whipsaws out the momentum right tail; ≥2.5 is a plateau |
|
||||||
|
| SPY 200d-MA regime overlay (block entries / go flat) | **Reject** | Halves return (315%→138%) with zero drawdown benefit — the ATR trail already manages downside, and the filter blocks the recovery-phase entries that make the money |
|
||||||
|
| Momentum lookback: 6-1, 3-1, 12-7 (Novy-Marx), composites | **Keep residual 12-1** | 6-1/3-1 rank-IC ≈ 0; 12-7 IC 0.045 / t 1.58 — weaker than residual 12-1 (0.055 / t 1.98) |
|
||||||
|
| Selection cutoff {70, 75, 85, 90} × book size {10, 15, 20} | **Keep cutoff 80; capacity reopened** | The older weekly replay favored 80 × 10, but its no-cap-pressure conclusion is superseded by 519 book-full rejections versus 472 trades under the current daily gate-reset control |
|
||||||
|
| Position sizing: equal-weight, inverse-vol, risk-% sweep | **Keep 1% fixed-fractional** | See the inverse-vol warning below |
|
||||||
|
| Post-stop re-entry: immediate, fixed 2–5 sessions, gate resets, confirmation filters | **Keep normal gate reset for the 10-position production book** | Sharpe 1.77 vs 1.67 immediate and 1.47 cooldown 5; rerun before changing portfolio capacity |
|
||||||
|
| FIP path-smoothness as an in-book tie-breaker/filter | **Reject** (but see the lead below) | Non-monotonic across FIP quintiles within the qualified set; either half of a median split underperforms the full book — thinning the entry stream costs more compounding than the tilt returns |
|
||||||
|
|
||||||
|
> **Capacity correction (2026-08-05):** the table's older weekly conclusion
|
||||||
|
> that the ten-slot cap never binds is superseded. Under the current daily
|
||||||
|
> gate-reset Phase A control, 472 trades were admitted and 519 qualified entries
|
||||||
|
> were rejected because the book was full (52.4% of admitted+blocked
|
||||||
|
> opportunities). Cutoff 80 remains the signal setting; portfolio capacity is
|
||||||
|
> reopened in the focused capacity-bracket study.
|
||||||
|
|
||||||
|
Two findings future sessions must not re-litigate:
|
||||||
|
|
||||||
|
- **The "inverse-vol sizing win" (July 2026) was mis-attributed — do not resurrect.** The diagnostic sized `notional = equity × 1% / vol_6m`, and the 20% notional cap bound on 95% of entries, so it actually measured "~5 positions × 20% notional each" — a concentration/risk-appetite bump economically equivalent to raising risk to 1.5%, not vol-managed sizing. Genuine inverse-vol sizing (risk budget × median-vol/vol) cuts max drawdown to −18.2% but costs ~58pp total return at flat Sharpe: a risk-preference trade, not edge.
|
||||||
|
- **`fip_id` — Da/Gurun/Warachka information discreteness over the 12-1 formation window — is the strongest cross-sectional signal on the *production* universe: IC −0.045, t = −2.91, correct sign (continuous-information winners outperform).** It clears the iron-rule bar in isolation but does not improve this book (the momentum gate already captures the effect in-sample). **Phase B (liquid-1500, research branch only):** unconditional fip fails iron rule (−0.017 / t −1.85); mom-conditional fip (−0.088 / t −4.58) is a *book-tilt candidate only* after a baseline breadth mom book is proven. Do **not** cite the orphaned 21:14 row (+0.0575) — it raced a partial `research.sqlite`. See `docs/research/fip-breadth-ic.md`.
|
||||||
|
|
||||||
|
### The iron rule for strategy changes
|
||||||
|
|
||||||
|
A signal earns its way into selection **only** through the factor harness:
|
||||||
|
|
||||||
|
1. Add it as a point-in-time function of past bars in `_signal_values()` (`backtest_service.py`).
|
||||||
|
2. Run the backtest (Admin → Jobs, or the weekly run) and read the **Signal edge** table (Signals → Track Record).
|
||||||
|
3. Wire it into the gate or ranking **only if** |mean IC| ≳ 0.03 with a consistent sign and `reliable: true` (≥ 12 non-overlapping windows).
|
||||||
|
|
||||||
|
Corollaries: never let an unvalidated score gate setups; the outcome evaluator must keep scoring **all** setups (unqualified ones are the control group); LLM output stays display-only in the quant path.
|
||||||
|
|
||||||
|
### Highest-value next experiments (in order)
|
||||||
|
|
||||||
|
> Check **[docs/research/](docs/research/README.md)** first — 12 strategy ideas have already been tested and rejected, including the obvious ones (take-profit exits, regime overlays, inverse-vol sizing, shorts).
|
||||||
|
|
||||||
|
1. **Forward monitor the promoted strategy** — the production UI now behaves like a portfolio monitor for the current strategy, with selectable lookbacks and SPY comparison. Forward paper-trade months are the only evidence the snapshot cannot provide; the July 2026 tuning pass closed every in-sample lead. (Trailing-stop sensitivity and the max-15 capacity check are done — see the tuning table above.)
|
||||||
|
2. **Signal context snapshots** — accumulate point-in-time composite/sentiment/fundamental context for every new setup so the discretionary overlay can be tested forward-only.
|
||||||
|
3. **Breadth is no longer free leverage** — Phase B found residual-mom t-stat *fell* on liquid-1500 vs the 505-name fingerprint (0.055/1.98 → 0.029/1.33). Any breadth book must clear a pre-registered baseline arm before fip tilts mean anything. (Deeper history was considered and declined.)
|
||||||
|
|
||||||
|
## Key Use Cases
|
||||||
|
|
||||||
|
- **Find today's best long setup.** On the **Dashboard**, the *Top Setups* table lists residual-gated qualified setups ranked by the production 80/20 residual/high-vol score, with the #1 flagged "Top pick". Each row opens the ticker page for its chart, Structural S/R, Gate Target Ladder targets and entry/stop.
|
||||||
|
- **Track a trade you took.** Mark a setup as a **paper trade**: it's marked-to-market against the latest close, auto-closed by the active exit policy (default: 3x ATR trail with a 30-trading-day max hold), and its sentiment stays fresh while open. *Signals → Track Record* shows the realized edge.
|
||||||
|
|
||||||
## Stack
|
## Stack
|
||||||
|
|
||||||
@@ -10,14 +301,14 @@ Investing-signal platform for NASDAQ stocks. Surfaces the best trading opportuni
|
|||||||
|---|---|
|
|---|---|
|
||||||
| Backend | Python 3.12+, FastAPI, Uvicorn, async SQLAlchemy, Alembic |
|
| Backend | Python 3.12+, FastAPI, Uvicorn, async SQLAlchemy, Alembic |
|
||||||
| Database | PostgreSQL (asyncpg) |
|
| Database | PostgreSQL (asyncpg) |
|
||||||
| Scheduler | APScheduler — OHLCV, sentiment, fundamentals, R:R scan |
|
| Scheduler | APScheduler — daily & intraday pipelines, fundamentals, alerts, regime, backtest |
|
||||||
| Frontend | React 18, TypeScript, Vite 5 |
|
| Frontend | React 18, TypeScript, Vite 5 |
|
||||||
| Styling | Tailwind CSS 3 with custom glassmorphism design system |
|
| Styling | Tailwind CSS 3 with custom glassmorphism design system |
|
||||||
| State | TanStack React Query v5 (server), Zustand (client/auth) |
|
| State | TanStack React Query v5 (server), Zustand (client/auth) |
|
||||||
| Charts | Canvas 2D candlestick chart with S/R overlays |
|
| Charts | Canvas 2D candlestick chart with S/R overlays |
|
||||||
| Routing | React Router v6 (SPA) |
|
| Routing | React Router v6 (SPA) |
|
||||||
| HTTP | Axios with JWT interceptor |
|
| HTTP | Axios with JWT interceptor |
|
||||||
| Data providers | Alpaca (OHLCV), OpenAI (sentiment, optional micro-batch), Fundamentals chain: FMP → Finnhub → Alpha Vantage |
|
| Data providers | Alpaca (OHLCV); OpenAI / Gemini / DeepSeek / xAI (sentiment, pluggable); Fundamentals chain: FMP → Finnhub → Alpha Vantage; FRED (regime); Telegram (alerts) |
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
@@ -26,20 +317,28 @@ Investing-signal platform for NASDAQ stocks. Surfaces the best trading opportuni
|
|||||||
- Universe bootstrap for `sp500`, `nasdaq100`, `nasdaq_all` via admin endpoint
|
- Universe bootstrap for `sp500`, `nasdaq100`, `nasdaq_all` via admin endpoint
|
||||||
- OHLCV price storage with upsert and validation
|
- OHLCV price storage with upsert and validation
|
||||||
- Technical indicators: ADX, EMA, RSI, ATR, Volume Profile, Pivot Points, EMA Cross
|
- Technical indicators: ADX, EMA, RSI, ATR, Volume Profile, Pivot Points, EMA Cross
|
||||||
- Support/Resistance detection with strength scoring and merge-within-tolerance
|
- Structural Support/Resistance detection with rejection/recency strength, ATR-adaptive merging and a hard cap; persisted for charts and alerts
|
||||||
|
- Transient Gate Target Ladder — volume-free range grid plus pivots, used only for nominal targets, reach-probability and gate R:R
|
||||||
- Sentiment analysis with time-decay weighted scoring
|
- Sentiment analysis with time-decay weighted scoring
|
||||||
- Fundamental data tracking (P/E, revenue growth, earnings surprise, market cap)
|
- Fundamental data tracking (P/E, revenue growth, earnings surprise, market cap)
|
||||||
- 5-dimension scoring engine (technical, S/R quality, sentiment, fundamental, momentum) with configurable weights
|
- 5-dimension scoring engine (technical, S/R quality, sentiment, fundamental, momentum) with configurable weights
|
||||||
- Risk:Reward scanner — long and short setups, ATR-based stops, configurable R:R threshold (default 1.5:1)
|
- Risk:Reward scanner — long and short setups, 1.5x ATR stops, Gate Target Ladder nominal targets, configurable scan R:R threshold (default 1.5:1 — distinct from the activation floor below)
|
||||||
- Auto-populated watchlist (top-10 by composite score) + manual entries (cap: 20)
|
- Activation gate — qualifies setups on a residual-momentum percentile floor (the actual selection), a headline gate-target R:R floor (prod: 2.0) and a 20% primary-target reach-probability floor (validated long-only edge)
|
||||||
|
- Recommendation layer — directional confidence, conflict detection, per-target reach-probability
|
||||||
|
- Paper trading — take a setup, mark-to-market vs. latest close, auto-close per the exit policy (default: 3x ATR trail with a 30-trading-day max hold; time / percent-trailing / target-stop selectable), realized track record + outcome evaluation
|
||||||
|
- Market-regime guard + observational State/Warning monitor (fixed-basket breadth, VIX, credit level + impulse) with a manual chronological correction study
|
||||||
|
- Telegram alerts (e.g. regime-quadrant changes)
|
||||||
|
- User-curated watchlist (cap: 20), enriched with composite score, R:R and S/R summary
|
||||||
- JWT auth with admin role, configurable registration, user access control
|
- JWT auth with admin role, configurable registration, user access control
|
||||||
- Scheduled jobs with enable/disable control and status monitoring
|
- Cron-scheduled pipelines (admin-configurable) with per-job enable/disable and live status monitoring
|
||||||
- Admin panel: user management, data cleanup, job control, system settings
|
- Admin panel: user management, data cleanup, job control, system settings
|
||||||
|
|
||||||
### Frontend
|
### Frontend
|
||||||
- Glassmorphism UI with frosted glass panels, gradient text, ambient glow effects, mesh gradient background
|
- Glassmorphism UI with frosted glass panels, gradient text, ambient glow effects, mesh gradient background
|
||||||
- Interactive candlestick chart (Canvas 2D) with hover tooltips showing OHLCV values
|
- Interactive candlestick chart (Canvas 2D) with hover tooltips showing OHLCV values
|
||||||
- Support/Resistance level overlays on chart (top 6 by strength, dashed lines with labels)
|
- Support/Resistance level overlays on chart (top 6 by strength, dashed lines with labels)
|
||||||
|
- Optional GTL price-traffic profile on the ticker chart (right-edge diagnostic; explicitly not volume)
|
||||||
|
- Production-rank strip below the ticker chart (80/20 contribution ledger plus separate momentum and volatility percentiles)
|
||||||
- Data freshness bar showing availability and recency of each data source
|
- Data freshness bar showing availability and recency of each data source
|
||||||
- Watchlist with composite scores, R:R ratios, and S/R summaries
|
- Watchlist with composite scores, R:R ratios, and S/R summaries
|
||||||
- Ticker detail page: chart, scores, sentiment breakdown, fundamentals, technical indicators, S/R table
|
- Ticker detail page: chart, scores, sentiment breakdown, fundamentals, technical indicators, S/R table
|
||||||
@@ -56,12 +355,15 @@ Investing-signal platform for NASDAQ stocks. Surfaces the best trading opportuni
|
|||||||
|---|---|---|
|
|---|---|---|
|
||||||
| `/login` | Login | Public |
|
| `/login` | Login | Public |
|
||||||
| `/register` | Register | Public (when enabled) |
|
| `/register` | Register | Public (when enabled) |
|
||||||
| `/watchlist` | Watchlist (default) | Authenticated |
|
| `/` | Dashboard — top setups, open trades, regime (default) | Authenticated |
|
||||||
|
| `/market` | Market — watchlist + rankings tabs | Authenticated |
|
||||||
|
| `/signals` | Signals — scanner + track record tabs | Authenticated |
|
||||||
|
| `/regime` | Market Regime | Authenticated |
|
||||||
| `/ticker/:symbol` | Ticker Detail | Authenticated |
|
| `/ticker/:symbol` | Ticker Detail | Authenticated |
|
||||||
| `/scanner` | Trade Scanner | Authenticated |
|
|
||||||
| `/rankings` | Rankings | Authenticated |
|
|
||||||
| `/admin` | Admin Panel | Admin only |
|
| `/admin` | Admin Panel | Admin only |
|
||||||
|
|
||||||
|
Legacy routes redirect: `/watchlist` → `/market`, `/rankings` → `/market?tab=rankings`, `/scanner` → `/signals`, `/performance` → `/signals?tab=track`.
|
||||||
|
|
||||||
## API Endpoints
|
## API Endpoints
|
||||||
|
|
||||||
All under `/api/v1/`. Interactive docs at `/docs` (Swagger) and `/redoc`.
|
All under `/api/v1/`. Interactive docs at `/docs` (Swagger) and `/redoc`.
|
||||||
@@ -75,10 +377,14 @@ All under `/api/v1/`. Interactive docs at `/docs` (Swagger) and `/redoc`.
|
|||||||
| Ingestion | `POST /ingestion/fetch/{symbol}` |
|
| Ingestion | `POST /ingestion/fetch/{symbol}` |
|
||||||
| Indicators | `GET /indicators/{symbol}/{type}`, `GET /indicators/{symbol}/ema-cross` |
|
| Indicators | `GET /indicators/{symbol}/{type}`, `GET /indicators/{symbol}/ema-cross` |
|
||||||
| S/R Levels | `GET /sr-levels/{symbol}` |
|
| S/R Levels | `GET /sr-levels/{symbol}` |
|
||||||
|
| Gate Target Ladder | `GET /gate-target-ladder/{symbol}` |
|
||||||
| Sentiment | `GET /sentiment/{symbol}` |
|
| Sentiment | `GET /sentiment/{symbol}` |
|
||||||
| Fundamentals | `GET /fundamentals/{symbol}` |
|
| Fundamentals | `GET /fundamentals/{symbol}` |
|
||||||
| Scores | `GET /scores/{symbol}`, `GET /rankings`, `PUT /scores/weights` |
|
| Scores | `GET /scores/{symbol}`, `GET /rankings`, `PUT /scores/weights` |
|
||||||
| Trades | `GET /trades` |
|
| Trades | `GET /trades`, `GET /trades/{symbol}`, `GET /trades/{symbol}/history`, `GET /trades/activation`, `GET /trades/performance` |
|
||||||
|
| Paper Trades | `GET /paper-trades`, `POST /paper-trades`, `POST /paper-trades/{id}/close` |
|
||||||
|
| Market / Regime | `GET /market/regime`, `GET /regime/monitor`, `GET/PUT /regime/config`, `GET /regime/history`, `GET /regime/event-study`, `GET/PUT /regime/fundamentals`, `GET /backtest/report` |
|
||||||
|
| Jobs | `GET /jobs/running` |
|
||||||
| Watchlist | `GET /watchlist`, `POST /watchlist/{symbol}`, `DELETE /watchlist/{symbol}` |
|
| Watchlist | `GET /watchlist`, `POST /watchlist/{symbol}`, `DELETE /watchlist/{symbol}` |
|
||||||
| Admin | `GET /admin/users`, `POST /admin/users`, `PUT /admin/users/{id}/access`, `PUT /admin/users/{id}/password`, `PUT /admin/settings/registration`, `GET /admin/settings`, `PUT /admin/settings/{key}`, `GET/PUT /admin/settings/recommendations`, `GET/PUT /admin/settings/ticker-universe`, `POST /admin/tickers/bootstrap`, `POST /admin/data/cleanup`, `GET /admin/jobs`, `POST /admin/jobs/{name}/trigger`, `PUT /admin/jobs/{name}/toggle`, `GET /admin/pipeline/readiness` |
|
| Admin | `GET /admin/users`, `POST /admin/users`, `PUT /admin/users/{id}/access`, `PUT /admin/users/{id}/password`, `PUT /admin/settings/registration`, `GET /admin/settings`, `PUT /admin/settings/{key}`, `GET/PUT /admin/settings/recommendations`, `GET/PUT /admin/settings/ticker-universe`, `POST /admin/tickers/bootstrap`, `POST /admin/data/cleanup`, `GET /admin/jobs`, `POST /admin/jobs/{name}/trigger`, `PUT /admin/jobs/{name}/toggle`, `GET /admin/pipeline/readiness` |
|
||||||
|
|
||||||
@@ -140,11 +446,134 @@ npm run preview # Preview the production build locally
|
|||||||
# Backend tests (in-memory SQLite — no PostgreSQL needed)
|
# Backend tests (in-memory SQLite — no PostgreSQL needed)
|
||||||
pytest tests/ -v
|
pytest tests/ -v
|
||||||
|
|
||||||
# Frontend tests
|
# Frontend: there is no test suite — `npm test` calls vitest, which is not
|
||||||
|
# installed. The frontend check is the full TypeScript build:
|
||||||
cd frontend
|
cd frontend
|
||||||
npm test
|
npm run build
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Local Backtest Snapshots
|
||||||
|
|
||||||
|
For research loops, run the production backtest locally from a SQLite snapshot
|
||||||
|
instead of deploying and clicking the Admin job. The snapshot contains only the
|
||||||
|
tables needed by `run_backtest`: tickers, OHLCV bars, SPY benchmark closes, and
|
||||||
|
the activation / recommendation / paper-exit settings. Secrets and cached reports
|
||||||
|
are not copied.
|
||||||
|
|
||||||
|
> The `paper_%` settings **must** be copied: the portfolio monitor's Production row
|
||||||
|
> replays the *runtime* exit policy via `get_exit_policy()`. Without them a snapshot
|
||||||
|
> silently falls back to the code defaults, so a live-tuned exit would not be
|
||||||
|
> reflected and the local run would disagree with prod for no visible reason.
|
||||||
|
|
||||||
|
1. Open an SSH tunnel to the production Postgres instance:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ssh -N -L 15432:127.0.0.1:5432 deploy@your-server
|
||||||
|
```
|
||||||
|
|
||||||
|
2. In another terminal, create or refresh the snapshot:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# macOS/Linux
|
||||||
|
python scripts/create_backtest_snapshot.py \
|
||||||
|
--database-url "postgresql+asyncpg://stock_backend:PASSWORD@127.0.0.1:15432/stock_data_backend" \
|
||||||
|
--output backtest_snapshots/prod.sqlite \
|
||||||
|
--force
|
||||||
|
|
||||||
|
# Windows PowerShell
|
||||||
|
.venv\Scripts\python.exe scripts\create_backtest_snapshot.py `
|
||||||
|
--database-url "postgresql+asyncpg://stock_backend:PASSWORD@127.0.0.1:15432/stock_data_backend" `
|
||||||
|
--output backtest_snapshots\prod.sqlite `
|
||||||
|
--force
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Run the backtest fully offline from the snapshot:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# macOS/Linux
|
||||||
|
python scripts/run_backtest_snapshot.py backtest_snapshots/prod.sqlite --workers 6
|
||||||
|
|
||||||
|
# Windows PowerShell
|
||||||
|
.venv\Scripts\python.exe scripts\run_backtest_snapshot.py backtest_snapshots\prod.sqlite --workers 6 --allow-spawn
|
||||||
|
```
|
||||||
|
|
||||||
|
Weekly remains the resource-safe default. Add `--cadence daily` for live-like daily entry opportunities; this performs roughly five times as many setup evaluations. To generate the complete weekly/daily × immediate/gate-reset comparison in one invocation, use:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/run_backtest_cadence_comparison.py backtest_snapshots/prod.sqlite --workers 7
|
||||||
|
```
|
||||||
|
|
||||||
|
On Windows, add `--allow-spawn`. The comparison runner writes the two full cadence reports plus one compact four-arm report. For the larger nine-policy daily research matrix used in the post-stop decision, see `scripts/run_daily_reentry_matrix.py` and the [research record](docs/research/post-stop-reentry.md).
|
||||||
|
|
||||||
|
On an 8-thread machine, `--workers 6` is a good starting point: it leaves a
|
||||||
|
couple of threads for Windows, the shell, and browser/UI work while still using
|
||||||
|
most of the CPU.
|
||||||
|
|
||||||
|
The runner writes `reports/backtest-<timestamp>.json` and prints the headline
|
||||||
|
metrics. Keep the SSH tunnel open only while creating the snapshot; the backtest
|
||||||
|
run itself is local/offline. `backtest_snapshots/` and generated backtest reports
|
||||||
|
are git-ignored.
|
||||||
|
|
||||||
|
The local runner, scheduled job, and Admin UI all default to
|
||||||
|
`production_gtl`, matching the live scanner's target path. For a deliberate
|
||||||
|
comparison, select **Structural S/R (comparison)** in the UI or pass
|
||||||
|
`--target-model structural_sr` locally. Every report records the selected model
|
||||||
|
and whether it is the production path.
|
||||||
|
|
||||||
|
### Archived GTL tuning decision
|
||||||
|
|
||||||
|
The completed replacement, cohort-composition, and strength-sensitivity
|
||||||
|
matrices found no stable improvement over the frozen Gate Target Ladder. The
|
||||||
|
temporary matrix runners and tuning hooks have been retired; their three compact
|
||||||
|
consolidated report pairs remain in `reports/` as the decision audit. Keep the
|
||||||
|
GTL unchanged and evaluate any future challenger only on new forward data. See
|
||||||
|
the [full research record](docs/research/sr-levels-and-exits.md#gtl-tuning-matrix).
|
||||||
|
|
||||||
|
### Reading a local backtest report
|
||||||
|
|
||||||
|
The deployed **Signals → Track Record** page is deliberately trimmed to validation
|
||||||
|
(portfolio monitor vs SPY, realized paper trades) and how-to-trade. The
|
||||||
|
strategy-tuning tables that used to live there now live **only** in the local
|
||||||
|
report — inspect these `reports/backtest-<timestamp>.json` sections and produce the
|
||||||
|
matching decision. Every change still goes through the factor harness first (see
|
||||||
|
**The iron rule for strategy changes** above).
|
||||||
|
|
||||||
|
| Report section | What to read | Decision it drives |
|
||||||
|
|---|---|---|
|
||||||
|
| `overall_qualified` vs `overall_all` | Is qualified net expectancy above the all-setups baseline? | Sanity — is the gate adding anything at all |
|
||||||
|
| `sweep` | Net avg R and trade count at each residual-momentum cutoff | Where to set the momentum percentile (Admin → Settings → Activation) |
|
||||||
|
| `gate_ablation` | Net expectancy with each floor removed | Drop a floor only if removing it doesn't hurt net expectancy |
|
||||||
|
| `time_exit_sweep` | Net avg R / net R-per-day by hold length | Whether a fixed time exit beats the promoted ATR trail |
|
||||||
|
| `portfolio_monitor`, `portfolio_sim`, `strategy_variants` | CAGR, Sharpe, max drawdown, per-year returns | Promote a strategy only if it beats the current baseline on CAGR/Sharpe/DD |
|
||||||
|
| `production_cadence_comparison` | Immediate vs production gate reset at the selected weekly or daily cadence | Isolates the re-entry rule while keeping gate, rank, exit, fees, sizing, and capacity fixed |
|
||||||
|
| `signal_eval` | Mean IC, t-stat, IC>0 %, `reliable` | Iron rule: wire a new factor in only if \|IC\| ≳ 0.03 with a consistent sign and `reliable: true` |
|
||||||
|
| `holdout` (opt-in) | Train vs test books, split by entry date | **The only honest OOS read.** Set `BACKTEST_HOLDOUT_SPLIT=YYYY-MM-DD` |
|
||||||
|
| `recommendation`, `research_recommendation` | The report's own headline read | A starting point, not a substitute for the sections above |
|
||||||
|
|
||||||
|
**Out-of-sample validation.** The `portfolio_monitor` lookbacks (6m / 1y / 3y / 5y / all) are
|
||||||
|
**nested windows that all end today** — every one of them overlaps the data an idea was found
|
||||||
|
on, so none of them is a holdout. A rule that looks good across all five can still be an
|
||||||
|
in-sample artifact (this exact trap ate the clear-air experiment; see the research log). For a
|
||||||
|
real train/test split by entry date:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
BACKTEST_HOLDOUT_SPLIT=2024-07-01 python scripts/run_backtest_snapshot.py \
|
||||||
|
backtest_snapshots/prod.sqlite --workers 7 --allow-spawn
|
||||||
|
```
|
||||||
|
|
||||||
|
Research-only flags, all off by default (the default report is byte-identical to the shipped baseline):
|
||||||
|
|
||||||
|
| Flag | What it does |
|
||||||
|
|---|---|
|
||||||
|
| `BACKTEST_HOLDOUT_SPLIT=YYYY-MM-DD` | Adds a `holdout` section: train (entries before) vs test (entries on/after), as disjoint books |
|
||||||
|
| `BACKTEST_MIN_RR_SWEEP=1` | Sweeps the activation R:R floor against portfolio Sharpe. Combine with `BACKTEST_HOLDOUT_SPLIT` to sweep out-of-sample |
|
||||||
|
| `BACKTEST_RESEARCH_EXITS=1` | Adds the rejected take-profit exit rows to the exit comparison |
|
||||||
|
| `BACKTEST_ATR_TARGET_FALLBACK=k` | Synthesizes a k×ATR target where S/R offers none |
|
||||||
|
| `BACKTEST_FALLBACK_CLEAR_AIR_ONLY=1` | Restricts that fallback to setups with genuinely no structure ahead |
|
||||||
|
|
||||||
|
`recommendation` is the one section surfaced on the deployed page ("What this
|
||||||
|
backtest recommends"); everything else in this table is intentionally local-only.
|
||||||
|
|
||||||
## Environment Variables
|
## Environment Variables
|
||||||
|
|
||||||
Configure in `.env` (copy from `.env.example`):
|
Configure in `.env` (copy from `.env.example`):
|
||||||
@@ -164,83 +593,86 @@ Configure in `.env` (copy from `.env.example`):
|
|||||||
| `FMP_API_KEY` | Optional (fundamentals) | — | Financial Modeling Prep API key (first provider in chain) |
|
| `FMP_API_KEY` | Optional (fundamentals) | — | Financial Modeling Prep API key (first provider in chain) |
|
||||||
| `FINNHUB_API_KEY` | Optional (fundamentals) | — | Finnhub API key (fallback provider) |
|
| `FINNHUB_API_KEY` | Optional (fundamentals) | — | Finnhub API key (fallback provider) |
|
||||||
| `ALPHA_VANTAGE_API_KEY` | Optional (fundamentals) | — | Alpha Vantage API key (fallback provider) |
|
| `ALPHA_VANTAGE_API_KEY` | Optional (fundamentals) | — | Alpha Vantage API key (fallback provider) |
|
||||||
| `DATA_COLLECTOR_FREQUENCY` | No | `daily` | OHLCV collection schedule |
|
| `FRED_API_KEY` | Optional (regime) | — | FRED key for the regime monitor (VIX, credit spreads) |
|
||||||
|
| `TELEGRAM_BOT_TOKEN` | Optional (alerts) | — | Telegram bot token for alerts (can also be set in Admin) |
|
||||||
|
| `TELEGRAM_CHAT_ID` | Optional (alerts) | — | Telegram chat id for alerts |
|
||||||
|
| `DATA_COLLECTOR_FREQUENCY` | No | `daily` | OHLCV collection schedule (legacy — see note below) |
|
||||||
| `SENTIMENT_POLL_INTERVAL_MINUTES` | No | `30` | Sentiment polling interval |
|
| `SENTIMENT_POLL_INTERVAL_MINUTES` | No | `30` | Sentiment polling interval |
|
||||||
| `FUNDAMENTAL_FETCH_FREQUENCY` | No | `daily` | Fundamentals fetch schedule |
|
| `FUNDAMENTAL_FETCH_FREQUENCY` | No | `weekly` | Fundamentals fetch cadence |
|
||||||
| `RR_SCAN_FREQUENCY` | No | `daily` | R:R scanner schedule |
|
| `RR_SCAN_FREQUENCY` | No | `daily` | R:R scanner schedule |
|
||||||
| `FUNDAMENTAL_RATE_LIMIT_RETRIES` | No | `3` | Retries per ticker on fundamentals rate-limit |
|
| `FUNDAMENTAL_RATE_LIMIT_RETRIES` | No | `3` | Retries per ticker on fundamentals rate-limit |
|
||||||
| `FUNDAMENTAL_RATE_LIMIT_BACKOFF_SECONDS` | No | `15` | Base backoff seconds for fundamentals retry (exponential) |
|
| `FUNDAMENTAL_RATE_LIMIT_BACKOFF_SECONDS` | No | `15` | Base backoff seconds for fundamentals retry (exponential) |
|
||||||
| `DEFAULT_WATCHLIST_AUTO_SIZE` | No | `10` | Auto-watchlist size |
|
| `DEFAULT_WATCHLIST_AUTO_SIZE` | No | `10` | Auto-watchlist size |
|
||||||
| `DEFAULT_RR_THRESHOLD` | No | `3.0` | Minimum R:R ratio for setups |
|
| `DEFAULT_RR_THRESHOLD` | No | `1.5` | Minimum R:R ratio for setups |
|
||||||
| `DB_POOL_SIZE` | No | `5` | Database connection pool size |
|
| `DB_POOL_SIZE` | No | `5` | Database connection pool size |
|
||||||
| `LOG_LEVEL` | No | `INFO` | Logging level |
|
| `LOG_LEVEL` | No | `INFO` | Logging level |
|
||||||
|
|
||||||
|
> **Note:** Pipeline timing (daily / intraday / fundamentals cron, timezone) is configured at runtime in **Admin → Jobs** and stored in the DB — the `*_FREQUENCY` env vars are legacy fallbacks for the few jobs still on interval triggers (alerts, universe sync).
|
||||||
|
|
||||||
## Production Deployment (Debian 12)
|
## Production Deployment (Debian 12)
|
||||||
|
|
||||||
|
**Ongoing deploys are automated.** Every push to `main` triggers the Gitea Actions pipeline (`.gitea/workflows/deploy.yml`): lint → test → rsync to the server → `pip install` → `alembic upgrade head` → restart `signalplatform.service` → health check. There is no manual deploy step; the steps below are only for provisioning a new server.
|
||||||
|
|
||||||
### 1. Install dependencies
|
### 1. Install dependencies
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
sudo apt update && sudo apt install -y python3.12 python3.12-venv postgresql nginx nodejs npm
|
sudo apt update && sudo apt install -y python3.12 python3.12-venv postgresql nginx rsync
|
||||||
```
|
```
|
||||||
|
|
||||||
### 2. Create service user
|
### 2. Create the deploy user
|
||||||
|
|
||||||
|
The pipeline connects over SSH as this user; it owns the app directory and needs passwordless permission to restart the service:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
sudo useradd -r -s /usr/sbin/nologin stockdata
|
sudo useradd -m deploy
|
||||||
|
sudo mkdir -p /opt/signalplatform
|
||||||
|
sudo chown deploy:deploy /opt/signalplatform
|
||||||
|
echo 'deploy ALL=(root) NOPASSWD: /usr/bin/systemctl restart signalplatform.service' | sudo tee /etc/sudoers.d/deploy-restart
|
||||||
```
|
```
|
||||||
|
|
||||||
### 3. Deploy application
|
### 3. Configure the pipeline (Gitea repo settings)
|
||||||
|
|
||||||
|
Variables: `DEPLOY_HOST`, `DEPLOY_USER` (`deploy`), `DEPLOY_PATH` (`/opt/signalplatform`), `SSH_KNOWN_HOSTS` (host fingerprint), `SSH_PORT`. Secret: `SSH_PRIVATE_KEY` (matching the deploy user's authorized key).
|
||||||
|
|
||||||
|
### 4. Configure the app
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
sudo mkdir -p /opt/stock-data-backend
|
# After the first pipeline run has synced the files:
|
||||||
# Copy project files to /opt/stock-data-backend
|
cp /opt/signalplatform/.env.example /opt/signalplatform/.env
|
||||||
cd /opt/stock-data-backend
|
|
||||||
python3.12 -m venv .venv
|
|
||||||
source .venv/bin/activate
|
|
||||||
pip install .
|
|
||||||
```
|
|
||||||
|
|
||||||
### 4. Configure
|
|
||||||
|
|
||||||
```bash
|
|
||||||
sudo cp .env.example /opt/stock-data-backend/.env
|
|
||||||
sudo chown stockdata:stockdata /opt/stock-data-backend/.env
|
|
||||||
# Edit .env with production values (strong JWT_SECRET, real API keys, etc.)
|
# Edit .env with production values (strong JWT_SECRET, real API keys, etc.)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
`.env` is excluded from the rsync, so it survives every deploy.
|
||||||
|
|
||||||
### 5. Database
|
### 5. Database
|
||||||
|
|
||||||
|
Either trigger the workflow manually (workflow_dispatch) with `run_setup_db: true` — the deploy then runs `deploy/setup_db.sh` instead of plain migrations — or run it once by hand:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
DB_NAME=stock_data_backend DB_USER=stock_backend DB_PASS=strong_password ./deploy/setup_db.sh
|
DB_NAME=stock_data_backend DB_USER=stock_backend DB_PASS=strong_password ./deploy/setup_db.sh
|
||||||
```
|
```
|
||||||
|
|
||||||
### 6. Build frontend
|
### 6. Systemd service
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cd frontend
|
sudo cp deploy/signalplatform.service /etc/systemd/system/
|
||||||
npm ci
|
|
||||||
npm run build
|
|
||||||
```
|
|
||||||
|
|
||||||
### 7. Systemd service
|
|
||||||
|
|
||||||
```bash
|
|
||||||
sudo cp deploy/stock-data-backend.service /etc/systemd/system/
|
|
||||||
sudo systemctl daemon-reload
|
sudo systemctl daemon-reload
|
||||||
sudo systemctl enable --now stock-data-backend
|
sudo systemctl enable --now signalplatform
|
||||||
```
|
```
|
||||||
|
|
||||||
### 8. Nginx reverse proxy
|
The unit runs uvicorn on `127.0.0.1:8998` as the `deploy` user, with `WorkingDirectory=/opt/signalplatform`.
|
||||||
|
|
||||||
|
### 7. Nginx reverse proxy
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
sudo cp deploy/nginx.conf /etc/nginx/sites-available/stock-data-backend
|
sudo cp deploy/nginx.conf /etc/nginx/sites-available/signalplatform
|
||||||
sudo ln -s /etc/nginx/sites-available/stock-data-backend /etc/nginx/sites-enabled/
|
sudo ln -s /etc/nginx/sites-available/signalplatform /etc/nginx/sites-enabled/
|
||||||
sudo nginx -t && sudo systemctl reload nginx
|
sudo nginx -t && sudo systemctl reload nginx
|
||||||
```
|
```
|
||||||
|
|
||||||
Nginx serves the frontend static files from `frontend/dist/` and proxies `/api/v1/` to the backend.
|
Nginx serves the frontend static files from `frontend/dist/` (built on the CI runner and rsynced) and proxies `/api/v1/` to the backend.
|
||||||
|
|
||||||
### 9. SSL (recommended)
|
### 8. SSL (recommended)
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
sudo apt install certbot python3-certbot-nginx
|
sudo apt install certbot python3-certbot-nginx
|
||||||
@@ -292,10 +724,17 @@ frontend/
|
|||||||
│ └── watchlist/ # Watchlist table, add ticker form
|
│ └── watchlist/ # Watchlist table, add ticker form
|
||||||
├── hooks/ # React Query hooks (one per resource)
|
├── hooks/ # React Query hooks (one per resource)
|
||||||
├── lib/ # Types, formatting utilities
|
├── lib/ # Types, formatting utilities
|
||||||
├── pages/ # Page components (7 pages)
|
├── pages/ # Page components (Login, Register, Dashboard, Market, Signals, Regime, Ticker, Admin)
|
||||||
├── stores/ # Zustand auth store
|
├── stores/ # Zustand auth store
|
||||||
└── styles/ # Global CSS with glassmorphism classes
|
└── styles/ # Global CSS with glassmorphism classes
|
||||||
|
|
||||||
|
docs/
|
||||||
|
└── research/ # Experiment log: what was tested, the result, the decision
|
||||||
|
├── README.md # Overview — start here before proposing a strategy change
|
||||||
|
└── sr-levels-and-exits.md
|
||||||
|
|
||||||
|
reports/ # Committed backtest reports (JSON) + compare_reports.py
|
||||||
|
|
||||||
deploy/
|
deploy/
|
||||||
├── nginx.conf # Reverse proxy + static file serving
|
├── nginx.conf # Reverse proxy + static file serving
|
||||||
├── setup_db.sh # Idempotent DB setup script
|
├── setup_db.sh # Idempotent DB setup script
|
||||||
@@ -306,3 +745,66 @@ tests/
|
|||||||
├── unit/ # Unit tests
|
├── unit/ # Unit tests
|
||||||
└── property/ # Property-based tests (Hypothesis)
|
└── property/ # Property-based tests (Hypothesis)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Maintainer Guide
|
||||||
|
|
||||||
|
Context for whoever — human or AI — continues this work. The owner pushes straight to `main` on a self-hosted Gitea remote (no PRs); deployment is automated by the Gitea Actions workflow at `.gitea/workflows/deploy.yml`.
|
||||||
|
|
||||||
|
### Invariants — do not break these
|
||||||
|
|
||||||
|
- **`app/services/qualification.py` is mirrored in `frontend/src/lib/qualification.ts`.** Any gate change must land in both, or the UI's "qualified" flags silently disagree with the server.
|
||||||
|
- **Live scan and backtest share the same pure functions.** The backtest replays production logic through DB-free functions (`compute_technical_from_arrays`, `compute_momentum_from_closes`, `detect_sr_levels`, `detect_gate_target_ladder`, the recommendation helpers). New strategy logic must stay in pure functions consumed by both paths, or the backtest stops measuring what production actually does.
|
||||||
|
- **Keep the two price-level models separate.** `detect_sr_levels` produces persisted Structural S/R for charts and alerts. `detect_gate_target_ladder` produces transient screening proposals and must never be persisted or presented as market structure. The scanner must not read `SRLevel` rows for target generation.
|
||||||
|
- **The Gate Target Ladder target is a gate input, never an exit.** `_atr_trailing_close()` does not take it as a parameter, and it must stay that way — take-profit exits were tested and halve CAGR. Any UI or alert that implies the trade exits at the target is a bug ([research](docs/research/sr-levels-and-exits.md#explicit-gate-target-ladder)).
|
||||||
|
- **The outcome evaluator evaluates ALL setups**, not just qualified ones — unqualified setups are the control group that makes the Track Record meaningful.
|
||||||
|
- **`SystemSetting` access goes through `app/services/settings_store.py`** — don't query the model directly.
|
||||||
|
- **Time-series data gets a real table** (see `benchmark_prices`, `regime_snapshots`); `SystemSetting` JSON is only for config and cached reports.
|
||||||
|
- **Discretionary overlay data is forward-only.** `signal_context_snapshots` captures composite/dimension/sentiment/fundamental context for new setups. Do not approximate historical sentiment/fundamental snapshots from today's data.
|
||||||
|
- Style: surgical changes, minimal new files; extend existing services rather than adding parallel ones.
|
||||||
|
|
||||||
|
### Where the strategy lives
|
||||||
|
|
||||||
|
| Concern | File |
|
||||||
|
|---|---|
|
||||||
|
| Composite + 5 dimension scores, weights | `app/services/scoring_service.py` |
|
||||||
|
| Residual 12-1 momentum ranking (the validated activation factor) | `app/services/momentum_service.py` |
|
||||||
|
| Setup construction (ATR stop, Gate Target Ladder targets) | `app/services/rr_scanner_service.py` |
|
||||||
|
| Confidence, targets, reach-probability, action | `app/services/recommendation_service.py` |
|
||||||
|
| Activation gate predicate (mirrored in TS) | `app/services/qualification.py` |
|
||||||
|
| Gate defaults / admin config | `app/services/admin_service.py` (`ACTIVATION_DEFAULTS`) |
|
||||||
|
| Backtest + factor rank-IC harness ("Signal edge") | `app/services/backtest_service.py` |
|
||||||
|
| Outcome resolution (target/stop/expired/ambiguous) | `app/services/outcome_service.py` |
|
||||||
|
| Paper trades + time/trailing/target auto-exit | `app/services/paper_trade_service.py` |
|
||||||
|
| Point-in-time setup context snapshots | `app/models/signal_context_snapshot.py` + `app/services/rr_scanner_service.py` |
|
||||||
|
| Structural S/R detection, Gate Target Ladder & zone clustering | `app/services/sr_service.py` |
|
||||||
|
| **Research log — what's been tested and rejected** | **`docs/research/`** |
|
||||||
|
| SPY benchmark for residual momentum + paper-trade alpha | `app/services/benchmark_service.py` |
|
||||||
|
| Pipelines & job registration | `app/scheduler.py` |
|
||||||
|
|
||||||
|
### Verifying changes
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pytest tests/ -q # backend; in-memory SQLite, no Postgres needed
|
||||||
|
cd frontend && npm run build # full tsc check — this IS the frontend "test"
|
||||||
|
```
|
||||||
|
|
||||||
|
- `npm test` in `frontend/` is dead (vitest isn't installed; there are no frontend test files). Use `npm run build`.
|
||||||
|
- Backend tests that exercise services which `commit()` need a plain session fixture, not the rolling-back `db_session` — copy the pattern in `tests/unit/test_rr_scanner_integration.py`.
|
||||||
|
- `ruff` reports ~11 pre-existing errors in old test files; those are not regressions.
|
||||||
|
|
||||||
|
### Deploying
|
||||||
|
|
||||||
|
Automated by Gitea Actions (`.gitea/workflows/deploy.yml`) on every push to `main`: **lint** (`ruff check app/`) → **test** (pytest; `alembic upgrade head` validated against a real Postgres 16 service; frontend `npm ci && npm run build`) → **deploy** (frontend built on the runner, rsync to the server excluding `.env`, `pip install -e .`, `alembic upgrade head`, restart `signalplatform.service`, health check on `127.0.0.1:8998`). Deploys are serialized by a concurrency group so overlapping pushes can't race.
|
||||||
|
|
||||||
|
Practical consequences:
|
||||||
|
|
||||||
|
- **A `ruff` error in `app/` or any failing backend test blocks the deploy.** (CI lints only `app/`, so the pre-existing ruff noise in old test files doesn't.)
|
||||||
|
- **Migrations run automatically on deploy** — no manual `alembic` step. A migration that only works on SQLite will fail CI against Postgres, by design.
|
||||||
|
- Pushing to `main` **is** deploying to production — there is no separate release step.
|
||||||
|
- After a gate or scanner change ships, trigger an R:R scan (Admin → Jobs) so live setups pick up new fields.
|
||||||
|
|
||||||
|
### Roadmap (agreed June 2026)
|
||||||
|
|
||||||
|
1. **Forward paper-test the momentum book** — the out-of-sample proof the backtest can't give. Watch Signals → Track Record (live vs backtest).
|
||||||
|
2. **Full IBKR integration** — read real positions, overlay entries/stops on charts, alert on holdings' score deterioration. (Paper trading, the lighter alternative, is done.)
|
||||||
|
3. Strategy experiments in the order listed under **Strategy Status** above — each one goes through the factor harness first.
|
||||||
|
|||||||
@@ -0,0 +1,41 @@
|
|||||||
|
"""add benchmark_prices
|
||||||
|
|
||||||
|
Stores daily closes for a benchmark index (SPY) so paper-trade alpha — trade
|
||||||
|
return minus the benchmark's return over the same holding period — can be
|
||||||
|
computed. Kept separate from the tradeable universe: the benchmark is not a
|
||||||
|
Ticker, so it never enters the scanner, momentum ranking, or rankings.
|
||||||
|
|
||||||
|
Revision ID: 012
|
||||||
|
Revises: 011
|
||||||
|
Create Date: 2026-06-28 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "012"
|
||||||
|
down_revision: Union[str, None] = "011"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"benchmark_prices",
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("symbol", sa.String(length=20), nullable=False),
|
||||||
|
sa.Column("date", sa.Date(), nullable=False),
|
||||||
|
sa.Column("close", sa.Float(), nullable=False),
|
||||||
|
sa.PrimaryKeyConstraint("id"),
|
||||||
|
sa.UniqueConstraint("symbol", "date", name="uq_benchmark_symbol_date"),
|
||||||
|
)
|
||||||
|
op.create_index("ix_benchmark_prices_symbol", "benchmark_prices", ["symbol"])
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_benchmark_prices_symbol", table_name="benchmark_prices")
|
||||||
|
op.drop_table("benchmark_prices")
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
"""add close_reason to paper_trades
|
||||||
|
|
||||||
|
Records how an open paper trade was closed (trailing | stop | target | manual) so
|
||||||
|
the close alert can summarise it and the UI can show why a position exited.
|
||||||
|
|
||||||
|
Revision ID: 013
|
||||||
|
Revises: 012
|
||||||
|
Create Date: 2026-06-30 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "013"
|
||||||
|
down_revision: Union[str, None] = "012"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column("paper_trades", sa.Column("close_reason", sa.String(length=10), nullable=True))
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("paper_trades", "close_reason")
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
"""add name to tickers
|
||||||
|
|
||||||
|
Company name (e.g. "Biogen Inc."), backfilled from Alpaca so the UI can show which
|
||||||
|
company is behind a symbol. Nullable — symbols Alpaca doesn't cover stay name-less.
|
||||||
|
|
||||||
|
Revision ID: 014
|
||||||
|
Revises: 013
|
||||||
|
Create Date: 2026-07-01 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "014"
|
||||||
|
down_revision: Union[str, None] = "013"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column("tickers", sa.Column("name", sa.String(length=120), nullable=True))
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("tickers", "name")
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
"""Phase 3 strategy adoption: time-based exit + confidence floor removed.
|
||||||
|
|
||||||
|
The July 2026 backtest (gate ablation graded under both exit models, plus the
|
||||||
|
capital-constrained portfolio simulation) concluded:
|
||||||
|
|
||||||
|
- The best exit is hold-to-horizon: keep the initial ATR stop and exit at the
|
||||||
|
30th trading day's close (+0.50R net/trade vs +0.13R for the S/R target
|
||||||
|
exit; simulated book +31.9% vs +23.7% CAGR at the same drawdown). The paper
|
||||||
|
trade exit-policy default is now ``time`` (30 trading days).
|
||||||
|
- The confidence floor adds nothing (identical net/trade with it removed,
|
||||||
|
under both exit models) while cutting ~25% of qualified trades. Its default
|
||||||
|
is now 0 (off).
|
||||||
|
|
||||||
|
Stored rows for these two settings were written under the old semantics, so
|
||||||
|
they are cleared here and the new code defaults take effect. Re-tune in
|
||||||
|
Admin -> Activation / Paper-Trade Exit if desired. Note: this changes which
|
||||||
|
setups qualify and how paper trades close, so Track Record comparability
|
||||||
|
resets from this deploy.
|
||||||
|
|
||||||
|
Revision ID: 015
|
||||||
|
Revises: 014
|
||||||
|
Create Date: 2026-07-02 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
# revision identifiers, used by Alembic.
|
||||||
|
revision: str = "015"
|
||||||
|
down_revision: Union[str, None] = "014"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.execute(
|
||||||
|
sa.text(
|
||||||
|
"DELETE FROM system_settings "
|
||||||
|
"WHERE key IN ('activation_min_confidence', 'paper_exit_mode')"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# One-way data reset: the old per-key values aren't recoverable. Code
|
||||||
|
# defaults apply until re-tuned, so there is nothing to restore.
|
||||||
|
pass
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
"""add signal context snapshots
|
||||||
|
|
||||||
|
Revision ID: 016
|
||||||
|
Revises: 015
|
||||||
|
Create Date: 2026-07-02 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "016"
|
||||||
|
down_revision: Union[str, None] = "015"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"signal_context_snapshots",
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("trade_setup_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("ticker_id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("detected_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("strategy_version", sa.String(length=80), nullable=False),
|
||||||
|
sa.Column("direction", sa.String(length=10), nullable=False),
|
||||||
|
sa.Column("entry_price", sa.Float(), nullable=False),
|
||||||
|
sa.Column("stop_loss", sa.Float(), nullable=False),
|
||||||
|
sa.Column("target", sa.Float(), nullable=False),
|
||||||
|
sa.Column("rr_ratio", sa.Float(), nullable=False),
|
||||||
|
sa.Column("confidence_score", sa.Float(), nullable=True),
|
||||||
|
sa.Column("recommended_action", sa.String(length=20), nullable=True),
|
||||||
|
sa.Column("risk_level", sa.String(length=10), nullable=True),
|
||||||
|
sa.Column("momentum_percentile", sa.Float(), nullable=True),
|
||||||
|
sa.Column("score_context_json", sa.Text(), nullable=False),
|
||||||
|
sa.Column("sentiment_context_json", sa.Text(), nullable=False),
|
||||||
|
sa.Column("fundamental_context_json", sa.Text(), nullable=False),
|
||||||
|
sa.ForeignKeyConstraint(["ticker_id"], ["tickers.id"], ondelete="CASCADE"),
|
||||||
|
sa.ForeignKeyConstraint(["trade_setup_id"], ["trade_setups.id"], ondelete="CASCADE"),
|
||||||
|
sa.PrimaryKeyConstraint("id"),
|
||||||
|
sa.UniqueConstraint("trade_setup_id", name="uq_signal_context_trade_setup"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_signal_context_ticker_detected",
|
||||||
|
"signal_context_snapshots",
|
||||||
|
["ticker_id", "detected_at"],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_signal_context_ticker_detected", table_name="signal_context_snapshots")
|
||||||
|
op.drop_table("signal_context_snapshots")
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
"""Add production strategy rank fields to trade setups.
|
||||||
|
|
||||||
|
Revision ID: 017
|
||||||
|
Revises: 016
|
||||||
|
Create Date: 2026-07-03 20:15:00.000000
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision = "017"
|
||||||
|
down_revision = "016"
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column("trade_setups", sa.Column("strategy_rank", sa.Float(), nullable=True))
|
||||||
|
op.add_column("trade_setups", sa.Column("volatility_percentile", sa.Float(), nullable=True))
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("trade_setups", "volatility_percentile")
|
||||||
|
op.drop_column("trade_setups", "strategy_rank")
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
"""Clear the stored paper exit mode so the promoted ATR-trail default applies.
|
||||||
|
|
||||||
|
The July 2026 promotion moved the paper-trade auto-exit default from ``time``
|
||||||
|
(30-day hold on the initial stop) to ``atr_trailing`` (initial stop + 3x ATR
|
||||||
|
trailing stop, max 30 trading days) — the exit the production backtest
|
||||||
|
validated, now implemented live in ``paper_trade_service``.
|
||||||
|
|
||||||
|
``get_exit_policy`` reads a stored ``paper_exit_mode`` row before falling back
|
||||||
|
to the code default, so any environment that persisted the old ``time`` value
|
||||||
|
(e.g. via Admin -> Paper-Trade Exit, or an earlier deploy) would silently keep
|
||||||
|
the old exit and never run the promoted ATR trail. Following the precedent of
|
||||||
|
migration 015, the stored row is cleared here so the new code default takes
|
||||||
|
effect. Admins can re-tune in Admin -> Paper-Trade Exit if desired. Note: this
|
||||||
|
changes how open paper trades close, so Track Record comparability resets from
|
||||||
|
this deploy.
|
||||||
|
|
||||||
|
Revision ID: 018
|
||||||
|
Revises: 017
|
||||||
|
Create Date: 2026-07-04 00:00:00.000000
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision = "018"
|
||||||
|
down_revision = "017"
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.execute(
|
||||||
|
sa.text("DELETE FROM system_settings WHERE key = 'paper_exit_mode'")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# One-way data reset: the prior stored value isn't recoverable. The code
|
||||||
|
# default applies until re-tuned, so there is nothing to restore.
|
||||||
|
pass
|
||||||
@@ -0,0 +1,61 @@
|
|||||||
|
"""Enforce singleton score and fundamental snapshots.
|
||||||
|
|
||||||
|
Revision ID: 019
|
||||||
|
Revises: 018
|
||||||
|
Create Date: 2026-07-11 00:00:00.000000
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision = "019"
|
||||||
|
down_revision = "018"
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
def _remove_duplicates(table: str, partition_by: str, order_by: str) -> None:
|
||||||
|
op.execute(
|
||||||
|
sa.text(
|
||||||
|
f"""
|
||||||
|
DELETE FROM {table}
|
||||||
|
WHERE id IN (
|
||||||
|
SELECT id FROM (
|
||||||
|
SELECT id, ROW_NUMBER() OVER (
|
||||||
|
PARTITION BY {partition_by}
|
||||||
|
ORDER BY {order_by} DESC, id DESC
|
||||||
|
) AS row_number
|
||||||
|
FROM {table}
|
||||||
|
) AS ranked
|
||||||
|
WHERE row_number > 1
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
_remove_duplicates("dimension_scores", "ticker_id, dimension", "computed_at")
|
||||||
|
_remove_duplicates("composite_scores", "ticker_id", "computed_at")
|
||||||
|
_remove_duplicates("fundamental_data", "ticker_id", "fetched_at")
|
||||||
|
|
||||||
|
op.create_unique_constraint(
|
||||||
|
"uq_dimension_score_ticker_dimension",
|
||||||
|
"dimension_scores",
|
||||||
|
["ticker_id", "dimension"],
|
||||||
|
)
|
||||||
|
op.create_unique_constraint("uq_composite_score_ticker", "composite_scores", ["ticker_id"])
|
||||||
|
op.create_unique_constraint("uq_fundamental_data_ticker", "fundamental_data", ["ticker_id"])
|
||||||
|
op.create_index("ix_sr_levels_ticker_id", "sr_levels", ["ticker_id"])
|
||||||
|
op.create_index("ix_trade_setups_ticker_rr", "trade_setups", ["ticker_id", "rr_ratio"])
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_trade_setups_ticker_rr", table_name="trade_setups")
|
||||||
|
op.drop_index("ix_sr_levels_ticker_id", table_name="sr_levels")
|
||||||
|
op.drop_constraint("uq_fundamental_data_ticker", "fundamental_data", type_="unique")
|
||||||
|
op.drop_constraint("uq_composite_score_ticker", "composite_scores", type_="unique")
|
||||||
|
op.drop_constraint("uq_dimension_score_ticker_dimension", "dimension_scores", type_="unique")
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
"""Drop the orphaned EV-gate activation settings.
|
||||||
|
|
||||||
|
``activation_min_expected_value`` and ``activation_min_target_probability`` are
|
||||||
|
leftovers from the June 2026 EV-gate redesign (migration 009). That gate was
|
||||||
|
superseded by the residual-momentum gate, and the current code reads neither key:
|
||||||
|
``admin_service._ACTIVATION_FLOAT_KEYS`` exposes only ``min_momentum_percentile``,
|
||||||
|
``min_rr`` and ``min_confidence``, and ``qualification.setup_qualifies`` gates on
|
||||||
|
those plus the hardcoded ``MIN_TARGET_PROBABILITY`` floor.
|
||||||
|
|
||||||
|
The rows are therefore inert but actively misleading: prod carries
|
||||||
|
``activation_min_target_probability = 50.0``, so anyone reading the DB (or an
|
||||||
|
Admin screen rendering it) would reasonably believe a 50% probability floor is
|
||||||
|
enforced. It is not — the real floor is the 20% constant in ``qualification.py``.
|
||||||
|
|
||||||
|
Reads never recreate them (``settings_store.get_value`` returns a default without
|
||||||
|
persisting), and the current Admin write path no longer emits these keys, so the
|
||||||
|
delete is permanent. Follows the precedent of migrations 009, 015 and 018.
|
||||||
|
|
||||||
|
Revision ID: 020
|
||||||
|
Revises: 019
|
||||||
|
Create Date: 2026-07-12 00:00:00.000000
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision = "020"
|
||||||
|
down_revision = "019"
|
||||||
|
branch_labels = None
|
||||||
|
depends_on = None
|
||||||
|
|
||||||
|
|
||||||
|
ORPHANED_KEYS = (
|
||||||
|
"activation_min_expected_value",
|
||||||
|
"activation_min_target_probability",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.execute(
|
||||||
|
sa.text(
|
||||||
|
"DELETE FROM system_settings WHERE key IN "
|
||||||
|
"('activation_min_expected_value', 'activation_min_target_probability')"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
# Restore the values prod carried before the delete. They are inert either
|
||||||
|
# way — no code path reads them — but this keeps the downgrade faithful.
|
||||||
|
# ``updated_at`` is NOT NULL with only a Python-side default, so raw SQL must
|
||||||
|
# supply it explicitly.
|
||||||
|
op.execute(
|
||||||
|
sa.text(
|
||||||
|
"INSERT INTO system_settings (key, value, updated_at) VALUES "
|
||||||
|
"('activation_min_expected_value', '0.01', CURRENT_TIMESTAMP), "
|
||||||
|
"('activation_min_target_probability', '50.0', CURRENT_TIMESTAMP) "
|
||||||
|
"ON CONFLICT (key) DO NOTHING"
|
||||||
|
)
|
||||||
|
)
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
"""add system_events table for operational warnings/errors
|
||||||
|
|
||||||
|
Revision ID: 021
|
||||||
|
Revises: 020
|
||||||
|
Create Date: 2026-07-14 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "021"
|
||||||
|
down_revision: Union[str, None] = "020"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"system_events",
|
||||||
|
sa.Column("id", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("severity", sa.String(length=16), nullable=False),
|
||||||
|
sa.Column("source", sa.String(length=64), nullable=False),
|
||||||
|
sa.Column("code", sa.String(length=64), nullable=False),
|
||||||
|
sa.Column("message", sa.Text(), nullable=False),
|
||||||
|
sa.Column("symbol", sa.String(length=20), nullable=True),
|
||||||
|
sa.Column("dedup_key", sa.String(length=200), nullable=True),
|
||||||
|
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("acknowledged_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
sa.PrimaryKeyConstraint("id"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_system_events_created_at",
|
||||||
|
"system_events",
|
||||||
|
["created_at"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_system_events_ack_created",
|
||||||
|
"system_events",
|
||||||
|
["acknowledged_at", "created_at"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_system_events_dedup_created",
|
||||||
|
"system_events",
|
||||||
|
["dedup_key", "created_at"],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_system_events_dedup_created", table_name="system_events")
|
||||||
|
op.drop_index("ix_system_events_ack_created", table_name="system_events")
|
||||||
|
op.drop_index("ix_system_events_created_at", table_name="system_events")
|
||||||
|
op.drop_table("system_events")
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
"""add persistent post-stop gate-reset observation
|
||||||
|
|
||||||
|
Revision ID: 022
|
||||||
|
Revises: 021
|
||||||
|
Create Date: 2026-07-17 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "022"
|
||||||
|
down_revision: Union[str, None] = "021"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"paper_trades",
|
||||||
|
sa.Column("reentry_gate_failed_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
)
|
||||||
|
op.add_column(
|
||||||
|
"paper_trades",
|
||||||
|
sa.Column(
|
||||||
|
"reentry_gate_requalified_at",
|
||||||
|
sa.DateTime(timezone=True),
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
# The policy starts at this deployment. Historical NULL values mean the
|
||||||
|
# scanner never recorded reset observations, not that those old episodes
|
||||||
|
# are still active. Mark both transitions complete so only stops created
|
||||||
|
# after the migration can open a re-entry lock.
|
||||||
|
op.execute(
|
||||||
|
sa.text(
|
||||||
|
"""
|
||||||
|
UPDATE paper_trades
|
||||||
|
SET reentry_gate_failed_at = closed_at,
|
||||||
|
reentry_gate_requalified_at = closed_at
|
||||||
|
WHERE status = 'closed'
|
||||||
|
AND close_reason = 'stop'
|
||||||
|
AND closed_at IS NOT NULL
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("paper_trades", "reentry_gate_requalified_at")
|
||||||
|
op.drop_column("paper_trades", "reentry_gate_failed_at")
|
||||||
@@ -0,0 +1,73 @@
|
|||||||
|
"""near-close schedule cutover + paper trade fill_mode era tag
|
||||||
|
|
||||||
|
Revision ID: 023
|
||||||
|
Revises: 022
|
||||||
|
Create Date: 2026-07-18 00:00:00.000000
|
||||||
|
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "023"
|
||||||
|
down_revision: Union[str, None] = "022"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
# Deliberate schedule rewrite (not a soft defaults refresh). Old stored values
|
||||||
|
# are logged then replaced so prod does not keep scanning at 07:00 Berlin.
|
||||||
|
_SCHEDULE_REWRITE: dict[str, str] = {
|
||||||
|
"schedule_timezone": "America/New_York",
|
||||||
|
"schedule_daily_pipeline_cron": "0 2 * * *",
|
||||||
|
"schedule_near_close_pipeline_cron": "30 15 * * 1-5",
|
||||||
|
"schedule_after_close_pipeline_cron": "45 16 * * 1-5",
|
||||||
|
"schedule_intraday_pipeline_cron": "0 10-15 * * 1-5",
|
||||||
|
"schedule_fundamentals_cron": "0 1 * * 1",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"paper_trades",
|
||||||
|
sa.Column("fill_mode", sa.String(length=20), nullable=True),
|
||||||
|
)
|
||||||
|
|
||||||
|
conn = op.get_bind()
|
||||||
|
settings = sa.table(
|
||||||
|
"system_settings",
|
||||||
|
sa.column("id", sa.Integer),
|
||||||
|
sa.column("key", sa.String),
|
||||||
|
sa.column("value", sa.Text),
|
||||||
|
sa.column("updated_at", sa.DateTime(timezone=True)),
|
||||||
|
)
|
||||||
|
now = sa.func.now()
|
||||||
|
|
||||||
|
for key, new_value in _SCHEDULE_REWRITE.items():
|
||||||
|
row = conn.execute(
|
||||||
|
sa.select(settings.c.value).where(settings.c.key == key)
|
||||||
|
).fetchone()
|
||||||
|
old_value = row[0] if row is not None else None
|
||||||
|
# Always log so ops can recover the pre-cutover schedule from migration output.
|
||||||
|
print(
|
||||||
|
f"schedule_cutover {key}: {old_value!r} -> {new_value!r}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
if row is None:
|
||||||
|
conn.execute(
|
||||||
|
sa.insert(settings).values(
|
||||||
|
key=key, value=new_value, updated_at=now
|
||||||
|
)
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
conn.execute(
|
||||||
|
sa.update(settings)
|
||||||
|
.where(settings.c.key == key)
|
||||||
|
.values(value=new_value, updated_at=now)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("paper_trades", "fill_mode")
|
||||||
|
# Do not restore old crons — unknown prior values; leave stored schedule as-is.
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
"""paper trade book tag (manual vs shadow) + weekday cron repair
|
||||||
|
|
||||||
|
Revision ID: 024
|
||||||
|
Revises: 023
|
||||||
|
Create Date: 2026-07-20 00:00:00.000000
|
||||||
|
|
||||||
|
Two things ship together because both are corrections to 023's stored state.
|
||||||
|
|
||||||
|
1. ``paper_trades.book`` separates the discretionary book from the automatic
|
||||||
|
shadow book. Everything that exists today was opened by hand, so the
|
||||||
|
backfill value is "manual".
|
||||||
|
|
||||||
|
2. 023 wrote weekday crons with a numeric day-of-week. APScheduler's
|
||||||
|
from_crontab() feeds field 5 to its own day_of_week where 0=Monday, so
|
||||||
|
"1-5" resolved to Tue-Sat: every Monday was skipped and the scanner ran on
|
||||||
|
Saturdays against stale data. Rewrite only the rows that still hold the
|
||||||
|
broken numeric form, so a hand-corrected setting is never clobbered.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "024"
|
||||||
|
down_revision: Union[str, None] = "023"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
# key -> (broken numeric form written by 023, corrected named form)
|
||||||
|
_CRON_REPAIR: dict[str, tuple[str, str]] = {
|
||||||
|
"schedule_near_close_pipeline_cron": ("30 15 * * 1-5", "30 15 * * mon-fri"),
|
||||||
|
"schedule_after_close_pipeline_cron": ("45 16 * * 1-5", "45 16 * * mon-fri"),
|
||||||
|
"schedule_intraday_pipeline_cron": ("0 10-15 * * 1-5", "0 10-15 * * mon-fri"),
|
||||||
|
"schedule_fundamentals_cron": ("0 1 * * 1", "0 1 * * mon"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
# server_default backfills existing rows, so no separate UPDATE is needed.
|
||||||
|
op.add_column(
|
||||||
|
"paper_trades",
|
||||||
|
sa.Column("book", sa.String(length=10), nullable=False, server_default="manual"),
|
||||||
|
)
|
||||||
|
|
||||||
|
# Literals are inlined rather than bound because bound parameters render as
|
||||||
|
# NULL under `alembic upgrade --sql`, which would silently produce a script
|
||||||
|
# that matches nothing. Every value here is a constant defined above.
|
||||||
|
for key, (broken, fixed) in _CRON_REPAIR.items():
|
||||||
|
op.execute(
|
||||||
|
f"UPDATE system_settings SET value = '{fixed}' " # noqa: S608
|
||||||
|
f"WHERE key = '{key}' AND value = '{broken}'"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("paper_trades", "book")
|
||||||
|
# Crons are deliberately left corrected — restoring the numeric form would
|
||||||
|
# reintroduce the skipped-Monday bug.
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
"""trade_setup scan_run_id — identity of the producing scan run
|
||||||
|
|
||||||
|
Revision ID: 025
|
||||||
|
Revises: 024
|
||||||
|
Create Date: 2026-07-21 00:00:00.000000
|
||||||
|
|
||||||
|
The shadow book must select the exact batch produced by its pipeline's scan.
|
||||||
|
Matching the scan-completion marker's run id proves which scan wrote last, but
|
||||||
|
setup selection was still a detected_at window that a concurrent manual scan
|
||||||
|
could write rows into. Stamping each row with its scan's run id lets the shadow
|
||||||
|
book select by identity instead. Existing rows are null (they predate the
|
||||||
|
column and are never traded by the shadow book).
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "025"
|
||||||
|
down_revision: Union[str, None] = "024"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"trade_setups",
|
||||||
|
sa.Column("scan_run_id", sa.String(length=32), nullable=True),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_trade_setups_scan_run_id", "trade_setups", ["scan_run_id"]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_trade_setups_scan_run_id", table_name="trade_setups")
|
||||||
|
op.drop_column("trade_setups", "scan_run_id")
|
||||||
@@ -0,0 +1,145 @@
|
|||||||
|
"""Dolt/SEC fundamentals schema — workstream A
|
||||||
|
|
||||||
|
Revision ID: 026
|
||||||
|
Revises: 025
|
||||||
|
Create Date: 2026-07-21 00:00:00.000000
|
||||||
|
|
||||||
|
Foundational schema for the Dolt bulk-data integration (workstream A): the
|
||||||
|
batch import-run audit table, the SEC-sourced immutable fundamental snapshots
|
||||||
|
(CIK-keyed, one row per accession), the Dolt earnings calendar/history, and the
|
||||||
|
SEC issuer identity columns on ``tickers``. No data is populated here — the
|
||||||
|
importers land in a later phase. ``fundamental_data`` is left untouched; its
|
||||||
|
cutover is gated separately (phase A5). ``data_import_runs`` is created first
|
||||||
|
because the other two tables carry an ``import_run_id`` FK to it.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "026"
|
||||||
|
down_revision: Union[str, None] = "025"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"data_import_runs",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column("source", sa.String(length=32), nullable=False),
|
||||||
|
sa.Column("revision", sa.String(length=64), nullable=True),
|
||||||
|
sa.Column("status", sa.String(length=16), nullable=False),
|
||||||
|
sa.Column("source_max_date", sa.Date(), nullable=True),
|
||||||
|
sa.Column("row_counts_json", sa.Text(), nullable=True),
|
||||||
|
sa.Column("validation_json", sa.Text(), nullable=True),
|
||||||
|
sa.Column("started_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
sa.Column("error_details", sa.Text(), nullable=True),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_data_import_runs_source_started", "data_import_runs", ["source", "started_at"]
|
||||||
|
)
|
||||||
|
|
||||||
|
op.create_table(
|
||||||
|
"fundamental_snapshots",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column("cik", sa.String(length=10), nullable=False),
|
||||||
|
sa.Column("accession", sa.String(length=25), nullable=False),
|
||||||
|
sa.Column("form", sa.String(length=12), nullable=False),
|
||||||
|
sa.Column("filed_date", sa.Date(), nullable=False),
|
||||||
|
sa.Column("accepted_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("period_start", sa.Date(), nullable=True),
|
||||||
|
sa.Column("period_end", sa.Date(), nullable=False),
|
||||||
|
sa.Column("fiscal_year", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("fiscal_period", sa.String(length=4), nullable=False),
|
||||||
|
# duration facts — cumulative YTD/FY
|
||||||
|
sa.Column("revenue", sa.Float(), nullable=True),
|
||||||
|
sa.Column("net_income", sa.Float(), nullable=True),
|
||||||
|
sa.Column("operating_income", sa.Float(), nullable=True),
|
||||||
|
sa.Column("diluted_eps", sa.Float(), nullable=True),
|
||||||
|
sa.Column("cfo", sa.Float(), nullable=True),
|
||||||
|
sa.Column("capex", sa.Float(), nullable=True),
|
||||||
|
sa.Column("depreciation_amortization", sa.Float(), nullable=True),
|
||||||
|
# balance-sheet facts — period-end
|
||||||
|
sa.Column("cash_and_st_investments", sa.Float(), nullable=True),
|
||||||
|
sa.Column("total_debt", sa.Float(), nullable=True),
|
||||||
|
sa.Column("shares_outstanding", sa.Float(), nullable=True),
|
||||||
|
sa.Column("shares_outstanding_date", sa.Date(), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"import_run_id",
|
||||||
|
sa.Integer(),
|
||||||
|
sa.ForeignKey("data_import_runs.id", ondelete="SET NULL"),
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.UniqueConstraint("accession", name="uq_fundamental_snapshots_accession"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period",
|
||||||
|
"fundamental_snapshots",
|
||||||
|
["cik", "fiscal_year", "fiscal_period"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period_end",
|
||||||
|
"fundamental_snapshots",
|
||||||
|
["cik", "period_end"],
|
||||||
|
)
|
||||||
|
|
||||||
|
op.create_table(
|
||||||
|
"earnings_events",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"ticker_id",
|
||||||
|
sa.Integer(),
|
||||||
|
sa.ForeignKey("tickers.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("announce_date", sa.Date(), nullable=False),
|
||||||
|
sa.Column("session", sa.String(length=10), nullable=False),
|
||||||
|
sa.Column("period_end", sa.Date(), nullable=True),
|
||||||
|
sa.Column("eps_estimate", sa.Float(), nullable=True),
|
||||||
|
sa.Column("eps_actual", sa.Float(), nullable=True),
|
||||||
|
sa.Column("source", sa.String(length=32), nullable=False),
|
||||||
|
sa.Column(
|
||||||
|
"import_run_id",
|
||||||
|
sa.Integer(),
|
||||||
|
sa.ForeignKey("data_import_runs.id", ondelete="SET NULL"),
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.UniqueConstraint("ticker_id", "announce_date", name="uq_earnings_ticker_announce"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_earnings_events_announce_date", "earnings_events", ["announce_date"]
|
||||||
|
)
|
||||||
|
|
||||||
|
# SEC issuer identity on tickers (nullable; the only ticker<->issuer join point).
|
||||||
|
op.add_column("tickers", sa.Column("cik", sa.String(length=10), nullable=True))
|
||||||
|
op.add_column("tickers", sa.Column("sic", sa.String(length=4), nullable=True))
|
||||||
|
op.add_column(
|
||||||
|
"tickers", sa.Column("sic_description", sa.String(length=160), nullable=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("tickers", "sic_description")
|
||||||
|
op.drop_column("tickers", "sic")
|
||||||
|
op.drop_column("tickers", "cik")
|
||||||
|
|
||||||
|
op.drop_index("ix_earnings_events_announce_date", table_name="earnings_events")
|
||||||
|
op.drop_table("earnings_events")
|
||||||
|
|
||||||
|
op.drop_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period_end", table_name="fundamental_snapshots"
|
||||||
|
)
|
||||||
|
op.drop_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period", table_name="fundamental_snapshots"
|
||||||
|
)
|
||||||
|
op.drop_table("fundamental_snapshots")
|
||||||
|
|
||||||
|
op.drop_index(
|
||||||
|
"ix_data_import_runs_source_started", table_name="data_import_runs"
|
||||||
|
)
|
||||||
|
op.drop_table("data_import_runs")
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
"""fundamental_snapshots.weighted_avg_diluted_shares — market-cap fallback
|
||||||
|
|
||||||
|
Revision ID: 027
|
||||||
|
Revises: 026
|
||||||
|
Create Date: 2026-07-24 00:00:00.000000
|
||||||
|
|
||||||
|
Multi-class issuers report the cover-page share count per share class. That is a
|
||||||
|
dimensional fact and Company Facts is non-dimensional, so it is absent entirely:
|
||||||
|
META has never tagged it, CMCSA stops in 2009, BRK-B in 2011, CHTR in 2016 (when
|
||||||
|
the Time Warner Cable deal made it multi-class). `shares_outstanding` is
|
||||||
|
therefore null for a large slice of the mega-cap universe, which silently removes
|
||||||
|
both `market_cap_est` and `fcf_yield`.
|
||||||
|
|
||||||
|
The weighted-average diluted count is always present (EPS requires it) and is
|
||||||
|
consolidated across classes. Measured against issuers where the true
|
||||||
|
point-in-time count IS available, it lands within ~0.6%: GOOGL 0.9936, MRNA
|
||||||
|
1.0045, AAPL 0.9974, MSFT 0.9978.
|
||||||
|
|
||||||
|
Stored as its own column rather than backfilled into `shares_outstanding`, so the
|
||||||
|
point-in-time column keeps its strict meaning and the fallback stays an explicit,
|
||||||
|
labelled read-time decision. Existing rows are null until a reparse.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "027"
|
||||||
|
down_revision: Union[str, None] = "026"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.add_column(
|
||||||
|
"fundamental_snapshots",
|
||||||
|
sa.Column("weighted_avg_diluted_shares", sa.Float(), nullable=True),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("fundamental_snapshots", "weighted_avg_diluted_shares")
|
||||||
@@ -0,0 +1,147 @@
|
|||||||
|
"""SEC filing retry queue and setup-quality gate
|
||||||
|
|
||||||
|
Revision ID: 028
|
||||||
|
Revises: 027
|
||||||
|
Create Date: 2026-08-03 00:00:00.000000
|
||||||
|
"""
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
import json
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "028"
|
||||||
|
down_revision: Union[str, None] = "027"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"sec_filing_gaps",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column("cik", sa.String(length=10), nullable=False),
|
||||||
|
sa.Column("accession", sa.String(length=25), nullable=False),
|
||||||
|
sa.Column("form", sa.String(length=12), nullable=True),
|
||||||
|
sa.Column("index_date", sa.Date(), nullable=True),
|
||||||
|
sa.Column("reason", sa.String(length=64), nullable=False),
|
||||||
|
sa.Column("coregistrant_ciks_json", sa.Text(), nullable=True),
|
||||||
|
sa.Column("first_seen_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("last_attempted_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("escalated_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
sa.UniqueConstraint("accession", name="uq_sec_filing_gaps_accession"),
|
||||||
|
)
|
||||||
|
op.create_index("ix_sec_filing_gaps_cik", "sec_filing_gaps", ["cik"])
|
||||||
|
_backfill_retry_queue()
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_index("ix_sec_filing_gaps_cik", table_name="sec_filing_gaps")
|
||||||
|
op.drop_table("sec_filing_gaps")
|
||||||
|
|
||||||
|
|
||||||
|
def _as_date(value) -> date | None:
|
||||||
|
if isinstance(value, datetime):
|
||||||
|
return value.date()
|
||||||
|
if isinstance(value, date):
|
||||||
|
return value
|
||||||
|
if isinstance(value, str):
|
||||||
|
try:
|
||||||
|
return date.fromisoformat(value)
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _backfill_retry_queue() -> None:
|
||||||
|
"""Materialize pre-queue promoted gaps once; runtime never scans history."""
|
||||||
|
bind = op.get_bind()
|
||||||
|
runs = sa.table(
|
||||||
|
"data_import_runs",
|
||||||
|
sa.column("source", sa.String()),
|
||||||
|
sa.column("status", sa.String()),
|
||||||
|
sa.column("validation_json", sa.Text()),
|
||||||
|
sa.column("source_max_date", sa.Date()),
|
||||||
|
sa.column("started_at", sa.DateTime(timezone=True)),
|
||||||
|
)
|
||||||
|
snapshots = sa.table(
|
||||||
|
"fundamental_snapshots",
|
||||||
|
sa.column("cik", sa.String()),
|
||||||
|
sa.column("accession", sa.String()),
|
||||||
|
sa.column("filed_date", sa.Date()),
|
||||||
|
)
|
||||||
|
gaps = sa.table(
|
||||||
|
"sec_filing_gaps",
|
||||||
|
sa.column("cik", sa.String()),
|
||||||
|
sa.column("accession", sa.String()),
|
||||||
|
sa.column("form", sa.String()),
|
||||||
|
sa.column("index_date", sa.Date()),
|
||||||
|
sa.column("reason", sa.String()),
|
||||||
|
sa.column("coregistrant_ciks_json", sa.Text()),
|
||||||
|
sa.column("first_seen_at", sa.DateTime(timezone=True)),
|
||||||
|
sa.column("last_attempted_at", sa.DateTime(timezone=True)),
|
||||||
|
sa.column("escalated_at", sa.DateTime(timezone=True)),
|
||||||
|
)
|
||||||
|
|
||||||
|
snapshot_rows = bind.execute(
|
||||||
|
sa.select(snapshots.c.cik, snapshots.c.accession, snapshots.c.filed_date)
|
||||||
|
).all()
|
||||||
|
resolved_accessions = {row.accession for row in snapshot_rows}
|
||||||
|
latest_filed_by_cik: dict[str, date] = {}
|
||||||
|
for row in snapshot_rows:
|
||||||
|
if row.filed_date is not None:
|
||||||
|
current = latest_filed_by_cik.get(row.cik)
|
||||||
|
if current is None or row.filed_date > current:
|
||||||
|
latest_filed_by_cik[row.cik] = row.filed_date
|
||||||
|
|
||||||
|
audit_rows = bind.execute(
|
||||||
|
sa.select(
|
||||||
|
runs.c.validation_json,
|
||||||
|
runs.c.source_max_date,
|
||||||
|
runs.c.started_at,
|
||||||
|
).where(
|
||||||
|
runs.c.source == "sec_facts",
|
||||||
|
runs.c.status == "promoted",
|
||||||
|
runs.c.validation_json.is_not(None),
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
now = datetime.now(timezone.utc)
|
||||||
|
candidates: dict[str, dict] = {}
|
||||||
|
for audit in audit_rows:
|
||||||
|
try:
|
||||||
|
summary = json.loads(audit.validation_json)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
continue
|
||||||
|
if not isinstance(summary, dict):
|
||||||
|
continue
|
||||||
|
for item in summary.get("missing_xbrl") or []:
|
||||||
|
accession = item.get("accession")
|
||||||
|
raw_cik = item.get("cik")
|
||||||
|
if not accession or raw_cik is None or accession in resolved_accessions:
|
||||||
|
continue
|
||||||
|
cik = str(raw_cik).zfill(10)
|
||||||
|
index_date = _as_date(item.get("index_date")) or _as_date(
|
||||||
|
audit.source_max_date
|
||||||
|
)
|
||||||
|
later_filed = latest_filed_by_cik.get(cik)
|
||||||
|
if index_date is not None and later_filed is not None and later_filed > index_date:
|
||||||
|
continue
|
||||||
|
first_seen = audit.started_at or now
|
||||||
|
existing = candidates.get(accession)
|
||||||
|
if existing is not None and existing["first_seen_at"] <= first_seen:
|
||||||
|
continue
|
||||||
|
candidates[accession] = {
|
||||||
|
"cik": cik,
|
||||||
|
"accession": accession,
|
||||||
|
"form": item.get("form"),
|
||||||
|
"index_date": index_date,
|
||||||
|
"reason": item.get("reason") or "not_in_companyfacts",
|
||||||
|
"coregistrant_ciks_json": json.dumps(item.get("coregistrants") or []),
|
||||||
|
"first_seen_at": first_seen,
|
||||||
|
"last_attempted_at": first_seen,
|
||||||
|
"escalated_at": None,
|
||||||
|
}
|
||||||
|
if candidates:
|
||||||
|
op.bulk_insert(gaps, list(candidates.values()))
|
||||||
+31
-3
@@ -37,6 +37,34 @@ class Settings(BaseSettings):
|
|||||||
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
||||||
alpha_vantage_api_key: str = ""
|
alpha_vantage_api_key: str = ""
|
||||||
|
|
||||||
|
# Dolt bulk-data — local clone of post-no-preference/earnings (workstream A).
|
||||||
|
# dolt_binary: full path when not on PATH (dev/Windows install). dolt_data_dir
|
||||||
|
# holds the clones; in production it MUST be outside the deploy tree (deploy is
|
||||||
|
# rsync --delete) — set DOLT_DATA_DIR to a persistent path. The earnings clone
|
||||||
|
# lives at <dolt_data_dir>/<dolt_earnings_subdir>.
|
||||||
|
dolt_binary: str = "dolt"
|
||||||
|
dolt_data_dir: str = "dolt-data"
|
||||||
|
dolt_earnings_subdir: str = "earnings"
|
||||||
|
# Headroom above the ~1.7 GB earnings clone (grows with pulls); 5 GB is a safe
|
||||||
|
# production floor — override lower only in a space-constrained dev box.
|
||||||
|
dolt_min_free_disk_gb: float = 5.0
|
||||||
|
# Bound every dolt subprocess so a hung pull/sql can't pin the import's
|
||||||
|
# connection + advisory lock indefinitely.
|
||||||
|
dolt_command_timeout_seconds: float = 600.0
|
||||||
|
|
||||||
|
# SEC EDGAR (workstream A, fundamentals). Fair-access policy REQUIRES an
|
||||||
|
# identifying User-Agent with a contact email — set a real one. Stay well
|
||||||
|
# under 10 req/s (spacing below); 403 means the UA/pattern is wrong → the
|
||||||
|
# client alerts and stops rather than retry-looping.
|
||||||
|
sec_user_agent: str = "signal-platform/1.0 (contact: set-a-real-email@example.com)"
|
||||||
|
sec_request_spacing_seconds: float = 0.2
|
||||||
|
sec_max_retries: int = 4
|
||||||
|
sec_request_timeout_seconds: float = 30.0
|
||||||
|
|
||||||
|
# A5 read-only comparison artifacts. Production must keep this outside the
|
||||||
|
# rsync deployment tree so the 5-7 day review window survives deploys.
|
||||||
|
fundamentals_parity_report_dir: str = "reports/fundamentals-parity"
|
||||||
|
|
||||||
# Regime Monitor — FRED (VIX level + HY credit spreads). Optional: without it
|
# Regime Monitor — FRED (VIX level + HY credit spreads). Optional: without it
|
||||||
# the volatility (P5) and credit-spread (F2) signals are reported as n/a.
|
# the volatility (P5) and credit-spread (F2) signals are reported as n/a.
|
||||||
fred_api_key: str = ""
|
fred_api_key: str = ""
|
||||||
@@ -51,7 +79,7 @@ class Settings(BaseSettings):
|
|||||||
# Sentiment search-budget controls (Gemini grounding free tier = 5000/month).
|
# Sentiment search-budget controls (Gemini grounding free tier = 5000/month).
|
||||||
# Scope (see _get_sentiment_priority_tickers): everything that matters is always
|
# Scope (see _get_sentiment_priority_tickers): everything that matters is always
|
||||||
# refreshed in full — open paper trades + the curated watchlist + top-pick
|
# refreshed in full — open paper trades + the curated watchlist + top-pick
|
||||||
# feeders (momentum leaders with a tradeable long setup) — plus a top-N composite
|
# feeders (residual-momentum leaders with a tradeable long setup) — plus a top-N composite
|
||||||
# discovery net. No per-run cap: the set is naturally bounded (watchlist <= 20,
|
# discovery net. No per-run cap: the set is naturally bounded (watchlist <= 20,
|
||||||
# composite <= top_composite), so a full refresh stays well inside the free tier.
|
# composite <= top_composite), so a full refresh stays well inside the free tier.
|
||||||
# Skip anything refreshed within fresh_hours (5 days: sentiment shifts slowly and
|
# Skip anything refreshed within fresh_hours (5 days: sentiment shifts slowly and
|
||||||
@@ -59,8 +87,8 @@ class Settings(BaseSettings):
|
|||||||
sentiment_fresh_hours: int = 120
|
sentiment_fresh_hours: int = 120
|
||||||
sentiment_top_composite: int = 30
|
sentiment_top_composite: int = 30
|
||||||
fundamental_fetch_frequency: str = "weekly" # quarterly-ish data; weekly conserves API quota
|
fundamental_fetch_frequency: str = "weekly" # quarterly-ish data; weekly conserves API quota
|
||||||
rr_scan_frequency: str = "daily"
|
rr_scan_frequency: str = "daily" # legacy label; qualifying scan is cron near-close
|
||||||
alerts_frequency: str = "hourly"
|
# alerts_frequency removed: alerts fire only via morning + near-close pipelines
|
||||||
fundamental_rate_limit_retries: int = 3
|
fundamental_rate_limit_retries: int = 3
|
||||||
fundamental_rate_limit_backoff_seconds: int = 15
|
fundamental_rate_limit_backoff_seconds: int = 15
|
||||||
# Pause between tickers in the bulk fundamentals job. Free tiers throttle
|
# Pause between tickers in the bulk fundamentals job. Free tiers throttle
|
||||||
|
|||||||
@@ -1,5 +1,8 @@
|
|||||||
from collections.abc import AsyncGenerator
|
from collections.abc import AsyncGenerator
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from sqlalchemy.dialects.postgresql import insert as postgresql_insert
|
||||||
|
from sqlalchemy.dialects.sqlite import insert as sqlite_insert
|
||||||
from sqlalchemy.ext.asyncio import (
|
from sqlalchemy.ext.asyncio import (
|
||||||
AsyncSession,
|
AsyncSession,
|
||||||
async_sessionmaker,
|
async_sessionmaker,
|
||||||
@@ -28,6 +31,13 @@ class Base(DeclarativeBase):
|
|||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def insert_for_session(session: AsyncSession, table: Any) -> Any:
|
||||||
|
"""Build a dialect-native INSERT that supports conflict handling."""
|
||||||
|
if session.get_bind().dialect.name == "postgresql":
|
||||||
|
return postgresql_insert(table)
|
||||||
|
return sqlite_insert(table)
|
||||||
|
|
||||||
|
|
||||||
async def get_session() -> AsyncGenerator[AsyncSession, None]:
|
async def get_session() -> AsyncGenerator[AsyncSession, None]:
|
||||||
async with async_session_factory() as session:
|
async with async_session_factory() as session:
|
||||||
yield session
|
yield session
|
||||||
|
|||||||
+2
-49
@@ -3,56 +3,9 @@
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# SSL + proxy injection — MUST happen before any HTTP client imports
|
# SSL + proxy injection — MUST happen before any HTTP client imports
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
import os as _os
|
from app.ssl_bootstrap import bootstrap_ssl
|
||||||
import ssl as _ssl
|
|
||||||
from pathlib import Path as _Path
|
|
||||||
|
|
||||||
_COMBINED_CERT = _Path(__file__).resolve().parent.parent / "combined-ca-bundle.pem"
|
bootstrap_ssl()
|
||||||
|
|
||||||
if _COMBINED_CERT.exists():
|
|
||||||
_cert_path = str(_COMBINED_CERT)
|
|
||||||
# Env vars for libraries that respect them (requests, urllib3)
|
|
||||||
_os.environ["SSL_CERT_FILE"] = _cert_path
|
|
||||||
_os.environ["REQUESTS_CA_BUNDLE"] = _cert_path
|
|
||||||
_os.environ["CURL_CA_BUNDLE"] = _cert_path
|
|
||||||
|
|
||||||
# Monkey-patch ssl.create_default_context so that ALL libraries
|
|
||||||
# (aiohttp, httpx, google-genai, alpaca-py, etc.) automatically
|
|
||||||
# use our combined CA bundle that includes the corporate root cert.
|
|
||||||
_original_create_default_context = _ssl.create_default_context
|
|
||||||
|
|
||||||
def _patched_create_default_context(
|
|
||||||
purpose=_ssl.Purpose.SERVER_AUTH, *, cafile=None, capath=None, cadata=None
|
|
||||||
):
|
|
||||||
ctx = _original_create_default_context(
|
|
||||||
purpose, cafile=cafile, capath=capath, cadata=cadata
|
|
||||||
)
|
|
||||||
# Always load our combined bundle on top of whatever was loaded
|
|
||||||
ctx.load_verify_locations(cafile=_cert_path)
|
|
||||||
return ctx
|
|
||||||
|
|
||||||
_ssl.create_default_context = _patched_create_default_context
|
|
||||||
|
|
||||||
# Also patch aiohttp's cached SSL context objects directly, since
|
|
||||||
# aiohttp creates them at import time and may have already cached
|
|
||||||
# a context without our corporate CA bundle.
|
|
||||||
try:
|
|
||||||
import aiohttp.connector as _aio_conn
|
|
||||||
if hasattr(_aio_conn, '_SSL_CONTEXT_VERIFIED') and _aio_conn._SSL_CONTEXT_VERIFIED is not None:
|
|
||||||
_aio_conn._SSL_CONTEXT_VERIFIED.load_verify_locations(cafile=_cert_path)
|
|
||||||
if hasattr(_aio_conn, '_SSL_CONTEXT_UNVERIFIED') and _aio_conn._SSL_CONTEXT_UNVERIFIED is not None:
|
|
||||||
_aio_conn._SSL_CONTEXT_UNVERIFIED.load_verify_locations(cafile=_cert_path)
|
|
||||||
except ImportError:
|
|
||||||
pass
|
|
||||||
|
|
||||||
# Corporate proxy — needed when Kiro spawns the process (no .zshrc sourced)
|
|
||||||
# Only enable this if explicitly requested via environment variable.
|
|
||||||
if _os.environ.get("USE_CORP_PROXY", "0") == "1":
|
|
||||||
_PROXY = "http://aproxy.corproot.net:8080"
|
|
||||||
_NO_PROXY = "corproot.net,sharedtcs.net,127.0.0.1,localhost,bix.swisscom.com,swisscom.com"
|
|
||||||
_os.environ.setdefault("HTTP_PROXY", _PROXY)
|
|
||||||
_os.environ.setdefault("HTTPS_PROXY", _PROXY)
|
|
||||||
_os.environ.setdefault("NO_PROXY", _NO_PROXY)
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import sys
|
import sys
|
||||||
|
|||||||
@@ -3,6 +3,9 @@ from app.models.ohlcv import OHLCVRecord
|
|||||||
from app.models.user import User
|
from app.models.user import User
|
||||||
from app.models.sentiment import SentimentScore
|
from app.models.sentiment import SentimentScore
|
||||||
from app.models.fundamental import FundamentalData
|
from app.models.fundamental import FundamentalData
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
from app.models.score import DimensionScore, CompositeScore
|
from app.models.score import DimensionScore, CompositeScore
|
||||||
from app.models.sr_level import SRLevel
|
from app.models.sr_level import SRLevel
|
||||||
from app.models.trade_setup import TradeSetup
|
from app.models.trade_setup import TradeSetup
|
||||||
@@ -11,6 +14,10 @@ from app.models.settings import SystemSetting, IngestionProgress
|
|||||||
from app.models.alert import AlertLog
|
from app.models.alert import AlertLog
|
||||||
from app.models.paper_trade import PaperTrade
|
from app.models.paper_trade import PaperTrade
|
||||||
from app.models.regime_snapshot import RegimeSnapshot
|
from app.models.regime_snapshot import RegimeSnapshot
|
||||||
|
from app.models.benchmark_price import BenchmarkPrice
|
||||||
|
from app.models.signal_context_snapshot import SignalContextSnapshot
|
||||||
|
from app.models.system_event import SystemEvent
|
||||||
|
from app.models.sec_filing_gap import SecFilingGap
|
||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
"Ticker",
|
"Ticker",
|
||||||
@@ -18,6 +25,9 @@ __all__ = [
|
|||||||
"User",
|
"User",
|
||||||
"SentimentScore",
|
"SentimentScore",
|
||||||
"FundamentalData",
|
"FundamentalData",
|
||||||
|
"FundamentalSnapshot",
|
||||||
|
"EarningsEvent",
|
||||||
|
"DataImportRun",
|
||||||
"DimensionScore",
|
"DimensionScore",
|
||||||
"CompositeScore",
|
"CompositeScore",
|
||||||
"SRLevel",
|
"SRLevel",
|
||||||
@@ -28,4 +38,8 @@ __all__ = [
|
|||||||
"AlertLog",
|
"AlertLog",
|
||||||
"PaperTrade",
|
"PaperTrade",
|
||||||
"RegimeSnapshot",
|
"RegimeSnapshot",
|
||||||
|
"BenchmarkPrice",
|
||||||
|
"SignalContextSnapshot",
|
||||||
|
"SystemEvent",
|
||||||
|
"SecFilingGap",
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
from datetime import date as date_type
|
||||||
|
|
||||||
|
from sqlalchemy import Date, Float, String, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class BenchmarkPrice(Base):
|
||||||
|
"""Daily close for a benchmark index (e.g. SPY), used to compute trade alpha.
|
||||||
|
|
||||||
|
A standalone price series, deliberately NOT a tracked ``Ticker`` — so the
|
||||||
|
benchmark never becomes a trade candidate or rankings-table row. Its closes
|
||||||
|
are used for residual momentum and trade alpha. One row per (symbol, date).
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "benchmark_prices"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("symbol", "date", name="uq_benchmark_symbol_date"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
symbol: Mapped[str] = mapped_column(String(20), nullable=False, index=True)
|
||||||
|
date: Mapped[date_type] = mapped_column(Date, nullable=False)
|
||||||
|
close: Mapped[float] = mapped_column(Float, nullable=False)
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
from datetime import date, datetime
|
||||||
|
|
||||||
|
from sqlalchemy import Date, DateTime, Index, String, Text
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class DataImportRun(Base):
|
||||||
|
"""One row per bulk-import attempt (SEC facts / Dolt earnings / Dolt stocks).
|
||||||
|
|
||||||
|
Lean audit record for the batch import framework: every attempt is logged,
|
||||||
|
whether it promoted, was a ``no_op`` (unchanged revision), was ``deferred``
|
||||||
|
for an expected retry, or ``failed``.
|
||||||
|
``row_counts`` and ``validation`` hold JSON strings (repo convention — see
|
||||||
|
``fundamental_data.unavailable_fields_json``), not JSONB; the validation
|
||||||
|
blob carries reconciliation/discrepancy summaries so no separate conflicts
|
||||||
|
table is needed. One run per source at a time is enforced at write time by a
|
||||||
|
Postgres advisory lock keyed by ``source``.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "data_import_runs"
|
||||||
|
__table_args__ = (
|
||||||
|
Index("ix_data_import_runs_source_started", "source", "started_at"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
# sec_facts | dolt_earnings | dolt_stocks
|
||||||
|
source: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||||
|
# Dolt commit hash, or SEC archive SHA-256. Null until known.
|
||||||
|
revision: Mapped[str | None] = mapped_column(String(64), nullable=True)
|
||||||
|
# running | validated | promoted | no_op | deferred | failed
|
||||||
|
status: Mapped[str] = mapped_column(String(16), nullable=False)
|
||||||
|
source_max_date: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
row_counts_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
validation_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
started_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
|
)
|
||||||
|
completed_at: Mapped[datetime | None] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=True
|
||||||
|
)
|
||||||
|
# Failure detail, or the non-error reason when status is deferred.
|
||||||
|
error_details: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
@@ -0,0 +1,42 @@
|
|||||||
|
from datetime import date, datetime
|
||||||
|
|
||||||
|
from sqlalchemy import Date, DateTime, Float, ForeignKey, Index, String, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsEvent(Base):
|
||||||
|
"""Earnings calendar + surprise history, sourced from the DoltHub earnings repo.
|
||||||
|
|
||||||
|
Forward rows (``announce_date`` > today) are the calendar; past rows are
|
||||||
|
results. Rescheduling is handled in the importer's promotion transaction:
|
||||||
|
this source's future-dated rows are deleted and re-inserted from the new
|
||||||
|
snapshot so moved/cancelled dates never linger; past rows are never deleted.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "earnings_events"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("ticker_id", "announce_date", name="uq_earnings_ticker_announce"),
|
||||||
|
Index("ix_earnings_events_announce_date", "announce_date"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("tickers.id", ondelete="CASCADE"), nullable=False
|
||||||
|
)
|
||||||
|
announce_date: Mapped[date] = mapped_column(Date, nullable=False)
|
||||||
|
# bmo | amc | unknown (source coverage is partial)
|
||||||
|
session: Mapped[str] = mapped_column(String(10), nullable=False, default="unknown")
|
||||||
|
period_end: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
eps_estimate: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
eps_actual: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
source: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||||
|
import_run_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True
|
||||||
|
)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
|
)
|
||||||
|
|
||||||
|
ticker = relationship("Ticker", back_populates="earnings_events")
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
from datetime import date, datetime
|
from datetime import date, datetime
|
||||||
|
|
||||||
from sqlalchemy import Date, DateTime, Float, ForeignKey, Text
|
from sqlalchemy import Date, DateTime, Float, ForeignKey, Text, UniqueConstraint
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
from app.database import Base
|
from app.database import Base
|
||||||
@@ -8,6 +8,9 @@ from app.database import Base
|
|||||||
|
|
||||||
class FundamentalData(Base):
|
class FundamentalData(Base):
|
||||||
__tablename__ = "fundamental_data"
|
__tablename__ = "fundamental_data"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("ticker_id", name="uq_fundamental_data_ticker"),
|
||||||
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(primary_key=True)
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
ticker_id: Mapped[int] = mapped_column(
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
|
|||||||
@@ -0,0 +1,87 @@
|
|||||||
|
from datetime import date, datetime
|
||||||
|
|
||||||
|
from sqlalchemy import Date, DateTime, Float, ForeignKey, Index, String, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class FundamentalSnapshot(Base):
|
||||||
|
"""CIK-keyed, one immutable row per SEC accession.
|
||||||
|
|
||||||
|
Keyed by issuer (CIK), not ticker — multi-class issuers (GOOG/GOOGL) share
|
||||||
|
one CIK and one set of fundamentals; the ``tickers.cik`` column is the only
|
||||||
|
join point. Amendments are retained: every accession is a distinct immutable
|
||||||
|
row, and readers resolve (cik, fiscal_year, fiscal_period) at read time by
|
||||||
|
taking the newest ``accepted_at`` **per field**, falling back to the newest
|
||||||
|
accession that actually reports one — a partial amendment (a 10-K/A adding
|
||||||
|
Part III reports no financial facts) must not blank the period — no flags, no mutation.
|
||||||
|
|
||||||
|
**Facts are stored as the filing reports them, never as derived quarters.**
|
||||||
|
Duration facts (revenue, net_income, operating_income, diluted_eps, cfo,
|
||||||
|
capex, depreciation_amortization) hold the filing's normalized **cumulative
|
||||||
|
YTD/FY** value over (period_start -> period_end). Balance-sheet facts
|
||||||
|
(cash_and_st_investments, total_debt, shares_outstanding) are **period-end**
|
||||||
|
values. ``shares_outstanding`` is a single consolidated point-in-time count —
|
||||||
|
the ``dei:EntityCommonStockSharesOutstanding`` cover-page fact, or
|
||||||
|
``us-gaap:CommonStockSharesOutstanding`` at period end when no dei fact exists
|
||||||
|
(e.g. Alphabet). It is never a class sum (companyfacts is non-dimensional) nor
|
||||||
|
the weighted-average diluted count, since both consumers (estimated market cap,
|
||||||
|
YoY dilution read) want a point-in-time value. Discrete quarters (10-Q YTD
|
||||||
|
deltas, Q4 = FY - Q1..Q3), TTM, YoY and
|
||||||
|
the quarter tape are all derived at read time — so non-calendar fiscal years
|
||||||
|
resolve correctly and a later amendment never leaves a stale frozen quarter.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "fundamental_snapshots"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("accession", name="uq_fundamental_snapshots_accession"),
|
||||||
|
Index("ix_fundamental_snapshots_cik_period", "cik", "fiscal_year", "fiscal_period"),
|
||||||
|
Index("ix_fundamental_snapshots_cik_period_end", "cik", "period_end"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
cik: Mapped[str] = mapped_column(String(10), nullable=False)
|
||||||
|
accession: Mapped[str] = mapped_column(String(25), nullable=False)
|
||||||
|
form: Mapped[str] = mapped_column(String(12), nullable=False) # 10-Q, 10-K, 10-K/A ...
|
||||||
|
filed_date: Mapped[date] = mapped_column(Date, nullable=False)
|
||||||
|
# Kept although PIT enforcement is deferred (one timestamp now vs painful retrofit).
|
||||||
|
accepted_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
||||||
|
|
||||||
|
# Period identity — required to align non-calendar fiscal years and to derive
|
||||||
|
# discrete quarters from cumulative facts.
|
||||||
|
period_start: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
period_end: Mapped[date] = mapped_column(Date, nullable=False)
|
||||||
|
fiscal_year: Mapped[int] = mapped_column(nullable=False)
|
||||||
|
fiscal_period: Mapped[str] = mapped_column(String(4), nullable=False) # Q1|Q2|Q3|Q4|FY
|
||||||
|
|
||||||
|
# Duration facts — cumulative YTD/FY over (period_start -> period_end).
|
||||||
|
revenue: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
net_income: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
operating_income: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
diluted_eps: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
cfo: Mapped[float | None] = mapped_column(Float, nullable=True) # cash flow from operations
|
||||||
|
capex: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
depreciation_amortization: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
|
||||||
|
# Balance-sheet facts — period-end values.
|
||||||
|
cash_and_st_investments: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
total_debt: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
shares_outstanding: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
# The cover-page share count (dei:EntityCommonStockSharesOutstanding) is
|
||||||
|
# reported "as of" its own date, which can differ from period_end — store it
|
||||||
|
# so market cap uses the right point-in-time count.
|
||||||
|
shares_outstanding_date: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
# Weighted-average diluted count for the filing's most recent quarter — the
|
||||||
|
# market-cap fallback when the cover-page count is absent, which it always is
|
||||||
|
# for multi-class issuers (per-class facts are dimensional, and companyfacts
|
||||||
|
# is not). An average is not cumulative, so unlike the duration facts above
|
||||||
|
# this is NOT a YTD value: it is the shortest-span fact ending at period_end.
|
||||||
|
weighted_avg_diluted_shares: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
|
||||||
|
import_run_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True
|
||||||
|
)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
|
)
|
||||||
@@ -34,3 +34,27 @@ class PaperTrade(Base):
|
|||||||
)
|
)
|
||||||
close_price: Mapped[float | None] = mapped_column(Float, nullable=True)
|
close_price: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
closed_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
closed_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||||
|
# How the trade was closed: "time" | "trailing" | "stop" | "target" | "manual".
|
||||||
|
close_reason: Mapped[str | None] = mapped_column(String(10), nullable=True)
|
||||||
|
# A trade stopped at its initial stop starts a re-entry gate-reset episode.
|
||||||
|
# The daily full-universe scanner records both state transitions: the first
|
||||||
|
# failed gate observation and a later fresh qualification. Re-entry remains
|
||||||
|
# non-actionable until both timestamps exist.
|
||||||
|
reentry_gate_failed_at: Mapped[datetime | None] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=True
|
||||||
|
)
|
||||||
|
reentry_gate_requalified_at: Mapped[datetime | None] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=True
|
||||||
|
)
|
||||||
|
# Execution era for forward vs backtest comparison:
|
||||||
|
# null/legacy = pre-cutover morning-scan, "near_close" = post near-close cutover.
|
||||||
|
fill_mode: Mapped[str | None] = mapped_column(String(20), nullable=True)
|
||||||
|
# Which book this trade belongs to:
|
||||||
|
# "manual" — discretionary, opened by the user from a qualified setup
|
||||||
|
# "shadow" — opened automatically by the validated strategy (top-ranked
|
||||||
|
# qualified up to capacity, 1% risk). The shadow book is the
|
||||||
|
# faithful live twin of the backtest; the two books share the
|
||||||
|
# same exit policy so the only difference is *selection*.
|
||||||
|
# Gate-reset re-entry state is tracked per book — the books diverge as soon
|
||||||
|
# as their entries differ, and each must see its own trade history.
|
||||||
|
book: Mapped[str] = mapped_column(String(10), nullable=False, default="manual")
|
||||||
|
|||||||
@@ -8,12 +8,12 @@ from app.database import Base
|
|||||||
|
|
||||||
|
|
||||||
class RegimeSnapshot(Base):
|
class RegimeSnapshot(Base):
|
||||||
"""Daily snapshot of the AI/Tech regime-change index.
|
"""Daily point-in-time snapshot of the AI/Tech Regime Monitor.
|
||||||
|
|
||||||
One row per calendar date (unique). ``breakdown_json`` holds the full
|
One row per calendar date (unique). ``breakdown_json`` holds the full
|
||||||
per-signal breakdown plus the raw inputs, so reads need no recomputation and
|
``breakdown_json`` is authoritative for v2 State, Warning, source dates,
|
||||||
the 7/30-day trend is just a query over ``total_score``. Decoupled from the
|
coverage, and fixed-basket metadata. ``total_score``/``band`` retain the v2
|
||||||
rest of the platform: nothing reads this to gate or score trades.
|
State reading for schema compatibility. Nothing reads this to gate trades.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
__tablename__ = "regime_snapshots"
|
__tablename__ = "regime_snapshots"
|
||||||
|
|||||||
+7
-1
@@ -1,6 +1,6 @@
|
|||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import Boolean, DateTime, Float, ForeignKey, String, Text
|
from sqlalchemy import Boolean, DateTime, Float, ForeignKey, String, Text, UniqueConstraint
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
from app.database import Base
|
from app.database import Base
|
||||||
@@ -8,6 +8,9 @@ from app.database import Base
|
|||||||
|
|
||||||
class DimensionScore(Base):
|
class DimensionScore(Base):
|
||||||
__tablename__ = "dimension_scores"
|
__tablename__ = "dimension_scores"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("ticker_id", "dimension", name="uq_dimension_score_ticker_dimension"),
|
||||||
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(primary_key=True)
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
ticker_id: Mapped[int] = mapped_column(
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
@@ -25,6 +28,9 @@ class DimensionScore(Base):
|
|||||||
|
|
||||||
class CompositeScore(Base):
|
class CompositeScore(Base):
|
||||||
__tablename__ = "composite_scores"
|
__tablename__ = "composite_scores"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("ticker_id", name="uq_composite_score_ticker"),
|
||||||
|
)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(primary_key=True)
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
ticker_id: Mapped[int] = mapped_column(
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
|
|||||||
@@ -0,0 +1,32 @@
|
|||||||
|
from datetime import date, datetime
|
||||||
|
|
||||||
|
from sqlalchemy import Date, DateTime, Index, String, Text, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class SecFilingGap(Base):
|
||||||
|
"""Active SEC filing that could not yet be reconstructed.
|
||||||
|
|
||||||
|
Rows form a small retry queue. Successful snapshot ingestion deletes the
|
||||||
|
matching row; a later valid filing supersedes it. While a current row remains,
|
||||||
|
tickers mapped to its CIK are not eligible for actionable trade setups.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "sec_filing_gaps"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("accession", name="uq_sec_filing_gaps_accession"),
|
||||||
|
Index("ix_sec_filing_gaps_cik", "cik"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
cik: Mapped[str] = mapped_column(String(10), nullable=False)
|
||||||
|
accession: Mapped[str] = mapped_column(String(25), nullable=False)
|
||||||
|
form: Mapped[str | None] = mapped_column(String(12), nullable=True)
|
||||||
|
index_date: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
reason: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
|
coregistrant_ciks_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
first_seen_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
||||||
|
last_attempted_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
||||||
|
escalated_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import DateTime, Float, ForeignKey, String, Text
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class SignalContextSnapshot(Base):
|
||||||
|
"""Point-in-time context captured when a trade setup is generated.
|
||||||
|
|
||||||
|
This stores the discretionary overlay inputs (scores, sentiment,
|
||||||
|
fundamentals) as they looked at detection time, so future analysis can test
|
||||||
|
whether human filtering improved or hurt the qualified-list strategy.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "signal_context_snapshots"
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
trade_setup_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("trade_setups.id", ondelete="CASCADE"), nullable=False, unique=True
|
||||||
|
)
|
||||||
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("tickers.id", ondelete="CASCADE"), nullable=False
|
||||||
|
)
|
||||||
|
detected_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
||||||
|
strategy_version: Mapped[str] = mapped_column(String(80), nullable=False)
|
||||||
|
|
||||||
|
direction: Mapped[str] = mapped_column(String(10), nullable=False)
|
||||||
|
entry_price: Mapped[float] = mapped_column(Float, nullable=False)
|
||||||
|
stop_loss: Mapped[float] = mapped_column(Float, nullable=False)
|
||||||
|
target: Mapped[float] = mapped_column(Float, nullable=False)
|
||||||
|
rr_ratio: Mapped[float] = mapped_column(Float, nullable=False)
|
||||||
|
confidence_score: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
recommended_action: Mapped[str | None] = mapped_column(String(20), nullable=True)
|
||||||
|
risk_level: Mapped[str | None] = mapped_column(String(10), nullable=True)
|
||||||
|
momentum_percentile: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
|
||||||
|
score_context_json: Mapped[str] = mapped_column(Text, nullable=False, default="{}")
|
||||||
|
sentiment_context_json: Mapped[str] = mapped_column(Text, nullable=False, default="{}")
|
||||||
|
fundamental_context_json: Mapped[str] = mapped_column(Text, nullable=False, default="{}")
|
||||||
|
|
||||||
|
trade_setup = relationship("TradeSetup")
|
||||||
|
ticker = relationship("Ticker")
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import DateTime, Float, ForeignKey, Integer, String
|
from sqlalchemy import DateTime, Float, ForeignKey, Index, Integer, String
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
from app.database import Base
|
from app.database import Base
|
||||||
@@ -8,6 +8,7 @@ from app.database import Base
|
|||||||
|
|
||||||
class SRLevel(Base):
|
class SRLevel(Base):
|
||||||
__tablename__ = "sr_levels"
|
__tablename__ = "sr_levels"
|
||||||
|
__table_args__ = (Index("ix_sr_levels_ticker_id", "ticker_id"),)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(primary_key=True)
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
ticker_id: Mapped[int] = mapped_column(
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
|
|||||||
@@ -0,0 +1,37 @@
|
|||||||
|
"""Operational system events (warnings/errors) for the admin UI and top-nav badge."""
|
||||||
|
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from sqlalchemy import DateTime, Index, String, Text
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class SystemEvent(Base):
|
||||||
|
"""Durable warning/error record (jobs, ingestion, pipelines).
|
||||||
|
|
||||||
|
``acknowledged_at`` is set when a user dismisses the nav badge / clears
|
||||||
|
events — history still shows in Admin → Jobs for the retention window.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "system_events"
|
||||||
|
__table_args__ = (
|
||||||
|
Index("ix_system_events_created_at", "created_at"),
|
||||||
|
Index("ix_system_events_ack_created", "acknowledged_at", "created_at"),
|
||||||
|
Index("ix_system_events_dedup_created", "dedup_key", "created_at"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
severity: Mapped[str] = mapped_column(String(16), nullable=False) # warning | error
|
||||||
|
source: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
|
code: Mapped[str] = mapped_column(String(64), nullable=False)
|
||||||
|
message: Mapped[str] = mapped_column(Text, nullable=False)
|
||||||
|
symbol: Mapped[str | None] = mapped_column(String(20), nullable=True)
|
||||||
|
dedup_key: Mapped[str | None] = mapped_column(String(200), nullable=True)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
|
)
|
||||||
|
acknowledged_at: Mapped[datetime | None] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=True
|
||||||
|
)
|
||||||
@@ -11,6 +11,16 @@ class Ticker(Base):
|
|||||||
|
|
||||||
id: Mapped[int] = mapped_column(primary_key=True)
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
symbol: Mapped[str] = mapped_column(String(10), unique=True, nullable=False)
|
symbol: Mapped[str] = mapped_column(String(10), unique=True, nullable=False)
|
||||||
|
# Company name (e.g. "Biogen Inc."); backfilled from Alpaca, nullable for
|
||||||
|
# symbols Alpaca doesn't know.
|
||||||
|
name: Mapped[str | None] = mapped_column(String(120), nullable=True)
|
||||||
|
# SEC issuer identity, refreshed by the SEC fundamentals import from
|
||||||
|
# company_tickers.json / submissions. The only ticker<->issuer join point;
|
||||||
|
# multi-class tickers (GOOG/GOOGL) share these values. Nullable: not every
|
||||||
|
# symbol resolves to a CIK (e.g. ADRs, foreign issuers not in SEC data).
|
||||||
|
cik: Mapped[str | None] = mapped_column(String(10), nullable=True)
|
||||||
|
sic: Mapped[str | None] = mapped_column(String(4), nullable=True)
|
||||||
|
sic_description: Mapped[str | None] = mapped_column(String(160), nullable=True)
|
||||||
created_at: Mapped[datetime] = mapped_column(
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
)
|
)
|
||||||
@@ -25,3 +35,4 @@ class Ticker(Base):
|
|||||||
trade_setups = relationship("TradeSetup", back_populates="ticker", cascade="all, delete-orphan")
|
trade_setups = relationship("TradeSetup", back_populates="ticker", cascade="all, delete-orphan")
|
||||||
watchlist_entries = relationship("WatchlistEntry", back_populates="ticker", cascade="all, delete-orphan")
|
watchlist_entries = relationship("WatchlistEntry", back_populates="ticker", cascade="all, delete-orphan")
|
||||||
ingestion_progress = relationship("IngestionProgress", back_populates="ticker", cascade="all, delete-orphan", uselist=False)
|
ingestion_progress = relationship("IngestionProgress", back_populates="ticker", cascade="all, delete-orphan", uselist=False)
|
||||||
|
earnings_events = relationship("EarningsEvent", back_populates="ticker", cascade="all, delete-orphan")
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ from datetime import date, datetime
|
|||||||
|
|
||||||
import json
|
import json
|
||||||
|
|
||||||
from sqlalchemy import Date, DateTime, Float, ForeignKey, String, Text
|
from sqlalchemy import Date, DateTime, Float, ForeignKey, Index, String, Text
|
||||||
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
from app.database import Base
|
from app.database import Base
|
||||||
@@ -10,6 +10,7 @@ from app.database import Base
|
|||||||
|
|
||||||
class TradeSetup(Base):
|
class TradeSetup(Base):
|
||||||
__tablename__ = "trade_setups"
|
__tablename__ = "trade_setups"
|
||||||
|
__table_args__ = (Index("ix_trade_setups_ticker_rr", "ticker_id", "rr_ratio"),)
|
||||||
|
|
||||||
id: Mapped[int] = mapped_column(primary_key=True)
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
ticker_id: Mapped[int] = mapped_column(
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
@@ -26,9 +27,14 @@ class TradeSetup(Base):
|
|||||||
)
|
)
|
||||||
|
|
||||||
confidence_score: Mapped[float | None] = mapped_column(Float, nullable=True)
|
confidence_score: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
# Ticker's 12-1 momentum percentile across the universe at detection time
|
# Ticker's activation momentum percentile across the universe at detection
|
||||||
# (0–100, 100 = strongest). Drives the activation gate's core selection.
|
# time. Since July 2026 this is residual 12-1 momentum when benchmark data is
|
||||||
|
# available, with raw 12-1 as a fallback.
|
||||||
momentum_percentile: Mapped[float | None] = mapped_column(Float, nullable=True)
|
momentum_percentile: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
# Production ordering score. July 2026 promotion: residual momentum remains
|
||||||
|
# the gate, while this rank blends residual momentum with realized volatility.
|
||||||
|
strategy_rank: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
volatility_percentile: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
targets_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
targets_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
conflict_flags_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
conflict_flags_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
recommended_action: Mapped[str | None] = mapped_column(String(20), nullable=True)
|
recommended_action: Mapped[str | None] = mapped_column(String(20), nullable=True)
|
||||||
@@ -39,6 +45,11 @@ class TradeSetup(Base):
|
|||||||
DateTime(timezone=True), nullable=True
|
DateTime(timezone=True), nullable=True
|
||||||
)
|
)
|
||||||
outcome_date: Mapped[date | None] = mapped_column(Date, nullable=True)
|
outcome_date: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
# Identity of the scan run that produced this row. The shadow book selects
|
||||||
|
# its batch by this id, not by a detected_at window, so a concurrent manual
|
||||||
|
# scan writing rows in the same time window is excluded by identity. Null on
|
||||||
|
# rows predating the column and on any non-scan creator.
|
||||||
|
scan_run_id: Mapped[str | None] = mapped_column(String(32), nullable=True)
|
||||||
|
|
||||||
ticker = relationship("Ticker", back_populates="trade_setups")
|
ticker = relationship("Ticker", back_populates="trade_setups")
|
||||||
|
|
||||||
|
|||||||
+31
-3
@@ -4,7 +4,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
import logging
|
import logging
|
||||||
from datetime import date
|
from datetime import date, datetime, time, timedelta, timezone
|
||||||
|
|
||||||
from alpaca.data.historical import StockHistoricalDataClient
|
from alpaca.data.historical import StockHistoricalDataClient
|
||||||
from alpaca.data.requests import StockBarsRequest
|
from alpaca.data.requests import StockBarsRequest
|
||||||
@@ -16,6 +16,11 @@ from app.providers.protocol import OHLCVData
|
|||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Free plans may not query data from the most recent ~15 minutes, and a window
|
||||||
|
# reaching into it fails the *entire* request — which would silently leave the
|
||||||
|
# near-close scan on yesterday's close. Margin over the documented boundary.
|
||||||
|
_RECENT_DATA_CUTOFF = timedelta(minutes=20)
|
||||||
|
|
||||||
|
|
||||||
class AlpacaOHLCVProvider:
|
class AlpacaOHLCVProvider:
|
||||||
"""Fetches daily OHLCV bars from Alpaca Markets Data API."""
|
"""Fetches daily OHLCV bars from Alpaca Markets Data API."""
|
||||||
@@ -25,6 +30,26 @@ class AlpacaOHLCVProvider:
|
|||||||
raise ProviderError("Alpaca API key and secret are required")
|
raise ProviderError("Alpaca API key and secret are required")
|
||||||
self._client = StockHistoricalDataClient(api_key, api_secret)
|
self._client = StockHistoricalDataClient(api_key, api_secret)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _resolve_window(start_date: date, end_date: date) -> tuple[datetime, datetime]:
|
||||||
|
"""Return the instants covering ``start_date``..``end_date`` inclusive.
|
||||||
|
|
||||||
|
Two boundaries have to be right or today's bar disappears:
|
||||||
|
|
||||||
|
* Daily bars are stamped at the session start in UTC (04:00Z under EDT),
|
||||||
|
so an ``end`` of midnight on ``end_date`` lands *before* that day's bar
|
||||||
|
and silently drops it — extend to the following midnight instead.
|
||||||
|
* The window must stay out of the delayed-data period, otherwise the
|
||||||
|
request is rejected outright with "subscription does not permit
|
||||||
|
querying recent SIP data". Clamping keeps today's in-progress bar
|
||||||
|
available, roughly 20 minutes behind live.
|
||||||
|
"""
|
||||||
|
start = datetime.combine(start_date, time.min, tzinfo=timezone.utc)
|
||||||
|
end = datetime.combine(
|
||||||
|
end_date + timedelta(days=1), time.min, tzinfo=timezone.utc
|
||||||
|
)
|
||||||
|
return start, min(end, datetime.now(timezone.utc) - _RECENT_DATA_CUTOFF)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _to_alpaca_symbol(symbol: str) -> str:
|
def _to_alpaca_symbol(symbol: str) -> str:
|
||||||
"""Convert internal symbol format (BRK-B) to Alpaca format (BRK.B)."""
|
"""Convert internal symbol format (BRK-B) to Alpaca format (BRK.B)."""
|
||||||
@@ -40,12 +65,15 @@ class AlpacaOHLCVProvider:
|
|||||||
) -> list[OHLCVData]:
|
) -> list[OHLCVData]:
|
||||||
"""Fetch daily OHLCV bars for *ticker* between *start_date* and *end_date*."""
|
"""Fetch daily OHLCV bars for *ticker* between *start_date* and *end_date*."""
|
||||||
alpaca_symbol = self._to_alpaca_symbol(ticker)
|
alpaca_symbol = self._to_alpaca_symbol(ticker)
|
||||||
|
start, end = self._resolve_window(start_date, end_date)
|
||||||
|
if end <= start:
|
||||||
|
return []
|
||||||
try:
|
try:
|
||||||
request = StockBarsRequest(
|
request = StockBarsRequest(
|
||||||
symbol_or_symbols=alpaca_symbol,
|
symbol_or_symbols=alpaca_symbol,
|
||||||
timeframe=TimeFrame.Day,
|
timeframe=TimeFrame.Day,
|
||||||
start=start_date,
|
start=start,
|
||||||
end=end_date,
|
end=end,
|
||||||
adjustment=Adjustment.SPLIT,
|
adjustment=Adjustment.SPLIT,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -101,7 +101,10 @@ class FinnhubFundamentalProvider:
|
|||||||
earnings_payload = earnings_resp.json() if earnings_resp.text else []
|
earnings_payload = earnings_resp.json() if earnings_resp.text else []
|
||||||
|
|
||||||
metrics = metric_payload.get("metric", {}) if isinstance(metric_payload, dict) else {}
|
metrics = metric_payload.get("metric", {}) if isinstance(metric_payload, dict) else {}
|
||||||
market_cap = _safe_float((profile_payload or {}).get("marketCapitalization"))
|
# Finnhub profile2 marketCapitalization is in millions of USD.
|
||||||
|
# Normalize to absolute dollars so cap bands / formatters match FMP & Alpha Vantage.
|
||||||
|
market_cap_millions = _safe_float((profile_payload or {}).get("marketCapitalization"))
|
||||||
|
market_cap = market_cap_millions * 1_000_000.0 if market_cap_millions is not None else None
|
||||||
pe_ratio = _safe_float(metrics.get("peTTM") or metrics.get("peNormalizedAnnual"))
|
pe_ratio = _safe_float(metrics.get("peTTM") or metrics.get("peNormalizedAnnual"))
|
||||||
revenue_growth = _safe_float(metrics.get("revenueGrowthTTMYoy") or metrics.get("revenueGrowth5Y"))
|
revenue_growth = _safe_float(metrics.get("revenueGrowthTTMYoy") or metrics.get("revenueGrowth5Y"))
|
||||||
|
|
||||||
|
|||||||
+166
-3
@@ -6,17 +6,21 @@ All endpoints require admin role.
|
|||||||
from fastapi import APIRouter, Depends, Query
|
from fastapi import APIRouter, Depends, Query
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from app.dependencies import get_db, require_admin
|
from app.dependencies import get_db, require_access, require_admin
|
||||||
from app.models.user import User
|
from app.models.user import User
|
||||||
from app.schemas.admin import (
|
from app.schemas.admin import (
|
||||||
ActivationConfigUpdate,
|
ActivationConfigUpdate,
|
||||||
AlertConfigUpdate,
|
AlertConfigUpdate,
|
||||||
CreateUserRequest,
|
CreateUserRequest,
|
||||||
DataCleanupRequest,
|
DataCleanupRequest,
|
||||||
|
FundamentalsCutoverConfigUpdate,
|
||||||
|
JobTriggerRequest,
|
||||||
JobToggle,
|
JobToggle,
|
||||||
RecommendationConfigUpdate,
|
RecommendationConfigUpdate,
|
||||||
|
PerformanceConfigUpdate,
|
||||||
ScheduleConfigUpdate,
|
ScheduleConfigUpdate,
|
||||||
SentimentConfigUpdate,
|
SentimentConfigUpdate,
|
||||||
|
ShadowBookConfigUpdate,
|
||||||
SentimentTestRequest,
|
SentimentTestRequest,
|
||||||
PasswordReset,
|
PasswordReset,
|
||||||
RegistrationToggle,
|
RegistrationToggle,
|
||||||
@@ -28,6 +32,7 @@ from app.schemas.common import APIEnvelope
|
|||||||
from app.services import admin_service
|
from app.services import admin_service
|
||||||
from app.services import alert_service
|
from app.services import alert_service
|
||||||
from app.services import sentiment_provider_service
|
from app.services import sentiment_provider_service
|
||||||
|
from app.services import system_event_service
|
||||||
from app.services import ticker_universe_service
|
from app.services import ticker_universe_service
|
||||||
|
|
||||||
router = APIRouter(tags=["admin"])
|
router = APIRouter(tags=["admin"])
|
||||||
@@ -133,6 +138,27 @@ async def list_settings(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/admin/settings/fundamentals-cutover", response_model=APIEnvelope)
|
||||||
|
async def get_fundamentals_cutover_settings(
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
config = await admin_service.get_fundamentals_cutover_config(db)
|
||||||
|
return APIEnvelope(status="success", data=config)
|
||||||
|
|
||||||
|
|
||||||
|
@router.put("/admin/settings/fundamentals-cutover", response_model=APIEnvelope)
|
||||||
|
async def update_fundamentals_cutover_settings(
|
||||||
|
body: FundamentalsCutoverConfigUpdate,
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
config = await admin_service.update_fundamentals_cutover_config(
|
||||||
|
db, body.enabled
|
||||||
|
)
|
||||||
|
return APIEnvelope(status="success", data=config)
|
||||||
|
|
||||||
|
|
||||||
@router.get("/admin/settings/recommendations", response_model=APIEnvelope)
|
@router.get("/admin/settings/recommendations", response_model=APIEnvelope)
|
||||||
async def get_recommendation_settings(
|
async def get_recommendation_settings(
|
||||||
_admin: User = Depends(require_admin),
|
_admin: User = Depends(require_admin),
|
||||||
@@ -199,6 +225,50 @@ async def update_schedule_settings(
|
|||||||
return APIEnvelope(status="success", data=updated)
|
return APIEnvelope(status="success", data=updated)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/admin/settings/performance", response_model=APIEnvelope)
|
||||||
|
async def get_performance_settings(
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
return APIEnvelope(
|
||||||
|
status="success", data=await admin_service.get_performance_config(db)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.put("/admin/settings/performance", response_model=APIEnvelope)
|
||||||
|
async def update_performance_settings(
|
||||||
|
body: PerformanceConfigUpdate,
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
updated = await admin_service.update_performance_config(
|
||||||
|
db, body.model_dump(exclude_unset=True)
|
||||||
|
)
|
||||||
|
return APIEnvelope(status="success", data=updated)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/admin/settings/shadow-book", response_model=APIEnvelope)
|
||||||
|
async def get_shadow_book_settings(
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
return APIEnvelope(
|
||||||
|
status="success", data=await admin_service.get_shadow_book_config(db)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.put("/admin/settings/shadow-book", response_model=APIEnvelope)
|
||||||
|
async def update_shadow_book_settings(
|
||||||
|
body: ShadowBookConfigUpdate,
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
updated = await admin_service.update_shadow_book_config(
|
||||||
|
db, body.model_dump(exclude_unset=True, exclude_none=True)
|
||||||
|
)
|
||||||
|
return APIEnvelope(status="success", data=updated)
|
||||||
|
|
||||||
|
|
||||||
@router.get("/admin/settings/sentiment", response_model=APIEnvelope)
|
@router.get("/admin/settings/sentiment", response_model=APIEnvelope)
|
||||||
async def get_sentiment_settings(
|
async def get_sentiment_settings(
|
||||||
_admin: User = Depends(require_admin),
|
_admin: User = Depends(require_admin),
|
||||||
@@ -315,6 +385,16 @@ async def bootstrap_tickers(
|
|||||||
return APIEnvelope(status="success", data=result)
|
return APIEnvelope(status="success", data=result)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/admin/tickers/backfill-names", response_model=APIEnvelope)
|
||||||
|
async def backfill_ticker_names(
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
"""Fill in company names for tracked tickers (one Alpaca call)."""
|
||||||
|
result = await ticker_universe_service.backfill_ticker_names(db)
|
||||||
|
return APIEnvelope(status="success", data=result)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Data cleanup
|
# Data cleanup
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -366,11 +446,17 @@ async def get_pipeline_readiness(
|
|||||||
@router.post("/admin/jobs/{job_name}/trigger", response_model=APIEnvelope)
|
@router.post("/admin/jobs/{job_name}/trigger", response_model=APIEnvelope)
|
||||||
async def trigger_job(
|
async def trigger_job(
|
||||||
job_name: str,
|
job_name: str,
|
||||||
|
body: JobTriggerRequest | None = None,
|
||||||
_admin: User = Depends(require_admin),
|
_admin: User = Depends(require_admin),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
):
|
):
|
||||||
"""Trigger a manual job run (placeholder)."""
|
"""Trigger a manual job run, optionally with one-run parameters."""
|
||||||
result = await admin_service.trigger_job(db, job_name)
|
result = await admin_service.trigger_job(
|
||||||
|
db,
|
||||||
|
job_name,
|
||||||
|
target_model=body.target_model if body is not None else None,
|
||||||
|
cadence=body.cadence if body is not None else None,
|
||||||
|
)
|
||||||
return APIEnvelope(status="success", data=result)
|
return APIEnvelope(status="success", data=result)
|
||||||
|
|
||||||
|
|
||||||
@@ -387,3 +473,80 @@ async def toggle_job(
|
|||||||
status="success",
|
status="success",
|
||||||
data={"key": setting.key, "value": setting.value},
|
data={"key": setting.key, "value": setting.value},
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/admin/fundamentals-parity", response_model=APIEnvelope)
|
||||||
|
async def get_fundamentals_parity_report(
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
):
|
||||||
|
"""Latest read-only A5 source/score comparison, or null before first run."""
|
||||||
|
return APIEnvelope(
|
||||||
|
status="success", data=admin_service.get_fundamentals_parity_report()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/admin/fundamentals-parity/csv", response_model=APIEnvelope)
|
||||||
|
async def get_fundamentals_parity_csv(
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
):
|
||||||
|
"""Latest flattened A5 report for an authenticated browser download."""
|
||||||
|
artifact = admin_service.get_fundamentals_parity_csv()
|
||||||
|
data = None if artifact is None else {"filename": artifact[0], "content": artifact[1]}
|
||||||
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/admin/fundamentals-parity/json", response_model=APIEnvelope)
|
||||||
|
async def get_fundamentals_parity_json(
|
||||||
|
_admin: User = Depends(require_admin),
|
||||||
|
):
|
||||||
|
"""Canonical A5 JSON artifact for an authenticated browser download."""
|
||||||
|
artifact = admin_service.get_fundamentals_parity_json()
|
||||||
|
data = None if artifact is None else {"filename": artifact[0], "content": artifact[1]}
|
||||||
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# System events (operational warnings / errors)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
@router.get("/admin/system-events", response_model=APIEnvelope)
|
||||||
|
async def list_system_events(
|
||||||
|
days: int = Query(7, ge=1, le=30),
|
||||||
|
severity: str | None = Query(None, description="warning | error"),
|
||||||
|
unacknowledged_only: bool = Query(False),
|
||||||
|
_user: User = Depends(require_access),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
"""List recent system events (default last 7 days)."""
|
||||||
|
rows = await system_event_service.list_events(
|
||||||
|
db,
|
||||||
|
days=days,
|
||||||
|
severity=severity,
|
||||||
|
unacknowledged_only=unacknowledged_only,
|
||||||
|
)
|
||||||
|
return APIEnvelope(
|
||||||
|
status="success",
|
||||||
|
data=[system_event_service.event_to_dict(r) for r in rows],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/admin/system-events/summary", response_model=APIEnvelope)
|
||||||
|
async def system_events_summary(
|
||||||
|
days: int = Query(7, ge=1, le=30),
|
||||||
|
_user: User = Depends(require_access),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
"""Badge counts for the top nav."""
|
||||||
|
data = await system_event_service.summary(db, days=days)
|
||||||
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/admin/system-events/acknowledge", response_model=APIEnvelope)
|
||||||
|
async def acknowledge_system_events(
|
||||||
|
days: int = Query(7, ge=1, le=30),
|
||||||
|
_user: User = Depends(require_access),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
):
|
||||||
|
"""Dismiss unacknowledged events in the lookback window (clears the badge)."""
|
||||||
|
count = await system_event_service.acknowledge_all(db, days=days)
|
||||||
|
return APIEnvelope(status="success", data={"acknowledged": count})
|
||||||
|
|||||||
@@ -9,6 +9,8 @@ from app.dependencies import get_db, require_access
|
|||||||
from app.schemas.common import APIEnvelope
|
from app.schemas.common import APIEnvelope
|
||||||
from app.schemas.fundamental import FundamentalResponse
|
from app.schemas.fundamental import FundamentalResponse
|
||||||
from app.services.fundamental_service import get_fundamental
|
from app.services.fundamental_service import get_fundamental
|
||||||
|
from app.services.fundamentals_api_service import build_fundamentals_v1
|
||||||
|
from app.services import fundamentals_quality_service
|
||||||
|
|
||||||
router = APIRouter(tags=["fundamentals"])
|
router = APIRouter(tags=["fundamentals"])
|
||||||
|
|
||||||
@@ -30,14 +32,14 @@ async def read_fundamentals(
|
|||||||
_user=Depends(require_access),
|
_user=Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Get latest fundamental data for a symbol."""
|
"""Get latest fundamental data for a symbol (legacy fields + additive v1)."""
|
||||||
record = await get_fundamental(db, symbol)
|
record = await get_fundamental(db, symbol)
|
||||||
|
v1 = await build_fundamentals_v1(db, symbol)
|
||||||
|
quality = await fundamentals_quality_service.ticker_quality(db, symbol)
|
||||||
|
|
||||||
if record is None:
|
legacy: dict = {}
|
||||||
data = FundamentalResponse(symbol=symbol.strip().upper())
|
if record is not None:
|
||||||
else:
|
legacy = dict(
|
||||||
data = FundamentalResponse(
|
|
||||||
symbol=symbol.strip().upper(),
|
|
||||||
pe_ratio=record.pe_ratio,
|
pe_ratio=record.pe_ratio,
|
||||||
revenue_growth=record.revenue_growth,
|
revenue_growth=record.revenue_growth,
|
||||||
earnings_surprise=record.earnings_surprise,
|
earnings_surprise=record.earnings_surprise,
|
||||||
@@ -47,4 +49,12 @@ async def read_fundamentals(
|
|||||||
unavailable_fields=_parse_unavailable_fields(record.unavailable_fields_json),
|
unavailable_fields=_parse_unavailable_fields(record.unavailable_fields_json),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
data = FundamentalResponse(
|
||||||
|
symbol=symbol.strip().upper(),
|
||||||
|
setup_eligible=quality.eligible,
|
||||||
|
setup_block_code=quality.code,
|
||||||
|
setup_block_reason=quality.message,
|
||||||
|
**legacy,
|
||||||
|
**v1,
|
||||||
|
)
|
||||||
return APIEnvelope(status="success", data=data.model_dump())
|
return APIEnvelope(status="success", data=data.model_dump())
|
||||||
|
|||||||
@@ -19,11 +19,15 @@ from app.dependencies import get_db, require_access
|
|||||||
from app.exceptions import ProviderError
|
from app.exceptions import ProviderError
|
||||||
from app.models.ohlcv import OHLCVRecord
|
from app.models.ohlcv import OHLCVRecord
|
||||||
from app.models.settings import IngestionProgress
|
from app.models.settings import IngestionProgress
|
||||||
|
from app.models.sr_level import SRLevel
|
||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
from app.models.user import User
|
from app.models.user import User
|
||||||
from app.providers.alpaca import AlpacaOHLCVProvider
|
from app.providers.alpaca import AlpacaOHLCVProvider
|
||||||
from app.providers.fundamentals_chain import build_fundamental_provider_chain
|
from app.providers.fundamentals_chain import build_fundamental_provider_chain
|
||||||
from app.services.rr_scanner_service import scan_ticker
|
from app.services.rr_scanner_service import (
|
||||||
|
resolve_activation_ranks_for_symbol,
|
||||||
|
scan_ticker,
|
||||||
|
)
|
||||||
from app.services.sentiment_provider_service import build_sentiment_provider
|
from app.services.sentiment_provider_service import build_sentiment_provider
|
||||||
from app.schemas.common import APIEnvelope
|
from app.schemas.common import APIEnvelope
|
||||||
from app.services import (
|
from app.services import (
|
||||||
@@ -102,8 +106,13 @@ async def fetch_symbol(
|
|||||||
await db.execute(
|
await db.execute(
|
||||||
delete(IngestionProgress).where(IngestionProgress.ticker_id == ticker_obj.id)
|
delete(IngestionProgress).where(IngestionProgress.ticker_id == ticker_obj.id)
|
||||||
)
|
)
|
||||||
|
# Drop Structural S/R with the bars; a failed re-fetch must not
|
||||||
|
# leave zones computed from deleted history.
|
||||||
|
await db.execute(
|
||||||
|
delete(SRLevel).where(SRLevel.ticker_id == ticker_obj.id)
|
||||||
|
)
|
||||||
await db.commit()
|
await db.commit()
|
||||||
logger.info("force_refetch: cleared OHLCV and progress for %s", symbol_upper)
|
logger.info("force_refetch: cleared OHLCV, S/R, and progress for %s", symbol_upper)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
logger.error("force_refetch cleanup failed for %s: %s", symbol_upper, exc)
|
logger.error("force_refetch cleanup failed for %s: %s", symbol_upper, exc)
|
||||||
|
|
||||||
@@ -114,11 +123,32 @@ async def fetch_symbol(
|
|||||||
result = await ingestion_service.fetch_and_ingest(
|
result = await ingestion_service.fetch_and_ingest(
|
||||||
db, provider, symbol_upper, start_date, end_date
|
db, provider, symbol_upper, start_date, end_date
|
||||||
)
|
)
|
||||||
|
# "stale" = provider returned nothing but our last bar is old
|
||||||
|
# (rename/delist/halt) — must not look like a successful refresh.
|
||||||
|
status_map = {
|
||||||
|
"complete": "ok",
|
||||||
|
"partial": "ok",
|
||||||
|
"no_data": "warning",
|
||||||
|
"stale": "warning",
|
||||||
|
}
|
||||||
sources_out["ohlcv"] = {
|
sources_out["ohlcv"] = {
|
||||||
"status": "ok" if result.status in ("complete", "partial") else "error",
|
"status": status_map.get(result.status, "error"),
|
||||||
"records": result.records_ingested,
|
"records": result.records_ingested,
|
||||||
"message": result.message,
|
"message": result.message,
|
||||||
|
"last_date": result.last_date.isoformat() if result.last_date else None,
|
||||||
}
|
}
|
||||||
|
if result.status in ("stale", "no_data", "error"):
|
||||||
|
from app.services.system_event_service import log_event
|
||||||
|
|
||||||
|
await log_event(
|
||||||
|
db,
|
||||||
|
severity="warning" if result.status != "error" else "error",
|
||||||
|
source="ingestion",
|
||||||
|
code=f"ohlcv_{result.status}",
|
||||||
|
message=result.message or f"OHLCV fetch {result.status} for {symbol_upper}",
|
||||||
|
symbol=symbol_upper,
|
||||||
|
dedup_key=f"ohlcv_{result.status}:{symbol_upper}",
|
||||||
|
)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
logger.error("OHLCV fetch failed for %s: %s", symbol_upper, exc)
|
logger.error("OHLCV fetch failed for %s: %s", symbol_upper, exc)
|
||||||
sources_out["ohlcv"] = {"status": "error", "records": 0, "message": str(exc)}
|
sources_out["ohlcv"] = {"status": "error", "records": 0, "message": str(exc)}
|
||||||
@@ -215,15 +245,23 @@ async def fetch_symbol(
|
|||||||
sources_out["scores"] = {"status": "error", "message": str(exc)}
|
sources_out["scores"] = {"status": "error", "message": str(exc)}
|
||||||
|
|
||||||
# --- Derived pipeline: scanner (free, always) ---
|
# --- Derived pipeline: scanner (free, always) ---
|
||||||
|
# Attach the same residual-momentum / strategy ranks the daily scan writes.
|
||||||
|
# Without them the new setup lands with null momentum_percentile and fails
|
||||||
|
# the activation gate (missing ranks do not qualify).
|
||||||
try:
|
try:
|
||||||
|
ranks = await resolve_activation_ranks_for_symbol(db, symbol_upper)
|
||||||
setups = await scan_ticker(
|
setups = await scan_ticker(
|
||||||
db,
|
db,
|
||||||
symbol_upper,
|
symbol_upper,
|
||||||
rr_threshold=settings.default_rr_threshold,
|
rr_threshold=settings.default_rr_threshold,
|
||||||
|
momentum_percentile=ranks.get("momentum_percentile"),
|
||||||
|
strategy_rank=ranks.get("strategy_rank"),
|
||||||
|
volatility_percentile=ranks.get("volatility_percentile"),
|
||||||
)
|
)
|
||||||
sources_out["scanner"] = {
|
sources_out["scanner"] = {
|
||||||
"status": "ok",
|
"status": "ok",
|
||||||
"setups_found": len(setups),
|
"setups_found": len(setups),
|
||||||
|
"momentum_percentile": ranks.get("momentum_percentile"),
|
||||||
"message": None,
|
"message": None,
|
||||||
}
|
}
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
|
|||||||
+31
-17
@@ -1,7 +1,9 @@
|
|||||||
"""Market-level endpoints (benchmark regime + AI/Tech regime-change monitor)."""
|
"""Market-level endpoints (benchmark regime + AI/Tech regime-change monitor)."""
|
||||||
|
|
||||||
|
from typing import Literal
|
||||||
|
|
||||||
from fastapi import APIRouter, Depends, Query
|
from fastapi import APIRouter, Depends, Query
|
||||||
from pydantic import BaseModel
|
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from app.dependencies import get_db, require_access, require_admin
|
from app.dependencies import get_db, require_access, require_admin
|
||||||
@@ -40,17 +42,25 @@ async def backtest_report(
|
|||||||
|
|
||||||
|
|
||||||
class RegimeConfigUpdate(BaseModel):
|
class RegimeConfigUpdate(BaseModel):
|
||||||
weights: dict[str, float] | None = None
|
breadth_basket: list[str] | None = Field(default=None, min_length=20, max_length=100)
|
||||||
alert_threshold: float | None = None
|
fundamental_staleness_days: int | None = Field(default=None, ge=30, le=180)
|
||||||
tickers: dict | None = None
|
|
||||||
leader_weight: float | None = None
|
@field_validator("breadth_basket")
|
||||||
rs_lookback: int | None = None
|
@classmethod
|
||||||
fundamental_staleness_days: int | None = None
|
def normalise_basket(cls, value: list[str] | None) -> list[str] | None:
|
||||||
|
if value is None:
|
||||||
|
return None
|
||||||
|
cleaned = [symbol.strip().upper().replace(".", "-") for symbol in value if symbol.strip()]
|
||||||
|
if len(cleaned) != len(set(cleaned)):
|
||||||
|
raise ValueError("breadth basket symbols must be unique")
|
||||||
|
return cleaned
|
||||||
|
|
||||||
|
|
||||||
class RegimeFundamentalsUpdate(BaseModel):
|
class RegimeFundamentalsUpdate(BaseModel):
|
||||||
f1_score: float | None = None
|
model_config = ConfigDict(extra="forbid")
|
||||||
f3_score: float | None = None
|
|
||||||
|
capex: dict[str, Literal["raising", "holding", "cutting", "unknown"]] | None = None
|
||||||
|
good_news_stock_down: Literal["yes", "no", "mixed"] | None = None
|
||||||
locked: bool | None = None
|
locked: bool | None = None
|
||||||
|
|
||||||
|
|
||||||
@@ -59,7 +69,7 @@ async def regime_monitor(
|
|||||||
_user: User = Depends(require_access),
|
_user: User = Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Latest AI/Tech regime-change index (0-100) + per-signal breakdown + trend."""
|
"""Latest v2 State and Warning risk-thermometer readings."""
|
||||||
data = await regime_monitor_service.get_regime_monitor(db)
|
data = await regime_monitor_service.get_regime_monitor(db)
|
||||||
return APIEnvelope(status="success", data=data)
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
@@ -69,7 +79,7 @@ async def regime_config(
|
|||||||
_admin: User = Depends(require_admin),
|
_admin: User = Depends(require_admin),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Editable weights / thresholds / ticker lists for the regime monitor."""
|
"""Editable fixed breadth basket and fundamental freshness window."""
|
||||||
data = await regime_monitor_service.get_regime_config(db)
|
data = await regime_monitor_service.get_regime_config(db)
|
||||||
return APIEnvelope(status="success", data=data)
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
@@ -80,7 +90,7 @@ async def update_regime_config(
|
|||||||
_admin: User = Depends(require_admin),
|
_admin: User = Depends(require_admin),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Merge the supplied fields into the stored regime-monitor config."""
|
"""Update the deliberately small v2 operator configuration."""
|
||||||
updates = body.model_dump(exclude_none=True)
|
updates = body.model_dump(exclude_none=True)
|
||||||
data = await regime_monitor_service.update_regime_config(db, updates)
|
data = await regime_monitor_service.update_regime_config(db, updates)
|
||||||
return APIEnvelope(status="success", data=data)
|
return APIEnvelope(status="success", data=data)
|
||||||
@@ -102,9 +112,12 @@ async def update_regime_fundamentals(
|
|||||||
_admin: User = Depends(require_admin),
|
_admin: User = Depends(require_admin),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Manually override F1/F3 (locks out the LLM refresh until unlocked)."""
|
"""Manually override categorical F1/F3 observations."""
|
||||||
data = await regime_monitor_service.set_fundamental_overrides(
|
data = await regime_monitor_service.set_fundamental_overrides(
|
||||||
db, f1_score=body.f1_score, f3_score=body.f3_score, locked=body.locked
|
db,
|
||||||
|
capex=body.capex,
|
||||||
|
good_news_stock_down=body.good_news_stock_down,
|
||||||
|
locked=body.locked,
|
||||||
)
|
)
|
||||||
return APIEnvelope(status="success", data=data)
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
@@ -114,8 +127,9 @@ async def refresh_regime_fundamentals(
|
|||||||
_admin: User = Depends(require_admin),
|
_admin: User = Depends(require_admin),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Ask the configured LLM to re-estimate F1/F3 now (forces past a lock)."""
|
"""Refresh F1/F3 via LLM, then recompute the latest eligible snapshot."""
|
||||||
data = await regime_monitor_service.refresh_fundamental_overrides(db, force=True)
|
data = await regime_monitor_service.refresh_fundamental_overrides(db, force=True)
|
||||||
|
await regime_monitor_service.update_regime_monitor(db)
|
||||||
return APIEnvelope(status="success", data=data)
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
|
|
||||||
@@ -133,10 +147,10 @@ async def regime_event_study(
|
|||||||
|
|
||||||
@router.get("/regime/history", response_model=APIEnvelope)
|
@router.get("/regime/history", response_model=APIEnvelope)
|
||||||
async def regime_history(
|
async def regime_history(
|
||||||
days: int = Query(default=400, ge=7, le=2000),
|
days: int = Query(default=800, ge=7, le=2000),
|
||||||
_user: User = Depends(require_access),
|
_user: User = Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Daily history of the index / early-warning / combined scores (for the chart)."""
|
"""Point-in-time v2 State/Warning history. Legacy rows are excluded."""
|
||||||
data = await regime_monitor_service.get_regime_history(db, days=days)
|
data = await regime_monitor_service.get_regime_history(db, days=days)
|
||||||
return APIEnvelope(status="success", data=data)
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|||||||
@@ -3,10 +3,15 @@
|
|||||||
from fastapi import APIRouter, Depends, Query
|
from fastapi import APIRouter, Depends, Query
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from app.dependencies import get_db, require_access
|
from app.dependencies import get_db, require_access, require_admin
|
||||||
from app.models.user import User
|
from app.models.user import User
|
||||||
from app.schemas.common import APIEnvelope
|
from app.schemas.common import APIEnvelope
|
||||||
from app.schemas.paper_trade import PaperTradeClose, PaperTradeCreate, PaperTradeResponse
|
from app.schemas.paper_trade import (
|
||||||
|
ExitPolicyUpdate,
|
||||||
|
PaperTradeClose,
|
||||||
|
PaperTradeCreate,
|
||||||
|
PaperTradeResponse,
|
||||||
|
)
|
||||||
from app.services import paper_trade_service
|
from app.services import paper_trade_service
|
||||||
|
|
||||||
router = APIRouter(tags=["paper-trades"])
|
router = APIRouter(tags=["paper-trades"])
|
||||||
@@ -40,6 +45,55 @@ async def list_paper_trades(
|
|||||||
return APIEnvelope(status="success", data=data)
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/paper-trades/exit-policy", response_model=APIEnvelope)
|
||||||
|
async def read_exit_policy(
|
||||||
|
_user: User = Depends(require_access),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
) -> APIEnvelope:
|
||||||
|
"""The active auto-exit policy for open paper trades (shown in the UI)."""
|
||||||
|
return APIEnvelope(status="success", data=await paper_trade_service.get_exit_policy(db))
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/paper-trades/equity-curve", response_model=APIEnvelope)
|
||||||
|
async def paper_trade_equity_curve(
|
||||||
|
user: User = Depends(require_access),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
) -> APIEnvelope:
|
||||||
|
"""Daily cumulative P&L of the paper book vs the same dollars riding SPY."""
|
||||||
|
return APIEnvelope(
|
||||||
|
status="success", data=await paper_trade_service.equity_curve(db, user.id)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/paper-trades/performance", response_model=APIEnvelope)
|
||||||
|
async def paper_trade_performance(
|
||||||
|
user: User = Depends(require_access),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
) -> APIEnvelope:
|
||||||
|
"""Shadow book vs discretionary book vs SPY since the configured start date."""
|
||||||
|
return APIEnvelope(
|
||||||
|
status="success",
|
||||||
|
data=await paper_trade_service.performance_summary(db, user.id),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@router.put("/paper-trades/exit-policy", response_model=APIEnvelope)
|
||||||
|
async def write_exit_policy(
|
||||||
|
body: ExitPolicyUpdate,
|
||||||
|
_user: User = Depends(require_admin),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
) -> APIEnvelope:
|
||||||
|
"""Change the auto-exit policy (admin)."""
|
||||||
|
data = await paper_trade_service.set_exit_policy(
|
||||||
|
db,
|
||||||
|
mode=body.mode,
|
||||||
|
trailing_pct=body.trailing_pct,
|
||||||
|
atr_multiplier=body.atr_multiplier,
|
||||||
|
hold_days=body.hold_days,
|
||||||
|
)
|
||||||
|
return APIEnvelope(status="success", data=data)
|
||||||
|
|
||||||
|
|
||||||
@router.post("/paper-trades", response_model=APIEnvelope, status_code=201)
|
@router.post("/paper-trades", response_model=APIEnvelope, status_code=201)
|
||||||
async def create_paper_trade(
|
async def create_paper_trade(
|
||||||
body: PaperTradeCreate,
|
body: PaperTradeCreate,
|
||||||
|
|||||||
@@ -39,6 +39,10 @@ def _map_composite_breakdown(raw: dict | None) -> CompositeBreakdownResponse | N
|
|||||||
missing_dimensions=raw["missing_dimensions"],
|
missing_dimensions=raw["missing_dimensions"],
|
||||||
renormalized_weights=raw["renormalized_weights"],
|
renormalized_weights=raw["renormalized_weights"],
|
||||||
formula=raw["formula"],
|
formula=raw["formula"],
|
||||||
|
base_score=raw.get("base_score"),
|
||||||
|
sentiment_score=raw.get("sentiment_score"),
|
||||||
|
sentiment_adjustment=raw.get("sentiment_adjustment"),
|
||||||
|
max_sentiment_adjustment=raw.get("max_sentiment_adjustment"),
|
||||||
)
|
)
|
||||||
|
|
||||||
router = APIRouter(tags=["scores"])
|
router = APIRouter(tags=["scores"])
|
||||||
@@ -50,7 +54,7 @@ async def read_score(
|
|||||||
_user=Depends(require_access),
|
_user=Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Get composite + dimension scores for a symbol. Recomputes stale scores."""
|
"""Get the latest persisted composite + dimension scores for a symbol."""
|
||||||
result = await get_score(db, symbol)
|
result = await get_score(db, symbol)
|
||||||
|
|
||||||
data = ScoreResponse(
|
data = ScoreResponse(
|
||||||
@@ -90,6 +94,7 @@ async def read_rankings(
|
|||||||
RankingEntry(
|
RankingEntry(
|
||||||
symbol=r["symbol"],
|
symbol=r["symbol"],
|
||||||
composite_score=r["composite_score"],
|
composite_score=r["composite_score"],
|
||||||
|
composite_stale=r.get("composite_stale", False),
|
||||||
dimensions=[
|
dimensions=[
|
||||||
DimensionScoreResponse(**d) for d in r["dimensions"]
|
DimensionScoreResponse(**d) for d in r["dimensions"]
|
||||||
],
|
],
|
||||||
|
|||||||
@@ -5,17 +5,81 @@ from sqlalchemy.ext.asyncio import AsyncSession
|
|||||||
|
|
||||||
from app.dependencies import get_db, require_access
|
from app.dependencies import get_db, require_access
|
||||||
from app.schemas.common import APIEnvelope
|
from app.schemas.common import APIEnvelope
|
||||||
from app.schemas.sr_level import SRLevelResponse, SRLevelResult, SRZoneResult
|
from app.schemas.sr_level import (
|
||||||
|
GateTargetLadderResponse,
|
||||||
|
GateTargetLevelResult,
|
||||||
|
SRLevelResponse,
|
||||||
|
SRLevelResult,
|
||||||
|
SRZoneResult,
|
||||||
|
)
|
||||||
from app.services.price_service import query_ohlcv
|
from app.services.price_service import query_ohlcv
|
||||||
from app.services.sr_service import cluster_sr_zones, get_sr_levels
|
from app.services.sr_service import (
|
||||||
|
cluster_sr_zones,
|
||||||
|
detect_gate_target_ladder,
|
||||||
|
get_sr_levels,
|
||||||
|
)
|
||||||
|
|
||||||
router = APIRouter(tags=["sr-levels"])
|
router = APIRouter(tags=["sr-levels"])
|
||||||
|
|
||||||
|
|
||||||
|
@router.get("/gate-target-ladder/{symbol}", response_model=APIEnvelope)
|
||||||
|
async def read_gate_target_ladder(
|
||||||
|
symbol: str,
|
||||||
|
_user=Depends(require_access),
|
||||||
|
db: AsyncSession = Depends(get_db),
|
||||||
|
) -> APIEnvelope:
|
||||||
|
"""Return the transient, volume-free GTL for chart diagnostics.
|
||||||
|
|
||||||
|
These proposals are not persisted ``SRLevel`` rows and must not be
|
||||||
|
presented as structural support/resistance.
|
||||||
|
"""
|
||||||
|
records = await query_ohlcv(db, symbol)
|
||||||
|
if not records:
|
||||||
|
data = GateTargetLadderResponse(
|
||||||
|
symbol=symbol.upper(),
|
||||||
|
levels=[],
|
||||||
|
count=0,
|
||||||
|
lookback_bars=0,
|
||||||
|
)
|
||||||
|
return APIEnvelope(status="success", data=data.model_dump())
|
||||||
|
|
||||||
|
highs = [float(record.high) for record in records]
|
||||||
|
lows = [float(record.low) for record in records]
|
||||||
|
closes = [float(record.close) for record in records]
|
||||||
|
detected = detect_gate_target_ladder(highs, lows, closes)
|
||||||
|
levels = [
|
||||||
|
GateTargetLevelResult(
|
||||||
|
price_level=float(level["price_level"]),
|
||||||
|
type=level["type"],
|
||||||
|
strength=int(level["strength"]),
|
||||||
|
detection_method=str(level.get("detection_method", "unknown")),
|
||||||
|
sources=list(level.get("sources") or []),
|
||||||
|
traffic_count=int(level.get("rejection_count", 0) or 0),
|
||||||
|
)
|
||||||
|
for level in sorted(detected, key=lambda row: float(row["price_level"]))
|
||||||
|
]
|
||||||
|
data = GateTargetLadderResponse(
|
||||||
|
symbol=symbol.upper(),
|
||||||
|
levels=levels,
|
||||||
|
count=len(levels),
|
||||||
|
lookback_bars=len(records),
|
||||||
|
)
|
||||||
|
return APIEnvelope(status="success", data=data.model_dump())
|
||||||
|
|
||||||
|
|
||||||
@router.get("/sr-levels/{symbol}", response_model=APIEnvelope)
|
@router.get("/sr-levels/{symbol}", response_model=APIEnvelope)
|
||||||
async def read_sr_levels(
|
async def read_sr_levels(
|
||||||
symbol: str,
|
symbol: str,
|
||||||
tolerance: float = Query(0.005, ge=0, le=0.1, description="Merge tolerance (default 0.5%)"),
|
tolerance: float | None = Query(
|
||||||
|
None,
|
||||||
|
ge=0,
|
||||||
|
le=0.1,
|
||||||
|
description=(
|
||||||
|
"Merge tolerance as fraction of price. Omit to return persisted levels "
|
||||||
|
"(ATR-adaptive at last recalculation). When set, returns a transient "
|
||||||
|
"detect with this tolerance (not written to the DB)."
|
||||||
|
),
|
||||||
|
),
|
||||||
max_zones: int = Query(6, ge=0, description="Max S/R zones to return (default 6)"),
|
max_zones: int = Query(6, ge=0, description="Max S/R zones to return (default 6)"),
|
||||||
_user=Depends(require_access),
|
_user=Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
|
|||||||
+17
-8
@@ -25,7 +25,7 @@ async def list_trade_setups(
|
|||||||
None,
|
None,
|
||||||
description="Filter by action: LONG_HIGH, LONG_MODERATE, SHORT_HIGH, SHORT_MODERATE, NEUTRAL",
|
description="Filter by action: LONG_HIGH, LONG_MODERATE, SHORT_HIGH, SHORT_MODERATE, NEUTRAL",
|
||||||
),
|
),
|
||||||
_user=Depends(require_access),
|
user=Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Get latest trade setups with recommendation data."""
|
"""Get latest trade setups with recommendation data."""
|
||||||
@@ -34,6 +34,10 @@ async def list_trade_setups(
|
|||||||
direction=direction,
|
direction=direction,
|
||||||
min_confidence=min_confidence,
|
min_confidence=min_confidence,
|
||||||
recommended_action=recommended_action,
|
recommended_action=recommended_action,
|
||||||
|
live_recommendation=True,
|
||||||
|
exclude_open_trade_tickers=True,
|
||||||
|
exclude_open_trade_user_id=user.id,
|
||||||
|
exclude_reentry_gate_locked_tickers=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
data = []
|
data = []
|
||||||
@@ -73,13 +77,13 @@ async def get_trade_performance(
|
|||||||
_user=Depends(require_access),
|
_user=Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Aggregate outcome statistics over evaluated trade setups.
|
"""Aggregate setup-outcome statistics (gate barrier diagnostic).
|
||||||
|
|
||||||
Outcomes are written by the nightly outcome_evaluator job (win = target
|
Outcomes come from the nightly outcome_evaluator: win = gate target first,
|
||||||
hit first, loss = stop hit first, expired = neither within the window).
|
loss = stop first, expired = neither in the window. This is **not** the
|
||||||
With qualified_only, the overall/direction/action breakdowns cover only
|
production ATR-trail book; it checks setup grading plumbing only.
|
||||||
setups clearing the activation gate; the confidence breakdown always
|
With qualified_only, overall/direction/action cover only gate-clearing
|
||||||
covers all setups so the gate can be validated against it.
|
setups; the confidence breakdown always covers all setups.
|
||||||
"""
|
"""
|
||||||
config = await admin_service.get_activation_config(db) if qualified_only else None
|
config = await admin_service.get_activation_config(db) if qualified_only else None
|
||||||
stats = await get_performance_stats(db, config=config)
|
stats = await get_performance_stats(db, config=config)
|
||||||
@@ -92,7 +96,12 @@ async def get_ticker_trade_setups(
|
|||||||
_user=Depends(require_access),
|
_user=Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
rows = await get_trade_setups(db, symbol=symbol)
|
rows = await get_trade_setups(
|
||||||
|
db,
|
||||||
|
symbol=symbol,
|
||||||
|
live_recommendation=True,
|
||||||
|
include_reentry_gate_lock=True,
|
||||||
|
)
|
||||||
data = []
|
data = []
|
||||||
for row in rows:
|
for row in rows:
|
||||||
summary = RecommendationSummaryResponse(
|
summary = RecommendationSummaryResponse(
|
||||||
|
|||||||
+698
-42
@@ -33,9 +33,34 @@ from app.exceptions import ProviderError
|
|||||||
from app.providers.alpaca import AlpacaOHLCVProvider
|
from app.providers.alpaca import AlpacaOHLCVProvider
|
||||||
from app.providers.fundamentals_chain import build_fundamental_provider_chain
|
from app.providers.fundamentals_chain import build_fundamental_provider_chain
|
||||||
from app.providers.protocol import SentimentData
|
from app.providers.protocol import SentimentData
|
||||||
from app.services import fundamental_service, ingestion_service, sentiment_service, settings_store
|
from app.services import (
|
||||||
|
fundamental_service,
|
||||||
|
ingestion_service,
|
||||||
|
pipeline_run,
|
||||||
|
sentiment_service,
|
||||||
|
settings_store,
|
||||||
|
shadow_book_service,
|
||||||
|
fundamentals_parity_service,
|
||||||
|
fundamental_data_refresh_service,
|
||||||
|
)
|
||||||
|
from app.services.data_import import (
|
||||||
|
STATUS_DEFERRED,
|
||||||
|
STATUS_FAILED,
|
||||||
|
SourceImporter,
|
||||||
|
run_import,
|
||||||
|
)
|
||||||
|
from app.services.dolt_earnings_importer import DoltEarningsImporter
|
||||||
|
from app.services.sec_fundamentals_importer import SecFundamentalsImporter
|
||||||
from app.services.alert_service import dispatch_alerts
|
from app.services.alert_service import dispatch_alerts
|
||||||
from app.services.backtest_service import run_and_store as run_backtest_and_store
|
from app.services.backtest_service import (
|
||||||
|
BACKTEST_TARGET_MODELS,
|
||||||
|
DEFAULT_BACKTEST_CADENCE,
|
||||||
|
PRODUCTION_GTL_TARGET_MODEL,
|
||||||
|
run_and_store as run_backtest_and_store,
|
||||||
|
validate_backtest_cadence,
|
||||||
|
validate_backtest_target_model,
|
||||||
|
)
|
||||||
|
from app.services.benchmark_service import refresh_benchmark_prices
|
||||||
from app.services.market_regime_service import update_market_regime
|
from app.services.market_regime_service import update_market_regime
|
||||||
from app.services.regime_monitor_service import update_regime_monitor
|
from app.services.regime_monitor_service import update_regime_monitor
|
||||||
from app.services.event_study_service import run_and_store as run_event_study_and_store
|
from app.services.event_study_service import run_and_store as run_event_study_and_store
|
||||||
@@ -78,6 +103,9 @@ _JOB_NAMES = [
|
|||||||
"data_backfill",
|
"data_backfill",
|
||||||
"sentiment_collector",
|
"sentiment_collector",
|
||||||
"fundamental_collector",
|
"fundamental_collector",
|
||||||
|
"dolt_earnings_import",
|
||||||
|
"sec_fundamentals_import",
|
||||||
|
"fundamentals_parity_report",
|
||||||
"rr_scanner",
|
"rr_scanner",
|
||||||
"ticker_universe_sync",
|
"ticker_universe_sync",
|
||||||
"alerts",
|
"alerts",
|
||||||
@@ -85,7 +113,9 @@ _JOB_NAMES = [
|
|||||||
"regime_monitor",
|
"regime_monitor",
|
||||||
"event_study",
|
"event_study",
|
||||||
"backtest",
|
"backtest",
|
||||||
"daily_pipeline",
|
"daily_pipeline", # morning: OHLCV/sentiment/regime — no qualifying scan
|
||||||
|
"near_close_pipeline", # OHLCV fetch → R:R scan → Telegram alerts
|
||||||
|
"after_close_pipeline", # OHLCV fetch → outcome eval (final bar)
|
||||||
"intraday_pipeline",
|
"intraday_pipeline",
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -105,6 +135,8 @@ def _idle_runtime() -> dict[str, object]:
|
|||||||
|
|
||||||
|
|
||||||
_job_runtime: dict[str, dict[str, object]] = {name: _idle_runtime() for name in _JOB_NAMES}
|
_job_runtime: dict[str, dict[str, object]] = {name: _idle_runtime() for name in _JOB_NAMES}
|
||||||
|
_next_backtest_target_model = PRODUCTION_GTL_TARGET_MODEL
|
||||||
|
_next_backtest_cadence = DEFAULT_BACKTEST_CADENCE
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -112,6 +144,47 @@ _job_runtime: dict[str, dict[str, object]] = {name: _idle_runtime() for name in
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def queue_backtest_options(
|
||||||
|
target_model: str | None,
|
||||||
|
cadence: str | None,
|
||||||
|
) -> tuple[str, str]:
|
||||||
|
"""Select model and cadence for the next manual backtest run only.
|
||||||
|
|
||||||
|
Scheduled and subsequent manual runs return to production GTL at the
|
||||||
|
resource-safe weekly cadence.
|
||||||
|
"""
|
||||||
|
global _next_backtest_target_model, _next_backtest_cadence
|
||||||
|
selected_model = validate_backtest_target_model(
|
||||||
|
target_model or PRODUCTION_GTL_TARGET_MODEL
|
||||||
|
)
|
||||||
|
selected_cadence = validate_backtest_cadence(
|
||||||
|
cadence or DEFAULT_BACKTEST_CADENCE
|
||||||
|
)
|
||||||
|
_next_backtest_target_model = selected_model
|
||||||
|
_next_backtest_cadence = selected_cadence
|
||||||
|
return selected_model, selected_cadence
|
||||||
|
|
||||||
|
|
||||||
|
def queue_backtest_target_model(target_model: str | None) -> str:
|
||||||
|
"""Compatibility wrapper for callers selecting only the target model."""
|
||||||
|
selected, _ = queue_backtest_options(target_model, DEFAULT_BACKTEST_CADENCE)
|
||||||
|
return selected
|
||||||
|
|
||||||
|
|
||||||
|
def _consume_backtest_options() -> tuple[str, str]:
|
||||||
|
global _next_backtest_target_model, _next_backtest_cadence
|
||||||
|
selected = (_next_backtest_target_model, _next_backtest_cadence)
|
||||||
|
_next_backtest_target_model = PRODUCTION_GTL_TARGET_MODEL
|
||||||
|
_next_backtest_cadence = DEFAULT_BACKTEST_CADENCE
|
||||||
|
return selected
|
||||||
|
|
||||||
|
|
||||||
|
def _consume_backtest_target_model() -> str:
|
||||||
|
"""Compatibility wrapper consuming all queued one-run options."""
|
||||||
|
selected, _ = _consume_backtest_options()
|
||||||
|
return selected
|
||||||
|
|
||||||
|
|
||||||
def _log_event(level: int, event: str, **fields: object) -> None:
|
def _log_event(level: int, event: str, **fields: object) -> None:
|
||||||
"""Emit a structured JSON log line: {"event": ..., **fields}."""
|
"""Emit a structured JSON log line: {"event": ..., **fields}."""
|
||||||
logger.log(level, json.dumps({"event": event, **fields}))
|
logger.log(level, json.dumps({"event": event, **fields}))
|
||||||
@@ -125,6 +198,28 @@ def _log_job_error(job_name: str, ticker: str, error: Exception) -> None:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def _record_system_event(
|
||||||
|
*,
|
||||||
|
severity: str,
|
||||||
|
source: str,
|
||||||
|
code: str,
|
||||||
|
message: str,
|
||||||
|
symbol: str | None = None,
|
||||||
|
dedup_key: str | None = None,
|
||||||
|
) -> None:
|
||||||
|
"""Best-effort durable event for Admin → Jobs and the top-nav badge."""
|
||||||
|
from app.services.system_event_service import log_event_standalone
|
||||||
|
|
||||||
|
await log_event_standalone(
|
||||||
|
severity=severity,
|
||||||
|
source=source,
|
||||||
|
code=code,
|
||||||
|
message=message,
|
||||||
|
symbol=symbol,
|
||||||
|
dedup_key=dedup_key,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _runtime_start(job_name: str, total: int | None = None, message: str | None = None) -> None:
|
def _runtime_start(job_name: str, total: int | None = None, message: str | None = None) -> None:
|
||||||
_job_runtime[job_name] = {
|
_job_runtime[job_name] = {
|
||||||
**_idle_runtime(),
|
**_idle_runtime(),
|
||||||
@@ -179,6 +274,22 @@ def _runtime_finish(
|
|||||||
"message": message,
|
"message": message,
|
||||||
})
|
})
|
||||||
_job_runtime[job_name] = runtime
|
_job_runtime[job_name] = runtime
|
||||||
|
# Durable event for error / rate-limit finishes (badge + Admin → Jobs panel).
|
||||||
|
if status in ("error", "rate_limited"):
|
||||||
|
severity = "error" if status == "error" else "warning"
|
||||||
|
try:
|
||||||
|
loop = asyncio.get_running_loop()
|
||||||
|
loop.create_task(
|
||||||
|
_record_system_event(
|
||||||
|
severity=severity,
|
||||||
|
source=job_name,
|
||||||
|
code=f"job_{status}",
|
||||||
|
message=message or f"Job {job_name} finished with status {status}",
|
||||||
|
dedup_key=f"job:{job_name}:{status}:{(message or '')[:80]}"[:200],
|
||||||
|
)
|
||||||
|
)
|
||||||
|
except RuntimeError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
def get_job_runtime_snapshot(job_name: str | None = None) -> dict[str, dict[str, object]] | dict[str, object]:
|
def get_job_runtime_snapshot(job_name: str | None = None) -> dict[str, dict[str, object]] | dict[str, object]:
|
||||||
@@ -221,11 +332,11 @@ async def _get_ohlcv_priority_tickers(db: AsyncSession) -> list[str]:
|
|||||||
async def _get_top_pick_feeder_ids(db: AsyncSession) -> set[int]:
|
async def _get_top_pick_feeder_ids(db: AsyncSession) -> set[int]:
|
||||||
"""Ticker ids whose latest LONG setup makes them a top-pick feeder.
|
"""Ticker ids whose latest LONG setup makes them a top-pick feeder.
|
||||||
|
|
||||||
A dashboard 'top pick' is the highest-momentum *qualified* setup. Sentiment
|
A dashboard 'top pick' is the highest residual-momentum *qualified* setup.
|
||||||
can never move a ticker's momentum percentile (the gate's core axis) — only
|
Sentiment can never move a ticker's activation percentile (the gate's core
|
||||||
its confidence and EV ranking. So the only tickers that are, or could become
|
axis) — only its confidence and EV ranking. So the only tickers that are, or
|
||||||
with positive sentiment, a top pick are momentum leaders that already have a
|
could become with positive sentiment, a top pick are residual-momentum leaders
|
||||||
tradeable long setup clearing the R:R floor. That set is exactly:
|
that already have a tradeable long setup clearing the R:R floor. That set is exactly:
|
||||||
|
|
||||||
latest long setup with momentum_percentile >= gate AND rr_ratio >= floor.
|
latest long setup with momentum_percentile >= gate AND rr_ratio >= floor.
|
||||||
|
|
||||||
@@ -310,7 +421,7 @@ async def _get_sentiment_priority_tickers(db: AsyncSession) -> list[str]:
|
|||||||
is always fully covered. The two tiers only affect ORDER, so a mid-run provider
|
is always fully covered. The two tiers only affect ORDER, so a mid-run provider
|
||||||
rate limit still lands the names we care about first:
|
rate limit still lands the names we care about first:
|
||||||
|
|
||||||
Priority: top-pick feeders (momentum leaders with a tradeable long setup, see
|
Priority: top-pick feeders (residual-momentum leaders with a tradeable long setup, see
|
||||||
``_get_top_pick_feeder_ids``) + the curated watchlist + open paper trades —
|
``_get_top_pick_feeder_ids``) + the curated watchlist + open paper trades —
|
||||||
the set we never want shown without sentiment.
|
the set we never want shown without sentiment.
|
||||||
Filler: top-N by composite — a cheap discovery net for names not yet covered.
|
Filler: top-N by composite — a cheap discovery net for names not yet covered.
|
||||||
@@ -396,19 +507,29 @@ def _chunked(symbols: list[str], chunk_size: int) -> list[list[str]]:
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
async def collect_ohlcv(full_backfill: bool = False, job_name: str = "data_collector") -> None:
|
async def collect_ohlcv(
|
||||||
|
full_backfill: bool = False,
|
||||||
|
job_name: str = "data_collector",
|
||||||
|
*,
|
||||||
|
refetch_days: int = 0,
|
||||||
|
refresh_sr: bool = True,
|
||||||
|
) -> None:
|
||||||
"""Fetch latest daily OHLCV for all tracked tickers.
|
"""Fetch latest daily OHLCV for all tracked tickers.
|
||||||
|
|
||||||
Uses AlpacaOHLCVProvider. Processes each ticker independently.
|
Uses AlpacaOHLCVProvider. Processes each ticker independently.
|
||||||
On rate limit, records last successful ticker for resume.
|
On rate limit, records last successful ticker for resume.
|
||||||
Start date is resolved by ingestion progress:
|
Start date is resolved by ingestion progress:
|
||||||
- existing ticker: resume from last_ingested_date + 1
|
- existing ticker: overlap last_ingested_date so partial bars refresh
|
||||||
- new ticker: backfill the configured history window
|
- new ticker: backfill the configured history window
|
||||||
|
|
||||||
``full_backfill`` forces every ticker to re-fetch the full
|
``full_backfill`` forces every ticker to re-fetch the full
|
||||||
``settings.ohlcv_history_days`` window (ignoring incremental resume) — used by
|
``settings.ohlcv_history_days`` window (ignoring incremental resume) — used by
|
||||||
the manual data_backfill job to deepen shallow histories. ``job_name`` lets the
|
the manual data_backfill job to deepen shallow histories. ``job_name`` lets the
|
||||||
backfill report its own runtime/resume state separate from data_collector.
|
backfill report its own runtime/resume state separate from data_collector.
|
||||||
|
|
||||||
|
``refetch_days`` re-pulls the last N days regardless of ingestion progress —
|
||||||
|
the after-close run uses it to overwrite the day's partial intraday bar, which
|
||||||
|
resume logic would otherwise skip as "already up to date".
|
||||||
"""
|
"""
|
||||||
_log_event(logging.INFO, "job_start", job=job_name)
|
_log_event(logging.INFO, "job_start", job=job_name)
|
||||||
_runtime_start(job_name)
|
_runtime_start(job_name)
|
||||||
@@ -445,11 +566,14 @@ async def collect_ohlcv(full_backfill: bool = False, job_name: str = "data_colle
|
|||||||
return
|
return
|
||||||
|
|
||||||
end_date = date.today()
|
end_date = date.today()
|
||||||
# Full backfill: pass an explicit start_date so fetch_and_ingest re-pulls
|
# An explicit start_date makes fetch_and_ingest re-pull that window instead
|
||||||
# the whole window instead of resuming from the last stored bar.
|
# of resuming from the last stored bar (upsert overwrites, so this is safe).
|
||||||
backfill_start = (
|
if full_backfill:
|
||||||
end_date - timedelta(days=settings.ohlcv_history_days) if full_backfill else None
|
backfill_start = end_date - timedelta(days=settings.ohlcv_history_days)
|
||||||
)
|
elif refetch_days:
|
||||||
|
backfill_start = end_date - timedelta(days=refetch_days)
|
||||||
|
else:
|
||||||
|
backfill_start = None
|
||||||
|
|
||||||
for symbol in symbols:
|
for symbol in symbols:
|
||||||
_runtime_progress(job_name, processed=processed, total=total, current_ticker=symbol)
|
_runtime_progress(job_name, processed=processed, total=total, current_ticker=symbol)
|
||||||
@@ -457,11 +581,21 @@ async def collect_ohlcv(full_backfill: bool = False, job_name: str = "data_colle
|
|||||||
try:
|
try:
|
||||||
result = await ingestion_service.fetch_and_ingest(
|
result = await ingestion_service.fetch_and_ingest(
|
||||||
db, provider, symbol, start_date=backfill_start, end_date=end_date,
|
db, provider, symbol, start_date=backfill_start, end_date=end_date,
|
||||||
|
refresh_sr=refresh_sr,
|
||||||
)
|
)
|
||||||
_last_successful[job_name] = symbol
|
_last_successful[job_name] = symbol
|
||||||
processed += 1
|
processed += 1
|
||||||
_runtime_progress(job_name, processed=processed, total=total, current_ticker=symbol)
|
_runtime_progress(job_name, processed=processed, total=total, current_ticker=symbol)
|
||||||
_log_event(logging.INFO, "ticker_collected", job=job_name, ticker=symbol, status=result.status, records=result.records_ingested)
|
_log_event(logging.INFO, "ticker_collected", job=job_name, ticker=symbol, status=result.status, records=result.records_ingested)
|
||||||
|
if result.status == "stale":
|
||||||
|
await _record_system_event(
|
||||||
|
severity="warning",
|
||||||
|
source=job_name,
|
||||||
|
code="ohlcv_stale",
|
||||||
|
message=result.message or f"No new OHLCV bars for {symbol}",
|
||||||
|
symbol=symbol,
|
||||||
|
dedup_key=f"ohlcv_stale:{symbol}",
|
||||||
|
)
|
||||||
if result.status == "partial":
|
if result.status == "partial":
|
||||||
# Rate limited — stop and resume next run
|
# Rate limited — stop and resume next run
|
||||||
_log_event(logging.WARNING, "rate_limited", job=job_name, ticker=symbol, processed=processed)
|
_log_event(logging.WARNING, "rate_limited", job=job_name, ticker=symbol, processed=processed)
|
||||||
@@ -469,6 +603,14 @@ async def collect_ohlcv(full_backfill: bool = False, job_name: str = "data_colle
|
|||||||
return
|
return
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
_log_job_error(job_name, symbol, exc)
|
_log_job_error(job_name, symbol, exc)
|
||||||
|
await _record_system_event(
|
||||||
|
severity="error",
|
||||||
|
source=job_name,
|
||||||
|
code="job_ticker_error",
|
||||||
|
message=f"{type(exc).__name__}: {exc}",
|
||||||
|
symbol=symbol,
|
||||||
|
dedup_key=f"job_ticker_error:{job_name}:{symbol}:{type(exc).__name__}",
|
||||||
|
)
|
||||||
|
|
||||||
# Reset resume pointer on full completion
|
# Reset resume pointer on full completion
|
||||||
_last_successful[job_name] = None
|
_last_successful[job_name] = None
|
||||||
@@ -479,6 +621,11 @@ async def collect_ohlcv(full_backfill: bool = False, job_name: str = "data_colle
|
|||||||
_runtime_finish(job_name, "error", processed=processed, total=total, message=str(exc))
|
_runtime_finish(job_name, "error", processed=processed, total=total, message=str(exc))
|
||||||
|
|
||||||
|
|
||||||
|
async def collect_ohlcv_for_scan() -> None:
|
||||||
|
"""Near-close fetch; the scanner immediately rebuilds S/R per ticker."""
|
||||||
|
await collect_ohlcv(refresh_sr=False)
|
||||||
|
|
||||||
|
|
||||||
async def backfill_ohlcv() -> None:
|
async def backfill_ohlcv() -> None:
|
||||||
"""Deep historical backfill: re-fetch the full ``settings.ohlcv_history_days``
|
"""Deep historical backfill: re-fetch the full ``settings.ohlcv_history_days``
|
||||||
window for every ticker, ignoring incremental resume.
|
window for every ticker, ignoring incremental resume.
|
||||||
@@ -490,6 +637,76 @@ async def backfill_ohlcv() -> None:
|
|||||||
await collect_ohlcv(full_backfill=True, job_name="data_backfill")
|
await collect_ohlcv(full_backfill=True, job_name="data_backfill")
|
||||||
|
|
||||||
|
|
||||||
|
async def run_shadow_book() -> None:
|
||||||
|
"""Open the strategy's own positions from the latest qualifying scan.
|
||||||
|
|
||||||
|
The shadow book is the faithful live twin of the backtest: top-ranked
|
||||||
|
qualified setups, up to capacity, 1% risk, no human input. It runs straight
|
||||||
|
after the near-close scan so its entries are marked at the same near-close
|
||||||
|
prices the discretionary book sees, leaving *selection* as the only
|
||||||
|
difference between the two books.
|
||||||
|
|
||||||
|
When run as a pipeline step it acts only on the scan that stamped *this
|
||||||
|
pipeline's* run id (``expected_run_id``): if the pipeline's own scan was
|
||||||
|
disabled or failed, the stored run id is some other scan's — including a
|
||||||
|
manual scan that overlapped and finished last — and shadow refuses.
|
||||||
|
Triggered directly from Admin (no pipeline context) it falls back to the
|
||||||
|
scan-freshness window — an explicit operator action.
|
||||||
|
|
||||||
|
Opt-in (``shadow_book_enabled``) because it writes live trades.
|
||||||
|
"""
|
||||||
|
job_name = "shadow_book"
|
||||||
|
expected_run_id = pipeline_run.current()
|
||||||
|
_log_event(logging.INFO, "job_start", job=job_name)
|
||||||
|
_runtime_start(job_name, total=1)
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with async_session_factory() as db:
|
||||||
|
if not await _is_job_enabled(db, job_name):
|
||||||
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="disabled")
|
||||||
|
_runtime_finish(job_name, "skipped", processed=0, total=1, message="Disabled")
|
||||||
|
return False
|
||||||
|
if not await shadow_book_service.is_enabled(db):
|
||||||
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="not enabled in settings")
|
||||||
|
_runtime_finish(job_name, "skipped", processed=0, total=1, message="Not enabled")
|
||||||
|
return
|
||||||
|
|
||||||
|
from app.services.admin_service import get_activation_config
|
||||||
|
|
||||||
|
activation_config = await get_activation_config(db)
|
||||||
|
summary = await shadow_book_service.open_shadow_positions(
|
||||||
|
db,
|
||||||
|
activation_config=activation_config,
|
||||||
|
expected_run_id=expected_run_id,
|
||||||
|
)
|
||||||
|
symbols = await shadow_book_service.symbols_for(db, summary["symbols"])
|
||||||
|
|
||||||
|
_runtime_progress(job_name, processed=1, total=1)
|
||||||
|
_runtime_finish(
|
||||||
|
job_name, "completed", processed=1, total=1,
|
||||||
|
message=(
|
||||||
|
f"Opened {summary['opened']} ({', '.join(symbols) if symbols else 'none'}); "
|
||||||
|
f"held {summary['skipped_held']}, gate-locked {summary['skipped_locked']}"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
_log_event(logging.INFO, "job_complete", job=job_name, opened=summary["opened"], symbols=symbols)
|
||||||
|
except Exception as exc:
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=str(exc))
|
||||||
|
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
||||||
|
|
||||||
|
|
||||||
|
async def collect_ohlcv_final() -> None:
|
||||||
|
"""After-close OHLCV refresh that replaces the day's partial bar.
|
||||||
|
|
||||||
|
Intraday runs store today's bar while the session is still open, so ingestion
|
||||||
|
progress already reads "today" and incremental resume would skip the day
|
||||||
|
entirely — leaving a partial bar as the permanent record. ``refetch_days``
|
||||||
|
forces the last few sessions to be re-pulled so outcome evaluation and
|
||||||
|
fill-quality checks grade against the real close.
|
||||||
|
"""
|
||||||
|
await collect_ohlcv(refetch_days=_FINAL_REFETCH_DAYS)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Job: Sentiment Collector
|
# Job: Sentiment Collector
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -621,6 +838,22 @@ async def collect_fundamentals() -> None:
|
|||||||
_log_event(logging.INFO, "job_skipped", job=job_name, reason="disabled")
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="disabled")
|
||||||
_runtime_finish(job_name, "skipped", processed=0, total=0, message="Disabled")
|
_runtime_finish(job_name, "skipped", processed=0, total=0, message="Disabled")
|
||||||
return
|
return
|
||||||
|
if await fundamental_data_refresh_service.is_enabled(db):
|
||||||
|
message = "SEC + Dolt fundamentals cutover is active"
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"job_skipped",
|
||||||
|
job=job_name,
|
||||||
|
reason="sec_dolt_cutover_active",
|
||||||
|
)
|
||||||
|
_runtime_finish(
|
||||||
|
job_name,
|
||||||
|
"skipped",
|
||||||
|
processed=0,
|
||||||
|
total=0,
|
||||||
|
message=message,
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
symbols = await _get_fundamental_priority_tickers(db)
|
symbols = await _get_fundamental_priority_tickers(db)
|
||||||
if not symbols:
|
if not symbols:
|
||||||
@@ -715,6 +948,184 @@ async def collect_fundamentals() -> None:
|
|||||||
_runtime_finish(job_name, "error", processed=processed, total=total, message=str(exc))
|
_runtime_finish(job_name, "error", processed=processed, total=total, message=str(exc))
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Jobs: shadow fundamentals sources
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_shadow_import(job_name: str, importer: SourceImporter) -> bool:
|
||||||
|
"""Run an importer and return whether its scheduled job was enabled.
|
||||||
|
|
||||||
|
The SEC wrapper uses the return value to run its activated local cache step
|
||||||
|
after deferred, failed, no-op, promoted, or source-locked attempts while honoring
|
||||||
|
the job-level disable switch.
|
||||||
|
"""
|
||||||
|
_log_event(logging.INFO, "job_start", job=job_name)
|
||||||
|
_runtime_start(job_name, total=1)
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with async_session_factory() as db:
|
||||||
|
if not await _is_job_enabled(db, job_name):
|
||||||
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="disabled")
|
||||||
|
_runtime_finish(job_name, "skipped", processed=0, total=1, message="Disabled")
|
||||||
|
return
|
||||||
|
|
||||||
|
run = await run_import(importer)
|
||||||
|
if run is None:
|
||||||
|
message = "Another import for this source is already running"
|
||||||
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="source_locked")
|
||||||
|
_runtime_finish(job_name, "skipped", processed=0, total=1, message=message)
|
||||||
|
return True
|
||||||
|
|
||||||
|
revision = f" · {run.revision[:12]}" if run.revision else ""
|
||||||
|
message = f"{run.status}{revision}"
|
||||||
|
if run.status == STATUS_DEFERRED:
|
||||||
|
message = run.error_details or message
|
||||||
|
_log_event(logging.INFO, "job_deferred", job=job_name, message=message)
|
||||||
|
_runtime_finish(job_name, "deferred", processed=0, total=1, message=message)
|
||||||
|
return True
|
||||||
|
if run.status == STATUS_FAILED:
|
||||||
|
message = run.error_details or message
|
||||||
|
_log_event(logging.ERROR, "job_error", job=job_name, message=message)
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=message)
|
||||||
|
return True
|
||||||
|
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"job_complete",
|
||||||
|
job=job_name,
|
||||||
|
import_status=run.status,
|
||||||
|
revision=run.revision,
|
||||||
|
)
|
||||||
|
_runtime_finish(job_name, "completed", processed=1, total=1, message=message)
|
||||||
|
return True
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message="Cancelled")
|
||||||
|
raise
|
||||||
|
except Exception as exc:
|
||||||
|
_log_event(
|
||||||
|
logging.ERROR,
|
||||||
|
"job_error",
|
||||||
|
job=job_name,
|
||||||
|
error_type=type(exc).__name__,
|
||||||
|
message=str(exc),
|
||||||
|
)
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=str(exc))
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
async def run_dolt_earnings_import() -> None:
|
||||||
|
"""Pull and import the Dolt earnings calendar/results feed in shadow."""
|
||||||
|
await _run_shadow_import("dolt_earnings_import", DoltEarningsImporter())
|
||||||
|
|
||||||
|
|
||||||
|
async def run_sec_fundamentals_import() -> None:
|
||||||
|
"""Import SEC facts, then run the activated local compat-cache refresh.
|
||||||
|
|
||||||
|
The refresh is deliberately separate from the network import result. Once
|
||||||
|
activated it therefore still runs from stored snapshots/earnings/prices when
|
||||||
|
SEC is unavailable, unchanged, or another SEC import owns the source lock.
|
||||||
|
"""
|
||||||
|
job_name = "sec_fundamentals_import"
|
||||||
|
job_enabled = await _run_shadow_import(job_name, SecFundamentalsImporter())
|
||||||
|
if not job_enabled:
|
||||||
|
return
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with async_session_factory() as db:
|
||||||
|
summary = await fundamental_data_refresh_service.refresh_if_enabled(db)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
_runtime_finish(
|
||||||
|
job_name, "error", processed=0, total=1, message="Cancelled"
|
||||||
|
)
|
||||||
|
raise
|
||||||
|
except Exception as exc:
|
||||||
|
message = f"Local fundamental_data refresh failed: {exc}"
|
||||||
|
_log_event(
|
||||||
|
logging.ERROR,
|
||||||
|
"fundamental_data_refresh_error",
|
||||||
|
job=job_name,
|
||||||
|
error_type=type(exc).__name__,
|
||||||
|
message=str(exc),
|
||||||
|
)
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=message)
|
||||||
|
return
|
||||||
|
|
||||||
|
if not summary["enabled"]:
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"fundamental_data_refresh_skipped",
|
||||||
|
job=job_name,
|
||||||
|
reason="cutover_disabled",
|
||||||
|
setting=fundamental_data_refresh_service.ACTIVATION_KEY,
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"fundamental_data_refresh_complete",
|
||||||
|
job=job_name,
|
||||||
|
**summary,
|
||||||
|
)
|
||||||
|
runtime = get_job_runtime_snapshot(job_name)
|
||||||
|
if runtime.get("status") == "completed":
|
||||||
|
import_message = runtime.get("message") or "import completed"
|
||||||
|
cache_message = (
|
||||||
|
f"cache {summary['refreshed']} · "
|
||||||
|
f"{summary['score_inputs_changed']} score inputs changed"
|
||||||
|
)
|
||||||
|
_runtime_finish(
|
||||||
|
job_name,
|
||||||
|
"completed",
|
||||||
|
processed=1,
|
||||||
|
total=1,
|
||||||
|
message=f"{import_message} · {cache_message}",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def run_fundamentals_parity_report() -> None:
|
||||||
|
"""Generate the A5 comparison bundle without mutating live fundamentals/scores."""
|
||||||
|
job_name = "fundamentals_parity_report"
|
||||||
|
_log_event(logging.INFO, "job_start", job=job_name)
|
||||||
|
_runtime_start(job_name, total=1)
|
||||||
|
try:
|
||||||
|
async with async_session_factory() as db:
|
||||||
|
if not await _is_job_enabled(db, job_name):
|
||||||
|
_runtime_finish(
|
||||||
|
job_name, "skipped", processed=0, total=1, message="Disabled"
|
||||||
|
)
|
||||||
|
return
|
||||||
|
report, artifacts = await fundamentals_parity_service.generate_and_store(
|
||||||
|
db, settings.fundamentals_parity_report_dir
|
||||||
|
)
|
||||||
|
summary = report["summary"]
|
||||||
|
message = (
|
||||||
|
f"{summary['universe_count']} tickers · "
|
||||||
|
f"{summary['fundamental_score_material_changes']} material score changes"
|
||||||
|
)
|
||||||
|
_runtime_finish(job_name, "completed", processed=1, total=1, message=message)
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"job_complete",
|
||||||
|
job=job_name,
|
||||||
|
generated_at=report["generated_at"],
|
||||||
|
json_path=artifacts["json"],
|
||||||
|
csv_path=artifacts["csv"],
|
||||||
|
)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message="Cancelled")
|
||||||
|
raise
|
||||||
|
except Exception as exc:
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=str(exc))
|
||||||
|
_log_event(
|
||||||
|
logging.ERROR,
|
||||||
|
"job_error",
|
||||||
|
job=job_name,
|
||||||
|
error_type=type(exc).__name__,
|
||||||
|
message=str(exc),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Job: R:R Scanner
|
# Job: R:R Scanner
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -866,6 +1277,34 @@ async def compute_market_regime() -> None:
|
|||||||
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Job: Benchmark Collector (SPY closes for paper-trade alpha)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
async def collect_benchmark() -> None:
|
||||||
|
"""Refresh the stored benchmark (SPY) daily closes used for paper-trade alpha."""
|
||||||
|
job_name = "benchmark_collector"
|
||||||
|
_log_event(logging.INFO, "job_start", job=job_name)
|
||||||
|
_runtime_start(job_name, total=1)
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with async_session_factory() as db:
|
||||||
|
if not await _is_job_enabled(db, job_name):
|
||||||
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="disabled")
|
||||||
|
_runtime_finish(job_name, "skipped", processed=0, total=1, message="Disabled")
|
||||||
|
return
|
||||||
|
|
||||||
|
written = await refresh_benchmark_prices(db)
|
||||||
|
|
||||||
|
_runtime_progress(job_name, processed=1, total=1)
|
||||||
|
_runtime_finish(job_name, "completed", processed=1, total=1, message=f"{written} rows")
|
||||||
|
_log_event(logging.INFO, "job_complete", job=job_name, rows=written)
|
||||||
|
except Exception as exc:
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=str(exc))
|
||||||
|
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Job: Regime Monitor
|
# Job: Regime Monitor
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -891,12 +1330,20 @@ async def compute_regime_monitor() -> None:
|
|||||||
|
|
||||||
result = await update_regime_monitor(db)
|
result = await update_regime_monitor(db)
|
||||||
|
|
||||||
|
state = result.get("state") or {}
|
||||||
|
warning = result.get("warning") or {}
|
||||||
_runtime_progress(job_name, processed=1, total=1)
|
_runtime_progress(job_name, processed=1, total=1)
|
||||||
_runtime_finish(
|
_runtime_finish(
|
||||||
job_name, "completed", processed=1, total=1,
|
job_name, "completed", processed=1, total=1,
|
||||||
message=f"Index: {result.get('total_score')} ({result.get('band')})",
|
message=f"State: {state.get('score')} · Warning: {warning.get('score')}",
|
||||||
|
)
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"job_complete",
|
||||||
|
job=job_name,
|
||||||
|
state=state.get("score"),
|
||||||
|
warning=warning.get("score"),
|
||||||
)
|
)
|
||||||
_log_event(logging.INFO, "job_complete", job=job_name, score=result.get("total_score"))
|
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
_runtime_finish(job_name, "error", processed=0, total=1, message=str(exc))
|
_runtime_finish(job_name, "error", processed=0, total=1, message=str(exc))
|
||||||
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
||||||
@@ -910,7 +1357,14 @@ async def compute_regime_monitor() -> None:
|
|||||||
async def run_backtest_job() -> None:
|
async def run_backtest_job() -> None:
|
||||||
"""Replay the price-derived engine over history and cache the report."""
|
"""Replay the price-derived engine over history and cache the report."""
|
||||||
job_name = "backtest"
|
job_name = "backtest"
|
||||||
_log_event(logging.INFO, "job_start", job=job_name)
|
target_model, cadence = _consume_backtest_options()
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"job_start",
|
||||||
|
job=job_name,
|
||||||
|
target_model=target_model,
|
||||||
|
cadence=cadence,
|
||||||
|
)
|
||||||
_runtime_start(job_name)
|
_runtime_start(job_name)
|
||||||
|
|
||||||
def _on_progress(done: int, count: int, symbol: str) -> None:
|
def _on_progress(done: int, count: int, symbol: str) -> None:
|
||||||
@@ -923,12 +1377,22 @@ async def run_backtest_job() -> None:
|
|||||||
_runtime_finish(job_name, "skipped", processed=0, total=0, message="Disabled")
|
_runtime_finish(job_name, "skipped", processed=0, total=0, message="Disabled")
|
||||||
return
|
return
|
||||||
|
|
||||||
report = await run_backtest_and_store(db, _on_progress)
|
report = await run_backtest_and_store(
|
||||||
|
db,
|
||||||
|
_on_progress,
|
||||||
|
target_model=target_model,
|
||||||
|
cadence=cadence,
|
||||||
|
)
|
||||||
|
|
||||||
_runtime_finish(
|
_runtime_finish(
|
||||||
job_name, "completed",
|
job_name, "completed",
|
||||||
processed=report.get("tickers", 0), total=report.get("tickers", 0),
|
processed=report.get("tickers", 0), total=report.get("tickers", 0),
|
||||||
message=f"{report.get('candidates', 0)} setups, {report.get('qualified', 0)} qualified",
|
message=(
|
||||||
|
f"{BACKTEST_TARGET_MODELS[target_model]}: "
|
||||||
|
f"{cadence} cadence, "
|
||||||
|
f"{report.get('candidates', 0)} setups, "
|
||||||
|
f"{report.get('qualified', 0)} qualified"
|
||||||
|
),
|
||||||
)
|
)
|
||||||
_log_event(logging.INFO, "job_complete", job=job_name, candidates=report.get("candidates"))
|
_log_event(logging.INFO, "job_complete", job=job_name, candidates=report.get("candidates"))
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -962,7 +1426,11 @@ async def run_event_study_job() -> None:
|
|||||||
|
|
||||||
_runtime_progress(job_name, processed=1, total=1)
|
_runtime_progress(job_name, processed=1, total=1)
|
||||||
if report.get("available"):
|
if report.get("available"):
|
||||||
msg = f"{len(report.get('events', []))} events, lead Δ {report.get('lead_delta_days')}d"
|
metrics = report.get("metrics") or {}
|
||||||
|
msg = (
|
||||||
|
f"{metrics.get('events_warned', 0)}/{metrics.get('events', 0)} warned, "
|
||||||
|
f"{metrics.get('false_alarms_per_year', 0)} false alarms/year"
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
msg = report.get("reason", "no data")
|
msg = report.get("reason", "no data")
|
||||||
_runtime_finish(job_name, "completed", processed=1, total=1, message=msg)
|
_runtime_finish(job_name, "completed", processed=1, total=1, message=msg)
|
||||||
@@ -1014,14 +1482,52 @@ async def sync_ticker_universe() -> None:
|
|||||||
# updates its own runtime status while the pipeline runs.
|
# updates its own runtime status while the pipeline runs.
|
||||||
#
|
#
|
||||||
# Daily (full): the complete data→signal refresh, once a day.
|
# Daily (full): the complete data→signal refresh, once a day.
|
||||||
|
# Morning (America/New_York ~02:00): refresh data + display context. No R:R scan
|
||||||
|
# — the qualifying full-universe scan runs once near the US close so post-stop
|
||||||
|
# gate-reset sees one observation per trading day (plus the trade_policy
|
||||||
|
# distinct-day guard for manual re-scans).
|
||||||
|
# Sessions re-pulled by the after-close fetch so the consolidated bar overwrites
|
||||||
|
# the intraday partial one (covers a long weekend / holiday gap).
|
||||||
|
_FINAL_REFETCH_DAYS = 5
|
||||||
|
|
||||||
_DAILY_PIPELINE_STEPS = [
|
_DAILY_PIPELINE_STEPS = [
|
||||||
("data_collector", "collect_ohlcv"),
|
("data_collector", "collect_ohlcv"),
|
||||||
|
("benchmark_collector", "collect_benchmark"),
|
||||||
("sentiment_collector", "collect_sentiment"),
|
("sentiment_collector", "collect_sentiment"),
|
||||||
("rr_scanner", "scan_rr"),
|
|
||||||
("outcome_evaluator", "evaluate_outcomes"),
|
|
||||||
("market_regime", "compute_market_regime"),
|
("market_regime", "compute_market_regime"),
|
||||||
# Observational only — runs here for scheduling; its output feeds nothing else.
|
# Observational only — display/alerts; not trade selection.
|
||||||
("regime_monitor", "compute_regime_monitor"),
|
("regime_monitor", "compute_regime_monitor"),
|
||||||
|
# Alerts after regime so quadrant changes reach Telegram in the morning.
|
||||||
|
# Dispatcher is change-driven; quiet days stay quiet. Setup alerts still
|
||||||
|
# fire on the near-close pipeline after the qualifying scan.
|
||||||
|
("alerts", "dispatch_alerts_job"),
|
||||||
|
]
|
||||||
|
|
||||||
|
# Near-close (~15:30 ET Mon–Fri): refresh in-progress day-t bars (incremental
|
||||||
|
# ingestion overlaps the latest stored session), then the only daily
|
||||||
|
# qualifying R:R scan, then Telegram immediately so manual fills can still hit
|
||||||
|
# MOC cutoffs (~15:50/15:55). Under a 15-minute delayed SIP feed a 15:30 scan
|
||||||
|
# may see ~15:15 prices — immaterial for a 12-1 momentum signal.
|
||||||
|
#
|
||||||
|
# US early-close days (~3/year, 13:00 ET close): this job runs post-close and
|
||||||
|
# entries behave like stale_close (still acceptable per execution-recovery matrix).
|
||||||
|
# No exchange calendar dependency.
|
||||||
|
_NEAR_CLOSE_PIPELINE_STEPS = [
|
||||||
|
# Must land today's in-progress bar (~20 min behind live), or the scan falls
|
||||||
|
# back to the previous close and execution degrades to the stale_close floor.
|
||||||
|
("data_collector", "collect_ohlcv_for_scan"),
|
||||||
|
("rr_scanner", "scan_rr"),
|
||||||
|
# Straight after the scan so shadow entries mark at the same near-close
|
||||||
|
# prices the discretionary book is looking at.
|
||||||
|
("shadow_book", "run_shadow_book"),
|
||||||
|
("alerts", "dispatch_alerts_job"),
|
||||||
|
]
|
||||||
|
|
||||||
|
# After close (~16:45 ET Mon–Fri): fresh OHLCV fetch so outcomes resolve on the
|
||||||
|
# final bar, not the near-close partial bar, then outcome/paper close.
|
||||||
|
_AFTER_CLOSE_PIPELINE_STEPS = [
|
||||||
|
("data_collector", "collect_ohlcv_final"),
|
||||||
|
("outcome_evaluator", "evaluate_outcomes"),
|
||||||
]
|
]
|
||||||
|
|
||||||
# Intraday (light): keep prices current and resolve outcomes through the day,
|
# Intraday (light): keep prices current and resolve outcomes through the day,
|
||||||
@@ -1033,12 +1539,21 @@ _INTRADAY_PIPELINE_STEPS = [
|
|||||||
("outcome_evaluator", "evaluate_outcomes"),
|
("outcome_evaluator", "evaluate_outcomes"),
|
||||||
]
|
]
|
||||||
|
|
||||||
|
# Warn if near-close fetch+scan+alert drifts past this — entries leave the close
|
||||||
|
# and the stale_close floor quietly becomes the ceiling.
|
||||||
|
_NEAR_CLOSE_DURATION_WARN_SECONDS = 600
|
||||||
|
|
||||||
|
|
||||||
async def _run_pipeline(job_name: str, steps: list[tuple[str, str]]) -> None:
|
async def _run_pipeline(job_name: str, steps: list[tuple[str, str]]) -> None:
|
||||||
"""Run an ordered list of (step_name, coroutine_name) steps.
|
"""Run an ordered list of (step_name, coroutine_name) steps.
|
||||||
|
|
||||||
Each step respects its own enable flag and manages its own runtime status; a
|
Each step respects its own enable flag and manages its own runtime status; a
|
||||||
failing step is logged and the pipeline continues with the next one.
|
failing step is logged and the pipeline continues with the next one.
|
||||||
|
|
||||||
|
A unique run id is bound for the invocation and visible to every step via the
|
||||||
|
shared task context: the scan step stamps it into its completion markers and
|
||||||
|
the shadow step requires an exact match, so only a scan that ran inside this
|
||||||
|
pipeline can drive the shadow book.
|
||||||
"""
|
"""
|
||||||
_log_event(logging.INFO, "job_start", job=job_name)
|
_log_event(logging.INFO, "job_start", job=job_name)
|
||||||
async with async_session_factory() as db:
|
async with async_session_factory() as db:
|
||||||
@@ -1052,6 +1567,7 @@ async def _run_pipeline(job_name: str, steps: list[tuple[str, str]]) -> None:
|
|||||||
|
|
||||||
funcs = globals()
|
funcs = globals()
|
||||||
done = 0
|
done = 0
|
||||||
|
token = pipeline_run.bind(pipeline_run.new_run_id())
|
||||||
try:
|
try:
|
||||||
for step_name, func_name in steps:
|
for step_name, func_name in steps:
|
||||||
_runtime_progress(job_name, processed=done, total=total, current_ticker=step_name)
|
_runtime_progress(job_name, processed=done, total=total, current_ticker=step_name)
|
||||||
@@ -1065,14 +1581,49 @@ async def _run_pipeline(job_name: str, steps: list[tuple[str, str]]) -> None:
|
|||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
_runtime_finish(job_name, "error", processed=done, total=total, message=str(exc))
|
_runtime_finish(job_name, "error", processed=done, total=total, message=str(exc))
|
||||||
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
||||||
|
finally:
|
||||||
|
pipeline_run.release(token)
|
||||||
|
|
||||||
|
|
||||||
async def run_daily_pipeline() -> None:
|
async def run_daily_pipeline() -> None:
|
||||||
"""Full daily flow: OHLCV → sentiment → R:R scan → outcome eval (+paper
|
"""Morning flow: OHLCV → benchmark → sentiment → market regime (no scan)."""
|
||||||
close) → market regime."""
|
|
||||||
await _run_pipeline("daily_pipeline", _DAILY_PIPELINE_STEPS)
|
await _run_pipeline("daily_pipeline", _DAILY_PIPELINE_STEPS)
|
||||||
|
|
||||||
|
|
||||||
|
async def run_near_close_pipeline() -> None:
|
||||||
|
"""Near-close flow: OHLCV fetch → R:R scan → Telegram alerts.
|
||||||
|
|
||||||
|
Logs wall duration; warn if past 10 minutes so operators notice close drift.
|
||||||
|
"""
|
||||||
|
import time
|
||||||
|
|
||||||
|
started = time.monotonic()
|
||||||
|
await _run_pipeline("near_close_pipeline", _NEAR_CLOSE_PIPELINE_STEPS)
|
||||||
|
elapsed = time.monotonic() - started
|
||||||
|
payload = {
|
||||||
|
"job": "near_close_pipeline",
|
||||||
|
"duration_seconds": round(elapsed, 1),
|
||||||
|
}
|
||||||
|
if elapsed > _NEAR_CLOSE_DURATION_WARN_SECONDS:
|
||||||
|
_log_event(
|
||||||
|
logging.WARNING,
|
||||||
|
"near_close_pipeline_slow",
|
||||||
|
**payload,
|
||||||
|
threshold_seconds=_NEAR_CLOSE_DURATION_WARN_SECONDS,
|
||||||
|
message=(
|
||||||
|
"Near-close pipeline exceeded 10 minutes — entries may drift from "
|
||||||
|
"the close toward the stale_close research floor"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
_log_event(logging.INFO, "near_close_pipeline_duration", **payload)
|
||||||
|
|
||||||
|
|
||||||
|
async def run_after_close_pipeline() -> None:
|
||||||
|
"""After-close flow: OHLCV fetch (final bar) → outcome eval (+paper close)."""
|
||||||
|
await _run_pipeline("after_close_pipeline", _AFTER_CLOSE_PIPELINE_STEPS)
|
||||||
|
|
||||||
|
|
||||||
async def run_intraday_pipeline() -> None:
|
async def run_intraday_pipeline() -> None:
|
||||||
"""Light intraday flow: refresh OHLCV → evaluate outcomes (+paper close)."""
|
"""Light intraday flow: refresh OHLCV → evaluate outcomes (+paper close)."""
|
||||||
await _run_pipeline("intraday_pipeline", _INTRADAY_PIPELINE_STEPS)
|
await _run_pipeline("intraday_pipeline", _INTRADAY_PIPELINE_STEPS)
|
||||||
@@ -1104,16 +1655,40 @@ def _parse_frequency(freq: str) -> dict[str, int]:
|
|||||||
# every process restart, so on a box that's redeployed often it can keep being
|
# every process restart, so on a box that's redeployed often it can keep being
|
||||||
# deferred and never fire. Cron fires at a fixed local time regardless.
|
# deferred and never fire. Cron fires at a fixed local time regardless.
|
||||||
|
|
||||||
|
# All wall times are America/New_York after the near-close execution cutover.
|
||||||
|
# Stored SystemSetting values shadow these defaults — deploy migration 023
|
||||||
|
# rewrites schedule_* keys so prod does not keep scanning at 07:00 Berlin.
|
||||||
|
# DAY-OF-WEEK MUST BE NAMES, NEVER NUMBERS. APScheduler's from_crontab() passes
|
||||||
|
# field 5 straight to its own day_of_week, where 0=Monday — so "1-5" resolves to
|
||||||
|
# Tue–Sat, silently skipping every Monday and scanning on Saturdays. Names are
|
||||||
|
# unambiguous in both dialects.
|
||||||
SCHEDULE_DEFAULTS: dict[str, str] = {
|
SCHEDULE_DEFAULTS: dict[str, str] = {
|
||||||
"schedule_timezone": "Europe/Berlin",
|
"schedule_timezone": "America/New_York",
|
||||||
"schedule_daily_pipeline_cron": "0 7 * * *", # full refresh, ready by ~8am
|
# Morning data/display refresh (no qualifying R:R scan).
|
||||||
"schedule_intraday_pipeline_cron": "0 14-22 * * 1-5", # hourly across the US session
|
"schedule_daily_pipeline_cron": "0 2 * * *",
|
||||||
"schedule_fundamentals_cron": "0 4 * * 1", # weekly, early Monday (slow job)
|
# Bulk source imports. The SEC job writes the legacy compat cache only after
|
||||||
|
# the explicit, default-off A5 cutover setting is enabled.
|
||||||
|
"schedule_dolt_earnings_cron": "30 2 * * *",
|
||||||
|
"schedule_sec_fundamentals_cron": "0 4 * * *",
|
||||||
|
"schedule_fundamentals_parity_cron": "30 5 * * *",
|
||||||
|
# Fetch in-progress bars → scan → Telegram (manual MOC window).
|
||||||
|
"schedule_near_close_pipeline_cron": "30 15 * * mon-fri",
|
||||||
|
# Fetch final bars → outcome eval (must not run on the partial near-close bar).
|
||||||
|
"schedule_after_close_pipeline_cron": "45 16 * * mon-fri",
|
||||||
|
# Hourly mid-session price + outcome (10:00–15:00 ET Mon–Fri).
|
||||||
|
"schedule_intraday_pipeline_cron": "0 10-15 * * mon-fri",
|
||||||
|
# Weekly fundamentals early Monday NY.
|
||||||
|
"schedule_fundamentals_cron": "0 1 * * mon",
|
||||||
}
|
}
|
||||||
|
|
||||||
# job id -> schedule setting key
|
# job id -> schedule setting key
|
||||||
_CRON_JOBS: dict[str, str] = {
|
_CRON_JOBS: dict[str, str] = {
|
||||||
"daily_pipeline": "schedule_daily_pipeline_cron",
|
"daily_pipeline": "schedule_daily_pipeline_cron",
|
||||||
|
"dolt_earnings_import": "schedule_dolt_earnings_cron",
|
||||||
|
"sec_fundamentals_import": "schedule_sec_fundamentals_cron",
|
||||||
|
"fundamentals_parity_report": "schedule_fundamentals_parity_cron",
|
||||||
|
"near_close_pipeline": "schedule_near_close_pipeline_cron",
|
||||||
|
"after_close_pipeline": "schedule_after_close_pipeline_cron",
|
||||||
"intraday_pipeline": "schedule_intraday_pipeline_cron",
|
"intraday_pipeline": "schedule_intraday_pipeline_cron",
|
||||||
"fundamental_collector": "schedule_fundamentals_cron",
|
"fundamental_collector": "schedule_fundamentals_cron",
|
||||||
}
|
}
|
||||||
@@ -1176,8 +1751,10 @@ def configure_scheduler(schedule_config: dict[str, str] | None = None) -> None:
|
|||||||
# interval job). They stay manually triggerable from Admin → Jobs.
|
# interval job). They stay manually triggerable from Admin → Jobs.
|
||||||
_members = [
|
_members = [
|
||||||
(collect_ohlcv, "data_collector", "Data Collector (OHLCV)"),
|
(collect_ohlcv, "data_collector", "Data Collector (OHLCV)"),
|
||||||
|
(collect_benchmark, "benchmark_collector", "Benchmark Collector"),
|
||||||
(collect_sentiment, "sentiment_collector", "Sentiment Collector"),
|
(collect_sentiment, "sentiment_collector", "Sentiment Collector"),
|
||||||
(scan_rr, "rr_scanner", "R:R Scanner"),
|
(scan_rr, "rr_scanner", "R:R Scanner"),
|
||||||
|
(run_shadow_book, "shadow_book", "Shadow Book (auto-traded strategy)"),
|
||||||
(evaluate_outcomes, "outcome_evaluator", "Outcome Evaluator"),
|
(evaluate_outcomes, "outcome_evaluator", "Outcome Evaluator"),
|
||||||
(compute_market_regime, "market_regime", "Market Regime"),
|
(compute_market_regime, "market_regime", "Market Regime"),
|
||||||
(compute_regime_monitor, "regime_monitor", "Regime Monitor"),
|
(compute_regime_monitor, "regime_monitor", "Regime Monitor"),
|
||||||
@@ -1192,7 +1769,62 @@ def configure_scheduler(schedule_config: dict[str, str] | None = None) -> None:
|
|||||||
scheduler.add_job(
|
scheduler.add_job(
|
||||||
run_daily_pipeline,
|
run_daily_pipeline,
|
||||||
_cron_trigger(cfg["schedule_daily_pipeline_cron"], tz, "schedule_daily_pipeline_cron"),
|
_cron_trigger(cfg["schedule_daily_pipeline_cron"], tz, "schedule_daily_pipeline_cron"),
|
||||||
id="daily_pipeline", name="Daily Pipeline", replace_existing=True,
|
id="daily_pipeline", name="Morning Pipeline", replace_existing=True,
|
||||||
|
)
|
||||||
|
scheduler.add_job(
|
||||||
|
run_dolt_earnings_import,
|
||||||
|
_cron_trigger(
|
||||||
|
cfg["schedule_dolt_earnings_cron"],
|
||||||
|
tz,
|
||||||
|
"schedule_dolt_earnings_cron",
|
||||||
|
),
|
||||||
|
id="dolt_earnings_import",
|
||||||
|
name="Dolt Earnings Import (shadow)",
|
||||||
|
replace_existing=True,
|
||||||
|
)
|
||||||
|
scheduler.add_job(
|
||||||
|
run_sec_fundamentals_import,
|
||||||
|
_cron_trigger(
|
||||||
|
cfg["schedule_sec_fundamentals_cron"],
|
||||||
|
tz,
|
||||||
|
"schedule_sec_fundamentals_cron",
|
||||||
|
),
|
||||||
|
id="sec_fundamentals_import",
|
||||||
|
name="SEC Fundamentals Import",
|
||||||
|
replace_existing=True,
|
||||||
|
)
|
||||||
|
scheduler.add_job(
|
||||||
|
run_fundamentals_parity_report,
|
||||||
|
_cron_trigger(
|
||||||
|
cfg["schedule_fundamentals_parity_cron"],
|
||||||
|
tz,
|
||||||
|
"schedule_fundamentals_parity_cron",
|
||||||
|
),
|
||||||
|
id="fundamentals_parity_report",
|
||||||
|
name="Fundamentals Parity Report (read-only)",
|
||||||
|
replace_existing=True,
|
||||||
|
)
|
||||||
|
scheduler.add_job(
|
||||||
|
run_near_close_pipeline,
|
||||||
|
_cron_trigger(
|
||||||
|
cfg["schedule_near_close_pipeline_cron"],
|
||||||
|
tz,
|
||||||
|
"schedule_near_close_pipeline_cron",
|
||||||
|
),
|
||||||
|
id="near_close_pipeline",
|
||||||
|
name="Near-Close Pipeline (scan+alert)",
|
||||||
|
replace_existing=True,
|
||||||
|
)
|
||||||
|
scheduler.add_job(
|
||||||
|
run_after_close_pipeline,
|
||||||
|
_cron_trigger(
|
||||||
|
cfg["schedule_after_close_pipeline_cron"],
|
||||||
|
tz,
|
||||||
|
"schedule_after_close_pipeline_cron",
|
||||||
|
),
|
||||||
|
id="after_close_pipeline",
|
||||||
|
name="After-Close Pipeline (outcome)",
|
||||||
|
replace_existing=True,
|
||||||
)
|
)
|
||||||
scheduler.add_job(
|
scheduler.add_job(
|
||||||
run_intraday_pipeline,
|
run_intraday_pipeline,
|
||||||
@@ -1212,10 +1844,12 @@ def configure_scheduler(schedule_config: dict[str, str] | None = None) -> None:
|
|||||||
sync_ticker_universe, "interval", hours=24,
|
sync_ticker_universe, "interval", hours=24,
|
||||||
id="ticker_universe_sync", name="Ticker Universe Sync", replace_existing=True,
|
id="ticker_universe_sync", name="Ticker Universe Sync", replace_existing=True,
|
||||||
)
|
)
|
||||||
alerts_interval = _parse_frequency(settings.alerts_frequency)
|
# Alerts auto-fire only via near_close_pipeline (scan → alert before MOC).
|
||||||
|
# Keep the job registered for Admin manual trigger; no independent interval.
|
||||||
scheduler.add_job(
|
scheduler.add_job(
|
||||||
dispatch_alerts_job, "interval", **alerts_interval,
|
dispatch_alerts_job, "interval", weeks=520,
|
||||||
id="alerts", name="Alerts Dispatcher", replace_existing=True,
|
id="alerts", name="Alerts Dispatcher",
|
||||||
|
replace_existing=True, next_run_time=None,
|
||||||
)
|
)
|
||||||
scheduler.add_job(
|
scheduler.add_job(
|
||||||
run_backtest_job, "interval", hours=168,
|
run_backtest_job, "interval", hours=168,
|
||||||
@@ -1235,10 +1869,32 @@ def configure_scheduler(schedule_config: dict[str, str] | None = None) -> None:
|
|||||||
replace_existing=True, next_run_time=None,
|
replace_existing=True, next_run_time=None,
|
||||||
)
|
)
|
||||||
|
|
||||||
_log_event(logging.INFO, "scheduler_configured", timezone=tz, daily_pipeline={
|
_log_event(
|
||||||
"cron": cfg["schedule_daily_pipeline_cron"],
|
logging.INFO,
|
||||||
"steps": [name for name, _ in _DAILY_PIPELINE_STEPS],
|
"scheduler_configured",
|
||||||
}, intraday_pipeline={
|
timezone=tz,
|
||||||
"cron": cfg["schedule_intraday_pipeline_cron"],
|
daily_pipeline={
|
||||||
"steps": [name for name, _ in _INTRADAY_PIPELINE_STEPS],
|
"cron": cfg["schedule_daily_pipeline_cron"],
|
||||||
}, fundamental_collector={"cron": cfg["schedule_fundamentals_cron"]}, independent=["ticker_universe_sync", "alerts", "backtest"])
|
"steps": [name for name, _ in _DAILY_PIPELINE_STEPS],
|
||||||
|
},
|
||||||
|
dolt_earnings_import={"cron": cfg["schedule_dolt_earnings_cron"]},
|
||||||
|
sec_fundamentals_import={"cron": cfg["schedule_sec_fundamentals_cron"]},
|
||||||
|
fundamentals_parity_report={
|
||||||
|
"cron": cfg["schedule_fundamentals_parity_cron"]
|
||||||
|
},
|
||||||
|
near_close_pipeline={
|
||||||
|
"cron": cfg["schedule_near_close_pipeline_cron"],
|
||||||
|
"steps": [name for name, _ in _NEAR_CLOSE_PIPELINE_STEPS],
|
||||||
|
},
|
||||||
|
after_close_pipeline={
|
||||||
|
"cron": cfg["schedule_after_close_pipeline_cron"],
|
||||||
|
"steps": [name for name, _ in _AFTER_CLOSE_PIPELINE_STEPS],
|
||||||
|
},
|
||||||
|
intraday_pipeline={
|
||||||
|
"cron": cfg["schedule_intraday_pipeline_cron"],
|
||||||
|
"steps": [name for name, _ in _INTRADAY_PIPELINE_STEPS],
|
||||||
|
},
|
||||||
|
fundamental_collector={"cron": cfg["schedule_fundamentals_cron"]},
|
||||||
|
independent=["ticker_universe_sync", "backtest"],
|
||||||
|
manual_only=["alerts", "data_backfill", "event_study"],
|
||||||
|
)
|
||||||
|
|||||||
+38
-1
@@ -43,6 +43,12 @@ class JobToggle(BaseModel):
|
|||||||
enabled: bool
|
enabled: bool
|
||||||
|
|
||||||
|
|
||||||
|
class JobTriggerRequest(BaseModel):
|
||||||
|
"""Optional parameters for a one-time manual job run."""
|
||||||
|
target_model: Literal["production_gtl", "structural_sr"] | None = None
|
||||||
|
cadence: Literal["weekly", "daily"] | None = None
|
||||||
|
|
||||||
|
|
||||||
class RecommendationConfigUpdate(BaseModel):
|
class RecommendationConfigUpdate(BaseModel):
|
||||||
high_confidence_threshold: float | None = Field(default=None, ge=0, le=100)
|
high_confidence_threshold: float | None = Field(default=None, ge=0, le=100)
|
||||||
moderate_confidence_threshold: float | None = Field(default=None, ge=0, le=100)
|
moderate_confidence_threshold: float | None = Field(default=None, ge=0, le=100)
|
||||||
@@ -64,17 +70,47 @@ class ActivationConfigUpdate(BaseModel):
|
|||||||
min_confidence: float | None = Field(default=None, ge=0, le=100)
|
min_confidence: float | None = Field(default=None, ge=0, le=100)
|
||||||
require_high_conviction: bool | None = None
|
require_high_conviction: bool | None = None
|
||||||
exclude_conflicts: bool | None = None
|
exclude_conflicts: bool | None = None
|
||||||
|
exclude_neutral: bool | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class FundamentalsCutoverConfigUpdate(BaseModel):
|
||||||
|
"""Switch the legacy fundamentals cache from quota APIs to SEC/Dolt."""
|
||||||
|
enabled: bool
|
||||||
|
|
||||||
|
|
||||||
class ScheduleConfigUpdate(BaseModel):
|
class ScheduleConfigUpdate(BaseModel):
|
||||||
"""Cron schedule for the pipelines + fundamentals. Crons are 5-field
|
"""Cron schedule for the pipelines + fundamentals. Crons are 5-field
|
||||||
(min hour dom month dow); timezone is an IANA name (e.g. Europe/Berlin)."""
|
(min hour dom month dow); timezone is an IANA name (e.g. America/New_York)."""
|
||||||
schedule_timezone: str | None = Field(default=None, max_length=64)
|
schedule_timezone: str | None = Field(default=None, max_length=64)
|
||||||
schedule_daily_pipeline_cron: str | None = Field(default=None, max_length=120)
|
schedule_daily_pipeline_cron: str | None = Field(default=None, max_length=120)
|
||||||
|
schedule_dolt_earnings_cron: str | None = Field(default=None, max_length=120)
|
||||||
|
schedule_sec_fundamentals_cron: str | None = Field(default=None, max_length=120)
|
||||||
|
schedule_fundamentals_parity_cron: str | None = Field(default=None, max_length=120)
|
||||||
|
schedule_near_close_pipeline_cron: str | None = Field(default=None, max_length=120)
|
||||||
|
schedule_after_close_pipeline_cron: str | None = Field(default=None, max_length=120)
|
||||||
schedule_intraday_pipeline_cron: str | None = Field(default=None, max_length=120)
|
schedule_intraday_pipeline_cron: str | None = Field(default=None, max_length=120)
|
||||||
schedule_fundamentals_cron: str | None = Field(default=None, max_length=120)
|
schedule_fundamentals_cron: str | None = Field(default=None, max_length=120)
|
||||||
|
|
||||||
|
|
||||||
|
class PerformanceConfigUpdate(BaseModel):
|
||||||
|
"""Window for the Performance comparison.
|
||||||
|
|
||||||
|
``start_date`` is an ISO date, or empty string to show all history. The
|
||||||
|
strategy has been revised repeatedly; pinning a start keeps the shadow-vs-
|
||||||
|
manual comparison inside one configuration instead of averaging across
|
||||||
|
rules that no longer exist.
|
||||||
|
"""
|
||||||
|
start_date: str | None = Field(default=None, max_length=10)
|
||||||
|
|
||||||
|
|
||||||
|
class ShadowBookConfigUpdate(BaseModel):
|
||||||
|
"""Auto-traded shadow book: the validated strategy with no human input."""
|
||||||
|
enabled: bool | None = None
|
||||||
|
capacity: int | None = Field(default=None, ge=1, le=100)
|
||||||
|
risk_pct: float | None = Field(default=None, gt=0, le=10)
|
||||||
|
start_equity: float | None = Field(default=None, ge=1000)
|
||||||
|
|
||||||
|
|
||||||
class SentimentConfigUpdate(BaseModel):
|
class SentimentConfigUpdate(BaseModel):
|
||||||
"""Runtime sentiment LLM config. api_key is write-only; omit/empty to keep
|
"""Runtime sentiment LLM config. api_key is write-only; omit/empty to keep
|
||||||
the stored key."""
|
the stored key."""
|
||||||
@@ -99,3 +135,4 @@ class AlertConfigUpdate(BaseModel):
|
|||||||
score_drop_enabled: bool | None = None
|
score_drop_enabled: bool | None = None
|
||||||
digest_enabled: bool | None = None
|
digest_enabled: bool | None = None
|
||||||
regime_quadrant_enabled: bool | None = None
|
regime_quadrant_enabled: bool | None = None
|
||||||
|
trade_closed_enabled: bool | None = None
|
||||||
|
|||||||
@@ -7,8 +7,75 @@ from datetime import date, datetime
|
|||||||
from pydantic import BaseModel
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
|
||||||
|
class MetricIndustry(BaseModel):
|
||||||
|
label: str
|
||||||
|
median: float
|
||||||
|
favorable_percentile: int # 0-100, polarity-aware (higher = more favorable)
|
||||||
|
peer_count: int
|
||||||
|
|
||||||
|
|
||||||
|
class MetricHistoryPoint(BaseModel):
|
||||||
|
period_end: str # YYYY-MM-DD
|
||||||
|
value: float | None
|
||||||
|
|
||||||
|
|
||||||
|
class MetricItem(BaseModel):
|
||||||
|
key: str
|
||||||
|
value: float | None = None
|
||||||
|
history: list[MetricHistoryPoint] = []
|
||||||
|
industry: MetricIndustry | None = None
|
||||||
|
period_end: str | None = None
|
||||||
|
filed_date: str | None = None
|
||||||
|
caveat: str | None = None
|
||||||
|
source: str = "sec"
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsNext(BaseModel):
|
||||||
|
date: str
|
||||||
|
session: str
|
||||||
|
days_until: int
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsRecent(BaseModel):
|
||||||
|
announce_date: str
|
||||||
|
period_end: str | None = None
|
||||||
|
eps_estimate: float | None = None
|
||||||
|
eps_actual: float | None = None
|
||||||
|
surprise_pct: float | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsObject(BaseModel):
|
||||||
|
next: EarningsNext | None = None
|
||||||
|
recent: list[EarningsRecent] = []
|
||||||
|
|
||||||
|
|
||||||
|
class Valuation(BaseModel):
|
||||||
|
pe: float | None = None
|
||||||
|
fcf_yield: float | None = None
|
||||||
|
market_cap_est: float | None = None
|
||||||
|
pe_industry: MetricIndustry | None = None
|
||||||
|
fcf_yield_industry: MetricIndustry | None = None
|
||||||
|
price_date: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class FundamentalsReads(BaseModel):
|
||||||
|
"""Deterministic text outputs, separate from the numeric metrics.
|
||||||
|
|
||||||
|
``by_key`` is a fixed map over every metric key plus ``pe`` and ``fcf_yield``,
|
||||||
|
each a read string or null. ``header`` is null when there is no read at all."""
|
||||||
|
|
||||||
|
header: str | None = None
|
||||||
|
by_key: dict[str, str | None] = {}
|
||||||
|
|
||||||
|
|
||||||
class FundamentalResponse(BaseModel):
|
class FundamentalResponse(BaseModel):
|
||||||
"""Envelope-ready fundamental data response."""
|
"""Envelope-ready fundamental data response.
|
||||||
|
|
||||||
|
Legacy fields are preserved unchanged (they come from ``fundamental_data`` /
|
||||||
|
the legacy providers). The additive v1 objects — earnings, metrics, valuation,
|
||||||
|
reads — are SEC/Dolt-derived and independent; a null legacy field is never
|
||||||
|
mapped onto the new SEC metrics and vice-versa.
|
||||||
|
"""
|
||||||
|
|
||||||
symbol: str
|
symbol: str
|
||||||
pe_ratio: float | None = None
|
pe_ratio: float | None = None
|
||||||
@@ -18,3 +85,12 @@ class FundamentalResponse(BaseModel):
|
|||||||
next_earnings_date: date | None = None
|
next_earnings_date: date | None = None
|
||||||
fetched_at: datetime | None = None
|
fetched_at: datetime | None = None
|
||||||
unavailable_fields: dict[str, str] = {}
|
unavailable_fields: dict[str, str] = {}
|
||||||
|
|
||||||
|
# --- additive v1 (always present; empty/null when unavailable) ---
|
||||||
|
earnings: EarningsObject | None = None
|
||||||
|
metrics: list[MetricItem] | None = None
|
||||||
|
valuation: Valuation | None = None
|
||||||
|
reads: FundamentalsReads | None = None
|
||||||
|
setup_eligible: bool = True
|
||||||
|
setup_block_code: str | None = None
|
||||||
|
setup_block_reason: str | None = None
|
||||||
|
|||||||
@@ -20,6 +20,14 @@ class PaperTradeClose(BaseModel):
|
|||||||
close_price: float | None = Field(default=None, gt=0)
|
close_price: float | None = Field(default=None, gt=0)
|
||||||
|
|
||||||
|
|
||||||
|
class ExitPolicyUpdate(BaseModel):
|
||||||
|
"""Auto-exit policy for open paper trades."""
|
||||||
|
mode: str | None = Field(default=None, pattern=r"^(time|trailing|atr_trailing|target)$")
|
||||||
|
trailing_pct: float | None = Field(default=None, ge=0.5, le=90)
|
||||||
|
atr_multiplier: float | None = Field(default=None, ge=0.5, le=10)
|
||||||
|
hold_days: int | None = Field(default=None, ge=2, le=250)
|
||||||
|
|
||||||
|
|
||||||
class PaperTradeResponse(BaseModel):
|
class PaperTradeResponse(BaseModel):
|
||||||
id: int
|
id: int
|
||||||
symbol: str
|
symbol: str
|
||||||
@@ -33,3 +41,19 @@ class PaperTradeResponse(BaseModel):
|
|||||||
close_price: float | None = None
|
close_price: float | None = None
|
||||||
closed_at: datetime | None = None
|
closed_at: datetime | None = None
|
||||||
current_price: float | None = None
|
current_price: float | None = None
|
||||||
|
# Alpha vs the S&P 500 (SPY) over the trade's holding period. None when the
|
||||||
|
# benchmark series doesn't cover the trade's open date yet.
|
||||||
|
benchmark_return_pct: float | None = None
|
||||||
|
alpha_pct: float | None = None
|
||||||
|
alpha_usd: float | None = None
|
||||||
|
close_reason: str | None = None
|
||||||
|
# Execution era: null = pre-cutover / unknown; "near_close" = post schedule cutover.
|
||||||
|
fill_mode: str | None = None
|
||||||
|
# Live trailing-stop level + how far price sits above it (% ), for open trades
|
||||||
|
# when the trailing exit policy is active.
|
||||||
|
trailing_stop: float | None = None
|
||||||
|
trailing_distance_pct: float | None = None
|
||||||
|
# Trading sessions represented by post-entry OHLCV bars. These are populated
|
||||||
|
# only while the active exit policy has a max-hold rule.
|
||||||
|
sessions_held: int | None = None
|
||||||
|
sessions_remaining: int | None = None
|
||||||
|
|||||||
@@ -33,6 +33,12 @@ class CompositeBreakdownResponse(BaseModel):
|
|||||||
missing_dimensions: list[str]
|
missing_dimensions: list[str]
|
||||||
renormalized_weights: dict[str, float]
|
renormalized_weights: dict[str, float]
|
||||||
formula: str
|
formula: str
|
||||||
|
# Sentiment is applied as a signed adjustment on top of the non-sentiment base
|
||||||
|
# rather than averaged in.
|
||||||
|
base_score: float | None = None
|
||||||
|
sentiment_score: float | None = None
|
||||||
|
sentiment_adjustment: float | None = None
|
||||||
|
max_sentiment_adjustment: float | None = None
|
||||||
|
|
||||||
|
|
||||||
class DimensionScoreResponse(BaseModel):
|
class DimensionScoreResponse(BaseModel):
|
||||||
@@ -72,6 +78,7 @@ class RankingEntry(BaseModel):
|
|||||||
|
|
||||||
symbol: str
|
symbol: str
|
||||||
composite_score: float
|
composite_score: float
|
||||||
|
composite_stale: bool = False
|
||||||
dimensions: list[DimensionScoreResponse] = []
|
dimensions: list[DimensionScoreResponse] = []
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+23
-1
@@ -15,7 +15,9 @@ class SRLevelResult(BaseModel):
|
|||||||
price_level: float
|
price_level: float
|
||||||
type: Literal["support", "resistance"]
|
type: Literal["support", "resistance"]
|
||||||
strength: int = Field(ge=0, le=100)
|
strength: int = Field(ge=0, le=100)
|
||||||
detection_method: Literal["volume_profile", "pivot_point", "merged"]
|
detection_method: Literal[
|
||||||
|
"volume_profile", "pivot_point", "merged", "round_number"
|
||||||
|
]
|
||||||
created_at: datetime
|
created_at: datetime
|
||||||
|
|
||||||
|
|
||||||
@@ -38,3 +40,23 @@ class SRLevelResponse(BaseModel):
|
|||||||
zones: list[SRZoneResult] = []
|
zones: list[SRZoneResult] = []
|
||||||
visible_levels: list[SRLevelResult] = []
|
visible_levels: list[SRLevelResult] = []
|
||||||
count: int
|
count: int
|
||||||
|
|
||||||
|
|
||||||
|
class GateTargetLevelResult(BaseModel):
|
||||||
|
"""A transient Gate Target Ladder proposal for diagnostic display."""
|
||||||
|
|
||||||
|
price_level: float
|
||||||
|
type: Literal["support", "resistance"]
|
||||||
|
strength: int = Field(ge=0, le=100)
|
||||||
|
detection_method: str
|
||||||
|
sources: list[str] = Field(default_factory=list)
|
||||||
|
traffic_count: int = Field(ge=0)
|
||||||
|
|
||||||
|
|
||||||
|
class GateTargetLadderResponse(BaseModel):
|
||||||
|
"""Volume-free Gate Target Ladder computed from current OHLCV history."""
|
||||||
|
|
||||||
|
symbol: str
|
||||||
|
levels: list[GateTargetLevelResult]
|
||||||
|
count: int
|
||||||
|
lookback_bars: int
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ class TickerCreate(BaseModel):
|
|||||||
class TickerResponse(BaseModel):
|
class TickerResponse(BaseModel):
|
||||||
id: int
|
id: int
|
||||||
symbol: str
|
symbol: str
|
||||||
|
name: str | None = None
|
||||||
created_at: datetime
|
created_at: datetime
|
||||||
|
|
||||||
model_config = {"from_attributes": True}
|
model_config = {"from_attributes": True}
|
||||||
|
|||||||
@@ -26,6 +26,14 @@ class RecommendationSummaryResponse(BaseModel):
|
|||||||
composite_score: float
|
composite_score: float
|
||||||
|
|
||||||
|
|
||||||
|
class TradeSetupContextAsOfResponse(BaseModel):
|
||||||
|
setup_detected_at: datetime
|
||||||
|
score_computed_at: datetime | None = None
|
||||||
|
sentiment_at: datetime | None = None
|
||||||
|
price_date: date | None = None
|
||||||
|
price_updated_at: datetime | None = None
|
||||||
|
|
||||||
|
|
||||||
class TradeSetupResponse(BaseModel):
|
class TradeSetupResponse(BaseModel):
|
||||||
"""A single trade setup detected by the R:R scanner."""
|
"""A single trade setup detected by the R:R scanner."""
|
||||||
|
|
||||||
@@ -49,4 +57,8 @@ class TradeSetupResponse(BaseModel):
|
|||||||
evaluated_at: datetime | None = None
|
evaluated_at: datetime | None = None
|
||||||
current_price: float | None = None
|
current_price: float | None = None
|
||||||
momentum_percentile: float | None = None
|
momentum_percentile: float | None = None
|
||||||
|
strategy_rank: float | None = None
|
||||||
|
volatility_percentile: float | None = None
|
||||||
|
reentry_gate_reset_required: bool = False
|
||||||
|
context_as_of: TradeSetupContextAsOfResponse | None = None
|
||||||
recommendation_summary: RecommendationSummaryResponse | None = None
|
recommendation_summary: RecommendationSummaryResponse | None = None
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ class WatchlistEntryResponse(BaseModel):
|
|||||||
dimensions: list[DimensionScoreSummary] = []
|
dimensions: list[DimensionScoreSummary] = []
|
||||||
rr_ratio: float | None = None
|
rr_ratio: float | None = None
|
||||||
rr_direction: str | None = None
|
rr_direction: str | None = None
|
||||||
|
momentum_percentile: float | None = None
|
||||||
sr_levels: list[SRLevelSummary] = []
|
sr_levels: list[SRLevelSummary] = []
|
||||||
last_close: float | None = None
|
last_close: float | None = None
|
||||||
change_pct: float | None = None
|
change_pct: float | None = None
|
||||||
|
|||||||
+198
-15
@@ -17,7 +17,7 @@ from app.models.settings import SystemSetting
|
|||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
from app.models.trade_setup import TradeSetup
|
from app.models.trade_setup import TradeSetup
|
||||||
from app.models.user import User
|
from app.models.user import User
|
||||||
from app.services import settings_store
|
from app.services import fundamental_data_refresh_service, settings_store
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -39,10 +39,9 @@ SUPPORTED_TICKER_UNIVERSES = {"sp500", "nasdaq100", "nasdaq_all"}
|
|||||||
# Track Record's qualified stats. The outcome evaluator deliberately ignores
|
# Track Record's qualified stats. The outcome evaluator deliberately ignores
|
||||||
# these — every setup is evaluated so the gate itself can be validated.
|
# these — every setup is evaluated so the gate itself can be validated.
|
||||||
#
|
#
|
||||||
# The core test is expected value (in R): probability-weighted asymmetry, so a
|
# The core selection is residual cross-sectional 12-1 momentum (top percentile
|
||||||
# fat-but-improbable target and a likely-but-thin one are both rejected. R:R and
|
# of the universe, long-only). R:R and confidence are floors; high-conviction /
|
||||||
# confidence are floors; high-conviction / clean-read / target-probability are
|
# clean-read are optional tighteners (off by default).
|
||||||
# optional tighteners (off by default — turn on to be more selective).
|
|
||||||
_ACTIVATION_FLOAT_KEYS: dict[str, str] = {
|
_ACTIVATION_FLOAT_KEYS: dict[str, str] = {
|
||||||
"min_momentum_percentile": "activation_min_momentum_percentile",
|
"min_momentum_percentile": "activation_min_momentum_percentile",
|
||||||
"min_rr": "activation_min_rr",
|
"min_rr": "activation_min_rr",
|
||||||
@@ -51,13 +50,22 @@ _ACTIVATION_FLOAT_KEYS: dict[str, str] = {
|
|||||||
_ACTIVATION_BOOL_KEYS: dict[str, str] = {
|
_ACTIVATION_BOOL_KEYS: dict[str, str] = {
|
||||||
"require_high_conviction": "activation_require_high_conviction",
|
"require_high_conviction": "activation_require_high_conviction",
|
||||||
"exclude_conflicts": "activation_exclude_conflicts",
|
"exclude_conflicts": "activation_exclude_conflicts",
|
||||||
|
"exclude_neutral": "activation_exclude_neutral",
|
||||||
}
|
}
|
||||||
ACTIVATION_DEFAULTS: dict[str, float | bool] = {
|
ACTIVATION_DEFAULTS: dict[str, float | bool] = {
|
||||||
"min_momentum_percentile": 80.0,
|
"min_momentum_percentile": 80.0,
|
||||||
"min_rr": 1.2,
|
# Production floor from the 2026-07-12 min_rr sweep (in-sample and OOS peak).
|
||||||
"min_confidence": 55.0,
|
# 1.2 was the old code default and the trough next to the spike — do not restore.
|
||||||
|
"min_rr": 2.0,
|
||||||
|
# 0 = off. The July 2026 gate ablation showed the confidence floor added
|
||||||
|
# nothing (identical net/trade with it removed, under both exit models)
|
||||||
|
# while cutting ~25% of qualified trades.
|
||||||
|
"min_confidence": 0.0,
|
||||||
"require_high_conviction": False,
|
"require_high_conviction": False,
|
||||||
"exclude_conflicts": False,
|
"exclude_conflicts": False,
|
||||||
|
# On by default: a NEUTRAL ("no clear setup") recommendation isn't an
|
||||||
|
# actionable signal, so it shouldn't qualify or be crowned a top pick.
|
||||||
|
"exclude_neutral": True,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -151,6 +159,28 @@ async def update_setting(db: AsyncSession, key: str, value: str) -> SystemSettin
|
|||||||
return setting
|
return setting
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Fundamentals source cutover
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
async def get_fundamentals_cutover_config(db: AsyncSession) -> dict[str, bool]:
|
||||||
|
"""Return the explicit A5 cache-cutover switch (default off)."""
|
||||||
|
return {"enabled": await fundamental_data_refresh_service.is_enabled(db)}
|
||||||
|
|
||||||
|
|
||||||
|
async def update_fundamentals_cutover_config(
|
||||||
|
db: AsyncSession, enabled: bool
|
||||||
|
) -> dict[str, bool]:
|
||||||
|
"""Activate or pause SEC/Dolt writes to the legacy fundamentals cache."""
|
||||||
|
await settings_store.upsert_setting(
|
||||||
|
db,
|
||||||
|
fundamental_data_refresh_service.ACTIVATION_KEY,
|
||||||
|
"true" if enabled else "false",
|
||||||
|
)
|
||||||
|
await db.commit()
|
||||||
|
return await get_fundamentals_cutover_config(db)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Activation thresholds
|
# Activation thresholds
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -196,6 +226,61 @@ async def update_activation_config(
|
|||||||
return await get_activation_config(db)
|
return await get_activation_config(db)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Performance window + shadow book
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
async def get_performance_config(db: AsyncSession) -> dict:
|
||||||
|
"""Start date for the Performance comparison ('' = all history)."""
|
||||||
|
from app.services.paper_trade_service import KEY_PERFORMANCE_START
|
||||||
|
|
||||||
|
return {"start_date": await settings_store.get_value(db, KEY_PERFORMANCE_START, "") or ""}
|
||||||
|
|
||||||
|
|
||||||
|
async def update_performance_config(db: AsyncSession, updates: dict) -> dict:
|
||||||
|
"""Set (or clear) the performance start date. Empty string means all history."""
|
||||||
|
from datetime import date as _date
|
||||||
|
|
||||||
|
from app.services.paper_trade_service import KEY_PERFORMANCE_START
|
||||||
|
|
||||||
|
if "start_date" in updates:
|
||||||
|
raw = (updates.get("start_date") or "").strip()
|
||||||
|
if raw:
|
||||||
|
try:
|
||||||
|
_date.fromisoformat(raw)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise ValidationError("start_date must be an ISO date (YYYY-MM-DD)") from exc
|
||||||
|
await update_setting(db, KEY_PERFORMANCE_START, raw)
|
||||||
|
return await get_performance_config(db)
|
||||||
|
|
||||||
|
|
||||||
|
async def get_shadow_book_config(db: AsyncSession) -> dict:
|
||||||
|
"""Shadow book switch + sizing, with the validated defaults filled in."""
|
||||||
|
from app.services import shadow_book_service
|
||||||
|
|
||||||
|
config = await shadow_book_service.get_config(db)
|
||||||
|
config["enabled"] = await shadow_book_service.is_enabled(db)
|
||||||
|
return config
|
||||||
|
|
||||||
|
|
||||||
|
async def update_shadow_book_config(db: AsyncSession, updates: dict) -> dict:
|
||||||
|
"""Update the shadow book. Enabling it starts automatic live entries."""
|
||||||
|
from app.services import shadow_book_service
|
||||||
|
|
||||||
|
if "enabled" in updates:
|
||||||
|
await update_setting(
|
||||||
|
db, shadow_book_service.KEY_ENABLED, "true" if updates["enabled"] else "false"
|
||||||
|
)
|
||||||
|
for key, storage_key in (
|
||||||
|
("capacity", shadow_book_service.KEY_CAPACITY),
|
||||||
|
("risk_pct", shadow_book_service.KEY_RISK_PCT),
|
||||||
|
("start_equity", shadow_book_service.KEY_START_EQUITY),
|
||||||
|
):
|
||||||
|
if key in updates:
|
||||||
|
await update_setting(db, storage_key, str(updates[key]))
|
||||||
|
return await get_shadow_book_config(db)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Pipeline schedule (cron)
|
# Pipeline schedule (cron)
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -310,14 +395,18 @@ async def update_ticker_universe_default(db: AsyncSession, universe: str) -> dic
|
|||||||
# Data cleanup
|
# Data cleanup
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
async def cleanup_data(db: AsyncSession, older_than_days: int) -> dict[str, int]:
|
async def cleanup_data(db: AsyncSession, older_than_days: int) -> dict:
|
||||||
"""Delete OHLCV, sentiment, and fundamental records older than N days.
|
"""Delete OHLCV, sentiment, and fundamental records older than N days.
|
||||||
|
|
||||||
Preserves tickers, users, and latest scores.
|
Preserves tickers, users, and latest scores. After OHLCV pruning, rebuilds
|
||||||
Returns a dict with counts of deleted records per table.
|
Structural S/R for every ticker so chart levels match the remaining history.
|
||||||
|
|
||||||
|
Returns deleted-row counts plus S/R refresh outcomes. A per-ticker S/R
|
||||||
|
failure rolls the session back (so later tickers still run) and is listed
|
||||||
|
in ``sr_refresh_failures`` rather than aborting the whole cleanup.
|
||||||
"""
|
"""
|
||||||
cutoff = datetime.now(timezone.utc) - timedelta(days=older_than_days)
|
cutoff = datetime.now(timezone.utc) - timedelta(days=older_than_days)
|
||||||
counts: dict[str, int] = {}
|
counts: dict = {}
|
||||||
|
|
||||||
# OHLCV — date column is a date, compare with cutoff date
|
# OHLCV — date column is a date, compare with cutoff date
|
||||||
result = await db.execute(
|
result = await db.execute(
|
||||||
@@ -338,6 +427,36 @@ async def cleanup_data(db: AsyncSession, older_than_days: int) -> dict[str, int]
|
|||||||
counts["fundamentals"] = result.rowcount # type: ignore[assignment]
|
counts["fundamentals"] = result.rowcount # type: ignore[assignment]
|
||||||
|
|
||||||
await db.commit()
|
await db.commit()
|
||||||
|
|
||||||
|
counts["sr_refresh_ok"] = 0
|
||||||
|
counts["sr_refresh_failed"] = 0
|
||||||
|
counts["sr_refresh_failures"] = []
|
||||||
|
|
||||||
|
# Structural S/R is derived from OHLCV; recompute after history shrinks.
|
||||||
|
if counts["ohlcv"]:
|
||||||
|
from app.services.sr_service import recalculate_sr_levels
|
||||||
|
|
||||||
|
symbols = list(
|
||||||
|
(await db.execute(select(Ticker.symbol).order_by(Ticker.symbol))).scalars().all()
|
||||||
|
)
|
||||||
|
for symbol in symbols:
|
||||||
|
try:
|
||||||
|
await recalculate_sr_levels(db, symbol)
|
||||||
|
counts["sr_refresh_ok"] += 1
|
||||||
|
except Exception as exc:
|
||||||
|
logger.exception("S/R refresh after cleanup failed for %s", symbol)
|
||||||
|
try:
|
||||||
|
await db.rollback()
|
||||||
|
except Exception:
|
||||||
|
logger.exception(
|
||||||
|
"Session rollback after S/R cleanup failure also failed for %s",
|
||||||
|
symbol,
|
||||||
|
)
|
||||||
|
counts["sr_refresh_failed"] += 1
|
||||||
|
counts["sr_refresh_failures"].append(
|
||||||
|
{"symbol": symbol, "error": f"{type(exc).__name__}: {exc}"}
|
||||||
|
)
|
||||||
|
|
||||||
return counts
|
return counts
|
||||||
|
|
||||||
|
|
||||||
@@ -512,8 +631,12 @@ async def get_pipeline_readiness(db: AsyncSession) -> list[dict]:
|
|||||||
VALID_JOB_NAMES = {
|
VALID_JOB_NAMES = {
|
||||||
"data_collector",
|
"data_collector",
|
||||||
"data_backfill",
|
"data_backfill",
|
||||||
|
"benchmark_collector",
|
||||||
"sentiment_collector",
|
"sentiment_collector",
|
||||||
"fundamental_collector",
|
"fundamental_collector",
|
||||||
|
"dolt_earnings_import",
|
||||||
|
"sec_fundamentals_import",
|
||||||
|
"fundamentals_parity_report",
|
||||||
"rr_scanner",
|
"rr_scanner",
|
||||||
"ticker_universe_sync",
|
"ticker_universe_sync",
|
||||||
"outcome_evaluator",
|
"outcome_evaluator",
|
||||||
@@ -523,14 +646,21 @@ VALID_JOB_NAMES = {
|
|||||||
"event_study",
|
"event_study",
|
||||||
"backtest",
|
"backtest",
|
||||||
"daily_pipeline",
|
"daily_pipeline",
|
||||||
|
"near_close_pipeline",
|
||||||
|
"after_close_pipeline",
|
||||||
"intraday_pipeline",
|
"intraday_pipeline",
|
||||||
|
"shadow_book",
|
||||||
}
|
}
|
||||||
|
|
||||||
JOB_LABELS = {
|
JOB_LABELS = {
|
||||||
"data_collector": "Data Collector (OHLCV)",
|
"data_collector": "Data Collector (OHLCV)",
|
||||||
"data_backfill": "Data Backfill (deep history)",
|
"data_backfill": "Data Backfill (deep history)",
|
||||||
|
"benchmark_collector": "Benchmark Collector",
|
||||||
"sentiment_collector": "Sentiment Collector",
|
"sentiment_collector": "Sentiment Collector",
|
||||||
"fundamental_collector": "Fundamental Collector",
|
"fundamental_collector": "Fundamental Collector",
|
||||||
|
"dolt_earnings_import": "Dolt Earnings Import (shadow)",
|
||||||
|
"sec_fundamentals_import": "SEC Fundamentals Import",
|
||||||
|
"fundamentals_parity_report": "Fundamentals Parity Report (read-only)",
|
||||||
"rr_scanner": "R:R Scanner",
|
"rr_scanner": "R:R Scanner",
|
||||||
"ticker_universe_sync": "Ticker Universe Sync",
|
"ticker_universe_sync": "Ticker Universe Sync",
|
||||||
"outcome_evaluator": "Outcome Evaluator",
|
"outcome_evaluator": "Outcome Evaluator",
|
||||||
@@ -539,18 +669,24 @@ JOB_LABELS = {
|
|||||||
"regime_monitor": "Regime Monitor",
|
"regime_monitor": "Regime Monitor",
|
||||||
"event_study": "Event Study",
|
"event_study": "Event Study",
|
||||||
"backtest": "Backtest",
|
"backtest": "Backtest",
|
||||||
"daily_pipeline": "Daily Pipeline",
|
"daily_pipeline": "Morning Pipeline",
|
||||||
|
"near_close_pipeline": "Near-Close Pipeline (scan+alert)",
|
||||||
|
"after_close_pipeline": "After-Close Pipeline (outcome)",
|
||||||
"intraday_pipeline": "Intraday Pipeline",
|
"intraday_pipeline": "Intraday Pipeline",
|
||||||
|
"shadow_book": "Shadow Book (auto-traded strategy)",
|
||||||
}
|
}
|
||||||
|
|
||||||
# Jobs driven by the daily_pipeline (in order) rather than their own timer.
|
# Jobs driven by a pipeline (in order) rather than their own auto timer.
|
||||||
PIPELINE_MEMBERS = {
|
PIPELINE_MEMBERS = {
|
||||||
"data_collector",
|
"data_collector",
|
||||||
|
"benchmark_collector",
|
||||||
"sentiment_collector",
|
"sentiment_collector",
|
||||||
"rr_scanner",
|
"rr_scanner",
|
||||||
"outcome_evaluator",
|
"outcome_evaluator",
|
||||||
|
"alerts",
|
||||||
"market_regime",
|
"market_regime",
|
||||||
"regime_monitor",
|
"regime_monitor",
|
||||||
|
"shadow_book",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -593,13 +729,23 @@ async def list_jobs(db: AsyncSession) -> list[dict]:
|
|||||||
return jobs_out
|
return jobs_out
|
||||||
|
|
||||||
|
|
||||||
async def trigger_job(db: AsyncSession, job_name: str) -> dict[str, str]:
|
async def trigger_job(
|
||||||
|
db: AsyncSession,
|
||||||
|
job_name: str,
|
||||||
|
*,
|
||||||
|
target_model: str | None = None,
|
||||||
|
cadence: str | None = None,
|
||||||
|
) -> dict[str, str]:
|
||||||
"""Trigger a manual job run via the scheduler.
|
"""Trigger a manual job run via the scheduler.
|
||||||
|
|
||||||
Runs the job immediately (in addition to its regular schedule).
|
Runs the job immediately (in addition to its regular schedule).
|
||||||
"""
|
"""
|
||||||
if job_name not in VALID_JOB_NAMES:
|
if job_name not in VALID_JOB_NAMES:
|
||||||
raise ValidationError(f"Unknown job: {job_name}. Valid jobs: {', '.join(sorted(VALID_JOB_NAMES))}")
|
raise ValidationError(f"Unknown job: {job_name}. Valid jobs: {', '.join(sorted(VALID_JOB_NAMES))}")
|
||||||
|
if target_model is not None and job_name != "backtest":
|
||||||
|
raise ValidationError("target_model is supported only for the backtest job")
|
||||||
|
if cadence is not None and job_name != "backtest":
|
||||||
|
raise ValidationError("cadence is supported only for the backtest job")
|
||||||
|
|
||||||
from app.scheduler import get_job_runtime_snapshot, scheduler
|
from app.scheduler import get_job_runtime_snapshot, scheduler
|
||||||
|
|
||||||
@@ -626,11 +772,21 @@ async def trigger_job(db: AsyncSession, job_name: str) -> dict[str, str]:
|
|||||||
if job is None:
|
if job is None:
|
||||||
return {"job": job_name, "status": "not_found", "message": f"Job '{job_name}' is not registered in the scheduler"}
|
return {"job": job_name, "status": "not_found", "message": f"Job '{job_name}' is not registered in the scheduler"}
|
||||||
|
|
||||||
|
if job_name == "backtest":
|
||||||
|
from app.scheduler import queue_backtest_options
|
||||||
|
|
||||||
|
target_model, cadence = queue_backtest_options(target_model, cadence)
|
||||||
|
|
||||||
job.modify(next_run_time=None) # Reset, then trigger immediately
|
job.modify(next_run_time=None) # Reset, then trigger immediately
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
job.modify(next_run_time=datetime.now(timezone.utc))
|
job.modify(next_run_time=datetime.now(timezone.utc))
|
||||||
|
|
||||||
return {"job": job_name, "status": "triggered", "message": f"Job '{job_name}' triggered for immediate execution"}
|
result = {"job": job_name, "status": "triggered", "message": f"Job '{job_name}' triggered for immediate execution"}
|
||||||
|
if target_model is not None:
|
||||||
|
result["target_model"] = target_model
|
||||||
|
if cadence is not None:
|
||||||
|
result["cadence"] = cadence
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
async def toggle_job(db: AsyncSession, job_name: str, enabled: bool) -> SystemSetting:
|
async def toggle_job(db: AsyncSession, job_name: str, enabled: bool) -> SystemSetting:
|
||||||
@@ -643,3 +799,30 @@ async def toggle_job(db: AsyncSession, job_name: str, enabled: bool) -> SystemSe
|
|||||||
|
|
||||||
key = f"job_{job_name}_enabled"
|
key = f"job_{job_name}_enabled"
|
||||||
return await update_setting(db, key, str(enabled).lower())
|
return await update_setting(db, key, str(enabled).lower())
|
||||||
|
|
||||||
|
|
||||||
|
def get_fundamentals_parity_report() -> dict | None:
|
||||||
|
"""Return the latest compact A5 summary, if the job has run."""
|
||||||
|
from app.config import settings
|
||||||
|
from app.services.fundamentals_parity_service import load_latest
|
||||||
|
|
||||||
|
report = load_latest(settings.fundamentals_parity_report_dir)
|
||||||
|
if report is not None:
|
||||||
|
report.pop("rows", None) # full per-ticker data is download-only
|
||||||
|
return report
|
||||||
|
|
||||||
|
|
||||||
|
def get_fundamentals_parity_csv() -> tuple[str, str] | None:
|
||||||
|
"""Return the latest A5 CSV filename and content for authenticated download."""
|
||||||
|
from app.config import settings
|
||||||
|
from app.services.fundamentals_parity_service import load_latest_csv
|
||||||
|
|
||||||
|
return load_latest_csv(settings.fundamentals_parity_report_dir)
|
||||||
|
|
||||||
|
|
||||||
|
def get_fundamentals_parity_json() -> tuple[str, str] | None:
|
||||||
|
"""Return the canonical A5 JSON artifact for authenticated download."""
|
||||||
|
from app.config import settings
|
||||||
|
from app.services.fundamentals_parity_service import load_latest_json
|
||||||
|
|
||||||
|
return load_latest_json(settings.fundamentals_parity_report_dir)
|
||||||
|
|||||||
+586
-86
@@ -16,16 +16,20 @@ precedence DB > env; the bot token is write-only (never returned on read).
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
import math
|
||||||
|
from collections import defaultdict
|
||||||
from datetime import datetime, timedelta, timezone
|
from datetime import datetime, timedelta, timezone
|
||||||
from types import SimpleNamespace
|
from types import SimpleNamespace
|
||||||
|
|
||||||
import httpx
|
import httpx
|
||||||
from sqlalchemy import select
|
from sqlalchemy import func, select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
from app.config import settings
|
from app.config import settings
|
||||||
from app.models.alert import AlertLog
|
from app.models.alert import AlertLog
|
||||||
from app.models.ohlcv import OHLCVRecord
|
from app.models.ohlcv import OHLCVRecord
|
||||||
|
from app.models.paper_trade import PaperTrade
|
||||||
|
from app.services.trade_policy import MANUAL_BOOK
|
||||||
from app.models.score import CompositeScore
|
from app.models.score import CompositeScore
|
||||||
from app.models.sr_level import SRLevel
|
from app.models.sr_level import SRLevel
|
||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
@@ -47,6 +51,7 @@ KEY_SR = "alerts_sr_proximity_enabled"
|
|||||||
KEY_SCORE_DROP = "alerts_score_drop_enabled"
|
KEY_SCORE_DROP = "alerts_score_drop_enabled"
|
||||||
KEY_DIGEST = "alerts_digest_enabled"
|
KEY_DIGEST = "alerts_digest_enabled"
|
||||||
KEY_REGIME_QUADRANT = "alerts_regime_quadrant_enabled"
|
KEY_REGIME_QUADRANT = "alerts_regime_quadrant_enabled"
|
||||||
|
KEY_TRADE_CLOSED = "alerts_trade_closed_enabled"
|
||||||
|
|
||||||
_BOOL_DEFAULTS = {
|
_BOOL_DEFAULTS = {
|
||||||
KEY_ENABLED: False,
|
KEY_ENABLED: False,
|
||||||
@@ -54,9 +59,23 @@ _BOOL_DEFAULTS = {
|
|||||||
KEY_SR: True,
|
KEY_SR: True,
|
||||||
KEY_SCORE_DROP: True,
|
KEY_SCORE_DROP: True,
|
||||||
KEY_DIGEST: True,
|
KEY_DIGEST: True,
|
||||||
KEY_REGIME_QUADRANT: True,
|
# Experimental human-facing thermometer: opt in explicitly. Existing stored
|
||||||
|
# true values remain true; only missing/reset configurations default off.
|
||||||
|
KEY_REGIME_QUADRANT: False,
|
||||||
|
KEY_TRADE_CLOSED: True,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Paper-trade auto-close alert: catch every close at least once (the job runs
|
||||||
|
# hourly), then never re-send the same trade (a huge cooldown ≈ once-per-trade).
|
||||||
|
CLOSED_LOOKBACK_HOURS = 26
|
||||||
|
CLOSED_ALERT_COOLDOWN_HOURS = 24 * 365 * 5
|
||||||
|
TRADE_CLOSED_TYPE = "trade_closed"
|
||||||
|
PAPER_BOOK_STARTING_CAPITAL = 10_000.0
|
||||||
|
QUALIFIED_TARGET_ZONE_PCT = 1.0
|
||||||
|
QUALIFIED_STATE_TYPE = "qualified_state"
|
||||||
|
QUALIFIED_ACTIVE = 1.0
|
||||||
|
QUALIFIED_INACTIVE = 0.0
|
||||||
|
|
||||||
# Tunables (kept as constants for now; promote to settings if needed)
|
# Tunables (kept as constants for now; promote to settings if needed)
|
||||||
SR_PROXIMITY_PCT = 2.0 # within this % of a strong zone → alert
|
SR_PROXIMITY_PCT = 2.0 # within this % of a strong zone → alert
|
||||||
SR_MIN_STRENGTH = 60 # only strong zones are alert-worthy
|
SR_MIN_STRENGTH = 60 # only strong zones are alert-worthy
|
||||||
@@ -66,22 +85,33 @@ COOLDOWN_HOURS = 72 # don't re-send the same key within this window
|
|||||||
DIGEST_HOUR_UTC = 22 # send the daily digest on the first run at/after this hour
|
DIGEST_HOUR_UTC = 22 # send the daily digest on the first run at/after this hour
|
||||||
|
|
||||||
WATERMARK_TYPE = "score_watermark"
|
WATERMARK_TYPE = "score_watermark"
|
||||||
|
SIGNAL_BUNDLE_ALERT_TYPES = ("qualified", "sr_proximity", "score_drop")
|
||||||
|
SIGNAL_BUNDLE_SECTIONS = (
|
||||||
|
("qualified", "Qualified setups"),
|
||||||
|
("sr_proximity", "Near support/resistance"),
|
||||||
|
("score_drop", "Score drops"),
|
||||||
|
)
|
||||||
|
SIGNAL_BUNDLE_MAX_CHARS = 3900 # Telegram limit is 4096; keep room for HTML parsing
|
||||||
|
|
||||||
# Regime quadrant-change alert: (regime index x early-warning) quadrant.
|
# Regime quadrant-change alert: (State x Warning) quadrant.
|
||||||
# Hysteresis (a deadband around each divider) stops a point sitting on a boundary
|
# Hysteresis (a deadband around each divider) stops a point sitting on a boundary
|
||||||
# from flip-flopping; the cooldown caps how often a genuine change can re-alert.
|
# from flip-flopping; the cooldown caps how often a genuine change can re-alert.
|
||||||
QUAD_TYPE = "regime_quadrant"
|
QUAD_TYPE = "regime_quadrant"
|
||||||
QUAD_X_DIV = 40.0 # regime index divider (matches the frontend quadrant)
|
QUAD_X_DIV = 50.0 # v3 State divider (backend response is authoritative)
|
||||||
QUAD_Y_DIV = 60.0 # early-warning divider
|
QUAD_Y_DIV = 40.0 # v3 Warning divider; the axes have different ranges
|
||||||
QUAD_MARGIN = 5.0 # half-width of the hysteresis deadband around each divider
|
QUAD_MARGIN = 5.0 # half-width of the hysteresis deadband around each divider
|
||||||
QUAD_COOLDOWN_DAYS = 3 # min days between quadrant-change alerts
|
QUAD_COOLDOWN_DAYS = 3 # min days between quadrant-change alerts
|
||||||
QUAD_LABELS = {
|
QUAD_LABELS = {
|
||||||
"1": "① Hot & brittle",
|
"1": "Early warning",
|
||||||
"2": "② Transition",
|
"2": "Active stress",
|
||||||
"3": "③ Healthy & broad",
|
"3": "Healthy",
|
||||||
"4": "④ Real downturn",
|
"4": "Stressed / stabilizing",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
AlertItem = tuple[str, str, str] # alert_type, dedup_key, text
|
||||||
|
AlertLogRef = tuple[str, str] # alert_type, dedup_key
|
||||||
|
ClosedTradeItem = tuple[str, str, float] # dedup_key, text, pnl_usd
|
||||||
|
|
||||||
|
|
||||||
def _as_bool(value: str | None, default: bool) -> bool:
|
def _as_bool(value: str | None, default: bool) -> bool:
|
||||||
if value is None:
|
if value is None:
|
||||||
@@ -90,7 +120,7 @@ def _as_bool(value: str | None, default: bool) -> bool:
|
|||||||
|
|
||||||
|
|
||||||
async def _resolve(db: AsyncSession) -> dict:
|
async def _resolve(db: AsyncSession) -> dict:
|
||||||
keys = [KEY_ENABLED, KEY_TOKEN, KEY_CHAT_ID, KEY_QUALIFIED, KEY_SR, KEY_SCORE_DROP, KEY_DIGEST, KEY_REGIME_QUADRANT]
|
keys = [KEY_ENABLED, KEY_TOKEN, KEY_CHAT_ID, KEY_QUALIFIED, KEY_SR, KEY_SCORE_DROP, KEY_DIGEST, KEY_REGIME_QUADRANT, KEY_TRADE_CLOSED]
|
||||||
stored = await settings_store.get_map(db, keys)
|
stored = await settings_store.get_map(db, keys)
|
||||||
|
|
||||||
db_token = (stored.get(KEY_TOKEN) or "").strip()
|
db_token = (stored.get(KEY_TOKEN) or "").strip()
|
||||||
@@ -113,6 +143,7 @@ async def _resolve(db: AsyncSession) -> dict:
|
|||||||
"score_drop": _as_bool(stored.get(KEY_SCORE_DROP), _BOOL_DEFAULTS[KEY_SCORE_DROP]),
|
"score_drop": _as_bool(stored.get(KEY_SCORE_DROP), _BOOL_DEFAULTS[KEY_SCORE_DROP]),
|
||||||
"digest": _as_bool(stored.get(KEY_DIGEST), _BOOL_DEFAULTS[KEY_DIGEST]),
|
"digest": _as_bool(stored.get(KEY_DIGEST), _BOOL_DEFAULTS[KEY_DIGEST]),
|
||||||
"regime_quadrant": _as_bool(stored.get(KEY_REGIME_QUADRANT), _BOOL_DEFAULTS[KEY_REGIME_QUADRANT]),
|
"regime_quadrant": _as_bool(stored.get(KEY_REGIME_QUADRANT), _BOOL_DEFAULTS[KEY_REGIME_QUADRANT]),
|
||||||
|
"trade_closed": _as_bool(stored.get(KEY_TRADE_CLOSED), _BOOL_DEFAULTS[KEY_TRADE_CLOSED]),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -129,6 +160,7 @@ async def get_alert_config(db: AsyncSession) -> dict:
|
|||||||
"score_drop_enabled": r["score_drop"],
|
"score_drop_enabled": r["score_drop"],
|
||||||
"digest_enabled": r["digest"],
|
"digest_enabled": r["digest"],
|
||||||
"regime_quadrant_enabled": r["regime_quadrant"],
|
"regime_quadrant_enabled": r["regime_quadrant"],
|
||||||
|
"trade_closed_enabled": r["trade_closed"],
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -143,6 +175,7 @@ async def update_alert_config(
|
|||||||
score_drop_enabled: bool | None = None,
|
score_drop_enabled: bool | None = None,
|
||||||
digest_enabled: bool | None = None,
|
digest_enabled: bool | None = None,
|
||||||
regime_quadrant_enabled: bool | None = None,
|
regime_quadrant_enabled: bool | None = None,
|
||||||
|
trade_closed_enabled: bool | None = None,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Persist config. An empty/omitted bot_token leaves the stored token intact."""
|
"""Persist config. An empty/omitted bot_token leaves the stored token intact."""
|
||||||
bool_updates = {
|
bool_updates = {
|
||||||
@@ -152,6 +185,7 @@ async def update_alert_config(
|
|||||||
KEY_SCORE_DROP: score_drop_enabled,
|
KEY_SCORE_DROP: score_drop_enabled,
|
||||||
KEY_DIGEST: digest_enabled,
|
KEY_DIGEST: digest_enabled,
|
||||||
KEY_REGIME_QUADRANT: regime_quadrant_enabled,
|
KEY_REGIME_QUADRANT: regime_quadrant_enabled,
|
||||||
|
KEY_TRADE_CLOSED: trade_closed_enabled,
|
||||||
}
|
}
|
||||||
for key, val in bool_updates.items():
|
for key, val in bool_updates.items():
|
||||||
if val is not None:
|
if val is not None:
|
||||||
@@ -203,6 +237,19 @@ async def _recently_alerted(
|
|||||||
return result.first() is not None
|
return result.first() is not None
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_qualified_states(db: AsyncSession) -> dict[str, bool]:
|
||||||
|
"""Latest active/inactive state per qualified setup opportunity."""
|
||||||
|
result = await db.execute(
|
||||||
|
select(AlertLog.dedup_key, AlertLog.value)
|
||||||
|
.where(AlertLog.alert_type == QUALIFIED_STATE_TYPE)
|
||||||
|
.order_by(AlertLog.created_at.asc(), AlertLog.id.asc())
|
||||||
|
)
|
||||||
|
states: dict[str, bool] = {}
|
||||||
|
for key, value in result.all():
|
||||||
|
states[key] = bool(value and value > 0)
|
||||||
|
return states
|
||||||
|
|
||||||
|
|
||||||
def _log_alert(db: AsyncSession, alert_type: str, key: str, value: float | None = None) -> None:
|
def _log_alert(db: AsyncSession, alert_type: str, key: str, value: float | None = None) -> None:
|
||||||
db.add(
|
db.add(
|
||||||
AlertLog(
|
AlertLog(
|
||||||
@@ -214,17 +261,6 @@ def _log_alert(db: AsyncSession, alert_type: str, key: str, value: float | None
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
async def _watermark(db: AsyncSession, symbol: str) -> float | None:
|
|
||||||
result = await db.execute(
|
|
||||||
select(AlertLog.value)
|
|
||||||
.where(AlertLog.alert_type == WATERMARK_TYPE, AlertLog.dedup_key == symbol)
|
|
||||||
.order_by(AlertLog.created_at.desc())
|
|
||||||
.limit(1)
|
|
||||||
)
|
|
||||||
row = result.first()
|
|
||||||
return row[0] if row else None
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Trigger collectors
|
# Trigger collectors
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -241,26 +277,79 @@ async def _watchlist_tickers(db: AsyncSession) -> list[tuple[int, str]]:
|
|||||||
|
|
||||||
|
|
||||||
async def _qualified_setups(db: AsyncSession) -> list[dict]:
|
async def _qualified_setups(db: AsyncSession) -> list[dict]:
|
||||||
setups = await get_trade_setups(db)
|
# live_recommendation: gate and format on current score/sentiment context,
|
||||||
|
# not the values frozen into the setup at scan time.
|
||||||
|
setups = await get_trade_setups(
|
||||||
|
db,
|
||||||
|
live_recommendation=True,
|
||||||
|
exclude_open_trade_tickers=True,
|
||||||
|
exclude_reentry_gate_locked_tickers=True,
|
||||||
|
)
|
||||||
config = await get_activation_config(db)
|
config = await get_activation_config(db)
|
||||||
return [s for s in setups if setup_qualifies(SimpleNamespace(**s), config)]
|
return [s for s in setups if setup_qualifies(SimpleNamespace(**s), config)]
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_price(value: float | int | None) -> str:
|
||||||
|
return "n/a" if value is None else f"{float(value):.2f}"
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_money(value: float | int | None) -> str:
|
||||||
|
if value is None:
|
||||||
|
return "n/a"
|
||||||
|
return f"${float(value):,.2f}"
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_signed_money(value: float | int | None) -> str:
|
||||||
|
if value is None:
|
||||||
|
return "n/a"
|
||||||
|
amount = float(value)
|
||||||
|
return f"{'+' if amount >= 0 else '-'}${abs(amount):,.2f}"
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_signed_pct(value: float | int | None) -> str:
|
||||||
|
if value is None:
|
||||||
|
return "n/a"
|
||||||
|
return f"{float(value):+.1f}%"
|
||||||
|
|
||||||
|
|
||||||
|
def _fmt_signed_move(from_price: float | int | None, to_price: float | int | None) -> str:
|
||||||
|
if from_price is None or to_price is None:
|
||||||
|
return "n/a"
|
||||||
|
from_float = float(from_price)
|
||||||
|
if from_float == 0:
|
||||||
|
return "n/a"
|
||||||
|
pct = (float(to_price) - from_float) / from_float * 100.0
|
||||||
|
return f"{pct:+.1f}%"
|
||||||
|
|
||||||
|
|
||||||
|
def _qualified_opportunity_key(s: dict) -> str:
|
||||||
|
"""Stable key for one alert per trade opportunity, not per scanner row."""
|
||||||
|
target = s.get("target")
|
||||||
|
if target is None or float(target) <= 0:
|
||||||
|
zone = "unknown"
|
||||||
|
else:
|
||||||
|
step = math.log1p(QUALIFIED_TARGET_ZONE_PCT / 100.0)
|
||||||
|
zone = str(round(math.log(float(target)) / step))
|
||||||
|
return f"qualified:{s['symbol']}:{s['direction']}:target-zone:{zone}"
|
||||||
|
|
||||||
|
|
||||||
def _format_qualified(s: dict) -> str:
|
def _format_qualified(s: dict) -> str:
|
||||||
prob = best_target_probability(SimpleNamespace(**s))
|
prob = best_target_probability(SimpleNamespace(**s))
|
||||||
arrow = "🟢" if s["direction"] == "long" else "🔴"
|
arrow = "🟢" if s["direction"] == "long" else "🔴"
|
||||||
|
current = s.get("current_price") or s.get("entry_price")
|
||||||
return (
|
return (
|
||||||
f"{arrow} <b>{s['symbol']} {s['direction'].upper()}</b> — qualified setup\n"
|
f"{arrow} <b>{s['symbol']} {s['direction'].upper()}</b> | "
|
||||||
f"entry {s['entry_price']:.2f} → target {s['target']:.2f} "
|
f"now {_fmt_price(current)} | entry {_fmt_price(s['entry_price'])} | "
|
||||||
f"(R:R {s['rr_ratio']:.1f}:1)\n"
|
f"target {_fmt_price(s['target'])} ({_fmt_signed_move(current, s['target'])}) | "
|
||||||
f"confidence {(s.get('confidence_score') or 0):.0f}% · P(target) {prob:.0f}%"
|
f"stop {_fmt_price(s['stop_loss'])} | R:R {s['rr_ratio']:.1f} | "
|
||||||
|
f"conf {(s.get('confidence_score') or 0):.0f}% | P(target) {prob:.0f}%"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
async def _collect_qualified(db: AsyncSession) -> list[tuple[str, str]]:
|
async def _collect_qualified(db: AsyncSession) -> list[tuple[str, str]]:
|
||||||
out: list[tuple[str, str]] = []
|
out: list[tuple[str, str]] = []
|
||||||
for s in await _qualified_setups(db):
|
for s in await _qualified_setups(db):
|
||||||
key = f"qualified:{s['symbol']}:{s['direction']}"
|
key = _qualified_opportunity_key(s)
|
||||||
out.append((key, _format_qualified(s)))
|
out.append((key, _format_qualified(s)))
|
||||||
return out
|
return out
|
||||||
|
|
||||||
@@ -276,6 +365,34 @@ async def _latest_close(db: AsyncSession, ticker_id: int) -> float | None:
|
|||||||
return float(row[0]) if row else None
|
return float(row[0]) if row else None
|
||||||
|
|
||||||
|
|
||||||
|
def _sr_zone_label(zone: dict) -> str:
|
||||||
|
return (
|
||||||
|
f"{zone['low']:.2f}–{zone['high']:.2f}"
|
||||||
|
if zone["level_count"] > 1
|
||||||
|
else f"{zone['midpoint']:.2f}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _sr_touch_price(zone: dict, current_price: float) -> float:
|
||||||
|
low = float(zone["low"])
|
||||||
|
high = float(zone["high"])
|
||||||
|
if current_price < low:
|
||||||
|
return low
|
||||||
|
if current_price > high:
|
||||||
|
return high
|
||||||
|
return float(zone["midpoint"])
|
||||||
|
|
||||||
|
|
||||||
|
def _format_sr_proximity(symbol: str, zone: dict, current_price: float) -> str:
|
||||||
|
touch_price = _sr_touch_price(zone, current_price)
|
||||||
|
return (
|
||||||
|
f"📍 <b>{symbol}</b> {zone['type']} | "
|
||||||
|
f"now {_fmt_price(current_price)} -> {_sr_zone_label(zone)} "
|
||||||
|
f"({_fmt_signed_move(current_price, touch_price)}) | "
|
||||||
|
f"strength {float(zone['strength']):.0f}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
async def _collect_sr_proximity(db: AsyncSession) -> list[tuple[str, str]]:
|
async def _collect_sr_proximity(db: AsyncSession) -> list[tuple[str, str]]:
|
||||||
"""One alert per watchlist ticker for the NEAREST strong S/R zone within range.
|
"""One alert per watchlist ticker for the NEAREST strong S/R zone within range.
|
||||||
|
|
||||||
@@ -284,17 +401,49 @@ async def _collect_sr_proximity(db: AsyncSession) -> list[tuple[str, str]]:
|
|||||||
single alert. Scoped to the watchlist only — qualified tickers already get
|
single alert. Scoped to the watchlist only — qualified tickers already get
|
||||||
their own 'qualified setup' alert, so S/R on them would be redundant.
|
their own 'qualified setup' alert, so S/R on them would be redundant.
|
||||||
"""
|
"""
|
||||||
|
watchlist = await _watchlist_tickers(db)
|
||||||
|
if not watchlist:
|
||||||
|
return []
|
||||||
|
|
||||||
|
ticker_ids = [ticker_id for ticker_id, _ in watchlist]
|
||||||
|
latest_dates = (
|
||||||
|
select(
|
||||||
|
OHLCVRecord.ticker_id,
|
||||||
|
func.max(OHLCVRecord.date).label("latest_date"),
|
||||||
|
)
|
||||||
|
.where(OHLCVRecord.ticker_id.in_(ticker_ids))
|
||||||
|
.group_by(OHLCVRecord.ticker_id)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
prices_result = await db.execute(
|
||||||
|
select(OHLCVRecord.ticker_id, OHLCVRecord.close).join(
|
||||||
|
latest_dates,
|
||||||
|
(OHLCVRecord.ticker_id == latest_dates.c.ticker_id)
|
||||||
|
& (OHLCVRecord.date == latest_dates.c.latest_date),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
prices = {ticker_id: float(close) for ticker_id, close in prices_result.all()}
|
||||||
|
|
||||||
|
levels_result = await db.execute(
|
||||||
|
select(SRLevel).where(SRLevel.ticker_id.in_(ticker_ids))
|
||||||
|
)
|
||||||
|
levels_by_ticker: dict[int, list[dict]] = defaultdict(list)
|
||||||
|
for level in levels_result.scalars():
|
||||||
|
levels_by_ticker[level.ticker_id].append(
|
||||||
|
{
|
||||||
|
"price_level": level.price_level,
|
||||||
|
"strength": level.strength,
|
||||||
|
"type": level.type,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
out: list[tuple[str, str]] = []
|
out: list[tuple[str, str]] = []
|
||||||
for tid, symbol in await _watchlist_tickers(db):
|
for tid, symbol in watchlist:
|
||||||
price = await _latest_close(db, tid)
|
price = prices.get(tid)
|
||||||
if not price:
|
if not price:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
levels_result = await db.execute(select(SRLevel).where(SRLevel.ticker_id == tid))
|
levels = levels_by_ticker[tid]
|
||||||
levels = [
|
|
||||||
{"price_level": lv.price_level, "strength": lv.strength, "type": lv.type}
|
|
||||||
for lv in levels_result.scalars().all()
|
|
||||||
]
|
|
||||||
if not levels:
|
if not levels:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
@@ -305,21 +454,12 @@ async def _collect_sr_proximity(db: AsyncSession) -> list[tuple[str, str]]:
|
|||||||
|
|
||||||
# Nearest strong zone only.
|
# Nearest strong zone only.
|
||||||
nearest = min(strong, key=lambda z: abs(price - z["midpoint"]))
|
nearest = min(strong, key=lambda z: abs(price - z["midpoint"]))
|
||||||
dist_pct = abs(price - nearest["midpoint"]) / price * 100
|
dist_pct = abs(price - _sr_touch_price(nearest, price)) / price * 100
|
||||||
if dist_pct > SR_PROXIMITY_PCT:
|
if dist_pct > SR_PROXIMITY_PCT:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
label = (
|
|
||||||
f"{nearest['low']:.2f}–{nearest['high']:.2f}"
|
|
||||||
if nearest["level_count"] > 1
|
|
||||||
else f"{nearest['midpoint']:.2f}"
|
|
||||||
)
|
|
||||||
key = f"sr:{symbol}:{nearest['type']}" # one per side per ticker per cooldown
|
key = f"sr:{symbol}:{nearest['type']}" # one per side per ticker per cooldown
|
||||||
out.append((
|
out.append((key, _format_sr_proximity(symbol, nearest, price)))
|
||||||
key,
|
|
||||||
f"📍 <b>{symbol}</b> approaching {nearest['type']} {label} "
|
|
||||||
f"(now {price:.2f}, {dist_pct:.1f}% away)",
|
|
||||||
))
|
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
@@ -331,17 +471,54 @@ async def _collect_score_drops(db: AsyncSession) -> list[tuple[str, str]]:
|
|||||||
doesn't re-fire; let the watermark rise with the score so the next drop is
|
doesn't re-fire; let the watermark rise with the score so the next drop is
|
||||||
measured from the new high.
|
measured from the new high.
|
||||||
"""
|
"""
|
||||||
out: list[tuple[str, str]] = []
|
watchlist = await _watchlist_tickers(db)
|
||||||
for tid, symbol in await _watchlist_tickers(db):
|
if not watchlist:
|
||||||
comp_result = await db.execute(
|
return []
|
||||||
select(CompositeScore.score).where(CompositeScore.ticker_id == tid)
|
|
||||||
)
|
|
||||||
row = comp_result.first()
|
|
||||||
if row is None or row[0] is None:
|
|
||||||
continue
|
|
||||||
current = float(row[0])
|
|
||||||
|
|
||||||
base = await _watermark(db, symbol)
|
ticker_ids = [ticker_id for ticker_id, _ in watchlist]
|
||||||
|
symbols = [symbol for _, symbol in watchlist]
|
||||||
|
scores_result = await db.execute(
|
||||||
|
select(CompositeScore.ticker_id, CompositeScore.score).where(
|
||||||
|
CompositeScore.ticker_id.in_(ticker_ids)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
scores = {ticker_id: float(score) for ticker_id, score in scores_result.all()}
|
||||||
|
|
||||||
|
ranked_watermarks = (
|
||||||
|
select(
|
||||||
|
AlertLog.dedup_key,
|
||||||
|
AlertLog.value,
|
||||||
|
func.row_number()
|
||||||
|
.over(
|
||||||
|
partition_by=AlertLog.dedup_key,
|
||||||
|
order_by=(AlertLog.created_at.desc(), AlertLog.id.desc()),
|
||||||
|
)
|
||||||
|
.label("rank"),
|
||||||
|
)
|
||||||
|
.where(
|
||||||
|
AlertLog.alert_type == WATERMARK_TYPE,
|
||||||
|
AlertLog.dedup_key.in_(symbols),
|
||||||
|
)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
watermarks_result = await db.execute(
|
||||||
|
select(ranked_watermarks.c.dedup_key, ranked_watermarks.c.value).where(
|
||||||
|
ranked_watermarks.c.rank == 1
|
||||||
|
)
|
||||||
|
)
|
||||||
|
watermarks = {
|
||||||
|
symbol: float(value)
|
||||||
|
for symbol, value in watermarks_result.all()
|
||||||
|
if value is not None
|
||||||
|
}
|
||||||
|
|
||||||
|
out: list[tuple[str, str]] = []
|
||||||
|
for tid, symbol in watchlist:
|
||||||
|
current = scores.get(tid)
|
||||||
|
if current is None:
|
||||||
|
continue
|
||||||
|
|
||||||
|
base = watermarks.get(symbol)
|
||||||
if base is None:
|
if base is None:
|
||||||
_log_alert(db, WATERMARK_TYPE, symbol, value=current) # seed, no alert
|
_log_alert(db, WATERMARK_TYPE, symbol, value=current) # seed, no alert
|
||||||
continue
|
continue
|
||||||
@@ -368,47 +545,227 @@ async def _collect_digest(db: AsyncSession) -> tuple[str, str] | None:
|
|||||||
lines = [f"📊 <b>Daily digest</b> — {now.date().isoformat()}"]
|
lines = [f"📊 <b>Daily digest</b> — {now.date().isoformat()}"]
|
||||||
if qualified:
|
if qualified:
|
||||||
top = sorted(qualified, key=lambda s: s["rr_ratio"], reverse=True)[:5]
|
top = sorted(qualified, key=lambda s: s["rr_ratio"], reverse=True)[:5]
|
||||||
lines.append(f"{len(qualified)} qualified setup(s):")
|
lines.append("")
|
||||||
|
lines.append(f"<b>Qualified setups</b> ({len(qualified)})")
|
||||||
for s in top:
|
for s in top:
|
||||||
lines.append(
|
lines.append(_format_qualified(s))
|
||||||
f"• {s['symbol']} {s['direction'].upper()} "
|
|
||||||
f"R:R {s['rr_ratio']:.1f}:1, conf {(s.get('confidence_score') or 0):.0f}%"
|
|
||||||
)
|
|
||||||
else:
|
else:
|
||||||
lines.append("No qualified setups today.")
|
lines.append("No qualified setups today.")
|
||||||
|
|
||||||
|
# Open paper trades: unrealized gain + the live trailing stop and how far away.
|
||||||
|
from app.services import paper_trade_service
|
||||||
|
|
||||||
|
open_trades = await paper_trade_service.list_trades(db, status="open")
|
||||||
|
if open_trades:
|
||||||
|
lines.append("")
|
||||||
|
lines.append(f"<b>Open trades</b> ({len(open_trades)})")
|
||||||
|
for t in open_trades:
|
||||||
|
entry = t["entry_price"]
|
||||||
|
cur = t.get("current_price")
|
||||||
|
sign = 1.0 if t["direction"] == "long" else -1.0
|
||||||
|
if cur and entry:
|
||||||
|
gain_pct = (cur - entry) / entry * 100.0 * sign
|
||||||
|
gain_usd = (cur - entry) * t["shares"] * sign
|
||||||
|
gain = f"{gain_pct:+.1f}% ({_fmt_signed_money(gain_usd)})"
|
||||||
|
else:
|
||||||
|
gain = "n/a"
|
||||||
|
ts = t.get("trailing_stop")
|
||||||
|
if ts is not None:
|
||||||
|
dist = t.get("trailing_distance_pct")
|
||||||
|
stop_txt = f"trail {ts:.2f}" + (f" ({dist:.1f}% away)" if dist is not None else "")
|
||||||
|
else:
|
||||||
|
stop_txt = f"stop {t['stop_loss']:.2f}"
|
||||||
|
lines.append(
|
||||||
|
f"💼 <b>{t['symbol']} {t['direction'].upper()} open</b> | "
|
||||||
|
f"now {_fmt_price(cur)} | entry {_fmt_price(entry)} | "
|
||||||
|
f"target {_fmt_price(t.get('target'))} ({_fmt_signed_move(cur, t.get('target'))}) | "
|
||||||
|
f"{stop_txt} | P&L {gain}"
|
||||||
|
)
|
||||||
|
|
||||||
return key, "\n".join(lines)
|
return key, "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Paper-trade close trigger (one summary per auto-closed trade)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _closed_trade_pnl(trade: PaperTrade) -> float:
|
||||||
|
sign = 1.0 if trade.direction == "long" else -1.0
|
||||||
|
entry = trade.entry_price
|
||||||
|
exit_price = trade.close_price if trade.close_price is not None else entry
|
||||||
|
per_share = (exit_price - entry) * sign
|
||||||
|
return per_share * trade.shares
|
||||||
|
|
||||||
|
|
||||||
|
def _format_closed_trade(trade: PaperTrade, symbol: str) -> str:
|
||||||
|
sign = 1.0 if trade.direction == "long" else -1.0
|
||||||
|
entry = trade.entry_price
|
||||||
|
exit_price = trade.close_price if trade.close_price is not None else entry
|
||||||
|
per_share = (exit_price - entry) * sign
|
||||||
|
pnl_pct = (per_share / entry * 100.0) if entry else 0.0
|
||||||
|
pnl_usd = _closed_trade_pnl(trade)
|
||||||
|
risk = abs(entry - trade.stop_loss)
|
||||||
|
r_mult = (per_share / risk) if risk > 0 else None
|
||||||
|
win = per_share > 0
|
||||||
|
money = _fmt_signed_money(pnl_usd)
|
||||||
|
r_txt = f" · {r_mult:+.2f}R" if r_mult is not None else ""
|
||||||
|
days = (trade.closed_at - trade.opened_at).days if (trade.closed_at and trade.opened_at) else None
|
||||||
|
held = f" · held {days}d" if days is not None else ""
|
||||||
|
reason = {"trailing": "trailing stop", "stop": "stop-loss", "target": "target", "time": "max hold"}.get(
|
||||||
|
trade.close_reason or "", trade.close_reason or "closed"
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
f"{'✅' if win else '🔴'} <b>{symbol} {trade.direction.upper()} closed</b> ({reason})\n"
|
||||||
|
f"{pnl_pct:+.1f}% · {money}{r_txt}{held}\n"
|
||||||
|
f"{entry:.2f} → {exit_price:.2f}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def _collect_closed_trades(db: AsyncSession) -> list[ClosedTradeItem]:
|
||||||
|
"""One alert item per auto-closed paper trade. Manual
|
||||||
|
closes are skipped — you already know about those. Dedup is by trade id."""
|
||||||
|
cutoff = datetime.now(timezone.utc) - timedelta(hours=CLOSED_LOOKBACK_HOURS)
|
||||||
|
result = await db.execute(
|
||||||
|
select(PaperTrade, Ticker.symbol)
|
||||||
|
.join(Ticker, PaperTrade.ticker_id == Ticker.id)
|
||||||
|
.where(
|
||||||
|
PaperTrade.status == "closed",
|
||||||
|
PaperTrade.closed_at.is_not(None),
|
||||||
|
PaperTrade.closed_at > cutoff,
|
||||||
|
PaperTrade.close_reason.in_(("trailing", "stop", "target", "time")),
|
||||||
|
# Your own positions only — shadow trades are a research record, not
|
||||||
|
# something you hold, and mixing them in unlabelled reads as if you
|
||||||
|
# were stopped out of a name you never took.
|
||||||
|
PaperTrade.book == MANUAL_BOOK,
|
||||||
|
)
|
||||||
|
.order_by(PaperTrade.closed_at.desc())
|
||||||
|
)
|
||||||
|
return [
|
||||||
|
(str(trade.id), _format_closed_trade(trade, symbol), _closed_trade_pnl(trade))
|
||||||
|
for trade, symbol in result.all()
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
async def _paper_book_value(db: AsyncSession) -> float:
|
||||||
|
"""Paper-trade equity: fixed capital plus realized/unrealized P&L.
|
||||||
|
|
||||||
|
Discretionary book only — the shadow book runs on its own notional equity
|
||||||
|
and folding it in would report a number matching neither book.
|
||||||
|
"""
|
||||||
|
result = await db.execute(
|
||||||
|
select(PaperTrade).where(PaperTrade.book == MANUAL_BOOK)
|
||||||
|
)
|
||||||
|
trades = list(result.scalars().all())
|
||||||
|
latest: dict[int, float | None] = {}
|
||||||
|
for trade in trades:
|
||||||
|
if trade.status == "open" and trade.ticker_id not in latest:
|
||||||
|
latest[trade.ticker_id] = await _latest_close(db, trade.ticker_id)
|
||||||
|
|
||||||
|
total_pnl = 0.0
|
||||||
|
for trade in trades:
|
||||||
|
ref = trade.close_price if trade.status == "closed" else latest.get(trade.ticker_id)
|
||||||
|
if ref is None:
|
||||||
|
ref = trade.entry_price
|
||||||
|
sign = 1.0 if trade.direction == "long" else -1.0
|
||||||
|
pnl = (float(ref) - trade.entry_price) * trade.shares * sign
|
||||||
|
total_pnl += pnl
|
||||||
|
return PAPER_BOOK_STARTING_CAPITAL + total_pnl
|
||||||
|
|
||||||
|
|
||||||
|
def _closed_trade_bundle(
|
||||||
|
items: list[ClosedTradeItem],
|
||||||
|
*,
|
||||||
|
current_book_value: float | None,
|
||||||
|
) -> tuple[list[AlertLogRef], str] | None:
|
||||||
|
if not items:
|
||||||
|
return None
|
||||||
|
total_pnl = sum(item[2] for item in items)
|
||||||
|
previous_book_value = (
|
||||||
|
current_book_value - total_pnl
|
||||||
|
if current_book_value is not None
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
change_pct = (
|
||||||
|
total_pnl / previous_book_value * 100.0
|
||||||
|
if previous_book_value not in (None, 0)
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
lines = [f"💼 <b>Paper trades closed</b> — {len(items)} trade(s)"]
|
||||||
|
if current_book_value is not None and previous_book_value is not None:
|
||||||
|
lines.append(
|
||||||
|
f"Paper book {_fmt_money(previous_book_value)} → {_fmt_money(current_book_value)} "
|
||||||
|
f"({_fmt_signed_money(total_pnl)}, {_fmt_signed_pct(change_pct)})"
|
||||||
|
)
|
||||||
|
lines.append("")
|
||||||
|
lines.append("\n\n".join(item[1] for item in items))
|
||||||
|
return ([(TRADE_CLOSED_TYPE, item[0]) for item in items], "\n".join(lines))
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Regime quadrant-change trigger (hysteresis + cooldown)
|
# Regime quadrant-change trigger (hysteresis + cooldown)
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
def _bools_to_quadrant(x_high: bool, y_high: bool) -> str:
|
def _bools_to_quadrant(x_high: bool, y_high: bool) -> str:
|
||||||
if y_high:
|
if y_high:
|
||||||
return "2" if x_high else "1" # ② Transition / ① Hot & brittle
|
return "2" if x_high else "1" # Active stress / Early warning
|
||||||
return "4" if x_high else "3" # ④ Real downturn / ③ Healthy & broad
|
return "4" if x_high else "3" # Stressed/stabilizing / Healthy
|
||||||
|
|
||||||
|
|
||||||
def _quadrant_to_bools(q: str) -> tuple[bool, bool]:
|
def _quadrant_to_bools(q: str) -> tuple[bool, bool]:
|
||||||
return {"1": (False, True), "2": (True, True), "3": (False, False), "4": (True, False)}[q]
|
return {"1": (False, True), "2": (True, True), "3": (False, False), "4": (True, False)}[q]
|
||||||
|
|
||||||
|
|
||||||
def _classify_quadrant(x: float, y: float, prev: str | None, margin: float = QUAD_MARGIN) -> str:
|
def _classify_quadrant(
|
||||||
"""Quadrant of (regime index x, early warning y), with per-axis hysteresis.
|
x: float,
|
||||||
|
y: float,
|
||||||
|
prev: str | None,
|
||||||
|
margin: float = QUAD_MARGIN,
|
||||||
|
x_div: float = QUAD_X_DIV,
|
||||||
|
y_div: float = QUAD_Y_DIV,
|
||||||
|
) -> str:
|
||||||
|
"""Quadrant of (State x, Warning y), with per-axis hysteresis.
|
||||||
|
|
||||||
Each axis only flips once the value crosses its divider by ``margin`` in the
|
Each axis only flips once the value crosses its divider by ``margin`` in the
|
||||||
new direction, so a point parked on a divider keeps its current quadrant
|
new direction, so a point parked on a divider keeps its current quadrant
|
||||||
instead of flip-flopping. ``prev`` None means a fresh (no-hysteresis) classify.
|
instead of flip-flopping. ``prev`` None means a fresh (no-hysteresis) classify.
|
||||||
"""
|
"""
|
||||||
if prev is None:
|
if prev is None:
|
||||||
return _bools_to_quadrant(x >= QUAD_X_DIV, y >= QUAD_Y_DIV)
|
return _bools_to_quadrant(x >= x_div, y >= y_div)
|
||||||
px, py = _quadrant_to_bools(prev)
|
px, py = _quadrant_to_bools(prev)
|
||||||
x_high = (x >= QUAD_X_DIV - margin) if px else (x >= QUAD_X_DIV + margin)
|
x_high = (x >= x_div - margin) if px else (x >= x_div + margin)
|
||||||
y_high = (y >= QUAD_Y_DIV - margin) if py else (y >= QUAD_Y_DIV + margin)
|
y_high = (y >= y_div - margin) if py else (y >= y_div + margin)
|
||||||
return _bools_to_quadrant(x_high, y_high)
|
return _bools_to_quadrant(x_high, y_high)
|
||||||
|
|
||||||
|
|
||||||
async def _last_quadrant(db: AsyncSession) -> tuple[str | None, datetime | None]:
|
def _quadrant_log_key(q: str, x: float, y: float, basket_hash: str | None = None) -> str:
|
||||||
|
return f"{basket_hash or 'legacy'}:{q}:{x:.1f}:{y:.1f}"
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_quadrant_log_key(
|
||||||
|
key: str | None,
|
||||||
|
) -> tuple[str | None, str | None, float | None, float | None]:
|
||||||
|
if not key:
|
||||||
|
return None, None, None, None
|
||||||
|
parts = key.split(":")
|
||||||
|
if parts[0] in QUAD_LABELS:
|
||||||
|
basket_hash, q, values = None, parts[0], parts[1:]
|
||||||
|
elif len(parts) >= 2:
|
||||||
|
basket_hash, q, values = parts[0], parts[1], parts[2:]
|
||||||
|
else:
|
||||||
|
return None, None, None, None
|
||||||
|
if q not in QUAD_LABELS:
|
||||||
|
return None, None, None, None
|
||||||
|
if len(values) >= 2:
|
||||||
|
try:
|
||||||
|
return basket_hash, q, float(values[0]), float(values[1])
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
return basket_hash, q, None, None
|
||||||
|
|
||||||
|
|
||||||
|
async def _last_quadrant(
|
||||||
|
db: AsyncSession,
|
||||||
|
) -> tuple[str | None, str | None, float | None, float | None, datetime | None]:
|
||||||
"""Most recently logged quadrant (and when), our baseline for change + cooldown."""
|
"""Most recently logged quadrant (and when), our baseline for change + cooldown."""
|
||||||
result = await db.execute(
|
result = await db.execute(
|
||||||
select(AlertLog.dedup_key, AlertLog.created_at)
|
select(AlertLog.dedup_key, AlertLog.created_at)
|
||||||
@@ -417,7 +774,10 @@ async def _last_quadrant(db: AsyncSession) -> tuple[str | None, datetime | None]
|
|||||||
.limit(1)
|
.limit(1)
|
||||||
)
|
)
|
||||||
row = result.first()
|
row = result.first()
|
||||||
return (row[0], row[1]) if row else (None, None)
|
if not row:
|
||||||
|
return None, None, None, None, None
|
||||||
|
basket_hash, prev_q, prev_x, prev_y = _parse_quadrant_log_key(row[0])
|
||||||
|
return basket_hash, prev_q, prev_x, prev_y, row[1]
|
||||||
|
|
||||||
|
|
||||||
async def _collect_regime_quadrant(db: AsyncSession) -> list[tuple[str, str]]:
|
async def _collect_regime_quadrant(db: AsyncSession) -> list[tuple[str, str]]:
|
||||||
@@ -428,43 +788,134 @@ async def _collect_regime_quadrant(db: AsyncSession) -> list[tuple[str, str]]:
|
|||||||
cooldown has elapsed. The dispatch loop logs the new quadrant on send, which
|
cooldown has elapsed. The dispatch loop logs the new quadrant on send, which
|
||||||
becomes the next baseline and resets the cooldown clock.
|
becomes the next baseline and resets the cooldown clock.
|
||||||
"""
|
"""
|
||||||
from app.services.regime_monitor_service import get_regime_monitor
|
from app.services.regime_monitor_service import get_regime_history, get_regime_monitor
|
||||||
|
|
||||||
data = await get_regime_monitor(db)
|
data = await get_regime_monitor(db)
|
||||||
if not data.get("available"):
|
if not data.get("available"):
|
||||||
return []
|
return []
|
||||||
x = data.get("total_score")
|
state = data.get("state") or {}
|
||||||
y = (data.get("early_warning") or {}).get("score")
|
warning = data.get("warning") or {}
|
||||||
|
x = state.get("score")
|
||||||
|
y = warning.get("score")
|
||||||
if x is None or y is None:
|
if x is None or y is None:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
prev, prev_time = await _last_quadrant(db)
|
quality = data.get("data_quality") or {}
|
||||||
if prev is None:
|
if (
|
||||||
_log_alert(db, QUAD_TYPE, _classify_quadrant(x, y, None)) # seed, no alert
|
float(state.get("coverage") or 0) < 75
|
||||||
|
or float(warning.get("coverage") or 0) < 75
|
||||||
|
or not quality.get("is_fresh")
|
||||||
|
):
|
||||||
return []
|
return []
|
||||||
|
|
||||||
new_q = _classify_quadrant(x, y, prev)
|
quadrant_cfg = data.get("quadrant_config") or {}
|
||||||
|
x_div = float(quadrant_cfg.get("state_divider", QUAD_X_DIV))
|
||||||
|
y_div = float(quadrant_cfg.get("warning_divider", QUAD_Y_DIV))
|
||||||
|
margin = float(quadrant_cfg.get("margin", QUAD_MARGIN))
|
||||||
|
basket_hash = str((data.get("basket") or {}).get("hash") or "unknown")
|
||||||
|
|
||||||
|
prev_hash, prev, prev_x, prev_y, prev_time = await _last_quadrant(db)
|
||||||
|
if prev is None or prev_hash != basket_hash:
|
||||||
|
seed = _classify_quadrant(x, y, None, margin, x_div, y_div)
|
||||||
|
_log_alert(db, QUAD_TYPE, _quadrant_log_key(seed, x, y, basket_hash))
|
||||||
|
return []
|
||||||
|
|
||||||
|
new_q = _classify_quadrant(x, y, prev, margin, x_div, y_div)
|
||||||
if new_q == prev:
|
if new_q == prev:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
|
history = await get_regime_history(db, days=14)
|
||||||
|
valid = [
|
||||||
|
point for point in history
|
||||||
|
if point.get("state") is not None
|
||||||
|
and point.get("warning") is not None
|
||||||
|
and float(point.get("state_coverage") or 0) >= 75
|
||||||
|
and float(point.get("warning_coverage") or 0) >= 75
|
||||||
|
]
|
||||||
|
if len(valid) < 2:
|
||||||
|
return []
|
||||||
|
prior = valid[-2]
|
||||||
|
prior_q = _classify_quadrant(
|
||||||
|
float(prior["state"]),
|
||||||
|
float(prior["warning"]),
|
||||||
|
prev,
|
||||||
|
margin,
|
||||||
|
x_div,
|
||||||
|
y_div,
|
||||||
|
)
|
||||||
|
if prior_q != new_q:
|
||||||
|
return []
|
||||||
|
|
||||||
if prev_time is not None:
|
if prev_time is not None:
|
||||||
if prev_time.tzinfo is None:
|
if prev_time.tzinfo is None:
|
||||||
prev_time = prev_time.replace(tzinfo=timezone.utc)
|
prev_time = prev_time.replace(tzinfo=timezone.utc)
|
||||||
if datetime.now(timezone.utc) - prev_time < timedelta(days=QUAD_COOLDOWN_DAYS):
|
if datetime.now(timezone.utc) - prev_time < timedelta(days=QUAD_COOLDOWN_DAYS):
|
||||||
return [] # genuine change, but inside the cooldown — stay quiet
|
return [] # genuine change, but inside the cooldown — stay quiet
|
||||||
|
|
||||||
|
if prev_x is not None and prev_y is not None:
|
||||||
|
metrics = (
|
||||||
|
f"State {prev_x:.0f} → {x:.0f} ({x - prev_x:+.0f}) · "
|
||||||
|
f"Warning {prev_y:.0f} → {y:.0f} ({y - prev_y:+.0f})"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
metrics = f"State {x:.0f} · Warning {y:.0f}"
|
||||||
text = (
|
text = (
|
||||||
f"🧭 <b>Regime quadrant change</b>\n"
|
f"🧭 <b>Regime quadrant change</b>\n"
|
||||||
f"{QUAD_LABELS.get(prev, prev)} → {QUAD_LABELS.get(new_q, new_q)}\n"
|
f"{QUAD_LABELS.get(prev, prev)} → {QUAD_LABELS.get(new_q, new_q)}\n"
|
||||||
f"regime {x:.0f} · early-warning {y:.0f}"
|
f"{metrics}\n"
|
||||||
|
f"coverage: state {state.get('coverage'):.0f}% / warning {warning.get('coverage'):.0f}%\n"
|
||||||
|
f"<i>Risk thermometer - not a trade signal.</i>"
|
||||||
)
|
)
|
||||||
return [(new_q, text)]
|
return [(_quadrant_log_key(new_q, x, y, basket_hash), text)]
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Dispatch
|
# Dispatch
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def _signal_bundle_messages(items: list[AlertItem]) -> list[tuple[list[AlertLogRef], str]]:
|
||||||
|
if not items:
|
||||||
|
return []
|
||||||
|
|
||||||
|
by_type: dict[str, list[AlertItem]] = {key: [] for key in SIGNAL_BUNDLE_ALERT_TYPES}
|
||||||
|
for item in items:
|
||||||
|
by_type.setdefault(item[0], []).append(item)
|
||||||
|
|
||||||
|
total = sum(len(group) for group in by_type.values())
|
||||||
|
header = f"📣 <b>Signal run</b> — {total} new alert(s)"
|
||||||
|
bundles: list[tuple[list[AlertLogRef], str]] = []
|
||||||
|
lines = [header]
|
||||||
|
logs: list[AlertLogRef] = []
|
||||||
|
current_section: str | None = None
|
||||||
|
|
||||||
|
def flush() -> None:
|
||||||
|
nonlocal lines, logs, current_section
|
||||||
|
if logs:
|
||||||
|
bundles.append((logs.copy(), "\n".join(lines)))
|
||||||
|
lines = [f"{header} (continued)"]
|
||||||
|
logs = []
|
||||||
|
current_section = None
|
||||||
|
|
||||||
|
for alert_type, section_title in SIGNAL_BUNDLE_SECTIONS:
|
||||||
|
for item_type, key, text in by_type.get(alert_type, []):
|
||||||
|
block: list[str] = []
|
||||||
|
if current_section != alert_type:
|
||||||
|
block.extend(["", f"<b>{section_title}</b>"])
|
||||||
|
block.append(text)
|
||||||
|
|
||||||
|
if logs and len("\n".join(lines + block)) > SIGNAL_BUNDLE_MAX_CHARS:
|
||||||
|
flush()
|
||||||
|
block = ["", f"<b>{section_title}</b>", text]
|
||||||
|
|
||||||
|
lines.extend(block)
|
||||||
|
logs.append((item_type, key))
|
||||||
|
current_section = alert_type
|
||||||
|
|
||||||
|
if logs:
|
||||||
|
bundles.append((logs.copy(), "\n".join(lines)))
|
||||||
|
return bundles
|
||||||
|
|
||||||
|
|
||||||
async def dispatch_alerts(db: AsyncSession) -> dict:
|
async def dispatch_alerts(db: AsyncSession) -> dict:
|
||||||
"""Gather all enabled triggers, dedup, and push to Telegram. Job entrypoint."""
|
"""Gather all enabled triggers, dedup, and push to Telegram. Job entrypoint."""
|
||||||
cfg = await _resolve(db)
|
cfg = await _resolve(db)
|
||||||
@@ -473,22 +924,33 @@ async def dispatch_alerts(db: AsyncSession) -> dict:
|
|||||||
if not cfg["token"] or not cfg["chat_id"]:
|
if not cfg["token"] or not cfg["chat_id"]:
|
||||||
return {"status": "no_credentials", "sent": 0}
|
return {"status": "no_credentials", "sent": 0}
|
||||||
|
|
||||||
outgoing: list[tuple[str, str, str]] = [] # (alert_type, key, text)
|
signal_outgoing: list[AlertItem] = []
|
||||||
|
outgoing: list[AlertItem] = []
|
||||||
|
closed_outgoing: list[ClosedTradeItem] = []
|
||||||
|
qualified_inactive: list[str] = []
|
||||||
|
|
||||||
if cfg["qualified"]:
|
if cfg["qualified"]:
|
||||||
for key, text in await _collect_qualified(db):
|
previous_qualified_states = await _latest_qualified_states(db)
|
||||||
if not await _recently_alerted(db, "qualified", key):
|
qualified_items = await _collect_qualified(db)
|
||||||
outgoing.append(("qualified", key, text))
|
current_qualified_keys = {key for key, _ in qualified_items}
|
||||||
|
for key, text in qualified_items:
|
||||||
|
if not previous_qualified_states.get(key, False):
|
||||||
|
signal_outgoing.append(("qualified", key, text))
|
||||||
|
qualified_inactive = [
|
||||||
|
key
|
||||||
|
for key, active in previous_qualified_states.items()
|
||||||
|
if active and key not in current_qualified_keys
|
||||||
|
]
|
||||||
|
|
||||||
if cfg["sr"]:
|
if cfg["sr"]:
|
||||||
for key, text in await _collect_sr_proximity(db):
|
for key, text in await _collect_sr_proximity(db):
|
||||||
if not await _recently_alerted(db, "sr_proximity", key):
|
if not await _recently_alerted(db, "sr_proximity", key):
|
||||||
outgoing.append(("sr_proximity", key, text))
|
signal_outgoing.append(("sr_proximity", key, text))
|
||||||
|
|
||||||
if cfg["score_drop"]:
|
if cfg["score_drop"]:
|
||||||
# also seeds/advances watermarks as a side effect
|
# also seeds/advances watermarks as a side effect
|
||||||
for key, text in await _collect_score_drops(db):
|
for key, text in await _collect_score_drops(db):
|
||||||
outgoing.append(("score_drop", key, text))
|
signal_outgoing.append(("score_drop", key, text))
|
||||||
|
|
||||||
if cfg["digest"]:
|
if cfg["digest"]:
|
||||||
digest = await _collect_digest(db)
|
digest = await _collect_digest(db)
|
||||||
@@ -500,9 +962,44 @@ async def dispatch_alerts(db: AsyncSession) -> dict:
|
|||||||
for key, text in await _collect_regime_quadrant(db):
|
for key, text in await _collect_regime_quadrant(db):
|
||||||
outgoing.append((QUAD_TYPE, key, text))
|
outgoing.append((QUAD_TYPE, key, text))
|
||||||
|
|
||||||
|
if cfg["trade_closed"]:
|
||||||
|
for key, text, pnl_usd in await _collect_closed_trades(db):
|
||||||
|
if not await _recently_alerted(db, TRADE_CLOSED_TYPE, key, cooldown_hours=CLOSED_ALERT_COOLDOWN_HOURS):
|
||||||
|
closed_outgoing.append((key, text, pnl_usd))
|
||||||
|
|
||||||
sent = 0
|
sent = 0
|
||||||
if outgoing:
|
candidates = len(signal_outgoing) + len(outgoing) + len(closed_outgoing)
|
||||||
|
closed_bundle = (
|
||||||
|
_closed_trade_bundle(
|
||||||
|
closed_outgoing,
|
||||||
|
current_book_value=await _paper_book_value(db),
|
||||||
|
)
|
||||||
|
if closed_outgoing
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
if signal_outgoing or outgoing or closed_bundle:
|
||||||
async with httpx.AsyncClient(timeout=15) as client:
|
async with httpx.AsyncClient(timeout=15) as client:
|
||||||
|
for log_refs, text in _signal_bundle_messages(signal_outgoing):
|
||||||
|
try:
|
||||||
|
await _send(client, cfg["token"], cfg["chat_id"], text)
|
||||||
|
for alert_type, key in log_refs:
|
||||||
|
_log_alert(db, alert_type, key)
|
||||||
|
if alert_type == "qualified":
|
||||||
|
_log_alert(db, QUALIFIED_STATE_TYPE, key, value=QUALIFIED_ACTIVE)
|
||||||
|
sent += 1
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Failed to send signal alert bundle")
|
||||||
|
|
||||||
|
if closed_bundle is not None:
|
||||||
|
log_refs, text = closed_bundle
|
||||||
|
try:
|
||||||
|
await _send(client, cfg["token"], cfg["chat_id"], text)
|
||||||
|
for alert_type, key in log_refs:
|
||||||
|
_log_alert(db, alert_type, key)
|
||||||
|
sent += 1
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Failed to send trade-closed alert bundle")
|
||||||
|
|
||||||
for alert_type, key, text in outgoing:
|
for alert_type, key, text in outgoing:
|
||||||
try:
|
try:
|
||||||
await _send(client, cfg["token"], cfg["chat_id"], text)
|
await _send(client, cfg["token"], cfg["chat_id"], text)
|
||||||
@@ -511,8 +1008,11 @@ async def dispatch_alerts(db: AsyncSession) -> dict:
|
|||||||
except Exception:
|
except Exception:
|
||||||
logger.exception("Failed to send alert %s", key)
|
logger.exception("Failed to send alert %s", key)
|
||||||
|
|
||||||
|
for key in qualified_inactive:
|
||||||
|
_log_alert(db, QUALIFIED_STATE_TYPE, key, value=QUALIFIED_INACTIVE)
|
||||||
|
|
||||||
await db.commit() # persist watermark seeds/advances and sent-logs
|
await db.commit() # persist watermark seeds/advances and sent-logs
|
||||||
return {"status": "ok", "sent": sent, "candidates": len(outgoing)}
|
return {"status": "ok", "sent": sent, "candidates": candidates}
|
||||||
|
|
||||||
|
|
||||||
async def send_test_alert(db: AsyncSession) -> dict:
|
async def send_test_alert(db: AsyncSession) -> dict:
|
||||||
|
|||||||
+4121
-81
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,101 @@
|
|||||||
|
"""Benchmark price store + alpha helpers.
|
||||||
|
|
||||||
|
Fetches the S&P 500 proxy (SPY) daily closes via Alpaca and persists them, so
|
||||||
|
paper-trade alpha — a trade's return minus the benchmark's return over the same
|
||||||
|
holding period — can be computed. The benchmark is a standalone series, NOT a
|
||||||
|
tracked ``Ticker``; its closes feed residual momentum and alpha, but it never
|
||||||
|
becomes a trade candidate or rankings-table row.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import bisect
|
||||||
|
import logging
|
||||||
|
from datetime import date, timedelta
|
||||||
|
|
||||||
|
from sqlalchemy import select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.config import settings
|
||||||
|
from app.models.benchmark_price import BenchmarkPrice
|
||||||
|
from app.providers.alpaca import AlpacaOHLCVProvider
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
BENCHMARK_SYMBOL = "SPY"
|
||||||
|
# ~800 calendar days ≈ 550 trading days — comfortably covers any realistic paper
|
||||||
|
# holding period plus a margin for the nearest-prior-trading-day lookup.
|
||||||
|
_HISTORY_DAYS = 800
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh_benchmark_prices(
|
||||||
|
db: AsyncSession, symbol: str = BENCHMARK_SYMBOL, days: int = _HISTORY_DAYS
|
||||||
|
) -> int:
|
||||||
|
"""Fetch the benchmark's daily closes and upsert them. Returns rows written.
|
||||||
|
|
||||||
|
Idempotent: inserts new dates, updates a close only if it changed (e.g. after
|
||||||
|
a split adjustment). Best-effort — returns 0 when Alpaca keys are unset.
|
||||||
|
"""
|
||||||
|
if not settings.alpaca_api_key or not settings.alpaca_api_secret:
|
||||||
|
logger.warning("Benchmark refresh skipped: Alpaca keys not configured")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
provider = AlpacaOHLCVProvider(settings.alpaca_api_key, settings.alpaca_api_secret)
|
||||||
|
end = date.today()
|
||||||
|
start = end - timedelta(days=days)
|
||||||
|
bars = await provider.fetch_ohlcv(symbol, start, end)
|
||||||
|
|
||||||
|
existing = {
|
||||||
|
row.date: row
|
||||||
|
for row in (
|
||||||
|
await db.execute(select(BenchmarkPrice).where(BenchmarkPrice.symbol == symbol))
|
||||||
|
).scalars()
|
||||||
|
}
|
||||||
|
|
||||||
|
written = 0
|
||||||
|
for bar in bars:
|
||||||
|
current = existing.get(bar.date)
|
||||||
|
if current is None:
|
||||||
|
db.add(BenchmarkPrice(symbol=symbol, date=bar.date, close=float(bar.close)))
|
||||||
|
written += 1
|
||||||
|
elif abs(current.close - float(bar.close)) > 1e-9:
|
||||||
|
current.close = float(bar.close)
|
||||||
|
written += 1
|
||||||
|
|
||||||
|
if written:
|
||||||
|
await db.commit()
|
||||||
|
logger.info("Benchmark %s refreshed: %d rows written", symbol, written)
|
||||||
|
return written
|
||||||
|
|
||||||
|
|
||||||
|
async def load_benchmark_closes(
|
||||||
|
db: AsyncSession, symbol: str = BENCHMARK_SYMBOL
|
||||||
|
) -> dict[date, float]:
|
||||||
|
"""Return ``{date: close}`` for the benchmark (empty if none stored yet)."""
|
||||||
|
rows = await db.execute(
|
||||||
|
select(BenchmarkPrice.date, BenchmarkPrice.close).where(BenchmarkPrice.symbol == symbol)
|
||||||
|
)
|
||||||
|
return {d: float(c) for d, c in rows.all()}
|
||||||
|
|
||||||
|
|
||||||
|
def benchmark_return_pct(
|
||||||
|
closes: dict[date, float], open_date: date, as_of_date: date
|
||||||
|
) -> float | None:
|
||||||
|
"""Benchmark % return between two dates, using the nearest close on/before each.
|
||||||
|
|
||||||
|
Returns ``None`` when there's no benchmark data at or before either endpoint
|
||||||
|
(e.g. a trade opened before the stored history, or the table is empty).
|
||||||
|
"""
|
||||||
|
if not closes:
|
||||||
|
return None
|
||||||
|
dates = sorted(closes)
|
||||||
|
|
||||||
|
def _close_on_or_before(target: date) -> float | None:
|
||||||
|
idx = bisect.bisect_right(dates, target) - 1
|
||||||
|
return closes[dates[idx]] if idx >= 0 else None
|
||||||
|
|
||||||
|
start = _close_on_or_before(open_date)
|
||||||
|
end = _close_on_or_before(as_of_date)
|
||||||
|
if start is None or end is None or start == 0:
|
||||||
|
return None
|
||||||
|
return (end - start) / start * 100.0
|
||||||
@@ -1,18 +1,19 @@
|
|||||||
"""Market-breadth early-warning indicator (from the stored universe OHLCV).
|
"""Market-breadth state and early-warning indicators.
|
||||||
|
|
||||||
Breadth is a genuinely *leading* construct: a few mega-caps can keep an index
|
Breadth is a genuinely *leading* construct: a few mega-caps can keep an index
|
||||||
rising while participation narrows underneath — the classic pre-top divergence.
|
rising while participation narrows underneath — the classic pre-top divergence.
|
||||||
We measure it from the OHLCV we already store for the whole universe, so it costs
|
V2 measures an explicit, frozen basket rather than every ticker currently stored
|
||||||
no new data source.
|
in the database. That keeps the live series reproducible when the wider product
|
||||||
|
universe changes.
|
||||||
|
|
||||||
Two layers:
|
Two layers:
|
||||||
- breadth = % of the universe trading above its own 200-DMA (0-100).
|
- breadth = % of the universe trading above its own 200-DMA (0-100).
|
||||||
- divergence = an early-warning score (0-100, high = fragile): the benchmark
|
- divergence = an early-warning score (0-100, high = fragile): the benchmark
|
||||||
price rising *while* breadth falls, plus a nudge for already-low breadth.
|
price holding/rising *while* breadth falls. Absolute low breadth stays in the
|
||||||
|
State index so it is not counted twice.
|
||||||
|
|
||||||
This module only *computes* the indicator. It is deliberately NOT wired into the
|
The live monitor uses the breadth level in State and the pure divergence in
|
||||||
live regime index yet — the event study measures whether it actually leads before
|
Warning. The event study evaluates the latter chronologically.
|
||||||
it earns any weight.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -31,9 +32,9 @@ logger = logging.getLogger(__name__)
|
|||||||
Series = list[tuple[date, float]]
|
Series = list[tuple[date, float]]
|
||||||
|
|
||||||
|
|
||||||
def _breadth_from_closes(
|
def _breadth_with_counts(
|
||||||
closes_by_symbol: dict[str, Series], window: int = 200, min_tickers: int = 20
|
closes_by_symbol: dict[str, Series], window: int = 200, min_tickers: int = 20
|
||||||
) -> dict[date, float]:
|
) -> tuple[dict[date, float], dict[date, int]]:
|
||||||
"""Pure core: % of symbols above their own rolling SMA(window), per date.
|
"""Pure core: % of symbols above their own rolling SMA(window), per date.
|
||||||
|
|
||||||
Each symbol's SMA is computed once with a sliding sum (O(bars)); dates with
|
Each symbol's SMA is computed once with a sliding sum (O(bars)); dates with
|
||||||
@@ -55,11 +56,31 @@ def _breadth_from_closes(
|
|||||||
entry[1] += 1
|
entry[1] += 1
|
||||||
if closes[i] > sma:
|
if closes[i] > sma:
|
||||||
entry[0] += 1
|
entry[0] += 1
|
||||||
return {
|
values = {
|
||||||
d: round(above / total * 100.0, 2)
|
d: round(above / total * 100.0, 2)
|
||||||
for d, (above, total) in counts.items()
|
for d, (above, total) in counts.items()
|
||||||
if total >= min_tickers
|
if total >= min_tickers
|
||||||
}
|
}
|
||||||
|
eligible = {d: total for d, (_, total) in counts.items() if total >= min_tickers}
|
||||||
|
return values, eligible
|
||||||
|
|
||||||
|
|
||||||
|
def _breadth_from_closes(
|
||||||
|
closes_by_symbol: dict[str, Series], window: int = 200, min_tickers: int = 20
|
||||||
|
) -> dict[date, float]:
|
||||||
|
"""Compatibility wrapper returning only the breadth percentage series."""
|
||||||
|
return _breadth_with_counts(closes_by_symbol, window, min_tickers)[0]
|
||||||
|
|
||||||
|
|
||||||
|
# Breadth deterioration counts fully when price masks it (true divergence, the
|
||||||
|
# dangerous pre-top case) and at CONFIRMED_FLOOR when price falls with it.
|
||||||
|
# v2 used a hard ``price_ret >= 0`` cliff, which zeroed the sensor during every
|
||||||
|
# decline -- so on 2026-07-24, with the basket shedding 10 percentage points
|
||||||
|
# above their 200-DMA in 20 sessions, Warning read exactly 0. Breadth *level*
|
||||||
|
# lives in State but breadth *velocity* appears nowhere else, so partial credit
|
||||||
|
# here is not double counting.
|
||||||
|
DIVERGENCE_CONFIRMED_FLOOR = 0.35
|
||||||
|
DIVERGENCE_TAPER_PCT = 3.0
|
||||||
|
|
||||||
|
|
||||||
def compute_divergence_series(
|
def compute_divergence_series(
|
||||||
@@ -67,10 +88,9 @@ def compute_divergence_series(
|
|||||||
) -> dict[date, float]:
|
) -> dict[date, float]:
|
||||||
"""Early-warning score (0-100, high = fragile) per date.
|
"""Early-warning score (0-100, high = fragile) per date.
|
||||||
|
|
||||||
Fragility rises when the benchmark price climbs over ``lookback`` days while
|
A 20 percentage-point breadth deterioration maps to 100 when the benchmark
|
||||||
breadth deteriorates over the same window, and is nudged up when the absolute
|
is flat or rising, tapering to ``DIVERGENCE_CONFIRMED_FLOOR`` of that once
|
||||||
breadth level is already low. It is the *divergence* (not the level) that
|
the benchmark is down ``DIVERGENCE_TAPER_PCT`` or more over the window.
|
||||||
makes this leading.
|
|
||||||
"""
|
"""
|
||||||
bench = {d: c for d, c in benchmark_closes}
|
bench = {d: c for d, c in benchmark_closes}
|
||||||
common = sorted(d for d in bench if d in breadth)
|
common = sorted(d for d in bench if d in breadth)
|
||||||
@@ -82,14 +102,20 @@ def compute_divergence_series(
|
|||||||
continue
|
continue
|
||||||
price_ret = (bench[d] / price_past - 1.0) * 100.0 # %
|
price_ret = (bench[d] / price_past - 1.0) * 100.0 # %
|
||||||
breadth_chg = breadth[d] - breadth[d0] # percentage points
|
breadth_chg = breadth[d] - breadth[d0] # percentage points
|
||||||
raw = price_ret - breadth_chg # price up & breadth down -> large
|
deterioration = max(0.0, -breadth_chg)
|
||||||
score = 50.0 + raw * 2.0 + (50.0 - breadth[d]) * 0.4
|
taper = max(0.0, min(1.0, (price_ret + DIVERGENCE_TAPER_PCT) / DIVERGENCE_TAPER_PCT))
|
||||||
out[d] = max(0.0, min(100.0, round(score, 2)))
|
gate = DIVERGENCE_CONFIRMED_FLOOR + (1.0 - DIVERGENCE_CONFIRMED_FLOOR) * taper
|
||||||
|
out[d] = max(0.0, min(100.0, round(deterioration * 5.0 * gate, 2)))
|
||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
async def _load_universe_closes(db: AsyncSession) -> dict[str, Series]:
|
async def _load_universe_closes(
|
||||||
result = await db.execute(select(Ticker).order_by(Ticker.symbol))
|
db: AsyncSession, symbols: list[str] | None = None
|
||||||
|
) -> dict[str, Series]:
|
||||||
|
stmt = select(Ticker).order_by(Ticker.symbol)
|
||||||
|
if symbols is not None:
|
||||||
|
stmt = stmt.where(Ticker.symbol.in_(symbols))
|
||||||
|
result = await db.execute(stmt)
|
||||||
closes_by_symbol: dict[str, Series] = {}
|
closes_by_symbol: dict[str, Series] = {}
|
||||||
for ticker in result.scalars().all():
|
for ticker in result.scalars().all():
|
||||||
try:
|
try:
|
||||||
@@ -103,13 +129,27 @@ async def _load_universe_closes(db: AsyncSession) -> dict[str, Series]:
|
|||||||
|
|
||||||
|
|
||||||
async def compute_breadth_series(
|
async def compute_breadth_series(
|
||||||
db: AsyncSession, window: int = 200, min_tickers: int = 20
|
db: AsyncSession,
|
||||||
|
window: int = 200,
|
||||||
|
min_tickers: int = 20,
|
||||||
|
symbols: list[str] | None = None,
|
||||||
) -> dict[date, float]:
|
) -> dict[date, float]:
|
||||||
"""Historical breadth series across the stored universe (for the event study)."""
|
"""Historical breadth series across an explicit basket (or all stored names)."""
|
||||||
closes_by_symbol = await _load_universe_closes(db)
|
closes_by_symbol = await _load_universe_closes(db, symbols)
|
||||||
return _breadth_from_closes(closes_by_symbol, window, min_tickers)
|
return _breadth_from_closes(closes_by_symbol, window, min_tickers)
|
||||||
|
|
||||||
|
|
||||||
|
async def compute_breadth_details(
|
||||||
|
db: AsyncSession,
|
||||||
|
symbols: list[str],
|
||||||
|
window: int = 200,
|
||||||
|
min_tickers: int = 20,
|
||||||
|
) -> tuple[dict[date, float], dict[date, int]]:
|
||||||
|
"""Breadth values plus the qualifying-member count for snapshot metadata."""
|
||||||
|
closes_by_symbol = await _load_universe_closes(db, symbols)
|
||||||
|
return _breadth_with_counts(closes_by_symbol, window, min_tickers)
|
||||||
|
|
||||||
|
|
||||||
async def compute_breadth_today(db: AsyncSession) -> float | None:
|
async def compute_breadth_today(db: AsyncSession) -> float | None:
|
||||||
"""Latest breadth reading (thin wrapper, for future live use)."""
|
"""Latest breadth reading (thin wrapper, for future live use)."""
|
||||||
series = await compute_breadth_series(db)
|
series = await compute_breadth_series(db)
|
||||||
|
|||||||
@@ -0,0 +1,348 @@
|
|||||||
|
"""Source-agnostic batch import framework (Dolt/SEC bulk data → PostgreSQL).
|
||||||
|
|
||||||
|
Every bulk importer (SEC facts, Dolt earnings, later Dolt stocks) plugs into
|
||||||
|
``run_import`` and gets, for free, the plan's non-negotiables:
|
||||||
|
|
||||||
|
- **One run per source at a time** — a Postgres *session-level* advisory lock
|
||||||
|
keyed by source. It is held on a single pinned connection for the whole run,
|
||||||
|
so it survives the intermediate commits (the ``running`` row, then the
|
||||||
|
promotion) and only releases at the end. No-op on non-Postgres (tests).
|
||||||
|
- **Idempotent per revision** — the cheap ``detect_revision`` probe is compared
|
||||||
|
against the last *promoted* run; an unchanged revision records a ``no_op``
|
||||||
|
with **zero row changes** (no expensive fetch, no writes).
|
||||||
|
- **Staging then atomic promotion** — the importer stages into an in-memory
|
||||||
|
object (no physical staging tables), validation reads it, and only a passing
|
||||||
|
run calls ``promote`` whose writes commit together with the run-row flip to
|
||||||
|
``promoted`` in a single transaction.
|
||||||
|
- **Failure is inert** — a failed validation or a mid-run exception marks the
|
||||||
|
run ``failed``, alerts via the system-events path, and leaves the live tables
|
||||||
|
exactly as they were (nothing is written before ``promote``).
|
||||||
|
|
||||||
|
Every attempt — promoted, no_op, or failed — is recorded in ``data_import_runs``.
|
||||||
|
KISS: no conflicts table (summaries go in ``validation_json``), no revision
|
||||||
|
table (idempotency queries the last run), no aggregate tables.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime, timedelta, timezone
|
||||||
|
from typing import Any, Protocol, runtime_checkable
|
||||||
|
|
||||||
|
from sqlalchemy import exists, select, text
|
||||||
|
from sqlalchemy.engine import Engine # noqa: F401 (typing only)
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncEngine, AsyncSession
|
||||||
|
|
||||||
|
from app.database import engine as app_engine
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
|
from app.services import system_event_service
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# data_import_runs.status values
|
||||||
|
STATUS_RUNNING = "running"
|
||||||
|
STATUS_VALIDATED = "validated"
|
||||||
|
STATUS_PROMOTED = "promoted"
|
||||||
|
STATUS_NO_OP = "no_op"
|
||||||
|
STATUS_DEFERRED = "deferred"
|
||||||
|
STATUS_FAILED = "failed"
|
||||||
|
|
||||||
|
_MAX_ERROR_LEN = 4000
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ValidationResult:
|
||||||
|
"""Outcome of an importer's validation gates.
|
||||||
|
|
||||||
|
``summary`` is serialized into ``validation_json`` (reconciliation /
|
||||||
|
discrepancy details live here — no separate conflicts table). ``validate``
|
||||||
|
MUST be read-only: it reads the staged object and, if needed, live tables
|
||||||
|
for comparison, but writes nothing — that invariant is what makes a failed
|
||||||
|
run leave the dataset untouched.
|
||||||
|
"""
|
||||||
|
|
||||||
|
ok: bool
|
||||||
|
summary: dict[str, Any] = field(default_factory=dict)
|
||||||
|
source_max_date: date | None = None
|
||||||
|
messages: list[str] = field(default_factory=list)
|
||||||
|
# Expected source-side lag: retry without an immediate error alert. Sources
|
||||||
|
# can bound the quiet period with deferred_alert_after_days. Only meaningful
|
||||||
|
# when ok=False.
|
||||||
|
retryable: bool = False
|
||||||
|
deferred_alert_after_days: int | None = None
|
||||||
|
deferred_alert_messages: list[str] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
@runtime_checkable
|
||||||
|
class SourceImporter(Protocol):
|
||||||
|
"""Interface a concrete bulk importer implements. All methods receive the
|
||||||
|
session bound to the lock-holding connection; ``stage`` and ``validate``
|
||||||
|
never write to live tables, only ``promote`` does."""
|
||||||
|
|
||||||
|
source: str # sec_facts | dolt_earnings | dolt_stocks
|
||||||
|
|
||||||
|
async def detect_revision(self, db: AsyncSession) -> str | None:
|
||||||
|
"""Cheap probe of the source revision (Dolt commit / SEC archive SHA).
|
||||||
|
|
||||||
|
Returns the revision id, or None when it can't be determined cheaply
|
||||||
|
(in which case idempotency is skipped and the run always stages)."""
|
||||||
|
...
|
||||||
|
|
||||||
|
async def stage(self, db: AsyncSession) -> Any:
|
||||||
|
"""Download/parse into an in-memory staged representation. No writes to
|
||||||
|
live tables."""
|
||||||
|
...
|
||||||
|
|
||||||
|
async def validate(self, db: AsyncSession, staged: Any) -> ValidationResult:
|
||||||
|
"""Run the source's validation gates against ``staged``. Read-only."""
|
||||||
|
...
|
||||||
|
|
||||||
|
async def promote(self, db: AsyncSession, staged: Any, run_id: int) -> dict[str, int]:
|
||||||
|
"""Apply ``staged`` to the live tables. Called inside the promotion
|
||||||
|
transaction; the caller commits. ``run_id`` is the current
|
||||||
|
``data_import_runs.id`` so written rows can be stamped with their
|
||||||
|
``import_run_id``. Returns row-count deltas."""
|
||||||
|
...
|
||||||
|
|
||||||
|
|
||||||
|
def _advisory_key(source: str) -> int:
|
||||||
|
"""Deterministic signed 64-bit key for a source's advisory lock."""
|
||||||
|
digest = hashlib.blake2b(source.encode("utf-8"), digest_size=8).digest()
|
||||||
|
return int.from_bytes(digest, "big", signed=True)
|
||||||
|
|
||||||
|
|
||||||
|
async def _last_promoted_revision(db: AsyncSession, source: str) -> str | None:
|
||||||
|
"""Revision of the most recent *promoted* run for ``source`` (the revision
|
||||||
|
currently loaded), or None if none has promoted yet."""
|
||||||
|
row = await db.execute(
|
||||||
|
select(DataImportRun.revision)
|
||||||
|
.where(
|
||||||
|
DataImportRun.source == source,
|
||||||
|
DataImportRun.status == STATUS_PROMOTED,
|
||||||
|
)
|
||||||
|
.order_by(DataImportRun.id.desc())
|
||||||
|
.limit(1)
|
||||||
|
)
|
||||||
|
return row.scalar_one_or_none()
|
||||||
|
|
||||||
|
|
||||||
|
async def _promotion_state_since(
|
||||||
|
db: AsyncSession, source: str, cutoff: datetime
|
||||||
|
) -> str:
|
||||||
|
promoted = (
|
||||||
|
DataImportRun.source == source,
|
||||||
|
DataImportRun.status == STATUS_PROMOTED,
|
||||||
|
)
|
||||||
|
ever, recent = (
|
||||||
|
await db.execute(
|
||||||
|
select(
|
||||||
|
exists().where(*promoted),
|
||||||
|
exists().where(*promoted, DataImportRun.started_at >= cutoff),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).one()
|
||||||
|
return "recent" if recent else "stale" if ever else "never"
|
||||||
|
|
||||||
|
|
||||||
|
def _now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
async def _alert(
|
||||||
|
db: AsyncSession,
|
||||||
|
source: str,
|
||||||
|
code: str,
|
||||||
|
messages: list[str],
|
||||||
|
*,
|
||||||
|
severity: str = "error",
|
||||||
|
dedup_hours: int = 24,
|
||||||
|
) -> None:
|
||||||
|
try:
|
||||||
|
await system_event_service.log_event(
|
||||||
|
db,
|
||||||
|
severity=severity,
|
||||||
|
source="data_import",
|
||||||
|
code=f"{source}_{code}",
|
||||||
|
message=(("; ".join(messages)) or code)[:_MAX_ERROR_LEN],
|
||||||
|
dedup_key=f"data_import:{source}:{code}",
|
||||||
|
dedup_hours=dedup_hours,
|
||||||
|
)
|
||||||
|
except Exception: # noqa: BLE001 — alerting must never mask the real outcome
|
||||||
|
logger.exception("Failed to emit data_import alert %s/%s", source, code)
|
||||||
|
|
||||||
|
|
||||||
|
async def run_import(
|
||||||
|
importer: SourceImporter,
|
||||||
|
*,
|
||||||
|
engine: AsyncEngine | None = None,
|
||||||
|
force: bool = False,
|
||||||
|
) -> DataImportRun | None:
|
||||||
|
"""Run one import for ``importer``.
|
||||||
|
|
||||||
|
Returns the recorded ``DataImportRun`` (promoted / no_op / deferred /
|
||||||
|
failed), or None
|
||||||
|
when the per-source advisory lock is already held (another run is active).
|
||||||
|
|
||||||
|
``force`` runs even when the revision is unchanged. The revision tracks the
|
||||||
|
*source*, so a re-import driven by a change on our side — a parser fix that
|
||||||
|
makes stored rows stale — is a no_op under the normal gate. Manually invoked
|
||||||
|
only; scheduled jobs must leave it False so an unchanged source stays a no_op.
|
||||||
|
"""
|
||||||
|
engine = engine or app_engine
|
||||||
|
source = importer.source
|
||||||
|
is_pg = engine.dialect.name == "postgresql"
|
||||||
|
key = _advisory_key(source)
|
||||||
|
|
||||||
|
async with engine.connect() as conn:
|
||||||
|
# Bind the session to this one connection so the session-level advisory
|
||||||
|
# lock persists across our commits. expire_on_commit must be set here —
|
||||||
|
# the app factory's setting doesn't carry to a directly-built session.
|
||||||
|
session = AsyncSession(bind=conn, expire_on_commit=False)
|
||||||
|
try:
|
||||||
|
if is_pg:
|
||||||
|
got = (
|
||||||
|
await session.execute(
|
||||||
|
text("SELECT pg_try_advisory_lock(:k)"), {"k": key}
|
||||||
|
)
|
||||||
|
).scalar()
|
||||||
|
await session.commit()
|
||||||
|
if not got:
|
||||||
|
logger.info("data_import %s: lock held, skipping", source)
|
||||||
|
return None
|
||||||
|
|
||||||
|
# Record the attempt FIRST — before the external revision probe, the
|
||||||
|
# most likely failure — so anything below is recorded and alerted and
|
||||||
|
# never escapes unrecorded. Revision is filled in once detected.
|
||||||
|
run = DataImportRun(
|
||||||
|
source=source,
|
||||||
|
status=STATUS_RUNNING,
|
||||||
|
started_at=_now(),
|
||||||
|
)
|
||||||
|
session.add(run)
|
||||||
|
await session.commit()
|
||||||
|
await session.refresh(run)
|
||||||
|
|
||||||
|
try:
|
||||||
|
revision = await importer.detect_revision(session)
|
||||||
|
run.revision = revision
|
||||||
|
last_rev = await _last_promoted_revision(session, source)
|
||||||
|
if not force and revision is not None and revision == last_rev:
|
||||||
|
run.status = STATUS_NO_OP
|
||||||
|
run.completed_at = _now()
|
||||||
|
await session.commit()
|
||||||
|
logger.info("data_import %s: no_op (revision %s)", source, revision)
|
||||||
|
return run
|
||||||
|
|
||||||
|
staged = await importer.stage(session)
|
||||||
|
result = await importer.validate(session, staged)
|
||||||
|
run.source_max_date = result.source_max_date
|
||||||
|
run.validation_json = json.dumps(result.summary, default=str)
|
||||||
|
|
||||||
|
if not result.ok:
|
||||||
|
run.error_details = ("; ".join(result.messages))[:_MAX_ERROR_LEN]
|
||||||
|
run.completed_at = _now()
|
||||||
|
if result.retryable:
|
||||||
|
run.status = STATUS_DEFERRED
|
||||||
|
await session.commit()
|
||||||
|
alert_days = result.deferred_alert_after_days
|
||||||
|
if alert_days is not None:
|
||||||
|
alert_days = max(1, alert_days)
|
||||||
|
cutoff = run.started_at - timedelta(days=alert_days)
|
||||||
|
promotion_state = await _promotion_state_since(
|
||||||
|
session, source, cutoff
|
||||||
|
)
|
||||||
|
if promotion_state != "recent":
|
||||||
|
history = (
|
||||||
|
f"{source} import has never promoted successfully"
|
||||||
|
if promotion_state == "never"
|
||||||
|
else f"{source} import has not promoted successfully "
|
||||||
|
f"within {alert_days} day(s)"
|
||||||
|
)
|
||||||
|
await _alert(
|
||||||
|
session,
|
||||||
|
source,
|
||||||
|
"deferred_stale",
|
||||||
|
[
|
||||||
|
f"{history}; import remains deferred",
|
||||||
|
*result.deferred_alert_messages,
|
||||||
|
f"Current deferral: "
|
||||||
|
f"{run.error_details or 'validation deferred'}",
|
||||||
|
],
|
||||||
|
severity="warning",
|
||||||
|
dedup_hours=alert_days * 24,
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"data_import %s: deferred for retry: %s",
|
||||||
|
source,
|
||||||
|
result.messages,
|
||||||
|
)
|
||||||
|
return run
|
||||||
|
|
||||||
|
run.status = STATUS_FAILED
|
||||||
|
await session.commit()
|
||||||
|
await _alert(session, source, "validation_failed", result.messages)
|
||||||
|
logger.warning(
|
||||||
|
"data_import %s: validation failed: %s",
|
||||||
|
source,
|
||||||
|
result.messages,
|
||||||
|
)
|
||||||
|
return run
|
||||||
|
|
||||||
|
# Promotion: importer writes + run-row flip in one transaction.
|
||||||
|
row_counts = await importer.promote(session, staged, run.id)
|
||||||
|
run.status = STATUS_PROMOTED
|
||||||
|
run.row_counts_json = json.dumps(row_counts, default=str)
|
||||||
|
run.completed_at = _now()
|
||||||
|
await session.commit()
|
||||||
|
await session.refresh(run)
|
||||||
|
logger.info(
|
||||||
|
"data_import %s: promoted (revision %s, rows %s)",
|
||||||
|
source,
|
||||||
|
revision,
|
||||||
|
row_counts,
|
||||||
|
)
|
||||||
|
return run
|
||||||
|
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
# Deploy / scheduler shutdown: best-effort mark failed so no
|
||||||
|
# ``running`` row lingers, then let the cancellation propagate —
|
||||||
|
# never swallow it.
|
||||||
|
try:
|
||||||
|
await session.rollback()
|
||||||
|
run.status = STATUS_FAILED
|
||||||
|
run.error_details = "cancelled"
|
||||||
|
run.completed_at = _now()
|
||||||
|
await session.commit()
|
||||||
|
except BaseException: # noqa: BLE001 — best-effort during teardown
|
||||||
|
logger.warning(
|
||||||
|
"data_import %s: could not record cancellation", source
|
||||||
|
)
|
||||||
|
raise
|
||||||
|
|
||||||
|
except Exception as exc: # noqa: BLE001 — record + alert, don't crash the job
|
||||||
|
await session.rollback()
|
||||||
|
run.status = STATUS_FAILED
|
||||||
|
run.error_details = repr(exc)[:_MAX_ERROR_LEN]
|
||||||
|
run.completed_at = _now()
|
||||||
|
try:
|
||||||
|
await session.commit()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
logger.exception("data_import %s: failed to record failure", source)
|
||||||
|
await _alert(session, source, "import_error", [repr(exc)])
|
||||||
|
logger.exception("data_import %s: import error", source)
|
||||||
|
return run
|
||||||
|
|
||||||
|
finally:
|
||||||
|
if is_pg:
|
||||||
|
try:
|
||||||
|
await session.execute(
|
||||||
|
text("SELECT pg_advisory_unlock(:k)"), {"k": key}
|
||||||
|
)
|
||||||
|
await session.commit()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
logger.exception("data_import %s: failed to release lock", source)
|
||||||
|
await session.close()
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
"""Minimal async client for a local Dolt clone.
|
||||||
|
|
||||||
|
The application never runs a long-lived Dolt sql-server; it shells out to the
|
||||||
|
`dolt` CLI against a persistent clone and reads results as CSV. Every call goes
|
||||||
|
through ``asyncio.create_subprocess_exec`` because the scheduler shares one event
|
||||||
|
loop with the API (`app/scheduler.py:73`) — a blocking `subprocess.run` here
|
||||||
|
would stall request handling.
|
||||||
|
|
||||||
|
Production keeps the clone in ``DOLT_DATA_DIR`` outside the deploy tree; the
|
||||||
|
binary path and data dir are configured (see ``app/config.py``). Read via
|
||||||
|
``dolt sql -r csv``; refresh with ``pull`` and record the resulting commit hash
|
||||||
|
as the import revision.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import logging
|
||||||
|
import shutil
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Default subprocess timeout. A hung `dolt pull`/`sql` would otherwise pin the
|
||||||
|
# import's connection and its advisory lock indefinitely, so every call is
|
||||||
|
# bounded; callers may override per operation.
|
||||||
|
DEFAULT_TIMEOUT = 600.0
|
||||||
|
|
||||||
|
|
||||||
|
class DoltError(RuntimeError):
|
||||||
|
"""A dolt subprocess failed, timed out, or exited non-zero."""
|
||||||
|
|
||||||
|
|
||||||
|
async def _run(
|
||||||
|
binary: str, args: list[str], *, cwd: Path, timeout: float = DEFAULT_TIMEOUT
|
||||||
|
) -> str:
|
||||||
|
proc = await asyncio.create_subprocess_exec(
|
||||||
|
binary,
|
||||||
|
*args,
|
||||||
|
cwd=str(cwd),
|
||||||
|
stdout=asyncio.subprocess.PIPE,
|
||||||
|
stderr=asyncio.subprocess.PIPE,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
stdout, stderr = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
proc.kill()
|
||||||
|
try:
|
||||||
|
await proc.wait()
|
||||||
|
except ProcessLookupError:
|
||||||
|
pass
|
||||||
|
raise DoltError(f"dolt {args[0] if args else ''} timed out after {timeout:.0f}s")
|
||||||
|
if proc.returncode != 0:
|
||||||
|
raise DoltError(
|
||||||
|
f"dolt {' '.join(args)} failed ({proc.returncode}): "
|
||||||
|
f"{stderr.decode('utf-8', 'replace').strip()[:500]}"
|
||||||
|
)
|
||||||
|
return stdout.decode("utf-8", "replace")
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_free_disk(path: Path, min_free_gb: float) -> None:
|
||||||
|
"""Raise if free space at ``path`` is below the threshold (checked before a
|
||||||
|
pull that could grow the clone). Uses the nearest existing ancestor so it
|
||||||
|
works before the clone dir exists."""
|
||||||
|
probe = path
|
||||||
|
while not probe.exists() and probe.parent != probe:
|
||||||
|
probe = probe.parent
|
||||||
|
free_gb = shutil.disk_usage(probe).free / (1024**3)
|
||||||
|
if free_gb < min_free_gb:
|
||||||
|
raise DoltError(
|
||||||
|
f"insufficient disk for dolt at {path}: {free_gb:.1f} GB free "
|
||||||
|
f"< {min_free_gb:.1f} GB required"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def pull(repo_dir: Path, *, binary: str, timeout: float = DEFAULT_TIMEOUT) -> None:
|
||||||
|
"""`dolt pull` the persistent clone to the latest upstream revision."""
|
||||||
|
await _run(binary, ["pull"], cwd=repo_dir, timeout=timeout)
|
||||||
|
|
||||||
|
|
||||||
|
async def current_commit(
|
||||||
|
repo_dir: Path, *, binary: str, timeout: float = DEFAULT_TIMEOUT
|
||||||
|
) -> str:
|
||||||
|
"""The HEAD commit hash of the clone — used as the import revision.
|
||||||
|
|
||||||
|
Uses ``DOLT_HASHOF('HEAD')`` (which formally identifies HEAD) rather than
|
||||||
|
ordering ``dolt_log`` by timestamp."""
|
||||||
|
rows = await query_csv(
|
||||||
|
repo_dir, "SELECT DOLT_HASHOF('HEAD') AS commit_hash", binary=binary, timeout=timeout
|
||||||
|
)
|
||||||
|
if not rows or not rows[0].get("commit_hash"):
|
||||||
|
raise DoltError("could not read HEAD commit hash")
|
||||||
|
return rows[0]["commit_hash"]
|
||||||
|
|
||||||
|
|
||||||
|
async def query_csv(
|
||||||
|
repo_dir: Path, sql: str, *, binary: str, timeout: float = DEFAULT_TIMEOUT
|
||||||
|
) -> list[dict[str, str]]:
|
||||||
|
"""Run a read query and parse the CSV result into a list of dict rows."""
|
||||||
|
out = await _run(binary, ["sql", "-q", sql, "-r", "csv"], cwd=repo_dir, timeout=timeout)
|
||||||
|
if not out.strip():
|
||||||
|
return []
|
||||||
|
return list(csv.DictReader(io.StringIO(out)))
|
||||||
@@ -0,0 +1,357 @@
|
|||||||
|
"""Production importer for the DoltHub post-no-preference/earnings calendar.
|
||||||
|
|
||||||
|
A ``SourceImporter`` (see ``app/services/data_import.py``) that pulls the local
|
||||||
|
Dolt clone, aligns the announcement calendar to the EPS history with the pure DP
|
||||||
|
in ``earnings_alignment`` (reused from the research script, not extending it),
|
||||||
|
and writes ``earnings_events`` for the tracked universe.
|
||||||
|
|
||||||
|
Shadow by construction: nothing reads ``earnings_events`` until the API/panel
|
||||||
|
lands (A4), so writing it does not touch production behavior.
|
||||||
|
|
||||||
|
**Promotion is destructive** — future-dated rows for this source are deleted and
|
||||||
|
re-inserted every run so reschedules/cancellations never linger. The forward
|
||||||
|
calendar is the project's acceptance gate, so ``validate`` is fail-closed: it
|
||||||
|
blocks promotion when the staged future set is empty or has collapsed relative
|
||||||
|
to what's already loaded.
|
||||||
|
|
||||||
|
Attribution: the earnings data is CC BY-SA 4.0 from post-no-preference/earnings.
|
||||||
|
See the repo ``NOTICE``. Internal use only — no redistribution.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from sqlalchemy import case, delete, func, select
|
||||||
|
|
||||||
|
from app.config import settings
|
||||||
|
from app.database import insert_for_session
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import dolt_client, earnings_alignment
|
||||||
|
from app.services.data_import import ValidationResult
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
SOURCE = "dolt_earnings"
|
||||||
|
|
||||||
|
# Earliest announcement date to import (matches the research backfill window).
|
||||||
|
WINDOW_START = date(2020, 1, 22)
|
||||||
|
# Alignment tolerances (research defaults): an announcement may lead its period
|
||||||
|
# end by up to 14 days or lag it by up to 90.
|
||||||
|
MAX_LAG_DAYS = 90
|
||||||
|
MAX_LEAD_DAYS = 14
|
||||||
|
# Fail promotion if the staged forward calendar drops below this fraction of the
|
||||||
|
# currently-loaded forward calendar (guards the destructive re-insert against a
|
||||||
|
# partial parse / symbol-mapping regression).
|
||||||
|
MIN_FUTURE_RATIO = 0.5
|
||||||
|
# Initial-load gates (when nothing is loaded yet — the ratio gate has no baseline).
|
||||||
|
# The source publishes a forward calendar; require a real horizon, not one stray
|
||||||
|
# future row. 21 days is a conservative floor under the ~35d horizon observed on
|
||||||
|
# the live clone.
|
||||||
|
MIN_FORWARD_HORIZON_DAYS = 21
|
||||||
|
# ...and require the symbol join to reach most of the tracked universe, so a
|
||||||
|
# broken/normalization-dropped join can't seed a hollow calendar.
|
||||||
|
MIN_INITIAL_COVERAGE = 0.5
|
||||||
|
|
||||||
|
_CAL_SQL = (
|
||||||
|
"SELECT act_symbol, `date`, `when` FROM earnings_calendar "
|
||||||
|
f"WHERE `date` >= '{WINDOW_START.isoformat()}'"
|
||||||
|
)
|
||||||
|
_HIST_SQL = (
|
||||||
|
"SELECT act_symbol, period_end_date, reported, estimate FROM eps_history "
|
||||||
|
f"WHERE period_end_date >= '{(WINDOW_START.replace(year=WINDOW_START.year - 1)).isoformat()}'"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class StagedEarnings:
|
||||||
|
rows: list[dict[str, Any]]
|
||||||
|
stats: dict[str, Any] = field(default_factory=dict)
|
||||||
|
future_count: int = 0
|
||||||
|
max_announce_date: date | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
class DoltEarningsImporter:
|
||||||
|
source = SOURCE
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
repo_dir: Path | str | None = None,
|
||||||
|
binary: str | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
do_pull: bool = True,
|
||||||
|
dolt: Any = dolt_client,
|
||||||
|
) -> None:
|
||||||
|
self.repo_dir = Path(
|
||||||
|
repo_dir
|
||||||
|
or (Path(settings.dolt_data_dir) / settings.dolt_earnings_subdir)
|
||||||
|
)
|
||||||
|
self.binary = binary or settings.dolt_binary
|
||||||
|
self.today = today or _now().date()
|
||||||
|
self.do_pull = do_pull
|
||||||
|
self._dolt = dolt # injectable for tests
|
||||||
|
|
||||||
|
# -- SourceImporter protocol -------------------------------------------
|
||||||
|
|
||||||
|
async def detect_revision(self, db) -> str | None:
|
||||||
|
timeout = settings.dolt_command_timeout_seconds
|
||||||
|
if self.do_pull:
|
||||||
|
dolt_client.ensure_free_disk(self.repo_dir, settings.dolt_min_free_disk_gb)
|
||||||
|
await self._dolt.pull(self.repo_dir, binary=self.binary, timeout=timeout)
|
||||||
|
return await self._dolt.current_commit(
|
||||||
|
self.repo_dir, binary=self.binary, timeout=timeout
|
||||||
|
)
|
||||||
|
|
||||||
|
async def stage(self, db) -> StagedEarnings:
|
||||||
|
universe = await self._load_universe(db) # {normalised symbol: ticker_id}
|
||||||
|
|
||||||
|
timeout = settings.dolt_command_timeout_seconds
|
||||||
|
cal_raw = await self._dolt.query_csv(
|
||||||
|
self.repo_dir, _CAL_SQL, binary=self.binary, timeout=timeout
|
||||||
|
)
|
||||||
|
hist_raw = await self._dolt.query_csv(
|
||||||
|
self.repo_dir, _HIST_SQL, binary=self.binary, timeout=timeout
|
||||||
|
)
|
||||||
|
_require_columns(cal_raw, {"act_symbol", "date", "when"}, "earnings_calendar")
|
||||||
|
_require_columns(
|
||||||
|
hist_raw, {"act_symbol", "period_end_date", "reported", "estimate"}, "eps_history"
|
||||||
|
)
|
||||||
|
|
||||||
|
cal_parsed = _parse_calendar(cal_raw, universe)
|
||||||
|
hist_parsed = _parse_history(hist_raw, universe)
|
||||||
|
calendar, cal_stats = earnings_alignment.dedup_calendar(cal_parsed)
|
||||||
|
history, hist_stats = earnings_alignment.dedup_history(hist_parsed)
|
||||||
|
|
||||||
|
period_lower = WINDOW_START.replace(year=WINDOW_START.year - 1)
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
matched = unmatched = 0
|
||||||
|
for symbol, events in calendar.items():
|
||||||
|
ticker_id = universe[symbol]
|
||||||
|
periods = [
|
||||||
|
p for p in history.get(symbol, []) if p["period_end_date"] >= period_lower
|
||||||
|
]
|
||||||
|
matches, unmatched_events, _ = earnings_alignment.align_symbol(
|
||||||
|
events, periods, max_lag_days=MAX_LAG_DAYS, max_lead_days=MAX_LEAD_DAYS
|
||||||
|
)
|
||||||
|
matched += len(matches)
|
||||||
|
unmatched += len(unmatched_events)
|
||||||
|
matched_by_event = {e: p for e, p in matches}
|
||||||
|
for e_idx, event in enumerate(events):
|
||||||
|
p_idx = matched_by_event.get(e_idx)
|
||||||
|
period = periods[p_idx] if p_idx is not None else None
|
||||||
|
rows.append(
|
||||||
|
{
|
||||||
|
"ticker_id": ticker_id,
|
||||||
|
"symbol": symbol,
|
||||||
|
"announce_date": event["announce_date"],
|
||||||
|
"session": event["session"],
|
||||||
|
"period_end": period["period_end_date"] if period else None,
|
||||||
|
"eps_estimate": period["eps_estimate"] if period else None,
|
||||||
|
"eps_actual": period["eps_actual"] if period else None,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
future_rows = [r for r in rows if r["announce_date"] > self.today]
|
||||||
|
tickers_with_future = {r["ticker_id"] for r in future_rows}
|
||||||
|
stats = {
|
||||||
|
"calendar": cal_stats,
|
||||||
|
"eps_history": hist_stats,
|
||||||
|
"universe_size": len(universe),
|
||||||
|
"symbols_with_calendar": len(calendar),
|
||||||
|
"matched_events": matched,
|
||||||
|
"unmatched_events": unmatched,
|
||||||
|
"tracked_tickers_with_future_date": len(tickers_with_future),
|
||||||
|
}
|
||||||
|
return StagedEarnings(
|
||||||
|
rows=rows,
|
||||||
|
stats=stats,
|
||||||
|
future_count=len(future_rows),
|
||||||
|
max_announce_date=max((r["announce_date"] for r in rows), default=None),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def validate(self, db, staged: StagedEarnings) -> ValidationResult:
|
||||||
|
# Promote deletes+reinserts the forward calendar, so this gate is
|
||||||
|
# fail-closed. The forward calendar is the project's acceptance gate.
|
||||||
|
messages: list[str] = []
|
||||||
|
current_future = await self._current_future_count(db)
|
||||||
|
universe_size = int(staged.stats.get("universe_size", 0) or 0)
|
||||||
|
coverage = (
|
||||||
|
staged.stats.get("symbols_with_calendar", 0) / universe_size
|
||||||
|
if universe_size
|
||||||
|
else 0.0
|
||||||
|
)
|
||||||
|
horizon_days = (
|
||||||
|
(staged.max_announce_date - self.today).days if staged.max_announce_date else 0
|
||||||
|
)
|
||||||
|
|
||||||
|
if staged.future_count == 0:
|
||||||
|
messages.append("no future-dated earnings rows staged")
|
||||||
|
elif current_future == 0:
|
||||||
|
# Initial load: no baseline for the ratio gate, so require a real
|
||||||
|
# forward horizon and broad universe coverage instead of one stray row.
|
||||||
|
if horizon_days < MIN_FORWARD_HORIZON_DAYS:
|
||||||
|
messages.append(
|
||||||
|
f"forward horizon only {horizon_days}d < {MIN_FORWARD_HORIZON_DAYS}d "
|
||||||
|
"on initial load"
|
||||||
|
)
|
||||||
|
if coverage < MIN_INITIAL_COVERAGE:
|
||||||
|
messages.append(
|
||||||
|
f"initial universe coverage {coverage:.0%} "
|
||||||
|
f"< {MIN_INITIAL_COVERAGE:.0%} — symbol join likely broken"
|
||||||
|
)
|
||||||
|
elif staged.future_count < current_future * MIN_FUTURE_RATIO:
|
||||||
|
messages.append(
|
||||||
|
f"forward calendar collapsed: staged {staged.future_count} future rows "
|
||||||
|
f"< {MIN_FUTURE_RATIO:.0%} of current {current_future}"
|
||||||
|
)
|
||||||
|
|
||||||
|
keys = [(r["ticker_id"], r["announce_date"]) for r in staged.rows]
|
||||||
|
if len(keys) != len(set(keys)):
|
||||||
|
messages.append("duplicate (ticker_id, announce_date) in staged set")
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
**staged.stats,
|
||||||
|
"staged_rows": len(staged.rows),
|
||||||
|
"future_rows": staged.future_count,
|
||||||
|
"current_future_rows": current_future,
|
||||||
|
"forward_horizon_days": horizon_days,
|
||||||
|
"universe_coverage": round(coverage, 3),
|
||||||
|
}
|
||||||
|
return ValidationResult(
|
||||||
|
ok=not messages,
|
||||||
|
summary=summary,
|
||||||
|
source_max_date=staged.max_announce_date,
|
||||||
|
messages=messages,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def promote(self, db, staged: StagedEarnings, run_id: int) -> dict[str, int]:
|
||||||
|
# Rescheduling: drop this source's future rows, then upsert the staged
|
||||||
|
# set. Past rows (results) are never deleted; moved/cancelled future
|
||||||
|
# dates simply don't reappear.
|
||||||
|
deleted = (
|
||||||
|
await db.execute(
|
||||||
|
delete(EarningsEvent).where(
|
||||||
|
EarningsEvent.source == SOURCE,
|
||||||
|
EarningsEvent.announce_date > self.today,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).rowcount or 0
|
||||||
|
|
||||||
|
now = _now()
|
||||||
|
for r in staged.rows:
|
||||||
|
stmt = insert_for_session(db, EarningsEvent).values(
|
||||||
|
ticker_id=r["ticker_id"],
|
||||||
|
announce_date=r["announce_date"],
|
||||||
|
session=r["session"],
|
||||||
|
period_end=r["period_end"],
|
||||||
|
eps_estimate=r["eps_estimate"],
|
||||||
|
eps_actual=r["eps_actual"],
|
||||||
|
source=SOURCE,
|
||||||
|
import_run_id=run_id,
|
||||||
|
created_at=now,
|
||||||
|
)
|
||||||
|
# Preserve a non-null prior EPS/period-end if a re-pairing comes back
|
||||||
|
# null; prefer a known session over 'unknown'.
|
||||||
|
stmt = stmt.on_conflict_do_update(
|
||||||
|
index_elements=["ticker_id", "announce_date"],
|
||||||
|
set_={
|
||||||
|
"session": case(
|
||||||
|
(stmt.excluded.session != "unknown", stmt.excluded.session),
|
||||||
|
else_=EarningsEvent.session,
|
||||||
|
),
|
||||||
|
"period_end": func.coalesce(
|
||||||
|
stmt.excluded.period_end, EarningsEvent.period_end
|
||||||
|
),
|
||||||
|
"eps_estimate": func.coalesce(
|
||||||
|
stmt.excluded.eps_estimate, EarningsEvent.eps_estimate
|
||||||
|
),
|
||||||
|
"eps_actual": func.coalesce(
|
||||||
|
stmt.excluded.eps_actual, EarningsEvent.eps_actual
|
||||||
|
),
|
||||||
|
"source": stmt.excluded.source,
|
||||||
|
"import_run_id": stmt.excluded.import_run_id,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
await db.execute(stmt)
|
||||||
|
|
||||||
|
return {"deleted_future": int(deleted), "upserted": len(staged.rows)}
|
||||||
|
|
||||||
|
# -- helpers -----------------------------------------------------------
|
||||||
|
|
||||||
|
async def _load_universe(self, db) -> dict[str, int]:
|
||||||
|
rows = (await db.execute(select(Ticker.id, Ticker.symbol))).all()
|
||||||
|
return {
|
||||||
|
earnings_alignment.normalise_symbol(symbol): tid
|
||||||
|
for tid, symbol in rows
|
||||||
|
if symbol
|
||||||
|
}
|
||||||
|
|
||||||
|
async def _current_future_count(self, db) -> int:
|
||||||
|
return (
|
||||||
|
await db.execute(
|
||||||
|
select(func.count())
|
||||||
|
.select_from(EarningsEvent)
|
||||||
|
.where(
|
||||||
|
EarningsEvent.source == SOURCE,
|
||||||
|
EarningsEvent.announce_date > self.today,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).scalar_one()
|
||||||
|
|
||||||
|
|
||||||
|
def _require_columns(rows: list[dict[str, str]], required: set[str], table: str) -> None:
|
||||||
|
"""Upstream schema-change gate: a missing column stops the run (→ failed)."""
|
||||||
|
if not rows:
|
||||||
|
return
|
||||||
|
present = set(rows[0].keys())
|
||||||
|
missing = required - present
|
||||||
|
if missing:
|
||||||
|
raise ValueError(f"{table}: upstream schema change, missing columns {sorted(missing)}")
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_calendar(raw: list[dict[str, str]], universe: dict[str, int]) -> list[dict[str, Any]]:
|
||||||
|
out: list[dict[str, Any]] = []
|
||||||
|
for row in raw:
|
||||||
|
symbol = earnings_alignment.normalise_symbol(row.get("act_symbol"))
|
||||||
|
raw_date = str(row.get("date") or "")[:10]
|
||||||
|
if symbol not in universe or not raw_date:
|
||||||
|
continue
|
||||||
|
announce_date = date.fromisoformat(raw_date)
|
||||||
|
if announce_date < WINDOW_START:
|
||||||
|
continue
|
||||||
|
out.append(
|
||||||
|
{
|
||||||
|
"symbol": symbol,
|
||||||
|
"announce_date": announce_date,
|
||||||
|
"session": earnings_alignment.normalise_session(row.get("when")),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_history(raw: list[dict[str, str]], universe: dict[str, int]) -> list[dict[str, Any]]:
|
||||||
|
out: list[dict[str, Any]] = []
|
||||||
|
for row in raw:
|
||||||
|
symbol = earnings_alignment.normalise_symbol(row.get("act_symbol"))
|
||||||
|
raw_date = str(row.get("period_end_date") or "")[:10]
|
||||||
|
if symbol not in universe or not raw_date:
|
||||||
|
continue
|
||||||
|
out.append(
|
||||||
|
{
|
||||||
|
"symbol": symbol,
|
||||||
|
"period_end_date": date.fromisoformat(raw_date),
|
||||||
|
"eps_actual": earnings_alignment.safe_number(row.get("reported")),
|
||||||
|
"eps_estimate": earnings_alignment.safe_number(row.get("estimate")),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return out
|
||||||
@@ -0,0 +1,207 @@
|
|||||||
|
"""Pure calendar<->EPS-history alignment for the DoltHub earnings source.
|
||||||
|
|
||||||
|
The earnings repo keeps the announcement calendar (`earnings_calendar`) and the
|
||||||
|
reported/estimate EPS history (`eps_history`) in separate tables with no shared
|
||||||
|
key — the calendar has announce dates, the history has period-end dates. This
|
||||||
|
module reproduces the research importer's **minimum-cost monotonic alignment**
|
||||||
|
(`scripts/import_dolthub_earnings.py`) as pure, DB-free, unit-testable functions
|
||||||
|
so the production importer can reuse it without extending that one-off script.
|
||||||
|
|
||||||
|
Constants and cost function are kept identical to the research script; the DP is
|
||||||
|
what pairs each announcement with the quarter it reported, tolerating gaps on
|
||||||
|
either side. Do not tune these without re-validating surprise-history pairing.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
from collections import defaultdict
|
||||||
|
from datetime import date
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
# Alignment costs — identical to scripts/import_dolthub_earnings.py.
|
||||||
|
SKIP_EVENT_COST = 45.0
|
||||||
|
SKIP_PERIOD_COST = 45.0
|
||||||
|
_TYPICAL_ANNOUNCE_LAG_DAYS = 30 # announcements land ~a month after period end
|
||||||
|
_MISSING_SESSION_PENALTY = 3.0
|
||||||
|
|
||||||
|
# Session normalization → the three values the schema/API promise.
|
||||||
|
_SESSION_ALIASES = {
|
||||||
|
"before market open": "bmo",
|
||||||
|
"before open": "bmo",
|
||||||
|
"bmo": "bmo",
|
||||||
|
"after market close": "amc",
|
||||||
|
"after close": "amc",
|
||||||
|
"amc": "amc",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def normalise_symbol(value: Any) -> str:
|
||||||
|
"""Upper-case, trim, and map dots to dashes so the DoltHub `act_symbol`
|
||||||
|
(`BF.B`) and the app's `tickers.symbol` join after the same normalization."""
|
||||||
|
return str(value or "").strip().upper().replace(".", "-")
|
||||||
|
|
||||||
|
|
||||||
|
def normalise_session(value: Any) -> str:
|
||||||
|
"""Map the source `when` text to bmo | amc | unknown. Anything not clearly a
|
||||||
|
pre-open or post-close session (including 'during market hours' and blanks)
|
||||||
|
collapses to 'unknown' — the schema/API only promise those three."""
|
||||||
|
cleaned = str(value or "").strip().lower().replace("_", " ").replace("-", " ")
|
||||||
|
return _SESSION_ALIASES.get(cleaned, "unknown")
|
||||||
|
|
||||||
|
|
||||||
|
def safe_number(value: Any) -> float | None:
|
||||||
|
if value is None or str(value).strip() == "":
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
result = float(value)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
return result if math.isfinite(result) else None
|
||||||
|
|
||||||
|
|
||||||
|
def dedup_calendar(
|
||||||
|
rows: list[dict[str, Any]],
|
||||||
|
) -> tuple[dict[str, list[dict[str, Any]]], dict[str, int]]:
|
||||||
|
"""Collapse to one row per (symbol, announce_date), preferring a known
|
||||||
|
session over 'unknown'. Rows must be pre-parsed:
|
||||||
|
{symbol, announce_date: date, session}. Returns {symbol: [events sorted by
|
||||||
|
date]} and dedup stats."""
|
||||||
|
by_key: dict[tuple[str, date], dict[str, Any]] = {}
|
||||||
|
duplicate_rows = 0
|
||||||
|
restated_rows = 0
|
||||||
|
for row in rows:
|
||||||
|
key = (row["symbol"], row["announce_date"])
|
||||||
|
previous = by_key.get(key)
|
||||||
|
if previous is None:
|
||||||
|
by_key[key] = row
|
||||||
|
continue
|
||||||
|
duplicate_rows += 1
|
||||||
|
prev_known = previous["session"] != "unknown"
|
||||||
|
new_known = row["session"] != "unknown"
|
||||||
|
if prev_known and new_known and previous["session"] != row["session"]:
|
||||||
|
restated_rows += 1
|
||||||
|
# Prefer a row that carries a known session.
|
||||||
|
if new_known:
|
||||||
|
by_key[key] = row
|
||||||
|
grouped: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||||
|
for row in by_key.values():
|
||||||
|
grouped[row["symbol"]].append(row)
|
||||||
|
for events in grouped.values():
|
||||||
|
events.sort(key=lambda item: item["announce_date"])
|
||||||
|
return grouped, {
|
||||||
|
"deduped_rows": len(by_key),
|
||||||
|
"duplicate_rows": duplicate_rows,
|
||||||
|
"restated_rows": restated_rows,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def dedup_history(
|
||||||
|
rows: list[dict[str, Any]],
|
||||||
|
) -> tuple[dict[str, list[dict[str, Any]]], dict[str, int]]:
|
||||||
|
"""Collapse to one row per (symbol, period_end_date), preferring the row with
|
||||||
|
more non-null EPS fields. Rows must be pre-parsed:
|
||||||
|
{symbol, period_end_date: date, eps_actual, eps_estimate}."""
|
||||||
|
fields = ("eps_actual", "eps_estimate")
|
||||||
|
by_key: dict[tuple[str, date], dict[str, Any]] = {}
|
||||||
|
duplicate_rows = 0
|
||||||
|
restated_rows = 0
|
||||||
|
for row in rows:
|
||||||
|
key = (row["symbol"], row["period_end_date"])
|
||||||
|
previous = by_key.get(key)
|
||||||
|
if previous is None:
|
||||||
|
by_key[key] = row
|
||||||
|
continue
|
||||||
|
duplicate_rows += 1
|
||||||
|
if any(
|
||||||
|
previous.get(f) is not None
|
||||||
|
and row.get(f) is not None
|
||||||
|
and previous[f] != row[f]
|
||||||
|
for f in fields
|
||||||
|
):
|
||||||
|
restated_rows += 1
|
||||||
|
prev_score = sum(previous.get(f) is not None for f in fields)
|
||||||
|
new_score = sum(row.get(f) is not None for f in fields)
|
||||||
|
if new_score >= prev_score:
|
||||||
|
by_key[key] = row
|
||||||
|
grouped: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||||
|
for row in by_key.values():
|
||||||
|
grouped[row["symbol"]].append(row)
|
||||||
|
for periods in grouped.values():
|
||||||
|
periods.sort(key=lambda item: item["period_end_date"])
|
||||||
|
return grouped, {
|
||||||
|
"deduped_rows": len(by_key),
|
||||||
|
"duplicate_rows": duplicate_rows,
|
||||||
|
"restated_rows": restated_rows,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def match_cost(event: dict[str, Any], period: dict[str, Any]) -> float:
|
||||||
|
delta = (event["announce_date"] - period["period_end_date"]).days
|
||||||
|
penalty = _MISSING_SESSION_PENALTY if event.get("session") == "unknown" else 0.0
|
||||||
|
return float(abs(delta - _TYPICAL_ANNOUNCE_LAG_DAYS)) + penalty
|
||||||
|
|
||||||
|
|
||||||
|
def align_symbol(
|
||||||
|
events: list[dict[str, Any]],
|
||||||
|
periods: list[dict[str, Any]],
|
||||||
|
*,
|
||||||
|
max_lag_days: int,
|
||||||
|
max_lead_days: int,
|
||||||
|
) -> tuple[list[tuple[int, int]], list[int], list[int]]:
|
||||||
|
"""Minimum-cost monotonic calendar-to-period alignment for one symbol.
|
||||||
|
|
||||||
|
Both lists must be sorted ascending (by announce_date / period_end_date). A
|
||||||
|
match is allowed only when ``-max_lead_days <= announce_date - period_end <=
|
||||||
|
max_lag_days``. Returns (matches, unmatched_event_indices,
|
||||||
|
unmatched_period_indices).
|
||||||
|
"""
|
||||||
|
n_events = len(events)
|
||||||
|
n_periods = len(periods)
|
||||||
|
scores = [[0.0] * (n_periods + 1) for _ in range(n_events + 1)]
|
||||||
|
choices = [[""] * (n_periods + 1) for _ in range(n_events + 1)]
|
||||||
|
for e in range(n_events - 1, -1, -1):
|
||||||
|
scores[e][n_periods] = scores[e + 1][n_periods] + SKIP_EVENT_COST
|
||||||
|
choices[e][n_periods] = "event"
|
||||||
|
for p in range(n_periods - 1, -1, -1):
|
||||||
|
scores[n_events][p] = scores[n_events][p + 1] + SKIP_PERIOD_COST
|
||||||
|
choices[n_events][p] = "period"
|
||||||
|
|
||||||
|
for e in range(n_events - 1, -1, -1):
|
||||||
|
for p in range(n_periods - 1, -1, -1):
|
||||||
|
options = [
|
||||||
|
(scores[e + 1][p] + SKIP_EVENT_COST, 2, "event"),
|
||||||
|
(scores[e][p + 1] + SKIP_PERIOD_COST, 1, "period"),
|
||||||
|
]
|
||||||
|
delta = (events[e]["announce_date"] - periods[p]["period_end_date"]).days
|
||||||
|
if -max_lead_days <= delta <= max_lag_days:
|
||||||
|
options.append(
|
||||||
|
(scores[e + 1][p + 1] + match_cost(events[e], periods[p]), 0, "match")
|
||||||
|
)
|
||||||
|
score, _, choice = min(options)
|
||||||
|
scores[e][p] = score
|
||||||
|
choices[e][p] = choice
|
||||||
|
|
||||||
|
matches: list[tuple[int, int]] = []
|
||||||
|
unmatched_events: list[int] = []
|
||||||
|
unmatched_periods: list[int] = []
|
||||||
|
e = p = 0
|
||||||
|
while e < n_events or p < n_periods:
|
||||||
|
if e >= n_events:
|
||||||
|
unmatched_periods.extend(range(p, n_periods))
|
||||||
|
break
|
||||||
|
if p >= n_periods:
|
||||||
|
unmatched_events.extend(range(e, n_events))
|
||||||
|
break
|
||||||
|
choice = choices[e][p]
|
||||||
|
if choice == "match":
|
||||||
|
matches.append((e, p))
|
||||||
|
e += 1
|
||||||
|
p += 1
|
||||||
|
elif choice == "period":
|
||||||
|
unmatched_periods.append(p)
|
||||||
|
p += 1
|
||||||
|
else:
|
||||||
|
unmatched_events.append(e)
|
||||||
|
e += 1
|
||||||
|
return matches, unmatched_events, unmatched_periods
|
||||||
+254
-236
@@ -1,21 +1,9 @@
|
|||||||
"""Event study: does a candidate indicator actually *lead* regime breaks?
|
"""Compact chronological validation for the Regime Monitor warning score.
|
||||||
|
|
||||||
This is a backtest-style measurement, but the unit of analysis is **events**
|
The study calls its outcome a 10% correction, uses the first 70% of sessions to
|
||||||
(historical drawdowns), not trades. For each candidate indicator it answers:
|
freeze an 80th-percentile warning threshold, and reports alarm episodes only on
|
||||||
- how many days of warning did it give before the break (event-centered)?
|
the final 30%. It is still labelled exploratory while the fixed breadth basket
|
||||||
- at what false-alarm cost (signal-centered precision/recall vs. the base rate)?
|
is reconstructed before its freeze date.
|
||||||
|
|
||||||
It compares the breadth-divergence early-warning candidate against a deterministic
|
|
||||||
**coincident** price composite (the existing regime price sub-scores), so you can
|
|
||||||
see whether the candidate crosses *earlier*. Everything is price/breadth only —
|
|
||||||
no LLM/FRED — so the result is reproducible.
|
|
||||||
|
|
||||||
Honest caveat: with only a handful of real drawdowns in ~5y, the sample is tiny
|
|
||||||
and the numbers are noisy. Read the median lead time as an order of magnitude, and
|
|
||||||
do NOT overfit thresholds to this history.
|
|
||||||
|
|
||||||
Report is cached in a SystemSetting (mirrors ``backtest_service``); a manual job
|
|
||||||
(Admin → Jobs) drives it.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -34,304 +22,334 @@ logger = logging.getLogger(__name__)
|
|||||||
|
|
||||||
KEY_REPORT = "regime_event_study"
|
KEY_REPORT = "regime_event_study"
|
||||||
|
|
||||||
# Defaults. The 15% threshold gave only 2 events in 5y (statistically useless),
|
EVENT_THRESHOLD_PCT = 10.0
|
||||||
# so the default is lower with a cooldown-based dedup to surface more, cleaner
|
EVENT_COOLDOWN_DAYS = 40
|
||||||
# events. Each indicator "warns" at its OWN 80th percentile rather than a shared
|
DRAWDOWN_LOOKBACK = 252
|
||||||
# absolute level, so the leading vs. coincident comparison is fair across scales.
|
HORIZON_DAYS = 20
|
||||||
EVENT_THRESHOLD_PCT = 10.0 # drawdown from the 52w high that counts as a "break"
|
WARN_PERCENTILE = 80.0
|
||||||
COOLDOWN_DAYS = 40 # min trading days between event onsets (dedup)
|
TRAIN_FRACTION = 0.70
|
||||||
DRAWDOWN_LOOKBACK = 252 # 52-week trailing high
|
# Below this many holdout corrections, recall is one event away from a very
|
||||||
HORIZON_DAYS = 20 # signal-centered prediction horizon
|
# different headline and should not be read as a property of the score.
|
||||||
WARN_PERCENTILE = 80.0 # each indicator warns at its own Nth percentile
|
MIN_EVENTS_FOR_CONFIDENCE = 8
|
||||||
PRE, POST = 60, 20 # event-centered window (trading days)
|
SENSOR_MISMATCH_TOLERANCE = 0.10
|
||||||
|
|
||||||
|
|
||||||
def _median(values: list[float]) -> float | None:
|
def _median(values: list[float]) -> float | None:
|
||||||
if not values:
|
if not values:
|
||||||
return None
|
return None
|
||||||
s = sorted(values)
|
ordered = sorted(values)
|
||||||
n = len(s)
|
middle = len(ordered) // 2
|
||||||
mid = n // 2
|
return (
|
||||||
return float(s[mid]) if n % 2 else (s[mid - 1] + s[mid]) / 2.0
|
float(ordered[middle])
|
||||||
|
if len(ordered) % 2
|
||||||
|
else (ordered[middle - 1] + ordered[middle]) / 2.0
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _percentile(values: list[float], pct: float) -> float | None:
|
def _percentile(values: list[float], pct: float) -> float | None:
|
||||||
"""Linear-interpolated percentile of the non-None values."""
|
ordered = sorted(v for v in values if v is not None)
|
||||||
vals = sorted(v for v in values if v is not None)
|
if not ordered:
|
||||||
if not vals:
|
|
||||||
return None
|
return None
|
||||||
k = (len(vals) - 1) * (pct / 100.0)
|
position = (len(ordered) - 1) * pct / 100.0
|
||||||
lo = int(k)
|
lower = int(position)
|
||||||
hi = min(lo + 1, len(vals) - 1)
|
upper = min(lower + 1, len(ordered) - 1)
|
||||||
return vals[lo] + (vals[hi] - vals[lo]) * (k - lo)
|
return ordered[lower] + (ordered[upper] - ordered[lower]) * (position - lower)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Event detection
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
def detect_events(
|
def detect_events(
|
||||||
closes: list[float],
|
closes: list[float],
|
||||||
dates: list[date],
|
dates: list[date],
|
||||||
threshold_pct: float = EVENT_THRESHOLD_PCT,
|
threshold_pct: float = EVENT_THRESHOLD_PCT,
|
||||||
lookback: int = DRAWDOWN_LOOKBACK,
|
lookback: int = DRAWDOWN_LOOKBACK,
|
||||||
cooldown: int = COOLDOWN_DAYS,
|
cooldown: int = EVENT_COOLDOWN_DAYS,
|
||||||
) -> list[dict]:
|
) -> list[dict]:
|
||||||
"""Drawdown events: ``t0`` = a day the drawdown from the trailing 52w high
|
"""Rising-edge corrections from the trailing 52-week high."""
|
||||||
crosses up through ``threshold_pct`` (rising edge). De-duplicated by a
|
|
||||||
``cooldown`` of trading days, so a continuous decline counts once but distinct
|
|
||||||
drawdowns separated by a recovery each register."""
|
|
||||||
events: list[dict] = []
|
events: list[dict] = []
|
||||||
prev_dd = 0.0
|
previous_drawdown = 0.0
|
||||||
last_event = -10**9
|
last_event = -10**9
|
||||||
for i in range(len(closes)):
|
for index, close in enumerate(closes):
|
||||||
window = closes[max(0, i - lookback + 1): i + 1]
|
high = max(closes[max(0, index - lookback + 1): index + 1])
|
||||||
hi = max(window)
|
drawdown = (high - close) / high * 100.0 if high > 0 else 0.0
|
||||||
dd = (hi - closes[i]) / hi * 100.0 if hi > 0 else 0.0
|
if (
|
||||||
if dd >= threshold_pct and prev_dd < threshold_pct and (i - last_event) >= cooldown:
|
drawdown >= threshold_pct
|
||||||
events.append({"date": dates[i].isoformat(), "index": i, "depth_pct": round(dd, 1)})
|
and previous_drawdown < threshold_pct
|
||||||
last_event = i
|
and index - last_event >= cooldown
|
||||||
prev_dd = dd
|
):
|
||||||
|
events.append({
|
||||||
|
"date": dates[index].isoformat(),
|
||||||
|
"index": index,
|
||||||
|
"depth_pct": round(drawdown, 1),
|
||||||
|
})
|
||||||
|
last_event = index
|
||||||
|
previous_drawdown = drawdown
|
||||||
return events
|
return events
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
def alarm_episodes(
|
||||||
# Event-centered: lead time + mean path
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
def _lead(indicator: dict[date, float], t0: int, dates: list[date], pre: int, threshold: float) -> int | None:
|
|
||||||
"""Earliest day within ``[t0-pre, t0]`` at which the indicator crosses
|
|
||||||
``threshold`` — i.e. how many days of warning before the event, or None."""
|
|
||||||
lead: int | None = None
|
|
||||||
for k in range(0, pre + 1):
|
|
||||||
idx = t0 - k
|
|
||||||
if idx < 0:
|
|
||||||
break
|
|
||||||
v = indicator.get(dates[idx])
|
|
||||||
if v is not None and v >= threshold:
|
|
||||||
lead = k # keep going: the largest k = earliest warning in the window
|
|
||||||
return lead
|
|
||||||
|
|
||||||
|
|
||||||
def event_centered(
|
|
||||||
indicator: dict[date, float],
|
indicator: dict[date, float],
|
||||||
events_idx: list[int],
|
|
||||||
dates: list[date],
|
dates: list[date],
|
||||||
pre: int = PRE,
|
threshold: float,
|
||||||
post: int = POST,
|
start_index: int = 1,
|
||||||
threshold: float = 60.0,
|
) -> list[int]:
|
||||||
) -> dict:
|
"""Indices where the warning crosses upward; it must reset below first."""
|
||||||
"""Align the indicator at each event's ``t0`` and measure how early it warned.
|
alarms: list[int] = []
|
||||||
|
was_high = False
|
||||||
|
if start_index > 0:
|
||||||
|
previous = indicator.get(dates[start_index - 1])
|
||||||
|
was_high = previous is not None and previous >= threshold
|
||||||
|
for index in range(start_index, len(dates)):
|
||||||
|
value = indicator.get(dates[index])
|
||||||
|
if value is None:
|
||||||
|
continue
|
||||||
|
high = value >= threshold
|
||||||
|
if high and not was_high:
|
||||||
|
alarms.append(index)
|
||||||
|
was_high = high
|
||||||
|
return alarms
|
||||||
|
|
||||||
Lead time is measured against ``threshold`` (each indicator gets its own,
|
|
||||||
derived from its distribution). Also returns the cross-event mean path.
|
def evaluate_alarms(
|
||||||
"""
|
alarm_indices: list[int],
|
||||||
|
event_indices: list[int],
|
||||||
|
dates: list[date],
|
||||||
|
horizon: int = HORIZON_DAYS,
|
||||||
|
) -> dict:
|
||||||
|
"""Event recall, episode false alarms, and lead time for one holdout."""
|
||||||
leads: list[float] = []
|
leads: list[float] = []
|
||||||
sums: dict[int, float] = {}
|
per_event: list[dict] = []
|
||||||
counts: dict[int, int] = {}
|
warned = 0
|
||||||
for t0 in events_idx:
|
for event_index in event_indices:
|
||||||
lead = _lead(indicator, t0, dates, pre, threshold)
|
matching = [
|
||||||
|
alarm for alarm in alarm_indices if 0 < event_index - alarm <= horizon
|
||||||
|
]
|
||||||
|
lead = max((event_index - alarm for alarm in matching), default=None)
|
||||||
if lead is not None:
|
if lead is not None:
|
||||||
leads.append(lead)
|
warned += 1
|
||||||
for rel in range(-pre, post + 1):
|
leads.append(float(lead))
|
||||||
idx = t0 + rel
|
per_event.append({
|
||||||
if 0 <= idx < len(dates):
|
"date": dates[event_index].isoformat(),
|
||||||
v = indicator.get(dates[idx])
|
"warned": lead is not None,
|
||||||
if v is not None:
|
"lead_days": lead,
|
||||||
sums[rel] = sums.get(rel, 0.0) + v
|
})
|
||||||
counts[rel] = counts.get(rel, 0) + 1
|
|
||||||
mean_path = [
|
false_alarms = sum(
|
||||||
{"rel_day": rel, "value": round(sums[rel] / counts[rel], 1)} for rel in sorted(sums)
|
1
|
||||||
]
|
for alarm in alarm_indices
|
||||||
|
if not any(0 < event - alarm <= horizon for event in event_indices)
|
||||||
|
)
|
||||||
return {
|
return {
|
||||||
|
"events": len(event_indices),
|
||||||
|
"events_warned": warned,
|
||||||
|
"events_missed": len(event_indices) - warned,
|
||||||
|
"alarm_episodes": len(alarm_indices),
|
||||||
|
"false_alarms": false_alarms,
|
||||||
"median_lead_days": _median(leads),
|
"median_lead_days": _median(leads),
|
||||||
"events_with_signal": len(leads),
|
"per_event": per_event,
|
||||||
"events_total": len(events_idx),
|
|
||||||
"warn_threshold": round(threshold, 1),
|
|
||||||
"mean_path": mean_path,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
def _warning_series(
|
||||||
# Signal-centered: precision / recall vs. base rate
|
prices: dict[str, rms.Series],
|
||||||
# ---------------------------------------------------------------------------
|
breadth_divergence: dict[date, float],
|
||||||
|
|
||||||
def signal_centered(
|
|
||||||
indicator: dict[date, float],
|
|
||||||
events_idx: list[int],
|
|
||||||
dates: list[date],
|
dates: list[date],
|
||||||
horizon: int = HORIZON_DAYS,
|
config: dict,
|
||||||
thresholds: list[float] | None = None,
|
oas_series: rms.Series | None = None,
|
||||||
) -> dict:
|
) -> tuple[dict[date, float], dict[date, int]]:
|
||||||
"""Treat ``indicator >= threshold`` as predicting a break within ``horizon``
|
"""Warning score per session plus how many sensors backed it.
|
||||||
days. Sweep thresholds → precision/recall/alarm count, plus the base rate."""
|
|
||||||
thresholds = thresholds or [50, 55, 60, 65, 70, 75, 80]
|
|
||||||
n = len(dates)
|
|
||||||
labels = [1 if any(i < e <= i + horizon for e in events_idx) else 0 for i in range(n)]
|
|
||||||
positives = sum(labels)
|
|
||||||
base_rate = positives / n if n else 0.0
|
|
||||||
|
|
||||||
rows: list[dict] = []
|
v2 re-derived this by hand from ``WARNING_WEIGHTS`` and so would have kept
|
||||||
for th in thresholds:
|
measuring the old construct after a scoring change. Since v3 dropped
|
||||||
tp = fp = fn = 0
|
fundamentals from the score, this is now exactly the live Warning score
|
||||||
for i in range(n):
|
rather than a technical-only approximation of it.
|
||||||
v = indicator.get(dates[i])
|
|
||||||
if v is None:
|
|
||||||
continue
|
|
||||||
pred = v >= th
|
|
||||||
if pred and labels[i]:
|
|
||||||
tp += 1
|
|
||||||
elif pred and not labels[i]:
|
|
||||||
fp += 1
|
|
||||||
elif not pred and labels[i]:
|
|
||||||
fn += 1
|
|
||||||
precision = tp / (tp + fp) if (tp + fp) else None
|
|
||||||
recall = tp / (tp + fn) if (tp + fn) else None
|
|
||||||
rows.append({
|
|
||||||
"threshold": th,
|
|
||||||
"precision": round(precision, 3) if precision is not None else None,
|
|
||||||
"recall": round(recall, 3) if recall is not None else None,
|
|
||||||
"alarms": tp + fp,
|
|
||||||
})
|
|
||||||
return {"base_rate": round(base_rate, 3), "horizon_days": horizon, "rows": rows}
|
|
||||||
|
|
||||||
|
The sensor count matters because the score renormalises over whatever is
|
||||||
# ---------------------------------------------------------------------------
|
available: a session backed by two sensors is not drawn from the same
|
||||||
# Coincident baseline (deterministic price composite, reusing the regime sub-scores)
|
distribution as one backed by three, and the frozen threshold assumes it is.
|
||||||
# ---------------------------------------------------------------------------
|
"""
|
||||||
|
tickers = config["tickers"]
|
||||||
def _coincident_series(prices: dict[str, list], dates: list[date], config: dict) -> dict[date, float]:
|
smh_full = prices.get(tickers["leaders"][0], [])
|
||||||
"""Mean of the available price sub-scores (P1-P4) as-of each date — the
|
spy_full = prices.get(tickers["market"], [])
|
||||||
coincident baseline the leading candidate must beat on lead time."""
|
|
||||||
lw = float(config.get("leader_weight", 2.0))
|
|
||||||
lb = int(config.get("rs_lookback", 60))
|
|
||||||
t = config["tickers"]
|
|
||||||
smh_full = prices.get(t["leaders"][0], []) if t["leaders"] else []
|
|
||||||
qqq_full = prices.get(t["confirm"][0], []) if t["confirm"] else []
|
|
||||||
spy_full = prices.get(t["market"], [])
|
|
||||||
out: dict[date, float] = {}
|
out: dict[date, float] = {}
|
||||||
for d in dates:
|
backing: dict[date, int] = {}
|
||||||
smh = rms._closes_asof(smh_full, d)
|
for session in dates:
|
||||||
qqq = rms._closes_asof(qqq_full, d)
|
sensors = rms.warning_sensor_scores(
|
||||||
spy = rms._closes_asof(spy_full, d)
|
breadth_divergence.get(session),
|
||||||
subs = [
|
rms._closes_asof(smh_full, session),
|
||||||
rms.p1_trend_break(smh, qqq, lw),
|
rms._closes_asof(spy_full, session),
|
||||||
rms.p2_death_cross(smh, qqq, lw),
|
rms._window_asof(oas_series, session, rms.HY_OAS_WINDOW_DAYS),
|
||||||
rms.p3_drawdown(smh, qqq),
|
)
|
||||||
rms.p4_relative_strength(smh, spy, lb),
|
score = rms.score_warning_sensors(sensors)
|
||||||
]
|
if score is not None:
|
||||||
vals = [v for v in subs if v is not None]
|
out[session] = round(score, 2)
|
||||||
if vals:
|
backing[session] = sum(1 for value in sensors.values() if value is not None)
|
||||||
out[d] = round(sum(vals) / len(vals), 2)
|
return out, backing
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
def _reliability(
|
||||||
# Orchestration
|
dates: list[date],
|
||||||
# ---------------------------------------------------------------------------
|
split: int,
|
||||||
|
backing: dict[date, int],
|
||||||
|
events_detected: int,
|
||||||
|
events_in_holdout: int,
|
||||||
|
) -> dict:
|
||||||
|
"""How far the headline metrics can actually be trusted.
|
||||||
|
|
||||||
|
Two things repeatedly invite over-reading this report:
|
||||||
|
|
||||||
|
* The holdout carries only the corrections that fall in the last 30% of the
|
||||||
|
sample. A "2/4" is one event away from "3/4", and in practice the events
|
||||||
|
that flip are decided by where the frozen threshold happens to land rather
|
||||||
|
than by whether the score saw anything.
|
||||||
|
* The score renormalises over available sensors, so a training window that
|
||||||
|
predates a sensor's history freezes a threshold on a different construct
|
||||||
|
than the holdout is measured against.
|
||||||
|
"""
|
||||||
|
expected = len(rms.WARNING_WEIGHTS)
|
||||||
|
train = [backing[d] for d in dates[:split] if d in backing]
|
||||||
|
holdout = [backing[d] for d in dates[split:] if d in backing]
|
||||||
|
train_full = sum(1 for n in train if n == expected) / len(train) if train else 0.0
|
||||||
|
holdout_full = sum(1 for n in holdout if n == expected) / len(holdout) if holdout else 0.0
|
||||||
|
return {
|
||||||
|
"events_detected": events_detected,
|
||||||
|
"events_in_holdout": events_in_holdout,
|
||||||
|
"minimum_events": MIN_EVENTS_FOR_CONFIDENCE,
|
||||||
|
"underpowered": events_in_holdout < MIN_EVENTS_FOR_CONFIDENCE,
|
||||||
|
"sensors_expected": expected,
|
||||||
|
"train_full_sensor_share": round(train_full * 100, 1),
|
||||||
|
"holdout_full_sensor_share": round(holdout_full * 100, 1),
|
||||||
|
"sensor_coverage_mismatch": abs(train_full - holdout_full) > SENSOR_MISMATCH_TOLERANCE,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
async def run_event_study(
|
async def run_event_study(
|
||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
threshold_pct: float = EVENT_THRESHOLD_PCT,
|
threshold_pct: float = EVENT_THRESHOLD_PCT,
|
||||||
horizon: int = HORIZON_DAYS,
|
horizon: int = HORIZON_DAYS,
|
||||||
cooldown: int = COOLDOWN_DAYS,
|
|
||||||
warn_percentile: float = WARN_PERCENTILE,
|
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Run the study: detect events on the benchmark, then measure breadth-divergence
|
|
||||||
vs. the coincident price composite. Best-effort; returns available=False on no data."""
|
|
||||||
config = await rms.get_regime_config(db)
|
config = await rms.get_regime_config(db)
|
||||||
end = date.today()
|
end = date.today()
|
||||||
start = end - timedelta(days=5 * 365 + 30)
|
start = end - timedelta(days=5 * 365 + 30)
|
||||||
|
|
||||||
prices = await rms._fetch_prices(config, start, end)
|
prices = await rms._fetch_prices(config, start, end)
|
||||||
leader = config["tickers"]["leaders"][0] if config["tickers"]["leaders"] else "SMH"
|
leader = config["tickers"]["leaders"][0]
|
||||||
bench = sorted(prices.get(leader, []), key=lambda x: x[0])
|
benchmark = sorted(prices.get(leader, []), key=lambda item: item[0])
|
||||||
if len(bench) < 260:
|
if len(benchmark) < 500:
|
||||||
return {"available": False, "reason": "insufficient benchmark history"}
|
return {"available": False, "reason": "insufficient benchmark history"}
|
||||||
|
|
||||||
dates = [d for d, _ in bench]
|
dates = [d for d, _ in benchmark]
|
||||||
closes = [c for _, c in bench]
|
closes = [value for _, value in benchmark]
|
||||||
events = detect_events(closes, dates, threshold_pct, cooldown=cooldown)
|
breadth, _ = await breadth_service.compute_breadth_details(
|
||||||
events_idx = [e["index"] for e in events]
|
db, config["breadth_basket"], window=200, min_tickers=20
|
||||||
|
)
|
||||||
|
divergence = breadth_service.compute_divergence_series(breadth, benchmark)
|
||||||
|
oas_series = await rms._fetch_fred_series("BAMLH0A0HYM2", start, end)
|
||||||
|
warning, backing = _warning_series(prices, divergence, dates, config, oas_series)
|
||||||
|
# The credit sensor cannot reach back as far as the price history does (the
|
||||||
|
# upstream series is capped at ~3 years), so the earlier part of the sample
|
||||||
|
# scores on W1+W2 alone via renormalisation. Report where W3 starts rather
|
||||||
|
# than letting the threshold quietly straddle two sensor sets.
|
||||||
|
credit_from = oas_series[0][0].isoformat() if oas_series else None
|
||||||
|
|
||||||
breadth = await breadth_service.compute_breadth_series(db)
|
split = max(1, min(len(dates) - 1, int(len(dates) * TRAIN_FRACTION)))
|
||||||
divergence = breadth_service.compute_divergence_series(breadth, bench)
|
train_values = [warning[d] for d in dates[:split] if d in warning]
|
||||||
coincident = _coincident_series(prices, dates, config)
|
warn_threshold = _percentile(train_values, WARN_PERCENTILE)
|
||||||
|
if warn_threshold is None:
|
||||||
|
return {"available": False, "reason": "insufficient warning history"}
|
||||||
|
|
||||||
# Each indicator warns at its OWN distribution's percentile, so a leading
|
all_events = detect_events(closes, dates, threshold_pct)
|
||||||
# indicator isn't penalised for living on a different scale than the baseline.
|
holdout_events = [event["index"] for event in all_events if event["index"] >= split]
|
||||||
warn = {
|
alarms = alarm_episodes(warning, dates, warn_threshold, start_index=split)
|
||||||
"breadth_divergence": _percentile(list(divergence.values()), warn_percentile) or 60.0,
|
metrics = evaluate_alarms(alarms, holdout_events, dates, horizon)
|
||||||
"coincident_price": _percentile(list(coincident.values()), warn_percentile) or 60.0,
|
holdout_sessions = max(1, len(dates) - split)
|
||||||
}
|
metrics["false_alarms_per_year"] = round(
|
||||||
series_by_key = {"breadth_divergence": divergence, "coincident_price": coincident}
|
metrics["false_alarms"] / (holdout_sessions / 252.0), 2
|
||||||
|
)
|
||||||
|
|
||||||
def _evaluate(series: dict[date, float], threshold: float) -> dict:
|
reliability = _reliability(dates, split, backing, len(all_events), len(holdout_events))
|
||||||
return {
|
|
||||||
**event_centered(series, events_idx, dates, threshold=threshold),
|
|
||||||
"signal": signal_centered(series, events_idx, dates, horizon),
|
|
||||||
}
|
|
||||||
|
|
||||||
indicators = {key: _evaluate(series_by_key[key], warn[key]) for key in series_by_key}
|
basket_asof = date.fromisoformat(config["basket_asof"])
|
||||||
|
retrospective = dates[split] < basket_asof
|
||||||
# Per-event comparison: which event, and each indicator's lead on THAT event —
|
evaluation = "exploratory" if retrospective else "holdout"
|
||||||
# so a median over a tiny sample can't hide an apples-to-oranges comparison.
|
lead_text = (
|
||||||
per_event = [
|
f"median lead {metrics['median_lead_days']:.0f} sessions"
|
||||||
{
|
if metrics["median_lead_days"] is not None
|
||||||
"date": e["date"],
|
else "no successful warning lead"
|
||||||
"depth_pct": e["depth_pct"],
|
)
|
||||||
"breadth_lead": _lead(divergence, e["index"], dates, PRE, warn["breadth_divergence"]),
|
summary = (
|
||||||
"coincident_lead": _lead(coincident, e["index"], dates, PRE, warn["coincident_price"]),
|
f"{evaluation.capitalize()} chronological test: warning episodes preceded "
|
||||||
}
|
f"{metrics['events_warned']}/{metrics['events']} 10% corrections; "
|
||||||
for e in events
|
f"{metrics['events_missed']} missed, {metrics['false_alarms_per_year']:.1f} "
|
||||||
]
|
f"false alarms/year, {lead_text}. "
|
||||||
|
f"{metrics['events']} of {reliability['events_detected']} detected corrections "
|
||||||
bd = indicators["breadth_divergence"]["median_lead_days"]
|
f"fall in the test period"
|
||||||
cd = indicators["coincident_price"]["median_lead_days"]
|
+ (
|
||||||
lead_delta = (bd - cd) if (bd is not None and cd is not None) else None
|
"; too few to read recall as a property of the score."
|
||||||
|
if reliability["underpowered"]
|
||||||
recent_breadth = [
|
else "."
|
||||||
{"date": d.isoformat(), "breadth": breadth[d], "divergence": divergence.get(d)}
|
)
|
||||||
for d in dates[-90:]
|
)
|
||||||
if d in breadth
|
per_event = metrics.pop("per_event")
|
||||||
]
|
|
||||||
|
|
||||||
report = {
|
report = {
|
||||||
"available": True,
|
"available": True,
|
||||||
|
"methodology": rms.METHODOLOGY,
|
||||||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"evaluation": evaluation,
|
||||||
|
"summary": summary,
|
||||||
"params": {
|
"params": {
|
||||||
"benchmark": leader,
|
"benchmark": leader,
|
||||||
|
"outcome": "10% correction from trailing 52-week high",
|
||||||
"event_threshold_pct": threshold_pct,
|
"event_threshold_pct": threshold_pct,
|
||||||
"cooldown_days": cooldown,
|
"event_cooldown_days": EVENT_COOLDOWN_DAYS,
|
||||||
"horizon_days": horizon,
|
"horizon_days": horizon,
|
||||||
"warn_percentile": warn_percentile,
|
"train_fraction": TRAIN_FRACTION,
|
||||||
|
"warn_percentile": WARN_PERCENTILE,
|
||||||
|
"warn_threshold": round(warn_threshold, 1),
|
||||||
|
"credit_sensor_from": credit_from,
|
||||||
|
"basket_hash": rms._basket_hash(config["breadth_basket"]),
|
||||||
|
"basket_asof": config["basket_asof"],
|
||||||
},
|
},
|
||||||
"events": events,
|
"sample": {
|
||||||
"indicators": indicators,
|
"start": dates[0].isoformat(),
|
||||||
"per_event": per_event,
|
"end": dates[-1].isoformat(),
|
||||||
"lead_delta_days": lead_delta,
|
"train_end": dates[split - 1].isoformat(),
|
||||||
"recent_breadth": recent_breadth,
|
"test_start": dates[split].isoformat(),
|
||||||
|
"sessions": len(dates),
|
||||||
|
"holdout_sessions": holdout_sessions,
|
||||||
|
},
|
||||||
|
"metrics": metrics,
|
||||||
|
"reliability": reliability,
|
||||||
|
"events": per_event,
|
||||||
|
"recent_breadth": [
|
||||||
|
{"date": d.isoformat(), "breadth": breadth[d], "warning": warning.get(d)}
|
||||||
|
for d in dates[-90:]
|
||||||
|
if d in breadth
|
||||||
|
],
|
||||||
}
|
}
|
||||||
logger.info(json.dumps({
|
logger.info(json.dumps({
|
||||||
"event": "event_study_complete", "events": len(events),
|
"event": "regime_event_study_complete",
|
||||||
"breadth_lead": bd, "coincident_lead": cd,
|
"evaluation": evaluation,
|
||||||
|
"events": metrics["events"],
|
||||||
|
"events_detected": reliability["events_detected"],
|
||||||
|
"warned": metrics["events_warned"],
|
||||||
|
"false_alarms_per_year": metrics["false_alarms_per_year"],
|
||||||
|
"underpowered": reliability["underpowered"],
|
||||||
|
"sensor_coverage_mismatch": reliability["sensor_coverage_mismatch"],
|
||||||
}))
|
}))
|
||||||
return report
|
return report
|
||||||
|
|
||||||
|
|
||||||
async def run_and_store(db: AsyncSession) -> dict:
|
async def run_and_store(db: AsyncSession) -> dict:
|
||||||
"""Run the event study and cache the report in a SystemSetting. Job entrypoint."""
|
|
||||||
report = await run_event_study(db)
|
report = await run_event_study(db)
|
||||||
await update_setting(db, KEY_REPORT, json.dumps(report))
|
await update_setting(db, KEY_REPORT, json.dumps(report))
|
||||||
return report
|
return report
|
||||||
|
|
||||||
|
|
||||||
async def get_event_study_report(db: AsyncSession) -> dict | None:
|
async def get_event_study_report(db: AsyncSession) -> dict | None:
|
||||||
"""Return the last cached event-study report, or None if never run."""
|
|
||||||
setting = await settings_store.get_setting(db, KEY_REPORT)
|
setting = await settings_store.get_setting(db, KEY_REPORT)
|
||||||
if setting is None:
|
if setting is None:
|
||||||
return None
|
return None
|
||||||
try:
|
try:
|
||||||
return json.loads(setting.value)
|
report = json.loads(setting.value)
|
||||||
except (TypeError, ValueError):
|
except (TypeError, ValueError):
|
||||||
return None
|
return None
|
||||||
|
return report if report.get("methodology") == rms.METHODOLOGY else None
|
||||||
|
|||||||
@@ -0,0 +1,179 @@
|
|||||||
|
"""A5 activation: refresh the legacy fundamentals cache from local bulk data."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from sqlalchemy import select, update
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.database import insert_for_session
|
||||||
|
from app.models.fundamental import FundamentalData
|
||||||
|
from app.models.score import CompositeScore, DimensionScore
|
||||||
|
from app.services import fundamentals_candidate_service, settings_store
|
||||||
|
|
||||||
|
|
||||||
|
# Absence is deliberately false. Production activation therefore requires one
|
||||||
|
# explicit, durable SystemSetting change after the A5 evidence is approved.
|
||||||
|
ACTIVATION_KEY = "fundamental_data_sec_dolt_cutover_enabled"
|
||||||
|
_SCORE_FIELDS = ("pe_ratio", "revenue_growth", "earnings_surprise")
|
||||||
|
|
||||||
|
|
||||||
|
async def is_enabled(db: AsyncSession) -> bool:
|
||||||
|
raw = await settings_store.get_value(db, ACTIVATION_KEY, "false")
|
||||||
|
return str(raw).strip().lower() == "true"
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh_if_enabled(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
now: datetime | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Refresh atomically when activated; otherwise perform no writes."""
|
||||||
|
if not await is_enabled(db):
|
||||||
|
return {
|
||||||
|
"enabled": False,
|
||||||
|
"refreshed": 0,
|
||||||
|
"score_inputs_changed": 0,
|
||||||
|
"dimension_scores_staled": 0,
|
||||||
|
"composite_scores_staled": 0,
|
||||||
|
}
|
||||||
|
return await refresh(db, now=now, today=today)
|
||||||
|
|
||||||
|
|
||||||
|
async def refresh(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
now: datetime | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Replace every ticker's compat-cache row in one database transaction.
|
||||||
|
|
||||||
|
Candidate values are assembled before the first write and use only local
|
||||||
|
PostgreSQL tables. A failure rolls the whole refresh back. Only changes to
|
||||||
|
the three scoring inputs invalidate cached scores; market cap and the next
|
||||||
|
earnings date are display-only.
|
||||||
|
"""
|
||||||
|
refreshed_at = now or datetime.now(timezone.utc)
|
||||||
|
candidates = await fundamentals_candidate_service.build_candidates(
|
||||||
|
db, today=today
|
||||||
|
)
|
||||||
|
ticker_ids = [candidate.ticker_id for candidate in candidates]
|
||||||
|
existing = await _existing_by_ticker(db, ticker_ids)
|
||||||
|
changed_ids = {
|
||||||
|
candidate.ticker_id
|
||||||
|
for candidate in candidates
|
||||||
|
if _score_inputs_changed(existing.get(candidate.ticker_id), candidate)
|
||||||
|
}
|
||||||
|
|
||||||
|
for candidate in candidates:
|
||||||
|
unavailable_json = json.dumps(
|
||||||
|
candidate.unavailable_fields, sort_keys=True
|
||||||
|
)
|
||||||
|
stmt = insert_for_session(db, FundamentalData).values(
|
||||||
|
ticker_id=candidate.ticker_id,
|
||||||
|
pe_ratio=candidate.pe_ratio,
|
||||||
|
revenue_growth=candidate.revenue_growth,
|
||||||
|
earnings_surprise=candidate.earnings_surprise,
|
||||||
|
market_cap=candidate.market_cap,
|
||||||
|
next_earnings_date=candidate.next_earnings_date,
|
||||||
|
fetched_at=refreshed_at,
|
||||||
|
unavailable_fields_json=unavailable_json,
|
||||||
|
)
|
||||||
|
await db.execute(
|
||||||
|
stmt.on_conflict_do_update(
|
||||||
|
index_elements=["ticker_id"],
|
||||||
|
set_={
|
||||||
|
"pe_ratio": stmt.excluded.pe_ratio,
|
||||||
|
"revenue_growth": stmt.excluded.revenue_growth,
|
||||||
|
"earnings_surprise": stmt.excluded.earnings_surprise,
|
||||||
|
"market_cap": stmt.excluded.market_cap,
|
||||||
|
"next_earnings_date": stmt.excluded.next_earnings_date,
|
||||||
|
"fetched_at": stmt.excluded.fetched_at,
|
||||||
|
"unavailable_fields_json": (
|
||||||
|
stmt.excluded.unavailable_fields_json
|
||||||
|
),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
dimension_ids = await _fundamental_dimension_ids(db, changed_ids)
|
||||||
|
composite_ids = await _composite_ids(db, changed_ids)
|
||||||
|
if dimension_ids:
|
||||||
|
await db.execute(
|
||||||
|
update(DimensionScore)
|
||||||
|
.where(DimensionScore.ticker_id.in_(dimension_ids))
|
||||||
|
.values(is_stale=True)
|
||||||
|
)
|
||||||
|
if composite_ids:
|
||||||
|
await db.execute(
|
||||||
|
update(CompositeScore)
|
||||||
|
.where(CompositeScore.ticker_id.in_(composite_ids))
|
||||||
|
.values(is_stale=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
await db.commit()
|
||||||
|
return {
|
||||||
|
"enabled": True,
|
||||||
|
"refreshed": len(candidates),
|
||||||
|
"score_inputs_changed": len(changed_ids),
|
||||||
|
"dimension_scores_staled": len(dimension_ids),
|
||||||
|
"composite_scores_staled": len(composite_ids),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def _existing_by_ticker(
|
||||||
|
db: AsyncSession, ticker_ids: list[int]
|
||||||
|
) -> dict[int, FundamentalData]:
|
||||||
|
if not ticker_ids:
|
||||||
|
return {}
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalData).where(
|
||||||
|
FundamentalData.ticker_id.in_(ticker_ids)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).scalars()
|
||||||
|
return {row.ticker_id: row for row in rows}
|
||||||
|
|
||||||
|
|
||||||
|
async def _fundamental_dimension_ids(
|
||||||
|
db: AsyncSession, ticker_ids: set[int]
|
||||||
|
) -> set[int]:
|
||||||
|
if not ticker_ids:
|
||||||
|
return set()
|
||||||
|
rows = await db.execute(
|
||||||
|
select(DimensionScore.ticker_id).where(
|
||||||
|
DimensionScore.ticker_id.in_(ticker_ids),
|
||||||
|
DimensionScore.dimension == "fundamental",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return set(rows.scalars())
|
||||||
|
|
||||||
|
|
||||||
|
async def _composite_ids(
|
||||||
|
db: AsyncSession, ticker_ids: set[int]
|
||||||
|
) -> set[int]:
|
||||||
|
if not ticker_ids:
|
||||||
|
return set()
|
||||||
|
rows = await db.execute(
|
||||||
|
select(CompositeScore.ticker_id).where(
|
||||||
|
CompositeScore.ticker_id.in_(ticker_ids)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return set(rows.scalars())
|
||||||
|
|
||||||
|
|
||||||
|
def _score_inputs_changed(
|
||||||
|
existing: FundamentalData | None,
|
||||||
|
candidate: fundamentals_candidate_service.CandidateFundamentals,
|
||||||
|
) -> bool:
|
||||||
|
if existing is None:
|
||||||
|
return True
|
||||||
|
return any(
|
||||||
|
getattr(existing, field) != getattr(candidate, field)
|
||||||
|
for field in _SCORE_FIELDS
|
||||||
|
)
|
||||||
@@ -10,9 +10,10 @@ import json
|
|||||||
import logging
|
import logging
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
from sqlalchemy import select
|
from sqlalchemy import select, update
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.database import insert_for_session
|
||||||
from app.exceptions import NotFoundError
|
from app.exceptions import NotFoundError
|
||||||
from app.models.fundamental import FundamentalData
|
from app.models.fundamental import FundamentalData
|
||||||
from app.models.score import DimensionScore
|
from app.models.score import DimensionScore
|
||||||
@@ -48,51 +49,45 @@ async def store_fundamental(
|
|||||||
"""
|
"""
|
||||||
ticker = await _get_ticker(db, symbol)
|
ticker = await _get_ticker(db, symbol)
|
||||||
|
|
||||||
# Check for existing record
|
|
||||||
result = await db.execute(
|
|
||||||
select(FundamentalData).where(FundamentalData.ticker_id == ticker.id)
|
|
||||||
)
|
|
||||||
existing = result.scalar_one_or_none()
|
|
||||||
|
|
||||||
now = datetime.now(timezone.utc)
|
now = datetime.now(timezone.utc)
|
||||||
unavailable_fields_json = json.dumps(unavailable_fields or {})
|
unavailable_fields_json = json.dumps(unavailable_fields or {})
|
||||||
|
|
||||||
if existing is not None:
|
stmt = insert_for_session(db, FundamentalData).values(
|
||||||
existing.pe_ratio = pe_ratio
|
ticker_id=ticker.id,
|
||||||
existing.revenue_growth = revenue_growth
|
pe_ratio=pe_ratio,
|
||||||
existing.earnings_surprise = earnings_surprise
|
revenue_growth=revenue_growth,
|
||||||
existing.market_cap = market_cap
|
earnings_surprise=earnings_surprise,
|
||||||
existing.next_earnings_date = next_earnings_date
|
market_cap=market_cap,
|
||||||
existing.fetched_at = now
|
next_earnings_date=next_earnings_date,
|
||||||
existing.unavailable_fields_json = unavailable_fields_json
|
fetched_at=now,
|
||||||
record = existing
|
unavailable_fields_json=unavailable_fields_json,
|
||||||
else:
|
)
|
||||||
record = FundamentalData(
|
stmt = stmt.on_conflict_do_update(
|
||||||
ticker_id=ticker.id,
|
index_elements=["ticker_id"],
|
||||||
pe_ratio=pe_ratio,
|
set_={
|
||||||
revenue_growth=revenue_growth,
|
"pe_ratio": stmt.excluded.pe_ratio,
|
||||||
earnings_surprise=earnings_surprise,
|
"revenue_growth": stmt.excluded.revenue_growth,
|
||||||
market_cap=market_cap,
|
"earnings_surprise": stmt.excluded.earnings_surprise,
|
||||||
next_earnings_date=next_earnings_date,
|
"market_cap": stmt.excluded.market_cap,
|
||||||
fetched_at=now,
|
"next_earnings_date": stmt.excluded.next_earnings_date,
|
||||||
unavailable_fields_json=unavailable_fields_json,
|
"fetched_at": stmt.excluded.fetched_at,
|
||||||
)
|
"unavailable_fields_json": stmt.excluded.unavailable_fields_json,
|
||||||
db.add(record)
|
},
|
||||||
|
).returning(FundamentalData)
|
||||||
|
record = (await db.execute(stmt)).scalar_one()
|
||||||
|
|
||||||
# Mark fundamental dimension score as stale if it exists
|
# Mark fundamental dimension score as stale if it exists
|
||||||
# TODO: Use DimensionScore service when built
|
# TODO: Use DimensionScore service when built
|
||||||
dim_result = await db.execute(
|
await db.execute(
|
||||||
select(DimensionScore).where(
|
update(DimensionScore)
|
||||||
|
.where(
|
||||||
DimensionScore.ticker_id == ticker.id,
|
DimensionScore.ticker_id == ticker.id,
|
||||||
DimensionScore.dimension == "fundamental",
|
DimensionScore.dimension == "fundamental",
|
||||||
)
|
)
|
||||||
|
.values(is_stale=True)
|
||||||
)
|
)
|
||||||
dim_score = dim_result.scalar_one_or_none()
|
|
||||||
if dim_score is not None:
|
|
||||||
dim_score.is_stale = True
|
|
||||||
|
|
||||||
await db.commit()
|
await db.commit()
|
||||||
await db.refresh(record)
|
|
||||||
return record
|
return record
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,344 @@
|
|||||||
|
"""Assemble the additive fundamentals API v1 objects (earnings, metrics,
|
||||||
|
valuation, reads) from SEC snapshots + Dolt earnings + the latest price.
|
||||||
|
|
||||||
|
Strictly additive: the router merges these into the existing FundamentalResponse
|
||||||
|
without touching legacy fields. Valuation ratios are computed at REQUEST TIME from
|
||||||
|
the stored snapshots + the latest ohlcv close (no stored valuation). Peer stats are
|
||||||
|
batched and CIK-deduplicated; invalid valuation inputs are guarded to null.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
from collections import defaultdict
|
||||||
|
from datetime import date, datetime
|
||||||
|
from typing import Any
|
||||||
|
from zoneinfo import ZoneInfo
|
||||||
|
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.ohlcv import OHLCVRecord
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import fundamentals_derivation as deriv
|
||||||
|
from app.services import fundamentals_peers as peers
|
||||||
|
from app.services import fundamentals_reads as reads
|
||||||
|
|
||||||
|
# The fixed metric row set — every key always present, value null when unavailable.
|
||||||
|
METRIC_KEYS = (
|
||||||
|
"revenue_growth_yoy", "eps_growth_yoy", "operating_margin", "fcf_margin",
|
||||||
|
"net_debt", "net_debt_to_ebitda", "share_count_change_yoy",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def build_fundamentals_v1(db: AsyncSession, symbol: str, *, today: date | None = None) -> dict[str, Any]:
|
||||||
|
today = today or _ny_today()
|
||||||
|
ticker = await _ticker_by_symbol(db, symbol)
|
||||||
|
|
||||||
|
earnings = await _build_earnings(db, ticker.id, today) if ticker else _empty_earnings()
|
||||||
|
if ticker is None or not ticker.cik:
|
||||||
|
# No SEC identity: metrics present but null, valuation null, empty reads.
|
||||||
|
return {"earnings": earnings, "metrics": _empty_metrics(), "valuation": None,
|
||||||
|
"reads": _empty_reads()}
|
||||||
|
|
||||||
|
subject_cik = ticker.cik
|
||||||
|
derived = deriv.derive((await _snapshots_for(db, [subject_cik])).get(subject_cik, []))
|
||||||
|
|
||||||
|
two = peers.two_digit_sic(ticker.sic)
|
||||||
|
peer_derived: dict[str, deriv.DerivedFundamentals] = {}
|
||||||
|
peer_price_by_cik: dict[str, tuple[float, date] | None] = {}
|
||||||
|
if two:
|
||||||
|
# Subject's representative is the REQUESTED ticker (so its price is used for
|
||||||
|
# the subject in the peer set); other issuers pick a deterministic-by-symbol rep.
|
||||||
|
group = await _peer_group(db, two, subject_cik, ticker.id)
|
||||||
|
peer_snaps = await _snapshots_for(db, list(group))
|
||||||
|
peer_derived = {cik: deriv.derive(rows) for cik, rows in peer_snaps.items()}
|
||||||
|
closes = await _latest_closes(db, set(group.values()))
|
||||||
|
peer_price_by_cik = {cik: closes.get(tid) for cik, tid in group.items()}
|
||||||
|
|
||||||
|
subject_price = await _latest_close(db, ticker.id)
|
||||||
|
metrics = _build_metrics(derived, peer_derived, two)
|
||||||
|
valuation = _build_valuation(derived, subject_price, peer_derived, peer_price_by_cik, two)
|
||||||
|
reads_obj = _build_reads(metrics, valuation)
|
||||||
|
return {"earnings": earnings, "metrics": metrics, "valuation": valuation, "reads": reads_obj}
|
||||||
|
|
||||||
|
|
||||||
|
# -- earnings ----------------------------------------------------------------
|
||||||
|
|
||||||
|
async def _build_earnings(db, ticker_id: int, today: date) -> dict[str, Any]:
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(EarningsEvent).where(EarningsEvent.ticker_id == ticker_id)
|
||||||
|
)).scalars().all()
|
||||||
|
# Same-day earnings are UPCOMING (days_until 0); recent is strictly earlier.
|
||||||
|
upcoming = sorted((e for e in rows if e.announce_date >= today), key=lambda e: e.announce_date)
|
||||||
|
past = sorted((e for e in rows if e.announce_date < today), key=lambda e: e.announce_date, reverse=True)
|
||||||
|
|
||||||
|
nxt = None
|
||||||
|
if upcoming:
|
||||||
|
e = upcoming[0]
|
||||||
|
nxt = {"date": e.announce_date.isoformat(), "session": e.session,
|
||||||
|
"days_until": (e.announce_date - today).days}
|
||||||
|
recent = [{
|
||||||
|
"announce_date": e.announce_date.isoformat(),
|
||||||
|
"period_end": _iso(e.period_end),
|
||||||
|
"eps_estimate": e.eps_estimate,
|
||||||
|
"eps_actual": e.eps_actual,
|
||||||
|
"surprise_pct": _surprise_pct(e.eps_estimate, e.eps_actual),
|
||||||
|
} for e in past[:4]]
|
||||||
|
return {"next": nxt, "recent": recent}
|
||||||
|
|
||||||
|
|
||||||
|
def _surprise_pct(estimate, actual):
|
||||||
|
if estimate is None or actual is None or estimate == 0:
|
||||||
|
return None
|
||||||
|
return round((actual - estimate) / abs(estimate) * 100.0, 2)
|
||||||
|
|
||||||
|
|
||||||
|
# -- metrics -----------------------------------------------------------------
|
||||||
|
|
||||||
|
def _build_metrics(derived, peer_derived, two: str | None) -> list[dict[str, Any]]:
|
||||||
|
out = []
|
||||||
|
for key in METRIC_KEYS:
|
||||||
|
series = derived.metrics.get(key)
|
||||||
|
value = series.value if series else None
|
||||||
|
history = [{"period_end": _iso(p.period_end), "value": p.value} for p in (series.history if series else [])]
|
||||||
|
industry = None
|
||||||
|
if two and peer_derived and key in peers.HIGHER_IS_BETTER:
|
||||||
|
group_values = [
|
||||||
|
(pd.metrics.get(key).value if pd.metrics.get(key) else None)
|
||||||
|
for pd in peer_derived.values()
|
||||||
|
]
|
||||||
|
stat = peers.peer_stat_for(key, value, group_values)
|
||||||
|
if stat:
|
||||||
|
industry = {"label": f"SIC {two} peers", "median": round(stat.median, 4),
|
||||||
|
"favorable_percentile": stat.favorable_percentile, "peer_count": stat.peer_count}
|
||||||
|
out.append({
|
||||||
|
"key": key,
|
||||||
|
"value": value,
|
||||||
|
"history": history,
|
||||||
|
"industry": industry,
|
||||||
|
"period_end": _iso(series.period_end) if series else None,
|
||||||
|
"filed_date": _iso(series.filed_date) if series else None,
|
||||||
|
"caveat": series.caveat if series else None,
|
||||||
|
"source": "sec",
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
# -- valuation (request-time) ------------------------------------------------
|
||||||
|
|
||||||
|
def _build_valuation(derived, subject_price, peer_derived, peer_price_by_cik, two) -> dict[str, Any] | None:
|
||||||
|
if derived.latest_period_end is None:
|
||||||
|
return None # no snapshots yet
|
||||||
|
price = subject_price[0] if subject_price else None
|
||||||
|
price_date = subject_price[1] if subject_price else None
|
||||||
|
if not _finite(price) or price <= 0:
|
||||||
|
return None # no usable price -> valuation null (approved contract)
|
||||||
|
|
||||||
|
pe = _pe(price, derived.ttm_diluted_eps)
|
||||||
|
market_cap = _market_cap(price, derived.shares_outstanding)
|
||||||
|
fcf_yield = _fcf_yield(derived.ttm_fcf, market_cap)
|
||||||
|
|
||||||
|
pe_industry = fcf_yield_industry = None
|
||||||
|
if two and peer_derived:
|
||||||
|
pe_values = [_pe(_p(peer_price_by_cik.get(cik)), pd.ttm_diluted_eps) for cik, pd in peer_derived.items()]
|
||||||
|
fy_values = [
|
||||||
|
_fcf_yield(pd.ttm_fcf, _market_cap(_p(peer_price_by_cik.get(cik)), pd.shares_outstanding))
|
||||||
|
for cik, pd in peer_derived.items()
|
||||||
|
]
|
||||||
|
pe_industry = _industry("pe", pe, pe_values, two)
|
||||||
|
fcf_yield_industry = _industry("fcf_yield", fcf_yield, fy_values, two)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"pe": _round(pe, 2),
|
||||||
|
"fcf_yield": _round(fcf_yield, 2),
|
||||||
|
"market_cap_est": _round(market_cap, 0),
|
||||||
|
# market_cap_est and fcf_yield both rest on the share count. When it came
|
||||||
|
# from the weighted-average diluted fallback (multi-class issuers, whose
|
||||||
|
# per-class cover-page count is absent from companyfacts), say so rather
|
||||||
|
# than presenting a period average as a point-in-time count.
|
||||||
|
"shares_estimated": bool(
|
||||||
|
market_cap is not None and derived.shares_outstanding_estimated
|
||||||
|
),
|
||||||
|
# A null P/E is ambiguous: no earnings data, or earnings we deliberately
|
||||||
|
# suppressed. Only the latter carries a caveat, so a split-contaminated
|
||||||
|
# TTM says why instead of looking like missing data.
|
||||||
|
"pe_caveat": derived.ttm_diluted_eps_caveat if pe is None else None,
|
||||||
|
"pe_industry": pe_industry,
|
||||||
|
"fcf_yield_industry": fcf_yield_industry,
|
||||||
|
"price_date": _iso(price_date),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _pe(price, ttm_eps):
|
||||||
|
if not _finite(price) or price <= 0 or not _finite(ttm_eps) or ttm_eps <= 0:
|
||||||
|
return None
|
||||||
|
return price / ttm_eps
|
||||||
|
|
||||||
|
|
||||||
|
def _market_cap(price, shares):
|
||||||
|
if not _finite(price) or price <= 0 or not _finite(shares) or shares <= 0:
|
||||||
|
return None
|
||||||
|
return price * shares
|
||||||
|
|
||||||
|
|
||||||
|
def _fcf_yield(ttm_fcf, market_cap):
|
||||||
|
if not _finite(ttm_fcf) or not _finite(market_cap) or market_cap <= 0:
|
||||||
|
return None
|
||||||
|
return ttm_fcf / market_cap * 100.0
|
||||||
|
|
||||||
|
|
||||||
|
def _industry(key, subject, group_values, two):
|
||||||
|
stat = peers.peer_stat_for(key, subject, group_values)
|
||||||
|
if stat is None:
|
||||||
|
return None
|
||||||
|
return {"label": f"SIC {two} peers", "median": round(stat.median, 4),
|
||||||
|
"favorable_percentile": stat.favorable_percentile, "peer_count": stat.peer_count}
|
||||||
|
|
||||||
|
|
||||||
|
# -- reads -------------------------------------------------------------------
|
||||||
|
|
||||||
|
_READ_KEYS = METRIC_KEYS + ("pe", "fcf_yield")
|
||||||
|
|
||||||
|
|
||||||
|
def _build_reads(metrics: list[dict], valuation: dict | None) -> dict[str, Any]:
|
||||||
|
by_metric = {m["key"]: m for m in metrics}
|
||||||
|
|
||||||
|
def hist(key):
|
||||||
|
return [_Pt(p["value"]) for p in by_metric.get(key, {}).get("history", [])]
|
||||||
|
|
||||||
|
growth = reads.growth_read(hist("revenue_growth_yoy"))
|
||||||
|
eps_growth = reads.growth_read(hist("eps_growth_yoy"))
|
||||||
|
op_margin = reads.margin_read(hist("operating_margin"))
|
||||||
|
fcf_margin = reads.margin_read(hist("fcf_margin"))
|
||||||
|
share = reads.share_count_read(by_metric.get("share_count_change_yoy", {}).get("value"))
|
||||||
|
leverage = reads.peer_read("net_debt_to_ebitda", _pct(by_metric.get("net_debt_to_ebitda", {}).get("industry")))
|
||||||
|
pe_read = reads.peer_read("pe", _pct(valuation.get("pe_industry"))) if valuation else None
|
||||||
|
fcf_yield_read = reads.peer_read("fcf_yield", _pct(valuation.get("fcf_yield_industry"))) if valuation else None
|
||||||
|
|
||||||
|
# Fixed by_key map over every metric + pe + fcf_yield (null where unavailable).
|
||||||
|
by_key: dict[str, str | None] = {k: None for k in _READ_KEYS}
|
||||||
|
by_key.update({
|
||||||
|
"revenue_growth_yoy": growth,
|
||||||
|
"eps_growth_yoy": eps_growth,
|
||||||
|
"operating_margin": op_margin,
|
||||||
|
"fcf_margin": fcf_margin,
|
||||||
|
"share_count_change_yoy": share,
|
||||||
|
"net_debt_to_ebitda": leverage,
|
||||||
|
"pe": pe_read,
|
||||||
|
"fcf_yield": fcf_yield_read,
|
||||||
|
})
|
||||||
|
header = reads.header_sentence(growth, op_margin, pe_read or fcf_yield_read) or None
|
||||||
|
return {"header": header, "by_key": by_key}
|
||||||
|
|
||||||
|
|
||||||
|
def _empty_reads() -> dict[str, Any]:
|
||||||
|
return {"header": None, "by_key": {k: None for k in _READ_KEYS}}
|
||||||
|
|
||||||
|
|
||||||
|
class _Pt:
|
||||||
|
__slots__ = ("value",)
|
||||||
|
|
||||||
|
def __init__(self, value):
|
||||||
|
self.value = value
|
||||||
|
|
||||||
|
|
||||||
|
def _pct(industry: dict | None):
|
||||||
|
return industry.get("favorable_percentile") if industry else None
|
||||||
|
|
||||||
|
|
||||||
|
# -- queries -----------------------------------------------------------------
|
||||||
|
|
||||||
|
async def _ticker_by_symbol(db, symbol: str) -> Ticker | None:
|
||||||
|
return (await db.execute(
|
||||||
|
select(Ticker).where(Ticker.symbol == symbol.strip().upper())
|
||||||
|
)).scalar_one_or_none()
|
||||||
|
|
||||||
|
|
||||||
|
async def _snapshots_for(db, ciks) -> dict[str, list]:
|
||||||
|
out: dict[str, list] = defaultdict(list)
|
||||||
|
if not ciks:
|
||||||
|
return out
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(FundamentalSnapshot).where(FundamentalSnapshot.cik.in_(list(ciks)))
|
||||||
|
)).scalars().all()
|
||||||
|
for r in rows:
|
||||||
|
out[r.cik].append(r)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _peer_group(db, two: str, subject_cik: str, subject_tid: int) -> dict[str, int]:
|
||||||
|
"""{cik: representative ticker_id} for tracked issuers in the 2-digit SIC group,
|
||||||
|
CIK-deduplicated. Each issuer's representative is its lexicographically-smallest
|
||||||
|
symbol (deterministic), EXCEPT the subject issuer, which uses the requested
|
||||||
|
ticker — so a multi-class subject (GOOGL) is priced by the requested class, not
|
||||||
|
an arbitrary sibling (GOOG)."""
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(Ticker.cik, Ticker.id, Ticker.symbol)
|
||||||
|
.where(Ticker.cik.is_not(None), func.substr(Ticker.sic, 1, 2) == two)
|
||||||
|
)).all()
|
||||||
|
rep: dict[str, tuple[int, str]] = {}
|
||||||
|
for cik, tid, sym in rows:
|
||||||
|
key = sym or ""
|
||||||
|
if cik not in rep or key < rep[cik][1]:
|
||||||
|
rep[cik] = (tid, key)
|
||||||
|
group = {cik: tid for cik, (tid, _) in rep.items()}
|
||||||
|
if subject_cik in group:
|
||||||
|
group[subject_cik] = subject_tid # requested ticker prices the subject
|
||||||
|
return group
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_closes(db, ticker_ids: set[int]) -> dict[int, tuple[float, date]]:
|
||||||
|
if not ticker_ids:
|
||||||
|
return {}
|
||||||
|
latest = (
|
||||||
|
select(OHLCVRecord.ticker_id, func.max(OHLCVRecord.date).label("d"))
|
||||||
|
.where(OHLCVRecord.ticker_id.in_(list(ticker_ids)))
|
||||||
|
.group_by(OHLCVRecord.ticker_id)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(OHLCVRecord.ticker_id, OHLCVRecord.close, OHLCVRecord.date).join(
|
||||||
|
latest, (OHLCVRecord.ticker_id == latest.c.ticker_id) & (OHLCVRecord.date == latest.c.d)
|
||||||
|
)
|
||||||
|
)).all()
|
||||||
|
return {tid: (close, d) for tid, close, d in rows}
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_close(db, ticker_id: int) -> tuple[float, date] | None:
|
||||||
|
return (await _latest_closes(db, {ticker_id})).get(ticker_id)
|
||||||
|
|
||||||
|
|
||||||
|
# -- helpers -----------------------------------------------------------------
|
||||||
|
|
||||||
|
def _empty_metrics() -> list[dict[str, Any]]:
|
||||||
|
return [{"key": k, "value": None, "history": [], "industry": None,
|
||||||
|
"period_end": None, "filed_date": None, "caveat": None,
|
||||||
|
"source": "sec"} for k in METRIC_KEYS]
|
||||||
|
|
||||||
|
|
||||||
|
def _empty_earnings() -> dict[str, Any]:
|
||||||
|
return {"next": None, "recent": []}
|
||||||
|
|
||||||
|
|
||||||
|
def _p(price_tuple):
|
||||||
|
return price_tuple[0] if price_tuple else None
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(v) -> bool:
|
||||||
|
return isinstance(v, (int, float)) and not isinstance(v, bool) and math.isfinite(v)
|
||||||
|
|
||||||
|
|
||||||
|
def _round(v, ndigits):
|
||||||
|
return round(v, ndigits) if _finite(v) else None
|
||||||
|
|
||||||
|
|
||||||
|
def _iso(d) -> str | None:
|
||||||
|
return d.isoformat() if d else None
|
||||||
|
|
||||||
|
|
||||||
|
def _ny_today() -> date:
|
||||||
|
"""Today's New York calendar date — the market's day, not the server's."""
|
||||||
|
return datetime.now(ZoneInfo("America/New_York")).date()
|
||||||
@@ -0,0 +1,288 @@
|
|||||||
|
"""Local SEC/Dolt candidate values for the legacy fundamentals cache.
|
||||||
|
|
||||||
|
This is the single read path shared by the A5 parity report and the activated
|
||||||
|
``fundamental_data`` refresh. It never contacts SEC or Dolt: every input comes
|
||||||
|
from PostgreSQL, so price- and earnings-driven values can still refresh when an
|
||||||
|
upstream import is unchanged or unavailable.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
from collections import defaultdict
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime
|
||||||
|
from typing import Any
|
||||||
|
from zoneinfo import ZoneInfo
|
||||||
|
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.ohlcv import OHLCVRecord
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import fundamentals_derivation as deriv
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CandidateFundamentals:
|
||||||
|
ticker_id: int
|
||||||
|
symbol: str
|
||||||
|
cik: str | None
|
||||||
|
pe_ratio: float | None
|
||||||
|
revenue_growth: float | None
|
||||||
|
earnings_surprise: float | None
|
||||||
|
market_cap: float | None
|
||||||
|
next_earnings_date: date | None
|
||||||
|
price_date: date | None
|
||||||
|
unavailable_fields: dict[str, str] = field(default_factory=dict)
|
||||||
|
|
||||||
|
|
||||||
|
async def build_candidates(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
today: date | None = None,
|
||||||
|
) -> list[CandidateFundamentals]:
|
||||||
|
"""Derive current cache candidates using only already-stored data."""
|
||||||
|
today = today or datetime.now(ZoneInfo("America/New_York")).date()
|
||||||
|
tickers = list(
|
||||||
|
(await db.execute(select(Ticker).order_by(Ticker.symbol))).scalars()
|
||||||
|
)
|
||||||
|
if not tickers:
|
||||||
|
return []
|
||||||
|
|
||||||
|
ticker_ids = [ticker.id for ticker in tickers]
|
||||||
|
ciks = sorted({ticker.cik for ticker in tickers if ticker.cik})
|
||||||
|
derived_by_cik = await _derived_by_cik(db, ciks)
|
||||||
|
closes_by_ticker = await _latest_closes(db, ticker_ids)
|
||||||
|
surprise_by_ticker, next_by_ticker = await _earnings_values(
|
||||||
|
db, ticker_ids, today
|
||||||
|
)
|
||||||
|
|
||||||
|
out: list[CandidateFundamentals] = []
|
||||||
|
for ticker in tickers:
|
||||||
|
derived = derived_by_cik.get(ticker.cik) if ticker.cik else None
|
||||||
|
close = closes_by_ticker.get(ticker.id)
|
||||||
|
price = close[0] if close is not None else None
|
||||||
|
price_date = close[1] if close is not None else None
|
||||||
|
growth_series = (
|
||||||
|
derived.metrics.get("revenue_growth_yoy")
|
||||||
|
if derived is not None
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
|
||||||
|
pe_ratio = (
|
||||||
|
_pe(price, derived.ttm_diluted_eps)
|
||||||
|
if derived is not None
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
revenue_growth = (
|
||||||
|
float(growth_series.value)
|
||||||
|
if growth_series is not None and _finite(growth_series.value)
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
earnings_surprise = surprise_by_ticker.get(ticker.id)
|
||||||
|
market_cap = (
|
||||||
|
_market_cap(price, derived.shares_outstanding)
|
||||||
|
if derived is not None
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
next_earnings_date = next_by_ticker.get(ticker.id)
|
||||||
|
|
||||||
|
out.append(
|
||||||
|
CandidateFundamentals(
|
||||||
|
ticker_id=ticker.id,
|
||||||
|
symbol=ticker.symbol,
|
||||||
|
cik=ticker.cik,
|
||||||
|
pe_ratio=pe_ratio,
|
||||||
|
revenue_growth=revenue_growth,
|
||||||
|
earnings_surprise=earnings_surprise,
|
||||||
|
market_cap=market_cap,
|
||||||
|
next_earnings_date=next_earnings_date,
|
||||||
|
price_date=price_date,
|
||||||
|
unavailable_fields=_availability_metadata(
|
||||||
|
derived=derived,
|
||||||
|
price=price,
|
||||||
|
pe_ratio=pe_ratio,
|
||||||
|
revenue_growth=revenue_growth,
|
||||||
|
earnings_surprise=earnings_surprise,
|
||||||
|
market_cap=market_cap,
|
||||||
|
next_earnings_date=next_earnings_date,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _derived_by_cik(
|
||||||
|
db: AsyncSession, ciks: list[str]
|
||||||
|
) -> dict[str, deriv.DerivedFundamentals]:
|
||||||
|
if not ciks:
|
||||||
|
return {}
|
||||||
|
grouped: dict[str, list[FundamentalSnapshot]] = defaultdict(list)
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalSnapshot).where(FundamentalSnapshot.cik.in_(ciks))
|
||||||
|
)
|
||||||
|
).scalars()
|
||||||
|
for row in rows:
|
||||||
|
grouped[row.cik].append(row)
|
||||||
|
return {cik: deriv.derive(grouped.get(cik, [])) for cik in ciks}
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_closes(
|
||||||
|
db: AsyncSession, ticker_ids: list[int]
|
||||||
|
) -> dict[int, tuple[float, date]]:
|
||||||
|
latest = (
|
||||||
|
select(
|
||||||
|
OHLCVRecord.ticker_id,
|
||||||
|
func.max(OHLCVRecord.date).label("max_date"),
|
||||||
|
)
|
||||||
|
.where(OHLCVRecord.ticker_id.in_(ticker_ids))
|
||||||
|
.group_by(OHLCVRecord.ticker_id)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(
|
||||||
|
OHLCVRecord.ticker_id,
|
||||||
|
OHLCVRecord.close,
|
||||||
|
OHLCVRecord.date,
|
||||||
|
).join(
|
||||||
|
latest,
|
||||||
|
(OHLCVRecord.ticker_id == latest.c.ticker_id)
|
||||||
|
& (OHLCVRecord.date == latest.c.max_date),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
return {
|
||||||
|
ticker_id: (float(close), close_date)
|
||||||
|
for ticker_id, close, close_date in rows
|
||||||
|
if _finite(close)
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def _earnings_values(
|
||||||
|
db: AsyncSession,
|
||||||
|
ticker_ids: list[int],
|
||||||
|
today: date,
|
||||||
|
) -> tuple[dict[int, float], dict[int, date]]:
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(EarningsEvent)
|
||||||
|
.where(EarningsEvent.ticker_id.in_(ticker_ids))
|
||||||
|
.order_by(EarningsEvent.ticker_id, EarningsEvent.announce_date.desc())
|
||||||
|
)
|
||||||
|
).scalars()
|
||||||
|
surprises: dict[int, float] = {}
|
||||||
|
upcoming: dict[int, date] = {}
|
||||||
|
for row in rows:
|
||||||
|
if row.announce_date >= today:
|
||||||
|
current = upcoming.get(row.ticker_id)
|
||||||
|
if current is None or row.announce_date < current:
|
||||||
|
upcoming[row.ticker_id] = row.announce_date
|
||||||
|
continue
|
||||||
|
if row.ticker_id in surprises:
|
||||||
|
continue
|
||||||
|
surprise = _surprise(row.eps_estimate, row.eps_actual)
|
||||||
|
if surprise is not None:
|
||||||
|
surprises[row.ticker_id] = surprise
|
||||||
|
return surprises, upcoming
|
||||||
|
|
||||||
|
|
||||||
|
def _availability_metadata(
|
||||||
|
*,
|
||||||
|
derived: deriv.DerivedFundamentals | None,
|
||||||
|
price: float | None,
|
||||||
|
pe_ratio: float | None,
|
||||||
|
revenue_growth: float | None,
|
||||||
|
earnings_surprise: float | None,
|
||||||
|
market_cap: float | None,
|
||||||
|
next_earnings_date: date | None,
|
||||||
|
) -> dict[str, str]:
|
||||||
|
metadata: dict[str, str] = {}
|
||||||
|
|
||||||
|
if pe_ratio is not None:
|
||||||
|
metadata["source_pe_ratio"] = "sec_facts+ohlcv_records"
|
||||||
|
elif derived is None or derived.latest_period_end is None:
|
||||||
|
metadata["pe_ratio"] = "no SEC fundamental snapshots"
|
||||||
|
elif not _finite(price) or price <= 0:
|
||||||
|
metadata["pe_ratio"] = "no usable PostgreSQL close"
|
||||||
|
elif derived.ttm_diluted_eps_caveat:
|
||||||
|
metadata["pe_ratio"] = derived.ttm_diluted_eps_caveat
|
||||||
|
else:
|
||||||
|
metadata["pe_ratio"] = "no positive SEC-derived TTM diluted EPS"
|
||||||
|
|
||||||
|
if revenue_growth is not None:
|
||||||
|
metadata["source_revenue_growth"] = "sec_facts"
|
||||||
|
else:
|
||||||
|
metadata["revenue_growth"] = "SEC-derived TTM revenue growth unavailable"
|
||||||
|
|
||||||
|
if earnings_surprise is not None:
|
||||||
|
metadata["source_earnings_surprise"] = "dolt_earnings"
|
||||||
|
else:
|
||||||
|
metadata["earnings_surprise"] = (
|
||||||
|
"no completed earnings event with actual and nonzero estimate"
|
||||||
|
)
|
||||||
|
|
||||||
|
if market_cap is not None:
|
||||||
|
metadata["source_market_cap"] = "sec_facts+ohlcv_records"
|
||||||
|
if derived is not None and derived.shares_outstanding_estimated:
|
||||||
|
metadata["market_cap_estimated"] = (
|
||||||
|
"shares use the SEC weighted-average diluted fallback"
|
||||||
|
)
|
||||||
|
elif derived is None or derived.latest_period_end is None:
|
||||||
|
metadata["market_cap"] = "no SEC fundamental snapshots"
|
||||||
|
elif not _finite(price) or price <= 0:
|
||||||
|
metadata["market_cap"] = "no usable PostgreSQL close"
|
||||||
|
else:
|
||||||
|
metadata["market_cap"] = "SEC-derived shares outstanding unavailable"
|
||||||
|
|
||||||
|
if next_earnings_date is not None:
|
||||||
|
metadata["source_next_earnings_date"] = "dolt_earnings"
|
||||||
|
else:
|
||||||
|
metadata["next_earnings_date"] = "no upcoming earnings event"
|
||||||
|
return metadata
|
||||||
|
|
||||||
|
|
||||||
|
def _surprise(
|
||||||
|
estimate: float | None,
|
||||||
|
actual: float | None,
|
||||||
|
) -> float | None:
|
||||||
|
if not _finite(estimate) or not _finite(actual) or estimate == 0:
|
||||||
|
return None
|
||||||
|
return (float(actual) - float(estimate)) / abs(float(estimate)) * 100.0
|
||||||
|
|
||||||
|
|
||||||
|
def _pe(price: float | None, ttm_eps: float | None) -> float | None:
|
||||||
|
if (
|
||||||
|
not _finite(price)
|
||||||
|
or price <= 0
|
||||||
|
or not _finite(ttm_eps)
|
||||||
|
or ttm_eps <= 0
|
||||||
|
):
|
||||||
|
return None
|
||||||
|
return float(price) / float(ttm_eps)
|
||||||
|
|
||||||
|
|
||||||
|
def _market_cap(
|
||||||
|
price: float | None,
|
||||||
|
shares_outstanding: float | None,
|
||||||
|
) -> float | None:
|
||||||
|
if (
|
||||||
|
not _finite(price)
|
||||||
|
or price <= 0
|
||||||
|
or not _finite(shares_outstanding)
|
||||||
|
or shares_outstanding <= 0
|
||||||
|
):
|
||||||
|
return None
|
||||||
|
return float(price) * float(shares_outstanding)
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(value: Any) -> bool:
|
||||||
|
return (
|
||||||
|
isinstance(value, (int, float))
|
||||||
|
and not isinstance(value, bool)
|
||||||
|
and math.isfinite(value)
|
||||||
|
)
|
||||||
@@ -0,0 +1,396 @@
|
|||||||
|
"""Pure read-time derivation of fundamental metrics from stored snapshots.
|
||||||
|
|
||||||
|
`fundamental_snapshots` stores one immutable row per accession with **cumulative
|
||||||
|
YTD** duration facts and period-end balance-sheet instants (A3). This module
|
||||||
|
derives everything the UI/API shows — discrete quarters, Q4, TTM, YoY growth,
|
||||||
|
margins, leverage, dilution, and the quarter tape — at read time, per the plan's
|
||||||
|
schema decision. No I/O, no DB: it takes an issuer's snapshot rows (ORM rows or
|
||||||
|
any objects with the same attributes) and returns structured metrics.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- **Amendment selection:** for each (fiscal_year, fiscal_period), the newest
|
||||||
|
`accepted_at` wins **per field**, falling back to the newest row that actually
|
||||||
|
reports one. A partial amendment (a 10-K/A adding Part III carries no financial
|
||||||
|
facts) must not blank the period.
|
||||||
|
- **Discrete quarter** = YTD(Qn) − YTD(Qn−1); Q1 = YTD(Q1); **Q4 = YTD(FY) −
|
||||||
|
YTD(Q3)**. Any missing period → the derived value is null, never partial.
|
||||||
|
- **TTM** = sum of the trailing four discrete quarters ending at a period.
|
||||||
|
- Units follow app convention: percentages are percentage points (21.0 = 21%),
|
||||||
|
net-debt/EBITDA is a multiple, net debt is dollars.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date
|
||||||
|
from types import SimpleNamespace
|
||||||
|
from typing import Any, Iterable
|
||||||
|
|
||||||
|
_FP_TO_Q = {"Q1": 1, "Q2": 2, "Q3": 3, "FY": 4}
|
||||||
|
_Q_TO_FP = {1: "Q1", 2: "Q2", 3: "Q3", 4: "FY"}
|
||||||
|
_PREV_FP = {"Q2": "Q1", "Q3": "Q2", "FY": "Q3"}
|
||||||
|
TAPE_LEN = 4 # quarter-tape length
|
||||||
|
SPLIT_SUSPECT_SHARE_CHANGE_PCT = 25.0
|
||||||
|
SPLIT_SENSITIVE_CAVEAT = (
|
||||||
|
"Not comparable: share count changed at least 25%; possible split or "
|
||||||
|
"corporate action."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Duration (flow) fields differenced from YTD into discrete quarters + summed to TTM.
|
||||||
|
_FLOW_FIELDS = (
|
||||||
|
"revenue", "net_income", "operating_income", "diluted_eps", "cfo", "capex",
|
||||||
|
"depreciation_amortization",
|
||||||
|
)
|
||||||
|
# Reported facts resolved independently across a period's accessions (see
|
||||||
|
# _merge_amendments); period identity/provenance is taken from the newest one.
|
||||||
|
_MERGED_FIELDS = (
|
||||||
|
*_FLOW_FIELDS,
|
||||||
|
"cash_and_st_investments", "total_debt", "shares_outstanding",
|
||||||
|
"shares_outstanding_date", "weighted_avg_diluted_shares",
|
||||||
|
# period_start is set alongside revenue by the parser, so it follows the same
|
||||||
|
# fallback: a bare amendment reports neither and must not blank it.
|
||||||
|
"period_start",
|
||||||
|
)
|
||||||
|
_CARRIED_FIELDS = (
|
||||||
|
"fiscal_year", "fiscal_period", "period_end", "filed_date",
|
||||||
|
"accepted_at", "form", "accession", "cik",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class MetricPoint:
|
||||||
|
period_end: date
|
||||||
|
value: float | None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class MetricSeries:
|
||||||
|
value: float | None = None
|
||||||
|
history: list[MetricPoint] = field(default_factory=list) # oldest -> newest, <= TAPE_LEN
|
||||||
|
period_end: date | None = None
|
||||||
|
filed_date: date | None = None
|
||||||
|
caveat: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class DerivedFundamentals:
|
||||||
|
metrics: dict[str, MetricSeries] = field(default_factory=dict)
|
||||||
|
# request-time valuation inputs (ratios are computed in the API with price)
|
||||||
|
ttm_diluted_eps: float | None = None
|
||||||
|
# Set when ttm_diluted_eps was suppressed rather than simply unavailable.
|
||||||
|
ttm_diluted_eps_caveat: str | None = None
|
||||||
|
ttm_fcf: float | None = None
|
||||||
|
shares_outstanding: float | None = None
|
||||||
|
# True when shares_outstanding came from the weighted-average diluted count
|
||||||
|
# because the point-in-time cover-page count was absent (always so for
|
||||||
|
# multi-class issuers). Consumers must label anything derived from it as
|
||||||
|
# estimated — it is a period average, not a point-in-time count.
|
||||||
|
shares_outstanding_estimated: bool = False
|
||||||
|
latest_period_end: date | None = None
|
||||||
|
latest_filed_date: date | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _prev_q(fy: int, q: int) -> tuple[int, int]:
|
||||||
|
return (fy, q - 1) if q > 1 else (fy - 1, 4)
|
||||||
|
|
||||||
|
|
||||||
|
def derive(snapshots: Iterable[Any]) -> DerivedFundamentals:
|
||||||
|
selected = _select_latest_per_period(snapshots)
|
||||||
|
result = DerivedFundamentals()
|
||||||
|
if not selected:
|
||||||
|
return result
|
||||||
|
|
||||||
|
# Discrete quarter values per flow field: {field: {(fy, q): value}}.
|
||||||
|
discrete = {f: _discrete_quarters(selected, f) for f in _FLOW_FIELDS}
|
||||||
|
quarters = _ordered_quarters(selected) # chronological (fy, q) with a row
|
||||||
|
latest = quarters[-1]
|
||||||
|
latest_row = selected[(latest[0], _Q_TO_FP[latest[1]])]
|
||||||
|
|
||||||
|
result.latest_period_end = latest_row.period_end
|
||||||
|
result.latest_filed_date = latest_row.filed_date
|
||||||
|
result.shares_outstanding = getattr(latest_row, "shares_outstanding", None)
|
||||||
|
if result.shares_outstanding is None:
|
||||||
|
# Multi-class issuers (META, CMCSA, BRK-B, CHTR, FOXA, NWSA, LEN) report
|
||||||
|
# the cover-page count per class, which is dimensional and so absent from
|
||||||
|
# companyfacts — leaving market cap and FCF yield silently unavailable for
|
||||||
|
# some of the largest names. The weighted-average diluted count is always
|
||||||
|
# present and within ~0.6% of the true count where both exist, so fall
|
||||||
|
# back to it and mark the result estimated rather than show nothing.
|
||||||
|
result.shares_outstanding = getattr(latest_row, "weighted_avg_diluted_shares", None)
|
||||||
|
result.shares_outstanding_estimated = result.shares_outstanding is not None
|
||||||
|
result.ttm_diluted_eps = _ttm(discrete["diluted_eps"], *latest)
|
||||||
|
ttm_cfo = _ttm(discrete["cfo"], *latest)
|
||||||
|
ttm_capex = _ttm(discrete["capex"], *latest)
|
||||||
|
result.ttm_fcf = None if ttm_cfo is None or ttm_capex is None else ttm_cfo - ttm_capex
|
||||||
|
|
||||||
|
# tape = the CONSECUTIVE run of up to TAPE_LEN quarters ending at the latest,
|
||||||
|
# stopping at a gap — so trend text never compares non-adjacent periods.
|
||||||
|
tape = _consecutive_suffix(quarters, TAPE_LEN)
|
||||||
|
result.metrics = {
|
||||||
|
"revenue_growth_yoy": _yoy_growth_series(discrete["revenue"], selected, tape),
|
||||||
|
"eps_growth_yoy": _yoy_growth_series(discrete["diluted_eps"], selected, tape),
|
||||||
|
"operating_margin": _margin_series(discrete["operating_income"], discrete["revenue"], selected, tape),
|
||||||
|
"fcf_margin": _fcf_margin_series(discrete, selected, tape),
|
||||||
|
"net_debt": _instant_series(selected, tape, _net_debt),
|
||||||
|
"net_debt_to_ebitda": _leverage_series(selected, discrete, tape),
|
||||||
|
"share_count_change_yoy": _share_change_series(selected, tape),
|
||||||
|
}
|
||||||
|
# TTM EPS sums four quarters of *per-share* values, so a split inside that
|
||||||
|
# window mixes pre- and post-split units — the same distortion the guard
|
||||||
|
# already catches for the series, and the one that produced BKNG's P/E of
|
||||||
|
# 1.10. Left unguarded it does not merely mislead: a nonsense-low P/E clamps
|
||||||
|
# to a perfect 100 fundamental sub-score, so it must null out like the rest.
|
||||||
|
if _guard_split_sensitive_metrics(result.metrics):
|
||||||
|
result.ttm_diluted_eps = None
|
||||||
|
result.ttm_diluted_eps_caveat = SPLIT_SENSITIVE_CAVEAT
|
||||||
|
for series in result.metrics.values():
|
||||||
|
series.period_end = latest_row.period_end
|
||||||
|
series.filed_date = latest_row.filed_date
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
# -- period selection --------------------------------------------------------
|
||||||
|
|
||||||
|
def _select_latest_per_period(snapshots: Iterable[Any]) -> dict[tuple[int, str], Any]:
|
||||||
|
grouped: dict[tuple[int, str], list[Any]] = {}
|
||||||
|
for row in snapshots:
|
||||||
|
fp = getattr(row, "fiscal_period", None)
|
||||||
|
fy = getattr(row, "fiscal_year", None)
|
||||||
|
if fp not in _FP_TO_Q or fy is None:
|
||||||
|
continue
|
||||||
|
grouped.setdefault((fy, fp), []).append(row)
|
||||||
|
return {key: _merge_amendments(rows) for key, rows in grouped.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def _merge_amendments(rows: list[Any]) -> Any:
|
||||||
|
"""Resolve one period from its accessions: newest wins, per field.
|
||||||
|
|
||||||
|
Amendments are frequently partial — a 10-K/A filed only to add Part III
|
||||||
|
reports no financial facts at all. Taking the newest accession wholesale
|
||||||
|
would blank every field it omits and null the period downstream (and with
|
||||||
|
it TTM and YoY, which need an unbroken quarter chain), so each field falls
|
||||||
|
back to the newest accession that actually reports it.
|
||||||
|
|
||||||
|
Only rows sharing the newest row's ``period_end`` are merged. A same-key row
|
||||||
|
covering a *different* period is a mislabelled filing, not an amendment, and
|
||||||
|
blending the two would silently mix fiscal years.
|
||||||
|
"""
|
||||||
|
if len(rows) == 1:
|
||||||
|
return rows[0]
|
||||||
|
ordered = sorted(rows, key=_amendment_order, reverse=True) # newest first
|
||||||
|
newest = ordered[0]
|
||||||
|
same_period = [
|
||||||
|
row
|
||||||
|
for row in ordered
|
||||||
|
if getattr(row, "period_end", None) == getattr(newest, "period_end", None)
|
||||||
|
]
|
||||||
|
if len(same_period) == 1:
|
||||||
|
return newest
|
||||||
|
merged = SimpleNamespace(**{name: getattr(newest, name, None) for name in _CARRIED_FIELDS})
|
||||||
|
for name in _MERGED_FIELDS:
|
||||||
|
merged_value = None
|
||||||
|
for row in same_period: # newest first
|
||||||
|
value = getattr(row, name, None)
|
||||||
|
if value is not None:
|
||||||
|
merged_value = value
|
||||||
|
break
|
||||||
|
setattr(merged, name, merged_value)
|
||||||
|
return merged
|
||||||
|
|
||||||
|
|
||||||
|
def _amendment_order(row: Any) -> tuple[bool, Any]:
|
||||||
|
# (has-timestamp, timestamp) so a row without one sorts oldest instead of
|
||||||
|
# raising when compared against a row that has one.
|
||||||
|
accepted = _accepted(row)
|
||||||
|
return (accepted is not None, accepted)
|
||||||
|
|
||||||
|
|
||||||
|
def _accepted(row: Any):
|
||||||
|
return getattr(row, "accepted_at", None) or getattr(row, "filed_date", None)
|
||||||
|
|
||||||
|
|
||||||
|
def _ordered_quarters(selected: dict[tuple[int, str], Any]) -> list[tuple[int, int]]:
|
||||||
|
return sorted((fy, _FP_TO_Q[fp]) for (fy, fp) in selected)
|
||||||
|
|
||||||
|
|
||||||
|
def _consecutive_suffix(quarters: list[tuple[int, int]], n: int) -> list[tuple[int, int]]:
|
||||||
|
"""The run of up to n quarters ending at the latest, walking back only through
|
||||||
|
adjacent periods (stop at the first gap). Returned oldest -> newest."""
|
||||||
|
if not quarters:
|
||||||
|
return []
|
||||||
|
present = set(quarters)
|
||||||
|
run = [quarters[-1]]
|
||||||
|
cur = quarters[-1]
|
||||||
|
while len(run) < n:
|
||||||
|
prev = _prev_q(*cur)
|
||||||
|
if prev not in present:
|
||||||
|
break
|
||||||
|
run.append(prev)
|
||||||
|
cur = prev
|
||||||
|
run.reverse()
|
||||||
|
return run
|
||||||
|
|
||||||
|
|
||||||
|
# -- discrete + TTM ----------------------------------------------------------
|
||||||
|
|
||||||
|
def _discrete_quarters(selected: dict[tuple[int, str], Any], field_name: str) -> dict[tuple[int, int], float]:
|
||||||
|
out: dict[tuple[int, int], float] = {}
|
||||||
|
for (fy, fp), row in selected.items():
|
||||||
|
val = _discrete_value(selected, fy, fp, field_name)
|
||||||
|
if val is not None:
|
||||||
|
out[(fy, _FP_TO_Q[fp])] = val
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _discrete_value(selected, fy: int, fp: str, field_name: str) -> float | None:
|
||||||
|
cur = getattr(selected[(fy, fp)], field_name, None)
|
||||||
|
if cur is None:
|
||||||
|
return None
|
||||||
|
if fp == "Q1":
|
||||||
|
return cur
|
||||||
|
prev = selected.get((fy, _PREV_FP[fp]))
|
||||||
|
prev_val = getattr(prev, field_name, None) if prev is not None else None
|
||||||
|
if prev_val is None:
|
||||||
|
return None
|
||||||
|
return cur - prev_val
|
||||||
|
|
||||||
|
|
||||||
|
def _ttm(dq: dict[tuple[int, int], float], fy: int, q: int) -> float | None:
|
||||||
|
keys = [(fy, q)]
|
||||||
|
k = (fy, q)
|
||||||
|
for _ in range(3):
|
||||||
|
k = _prev_q(*k)
|
||||||
|
keys.append(k)
|
||||||
|
vals = [dq.get(kk) for kk in keys]
|
||||||
|
if any(v is None for v in vals):
|
||||||
|
return None
|
||||||
|
return sum(vals)
|
||||||
|
|
||||||
|
|
||||||
|
def _pct_change(cur: float | None, prior: float | None) -> float | None:
|
||||||
|
# A non-positive prior makes a YoY % meaningless (e.g. loss->profit), so null it.
|
||||||
|
if cur is None or prior is None or prior <= 0:
|
||||||
|
return None
|
||||||
|
return (cur / prior - 1.0) * 100.0
|
||||||
|
|
||||||
|
|
||||||
|
# -- per-metric series (value at latest + tape history) ----------------------
|
||||||
|
|
||||||
|
def _period_end(selected, fy: int, q: int) -> date | None:
|
||||||
|
row = selected.get((fy, _Q_TO_FP[q]))
|
||||||
|
return row.period_end if row is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _yoy_growth_series(dq, selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
cur, prior = _ttm(dq, fy, q), _ttm(dq, fy - 1, q)
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), _pct_change(cur, prior)))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _margin_series(num_dq, den_dq, selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
num, den = _ttm(num_dq, fy, q), _ttm(den_dq, fy, q)
|
||||||
|
val = None if num is None or not den else num / den * 100.0
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), val))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _fcf_margin_series(discrete, selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
cfo, capex, rev = _ttm(discrete["cfo"], fy, q), _ttm(discrete["capex"], fy, q), _ttm(discrete["revenue"], fy, q)
|
||||||
|
val = None if cfo is None or capex is None or not rev else (cfo - capex) / rev * 100.0
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), val))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _instant_series(selected, tape, fn) -> MetricSeries:
|
||||||
|
pts = [MetricPoint(_period_end(selected, fy, q), fn(selected.get((fy, _Q_TO_FP[q])))) for (fy, q) in tape]
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _leverage_series(selected, discrete, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
row = selected.get((fy, _Q_TO_FP[q]))
|
||||||
|
nd = _net_debt(row)
|
||||||
|
op, da = _ttm(discrete["operating_income"], fy, q), _ttm(discrete["depreciation_amortization"], fy, q)
|
||||||
|
ebitda = None if op is None or da is None else op + da
|
||||||
|
# Null when EBITDA <= 0: a negative denominator would flip polarity and a
|
||||||
|
# "lower is better" read would rank a distressed issuer as favorable.
|
||||||
|
val = None if nd is None or ebitda is None or ebitda <= 0 else nd / ebitda
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), val))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _share_change_series(selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
cur = _shares(selected.get((fy, _Q_TO_FP[q])))
|
||||||
|
prior = _shares(selected.get((fy - 1, _Q_TO_FP[q])))
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), _pct_change(cur, prior)))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _guard_split_sensitive_metrics(metrics: dict[str, MetricSeries]) -> bool:
|
||||||
|
"""Suppress historical comparisons likely distorted by a corporate action.
|
||||||
|
|
||||||
|
Company Facts has no point-in-time split factors. A large YoY share-count
|
||||||
|
move can therefore make both the point-in-time share comparison and
|
||||||
|
per-share EPS growth non-comparable. Keep the raw facts in snapshots, but
|
||||||
|
expose nulls plus an explicit caveat in the user-facing derived series.
|
||||||
|
|
||||||
|
Returns True when the *latest* period is suspect, so callers can apply the
|
||||||
|
same suppression to per-share scalars derived from that window.
|
||||||
|
"""
|
||||||
|
shares = metrics.get("share_count_change_yoy")
|
||||||
|
eps = metrics.get("eps_growth_yoy")
|
||||||
|
if shares is None or eps is None:
|
||||||
|
return False
|
||||||
|
|
||||||
|
suspect_periods = {
|
||||||
|
point.period_end
|
||||||
|
for point in shares.history
|
||||||
|
if point.value is not None
|
||||||
|
and abs(point.value) >= SPLIT_SUSPECT_SHARE_CHANGE_PCT
|
||||||
|
}
|
||||||
|
if not suspect_periods:
|
||||||
|
return False
|
||||||
|
|
||||||
|
latest_suspect = False
|
||||||
|
for series in (shares, eps):
|
||||||
|
latest_guarded = bool(
|
||||||
|
series.history and series.history[-1].period_end in suspect_periods
|
||||||
|
)
|
||||||
|
latest_suspect = latest_suspect or latest_guarded
|
||||||
|
for point in series.history:
|
||||||
|
if point.period_end in suspect_periods:
|
||||||
|
point.value = None
|
||||||
|
series.value = series.history[-1].value if series.history else None
|
||||||
|
if latest_guarded:
|
||||||
|
series.caveat = SPLIT_SENSITIVE_CAVEAT
|
||||||
|
return latest_suspect
|
||||||
|
|
||||||
|
|
||||||
|
def _net_debt(row: Any) -> float | None:
|
||||||
|
if row is None:
|
||||||
|
return None
|
||||||
|
cash = getattr(row, "cash_and_st_investments", None)
|
||||||
|
debt = getattr(row, "total_debt", None)
|
||||||
|
# Require BOTH components — treating a missing side as zero would produce a
|
||||||
|
# partial, misleading value.
|
||||||
|
if cash is None or debt is None:
|
||||||
|
return None
|
||||||
|
return debt - cash # positive = net debt
|
||||||
|
|
||||||
|
|
||||||
|
def _shares(row: Any) -> float | None:
|
||||||
|
return getattr(row, "shares_outstanding", None) if row is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _series(points: list[MetricPoint]) -> MetricSeries:
|
||||||
|
value = points[-1].value if points else None
|
||||||
|
return MetricSeries(value=value, history=points)
|
||||||
@@ -0,0 +1,498 @@
|
|||||||
|
"""Read-only A5 comparison of legacy and SEC/Dolt fundamental inputs.
|
||||||
|
|
||||||
|
The report deliberately does not write ``fundamental_data`` or score tables.
|
||||||
|
It reconstructs the current legacy and candidate fundamental scores, projects
|
||||||
|
their composite-score/rank effect with the active weights, and archives a
|
||||||
|
timestamped JSON + CSV bundle for explicit human approval.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import json
|
||||||
|
import math
|
||||||
|
import os
|
||||||
|
import statistics
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Iterable
|
||||||
|
from zoneinfo import ZoneInfo
|
||||||
|
|
||||||
|
from sqlalchemy import select, text
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
|
from app.models.fundamental import FundamentalData
|
||||||
|
from app.services import fundamentals_candidate_service as candidate_service
|
||||||
|
|
||||||
|
REPORT_VERSION = 1
|
||||||
|
APPROVAL_STATUS = "pending_explicit_approval"
|
||||||
|
FIELD_KEYS = ("pe_ratio", "revenue_growth", "earnings_surprise")
|
||||||
|
MIN_SCORE_METRICS = 2
|
||||||
|
|
||||||
|
# Materiality is a review aid, never an automatic cutover verdict. Definition
|
||||||
|
# changes remain visible even when a delta falls inside these bands.
|
||||||
|
FIELD_TOLERANCES = {
|
||||||
|
"pe_ratio": {"absolute": 1.0, "relative_pct": 10.0},
|
||||||
|
"revenue_growth": {"absolute": 2.0, "relative_pct": None},
|
||||||
|
"earnings_surprise": {"absolute": 2.0, "relative_pct": None},
|
||||||
|
}
|
||||||
|
DEFINITION_NOTES = {
|
||||||
|
"pe_ratio": (
|
||||||
|
"Legacy provider P/E convention versus latest close divided by "
|
||||||
|
"SEC-derived TTM diluted EPS."
|
||||||
|
),
|
||||||
|
"revenue_growth": (
|
||||||
|
"Legacy provider growth convention versus SEC-derived TTM revenue YoY."
|
||||||
|
),
|
||||||
|
"earnings_surprise": (
|
||||||
|
"Legacy provider latest surprise versus latest completed Dolt earnings "
|
||||||
|
"event with actual and estimate."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def fundamental_score(
|
||||||
|
pe_ratio: float | None,
|
||||||
|
revenue_growth: float | None,
|
||||||
|
earnings_surprise: float | None,
|
||||||
|
) -> float | None:
|
||||||
|
"""Match the production fundamental-dimension formula without persistence."""
|
||||||
|
scores: list[float] = []
|
||||||
|
if _finite(pe_ratio) and pe_ratio > 0:
|
||||||
|
scores.append(max(0.0, min(100.0, 100.0 - (pe_ratio - 15.0) * (100.0 / 30.0))))
|
||||||
|
if _finite(revenue_growth):
|
||||||
|
scores.append(max(0.0, min(100.0, 50.0 + revenue_growth * 2.5)))
|
||||||
|
if _finite(earnings_surprise):
|
||||||
|
scores.append(max(0.0, min(100.0, 50.0 + earnings_surprise * 5.0)))
|
||||||
|
return sum(scores) / len(scores) if len(scores) >= MIN_SCORE_METRICS else None
|
||||||
|
|
||||||
|
|
||||||
|
async def build_report(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
generated_at: datetime | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Build a point-in-time parity report from one database session."""
|
||||||
|
generated_at = generated_at or datetime.now(timezone.utc)
|
||||||
|
today = today or datetime.now(ZoneInfo("America/New_York")).date()
|
||||||
|
|
||||||
|
# A report must not mix rows from before and after a concurrent import
|
||||||
|
# promotion. The scheduled job provides a fresh session, so establish the
|
||||||
|
# production snapshot before its first query and have Postgres enforce the
|
||||||
|
# no-write contract as well. SQLite tests retain their normal transaction.
|
||||||
|
if db.get_bind().dialect.name == "postgresql":
|
||||||
|
connection = await db.connection(
|
||||||
|
execution_options={"isolation_level": "REPEATABLE READ"}
|
||||||
|
)
|
||||||
|
await connection.execute(text("SET TRANSACTION READ ONLY"))
|
||||||
|
|
||||||
|
candidates = await candidate_service.build_candidates(db, today=today)
|
||||||
|
ticker_ids = [candidate.ticker_id for candidate in candidates]
|
||||||
|
legacy_by_ticker = await _legacy_values(db, ticker_ids)
|
||||||
|
source_runs = await _source_runs(db)
|
||||||
|
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
for candidate in candidates:
|
||||||
|
legacy = legacy_by_ticker.get(candidate.ticker_id)
|
||||||
|
candidate_values = {
|
||||||
|
"pe_ratio": candidate.pe_ratio,
|
||||||
|
"revenue_growth": candidate.revenue_growth,
|
||||||
|
"earnings_surprise": candidate.earnings_surprise,
|
||||||
|
}
|
||||||
|
legacy_values = {
|
||||||
|
"pe_ratio": legacy.pe_ratio if legacy else None,
|
||||||
|
"revenue_growth": legacy.revenue_growth if legacy else None,
|
||||||
|
"earnings_surprise": legacy.earnings_surprise if legacy else None,
|
||||||
|
}
|
||||||
|
fields = {
|
||||||
|
key: _field_comparison(key, legacy_values[key], candidate_values[key])
|
||||||
|
for key in FIELD_KEYS
|
||||||
|
}
|
||||||
|
legacy_score = fundamental_score(**legacy_values)
|
||||||
|
candidate_score = fundamental_score(**candidate_values)
|
||||||
|
rows.append(
|
||||||
|
{
|
||||||
|
"symbol": candidate.symbol,
|
||||||
|
"cik": candidate.cik,
|
||||||
|
"legacy_fetched_at": _iso(legacy.fetched_at) if legacy else None,
|
||||||
|
"price_date": _iso(candidate.price_date),
|
||||||
|
"fields": fields,
|
||||||
|
"scores": {
|
||||||
|
"legacy_fundamental": _round(legacy_score),
|
||||||
|
"candidate_fundamental": _round(candidate_score),
|
||||||
|
"fundamental_delta": _delta(legacy_score, candidate_score),
|
||||||
|
"legacy_fundamental_rank": None,
|
||||||
|
"candidate_fundamental_rank": None,
|
||||||
|
"fundamental_rank_change": None,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
_attach_ranks(rows, "legacy_fundamental", "legacy_fundamental_rank")
|
||||||
|
_attach_ranks(rows, "candidate_fundamental", "candidate_fundamental_rank")
|
||||||
|
for row in rows:
|
||||||
|
scores = row["scores"]
|
||||||
|
scores["fundamental_rank_change"] = _rank_change(
|
||||||
|
scores["legacy_fundamental_rank"], scores["candidate_fundamental_rank"]
|
||||||
|
)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"report_version": REPORT_VERSION,
|
||||||
|
"generated_at": generated_at.isoformat(),
|
||||||
|
"as_of_date": today.isoformat(),
|
||||||
|
"approval_status": APPROVAL_STATUS,
|
||||||
|
"read_only": True,
|
||||||
|
"fundamental_score_formula": (
|
||||||
|
"Equal-weighted mean of 2+ available sub-scores: P/E = "
|
||||||
|
"clamp(100-(pe-15)*(100/30)); revenue growth = "
|
||||||
|
"clamp(50+growth*2.5); earnings surprise = "
|
||||||
|
"clamp(50+surprise*5)."
|
||||||
|
),
|
||||||
|
"source_runs": source_runs,
|
||||||
|
"definition_notes": DEFINITION_NOTES,
|
||||||
|
"materiality_notes": {
|
||||||
|
"fields": FIELD_TOLERANCES,
|
||||||
|
"fundamental_score_absolute": 5.0,
|
||||||
|
"automatic_cutover": False,
|
||||||
|
},
|
||||||
|
"summary": _summary(rows),
|
||||||
|
"rows": rows,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def store_report(report: dict[str, Any], report_dir: str | Path) -> dict[str, str]:
|
||||||
|
"""Atomically archive JSON/CSV artifacts and update the latest manifest."""
|
||||||
|
directory = Path(report_dir).expanduser().resolve()
|
||||||
|
directory.mkdir(parents=True, exist_ok=True)
|
||||||
|
stamp = _artifact_stamp(report["generated_at"])
|
||||||
|
json_name = f"fundamentals-parity-{stamp}.json"
|
||||||
|
csv_name = f"fundamentals-parity-{stamp}.csv"
|
||||||
|
json_path = directory / json_name
|
||||||
|
csv_path = directory / csv_name
|
||||||
|
|
||||||
|
_atomic_write(json_path, json.dumps(report, indent=2, sort_keys=True) + "\n")
|
||||||
|
_atomic_write(csv_path, report_csv(report))
|
||||||
|
manifest = {
|
||||||
|
"generated_at": report["generated_at"],
|
||||||
|
"json_file": json_name,
|
||||||
|
"csv_file": csv_name,
|
||||||
|
}
|
||||||
|
_atomic_write(
|
||||||
|
directory / "latest.json",
|
||||||
|
json.dumps(manifest, indent=2, sort_keys=True) + "\n",
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"json": str(json_path),
|
||||||
|
"csv": str(csv_path),
|
||||||
|
"manifest": str(directory / "latest.json"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def generate_and_store(
|
||||||
|
db: AsyncSession,
|
||||||
|
report_dir: str | Path,
|
||||||
|
*,
|
||||||
|
generated_at: datetime | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
) -> tuple[dict[str, Any], dict[str, str]]:
|
||||||
|
report = await build_report(db, generated_at=generated_at, today=today)
|
||||||
|
return report, store_report(report, report_dir)
|
||||||
|
|
||||||
|
|
||||||
|
def load_latest(report_dir: str | Path) -> dict[str, Any] | None:
|
||||||
|
manifest = _load_manifest(report_dir)
|
||||||
|
if manifest is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
path = _manifest_artifact(report_dir, manifest, "json_file")
|
||||||
|
loaded = json.loads(path.read_text(encoding="utf-8"))
|
||||||
|
except (OSError, json.JSONDecodeError, TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
return loaded if isinstance(loaded, dict) else None
|
||||||
|
|
||||||
|
|
||||||
|
def load_latest_csv(report_dir: str | Path) -> tuple[str, str] | None:
|
||||||
|
return _load_latest_text_artifact(report_dir, "csv_file")
|
||||||
|
|
||||||
|
|
||||||
|
def load_latest_json(report_dir: str | Path) -> tuple[str, str] | None:
|
||||||
|
return _load_latest_text_artifact(report_dir, "json_file")
|
||||||
|
|
||||||
|
|
||||||
|
def _load_latest_text_artifact(
|
||||||
|
report_dir: str | Path, manifest_key: str
|
||||||
|
) -> tuple[str, str] | None:
|
||||||
|
manifest = _load_manifest(report_dir)
|
||||||
|
if manifest is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
path = _manifest_artifact(report_dir, manifest, manifest_key)
|
||||||
|
return path.name, path.read_text(encoding="utf-8")
|
||||||
|
except (OSError, TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def report_csv(report: dict[str, Any]) -> str:
|
||||||
|
output = io.StringIO(newline="")
|
||||||
|
columns = [
|
||||||
|
"symbol",
|
||||||
|
"cik",
|
||||||
|
"legacy_fetched_at",
|
||||||
|
"price_date",
|
||||||
|
*(
|
||||||
|
f"{field}_{suffix}"
|
||||||
|
for field in FIELD_KEYS
|
||||||
|
for suffix in ("legacy", "candidate", "absolute_delta", "relative_delta_pct", "material")
|
||||||
|
),
|
||||||
|
"legacy_fundamental",
|
||||||
|
"candidate_fundamental",
|
||||||
|
"fundamental_delta",
|
||||||
|
"legacy_fundamental_rank",
|
||||||
|
"candidate_fundamental_rank",
|
||||||
|
"fundamental_rank_change",
|
||||||
|
]
|
||||||
|
writer = csv.DictWriter(output, fieldnames=columns)
|
||||||
|
writer.writeheader()
|
||||||
|
for row in report.get("rows", []):
|
||||||
|
flat = {
|
||||||
|
"symbol": row["symbol"],
|
||||||
|
"cik": row.get("cik"),
|
||||||
|
"legacy_fetched_at": row.get("legacy_fetched_at"),
|
||||||
|
"price_date": row.get("price_date"),
|
||||||
|
**row["scores"],
|
||||||
|
}
|
||||||
|
for field in FIELD_KEYS:
|
||||||
|
comparison = row["fields"][field]
|
||||||
|
for suffix in (
|
||||||
|
"legacy",
|
||||||
|
"candidate",
|
||||||
|
"absolute_delta",
|
||||||
|
"relative_delta_pct",
|
||||||
|
"material",
|
||||||
|
):
|
||||||
|
flat[f"{field}_{suffix}"] = comparison.get(suffix)
|
||||||
|
writer.writerow(flat)
|
||||||
|
return output.getvalue()
|
||||||
|
|
||||||
|
|
||||||
|
async def _legacy_values(
|
||||||
|
db: AsyncSession, ticker_ids: list[int]
|
||||||
|
) -> dict[int, FundamentalData]:
|
||||||
|
if not ticker_ids:
|
||||||
|
return {}
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalData).where(FundamentalData.ticker_id.in_(ticker_ids))
|
||||||
|
)
|
||||||
|
).scalars()
|
||||||
|
return {row.ticker_id: row for row in rows}
|
||||||
|
|
||||||
|
|
||||||
|
async def _source_runs(db: AsyncSession) -> dict[str, dict[str, Any] | None]:
|
||||||
|
sources = ("sec_facts", "dolt_earnings")
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(DataImportRun)
|
||||||
|
.where(
|
||||||
|
DataImportRun.source.in_(sources),
|
||||||
|
DataImportRun.status.in_(("promoted", "no_op")),
|
||||||
|
)
|
||||||
|
.order_by(DataImportRun.id.desc())
|
||||||
|
)
|
||||||
|
).scalars()
|
||||||
|
latest: dict[str, dict[str, Any] | None] = {source: None for source in sources}
|
||||||
|
for row in rows:
|
||||||
|
if latest[row.source] is None:
|
||||||
|
latest[row.source] = {
|
||||||
|
"run_id": row.id,
|
||||||
|
"status": row.status,
|
||||||
|
"revision": row.revision,
|
||||||
|
"source_max_date": _iso(row.source_max_date),
|
||||||
|
"completed_at": _iso(row.completed_at),
|
||||||
|
}
|
||||||
|
return latest
|
||||||
|
|
||||||
|
|
||||||
|
def _field_comparison(
|
||||||
|
key: str, legacy: float | None, candidate: float | None
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
legacy = float(legacy) if _finite(legacy) else None
|
||||||
|
candidate = float(candidate) if _finite(candidate) else None
|
||||||
|
absolute = _delta(legacy, candidate)
|
||||||
|
relative = (
|
||||||
|
None
|
||||||
|
if absolute is None or legacy in (None, 0)
|
||||||
|
else round(absolute / abs(legacy) * 100.0, 4)
|
||||||
|
)
|
||||||
|
tolerance = FIELD_TOLERANCES[key]
|
||||||
|
material = False
|
||||||
|
if absolute is not None:
|
||||||
|
material = abs(absolute) > tolerance["absolute"]
|
||||||
|
relative_limit = tolerance["relative_pct"]
|
||||||
|
if relative_limit is not None:
|
||||||
|
material = material and relative is not None and abs(relative) > relative_limit
|
||||||
|
return {
|
||||||
|
"legacy": _round(legacy),
|
||||||
|
"candidate": _round(candidate),
|
||||||
|
"absolute_delta": absolute,
|
||||||
|
"relative_delta_pct": relative,
|
||||||
|
"material": material,
|
||||||
|
"definition_changed": True,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _attach_ranks(rows: list[dict[str, Any]], value_key: str, rank_key: str) -> None:
|
||||||
|
values = [
|
||||||
|
row["scores"][value_key]
|
||||||
|
for row in rows
|
||||||
|
if _finite(row["scores"][value_key])
|
||||||
|
]
|
||||||
|
for row in rows:
|
||||||
|
value = row["scores"][value_key]
|
||||||
|
row["scores"][rank_key] = (
|
||||||
|
1 + sum(other > value for other in values) if _finite(value) else None
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _summary(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||||||
|
field_stats = {}
|
||||||
|
for key in FIELD_KEYS:
|
||||||
|
comparisons = [row["fields"][key] for row in rows]
|
||||||
|
deltas = [
|
||||||
|
abs(item["absolute_delta"])
|
||||||
|
for item in comparisons
|
||||||
|
if item["absolute_delta"] is not None
|
||||||
|
]
|
||||||
|
field_stats[key] = {
|
||||||
|
"legacy_available": sum(item["legacy"] is not None for item in comparisons),
|
||||||
|
"candidate_available": sum(
|
||||||
|
item["candidate"] is not None for item in comparisons
|
||||||
|
),
|
||||||
|
"both_available": len(deltas),
|
||||||
|
"material_differences": sum(item["material"] for item in comparisons),
|
||||||
|
"median_absolute_delta": _round(statistics.median(deltas) if deltas else None),
|
||||||
|
"p95_absolute_delta": _round(_percentile(deltas, 0.95)),
|
||||||
|
"max_absolute_delta": _round(max(deltas) if deltas else None),
|
||||||
|
}
|
||||||
|
|
||||||
|
fundamental_deltas = _score_deltas(rows, "fundamental_delta")
|
||||||
|
changed_rows = sorted(
|
||||||
|
(
|
||||||
|
{
|
||||||
|
"symbol": row["symbol"],
|
||||||
|
"fundamental_delta": row["scores"]["fundamental_delta"],
|
||||||
|
"fundamental_rank_change": row["scores"]["fundamental_rank_change"],
|
||||||
|
}
|
||||||
|
for row in rows
|
||||||
|
if row["scores"]["fundamental_delta"] is not None
|
||||||
|
),
|
||||||
|
key=lambda item: (
|
||||||
|
abs(item["fundamental_delta"] or 0),
|
||||||
|
),
|
||||||
|
reverse=True,
|
||||||
|
)[:20]
|
||||||
|
return {
|
||||||
|
"universe_count": len(rows),
|
||||||
|
"legacy_fundamental_score_available": _count_score(
|
||||||
|
rows, "legacy_fundamental"
|
||||||
|
),
|
||||||
|
"candidate_fundamental_score_available": _count_score(
|
||||||
|
rows, "candidate_fundamental"
|
||||||
|
),
|
||||||
|
"fundamental_scores_compared": len(fundamental_deltas),
|
||||||
|
"fundamental_score_material_changes": sum(
|
||||||
|
abs(delta) > 5.0 for delta in fundamental_deltas
|
||||||
|
),
|
||||||
|
"fundamental_rank_changes": _rank_change_count(
|
||||||
|
rows, "fundamental_rank_change"
|
||||||
|
),
|
||||||
|
"field_stats": field_stats,
|
||||||
|
"largest_changes": changed_rows,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _score_deltas(rows: Iterable[dict[str, Any]], key: str) -> list[float]:
|
||||||
|
return [
|
||||||
|
row["scores"][key]
|
||||||
|
for row in rows
|
||||||
|
if row["scores"][key] is not None
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _count_score(rows: Iterable[dict[str, Any]], key: str) -> int:
|
||||||
|
return sum(row["scores"][key] is not None for row in rows)
|
||||||
|
|
||||||
|
|
||||||
|
def _rank_change_count(rows: Iterable[dict[str, Any]], key: str) -> int:
|
||||||
|
return sum(
|
||||||
|
row["scores"][key] not in (None, 0)
|
||||||
|
for row in rows
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _rank_change(legacy: int | None, candidate: int | None) -> int | None:
|
||||||
|
# Positive means the candidate improved its rank.
|
||||||
|
return legacy - candidate if legacy is not None and candidate is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _delta(legacy: float | None, candidate: float | None) -> float | None:
|
||||||
|
if not _finite(legacy) or not _finite(candidate):
|
||||||
|
return None
|
||||||
|
return round(candidate - legacy, 4)
|
||||||
|
|
||||||
|
|
||||||
|
def _round(value: float | None, digits: int = 4) -> float | None:
|
||||||
|
return round(float(value), digits) if _finite(value) else None
|
||||||
|
|
||||||
|
|
||||||
|
def _percentile(values: list[float], quantile: float) -> float | None:
|
||||||
|
if not values:
|
||||||
|
return None
|
||||||
|
ordered = sorted(values)
|
||||||
|
index = max(0, math.ceil(quantile * len(ordered)) - 1)
|
||||||
|
return ordered[index]
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(value: Any) -> bool:
|
||||||
|
return (
|
||||||
|
isinstance(value, (int, float))
|
||||||
|
and not isinstance(value, bool)
|
||||||
|
and math.isfinite(value)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _iso(value: Any) -> str | None:
|
||||||
|
return value.isoformat() if value is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _artifact_stamp(raw: str) -> str:
|
||||||
|
parsed = datetime.fromisoformat(raw.replace("Z", "+00:00"))
|
||||||
|
return parsed.astimezone(timezone.utc).strftime("%Y%m%dT%H%M%S%fZ")
|
||||||
|
|
||||||
|
|
||||||
|
def _atomic_write(path: Path, content: str) -> None:
|
||||||
|
temp = path.with_name(f".{path.name}.{os.getpid()}.tmp")
|
||||||
|
temp.write_text(content, encoding="utf-8", newline="")
|
||||||
|
os.replace(temp, path)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_manifest(report_dir: str | Path) -> dict[str, Any] | None:
|
||||||
|
path = Path(report_dir).expanduser().resolve() / "latest.json"
|
||||||
|
try:
|
||||||
|
loaded = json.loads(path.read_text(encoding="utf-8"))
|
||||||
|
except (OSError, json.JSONDecodeError, TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
return loaded if isinstance(loaded, dict) else None
|
||||||
|
|
||||||
|
|
||||||
|
def _manifest_artifact(
|
||||||
|
report_dir: str | Path, manifest: dict[str, Any], key: str
|
||||||
|
) -> Path:
|
||||||
|
directory = Path(report_dir).expanduser().resolve()
|
||||||
|
name = Path(str(manifest.get(key, ""))).name
|
||||||
|
if not name:
|
||||||
|
raise ValueError(f"Latest parity manifest has no {key}")
|
||||||
|
return directory / name
|
||||||
@@ -0,0 +1,107 @@
|
|||||||
|
"""Pure peer comparison for fundamentals (read-time).
|
||||||
|
|
||||||
|
Peers are tracked-universe issuers sharing the **first two SIC digits**,
|
||||||
|
deduplicated by CIK (GOOG/GOOGL are one issuer, one observation). This module is
|
||||||
|
the pure statistics core: given a subject value and the peer group's values for a
|
||||||
|
metric, it returns median + polarity-aware favorable percentile + peer_count, or
|
||||||
|
None when there are fewer than the minimum valid peers (the caller then omits the
|
||||||
|
industry object entirely rather than show a misleading comparison).
|
||||||
|
|
||||||
|
Grouping (which issuers share a 2-digit SIC, CIK-dedup) is the API's job; this
|
||||||
|
module only does the math. **Absolute net_debt is size-dependent and must not get
|
||||||
|
a peer percentile** — leverage is compared via net_debt_to_ebitda.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
import statistics
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
MIN_PEERS = 5
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(v: Any) -> bool:
|
||||||
|
"""True for a finite number — excludes None, bool, NaN, ±inf (plan: null/invalid)."""
|
||||||
|
return isinstance(v, (int, float)) and not isinstance(v, bool) and math.isfinite(v)
|
||||||
|
|
||||||
|
# Metric -> is a higher value more favorable? (Peer-eligible metrics only;
|
||||||
|
# absolute net_debt is intentionally absent — size-dependent.)
|
||||||
|
HIGHER_IS_BETTER: dict[str, bool] = {
|
||||||
|
"revenue_growth_yoy": True,
|
||||||
|
"eps_growth_yoy": True,
|
||||||
|
"operating_margin": True,
|
||||||
|
"fcf_margin": True,
|
||||||
|
"fcf_yield": True,
|
||||||
|
"net_debt_to_ebitda": False, # lower leverage is better
|
||||||
|
"pe": False, # cheaper is better
|
||||||
|
"share_count_change_yoy": False, # dilution is bad
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class PeerStat:
|
||||||
|
median: float
|
||||||
|
favorable_percentile: int # 0-100, polarity-aware (higher = more favorable)
|
||||||
|
peer_count: int # valid issuers in the group
|
||||||
|
|
||||||
|
|
||||||
|
def peer_stat(
|
||||||
|
subject: float | None,
|
||||||
|
group_values: list[float | None],
|
||||||
|
*,
|
||||||
|
higher_is_better: bool,
|
||||||
|
min_peers: int = MIN_PEERS,
|
||||||
|
) -> PeerStat | None:
|
||||||
|
"""Median + favorable percentile for ``subject`` within its group.
|
||||||
|
|
||||||
|
``group_values`` is every issuer's value for the metric (including the
|
||||||
|
subject), CIK-deduplicated by the caller. Null/invalid (non-finite) values are
|
||||||
|
excluded. Returns None when fewer than ``min_peers`` valid values exist, or
|
||||||
|
the subject is null/invalid.
|
||||||
|
|
||||||
|
The percentile is a **tie-aware rank against the other issuers** —
|
||||||
|
``(worse + 0.5·tied) / (peers − 1)`` — so a whole group of equal values maps
|
||||||
|
to 50, not 100, and the median maps to 50.
|
||||||
|
"""
|
||||||
|
valid = [v for v in group_values if _finite(v)]
|
||||||
|
if not _finite(subject) or len(valid) < min_peers:
|
||||||
|
return None
|
||||||
|
median = statistics.median(valid)
|
||||||
|
|
||||||
|
others = valid.copy()
|
||||||
|
try:
|
||||||
|
others.remove(subject) # rank the subject against the OTHER issuers
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
denom = len(others)
|
||||||
|
if denom == 0:
|
||||||
|
return None
|
||||||
|
if higher_is_better:
|
||||||
|
worse = sum(1 for v in others if v < subject)
|
||||||
|
else:
|
||||||
|
worse = sum(1 for v in others if v > subject)
|
||||||
|
tied = sum(1 for v in others if v == subject)
|
||||||
|
percentile = round((worse + 0.5 * tied) / denom * 100)
|
||||||
|
return PeerStat(median=median, favorable_percentile=percentile, peer_count=len(valid))
|
||||||
|
|
||||||
|
|
||||||
|
def peer_stat_for(
|
||||||
|
metric_key: str, subject: float | None, group_values: list[float | None], **kwargs
|
||||||
|
) -> PeerStat | None:
|
||||||
|
"""Convenience wrapper that looks up polarity by metric key. Returns None for
|
||||||
|
metrics not eligible for peer comparison (e.g. absolute net_debt)."""
|
||||||
|
if metric_key not in HIGHER_IS_BETTER:
|
||||||
|
return None
|
||||||
|
return peer_stat(
|
||||||
|
subject, group_values, higher_is_better=HIGHER_IS_BETTER[metric_key], **kwargs
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def two_digit_sic(sic: str | None) -> str | None:
|
||||||
|
"""The 2-digit SIC prefix used for grouping, or None if unusable."""
|
||||||
|
if not sic:
|
||||||
|
return None
|
||||||
|
digits = str(sic).strip()
|
||||||
|
return digits[:2] if len(digits) >= 2 and digits[:2].isdigit() else None
|
||||||
@@ -0,0 +1,164 @@
|
|||||||
|
"""Actionability gate for incomplete SEC fundamentals."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from dataclasses import dataclass
|
||||||
|
|
||||||
|
from sqlalchemy import exists, func, select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.sec_filing_gap import SecFilingGap
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import fundamental_data_refresh_service
|
||||||
|
|
||||||
|
_SEC_FORMS = ("10-K", "10-Q", "10-K/A", "10-Q/A")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class SetupQuality:
|
||||||
|
eligible: bool
|
||||||
|
code: str | None = None
|
||||||
|
message: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
async def active_gaps(
|
||||||
|
db: AsyncSession,
|
||||||
|
ciks: set[str] | None = None,
|
||||||
|
) -> list[SecFilingGap]:
|
||||||
|
"""Unresolved gaps that have not been superseded by a later filing."""
|
||||||
|
matching_snapshot = exists().where(
|
||||||
|
FundamentalSnapshot.accession == SecFilingGap.accession
|
||||||
|
)
|
||||||
|
gap_date = func.coalesce(
|
||||||
|
SecFilingGap.index_date,
|
||||||
|
func.date(SecFilingGap.first_seen_at),
|
||||||
|
)
|
||||||
|
later_snapshot = exists().where(
|
||||||
|
FundamentalSnapshot.cik == SecFilingGap.cik,
|
||||||
|
FundamentalSnapshot.form.in_(_SEC_FORMS),
|
||||||
|
FundamentalSnapshot.filed_date > gap_date,
|
||||||
|
)
|
||||||
|
stmt = select(SecFilingGap).where(
|
||||||
|
~matching_snapshot,
|
||||||
|
~later_snapshot,
|
||||||
|
)
|
||||||
|
if ciks is not None:
|
||||||
|
if not ciks:
|
||||||
|
return []
|
||||||
|
stmt = stmt.where(SecFilingGap.cik.in_(ciks))
|
||||||
|
return list((await db.execute(stmt)).scalars().all())
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_validation(db: AsyncSession) -> dict:
|
||||||
|
payload = (
|
||||||
|
await db.execute(
|
||||||
|
select(DataImportRun.validation_json)
|
||||||
|
.where(
|
||||||
|
DataImportRun.source == "sec_facts",
|
||||||
|
DataImportRun.validation_json.is_not(None),
|
||||||
|
)
|
||||||
|
.order_by(DataImportRun.id.desc())
|
||||||
|
.limit(1)
|
||||||
|
)
|
||||||
|
).scalar_one_or_none()
|
||||||
|
if not payload:
|
||||||
|
return {}
|
||||||
|
try:
|
||||||
|
summary = json.loads(payload)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return {}
|
||||||
|
return summary if isinstance(summary, dict) else {}
|
||||||
|
|
||||||
|
|
||||||
|
async def blocked_reasons_by_cik(
|
||||||
|
db: AsyncSession,
|
||||||
|
ciks: set[str] | None = None,
|
||||||
|
) -> dict[str, str]:
|
||||||
|
"""Current SEC blocker code by CIK; no historical audit scan."""
|
||||||
|
if not await fundamental_data_refresh_service.is_enabled(db):
|
||||||
|
return {}
|
||||||
|
if ciks is not None and not ciks:
|
||||||
|
return {}
|
||||||
|
|
||||||
|
reasons = {
|
||||||
|
gap.cik: "sec_filing_gap" for gap in await active_gaps(db, ciks)
|
||||||
|
}
|
||||||
|
summary = await _latest_validation(db)
|
||||||
|
|
||||||
|
def wanted(cik: str) -> bool:
|
||||||
|
return ciks is None or cik in ciks
|
||||||
|
|
||||||
|
# New summaries carry the complete compact CIK set while the detailed lists
|
||||||
|
# stay capped for audit readability. Detailed entries supply the reason.
|
||||||
|
for cik in summary.get("setup_blocked_ciks") or []:
|
||||||
|
normalized = str(cik) if cik else ""
|
||||||
|
if normalized and wanted(normalized):
|
||||||
|
reasons.setdefault(normalized, "sec_filing_gap")
|
||||||
|
for item in summary.get("missing_xbrl") or []:
|
||||||
|
normalized = str(item.get("cik") or "")
|
||||||
|
if normalized and wanted(normalized):
|
||||||
|
reasons.setdefault(normalized, "sec_filing_gap")
|
||||||
|
for cik in summary.get("no_xbrl_ciks") or []:
|
||||||
|
normalized = str(cik) if cik else ""
|
||||||
|
if normalized and wanted(normalized):
|
||||||
|
reasons[normalized] = "no_xbrl_filings"
|
||||||
|
for item in summary.get("no_xbrl_filings") or []:
|
||||||
|
normalized = str(item.get("cik") or "")
|
||||||
|
if normalized and wanted(normalized):
|
||||||
|
reasons[normalized] = "no_xbrl_filings"
|
||||||
|
return reasons
|
||||||
|
|
||||||
|
|
||||||
|
async def blocked_ciks(db: AsyncSession) -> set[str]:
|
||||||
|
return set(await blocked_reasons_by_cik(db))
|
||||||
|
|
||||||
|
|
||||||
|
async def blocked_ticker_ids(db: AsyncSession) -> set[int]:
|
||||||
|
ciks = await blocked_ciks(db)
|
||||||
|
if not ciks:
|
||||||
|
return set()
|
||||||
|
rows = await db.execute(select(Ticker.id).where(Ticker.cik.in_(ciks)))
|
||||||
|
return {int(ticker_id) for ticker_id in rows.scalars()}
|
||||||
|
|
||||||
|
|
||||||
|
async def ticker_quality(db: AsyncSession, symbol: str) -> SetupQuality:
|
||||||
|
ticker = (
|
||||||
|
await db.execute(
|
||||||
|
select(Ticker).where(Ticker.symbol == symbol.strip().upper())
|
||||||
|
)
|
||||||
|
).scalar_one_or_none()
|
||||||
|
if ticker is None or not ticker.cik:
|
||||||
|
return SetupQuality(eligible=True)
|
||||||
|
reason = (await blocked_reasons_by_cik(db, {ticker.cik})).get(ticker.cik)
|
||||||
|
if reason == "no_xbrl_filings":
|
||||||
|
return SetupQuality(
|
||||||
|
eligible=False,
|
||||||
|
code=reason,
|
||||||
|
message=(
|
||||||
|
"No SEC 10-K/10-Q is available for this registrant, so new setups "
|
||||||
|
"are paused. New registrants clear automatically after their first "
|
||||||
|
"filing; a successor shell needs an SEC CIK override."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
if reason:
|
||||||
|
return SetupQuality(
|
||||||
|
eligible=False,
|
||||||
|
code=reason,
|
||||||
|
message=(
|
||||||
|
"A recent SEC filing is still being reconciled, so new setups are "
|
||||||
|
"paused. The scheduled fundamentals import retries it automatically."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
return SetupQuality(eligible=True)
|
||||||
|
|
||||||
|
|
||||||
|
async def ticker_is_eligible(db: AsyncSession, ticker_id: int) -> bool:
|
||||||
|
cik = (
|
||||||
|
await db.execute(select(Ticker.cik).where(Ticker.id == ticker_id))
|
||||||
|
).scalar_one_or_none()
|
||||||
|
if not cik:
|
||||||
|
return True
|
||||||
|
return cik not in await blocked_reasons_by_cik(db, {cik})
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
"""Deterministic text 'reads' for the fundamentals panel (pure, one rule set).
|
||||||
|
|
||||||
|
The tape reads and the header sentence use identical outputs — no LLM, no new
|
||||||
|
composite score. Thresholds are tunable named constants, not scattered literals
|
||||||
|
(plan: ±2pp growth, ±1pp margins, ±1% dilution, 60/40 peer bands, ≥3 periods).
|
||||||
|
|
||||||
|
Consumers pass metric series (value + dated history, from
|
||||||
|
``fundamentals_derivation``) and peer percentiles; these functions return short
|
||||||
|
strings or None (render "—", no read).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from statistics import mean
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
MIN_PERIODS = 3
|
||||||
|
GROWTH_ACCEL_PP = 2.0
|
||||||
|
MARGIN_MOVE_PP = 1.0
|
||||||
|
SHARE_DILUTION_PCT = 1.0
|
||||||
|
PEER_FAVORABLE = 60
|
||||||
|
PEER_ADVERSE = 40
|
||||||
|
|
||||||
|
|
||||||
|
def _latest_run(history: list[Any]) -> list[float]:
|
||||||
|
"""The consecutive non-null values ending at the latest point (oldest->newest).
|
||||||
|
A null latest, or an internal gap, truncates the run — so a read never reflects
|
||||||
|
a period whose displayed value is n/a."""
|
||||||
|
run: list[float] = []
|
||||||
|
for p in reversed(history):
|
||||||
|
if p.value is None:
|
||||||
|
break
|
||||||
|
run.append(p.value)
|
||||||
|
run.reverse()
|
||||||
|
return run
|
||||||
|
|
||||||
|
|
||||||
|
def growth_read(history: list[Any]) -> str | None:
|
||||||
|
"""Change in a YoY-growth series: latest − prior. Needs >= 3 consecutive
|
||||||
|
non-null values ending at the latest point."""
|
||||||
|
vals = _latest_run(history)
|
||||||
|
if len(vals) < MIN_PERIODS:
|
||||||
|
return None
|
||||||
|
delta = vals[-1] - vals[-2]
|
||||||
|
if delta >= GROWTH_ACCEL_PP:
|
||||||
|
return "accelerating"
|
||||||
|
if delta <= -GROWTH_ACCEL_PP:
|
||||||
|
return "decelerating"
|
||||||
|
return "steady"
|
||||||
|
|
||||||
|
|
||||||
|
def margin_read(history: list[Any]) -> str | None:
|
||||||
|
"""Latest margin vs the mean of prior periods (pp). Needs >= 3 consecutive
|
||||||
|
non-null values ending at the latest point."""
|
||||||
|
vals = _latest_run(history)
|
||||||
|
if len(vals) < MIN_PERIODS:
|
||||||
|
return None
|
||||||
|
delta = vals[-1] - mean(vals[:-1])
|
||||||
|
if delta >= MARGIN_MOVE_PP:
|
||||||
|
return "improving"
|
||||||
|
if delta <= -MARGIN_MOVE_PP:
|
||||||
|
return "deteriorating"
|
||||||
|
return "stable"
|
||||||
|
|
||||||
|
|
||||||
|
def share_count_read(value: float | None) -> str | None:
|
||||||
|
"""Share-count YoY %: >+1% dilution, <-1% buying back, else flat."""
|
||||||
|
if value is None:
|
||||||
|
return None
|
||||||
|
if value > SHARE_DILUTION_PCT:
|
||||||
|
return f"{value:.1f}% dilution"
|
||||||
|
if value < -SHARE_DILUTION_PCT:
|
||||||
|
return "buying back"
|
||||||
|
return "flat"
|
||||||
|
|
||||||
|
|
||||||
|
def peer_read(metric_key: str, favorable_percentile: int | None) -> str | None:
|
||||||
|
"""Peer-relative read for a metric, polarity already baked into the
|
||||||
|
percentile (higher = more favorable)."""
|
||||||
|
if favorable_percentile is None:
|
||||||
|
return None
|
||||||
|
if favorable_percentile >= PEER_FAVORABLE:
|
||||||
|
return _FAVORABLE.get(metric_key, "above peers")
|
||||||
|
if favorable_percentile <= PEER_ADVERSE:
|
||||||
|
return _ADVERSE.get(metric_key, "below peers")
|
||||||
|
return "in line"
|
||||||
|
|
||||||
|
|
||||||
|
_FAVORABLE = {
|
||||||
|
"pe": "attractively valued",
|
||||||
|
"fcf_yield": "above peers",
|
||||||
|
"net_debt_to_ebitda": "conservative leverage",
|
||||||
|
}
|
||||||
|
_ADVERSE = {
|
||||||
|
"pe": "priced above peers",
|
||||||
|
"fcf_yield": "below peers",
|
||||||
|
"net_debt_to_ebitda": "elevated leverage",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def header_sentence(
|
||||||
|
growth: str | None, margin: str | None, valuation: str | None
|
||||||
|
) -> str:
|
||||||
|
"""Join the growth / margin / peer-valuation reads with ' · ', omitting
|
||||||
|
segments with no read. Segment sources are fixed by the caller (growth =
|
||||||
|
revenue-growth read, margin = operating-margin read, valuation = P/E peer
|
||||||
|
read falling back to FCF yield)."""
|
||||||
|
parts = []
|
||||||
|
if growth:
|
||||||
|
parts.append(f"growth {growth}")
|
||||||
|
if margin:
|
||||||
|
parts.append(f"margins {margin}")
|
||||||
|
if valuation:
|
||||||
|
parts.append(f"valuation {valuation}")
|
||||||
|
return " · ".join(parts)
|
||||||
@@ -28,6 +28,7 @@ MIN_BARS: dict[str, int] = {
|
|||||||
"atr": 15,
|
"atr": 15,
|
||||||
"volume_profile": 20,
|
"volume_profile": 20,
|
||||||
"pivot_points": 5,
|
"pivot_points": 5,
|
||||||
|
"fip_id": 253, # 12-1 formation: need index i-252
|
||||||
}
|
}
|
||||||
|
|
||||||
DEFAULT_PERIODS: dict[str, int] = {
|
DEFAULT_PERIODS: dict[str, int] = {
|
||||||
@@ -256,6 +257,12 @@ def compute_volume_profile(
|
|||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Compute Volume Profile: POC, Value Area, HVN, LVN.
|
"""Compute Volume Profile: POC, Value Area, HVN, LVN.
|
||||||
|
|
||||||
|
Volume is assigned to the bin containing each bar's **close** (no
|
||||||
|
double-counting across the high–low span).
|
||||||
|
|
||||||
|
HVN = local peaks in the volume histogram (not every bin above mean).
|
||||||
|
LVN = local valleys in the histogram.
|
||||||
|
|
||||||
Score: proximity of latest close to POC (closer = higher).
|
Score: proximity of latest close to POC (closer = higher).
|
||||||
"""
|
"""
|
||||||
n = len(closes)
|
n = len(closes)
|
||||||
@@ -275,14 +282,18 @@ def compute_volume_profile(
|
|||||||
price_min + (i + 0.5) * bin_width for i in range(num_bins)
|
price_min + (i + 0.5) * bin_width for i in range(num_bins)
|
||||||
]
|
]
|
||||||
|
|
||||||
|
# Assign each bar's full volume to the close's bin only.
|
||||||
for i in range(n):
|
for i in range(n):
|
||||||
# Distribute volume across bins the bar spans
|
c = closes[i]
|
||||||
bar_low, bar_high = lows[i], highs[i]
|
if c <= price_min:
|
||||||
for b in range(num_bins):
|
b = 0
|
||||||
bl = price_min + b * bin_width
|
elif c >= price_max:
|
||||||
bh = bl + bin_width
|
b = num_bins - 1
|
||||||
if bar_high >= bl and bar_low <= bh:
|
else:
|
||||||
bins[b] += volumes[i]
|
b = int((c - price_min) / bin_width)
|
||||||
|
if b >= num_bins:
|
||||||
|
b = num_bins - 1
|
||||||
|
bins[b] += volumes[i]
|
||||||
|
|
||||||
total_vol = sum(bins)
|
total_vol = sum(bins)
|
||||||
if total_vol == 0:
|
if total_vol == 0:
|
||||||
@@ -304,10 +315,17 @@ def compute_volume_profile(
|
|||||||
va_low = round(price_min + min(va_indices) * bin_width, 4)
|
va_low = round(price_min + min(va_indices) * bin_width, 4)
|
||||||
va_high = round(price_min + (max(va_indices) + 1) * bin_width, 4)
|
va_high = round(price_min + (max(va_indices) + 1) * bin_width, 4)
|
||||||
|
|
||||||
# HVN / LVN: bins above/below average volume
|
# HVN / LVN: local peaks / valleys (require above/below mean to skip noise)
|
||||||
avg_vol = total_vol / num_bins
|
avg_vol = total_vol / num_bins
|
||||||
hvn = [round(bin_prices[i], 4) for i in range(num_bins) if bins[i] > avg_vol]
|
hvn: list[float] = []
|
||||||
lvn = [round(bin_prices[i], 4) for i in range(num_bins) if bins[i] < avg_vol]
|
lvn: list[float] = []
|
||||||
|
for i in range(num_bins):
|
||||||
|
left = bins[i - 1] if i > 0 else bins[i]
|
||||||
|
right = bins[i + 1] if i < num_bins - 1 else bins[i]
|
||||||
|
if bins[i] > left and bins[i] > right and bins[i] > avg_vol:
|
||||||
|
hvn.append(round(bin_prices[i], 4))
|
||||||
|
elif bins[i] < left and bins[i] < right and bins[i] < avg_vol:
|
||||||
|
lvn.append(round(bin_prices[i], 4))
|
||||||
|
|
||||||
# Score: proximity of latest close to POC
|
# Score: proximity of latest close to POC
|
||||||
latest = closes[-1]
|
latest = closes[-1]
|
||||||
@@ -333,10 +351,14 @@ def compute_pivot_points(
|
|||||||
lows: list[float],
|
lows: list[float],
|
||||||
closes: list[float],
|
closes: list[float],
|
||||||
window: int = 2,
|
window: int = 2,
|
||||||
|
min_prominence: float | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Detect swing highs/lows as pivot points.
|
"""Detect swing highs/lows as pivot points.
|
||||||
|
|
||||||
A swing high at index *i* means highs[i] >= all highs in [i-window, i+window].
|
A swing high at index *i* means highs[i] >= all highs in [i-window, i+window].
|
||||||
|
When *min_prominence* is set, only keep swings whose window range
|
||||||
|
(max high − min low) is at least that amount — filters tiny noise fractals.
|
||||||
|
|
||||||
Score: based on number of pivots near current price.
|
Score: based on number of pivots near current price.
|
||||||
"""
|
"""
|
||||||
n = len(closes)
|
n = len(closes)
|
||||||
@@ -349,12 +371,24 @@ def compute_pivot_points(
|
|||||||
swing_lows: list[float] = []
|
swing_lows: list[float] = []
|
||||||
|
|
||||||
for i in range(window, n - window):
|
for i in range(window, n - window):
|
||||||
|
lo = i - window
|
||||||
|
hi = i + window + 1
|
||||||
# Swing high
|
# Swing high
|
||||||
if all(highs[i] >= highs[j] for j in range(i - window, i + window + 1)):
|
if all(highs[i] >= highs[j] for j in range(lo, hi)):
|
||||||
swing_highs.append(round(highs[i], 4))
|
if min_prominence is None or min_prominence <= 0:
|
||||||
|
swing_highs.append(round(highs[i], 4))
|
||||||
|
else:
|
||||||
|
depth = highs[i] - min(lows[j] for j in range(lo, hi))
|
||||||
|
if depth >= min_prominence:
|
||||||
|
swing_highs.append(round(highs[i], 4))
|
||||||
# Swing low
|
# Swing low
|
||||||
if all(lows[i] <= lows[j] for j in range(i - window, i + window + 1)):
|
if all(lows[i] <= lows[j] for j in range(lo, hi)):
|
||||||
swing_lows.append(round(lows[i], 4))
|
if min_prominence is None or min_prominence <= 0:
|
||||||
|
swing_lows.append(round(lows[i], 4))
|
||||||
|
else:
|
||||||
|
depth = max(highs[j] for j in range(lo, hi)) - lows[i]
|
||||||
|
if depth >= min_prominence:
|
||||||
|
swing_lows.append(round(lows[i], 4))
|
||||||
|
|
||||||
all_pivots = swing_highs + swing_lows
|
all_pivots = swing_highs + swing_lows
|
||||||
latest = closes[-1]
|
latest = closes[-1]
|
||||||
@@ -374,6 +408,88 @@ def compute_pivot_points(
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# Path labels for display only. Calibrated on the ~505-name prod snapshot
|
||||||
|
# (2026-07, n=502 with full history): empirical p25 ≈ −0.082, p75 ≈ +0.004,
|
||||||
|
# mean ≈ −0.043. Paper-style |ID| ≳ 0.25 almost never appears in live equities
|
||||||
|
# (only ~0.2% of names); real momentum winners cluster around −0.04…−0.12.
|
||||||
|
# Thresholds are therefore ~quartile cutoffs, not ±0.25 textbook extremes.
|
||||||
|
# Distribution is left-skewed (bullish sample → more "continuous" than "discrete"),
|
||||||
|
# so the discrete band is not symmetric.
|
||||||
|
FIP_PATH_CONTINUOUS_MAX = -0.08 # ~p25: smoother quartile
|
||||||
|
FIP_PATH_DISCRETE_MIN = 0.00 # ~p75: less-continuous quartile
|
||||||
|
|
||||||
|
|
||||||
|
def compute_fip_id(closes: list[float], as_of_index: int | None = None) -> dict[str, Any]:
|
||||||
|
"""Da/Gurun/Warachka information discreteness over the 12-1 formation window.
|
||||||
|
|
||||||
|
Display / research context only — **not** used by the activation gate or
|
||||||
|
production rank. Same window as residual 12-1 momentum: cumulative return
|
||||||
|
from close[i-252] to close[i-21] (skip last month).
|
||||||
|
|
||||||
|
ID = sign(PRET) × (%neg − %pos)
|
||||||
|
|
||||||
|
Lower ID ⇒ smoother / more continuous path (for a winner: many small up days).
|
||||||
|
Higher ID ⇒ jumpy / discrete path (few large moves).
|
||||||
|
|
||||||
|
Zero-return days count in neither numerator but remain in the denominator
|
||||||
|
(paper definition). Quirk: a flat series with one big jump can still land
|
||||||
|
near zero ("mixed") because zeros dilute %pos/%neg — faithful to the paper
|
||||||
|
and to real equities (exact zero daily returns are rare). Synthetic jump
|
||||||
|
tests assert ordering vs a steady climber, not the discrete label itself.
|
||||||
|
"""
|
||||||
|
i = len(closes) - 1 if as_of_index is None else as_of_index
|
||||||
|
if i < 252 or closes[i - 252] <= 0 or closes[i - 21] <= 0:
|
||||||
|
raise ValidationError(
|
||||||
|
f"FIP ID requires at least 253 bars with positive formation closes, "
|
||||||
|
f"got {len(closes)}"
|
||||||
|
)
|
||||||
|
pret = closes[i - 21] / closes[i - 252] - 1.0
|
||||||
|
rets: list[float] = []
|
||||||
|
for k in range(i - 251, i - 20):
|
||||||
|
prev = closes[k - 1]
|
||||||
|
if prev <= 0:
|
||||||
|
raise ValidationError("FIP ID requires positive closes in the formation window")
|
||||||
|
rets.append(closes[k] / prev - 1.0)
|
||||||
|
if len(rets) < 200:
|
||||||
|
raise ValidationError(
|
||||||
|
f"FIP ID requires ≥200 daily returns in formation, got {len(rets)}"
|
||||||
|
)
|
||||||
|
n = len(rets)
|
||||||
|
pct_pos = sum(1 for r in rets if r > 0) / n
|
||||||
|
pct_neg = sum(1 for r in rets if r < 0) / n
|
||||||
|
if pret > 0:
|
||||||
|
sign = 1.0
|
||||||
|
elif pret < 0:
|
||||||
|
sign = -1.0
|
||||||
|
else:
|
||||||
|
sign = 0.0
|
||||||
|
fip = sign * (pct_neg - pct_pos)
|
||||||
|
# Map observed ID range (~[-0.3, 0.15]) loosely to 0–100 for the card chrome;
|
||||||
|
# lower ID (smoother) → higher score. Display only.
|
||||||
|
score = max(0.0, min(100.0, 50.0 * (1.0 - fip)))
|
||||||
|
if fip <= FIP_PATH_CONTINUOUS_MAX:
|
||||||
|
path = "continuous"
|
||||||
|
path_label = "smooth grind (continuous information)"
|
||||||
|
elif fip >= FIP_PATH_DISCRETE_MIN:
|
||||||
|
path = "discrete"
|
||||||
|
path_label = "jumpy path (discrete information)"
|
||||||
|
else:
|
||||||
|
path = "mixed"
|
||||||
|
path_label = "mixed path"
|
||||||
|
return {
|
||||||
|
"fip_id": round(fip, 4),
|
||||||
|
"pret_12_1": round(pret, 4),
|
||||||
|
"pct_up_days": round(pct_pos * 100.0, 1),
|
||||||
|
"pct_down_days": round(pct_neg * 100.0, 1),
|
||||||
|
"formation_days": n,
|
||||||
|
"path": path,
|
||||||
|
"path_label": path_label,
|
||||||
|
"display_only": True,
|
||||||
|
"note": "Not used by the production gate or rank — context only.",
|
||||||
|
"score": round(score, 4),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def compute_ema_cross(
|
def compute_ema_cross(
|
||||||
closes: list[float],
|
closes: list[float],
|
||||||
short_period: int = 20,
|
short_period: int = 20,
|
||||||
@@ -418,7 +534,15 @@ def compute_ema_cross(
|
|||||||
# Supported indicator types
|
# Supported indicator types
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
INDICATOR_TYPES = {"adx", "ema", "rsi", "atr", "volume_profile", "pivot_points"}
|
INDICATOR_TYPES = {
|
||||||
|
"adx",
|
||||||
|
"ema",
|
||||||
|
"rsi",
|
||||||
|
"atr",
|
||||||
|
"volume_profile",
|
||||||
|
"pivot_points",
|
||||||
|
"fip_id",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -481,6 +605,8 @@ async def get_indicator(
|
|||||||
result = compute_volume_profile(highs, lows, closes, volumes)
|
result = compute_volume_profile(highs, lows, closes, volumes)
|
||||||
elif indicator_type == "pivot_points":
|
elif indicator_type == "pivot_points":
|
||||||
result = compute_pivot_points(highs, lows, closes)
|
result = compute_pivot_points(highs, lows, closes)
|
||||||
|
elif indicator_type == "fip_id":
|
||||||
|
result = compute_fip_id(closes)
|
||||||
else:
|
else:
|
||||||
raise ValidationError(f"Unknown indicator type: {indicator_type}")
|
raise ValidationError(f"Unknown indicator type: {indicator_type}")
|
||||||
|
|
||||||
|
|||||||
@@ -23,6 +23,15 @@ from app.services import price_service
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
async def _refresh_structural_sr(db: AsyncSession, symbol: str) -> None:
|
||||||
|
"""Rebuild Structural S/R after batch OHLCV writes (best-effort).
|
||||||
|
|
||||||
|
Price bars are already committed; an S/R failure must not discard the
|
||||||
|
ingestion result. Shared with single-bar upsert via price_service.
|
||||||
|
"""
|
||||||
|
await price_service._refresh_structural_sr_best_effort(db, symbol)
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class IngestionResult:
|
class IngestionResult:
|
||||||
"""Result of an ingestion run."""
|
"""Result of an ingestion run."""
|
||||||
@@ -30,7 +39,7 @@ class IngestionResult:
|
|||||||
symbol: str
|
symbol: str
|
||||||
records_ingested: int
|
records_ingested: int
|
||||||
last_date: date | None
|
last_date: date | None
|
||||||
status: str # "complete" | "partial" | "error"
|
status: str # "complete" | "partial" | "error" | "no_data"
|
||||||
message: str | None = None
|
message: str | None = None
|
||||||
|
|
||||||
|
|
||||||
@@ -59,6 +68,19 @@ async def _get_ohlcv_bar_count(db: AsyncSession, ticker_id: int) -> int:
|
|||||||
return int(result.scalar() or 0)
|
return int(result.scalar() or 0)
|
||||||
|
|
||||||
|
|
||||||
|
async def _get_latest_ohlcv_date(db: AsyncSession, ticker_id: int) -> date | None:
|
||||||
|
result = await db.execute(
|
||||||
|
select(func.max(OHLCVRecord.date)).where(OHLCVRecord.ticker_id == ticker_id)
|
||||||
|
)
|
||||||
|
return result.scalar_one_or_none()
|
||||||
|
|
||||||
|
|
||||||
|
# If the provider returns no bars but our last stored session is older than this,
|
||||||
|
# treat the run as stale (not "success / up to date"). Common causes: ticker
|
||||||
|
# rename, delisting, or multi-day halt — SATS→ECHO is the canonical example.
|
||||||
|
_STALE_OHLCV_GAP_DAYS = 5
|
||||||
|
|
||||||
|
|
||||||
async def _update_progress(
|
async def _update_progress(
|
||||||
db: AsyncSession, ticker_id: int, last_date: date
|
db: AsyncSession, ticker_id: int, last_date: date
|
||||||
) -> None:
|
) -> None:
|
||||||
@@ -78,6 +100,8 @@ async def fetch_and_ingest(
|
|||||||
symbol: str,
|
symbol: str,
|
||||||
start_date: date | None = None,
|
start_date: date | None = None,
|
||||||
end_date: date | None = None,
|
end_date: date | None = None,
|
||||||
|
*,
|
||||||
|
refresh_sr: bool = True,
|
||||||
) -> IngestionResult:
|
) -> IngestionResult:
|
||||||
"""Fetch OHLCV data from provider and upsert into Price Store.
|
"""Fetch OHLCV data from provider and upsert into Price Store.
|
||||||
|
|
||||||
@@ -107,7 +131,12 @@ async def fetch_and_ingest(
|
|||||||
if bar_count < minimum_backfill_bars:
|
if bar_count < minimum_backfill_bars:
|
||||||
start_date = backfill_start
|
start_date = backfill_start
|
||||||
elif progress is not None:
|
elif progress is not None:
|
||||||
start_date = progress.last_ingested_date + timedelta(days=1)
|
# Re-fetch the latest stored session so an in-progress daily bar can
|
||||||
|
# be overwritten as the market moves. Starting one day later makes
|
||||||
|
# every subsequent intraday, near-close, and manual refresh skip
|
||||||
|
# today's bar once the first partial snapshot has been stored.
|
||||||
|
# The price-store upsert keeps this one-session overlap idempotent.
|
||||||
|
start_date = progress.last_ingested_date
|
||||||
else:
|
else:
|
||||||
start_date = backfill_start
|
start_date = backfill_start
|
||||||
|
|
||||||
@@ -143,6 +172,46 @@ async def fetch_and_ingest(
|
|||||||
message=str(exc),
|
message=str(exc),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Provider returned nothing. With no history at all this almost always means
|
||||||
|
# the provider doesn't cover this symbol (Alpaca = US listings only) — surface
|
||||||
|
# that instead of a misleading "success". With recent history, an empty window
|
||||||
|
# usually means weekends/holidays. With a multi-day gap, the symbol is likely
|
||||||
|
# halted, delisted, or *renamed* (e.g. SATS → ECHO) and we must not claim success.
|
||||||
|
if not records:
|
||||||
|
existing = await _get_ohlcv_bar_count(db, ticker.id)
|
||||||
|
if existing == 0:
|
||||||
|
return IngestionResult(
|
||||||
|
symbol=ticker.symbol,
|
||||||
|
records_ingested=0,
|
||||||
|
last_date=None,
|
||||||
|
status="no_data",
|
||||||
|
message=(
|
||||||
|
"No data returned by the provider — it may not cover this symbol "
|
||||||
|
"(Alpaca serves US-listed securities only)."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
latest = await _get_latest_ohlcv_date(db, ticker.id)
|
||||||
|
gap_days = (end_date - latest).days if latest is not None else None
|
||||||
|
if gap_days is not None and gap_days > _STALE_OHLCV_GAP_DAYS:
|
||||||
|
return IngestionResult(
|
||||||
|
symbol=ticker.symbol,
|
||||||
|
records_ingested=0,
|
||||||
|
last_date=latest,
|
||||||
|
status="stale",
|
||||||
|
message=(
|
||||||
|
f"No new bars since {latest.isoformat()} ({gap_days}d gap). "
|
||||||
|
"The symbol may be halted, delisted, or renamed under a new ticker — "
|
||||||
|
"check the listing and add/fetch the current symbol if it changed."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
return IngestionResult(
|
||||||
|
symbol=ticker.symbol,
|
||||||
|
records_ingested=0,
|
||||||
|
last_date=latest,
|
||||||
|
status="complete",
|
||||||
|
message="Already up to date — no new bars.",
|
||||||
|
)
|
||||||
|
|
||||||
# Sort records by date to ensure ordered ingestion
|
# Sort records by date to ensure ordered ingestion
|
||||||
records.sort(key=lambda r: r.date)
|
records.sort(key=lambda r: r.date)
|
||||||
|
|
||||||
@@ -160,6 +229,8 @@ async def fetch_and_ingest(
|
|||||||
low=record.low,
|
low=record.low,
|
||||||
close=record.close,
|
close=record.close,
|
||||||
volume=record.volume,
|
volume=record.volume,
|
||||||
|
# One S/R rebuild at the end of the batch, not per bar.
|
||||||
|
refresh_sr=False,
|
||||||
)
|
)
|
||||||
ingested_count += 1
|
ingested_count += 1
|
||||||
last_ingested = record.date
|
last_ingested = record.date
|
||||||
@@ -168,12 +239,15 @@ async def fetch_and_ingest(
|
|||||||
await _update_progress(db, ticker.id, record.date)
|
await _update_progress(db, ticker.id, record.date)
|
||||||
|
|
||||||
except RateLimitError:
|
except RateLimitError:
|
||||||
# Mid-ingestion rate limit — return partial progress
|
# Mid-ingestion rate limit — return partial progress after
|
||||||
|
# refreshing S/R from whatever bars we already wrote.
|
||||||
logger.warning(
|
logger.warning(
|
||||||
"Rate limited during ingestion for %s after %d records",
|
"Rate limited during ingestion for %s after %d records",
|
||||||
ticker.symbol,
|
ticker.symbol,
|
||||||
ingested_count,
|
ingested_count,
|
||||||
)
|
)
|
||||||
|
if ingested_count > 0 and refresh_sr:
|
||||||
|
await _refresh_structural_sr(db, ticker.symbol)
|
||||||
return IngestionResult(
|
return IngestionResult(
|
||||||
symbol=ticker.symbol,
|
symbol=ticker.symbol,
|
||||||
records_ingested=ingested_count,
|
records_ingested=ingested_count,
|
||||||
@@ -182,6 +256,28 @@ async def fetch_and_ingest(
|
|||||||
message=f"Rate limited. Ingested {ingested_count} records. Resume available.",
|
message=f"Rate limited. Ingested {ingested_count} records. Resume available.",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if ingested_count > 0 and refresh_sr:
|
||||||
|
await _refresh_structural_sr(db, ticker.symbol)
|
||||||
|
|
||||||
|
# Incremental fetches deliberately overlap the latest stored session so an
|
||||||
|
# in-progress bar can be updated. A halted/delisted symbol can therefore
|
||||||
|
# return one old bar forever; non-empty no longer means fresh. Judge stale
|
||||||
|
# state from the newest stored session after the upserts instead.
|
||||||
|
latest = await _get_latest_ohlcv_date(db, ticker.id)
|
||||||
|
gap_days = (end_date - latest).days if latest is not None else None
|
||||||
|
if gap_days is not None and gap_days > _STALE_OHLCV_GAP_DAYS:
|
||||||
|
return IngestionResult(
|
||||||
|
symbol=ticker.symbol,
|
||||||
|
records_ingested=ingested_count,
|
||||||
|
last_date=latest,
|
||||||
|
status="stale",
|
||||||
|
message=(
|
||||||
|
f"No new bars since {latest.isoformat()} ({gap_days}d gap). "
|
||||||
|
"The symbol may be halted, delisted, or renamed under a new ticker — "
|
||||||
|
"check the listing and add/fetch the current symbol if it changed."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
return IngestionResult(
|
return IngestionResult(
|
||||||
symbol=ticker.symbol,
|
symbol=ticker.symbol,
|
||||||
records_ingested=ingested_count,
|
records_ingested=ingested_count,
|
||||||
|
|||||||
@@ -1,17 +1,18 @@
|
|||||||
"""Cross-sectional 12-1 momentum ranking for the universe.
|
"""Cross-sectional residual 12-1 momentum ranking for the universe.
|
||||||
|
|
||||||
The activation gate selects the top ``min_momentum_percentile`` of the universe
|
The activation gate selects the top ``min_momentum_percentile`` of the universe
|
||||||
by 12-1 month momentum (return from ~12 months ago to ~1 month ago — the one
|
by residual 12-1 month momentum: the stock's 12-1 return after subtracting its
|
||||||
price signal the backtest showed sorts forward returns). The daily scan ranks
|
estimated benchmark beta contribution over the same formation window. The daily
|
||||||
every ticker and stores each setup's percentile (see ``rr_scanner_service``), so
|
scan ranks every ticker and stores each setup's percentile (see
|
||||||
the live list, the Track Record's qualified stats, and outcome evaluation all gate
|
``rr_scanner_service``), so the live list, the Track Record's qualified stats,
|
||||||
on the same value.
|
and outcome evaluation all gate on the same value.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
|
from datetime import date
|
||||||
|
|
||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
@@ -26,6 +27,34 @@ logger = logging.getLogger(__name__)
|
|||||||
_MOM_LOOKBACK = 252
|
_MOM_LOOKBACK = 252
|
||||||
_MOM_SKIP = 21
|
_MOM_SKIP = 21
|
||||||
|
|
||||||
|
# Promoted production ordering: strategy_rank blends the momentum and realized-
|
||||||
|
# volatility percentiles. Single source of truth — the backtest's production
|
||||||
|
# ranking key imports these so live and simulated ordering cannot drift.
|
||||||
|
STRATEGY_RANK_MOMENTUM_WEIGHT = 0.8
|
||||||
|
STRATEGY_RANK_VOL_WEIGHT = 1.0 - STRATEGY_RANK_MOMENTUM_WEIGHT
|
||||||
|
|
||||||
|
|
||||||
|
def blend_strategy_rank(
|
||||||
|
momentum_percentile: float | None,
|
||||||
|
volatility_percentile: float | None,
|
||||||
|
*,
|
||||||
|
momentum_weight: float = STRATEGY_RANK_MOMENTUM_WEIGHT,
|
||||||
|
) -> float | None:
|
||||||
|
"""80/20 production rank with mom-only fallback when vol is missing.
|
||||||
|
|
||||||
|
Live and backtest must share this policy: missing vol must not send a
|
||||||
|
residual-qualified name to the bottom of the book (that was the old
|
||||||
|
backtest behaviour when either leg was None).
|
||||||
|
"""
|
||||||
|
if momentum_percentile is not None and volatility_percentile is not None:
|
||||||
|
vol_weight = 1.0 - momentum_weight
|
||||||
|
return round(
|
||||||
|
float(momentum_percentile) * momentum_weight
|
||||||
|
+ float(volatility_percentile) * vol_weight,
|
||||||
|
2,
|
||||||
|
)
|
||||||
|
return float(momentum_percentile) if momentum_percentile is not None else None
|
||||||
|
|
||||||
|
|
||||||
def compute_12_1_momentum(closes: list[float]) -> float | None:
|
def compute_12_1_momentum(closes: list[float]) -> float | None:
|
||||||
"""Return over the window ending ~1 month ago, starting ~12 months ago.
|
"""Return over the window ending ~1 month ago, starting ~12 months ago.
|
||||||
@@ -35,29 +64,153 @@ def compute_12_1_momentum(closes: list[float]) -> float | None:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def compute_residual_12_1_momentum(
|
||||||
|
dates: list[date],
|
||||||
|
closes: list[float],
|
||||||
|
benchmark_closes: dict[date, float],
|
||||||
|
) -> float | None:
|
||||||
|
"""12-1 momentum after removing linear benchmark exposure.
|
||||||
|
|
||||||
|
Estimate beta from daily stock/benchmark returns over the standard 12-1
|
||||||
|
formation window, then sum stock return minus beta * benchmark return. No
|
||||||
|
intercept is subtracted: fitting an intercept over the same window would make
|
||||||
|
residuals sum to roughly zero and destroy the ranking signal.
|
||||||
|
"""
|
||||||
|
i = len(closes) - 1
|
||||||
|
if not benchmark_closes or len(dates) != len(closes) or i - _MOM_LOOKBACK < 0:
|
||||||
|
return None
|
||||||
|
|
||||||
|
stock_rets: list[float] = []
|
||||||
|
market_rets: list[float] = []
|
||||||
|
for k in range(i - _MOM_LOOKBACK + 1, i - _MOM_SKIP + 1):
|
||||||
|
prev_close = closes[k - 1]
|
||||||
|
bench_prev = benchmark_closes.get(dates[k - 1])
|
||||||
|
bench_cur = benchmark_closes.get(dates[k])
|
||||||
|
if prev_close <= 0 or bench_prev is None or bench_cur is None or bench_prev <= 0:
|
||||||
|
continue
|
||||||
|
stock_rets.append(closes[k] / prev_close - 1.0)
|
||||||
|
market_rets.append(bench_cur / bench_prev - 1.0)
|
||||||
|
|
||||||
|
if len(stock_rets) < 100:
|
||||||
|
return None
|
||||||
|
mean_market = sum(market_rets) / len(market_rets)
|
||||||
|
mean_stock = sum(stock_rets) / len(stock_rets)
|
||||||
|
var_market = sum((x - mean_market) ** 2 for x in market_rets)
|
||||||
|
if var_market <= 0:
|
||||||
|
return None
|
||||||
|
cov = sum(
|
||||||
|
(stock_rets[k] - mean_stock) * (market_rets[k] - mean_market)
|
||||||
|
for k in range(len(stock_rets))
|
||||||
|
)
|
||||||
|
beta = cov / var_market
|
||||||
|
return sum(stock_rets[k] - beta * market_rets[k] for k in range(len(stock_rets)))
|
||||||
|
|
||||||
|
|
||||||
|
async def _load_activation_benchmark(db: AsyncSession) -> dict[date, float]:
|
||||||
|
"""Load SPY closes for residual momentum; refresh once if the table is empty."""
|
||||||
|
try:
|
||||||
|
from app.services.benchmark_service import load_benchmark_closes, refresh_benchmark_prices
|
||||||
|
|
||||||
|
closes = await load_benchmark_closes(db)
|
||||||
|
if closes:
|
||||||
|
return closes
|
||||||
|
await refresh_benchmark_prices(db)
|
||||||
|
return await load_benchmark_closes(db)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Residual momentum benchmark load failed; falling back to raw momentum")
|
||||||
|
return {}
|
||||||
|
|
||||||
|
|
||||||
async def compute_momentum_percentiles(db: AsyncSession) -> dict[str, float]:
|
async def compute_momentum_percentiles(db: AsyncSession) -> dict[str, float]:
|
||||||
"""Compute each ticker's 12-1 momentum and rank the universe into a
|
"""Momentum leg only — thin view of ``compute_activation_ranks``.
|
||||||
``{symbol: percentile}`` map (0–100, 100 = strongest momentum). Tickers
|
|
||||||
without a full year of history are absent (can't be ranked)."""
|
Prefer ``compute_activation_ranks`` in new code (includes vol + strategy_rank).
|
||||||
|
Kept so tests/helpers that only need the residual/raw percentile map stay simple.
|
||||||
|
"""
|
||||||
|
ranks = await compute_activation_ranks(db)
|
||||||
|
return {
|
||||||
|
sym: float(row["momentum_percentile"])
|
||||||
|
for sym, row in ranks.items()
|
||||||
|
if row.get("momentum_percentile") is not None
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def compute_realized_vol_6m(closes: list[float]) -> float | None:
|
||||||
|
"""126-trading-day realized daily volatility. Higher = more volatile."""
|
||||||
|
if len(closes) < 127:
|
||||||
|
return None
|
||||||
|
rets = [
|
||||||
|
closes[k] / closes[k - 1] - 1.0
|
||||||
|
for k in range(len(closes) - 126, len(closes))
|
||||||
|
if closes[k - 1] > 0
|
||||||
|
]
|
||||||
|
if len(rets) < 2:
|
||||||
|
return None
|
||||||
|
mean = sum(rets) / len(rets)
|
||||||
|
var = sum((x - mean) ** 2 for x in rets) / (len(rets) - 1)
|
||||||
|
return var ** 0.5
|
||||||
|
|
||||||
|
|
||||||
|
def _percentiles(values: dict[str, float]) -> dict[str, float]:
|
||||||
|
ranked = sorted(values, key=lambda s: values[s])
|
||||||
|
n = len(ranked)
|
||||||
|
return {
|
||||||
|
sym: round((rank / (n - 1) * 100.0) if n > 1 else 100.0, 2)
|
||||||
|
for rank, sym in enumerate(ranked)
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def compute_activation_ranks(db: AsyncSession) -> dict[str, dict[str, float | None]]:
|
||||||
|
"""Compute production activation ranks for the live scanner.
|
||||||
|
|
||||||
|
``momentum_percentile`` remains the residual/raw 12-1 gate. ``strategy_rank``
|
||||||
|
is the promoted production ordering score: 80% activation momentum rank plus
|
||||||
|
20% 6-month realized-volatility percentile. Live ranks are universe-wide
|
||||||
|
before scanning; the research backtest ranked each weekly setup-candidate
|
||||||
|
cross-section, so this is the deliberate production approximation.
|
||||||
|
"""
|
||||||
result = await db.execute(select(Ticker).order_by(Ticker.symbol))
|
result = await db.execute(select(Ticker).order_by(Ticker.symbol))
|
||||||
tickers = list(result.scalars().all())
|
tickers = list(result.scalars().all())
|
||||||
|
|
||||||
momentum: dict[str, float] = {}
|
benchmark_closes = await _load_activation_benchmark(db)
|
||||||
|
using_residual = len(benchmark_closes) >= _MOM_LOOKBACK
|
||||||
|
|
||||||
|
momentum_values: dict[str, float] = {}
|
||||||
|
vol_values: dict[str, float] = {}
|
||||||
for ticker in tickers:
|
for ticker in tickers:
|
||||||
try:
|
try:
|
||||||
records = await query_ohlcv(db, ticker.symbol)
|
records = await query_ohlcv(db, ticker.symbol)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception("Momentum fetch failed for %s", ticker.symbol)
|
logger.exception("Activation rank fetch failed for %s", ticker.symbol)
|
||||||
continue
|
continue
|
||||||
m = compute_12_1_momentum([float(r.close) for r in records])
|
closes = [float(r.close) for r in records]
|
||||||
if m is not None:
|
momentum = (
|
||||||
momentum[ticker.symbol] = m
|
compute_residual_12_1_momentum([r.date for r in records], closes, benchmark_closes)
|
||||||
|
if using_residual
|
||||||
|
else compute_12_1_momentum(closes)
|
||||||
|
)
|
||||||
|
if momentum is not None:
|
||||||
|
momentum_values[ticker.symbol] = momentum
|
||||||
|
vol = compute_realized_vol_6m(closes)
|
||||||
|
if vol is not None:
|
||||||
|
vol_values[ticker.symbol] = vol
|
||||||
|
|
||||||
ranked = sorted(momentum, key=lambda s: momentum[s])
|
momentum_percentiles = _percentiles(momentum_values)
|
||||||
n = len(ranked)
|
vol_percentiles = _percentiles(vol_values)
|
||||||
percentiles = {
|
symbols = set(momentum_percentiles) | set(vol_percentiles)
|
||||||
sym: round((rank / (n - 1) * 100.0) if n > 1 else 100.0, 2)
|
ranks: dict[str, dict[str, float | None]] = {}
|
||||||
for rank, sym in enumerate(ranked)
|
for sym in symbols:
|
||||||
}
|
momentum_pct = momentum_percentiles.get(sym)
|
||||||
logger.info(json.dumps({"event": "momentum_ranked", "tickers": n}))
|
vol_pct = vol_percentiles.get(sym)
|
||||||
return percentiles
|
ranks[sym] = {
|
||||||
|
"momentum_percentile": momentum_pct,
|
||||||
|
"volatility_percentile": vol_pct,
|
||||||
|
"strategy_rank": blend_strategy_rank(momentum_pct, vol_pct),
|
||||||
|
}
|
||||||
|
|
||||||
|
logger.info(json.dumps({
|
||||||
|
"event": "activation_ranked",
|
||||||
|
"signal": "residual_12_1_plus_vol_80_20" if using_residual else "raw_12_1_plus_vol_80_20",
|
||||||
|
"tickers": len(ranks),
|
||||||
|
}))
|
||||||
|
return ranks
|
||||||
|
|||||||
@@ -1,11 +1,15 @@
|
|||||||
"""Trade setup outcome evaluation service.
|
"""Trade setup outcome evaluation service.
|
||||||
|
|
||||||
Closes the feedback loop on R:R scanner setups: walks daily OHLCV bars
|
Diagnostic barrier resolution for scanner setups: walks daily OHLCV bars
|
||||||
after detection and records whether the stop or the target was hit first.
|
after detection and records whether the gate target or the stop was hit first.
|
||||||
|
|
||||||
|
This is **not** the production exit model. Live paper trades and the portfolio
|
||||||
|
monitor use ATR trail / max hold and never exit at the gate target. Track-record
|
||||||
|
stats from this path measure gate-level plumbing, not ATR-trail book expectancy.
|
||||||
|
|
||||||
Outcome semantics (entry is the close at detection time, i.e. market entry):
|
Outcome semantics (entry is the close at detection time, i.e. market entry):
|
||||||
- target_hit: target reached before the stop
|
- target_hit: gate target reached before the stop
|
||||||
- stop_hit: stop reached before the target
|
- stop_hit: stop reached before the gate target
|
||||||
- ambiguous: stop AND target both within the same daily bar — with daily
|
- ambiguous: stop AND target both within the same daily bar — with daily
|
||||||
granularity the order is unknowable, counted as a loss in stats
|
granularity the order is unknowable, counted as a loss in stats
|
||||||
- expired: neither level hit within ``max_bars`` trading days
|
- expired: neither level hit within ``max_bars`` trading days
|
||||||
@@ -16,7 +20,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from datetime import date, datetime, timezone
|
from datetime import date, datetime, timedelta, timezone
|
||||||
|
|
||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
@@ -34,6 +38,13 @@ OUTCOME_EXPIRED = "expired"
|
|||||||
|
|
||||||
DEFAULT_MAX_BARS = 30
|
DEFAULT_MAX_BARS = 30
|
||||||
|
|
||||||
|
# A setup's outcome is only unbiased once its full evaluation window has elapsed:
|
||||||
|
# until then, near stops resolve as losses within days while far targets are still
|
||||||
|
# pending, so a young sample skews sharply negative. Only count setups detected at
|
||||||
|
# least this many CALENDAR days ago (~max_bars trading days, ×1.5 to cover
|
||||||
|
# weekends/holidays). Younger setups are reported separately as "maturing".
|
||||||
|
_MATURITY_DAYS = int(DEFAULT_MAX_BARS * 1.5)
|
||||||
|
|
||||||
# Confidence buckets for the performance breakdown
|
# Confidence buckets for the performance breakdown
|
||||||
_CONFIDENCE_BUCKETS = [
|
_CONFIDENCE_BUCKETS = [
|
||||||
("<50%", 0.0, 50.0),
|
("<50%", 0.0, 50.0),
|
||||||
@@ -183,7 +194,12 @@ async def get_performance_stats(
|
|||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
config: dict | None = None,
|
config: dict | None = None,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Aggregate outcome statistics over all evaluated trade setups.
|
"""Aggregate outcome statistics over the *matured* evaluated trade setups.
|
||||||
|
|
||||||
|
Only setups whose full evaluation window has elapsed (see ``_MATURITY_DAYS``)
|
||||||
|
are counted, so the headline isn't dominated by quick stop-outs while slower
|
||||||
|
winners are still in flight. ``maturing`` reports how many are excluded for
|
||||||
|
being too young.
|
||||||
|
|
||||||
avg_r is the expectancy per trade in R-multiples (win = +rr_ratio,
|
avg_r is the expectancy per trade in R-multiples (win = +rr_ratio,
|
||||||
loss = -1R, expired = 0R). A positive avg_r means the signals have
|
loss = -1R, expired = 0R). A positive avg_r means the signals have
|
||||||
@@ -197,13 +213,23 @@ async def get_performance_stats(
|
|||||||
result = await db.execute(
|
result = await db.execute(
|
||||||
select(TradeSetup).where(TradeSetup.actual_outcome.is_not(None))
|
select(TradeSetup).where(TradeSetup.actual_outcome.is_not(None))
|
||||||
)
|
)
|
||||||
evaluated = list(result.scalars().all())
|
evaluated_all = list(result.scalars().all())
|
||||||
|
|
||||||
|
# Matured cohort only — see _MATURITY_DAYS. Setups whose window hasn't fully
|
||||||
|
# elapsed are excluded so quick stop-outs can't drag the headline negative
|
||||||
|
# while their slower-to-resolve winners are still in flight.
|
||||||
|
cutoff_date = (datetime.now(timezone.utc) - timedelta(days=_MATURITY_DAYS)).date()
|
||||||
|
evaluated = [s for s in evaluated_all if s.detected_at.date() <= cutoff_date]
|
||||||
|
|
||||||
pending_result = await db.execute(
|
pending_result = await db.execute(
|
||||||
select(TradeSetup.id).where(TradeSetup.actual_outcome.is_(None))
|
select(TradeSetup.id).where(TradeSetup.actual_outcome.is_(None))
|
||||||
)
|
)
|
||||||
pending_count = len(pending_result.scalars().all())
|
pending_count = len(pending_result.scalars().all())
|
||||||
|
|
||||||
|
# Still inside their measurement window (excluded above so they can't bias the
|
||||||
|
# stats): young setups that already resolved + everything still pending.
|
||||||
|
maturing_count = (len(evaluated_all) - len(evaluated)) + pending_count
|
||||||
|
|
||||||
if config is not None:
|
if config is not None:
|
||||||
qualified = [s for s in evaluated if setup_qualifies(s, config)]
|
qualified = [s for s in evaluated if setup_qualifies(s, config)]
|
||||||
else:
|
else:
|
||||||
@@ -229,6 +255,7 @@ async def get_performance_stats(
|
|||||||
return {
|
return {
|
||||||
"overall": _bucket_stats(qualified),
|
"overall": _bucket_stats(qualified),
|
||||||
"pending": pending_count,
|
"pending": pending_count,
|
||||||
|
"maturing": maturing_count,
|
||||||
"by_direction": {k: _bucket_stats(v) for k, v in sorted(by_direction.items())},
|
"by_direction": {k: _bucket_stats(v) for k, v in sorted(by_direction.items())},
|
||||||
"by_action": {k: _bucket_stats(v) for k, v in sorted(by_action.items())},
|
"by_action": {k: _bucket_stats(v) for k, v in sorted(by_action.items())},
|
||||||
"by_confidence": {
|
"by_confidence": {
|
||||||
|
|||||||
@@ -2,7 +2,9 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from datetime import datetime, timezone
|
import bisect
|
||||||
|
import logging
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
from sqlalchemy import and_, func, select
|
from sqlalchemy import and_, func, select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
@@ -11,6 +13,7 @@ from app.exceptions import NotFoundError, ValidationError
|
|||||||
from app.models.ohlcv import OHLCVRecord
|
from app.models.ohlcv import OHLCVRecord
|
||||||
from app.models.paper_trade import PaperTrade
|
from app.models.paper_trade import PaperTrade
|
||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import benchmark_service, settings_store
|
||||||
from app.services.outcome_service import (
|
from app.services.outcome_service import (
|
||||||
OUTCOME_AMBIGUOUS,
|
OUTCOME_AMBIGUOUS,
|
||||||
OUTCOME_STOP_HIT,
|
OUTCOME_STOP_HIT,
|
||||||
@@ -18,6 +21,86 @@ from app.services.outcome_service import (
|
|||||||
Bar,
|
Bar,
|
||||||
evaluate_setup_against_bars,
|
evaluate_setup_against_bars,
|
||||||
)
|
)
|
||||||
|
from app.services.trade_policy import MANUAL_BOOK, SHADOW_BOOK, get_reentry_gate_locks
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Exit policy for OPEN paper trades (auto-close). Production defaults to the
|
||||||
|
# July 2026 promoted strategy: initial stop + 3x ATR trailing stop, with a max
|
||||||
|
# 30-trading-day hold. The older percent trail and target/stop modes remain
|
||||||
|
# selectable for comparison. Stored in SystemSetting so it's tunable and visible.
|
||||||
|
KEY_EXIT_MODE = "paper_exit_mode"
|
||||||
|
KEY_TRAILING_PCT = "paper_trailing_pct"
|
||||||
|
KEY_ATR_MULTIPLIER = "paper_atr_multiplier"
|
||||||
|
KEY_HOLD_DAYS = "paper_hold_days"
|
||||||
|
DEFAULT_EXIT_MODE = "atr_trailing"
|
||||||
|
DEFAULT_TRAILING_PCT = 12.0
|
||||||
|
DEFAULT_ATR_MULTIPLIER = 3.0
|
||||||
|
DEFAULT_HOLD_DAYS = 30
|
||||||
|
|
||||||
|
_VALID_EXIT_MODES = ("time", "trailing", "atr_trailing", "target")
|
||||||
|
|
||||||
|
|
||||||
|
async def get_exit_policy(db: AsyncSession) -> dict:
|
||||||
|
"""Active auto-exit policy:
|
||||||
|
{'mode': 'time'|'trailing'|'atr_trailing'|'target', ...}."""
|
||||||
|
mode = (await settings_store.get_value(db, KEY_EXIT_MODE, DEFAULT_EXIT_MODE)).strip().lower()
|
||||||
|
if mode not in _VALID_EXIT_MODES:
|
||||||
|
mode = DEFAULT_EXIT_MODE
|
||||||
|
raw = await settings_store.get_value(db, KEY_TRAILING_PCT, str(DEFAULT_TRAILING_PCT))
|
||||||
|
try:
|
||||||
|
pct = float(raw)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
pct = DEFAULT_TRAILING_PCT
|
||||||
|
pct = max(0.5, min(90.0, pct))
|
||||||
|
raw_atr = await settings_store.get_value(db, KEY_ATR_MULTIPLIER, str(DEFAULT_ATR_MULTIPLIER))
|
||||||
|
try:
|
||||||
|
atr_multiplier = float(raw_atr)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
atr_multiplier = DEFAULT_ATR_MULTIPLIER
|
||||||
|
atr_multiplier = max(0.5, min(10.0, atr_multiplier))
|
||||||
|
raw_days = await settings_store.get_value(db, KEY_HOLD_DAYS, str(DEFAULT_HOLD_DAYS))
|
||||||
|
try:
|
||||||
|
hold_days = int(float(raw_days))
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
hold_days = DEFAULT_HOLD_DAYS
|
||||||
|
hold_days = max(2, min(250, hold_days))
|
||||||
|
return {
|
||||||
|
"mode": mode,
|
||||||
|
"trailing_pct": pct,
|
||||||
|
"atr_multiplier": atr_multiplier,
|
||||||
|
"hold_days": hold_days,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def set_exit_policy(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
mode: str | None = None,
|
||||||
|
trailing_pct: float | None = None,
|
||||||
|
atr_multiplier: float | None = None,
|
||||||
|
hold_days: int | None = None,
|
||||||
|
) -> dict:
|
||||||
|
"""Persist the auto-exit policy (admin). Validates inputs."""
|
||||||
|
if mode is not None:
|
||||||
|
mode = mode.strip().lower()
|
||||||
|
if mode not in _VALID_EXIT_MODES:
|
||||||
|
raise ValidationError("mode must be 'time', 'trailing', 'atr_trailing' or 'target'")
|
||||||
|
await settings_store.upsert_setting(db, KEY_EXIT_MODE, mode)
|
||||||
|
if trailing_pct is not None:
|
||||||
|
if not 0.5 <= float(trailing_pct) <= 90.0:
|
||||||
|
raise ValidationError("trailing_pct must be between 0.5 and 90")
|
||||||
|
await settings_store.upsert_setting(db, KEY_TRAILING_PCT, str(float(trailing_pct)))
|
||||||
|
if atr_multiplier is not None:
|
||||||
|
if not 0.5 <= float(atr_multiplier) <= 10.0:
|
||||||
|
raise ValidationError("atr_multiplier must be between 0.5 and 10")
|
||||||
|
await settings_store.upsert_setting(db, KEY_ATR_MULTIPLIER, str(float(atr_multiplier)))
|
||||||
|
if hold_days is not None:
|
||||||
|
if not 2 <= int(hold_days) <= 250:
|
||||||
|
raise ValidationError("hold_days must be between 2 and 250")
|
||||||
|
await settings_store.upsert_setting(db, KEY_HOLD_DAYS, str(int(hold_days)))
|
||||||
|
await db.commit()
|
||||||
|
return await get_exit_policy(db)
|
||||||
|
|
||||||
|
|
||||||
async def _get_ticker(db: AsyncSession, symbol: str) -> Ticker:
|
async def _get_ticker(db: AsyncSession, symbol: str) -> Ticker:
|
||||||
@@ -50,6 +133,177 @@ async def _latest_closes(db: AsyncSession, ticker_ids: set[int]) -> dict[int, fl
|
|||||||
return {tid: float(close) for tid, close in result.all()}
|
return {tid: float(close) for tid, close in result.all()}
|
||||||
|
|
||||||
|
|
||||||
|
async def _max_high_after(db: AsyncSession, ticker_id: int, since: date) -> float | None:
|
||||||
|
"""Highest high strictly after ``since`` — the running peak for a trailing stop."""
|
||||||
|
result = await db.execute(
|
||||||
|
select(func.max(OHLCVRecord.high)).where(
|
||||||
|
OHLCVRecord.ticker_id == ticker_id, OHLCVRecord.date > since
|
||||||
|
)
|
||||||
|
)
|
||||||
|
v = result.scalar()
|
||||||
|
return float(v) if v is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _time_close(
|
||||||
|
direction: str, init_stop: float, hold_days: int, rows: list[tuple]
|
||||||
|
) -> tuple[float, date, str] | None:
|
||||||
|
"""Walk post-entry ``rows`` of (date, open, high, low, close); close at the
|
||||||
|
initial stop if hit (a gap through it fills at the open, matching the
|
||||||
|
backtest's fill model), else at the ``hold_days``-th bar's close ('time').
|
||||||
|
None while neither has happened."""
|
||||||
|
long = direction == "long"
|
||||||
|
for i, (d, open_, high, low, close) in enumerate(rows):
|
||||||
|
if (low <= init_stop) if long else (high >= init_stop):
|
||||||
|
fill = min(init_stop, open_) if long else max(init_stop, open_)
|
||||||
|
return float(fill), d, "stop"
|
||||||
|
if i + 1 >= hold_days:
|
||||||
|
return float(close), d, "time"
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _trailing_close(
|
||||||
|
direction: str, entry: float, init_stop: float, trail_frac: float, bars: list[Bar]
|
||||||
|
) -> tuple[float, date, str] | None:
|
||||||
|
"""Walk post-entry bars; return (price, date, reason) when the trailing or initial
|
||||||
|
stop is hit, else None. The stop only ratchets up: max(init_stop, peak*(1-trail))
|
||||||
|
for a long. reason = 'trailing' once it's above the initial stop, else 'stop'."""
|
||||||
|
long = direction == "long"
|
||||||
|
peak = entry
|
||||||
|
for b in bars:
|
||||||
|
if long:
|
||||||
|
level = max(init_stop, peak * (1 - trail_frac))
|
||||||
|
if b.low <= level:
|
||||||
|
return level, b.date, ("trailing" if level > init_stop else "stop")
|
||||||
|
if b.high > peak:
|
||||||
|
peak = b.high
|
||||||
|
else:
|
||||||
|
level = min(init_stop, peak * (1 + trail_frac))
|
||||||
|
if b.high >= level:
|
||||||
|
return level, b.date, ("trailing" if level < init_stop else "stop")
|
||||||
|
if b.low < peak:
|
||||||
|
peak = b.low
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _atr_series_from_rows(rows: list[tuple], period: int = 14) -> list[float | None]:
|
||||||
|
"""ATR at each index i, equal to ``compute_atr(rows[: i + 1])["atr"]`` but
|
||||||
|
computed in a single O(n) Wilder pass instead of re-smoothing the whole
|
||||||
|
prefix per bar. None where there are fewer than ``period + 1`` bars or the
|
||||||
|
rounded ATR is non-positive. ``period`` mirrors ``compute_atr``'s default;
|
||||||
|
keep them in sync.
|
||||||
|
|
||||||
|
Exactness: ``compute_atr`` keeps its running ATR unrounded through the
|
||||||
|
recurrence and rounds only at return, so storing ``round(running, 4)`` at
|
||||||
|
each index reproduces its per-prefix value bit-for-bit.
|
||||||
|
"""
|
||||||
|
n = len(rows)
|
||||||
|
out: list[float | None] = [None] * n
|
||||||
|
if n < period + 1:
|
||||||
|
return out
|
||||||
|
tr = [0.0] * n
|
||||||
|
for i in range(1, n):
|
||||||
|
high, low, prev_close = float(rows[i][2]), float(rows[i][3]), float(rows[i - 1][4])
|
||||||
|
tr[i] = max(high - low, abs(high - prev_close), abs(low - prev_close))
|
||||||
|
running = sum(tr[1 : period + 1]) / period
|
||||||
|
rounded = round(running, 4)
|
||||||
|
out[period] = rounded if rounded > 0 else None
|
||||||
|
for j in range(period + 1, n):
|
||||||
|
running = (running * (period - 1) + tr[j]) / period
|
||||||
|
rounded = round(running, 4)
|
||||||
|
out[j] = rounded if rounded > 0 else None
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _atr_trailing_level(
|
||||||
|
direction: str,
|
||||||
|
entry: float,
|
||||||
|
init_stop: float,
|
||||||
|
atr_multiplier: float,
|
||||||
|
rows: list[tuple],
|
||||||
|
opened_on: date,
|
||||||
|
) -> float:
|
||||||
|
"""Current ATR trailing stop level after all available post-entry closes."""
|
||||||
|
long = direction == "long"
|
||||||
|
stop = float(init_stop)
|
||||||
|
anchor = float(entry)
|
||||||
|
atr_by_idx = _atr_series_from_rows(rows)
|
||||||
|
for idx, (d, _, _, _, close) in enumerate(rows):
|
||||||
|
if d <= opened_on:
|
||||||
|
continue
|
||||||
|
close = float(close)
|
||||||
|
atr = atr_by_idx[idx]
|
||||||
|
if long:
|
||||||
|
anchor = max(anchor, close)
|
||||||
|
if atr is not None:
|
||||||
|
next_stop = anchor - atr_multiplier * atr
|
||||||
|
if next_stop < close:
|
||||||
|
stop = max(stop, next_stop)
|
||||||
|
else:
|
||||||
|
anchor = min(anchor, close)
|
||||||
|
if atr is not None:
|
||||||
|
next_stop = anchor + atr_multiplier * atr
|
||||||
|
if next_stop > close:
|
||||||
|
stop = min(stop, next_stop)
|
||||||
|
return stop
|
||||||
|
|
||||||
|
|
||||||
|
def _atr_trailing_close(
|
||||||
|
direction: str,
|
||||||
|
entry: float,
|
||||||
|
init_stop: float,
|
||||||
|
atr_multiplier: float,
|
||||||
|
hold_days: int,
|
||||||
|
rows: list[tuple],
|
||||||
|
opened_on: date,
|
||||||
|
) -> tuple[float, date, str] | None:
|
||||||
|
"""Initial stop + ATR trailing stop + max hold, matching the portfolio sim.
|
||||||
|
|
||||||
|
Stop checks happen before the same day's trailing update, so a newly ratcheted
|
||||||
|
stop becomes active on the next bar. Gaps through the stop fill at the open.
|
||||||
|
"""
|
||||||
|
long = direction == "long"
|
||||||
|
stop = float(init_stop)
|
||||||
|
anchor = float(entry)
|
||||||
|
bars_held = 0
|
||||||
|
atr_by_idx = _atr_series_from_rows(rows)
|
||||||
|
for idx, (d, open_, high, low, close) in enumerate(rows):
|
||||||
|
if d <= opened_on:
|
||||||
|
continue
|
||||||
|
open_ = float(open_)
|
||||||
|
high = float(high)
|
||||||
|
low = float(low)
|
||||||
|
close = float(close)
|
||||||
|
bars_held += 1
|
||||||
|
|
||||||
|
if long:
|
||||||
|
if low <= stop:
|
||||||
|
reason = "trailing" if stop > init_stop + 1e-9 else "stop"
|
||||||
|
return min(stop, open_), d, reason
|
||||||
|
else:
|
||||||
|
if high >= stop:
|
||||||
|
reason = "trailing" if stop < init_stop - 1e-9 else "stop"
|
||||||
|
return max(stop, open_), d, reason
|
||||||
|
|
||||||
|
if bars_held >= hold_days:
|
||||||
|
return close, d, "time"
|
||||||
|
|
||||||
|
atr = atr_by_idx[idx]
|
||||||
|
if long:
|
||||||
|
anchor = max(anchor, close)
|
||||||
|
if atr is not None:
|
||||||
|
next_stop = anchor - atr_multiplier * atr
|
||||||
|
if next_stop < close:
|
||||||
|
stop = max(stop, next_stop)
|
||||||
|
else:
|
||||||
|
anchor = min(anchor, close)
|
||||||
|
if atr is not None:
|
||||||
|
next_stop = anchor + atr_multiplier * atr
|
||||||
|
if next_stop > close:
|
||||||
|
stop = min(stop, next_stop)
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
async def create_trade(
|
async def create_trade(
|
||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
user_id: int,
|
user_id: int,
|
||||||
@@ -68,6 +322,10 @@ async def create_trade(
|
|||||||
raise ValidationError("shares and entry_price must be positive")
|
raise ValidationError("shares and entry_price must be positive")
|
||||||
|
|
||||||
ticker = await _get_ticker(db, symbol)
|
ticker = await _get_ticker(db, symbol)
|
||||||
|
if ticker.id in await get_reentry_gate_locks(db):
|
||||||
|
raise ValidationError(
|
||||||
|
f"{ticker.symbol} requires a post-stop gate reset before re-entry"
|
||||||
|
)
|
||||||
trade = PaperTrade(
|
trade = PaperTrade(
|
||||||
user_id=user_id,
|
user_id=user_id,
|
||||||
ticker_id=ticker.id,
|
ticker_id=ticker.id,
|
||||||
@@ -78,6 +336,9 @@ async def create_trade(
|
|||||||
target=target,
|
target=target,
|
||||||
status="open",
|
status="open",
|
||||||
opened_at=datetime.now(timezone.utc),
|
opened_at=datetime.now(timezone.utc),
|
||||||
|
# Near-close cutover era — Track Record must not mix with morning-scan
|
||||||
|
# fills or future broker-routed fills when comparing to backtests.
|
||||||
|
fill_mode="near_close",
|
||||||
)
|
)
|
||||||
db.add(trade)
|
db.add(trade)
|
||||||
await db.commit()
|
await db.commit()
|
||||||
@@ -85,7 +346,36 @@ async def create_trade(
|
|||||||
return trade
|
return trade
|
||||||
|
|
||||||
|
|
||||||
def _to_dict(trade: PaperTrade, symbol: str, current_price: float | None) -> dict:
|
def _to_dict(
|
||||||
|
trade: PaperTrade,
|
||||||
|
symbol: str,
|
||||||
|
current_price: float | None,
|
||||||
|
benchmark_closes: dict[date, float] | None = None,
|
||||||
|
trailing: tuple[float, float | None] | None = None,
|
||||||
|
holding_sessions: tuple[int, int] | None = None,
|
||||||
|
) -> dict:
|
||||||
|
# For open trades, mark to market; for closed, the realized exit price.
|
||||||
|
ref = current_price if trade.status == "open" else trade.close_price
|
||||||
|
|
||||||
|
# Alpha = trade return − benchmark (SPY) return over the same holding period.
|
||||||
|
benchmark_return = None
|
||||||
|
alpha_pct = None
|
||||||
|
alpha_usd = None
|
||||||
|
if ref is not None and trade.entry_price and benchmark_closes:
|
||||||
|
sign = 1.0 if trade.direction == "long" else -1.0
|
||||||
|
trade_return = (ref - trade.entry_price) / trade.entry_price * 100.0 * sign
|
||||||
|
as_of = (
|
||||||
|
trade.closed_at.date()
|
||||||
|
if trade.status == "closed" and trade.closed_at is not None
|
||||||
|
else date.today()
|
||||||
|
)
|
||||||
|
benchmark_return = benchmark_service.benchmark_return_pct(
|
||||||
|
benchmark_closes, trade.opened_at.date(), as_of
|
||||||
|
)
|
||||||
|
if benchmark_return is not None:
|
||||||
|
alpha_pct = trade_return - benchmark_return
|
||||||
|
alpha_usd = alpha_pct / 100.0 * trade.entry_price * trade.shares
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"id": trade.id,
|
"id": trade.id,
|
||||||
"symbol": symbol,
|
"symbol": symbol,
|
||||||
@@ -98,29 +388,143 @@ def _to_dict(trade: PaperTrade, symbol: str, current_price: float | None) -> dic
|
|||||||
"opened_at": trade.opened_at,
|
"opened_at": trade.opened_at,
|
||||||
"close_price": trade.close_price,
|
"close_price": trade.close_price,
|
||||||
"closed_at": trade.closed_at,
|
"closed_at": trade.closed_at,
|
||||||
# For open trades, mark to market; for closed, the realized exit price.
|
"current_price": ref,
|
||||||
"current_price": current_price if trade.status == "open" else trade.close_price,
|
"benchmark_return_pct": benchmark_return,
|
||||||
|
"alpha_pct": alpha_pct,
|
||||||
|
"alpha_usd": alpha_usd,
|
||||||
|
"close_reason": trade.close_reason,
|
||||||
|
"fill_mode": trade.fill_mode,
|
||||||
|
"trailing_stop": trailing[0] if trailing else None,
|
||||||
|
"trailing_distance_pct": trailing[1] if trailing else None,
|
||||||
|
"sessions_held": holding_sessions[0] if holding_sessions else None,
|
||||||
|
"sessions_remaining": holding_sessions[1] if holding_sessions else None,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
async def list_trades(
|
async def list_trades(
|
||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
user_id: int,
|
user_id: int | None = None,
|
||||||
status: str | None = None,
|
status: str | None = None,
|
||||||
|
book: str | None = MANUAL_BOOK,
|
||||||
) -> list[dict]:
|
) -> list[dict]:
|
||||||
|
"""Trades for the UI. Defaults to the discretionary book.
|
||||||
|
|
||||||
|
Shadow trades are attached to a user row for FK reasons only — they are not
|
||||||
|
that person's decisions. Listing them alongside manual trades would mix two
|
||||||
|
different books in one P&L and let the autonomous record be edited by hand.
|
||||||
|
Pass ``book=None`` to deliberately span both.
|
||||||
|
"""
|
||||||
stmt = (
|
stmt = (
|
||||||
select(PaperTrade, Ticker.symbol)
|
select(PaperTrade, Ticker.symbol)
|
||||||
.join(Ticker, PaperTrade.ticker_id == Ticker.id)
|
.join(Ticker, PaperTrade.ticker_id == Ticker.id)
|
||||||
.where(PaperTrade.user_id == user_id)
|
|
||||||
)
|
)
|
||||||
|
if user_id is not None: # None → all users (single-user app; used by the digest)
|
||||||
|
stmt = stmt.where(PaperTrade.user_id == user_id)
|
||||||
if status is not None:
|
if status is not None:
|
||||||
stmt = stmt.where(PaperTrade.status == status)
|
stmt = stmt.where(PaperTrade.status == status)
|
||||||
|
if book is not None:
|
||||||
|
stmt = stmt.where(PaperTrade.book == book)
|
||||||
stmt = stmt.order_by(PaperTrade.opened_at.desc())
|
stmt = stmt.order_by(PaperTrade.opened_at.desc())
|
||||||
|
|
||||||
rows = (await db.execute(stmt)).all()
|
rows = (await db.execute(stmt)).all()
|
||||||
open_ids = {t.ticker_id for t, _ in rows if t.status == "open"}
|
open_ids = {t.ticker_id for t, _ in rows if t.status == "open"}
|
||||||
prices = await _latest_closes(db, open_ids)
|
prices = await _latest_closes(db, open_ids)
|
||||||
return [_to_dict(t, sym, prices.get(t.ticker_id)) for t, sym in rows]
|
|
||||||
|
# Benchmark closes for alpha — populated by the daily/benchmark job. Empty until
|
||||||
|
# that runs once, in which case alpha is simply left unset (a read path never
|
||||||
|
# makes a provider call).
|
||||||
|
benchmark_closes = await benchmark_service.load_benchmark_closes(db)
|
||||||
|
|
||||||
|
# Current trailing-stop level + distance for open trades (when a trailing
|
||||||
|
# policy is active).
|
||||||
|
policy = await get_exit_policy(db)
|
||||||
|
holding_sessions: dict[int, tuple[int, int]] = {}
|
||||||
|
if policy["mode"] in ("time", "atr_trailing"):
|
||||||
|
hold_days = int(policy["hold_days"])
|
||||||
|
open_trades = [trade for trade, _ in rows if trade.status == "open"]
|
||||||
|
if open_trades:
|
||||||
|
ticker_ids = {trade.ticker_id for trade in open_trades}
|
||||||
|
earliest_opened = min(trade.opened_at.date() for trade in open_trades)
|
||||||
|
session_rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(OHLCVRecord.ticker_id, OHLCVRecord.date)
|
||||||
|
.where(
|
||||||
|
OHLCVRecord.ticker_id.in_(ticker_ids),
|
||||||
|
OHLCVRecord.date > earliest_opened,
|
||||||
|
)
|
||||||
|
.order_by(OHLCVRecord.ticker_id, OHLCVRecord.date)
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
dates_by_ticker: dict[int, list[date]] = {}
|
||||||
|
for ticker_id, session_date in session_rows:
|
||||||
|
dates_by_ticker.setdefault(int(ticker_id), []).append(session_date)
|
||||||
|
for trade in open_trades:
|
||||||
|
dates = dates_by_ticker.get(trade.ticker_id, [])
|
||||||
|
held = len(dates) - bisect.bisect_right(
|
||||||
|
dates, trade.opened_at.date()
|
||||||
|
)
|
||||||
|
# Do not clamp: a policy shortened below the current holding
|
||||||
|
# period must remain visible as overdue until the exit pass runs.
|
||||||
|
holding_sessions[trade.id] = (held, hold_days - held)
|
||||||
|
|
||||||
|
trailing_info: dict[int, tuple[float, float | None]] = {}
|
||||||
|
if policy["mode"] == "trailing":
|
||||||
|
trail_frac = policy["trailing_pct"] / 100.0
|
||||||
|
for t, _ in rows:
|
||||||
|
if t.status != "open":
|
||||||
|
continue
|
||||||
|
max_high = await _max_high_after(db, t.ticker_id, t.opened_at.date())
|
||||||
|
peak = max(t.entry_price, max_high) if max_high is not None else t.entry_price
|
||||||
|
long = t.direction == "long"
|
||||||
|
level = (
|
||||||
|
max(t.stop_loss, peak * (1 - trail_frac))
|
||||||
|
if long
|
||||||
|
else min(t.stop_loss, peak * (1 + trail_frac))
|
||||||
|
)
|
||||||
|
cur = prices.get(t.ticker_id)
|
||||||
|
dist = None
|
||||||
|
if cur:
|
||||||
|
dist = ((cur - level) / cur * 100.0) if long else ((level - cur) / cur * 100.0)
|
||||||
|
trailing_info[t.id] = (level, dist)
|
||||||
|
elif policy["mode"] == "atr_trailing":
|
||||||
|
atr_multiplier = float(policy["atr_multiplier"])
|
||||||
|
for t, _ in rows:
|
||||||
|
if t.status != "open":
|
||||||
|
continue
|
||||||
|
bars_result = await db.execute(
|
||||||
|
select(
|
||||||
|
OHLCVRecord.date, OHLCVRecord.open, OHLCVRecord.high,
|
||||||
|
OHLCVRecord.low, OHLCVRecord.close,
|
||||||
|
)
|
||||||
|
.where(OHLCVRecord.ticker_id == t.ticker_id)
|
||||||
|
.order_by(OHLCVRecord.date.asc())
|
||||||
|
)
|
||||||
|
level = _atr_trailing_level(
|
||||||
|
t.direction,
|
||||||
|
t.entry_price,
|
||||||
|
t.stop_loss,
|
||||||
|
atr_multiplier,
|
||||||
|
bars_result.all(),
|
||||||
|
t.opened_at.date(),
|
||||||
|
)
|
||||||
|
cur = prices.get(t.ticker_id)
|
||||||
|
dist = None
|
||||||
|
if cur:
|
||||||
|
long = t.direction == "long"
|
||||||
|
dist = ((cur - level) / cur * 100.0) if long else ((level - cur) / cur * 100.0)
|
||||||
|
trailing_info[t.id] = (level, dist)
|
||||||
|
|
||||||
|
return [
|
||||||
|
_to_dict(
|
||||||
|
t,
|
||||||
|
sym,
|
||||||
|
prices.get(t.ticker_id),
|
||||||
|
benchmark_closes,
|
||||||
|
trailing_info.get(t.id),
|
||||||
|
holding_sessions.get(t.id),
|
||||||
|
)
|
||||||
|
for t, sym in rows
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
async def close_trade(
|
async def close_trade(
|
||||||
@@ -138,6 +542,13 @@ async def close_trade(
|
|||||||
trade = result.scalar_one_or_none()
|
trade = result.scalar_one_or_none()
|
||||||
if trade is None:
|
if trade is None:
|
||||||
raise NotFoundError(f"Paper trade not found: {trade_id}")
|
raise NotFoundError(f"Paper trade not found: {trade_id}")
|
||||||
|
if trade.book == SHADOW_BOOK:
|
||||||
|
# The shadow book's value is that no human touched it. A hand-closed
|
||||||
|
# position would make its record something other than what the strategy
|
||||||
|
# would have done; it exits only via the automatic exit policy.
|
||||||
|
raise ValidationError(
|
||||||
|
"Shadow book trades are closed by the exit policy, not by hand"
|
||||||
|
)
|
||||||
if trade.status == "closed":
|
if trade.status == "closed":
|
||||||
raise ValidationError("Trade is already closed")
|
raise ValidationError("Trade is already closed")
|
||||||
|
|
||||||
@@ -149,6 +560,7 @@ async def close_trade(
|
|||||||
|
|
||||||
trade.status = "closed"
|
trade.status = "closed"
|
||||||
trade.close_price = float(close_price)
|
trade.close_price = float(close_price)
|
||||||
|
trade.close_reason = "manual"
|
||||||
trade.closed_at = datetime.now(timezone.utc)
|
trade.closed_at = datetime.now(timezone.utc)
|
||||||
await db.commit()
|
await db.commit()
|
||||||
await db.refresh(trade)
|
await db.refresh(trade)
|
||||||
@@ -156,49 +568,383 @@ async def close_trade(
|
|||||||
|
|
||||||
|
|
||||||
async def resolve_open_trades(db: AsyncSession) -> int:
|
async def resolve_open_trades(db: AsyncSession) -> int:
|
||||||
"""Auto-close open trades whose stop or target was hit in the daily bars.
|
"""Auto-close open trades per the active exit policy, from the daily bars.
|
||||||
|
|
||||||
Walks the bars after each trade's open (same logic as the outcome evaluator).
|
Walks the bars after each trade's open. 'atr_trailing' closes at the initial
|
||||||
Target hit → close at the target; stop (or an ambiguous same-bar touch) →
|
stop, a 3x-ATR-style trailing stop, or the hold_days-th close; 'time' closes
|
||||||
close at the stop. Trades that have hit neither stay open. Returns the count
|
at the initial stop or the hold_days-th close; 'trailing' uses the legacy
|
||||||
closed.
|
percent trail; 'target' uses the setup's target or stop. Trades that have hit
|
||||||
|
nothing stay open. Returns the count closed.
|
||||||
"""
|
"""
|
||||||
result = await db.execute(select(PaperTrade).where(PaperTrade.status == "open"))
|
result = await db.execute(select(PaperTrade).where(PaperTrade.status == "open"))
|
||||||
open_trades = list(result.scalars().all())
|
open_trades = list(result.scalars().all())
|
||||||
if not open_trades:
|
if not open_trades:
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
policy = await get_exit_policy(db)
|
||||||
|
mode = policy["mode"]
|
||||||
|
trail_frac = policy["trailing_pct"] / 100.0
|
||||||
|
atr_multiplier = float(policy["atr_multiplier"])
|
||||||
|
hold_days = policy["hold_days"]
|
||||||
|
|
||||||
closed = 0
|
closed = 0
|
||||||
for trade in open_trades:
|
for trade in open_trades:
|
||||||
bars_result = await db.execute(
|
bars_result = await db.execute(
|
||||||
select(OHLCVRecord.date, OHLCVRecord.high, OHLCVRecord.low)
|
select(
|
||||||
|
OHLCVRecord.date, OHLCVRecord.open, OHLCVRecord.high,
|
||||||
|
OHLCVRecord.low, OHLCVRecord.close,
|
||||||
|
)
|
||||||
.where(
|
.where(
|
||||||
OHLCVRecord.ticker_id == trade.ticker_id,
|
OHLCVRecord.ticker_id == trade.ticker_id,
|
||||||
OHLCVRecord.date > trade.opened_at.date(),
|
OHLCVRecord.date > trade.opened_at.date(),
|
||||||
)
|
)
|
||||||
.order_by(OHLCVRecord.date.asc())
|
.order_by(OHLCVRecord.date.asc())
|
||||||
)
|
)
|
||||||
bars = [Bar(date=d, high=h, low=lo) for d, h, lo in bars_result.all()]
|
post_rows = bars_result.all()
|
||||||
|
bars = [Bar(date=d, high=h, low=lo) for d, _, h, lo, _ in post_rows]
|
||||||
if not bars:
|
if not bars:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# max_bars beyond the data so a still-open trade returns undecided (not "expired").
|
if mode == "time":
|
||||||
outcome, outcome_date = evaluate_setup_against_bars(
|
hit = _time_close(trade.direction, trade.stop_loss, hold_days, post_rows)
|
||||||
trade.direction, trade.stop_loss, trade.target, bars, max_bars=len(bars) + 1
|
if hit is None:
|
||||||
)
|
continue # neither the stop nor the hold horizon reached yet
|
||||||
if outcome == OUTCOME_TARGET_HIT:
|
close_price, close_date, reason = hit
|
||||||
trade.close_price = trade.target
|
elif mode == "trailing":
|
||||||
elif outcome in (OUTCOME_STOP_HIT, OUTCOME_AMBIGUOUS):
|
hit = _trailing_close(trade.direction, trade.entry_price, trade.stop_loss, trail_frac, bars)
|
||||||
trade.close_price = trade.stop_loss
|
if hit is None:
|
||||||
|
continue # neither the trailing nor the initial stop reached yet
|
||||||
|
close_price, close_date, reason = hit
|
||||||
|
elif mode == "atr_trailing":
|
||||||
|
all_bars_result = await db.execute(
|
||||||
|
select(
|
||||||
|
OHLCVRecord.date, OHLCVRecord.open, OHLCVRecord.high,
|
||||||
|
OHLCVRecord.low, OHLCVRecord.close,
|
||||||
|
)
|
||||||
|
.where(OHLCVRecord.ticker_id == trade.ticker_id)
|
||||||
|
.order_by(OHLCVRecord.date.asc())
|
||||||
|
)
|
||||||
|
hit = _atr_trailing_close(
|
||||||
|
trade.direction,
|
||||||
|
trade.entry_price,
|
||||||
|
trade.stop_loss,
|
||||||
|
atr_multiplier,
|
||||||
|
hold_days,
|
||||||
|
all_bars_result.all(),
|
||||||
|
trade.opened_at.date(),
|
||||||
|
)
|
||||||
|
if hit is None:
|
||||||
|
continue
|
||||||
|
close_price, close_date, reason = hit
|
||||||
else:
|
else:
|
||||||
continue
|
# max_bars beyond the data so a still-open trade returns undecided (not "expired").
|
||||||
|
outcome, outcome_date = evaluate_setup_against_bars(
|
||||||
|
trade.direction, trade.stop_loss, trade.target, bars, max_bars=len(bars) + 1
|
||||||
|
)
|
||||||
|
if outcome == OUTCOME_TARGET_HIT:
|
||||||
|
close_price, close_date, reason = trade.target, outcome_date, "target"
|
||||||
|
elif outcome in (OUTCOME_STOP_HIT, OUTCOME_AMBIGUOUS):
|
||||||
|
close_price, close_date, reason = trade.stop_loss, outcome_date, "stop"
|
||||||
|
else:
|
||||||
|
continue
|
||||||
|
|
||||||
trade.status = "closed"
|
trade.status = "closed"
|
||||||
trade.closed_at = datetime.combine(
|
trade.close_price = float(close_price)
|
||||||
outcome_date, datetime.min.time(), tzinfo=timezone.utc
|
trade.close_reason = reason
|
||||||
)
|
trade.closed_at = datetime.combine(close_date, datetime.min.time(), tzinfo=timezone.utc)
|
||||||
closed += 1
|
closed += 1
|
||||||
|
|
||||||
if closed:
|
if closed:
|
||||||
await db.commit()
|
await db.commit()
|
||||||
return closed
|
return closed
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Equity curve — the paper book's cumulative P&L vs the same dollars in SPY.
|
||||||
|
|
||||||
|
|
||||||
|
def _value_on_or_before(
|
||||||
|
dates_sorted: list[date], closes: dict[date, float], target: date
|
||||||
|
) -> float | None:
|
||||||
|
"""Close on the nearest trading day at or before ``target`` (None if before history)."""
|
||||||
|
idx = bisect.bisect_right(dates_sorted, target) - 1
|
||||||
|
return closes[dates_sorted[idx]] if idx >= 0 else None
|
||||||
|
|
||||||
|
|
||||||
|
def build_equity_curve(
|
||||||
|
trades: list,
|
||||||
|
ticker_closes: dict[int, dict[date, float]],
|
||||||
|
benchmark_closes: dict[date, float],
|
||||||
|
) -> list[dict]:
|
||||||
|
"""Daily cumulative P&L of the paper book vs a benchmark counterfactual.
|
||||||
|
|
||||||
|
For every benchmark trading day since the first trade opened:
|
||||||
|
|
||||||
|
book_pnl = Σ realized P&L of trades closed by then
|
||||||
|
+ Σ mark-to-market P&L of trades still open (ticker close
|
||||||
|
on/before that day)
|
||||||
|
benchmark_pnl = Σ per trade: the SAME cost basis (entry x shares) riding
|
||||||
|
the benchmark over the SAME window (open → close/now).
|
||||||
|
Long-benchmark regardless of trade direction — the
|
||||||
|
question is "what if this money had just sat in SPY".
|
||||||
|
|
||||||
|
Pure function so the math is unit-testable; trades are duck-typed
|
||||||
|
(ticker_id, direction, entry_price, shares, status, opened_at, closed_at,
|
||||||
|
close_price). Trades opened before the stored benchmark history contribute
|
||||||
|
to book_pnl but not to benchmark_pnl (no baseline close to measure from).
|
||||||
|
"""
|
||||||
|
if not trades or not benchmark_closes:
|
||||||
|
return []
|
||||||
|
first = min(t.opened_at.date() for t in trades)
|
||||||
|
bench_dates = sorted(benchmark_closes)
|
||||||
|
days = [d for d in bench_dates if d >= first]
|
||||||
|
if not days:
|
||||||
|
return []
|
||||||
|
ticker_dates_sorted = {tid: sorted(c) for tid, c in ticker_closes.items()}
|
||||||
|
|
||||||
|
out: list[dict] = []
|
||||||
|
for d in days:
|
||||||
|
book = 0.0
|
||||||
|
bench = 0.0
|
||||||
|
any_priced = False
|
||||||
|
for t in trades:
|
||||||
|
opened = t.opened_at.date()
|
||||||
|
if opened > d:
|
||||||
|
continue
|
||||||
|
closed_on = (
|
||||||
|
t.closed_at.date()
|
||||||
|
if (t.status == "closed" and t.closed_at is not None)
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
window_end = min(d, closed_on) if closed_on is not None else d
|
||||||
|
|
||||||
|
if closed_on is not None and closed_on <= d and t.close_price is not None:
|
||||||
|
ref = float(t.close_price)
|
||||||
|
else:
|
||||||
|
closes = ticker_closes.get(t.ticker_id) or {}
|
||||||
|
ref_val = _value_on_or_before(
|
||||||
|
ticker_dates_sorted.get(t.ticker_id) or [], closes, d
|
||||||
|
)
|
||||||
|
if ref_val is None:
|
||||||
|
continue
|
||||||
|
ref = ref_val
|
||||||
|
per_share = (
|
||||||
|
ref - t.entry_price if t.direction == "long" else t.entry_price - ref
|
||||||
|
)
|
||||||
|
book += per_share * t.shares
|
||||||
|
any_priced = True
|
||||||
|
|
||||||
|
s0 = _value_on_or_before(bench_dates, benchmark_closes, opened)
|
||||||
|
s1 = _value_on_or_before(bench_dates, benchmark_closes, window_end)
|
||||||
|
if s0 and s1:
|
||||||
|
bench += (t.entry_price * t.shares) * (s1 - s0) / s0
|
||||||
|
if any_priced:
|
||||||
|
out.append(
|
||||||
|
{
|
||||||
|
"date": d.isoformat(),
|
||||||
|
"book_pnl": round(book, 2),
|
||||||
|
"benchmark_pnl": round(bench, 2),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
KEY_PERFORMANCE_START = "performance_start_date"
|
||||||
|
|
||||||
|
|
||||||
|
async def get_performance_start(db: AsyncSession) -> date | None:
|
||||||
|
"""Date the performance view starts from, or None for 'all history'.
|
||||||
|
|
||||||
|
The strategy has been revised repeatedly, so early trades were taken under
|
||||||
|
rules that no longer exist. Pinning a start date keeps the comparison inside
|
||||||
|
one regime instead of averaging across configurations that were replaced.
|
||||||
|
"""
|
||||||
|
raw = await settings_store.get_value(db, KEY_PERFORMANCE_START, "")
|
||||||
|
if not raw or not str(raw).strip():
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return date.fromisoformat(str(raw).strip())
|
||||||
|
except ValueError:
|
||||||
|
logger.warning("invalid %s: %r", KEY_PERFORMANCE_START, raw)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def trade_r_multiple(trade, mark: float | None) -> float | None:
|
||||||
|
"""Result in R — profit measured in units of the trade's own initial risk.
|
||||||
|
|
||||||
|
R is the only sizing-independent yardstick available here: the shadow book
|
||||||
|
sizes at a fixed 1% of equity while manual trades were sized by hand, so
|
||||||
|
currency P&L cannot compare them. Open trades are marked to ``mark``.
|
||||||
|
"""
|
||||||
|
risk_per_share = abs(trade.entry_price - trade.stop_loss)
|
||||||
|
if risk_per_share <= 0:
|
||||||
|
return None
|
||||||
|
exit_price = trade.close_price if trade.status == "closed" else mark
|
||||||
|
if exit_price is None:
|
||||||
|
return None
|
||||||
|
per_share = (
|
||||||
|
exit_price - trade.entry_price
|
||||||
|
if trade.direction == "long"
|
||||||
|
else trade.entry_price - exit_price
|
||||||
|
)
|
||||||
|
return per_share / risk_per_share
|
||||||
|
|
||||||
|
|
||||||
|
def book_stats(trades: list, marks: dict[int, float]) -> dict:
|
||||||
|
"""Sizing-independent summary of one book: counts, win rate, R-multiples."""
|
||||||
|
rs = [
|
||||||
|
r
|
||||||
|
for r in (trade_r_multiple(t, marks.get(t.ticker_id)) for t in trades)
|
||||||
|
if r is not None
|
||||||
|
]
|
||||||
|
closed = [t for t in trades if t.status == "closed"]
|
||||||
|
wins = [r for r in rs if r > 0]
|
||||||
|
return {
|
||||||
|
"trades": len(trades),
|
||||||
|
"closed": len(closed),
|
||||||
|
"open": len(trades) - len(closed),
|
||||||
|
"win_rate": round(100.0 * len(wins) / len(rs), 1) if rs else None,
|
||||||
|
"total_r": round(sum(rs), 2) if rs else 0.0,
|
||||||
|
"avg_r": round(sum(rs) / len(rs), 3) if rs else None,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def equity_curve(db: AsyncSession, user_id: int) -> list[dict]:
|
||||||
|
"""Equity-curve series for a user's paper book (empty without benchmark data)."""
|
||||||
|
trades = (
|
||||||
|
(await db.execute(select(PaperTrade).where(PaperTrade.user_id == user_id)))
|
||||||
|
.scalars()
|
||||||
|
.all()
|
||||||
|
)
|
||||||
|
if not trades:
|
||||||
|
return []
|
||||||
|
benchmark_closes = await benchmark_service.load_benchmark_closes(db)
|
||||||
|
if not benchmark_closes:
|
||||||
|
return []
|
||||||
|
first = min(t.opened_at.date() for t in trades)
|
||||||
|
ticker_ids = {t.ticker_id for t in trades}
|
||||||
|
rows = await db.execute(
|
||||||
|
select(OHLCVRecord.ticker_id, OHLCVRecord.date, OHLCVRecord.close).where(
|
||||||
|
OHLCVRecord.ticker_id.in_(ticker_ids),
|
||||||
|
OHLCVRecord.date >= first,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
ticker_closes: dict[int, dict[date, float]] = {}
|
||||||
|
for tid, day, close in rows.all():
|
||||||
|
ticker_closes.setdefault(tid, {})[day] = float(close)
|
||||||
|
return build_equity_curve(list(trades), ticker_closes, benchmark_closes)
|
||||||
|
|
||||||
|
|
||||||
|
def _cumulative_pnl(trades: list, ticker_closes: dict, days: list[date]) -> list[float]:
|
||||||
|
"""Cumulative realized + mark-to-market P&L of one book on each day."""
|
||||||
|
sorted_dates = {tid: sorted(c) for tid, c in ticker_closes.items()}
|
||||||
|
out: list[float] = []
|
||||||
|
for d in days:
|
||||||
|
total = 0.0
|
||||||
|
for t in trades:
|
||||||
|
if t.opened_at.date() > d:
|
||||||
|
continue
|
||||||
|
closed_on = (
|
||||||
|
t.closed_at.date()
|
||||||
|
if (t.status == "closed" and t.closed_at is not None)
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
if closed_on is not None and closed_on <= d and t.close_price is not None:
|
||||||
|
ref = float(t.close_price)
|
||||||
|
else:
|
||||||
|
ref = _value_on_or_before(
|
||||||
|
sorted_dates.get(t.ticker_id) or [],
|
||||||
|
ticker_closes.get(t.ticker_id) or {},
|
||||||
|
d,
|
||||||
|
)
|
||||||
|
if ref is None:
|
||||||
|
continue
|
||||||
|
per_share = (
|
||||||
|
ref - t.entry_price if t.direction == "long" else t.entry_price - ref
|
||||||
|
)
|
||||||
|
total += per_share * t.shares
|
||||||
|
out.append(round(total, 2))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def performance_summary(db: AsyncSession, user_id: int | None = None) -> dict:
|
||||||
|
"""Shadow book vs discretionary book vs SPY, from the configured start date.
|
||||||
|
|
||||||
|
Currency P&L is reported per book but is *not* the comparison — the books
|
||||||
|
size differently, so the honest read is the R-multiple stats. SPY is a plain
|
||||||
|
buy-and-hold reference over the same window rather than a per-trade
|
||||||
|
counterfactual, so one line serves both books.
|
||||||
|
"""
|
||||||
|
start = await get_performance_start(db)
|
||||||
|
stmt = select(PaperTrade)
|
||||||
|
if start is not None:
|
||||||
|
stmt = stmt.where(func.date(PaperTrade.opened_at) >= start)
|
||||||
|
if user_id is not None:
|
||||||
|
# "Your picks" must be *yours*. The shadow book is a single autonomous
|
||||||
|
# book with no owner, so it is never scoped to a user.
|
||||||
|
stmt = stmt.where(
|
||||||
|
(PaperTrade.book == SHADOW_BOOK) | (PaperTrade.user_id == user_id)
|
||||||
|
)
|
||||||
|
trades = list((await db.execute(stmt)).scalars().all())
|
||||||
|
|
||||||
|
benchmark_closes = await benchmark_service.load_benchmark_closes(db)
|
||||||
|
empty = {
|
||||||
|
"start_date": start.isoformat() if start else None,
|
||||||
|
"series": [],
|
||||||
|
"stats": {},
|
||||||
|
}
|
||||||
|
if not trades or not benchmark_closes:
|
||||||
|
return empty
|
||||||
|
|
||||||
|
first = min(t.opened_at.date() for t in trades)
|
||||||
|
if start is not None:
|
||||||
|
first = max(first, start)
|
||||||
|
days = [d for d in sorted(benchmark_closes) if d >= first]
|
||||||
|
if not days:
|
||||||
|
return empty
|
||||||
|
|
||||||
|
ticker_ids = {t.ticker_id for t in trades}
|
||||||
|
rows = await db.execute(
|
||||||
|
select(OHLCVRecord.ticker_id, OHLCVRecord.date, OHLCVRecord.close).where(
|
||||||
|
OHLCVRecord.ticker_id.in_(ticker_ids), OHLCVRecord.date >= first
|
||||||
|
)
|
||||||
|
)
|
||||||
|
ticker_closes: dict[int, dict[date, float]] = {}
|
||||||
|
for tid, day, close in rows.all():
|
||||||
|
ticker_closes.setdefault(tid, {})[day] = float(close)
|
||||||
|
|
||||||
|
books = {
|
||||||
|
MANUAL_BOOK: [t for t in trades if (t.book or MANUAL_BOOK) == MANUAL_BOOK],
|
||||||
|
SHADOW_BOOK: [t for t in trades if t.book == SHADOW_BOOK],
|
||||||
|
}
|
||||||
|
pnl = {
|
||||||
|
name: _cumulative_pnl(book_trades, ticker_closes, days)
|
||||||
|
for name, book_trades in books.items()
|
||||||
|
}
|
||||||
|
|
||||||
|
bench_dates = sorted(benchmark_closes)
|
||||||
|
spy0 = _value_on_or_before(bench_dates, benchmark_closes, days[0])
|
||||||
|
spy_pct = [
|
||||||
|
round(100.0 * (benchmark_closes[d] / spy0 - 1.0), 2) if spy0 else 0.0
|
||||||
|
for d in days
|
||||||
|
]
|
||||||
|
|
||||||
|
# Latest close per ticker, for marking open positions in the R stats.
|
||||||
|
marks = {
|
||||||
|
tid: closes[max(closes)] for tid, closes in ticker_closes.items() if closes
|
||||||
|
}
|
||||||
|
stats = {name: book_stats(bt, marks) for name, bt in books.items()}
|
||||||
|
for name in books:
|
||||||
|
stats[name]["pnl"] = pnl[name][-1] if pnl[name] else 0.0
|
||||||
|
stats["spy"] = {"pct": spy_pct[-1] if spy_pct else 0.0}
|
||||||
|
|
||||||
|
series = [
|
||||||
|
{
|
||||||
|
"date": d.isoformat(),
|
||||||
|
"manual_pnl": pnl[MANUAL_BOOK][i],
|
||||||
|
"shadow_pnl": pnl[SHADOW_BOOK][i],
|
||||||
|
"spy_pct": spy_pct[i],
|
||||||
|
}
|
||||||
|
for i, d in enumerate(days)
|
||||||
|
]
|
||||||
|
return {"start_date": start.isoformat() if start else None, "series": series, "stats": stats}
|
||||||
|
|||||||
@@ -0,0 +1,47 @@
|
|||||||
|
"""Per-invocation identity for pipeline runs.
|
||||||
|
|
||||||
|
A pipeline invocation stamps a unique run id into the task context. The scan it
|
||||||
|
runs records that id alongside its completion markers, and the shadow book
|
||||||
|
requires an *exact* match before acting on the scan's batch.
|
||||||
|
|
||||||
|
This is what timestamp comparison cannot provide. A manually triggered scan and
|
||||||
|
the scheduled near-close pipeline are separate APScheduler jobs, and
|
||||||
|
``max_instances=1`` only serialises a job against itself — not two different
|
||||||
|
jobs. So a manual scan can start just before the pipeline and finish just after
|
||||||
|
it began, leaving a completion timestamp later than the pipeline's start even
|
||||||
|
though its batch is unrelated. Matching on a run id generated by the pipeline,
|
||||||
|
and stamped only by the scan running inside that pipeline, removes the ambiguity.
|
||||||
|
|
||||||
|
Lives in its own module so the scheduler (which sets the id), the scanner (which
|
||||||
|
stamps it), and the shadow book (which checks it) can all import it without an
|
||||||
|
import cycle.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import contextvars
|
||||||
|
import uuid
|
||||||
|
|
||||||
|
_run_id: contextvars.ContextVar[str | None] = contextvars.ContextVar(
|
||||||
|
"pipeline_run_id", default=None
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def new_run_id() -> str:
|
||||||
|
"""A fresh, collision-free run id."""
|
||||||
|
return uuid.uuid4().hex
|
||||||
|
|
||||||
|
|
||||||
|
def current() -> str | None:
|
||||||
|
"""Run id of the pipeline invocation on the current task, if any."""
|
||||||
|
return _run_id.get()
|
||||||
|
|
||||||
|
|
||||||
|
def bind(run_id: str) -> contextvars.Token:
|
||||||
|
"""Set the current run id; pass the returned token to ``release``."""
|
||||||
|
return _run_id.set(run_id)
|
||||||
|
|
||||||
|
|
||||||
|
def release(token: contextvars.Token) -> None:
|
||||||
|
"""Restore the previous run id (call in a finally)."""
|
||||||
|
_run_id.reset(token)
|
||||||
@@ -1,15 +1,20 @@
|
|||||||
"""Price Store service: upsert and query OHLCV records."""
|
"""Price Store service: upsert and query OHLCV records."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
from datetime import date, datetime
|
from datetime import date, datetime
|
||||||
|
|
||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
from sqlalchemy.dialects.postgresql import insert as pg_insert
|
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.database import insert_for_session
|
||||||
from app.exceptions import NotFoundError, ValidationError
|
from app.exceptions import NotFoundError, ValidationError
|
||||||
from app.models.ohlcv import OHLCVRecord
|
from app.models.ohlcv import OHLCVRecord
|
||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
async def _get_ticker(db: AsyncSession, symbol: str) -> Ticker:
|
async def _get_ticker(db: AsyncSession, symbol: str) -> Ticker:
|
||||||
"""Look up a ticker by symbol. Raises NotFoundError if missing."""
|
"""Look up a ticker by symbol. Raises NotFoundError if missing."""
|
||||||
@@ -44,16 +49,27 @@ async def upsert_ohlcv(
|
|||||||
low: float,
|
low: float,
|
||||||
close: float,
|
close: float,
|
||||||
volume: int,
|
volume: int,
|
||||||
|
*,
|
||||||
|
refresh_sr: bool = True,
|
||||||
) -> OHLCVRecord:
|
) -> OHLCVRecord:
|
||||||
"""Insert or update an OHLCV record for (ticker, date).
|
"""Insert or update an OHLCV record for (ticker, date).
|
||||||
|
|
||||||
Validates business rules, resolves ticker, then uses
|
Validates business rules, resolves ticker, then uses
|
||||||
ON CONFLICT DO UPDATE on the (ticker_id, date) unique constraint.
|
ON CONFLICT DO UPDATE on the (ticker_id, date) unique constraint.
|
||||||
|
|
||||||
|
``refresh_sr`` (default True) recalculates persisted Structural S/R after
|
||||||
|
the write so chart levels stay current. Batch ingestion passes
|
||||||
|
``refresh_sr=False`` and refreshes once at the end of the ticker batch.
|
||||||
|
|
||||||
|
The OHLCV commit is authoritative: if S/R rebuild fails after a successful
|
||||||
|
price write, the error is logged, the session is rolled back to clear
|
||||||
|
poison, and the upsert still returns the persisted bar (caller can retry
|
||||||
|
S/R via the scanner/ingestion pipeline).
|
||||||
"""
|
"""
|
||||||
_validate_ohlcv(high, low, open_, close, volume, record_date)
|
_validate_ohlcv(high, low, open_, close, volume, record_date)
|
||||||
ticker = await _get_ticker(db, symbol)
|
ticker = await _get_ticker(db, symbol)
|
||||||
|
|
||||||
stmt = pg_insert(OHLCVRecord).values(
|
stmt = insert_for_session(db, OHLCVRecord).values(
|
||||||
ticker_id=ticker.id,
|
ticker_id=ticker.id,
|
||||||
date=record_date,
|
date=record_date,
|
||||||
open=open_,
|
open=open_,
|
||||||
@@ -64,7 +80,7 @@ async def upsert_ohlcv(
|
|||||||
created_at=datetime.utcnow(),
|
created_at=datetime.utcnow(),
|
||||||
)
|
)
|
||||||
stmt = stmt.on_conflict_do_update(
|
stmt = stmt.on_conflict_do_update(
|
||||||
constraint="uq_ohlcv_ticker_date",
|
index_elements=["ticker_id", "date"],
|
||||||
set_={
|
set_={
|
||||||
"open": stmt.excluded.open,
|
"open": stmt.excluded.open,
|
||||||
"high": stmt.excluded.high,
|
"high": stmt.excluded.high,
|
||||||
@@ -80,12 +96,36 @@ async def upsert_ohlcv(
|
|||||||
|
|
||||||
record = result.scalar_one()
|
record = result.scalar_one()
|
||||||
|
|
||||||
# TODO: Invalidate LRU cache entries for this ticker (Task 7.1)
|
from app.cache import indicator_cache
|
||||||
# TODO: Mark composite score as stale for this ticker (Task 10.1)
|
|
||||||
|
indicator_cache.invalidate_ticker(ticker.symbol)
|
||||||
|
|
||||||
|
if refresh_sr:
|
||||||
|
await _refresh_structural_sr_best_effort(db, ticker.symbol)
|
||||||
|
|
||||||
return record
|
return record
|
||||||
|
|
||||||
|
|
||||||
|
async def _refresh_structural_sr_best_effort(db: AsyncSession, symbol: str) -> bool:
|
||||||
|
"""Rebuild Structural S/R; never fail a successful OHLCV write.
|
||||||
|
|
||||||
|
Returns True on success. On failure rolls the session back so a later
|
||||||
|
operation on the same session is not poisoned by the failed unit of work.
|
||||||
|
"""
|
||||||
|
from app.services.sr_service import recalculate_sr_levels
|
||||||
|
|
||||||
|
try:
|
||||||
|
await recalculate_sr_levels(db, symbol)
|
||||||
|
return True
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Structural S/R refresh failed for %s after OHLCV write", symbol)
|
||||||
|
try:
|
||||||
|
await db.rollback()
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Session rollback after S/R failure also failed for %s", symbol)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
async def query_ohlcv(
|
async def query_ohlcv(
|
||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
symbol: str,
|
symbol: str,
|
||||||
|
|||||||
@@ -2,12 +2,14 @@
|
|||||||
|
|
||||||
A single predicate, driven by the admin activation config, used by the
|
A single predicate, driven by the admin activation config, used by the
|
||||||
performance stats (server) and mirrored on the frontend. The core selection is
|
performance stats (server) and mirrored on the frontend. The core selection is
|
||||||
cross-sectional momentum: a setup's ticker must rank in the top
|
residual cross-sectional momentum: a setup's ticker must rank in the top
|
||||||
``min_momentum_percentile`` of the universe by 12-1 month momentum — the one
|
``min_momentum_percentile`` of the universe by beta-adjusted 12-1 month momentum.
|
||||||
signal the backtest showed actually sorts forward returns. R:R and confidence
|
R:R and confidence remain as floors, and conviction/conflict survive as optional
|
||||||
remain as floors, and conviction/conflict survive as optional tighteners (off by
|
tighteners (off by default). Qualified setups must also have a primary target
|
||||||
default). The momentum percentile is computed across the universe and attached to
|
with at least ``MIN_TARGET_PROBABILITY`` reach probability: a primary below the
|
||||||
each setup upstream; when it's absent the gate falls back to the floors.
|
floor is a lottery target whose distance inflates R:R, so it would otherwise
|
||||||
|
game the min_rr gate (the model clamps probabilities at 3%, and far targets pin
|
||||||
|
there while their live R:R stays high forever).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -16,6 +18,23 @@ from typing import Any
|
|||||||
|
|
||||||
HIGH_CONVICTION_ACTIONS = {"LONG_HIGH", "SHORT_HIGH"}
|
HIGH_CONVICTION_ACTIONS = {"LONG_HIGH", "SHORT_HIGH"}
|
||||||
|
|
||||||
|
# Floor for the primary target's reach probability, shared with the primary
|
||||||
|
# target selection in recommendation_service and mirrored in the frontend
|
||||||
|
# (qualification.ts). Under the two-barrier model a fair-race 1.5:1 target sits
|
||||||
|
# near ~34% before drift adjustments, so 20% only excludes targets the model
|
||||||
|
# itself considers long shots.
|
||||||
|
MIN_TARGET_PROBABILITY = 20.0
|
||||||
|
|
||||||
|
|
||||||
|
def _action_direction(action: str | None) -> str:
|
||||||
|
if not action or action == "NEUTRAL":
|
||||||
|
return "neutral"
|
||||||
|
if action.startswith("LONG"):
|
||||||
|
return "long"
|
||||||
|
if action.startswith("SHORT"):
|
||||||
|
return "short"
|
||||||
|
return "neutral"
|
||||||
|
|
||||||
|
|
||||||
def best_target_probability(setup: Any) -> float:
|
def best_target_probability(setup: Any) -> float:
|
||||||
"""Highest probability among a setup's targets, 0 if none."""
|
"""Highest probability among a setup's targets, 0 if none."""
|
||||||
@@ -24,6 +43,18 @@ def best_target_probability(setup: Any) -> float:
|
|||||||
return max(probs, default=0.0)
|
return max(probs, default=0.0)
|
||||||
|
|
||||||
|
|
||||||
|
def primary_target_probability(setup: Any) -> float | None:
|
||||||
|
"""Probability of the primary/headline target, falling back to best target."""
|
||||||
|
targets = getattr(setup, "targets", None) or []
|
||||||
|
for target in targets:
|
||||||
|
if not isinstance(target, dict) or not target.get("is_primary"):
|
||||||
|
continue
|
||||||
|
probability = target.get("probability")
|
||||||
|
return float(probability) if probability is not None else None
|
||||||
|
best = best_target_probability(setup)
|
||||||
|
return best if best > 0 else None
|
||||||
|
|
||||||
|
|
||||||
def live_risk_reward(setup: Any, current_price: float) -> float | None:
|
def live_risk_reward(setup: Any, current_price: float) -> float | None:
|
||||||
"""R:R recomputed from the CURRENT price, not the (possibly stale) entry.
|
"""R:R recomputed from the CURRENT price, not the (possibly stale) entry.
|
||||||
|
|
||||||
@@ -48,10 +79,10 @@ def setup_qualifies(setup: Any, config: dict) -> bool:
|
|||||||
``setup`` is duck-typed: any object exposing rr_ratio, confidence_score,
|
``setup`` is duck-typed: any object exposing rr_ratio, confidence_score,
|
||||||
recommended_action, risk_level and a ``targets`` list of dicts.
|
recommended_action, risk_level and a ``targets`` list of dicts.
|
||||||
|
|
||||||
Gate order: R:R floor → freshness (live R:R) → confidence floor → momentum
|
Gate order: R:R floor, freshness (live R:R), target probability, confidence
|
||||||
percentile (the core selection) → optional conviction / conflict tighteners.
|
floor, momentum percentile (the core selection), then optional conviction /
|
||||||
``min_momentum_percentile`` defaults to 0 (off) for callers that pass a legacy
|
conflict tighteners. ``min_momentum_percentile`` defaults to 0 (off) for
|
||||||
config without the key.
|
callers that pass a legacy config without the key.
|
||||||
"""
|
"""
|
||||||
if setup.rr_ratio < config["min_rr"]:
|
if setup.rr_ratio < config["min_rr"]:
|
||||||
return False
|
return False
|
||||||
@@ -63,20 +94,32 @@ def setup_qualifies(setup: Any, config: dict) -> bool:
|
|||||||
live_rr = live_risk_reward(setup, float(current_price))
|
live_rr = live_risk_reward(setup, float(current_price))
|
||||||
if live_rr is not None and live_rr < config["min_rr"]:
|
if live_rr is not None and live_rr < config["min_rr"]:
|
||||||
return False
|
return False
|
||||||
|
target_probability = primary_target_probability(setup)
|
||||||
|
if target_probability is None or target_probability < MIN_TARGET_PROBABILITY:
|
||||||
|
return False
|
||||||
if (setup.confidence_score or 0.0) < config["min_confidence"]:
|
if (setup.confidence_score or 0.0) < config["min_confidence"]:
|
||||||
return False
|
return False
|
||||||
# Cross-sectional momentum: the core selection. A setup's ticker must rank in
|
# Residual cross-sectional momentum: the core selection. A setup's ticker
|
||||||
# the top ``min_momentum_percentile`` of the universe by 12-1 momentum. The
|
# must rank in the top ``min_momentum_percentile`` of the universe by
|
||||||
# validated edge is long-only, so while the gate is active shorts (which fight
|
# beta-adjusted 12-1 momentum. The validated edge is long-only, so while the
|
||||||
# the trend) never qualify. The percentile floor is only enforced when a
|
# gate is active shorts (which fight the trend) never qualify. Missing ranks
|
||||||
# percentile is attached (live setups / backtest); callers that don't attach
|
# do not qualify because the production edge depends on this cross-sectional
|
||||||
# it defer to the floors above.
|
# selection.
|
||||||
min_pct = float(config.get("min_momentum_percentile", 0.0))
|
min_pct = float(config.get("min_momentum_percentile", 0.0))
|
||||||
if min_pct > 0:
|
if min_pct > 0:
|
||||||
if (getattr(setup, "direction", "long") or "long") == "short":
|
if (getattr(setup, "direction", "long") or "long") == "short":
|
||||||
return False
|
return False
|
||||||
momentum_percentile = getattr(setup, "momentum_percentile", None)
|
momentum_percentile = getattr(setup, "momentum_percentile", None)
|
||||||
if momentum_percentile is not None and momentum_percentile < min_pct:
|
if momentum_percentile is None or momentum_percentile < min_pct:
|
||||||
|
return False
|
||||||
|
# A setup is actionable only when the live ticker action points in the same
|
||||||
|
# direction. NEUTRAL means no clear signal; an opposite action means the
|
||||||
|
# setup is counter-bias. ``exclude_neutral`` defaults on; callers that omit
|
||||||
|
# it keep legacy floor-only behavior.
|
||||||
|
if config.get("exclude_neutral"):
|
||||||
|
action_direction = _action_direction(getattr(setup, "recommended_action", None))
|
||||||
|
setup_direction = (getattr(setup, "direction", "long") or "long").lower()
|
||||||
|
if action_direction == "neutral" or action_direction != setup_direction:
|
||||||
return False
|
return False
|
||||||
if config.get("require_high_conviction"):
|
if config.get("require_high_conviction"):
|
||||||
if (setup.recommended_action or "") not in HIGH_CONVICTION_ACTIONS:
|
if (setup.recommended_action or "") not in HIGH_CONVICTION_ACTIONS:
|
||||||
|
|||||||
@@ -13,6 +13,7 @@ from app.models.settings import SystemSetting
|
|||||||
from app.models.sr_level import SRLevel
|
from app.models.sr_level import SRLevel
|
||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
from app.models.trade_setup import TradeSetup
|
from app.models.trade_setup import TradeSetup
|
||||||
|
from app.services.qualification import MIN_TARGET_PROBABILITY
|
||||||
from app.services.sr_service import cluster_sr_zones
|
from app.services.sr_service import cluster_sr_zones
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
@@ -44,12 +45,23 @@ _MODERATE_MAX_ATR = 4.6
|
|||||||
# the same tolerance the chart and alerts use, so S/R is one model app-wide.
|
# the same tolerance the chart and alerts use, so S/R is one model app-wide.
|
||||||
_SR_ZONE_TOLERANCE = 0.02
|
_SR_ZONE_TOLERANCE = 0.02
|
||||||
|
|
||||||
|
# Reach-probability estimates are clamped to this band; a target at the floor
|
||||||
|
# means "the model considers it essentially unreachable" and floor-pinned
|
||||||
|
# targets are mutually indistinguishable.
|
||||||
|
_PROBABILITY_CLAMP_LOW = 3.0
|
||||||
|
_PROBABILITY_CLAMP_HIGH = 95.0
|
||||||
|
|
||||||
|
|
||||||
def _clamp(value: float, low: float, high: float) -> float:
|
def _clamp(value: float, low: float, high: float) -> float:
|
||||||
return max(low, min(high, value))
|
return max(low, min(high, value))
|
||||||
|
|
||||||
|
|
||||||
def _zone_representative_levels(sr_levels: list[SRLevel], entry_price: float) -> list[Any]:
|
def _zone_representative_levels(
|
||||||
|
sr_levels: list[SRLevel],
|
||||||
|
entry_price: float,
|
||||||
|
*,
|
||||||
|
strength_mode: str = "sum",
|
||||||
|
) -> list[Any]:
|
||||||
"""Collapse near-duplicate S/R levels into one representative per zone.
|
"""Collapse near-duplicate S/R levels into one representative per zone.
|
||||||
|
|
||||||
Targets are generated from these representatives, so a clustered wall (e.g.
|
Targets are generated from these representatives, so a clustered wall (e.g.
|
||||||
@@ -64,11 +76,25 @@ def _zone_representative_levels(sr_levels: list[SRLevel], entry_price: float) ->
|
|||||||
if not sr_levels or entry_price <= 0:
|
if not sr_levels or entry_price <= 0:
|
||||||
return list(sr_levels)
|
return list(sr_levels)
|
||||||
|
|
||||||
level_dicts = [
|
level_dicts = []
|
||||||
{"price_level": float(lv.price_level), "strength": int(lv.strength), "type": lv.type}
|
for lv in sr_levels:
|
||||||
for lv in sr_levels
|
level_dicts.append({
|
||||||
]
|
"price_level": float(lv.price_level),
|
||||||
zones = cluster_sr_zones(level_dicts, entry_price, tolerance=_SR_ZONE_TOLERANCE)
|
"strength": int(lv.strength),
|
||||||
|
"type": lv.type,
|
||||||
|
"detection_method": getattr(lv, "detection_method", "unknown"),
|
||||||
|
"sources": list(getattr(lv, "sources", None) or [
|
||||||
|
getattr(lv, "detection_method", "unknown")
|
||||||
|
]),
|
||||||
|
"rejection_count": int(getattr(lv, "rejection_count", 0) or 0),
|
||||||
|
"last_rejection_age": getattr(lv, "last_rejection_age", None),
|
||||||
|
})
|
||||||
|
zones = cluster_sr_zones(
|
||||||
|
level_dicts,
|
||||||
|
entry_price,
|
||||||
|
tolerance=_SR_ZONE_TOLERANCE,
|
||||||
|
strength_mode=strength_mode,
|
||||||
|
)
|
||||||
|
|
||||||
reps: list[Any] = []
|
reps: list[Any] = []
|
||||||
for zone in zones:
|
for zone in zones:
|
||||||
@@ -87,6 +113,10 @@ def _zone_representative_levels(sr_levels: list[SRLevel], entry_price: float) ->
|
|||||||
price_level=float(near_edge),
|
price_level=float(near_edge),
|
||||||
type=zone["type"],
|
type=zone["type"],
|
||||||
strength=int(zone["strength"]),
|
strength=int(zone["strength"]),
|
||||||
|
detection_method=getattr(strongest, "detection_method", "unknown"),
|
||||||
|
sources=list(zone.get("sources") or []),
|
||||||
|
rejection_count=int(zone.get("rejection_count", 0)),
|
||||||
|
last_rejection_age=zone.get("last_rejection_age"),
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
return reps
|
return reps
|
||||||
@@ -305,6 +335,15 @@ class TargetGenerator:
|
|||||||
"classification": "Moderate",
|
"classification": "Moderate",
|
||||||
"sr_level_id": int(level.id),
|
"sr_level_id": int(level.id),
|
||||||
"sr_strength": float(level.strength),
|
"sr_strength": float(level.strength),
|
||||||
|
"sr_sources": list(getattr(level, "sources", None) or [
|
||||||
|
getattr(level, "detection_method", "unknown")
|
||||||
|
]),
|
||||||
|
"sr_rejection_count": int(
|
||||||
|
getattr(level, "rejection_count", 0) or 0
|
||||||
|
),
|
||||||
|
"sr_last_rejection_age": getattr(
|
||||||
|
level, "last_rejection_age", None
|
||||||
|
),
|
||||||
"quality": float(quality),
|
"quality": float(quality),
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
@@ -407,7 +446,7 @@ class ProbabilityEstimator:
|
|||||||
elif opposed:
|
elif opposed:
|
||||||
probability -= signal_weight * 100.0
|
probability -= signal_weight * 100.0
|
||||||
|
|
||||||
return round(_clamp(probability, 3.0, 95.0), 2)
|
return round(_clamp(probability, _PROBABILITY_CLAMP_LOW, _PROBABILITY_CLAMP_HIGH), 2)
|
||||||
|
|
||||||
|
|
||||||
signal_conflict_detector = SignalConflictDetector()
|
signal_conflict_detector = SignalConflictDetector()
|
||||||
@@ -524,41 +563,13 @@ def _build_reasoning(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
PRIMARY_TARGET_MIN_RR = 1.5
|
def build_recommendation_snapshot(
|
||||||
|
|
||||||
|
|
||||||
def _select_primary_target(targets: list[dict], min_rr: float = PRIMARY_TARGET_MIN_RR) -> dict | None:
|
|
||||||
"""Primary = the most LIKELY target that still offers real asymmetry.
|
|
||||||
|
|
||||||
Among targets clearing a minimal R:R floor, pick the highest probability
|
|
||||||
(tie-break by R:R). This fixes the old pick, which ignored probability and
|
|
||||||
could land on the furthest, least-likely 'lottery' level. Stronger-reward
|
|
||||||
levels remain in the table as stretch targets. Falls back to the highest-R:R
|
|
||||||
target if nothing clears the floor.
|
|
||||||
"""
|
|
||||||
if not targets:
|
|
||||||
return None
|
|
||||||
|
|
||||||
worthwhile = [t for t in targets if float(t.get("rr_ratio", 0.0)) >= min_rr]
|
|
||||||
pool = worthwhile or targets
|
|
||||||
return max(
|
|
||||||
pool,
|
|
||||||
key=lambda t: (float(t.get("probability", 0.0)), float(t.get("rr_ratio", 0.0))),
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
async def enhance_trade_setup(
|
|
||||||
db: AsyncSession,
|
|
||||||
ticker: Ticker,
|
|
||||||
setup: TradeSetup,
|
|
||||||
dimension_scores: dict[str, float],
|
dimension_scores: dict[str, float],
|
||||||
sr_levels: list[SRLevel],
|
|
||||||
sentiment_classification: str | None,
|
sentiment_classification: str | None,
|
||||||
atr_value: float,
|
config: dict[str, float],
|
||||||
available_directions: set[str] | None = None,
|
available_directions: set[str] | None = None,
|
||||||
) -> TradeSetup:
|
) -> dict[str, Any]:
|
||||||
config = await get_recommendation_config(db)
|
"""Build the ticker-level recommendation from the supplied live context."""
|
||||||
|
|
||||||
conflicts = signal_conflict_detector.detect_conflicts(
|
conflicts = signal_conflict_detector.detect_conflicts(
|
||||||
dimension_scores=dimension_scores,
|
dimension_scores=dimension_scores,
|
||||||
sentiment_classification=sentiment_classification,
|
sentiment_classification=sentiment_classification,
|
||||||
@@ -578,6 +589,118 @@ async def enhance_trade_setup(
|
|||||||
conflicts=conflicts,
|
conflicts=conflicts,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
action = _choose_recommended_action(
|
||||||
|
long_confidence, short_confidence, config, available_directions
|
||||||
|
)
|
||||||
|
reasoning = _build_reasoning(
|
||||||
|
action=action,
|
||||||
|
long_confidence=long_confidence,
|
||||||
|
short_confidence=short_confidence,
|
||||||
|
conflicts=conflicts,
|
||||||
|
dimension_scores=dimension_scores,
|
||||||
|
sentiment_classification=sentiment_classification,
|
||||||
|
config=config,
|
||||||
|
available_directions=available_directions,
|
||||||
|
)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"action": action,
|
||||||
|
"reasoning": reasoning,
|
||||||
|
"risk_level": _risk_level_from_conflicts(conflicts),
|
||||||
|
"long_confidence": long_confidence,
|
||||||
|
"short_confidence": short_confidence,
|
||||||
|
"conflicts": conflicts,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# Below this the target is a lottery ticket. Shared with the activation gate
|
||||||
|
# (qualification.MIN_TARGET_PROBABILITY) so the primary selection and the gate
|
||||||
|
# agree on what counts as a probability-backed target.
|
||||||
|
PRIMARY_TARGET_MIN_PROBABILITY = MIN_TARGET_PROBABILITY
|
||||||
|
|
||||||
|
# Primary-target selector floor (independent of the live activation min_rr).
|
||||||
|
# Live scanner and backtest setup replay must share this constant.
|
||||||
|
PRIMARY_TARGET_MIN_RR = 1.5
|
||||||
|
|
||||||
|
|
||||||
|
def _prune_floor_pinned_targets(targets: list[dict]) -> list[dict]:
|
||||||
|
"""Keep only the nearest target pinned at the probability clamp floor.
|
||||||
|
|
||||||
|
Floor-pinned targets are indistinguishable to the model (true probability
|
||||||
|
at/below the clamp), so farther ones add no information — they just fill
|
||||||
|
the table with duplicate "3%" rows whose inflated R:R invites lottery
|
||||||
|
picks. ``targets`` is distance-sorted by the generator, so the first
|
||||||
|
floor-pinned entry is the nearest (most reachable) representative.
|
||||||
|
"""
|
||||||
|
pruned: list[dict] = []
|
||||||
|
seen_floor = False
|
||||||
|
for target in targets:
|
||||||
|
if float(target.get("probability", 0.0)) <= _PROBABILITY_CLAMP_LOW:
|
||||||
|
if seen_floor:
|
||||||
|
continue
|
||||||
|
seen_floor = True
|
||||||
|
pruned.append(target)
|
||||||
|
return pruned
|
||||||
|
|
||||||
|
|
||||||
|
def _select_primary_target(
|
||||||
|
targets: list[dict],
|
||||||
|
min_rr: float,
|
||||||
|
min_probability: float = PRIMARY_TARGET_MIN_PROBABILITY,
|
||||||
|
) -> dict | None:
|
||||||
|
"""Primary = the most LIKELY target that still offers real asymmetry.
|
||||||
|
|
||||||
|
Among targets clearing BOTH floors (R:R >= min_rr and probability >=
|
||||||
|
min_probability), pick the highest probability (tie-break by R:R).
|
||||||
|
Stronger-reward levels remain in the table as stretch targets.
|
||||||
|
|
||||||
|
Degenerate case: after a run-up, every level with acceptable R:R can be a
|
||||||
|
far 'lottery' target (probability at/near the model's 3% clamp floor).
|
||||||
|
Previously the pick was restricted to the R:R pool, so such a lottery level
|
||||||
|
became the headline — its inflated R:R then sailed through the activation
|
||||||
|
gate's min_rr floor. Now we fall back to the most likely target overall:
|
||||||
|
the headline carries an honest (low) R:R and the gate rejects the setup on
|
||||||
|
real numbers instead of being gamed by an unreachable target.
|
||||||
|
"""
|
||||||
|
if not targets:
|
||||||
|
return None
|
||||||
|
|
||||||
|
worthwhile = [
|
||||||
|
t
|
||||||
|
for t in targets
|
||||||
|
if float(t.get("rr_ratio", 0.0)) >= min_rr
|
||||||
|
and float(t.get("probability", 0.0)) >= min_probability
|
||||||
|
]
|
||||||
|
pool = worthwhile or targets
|
||||||
|
return max(
|
||||||
|
pool,
|
||||||
|
key=lambda t: (float(t.get("probability", 0.0)), float(t.get("rr_ratio", 0.0))),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def enhance_trade_setup(
|
||||||
|
db: AsyncSession,
|
||||||
|
ticker: Ticker,
|
||||||
|
setup: TradeSetup,
|
||||||
|
dimension_scores: dict[str, float],
|
||||||
|
sr_levels: list[SRLevel],
|
||||||
|
sentiment_classification: str | None,
|
||||||
|
atr_value: float,
|
||||||
|
primary_min_rr: float,
|
||||||
|
available_directions: set[str] | None = None,
|
||||||
|
) -> TradeSetup:
|
||||||
|
config = await get_recommendation_config(db)
|
||||||
|
|
||||||
|
snapshot = build_recommendation_snapshot(
|
||||||
|
dimension_scores=dimension_scores,
|
||||||
|
sentiment_classification=sentiment_classification,
|
||||||
|
config=config,
|
||||||
|
available_directions=available_directions,
|
||||||
|
)
|
||||||
|
conflicts = list(snapshot["conflicts"])
|
||||||
|
long_confidence = float(snapshot["long_confidence"])
|
||||||
|
short_confidence = float(snapshot["short_confidence"])
|
||||||
|
|
||||||
direction = setup.direction.lower()
|
direction = setup.direction.lower()
|
||||||
confidence = long_confidence if direction == "long" else short_confidence
|
confidence = long_confidence if direction == "long" else short_confidence
|
||||||
|
|
||||||
@@ -604,11 +727,14 @@ async def enhance_trade_setup(
|
|||||||
# Label follows from the reach-probability: high prob = Conservative.
|
# Label follows from the reach-probability: high prob = Conservative.
|
||||||
target["classification"] = _classify_by_probability(target["probability"])
|
target["classification"] = _classify_by_probability(target["probability"])
|
||||||
|
|
||||||
|
# Collapse duplicate floor-pinned lottery targets to the nearest one.
|
||||||
|
targets = _prune_floor_pinned_targets(targets)
|
||||||
|
|
||||||
# Primary target = most-likely target with real asymmetry (see
|
# Primary target = most-likely target with real asymmetry (see
|
||||||
# _select_primary_target), not the old quality-score pick that ignored
|
# _select_primary_target), not the old quality-score pick that ignored
|
||||||
# probability. Sync the setup's headline target/rr_ratio so the chart, gate
|
# probability. Sync the setup's headline target/rr_ratio so the chart, gate
|
||||||
# and outcome eval all agree with the table's starred row.
|
# and outcome eval all agree with the table's starred row.
|
||||||
primary = _select_primary_target(targets)
|
primary = _select_primary_target(targets, min_rr=primary_min_rr)
|
||||||
if primary is not None:
|
if primary is not None:
|
||||||
for target in targets:
|
for target in targets:
|
||||||
target["is_primary"] = target is primary
|
target["is_primary"] = target is primary
|
||||||
@@ -622,19 +748,8 @@ async def enhance_trade_setup(
|
|||||||
|
|
||||||
# Action and reasoning are ticker-level: they consider both directions and
|
# Action and reasoning are ticker-level: they consider both directions and
|
||||||
# which directions are actually tradeable, and are identical on every setup.
|
# which directions are actually tradeable, and are identical on every setup.
|
||||||
action = _choose_recommended_action(
|
action = str(snapshot["action"])
|
||||||
long_confidence, short_confidence, config, available_directions
|
reasoning = str(snapshot["reasoning"])
|
||||||
)
|
|
||||||
reasoning = _build_reasoning(
|
|
||||||
action=action,
|
|
||||||
long_confidence=long_confidence,
|
|
||||||
short_confidence=short_confidence,
|
|
||||||
conflicts=conflicts,
|
|
||||||
dimension_scores=dimension_scores,
|
|
||||||
sentiment_classification=sentiment_classification,
|
|
||||||
config=config,
|
|
||||||
available_directions=available_directions,
|
|
||||||
)
|
|
||||||
|
|
||||||
setup.confidence_score = round(confidence, 2)
|
setup.confidence_score = round(confidence, 2)
|
||||||
setup.targets_json = json.dumps(targets)
|
setup.targets_json = json.dumps(targets)
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+116
-110
@@ -2,8 +2,8 @@
|
|||||||
|
|
||||||
Computes dimension scores (technical, sr_quality, sentiment, fundamental,
|
Computes dimension scores (technical, sr_quality, sentiment, fundamental,
|
||||||
momentum) each 0-100, composite score as weighted average of available
|
momentum) each 0-100, composite score as weighted average of available
|
||||||
dimensions with re-normalized weights, staleness marking/recomputation
|
dimensions with re-normalized weights, staleness marking, explicit refresh
|
||||||
on demand, and weight update triggers full recomputation.
|
paths, and weight update triggers full recomputation.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -16,6 +16,7 @@ from datetime import datetime, timezone
|
|||||||
from sqlalchemy import select
|
from sqlalchemy import select
|
||||||
from sqlalchemy.ext.asyncio import AsyncSession
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.database import insert_for_session
|
||||||
from app.exceptions import NotFoundError, ValidationError
|
from app.exceptions import NotFoundError, ValidationError
|
||||||
from app.models.score import CompositeScore, DimensionScore
|
from app.models.score import CompositeScore, DimensionScore
|
||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
@@ -28,13 +29,31 @@ DIMENSIONS = ["technical", "sr_quality", "sentiment", "fundamental", "momentum"]
|
|||||||
DEFAULT_WEIGHTS: dict[str, float] = {
|
DEFAULT_WEIGHTS: dict[str, float] = {
|
||||||
"technical": 0.25,
|
"technical": 0.25,
|
||||||
"sr_quality": 0.20,
|
"sr_quality": 0.20,
|
||||||
"sentiment": 0.15,
|
"sentiment": 0.10,
|
||||||
"fundamental": 0.20,
|
"fundamental": 0.20,
|
||||||
"momentum": 0.20,
|
"momentum": 0.20,
|
||||||
}
|
}
|
||||||
|
|
||||||
SCORING_WEIGHTS_KEY = "scoring_weights"
|
SCORING_WEIGHTS_KEY = "scoring_weights"
|
||||||
|
|
||||||
|
# Sentiment enters the composite as a signed adjustment around this neutral point,
|
||||||
|
# not as an averaged-in level (see _sentiment_adjustment / compute_composite_score).
|
||||||
|
NEUTRAL_SENTIMENT = 50.0
|
||||||
|
|
||||||
|
|
||||||
|
def _sentiment_adjustment(sentiment_score: float | None, sentiment_weight: float) -> float:
|
||||||
|
"""Signed points sentiment contributes to the base composite.
|
||||||
|
|
||||||
|
+MAX_ADJ at max-confidence bullish (score 100), 0 at neutral (50), -MAX_ADJ at
|
||||||
|
max-confidence bearish (score 0), where MAX_ADJ = sentiment weight * 100. A
|
||||||
|
50%-confidence call maps to score 50 → no effect (a coin flip carries no info),
|
||||||
|
so going from no sentiment to bullish can only ever help.
|
||||||
|
"""
|
||||||
|
if sentiment_score is None:
|
||||||
|
return 0.0
|
||||||
|
max_adj = sentiment_weight * 100.0
|
||||||
|
return max_adj * (sentiment_score - NEUTRAL_SENTIMENT) / 50.0
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Helpers
|
# Helpers
|
||||||
@@ -643,14 +662,23 @@ async def compute_dimension_score(
|
|||||||
# Can't compute — mark stale
|
# Can't compute — mark stale
|
||||||
existing.is_stale = True
|
existing.is_stale = True
|
||||||
elif score_val is not None:
|
elif score_val is not None:
|
||||||
dim = DimensionScore(
|
stmt = insert_for_session(db, DimensionScore).values(
|
||||||
ticker_id=ticker.id,
|
ticker_id=ticker.id,
|
||||||
dimension=dimension,
|
dimension=dimension,
|
||||||
score=score_val,
|
score=score_val,
|
||||||
is_stale=False,
|
is_stale=False,
|
||||||
computed_at=now,
|
computed_at=now,
|
||||||
)
|
)
|
||||||
db.add(dim)
|
await db.execute(
|
||||||
|
stmt.on_conflict_do_update(
|
||||||
|
index_elements=["ticker_id", "dimension"],
|
||||||
|
set_={
|
||||||
|
"score": stmt.excluded.score,
|
||||||
|
"is_stale": False,
|
||||||
|
"computed_at": stmt.excluded.computed_at,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
return score_val
|
return score_val
|
||||||
|
|
||||||
@@ -670,10 +698,15 @@ async def compute_composite_score(
|
|||||||
symbol: str,
|
symbol: str,
|
||||||
weights: dict[str, float] | None = None,
|
weights: dict[str, float] | None = None,
|
||||||
) -> tuple[float | None, list[str]]:
|
) -> tuple[float | None, list[str]]:
|
||||||
"""Compute composite score from available dimension scores.
|
"""Compute the composite score.
|
||||||
|
|
||||||
|
The non-sentiment dimensions form a re-normalized weighted-average *base*.
|
||||||
|
Sentiment is then applied as a signed adjustment around neutral (50), not
|
||||||
|
averaged in: neutral leaves the base unchanged, bullish adds and bearish
|
||||||
|
subtracts (scaled by confidence), so going from no sentiment to bullish can
|
||||||
|
only help. See _sentiment_adjustment.
|
||||||
|
|
||||||
Returns (composite_score, missing_dimensions).
|
Returns (composite_score, missing_dimensions).
|
||||||
Missing dimensions are excluded and weights re-normalized.
|
|
||||||
"""
|
"""
|
||||||
ticker = await _get_ticker(db, symbol)
|
ticker = await _get_ticker(db, symbol)
|
||||||
|
|
||||||
@@ -686,51 +719,53 @@ async def compute_composite_score(
|
|||||||
)
|
)
|
||||||
dim_scores = {ds.dimension: ds for ds in result.scalars().all()}
|
dim_scores = {ds.dimension: ds for ds in result.scalars().all()}
|
||||||
|
|
||||||
available: list[tuple[str, float, float]] = [] # (dim, weight, score)
|
def _live(dim: str) -> float | None:
|
||||||
missing: list[str] = []
|
|
||||||
|
|
||||||
for dim in DIMENSIONS:
|
|
||||||
w = weights.get(dim, 0.0)
|
|
||||||
if w <= 0:
|
|
||||||
continue
|
|
||||||
ds = dim_scores.get(dim)
|
ds = dim_scores.get(dim)
|
||||||
if ds is not None and not ds.is_stale and ds.score is not None:
|
if ds is not None and not ds.is_stale and ds.score is not None:
|
||||||
available.append((dim, w, ds.score))
|
return ds.score
|
||||||
else:
|
return None
|
||||||
missing.append(dim)
|
|
||||||
|
|
||||||
if not available:
|
missing = [dim for dim in DIMENSIONS if _live(dim) is None]
|
||||||
return None, missing
|
|
||||||
|
|
||||||
# Re-normalize weights
|
# Base: re-normalized weighted average of the non-sentiment dimensions.
|
||||||
total_weight = sum(w for _, w, _ in available)
|
base_available = [
|
||||||
if total_weight == 0:
|
(dim, weights.get(dim, 0.0), _live(dim))
|
||||||
return None, missing
|
for dim in DIMENSIONS
|
||||||
|
if dim != "sentiment" and weights.get(dim, 0.0) > 0 and _live(dim) is not None
|
||||||
|
]
|
||||||
|
sentiment_score = _live("sentiment")
|
||||||
|
|
||||||
composite = sum(w * s for _, w, s in available) / total_weight
|
if base_available:
|
||||||
composite = max(0.0, min(100.0, composite))
|
total_weight = sum(w for _, w, _ in base_available)
|
||||||
|
base = sum(w * s for _, w, s in base_available) / total_weight
|
||||||
|
elif sentiment_score is not None:
|
||||||
|
base = NEUTRAL_SENTIMENT # only sentiment present → neutral baseline
|
||||||
|
else:
|
||||||
|
return None, missing # nothing to score
|
||||||
|
|
||||||
|
delta = _sentiment_adjustment(sentiment_score, weights.get("sentiment", 0.0))
|
||||||
|
composite = max(0.0, min(100.0, base + delta))
|
||||||
|
|
||||||
# Persist composite score
|
# Persist composite score
|
||||||
now = datetime.now(timezone.utc)
|
now = datetime.now(timezone.utc)
|
||||||
comp_result = await db.execute(
|
stmt = insert_for_session(db, CompositeScore).values(
|
||||||
select(CompositeScore).where(CompositeScore.ticker_id == ticker.id)
|
ticker_id=ticker.id,
|
||||||
|
score=composite,
|
||||||
|
is_stale=False,
|
||||||
|
weights_json=json.dumps(weights),
|
||||||
|
computed_at=now,
|
||||||
)
|
)
|
||||||
existing = comp_result.scalar_one_or_none()
|
await db.execute(
|
||||||
|
stmt.on_conflict_do_update(
|
||||||
if existing is not None:
|
index_elements=["ticker_id"],
|
||||||
existing.score = composite
|
set_={
|
||||||
existing.is_stale = False
|
"score": stmt.excluded.score,
|
||||||
existing.weights_json = json.dumps(weights)
|
"is_stale": False,
|
||||||
existing.computed_at = now
|
"weights_json": stmt.excluded.weights_json,
|
||||||
else:
|
"computed_at": stmt.excluded.computed_at,
|
||||||
comp = CompositeScore(
|
},
|
||||||
ticker_id=ticker.id,
|
|
||||||
score=composite,
|
|
||||||
is_stale=False,
|
|
||||||
weights_json=json.dumps(weights),
|
|
||||||
computed_at=now,
|
|
||||||
)
|
)
|
||||||
db.add(comp)
|
)
|
||||||
|
|
||||||
return composite, missing
|
return composite, missing
|
||||||
|
|
||||||
@@ -739,73 +774,37 @@ async def compute_composite_score(
|
|||||||
async def get_score(
|
async def get_score(
|
||||||
db: AsyncSession, symbol: str
|
db: AsyncSession, symbol: str
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Get composite + all dimension scores for a ticker.
|
"""Read composite + dimension scores for a ticker without recomputing.
|
||||||
|
|
||||||
Recomputes stale dimensions on demand, then recomputes composite.
|
GET endpoints use this path, so it must not mutate persisted score context.
|
||||||
Returns a dict suitable for ScoreResponse, including dimension breakdowns
|
Scheduled/manual write paths are responsible for refreshing scores.
|
||||||
and composite breakdown with re-normalization info.
|
|
||||||
"""
|
"""
|
||||||
ticker = await _get_ticker(db, symbol)
|
ticker = await _get_ticker(db, symbol)
|
||||||
weights = await _get_weights(db)
|
weights = await _get_weights(db)
|
||||||
|
|
||||||
# Check for stale dimension scores and recompute them
|
|
||||||
result = await db.execute(
|
|
||||||
select(DimensionScore).where(DimensionScore.ticker_id == ticker.id)
|
|
||||||
)
|
|
||||||
dim_scores = {ds.dimension: ds for ds in result.scalars().all()}
|
|
||||||
|
|
||||||
for dim in DIMENSIONS:
|
|
||||||
ds = dim_scores.get(dim)
|
|
||||||
if ds is None or ds.is_stale:
|
|
||||||
await compute_dimension_score(db, symbol, dim)
|
|
||||||
|
|
||||||
# Check composite staleness
|
|
||||||
comp_result = await db.execute(
|
|
||||||
select(CompositeScore).where(CompositeScore.ticker_id == ticker.id)
|
|
||||||
)
|
|
||||||
comp = comp_result.scalar_one_or_none()
|
|
||||||
|
|
||||||
if comp is None or comp.is_stale:
|
|
||||||
await compute_composite_score(db, symbol, weights)
|
|
||||||
|
|
||||||
await db.commit()
|
|
||||||
|
|
||||||
# Re-fetch everything fresh
|
|
||||||
result = await db.execute(
|
result = await db.execute(
|
||||||
select(DimensionScore).where(DimensionScore.ticker_id == ticker.id)
|
select(DimensionScore).where(DimensionScore.ticker_id == ticker.id)
|
||||||
)
|
)
|
||||||
dim_scores_list = list(result.scalars().all())
|
dim_scores_list = list(result.scalars().all())
|
||||||
|
dim_scores = {ds.dimension: ds for ds in dim_scores_list}
|
||||||
|
|
||||||
comp_result = await db.execute(
|
comp_result = await db.execute(
|
||||||
select(CompositeScore).where(CompositeScore.ticker_id == ticker.id)
|
select(CompositeScore).where(CompositeScore.ticker_id == ticker.id)
|
||||||
)
|
)
|
||||||
comp = comp_result.scalar_one_or_none()
|
comp = comp_result.scalar_one_or_none()
|
||||||
|
|
||||||
# Compute breakdowns for each dimension by calling the dimension computers
|
|
||||||
breakdowns: dict[str, dict | None] = {}
|
|
||||||
for dim in DIMENSIONS:
|
|
||||||
try:
|
|
||||||
raw_result = await _DIMENSION_COMPUTERS[dim](db, symbol)
|
|
||||||
if isinstance(raw_result, tuple) and len(raw_result) == 2:
|
|
||||||
breakdowns[dim] = raw_result[1]
|
|
||||||
else:
|
|
||||||
breakdowns[dim] = None
|
|
||||||
except Exception:
|
|
||||||
breakdowns[dim] = None
|
|
||||||
|
|
||||||
# Build dimension entries with breakdowns
|
|
||||||
dimensions = []
|
dimensions = []
|
||||||
missing = []
|
missing = []
|
||||||
available_dims: list[str] = []
|
available_dims: list[str] = []
|
||||||
for dim in DIMENSIONS:
|
for dim in DIMENSIONS:
|
||||||
found = next((ds for ds in dim_scores_list if ds.dimension == dim), None)
|
found = dim_scores.get(dim)
|
||||||
if found is not None and not found.is_stale and found.score is not None:
|
if found is not None and not found.is_stale and found.score is not None:
|
||||||
dimensions.append({
|
dimensions.append({
|
||||||
"dimension": found.dimension,
|
"dimension": found.dimension,
|
||||||
"score": found.score,
|
"score": found.score,
|
||||||
"is_stale": found.is_stale,
|
"is_stale": found.is_stale,
|
||||||
"computed_at": found.computed_at,
|
"computed_at": found.computed_at,
|
||||||
"breakdown": breakdowns.get(dim),
|
"breakdown": None,
|
||||||
})
|
})
|
||||||
w = weights.get(dim, 0.0)
|
w = weights.get(dim, 0.0)
|
||||||
if w > 0:
|
if w > 0:
|
||||||
@@ -819,25 +818,50 @@ async def get_score(
|
|||||||
"score": found.score,
|
"score": found.score,
|
||||||
"is_stale": found.is_stale,
|
"is_stale": found.is_stale,
|
||||||
"computed_at": found.computed_at,
|
"computed_at": found.computed_at,
|
||||||
"breakdown": breakdowns.get(dim),
|
"breakdown": None,
|
||||||
})
|
})
|
||||||
|
|
||||||
# Build composite breakdown with re-normalization info
|
# Build composite breakdown: the non-sentiment base (re-normalized weighted
|
||||||
composite_breakdown = None
|
# average) plus sentiment as a signed adjustment around neutral.
|
||||||
available_weight_sum = sum(weights.get(d, 0.0) for d in available_dims)
|
base_dims = [d for d in available_dims if d != "sentiment"]
|
||||||
|
available_weight_sum = sum(weights.get(d, 0.0) for d in base_dims)
|
||||||
if available_weight_sum > 0:
|
if available_weight_sum > 0:
|
||||||
renormalized_weights = {
|
renormalized_weights = {
|
||||||
d: weights.get(d, 0.0) / available_weight_sum for d in available_dims
|
d: weights.get(d, 0.0) / available_weight_sum for d in base_dims
|
||||||
}
|
}
|
||||||
else:
|
else:
|
||||||
renormalized_weights = {}
|
renormalized_weights = {}
|
||||||
|
|
||||||
|
fresh = {
|
||||||
|
ds.dimension: ds.score
|
||||||
|
for ds in dim_scores_list
|
||||||
|
if not ds.is_stale and ds.score is not None
|
||||||
|
}
|
||||||
|
if renormalized_weights:
|
||||||
|
base_score = sum(renormalized_weights[d] * fresh[d] for d in base_dims)
|
||||||
|
elif "sentiment" in fresh:
|
||||||
|
base_score = NEUTRAL_SENTIMENT
|
||||||
|
else:
|
||||||
|
base_score = None
|
||||||
|
|
||||||
|
sentiment_val = fresh.get("sentiment")
|
||||||
|
sentiment_weight = weights.get("sentiment", 0.0)
|
||||||
|
sentiment_adjustment = _sentiment_adjustment(sentiment_val, sentiment_weight)
|
||||||
|
|
||||||
composite_breakdown = {
|
composite_breakdown = {
|
||||||
"weights": weights,
|
"weights": weights,
|
||||||
"available_dimensions": available_dims,
|
"available_dimensions": base_dims,
|
||||||
"missing_dimensions": missing,
|
"missing_dimensions": missing,
|
||||||
"renormalized_weights": renormalized_weights,
|
"renormalized_weights": renormalized_weights,
|
||||||
"formula": "Weighted average of available dimensions with re-normalized weights: sum(weight_i * score_i) / sum(weight_i)",
|
"base_score": base_score,
|
||||||
|
"sentiment_score": sentiment_val,
|
||||||
|
"sentiment_adjustment": sentiment_adjustment,
|
||||||
|
"max_sentiment_adjustment": sentiment_weight * 100.0,
|
||||||
|
"formula": (
|
||||||
|
"Base = re-normalized weighted average of the non-sentiment dimensions. "
|
||||||
|
"Composite = base + sentiment adjustment, where adjustment = "
|
||||||
|
"MAX_ADJ * (sentiment - 50) / 50 and MAX_ADJ = sentiment weight * 100."
|
||||||
|
),
|
||||||
}
|
}
|
||||||
|
|
||||||
return {
|
return {
|
||||||
@@ -874,31 +898,13 @@ async def get_rankings(db: AsyncSession) -> dict:
|
|||||||
dims[ds.ticker_id][ds.dimension] = ds
|
dims[ds.ticker_id][ds.dimension] = ds
|
||||||
return comps, dims
|
return comps, dims
|
||||||
|
|
||||||
# Two bulk reads instead of ~4 queries per ticker.
|
|
||||||
comps, dims_by_ticker = await _load_scores()
|
comps, dims_by_ticker = await _load_scores()
|
||||||
|
|
||||||
# Lazily recompute any stale/missing scores (kept fresh by the daily scan;
|
|
||||||
# this self-heals tickers that aged out between scans), committing once.
|
|
||||||
recomputed = False
|
|
||||||
for ticker in tickers:
|
|
||||||
comp = comps.get(ticker.id)
|
|
||||||
if comp is None or comp.is_stale:
|
|
||||||
dim_scores = dims_by_ticker.get(ticker.id, {})
|
|
||||||
for dim in DIMENSIONS:
|
|
||||||
ds = dim_scores.get(dim)
|
|
||||||
if ds is None or ds.is_stale:
|
|
||||||
await compute_dimension_score(db, ticker.symbol, dim)
|
|
||||||
await compute_composite_score(db, ticker.symbol, weights)
|
|
||||||
recomputed = True
|
|
||||||
|
|
||||||
if recomputed:
|
|
||||||
await db.commit()
|
|
||||||
comps, dims_by_ticker = await _load_scores()
|
|
||||||
|
|
||||||
rankings = [
|
rankings = [
|
||||||
{
|
{
|
||||||
"symbol": ticker.symbol,
|
"symbol": ticker.symbol,
|
||||||
"composite_score": comp.score,
|
"composite_score": comp.score,
|
||||||
|
"composite_stale": comp.is_stale,
|
||||||
"dimensions": [
|
"dimensions": [
|
||||||
{
|
{
|
||||||
"dimension": ds.dimension,
|
"dimension": ds.dimension,
|
||||||
|
|||||||
@@ -0,0 +1,398 @@
|
|||||||
|
"""Async SEC EDGAR client for the fundamentals importer (workstream A).
|
||||||
|
|
||||||
|
All access is batch (never at request time). This wraps the three SEC products
|
||||||
|
the A3 design uses — `company_tickers.json`, `submissions/`, `companyfacts/`, and
|
||||||
|
the daily filing index — behind one client that honors SEC's fair-access policy:
|
||||||
|
|
||||||
|
- an identifying ``User-Agent`` with a contact email on every request (config);
|
||||||
|
- request spacing well under the 10 req/s limit;
|
||||||
|
- exponential backoff + retry on 429;
|
||||||
|
- **403 → alert and stop** (raise ``SecForbiddenError``), never a retry-loop — a
|
||||||
|
403 means the UA or request pattern is wrong and retrying won't fix it. The one
|
||||||
|
exception is S3's ``AccessDenied`` on an ``/Archives/`` path, which is how the
|
||||||
|
bucket reports an absent file (``_is_absent_archive_key``).
|
||||||
|
|
||||||
|
Parsing lives here (index fixed-width, submissions pagination); DB writes and the
|
||||||
|
snapshot mapping live in the importer. No conditional GETs — the companyfacts
|
||||||
|
endpoint exposes no ETag/Last-Modified (verified), which is why the importer is
|
||||||
|
daily-index driven rather than polling archives.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
from datetime import date, datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from app.config import settings
|
||||||
|
from app.exceptions import ProviderError
|
||||||
|
from app.services.earnings_alignment import normalise_symbol
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
_WWW = "https://www.sec.gov"
|
||||||
|
_DATA = "https://data.sec.gov"
|
||||||
|
|
||||||
|
# Resolve CA bundle for explicit httpx verify (matches app/providers/fmp.py).
|
||||||
|
_CA = os.environ.get("SSL_CERT_FILE", "")
|
||||||
|
_CA_VERIFY: str | bool = _CA if _CA and Path(_CA).exists() else True
|
||||||
|
|
||||||
|
_FORMS_10 = frozenset({"10-K", "10-Q", "10-K/A", "10-Q/A"})
|
||||||
|
|
||||||
|
|
||||||
|
class SecError(ProviderError):
|
||||||
|
"""SEC request failed (403, exhausted 429/5xx, timeout, transport, parse)."""
|
||||||
|
|
||||||
|
|
||||||
|
class SecForbiddenError(SecError):
|
||||||
|
"""SEC returned 403 — User-Agent/pattern rejected. Alert and stop."""
|
||||||
|
|
||||||
|
|
||||||
|
class SecNotFoundError(SecError):
|
||||||
|
"""The resource does not exist (e.g. no daily index published for a day).
|
||||||
|
|
||||||
|
The *only* error a caller may treat as 'missing' — every other SecError
|
||||||
|
(fair-access rejection, exhausted retries, 5xx, timeout) must propagate so a
|
||||||
|
fetch failure is never mistaken for an empty result.
|
||||||
|
|
||||||
|
Raised for a 404, and for the one 403 that also means "absent": see
|
||||||
|
``_is_absent_archive_key``."""
|
||||||
|
|
||||||
|
|
||||||
|
def _is_absent_archive_key(url: str, resp: httpx.Response) -> bool:
|
||||||
|
"""True when a 403 means "this file does not exist", not "you are blocked".
|
||||||
|
|
||||||
|
``www.sec.gov/Archives`` is served straight out of an S3 bucket that grants
|
||||||
|
no ``s3:ListBucket``, so a missing key cannot be answered with 404 — S3
|
||||||
|
returns **403 with its ``AccessDenied`` XML** instead. SEC publishes a daily
|
||||||
|
index only for business days, so every weekend and market holiday inside an
|
||||||
|
incremental walk lands on exactly this response (verified 2026-07-30:
|
||||||
|
``form.20260725.idx``, a Saturday, 403s while the Friday and Monday files
|
||||||
|
return 200 on the same User-Agent).
|
||||||
|
|
||||||
|
A genuine fair-access rejection is distinguishable and must stay fatal: it is
|
||||||
|
SEC's WAF interstitial — ``text/html``, "Your Request Originates from an
|
||||||
|
Undeclared Automated Tool" — and it is returned for files that *do* exist,
|
||||||
|
on any path. Hence the narrow gate: the Archives prefix plus S3's own error
|
||||||
|
document. Nothing else may be downgraded to "missing"."""
|
||||||
|
try:
|
||||||
|
parsed = httpx.URL(url)
|
||||||
|
except (TypeError, ValueError): # pragma: no cover — url comes from us
|
||||||
|
return False
|
||||||
|
if parsed.host != "www.sec.gov" or not parsed.path.startswith("/Archives/"):
|
||||||
|
return False
|
||||||
|
if "xml" not in resp.headers.get("Content-Type", "").lower():
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
return "<Code>AccessDenied</Code>" in resp.text
|
||||||
|
except (UnicodeDecodeError, httpx.HTTPError): # pragma: no cover
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _looks_like_contact_email(ua: str) -> bool:
|
||||||
|
if "example.com" in ua.lower() or "set-a-real-email" in ua.lower():
|
||||||
|
return False
|
||||||
|
return re.search(r"[^@\s]+@[^@\s]+\.[^@\s]+", ua) is not None
|
||||||
|
|
||||||
|
|
||||||
|
# SEC asks callers to stay well under 10 req/s; enforce a floor on real clients.
|
||||||
|
_MIN_PROD_SPACING = 0.11
|
||||||
|
|
||||||
|
|
||||||
|
def cik10(cik: int | str) -> str:
|
||||||
|
"""Zero-pad a CIK to the 10-digit form SEC URLs use (320193 -> 0000320193)."""
|
||||||
|
return str(int(cik)).zfill(10)
|
||||||
|
|
||||||
|
|
||||||
|
class SecClient:
|
||||||
|
"""Fair-access SEC HTTP client. Use as ``async with SecClient() as c:``."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
user_agent: str | None = None,
|
||||||
|
spacing_seconds: float | None = None,
|
||||||
|
max_retries: int | None = None,
|
||||||
|
timeout: float | None = None,
|
||||||
|
transport: httpx.AsyncBaseTransport | None = None,
|
||||||
|
) -> None:
|
||||||
|
self._ua = user_agent or settings.sec_user_agent
|
||||||
|
self._spacing = (
|
||||||
|
spacing_seconds if spacing_seconds is not None else settings.sec_request_spacing_seconds
|
||||||
|
)
|
||||||
|
self._max_retries = (
|
||||||
|
max_retries if max_retries is not None else settings.sec_max_retries
|
||||||
|
)
|
||||||
|
self._timeout = timeout if timeout is not None else settings.sec_request_timeout_seconds
|
||||||
|
self._transport = transport # injectable for tests
|
||||||
|
self._client: httpx.AsyncClient | None = None
|
||||||
|
self._lock = asyncio.Lock()
|
||||||
|
self._last_request = 0.0
|
||||||
|
|
||||||
|
def _validate_fair_access(self) -> None:
|
||||||
|
"""On a real (non-mocked) client, enforce SEC fair-access preconditions
|
||||||
|
so we can't accidentally hammer SEC or get 403'd: a genuine contact-email
|
||||||
|
User-Agent and a spacing floor. Mock transports skip this (tests use 0)."""
|
||||||
|
if not _looks_like_contact_email(self._ua):
|
||||||
|
raise SecError(
|
||||||
|
"sec_user_agent must contain a real contact email (got "
|
||||||
|
f"{self._ua!r}) — SEC fair-access requires it"
|
||||||
|
)
|
||||||
|
if self._spacing < _MIN_PROD_SPACING:
|
||||||
|
raise SecError(
|
||||||
|
f"sec_request_spacing_seconds {self._spacing} is below the "
|
||||||
|
f"{_MIN_PROD_SPACING}s fair-access floor"
|
||||||
|
)
|
||||||
|
|
||||||
|
async def __aenter__(self) -> "SecClient":
|
||||||
|
if self._transport is None:
|
||||||
|
self._validate_fair_access()
|
||||||
|
self._client = httpx.AsyncClient(
|
||||||
|
headers={"User-Agent": self._ua, "Accept-Encoding": "gzip, deflate"},
|
||||||
|
timeout=self._timeout,
|
||||||
|
verify=_CA_VERIFY,
|
||||||
|
transport=self._transport,
|
||||||
|
)
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(self, *exc) -> None:
|
||||||
|
if self._client is not None:
|
||||||
|
await self._client.aclose()
|
||||||
|
self._client = None
|
||||||
|
|
||||||
|
async def _throttle(self) -> None:
|
||||||
|
async with self._lock:
|
||||||
|
now = asyncio.get_event_loop().time()
|
||||||
|
wait = self._spacing - (now - self._last_request)
|
||||||
|
if wait > 0:
|
||||||
|
await asyncio.sleep(wait)
|
||||||
|
self._last_request = asyncio.get_event_loop().time()
|
||||||
|
|
||||||
|
async def _get(self, url: str) -> httpx.Response:
|
||||||
|
assert self._client is not None, "use `async with SecClient()`"
|
||||||
|
attempt = 0
|
||||||
|
while True:
|
||||||
|
await self._throttle()
|
||||||
|
try:
|
||||||
|
resp = await self._client.get(url)
|
||||||
|
except (httpx.TimeoutException, httpx.TransportError) as exc:
|
||||||
|
attempt += 1
|
||||||
|
if attempt > self._max_retries:
|
||||||
|
raise SecError(f"SEC network error for {url}: {exc}") from exc
|
||||||
|
await asyncio.sleep(min(2.0**attempt, 30.0))
|
||||||
|
continue
|
||||||
|
|
||||||
|
code = resp.status_code
|
||||||
|
if code == 403:
|
||||||
|
if _is_absent_archive_key(url, resp):
|
||||||
|
raise SecNotFoundError(f"SEC 403/AccessDenied (absent) for {url}")
|
||||||
|
raise SecForbiddenError(
|
||||||
|
f"SEC 403 for {url} — User-Agent/pattern rejected; set a real "
|
||||||
|
"sec_user_agent contact email"
|
||||||
|
)
|
||||||
|
if code == 404:
|
||||||
|
raise SecNotFoundError(f"SEC 404 for {url}")
|
||||||
|
# 429 and 5xx are transient — retry with backoff, honoring Retry-After.
|
||||||
|
if code == 429 or 500 <= code < 600:
|
||||||
|
attempt += 1
|
||||||
|
if attempt > self._max_retries:
|
||||||
|
raise SecError(f"SEC {code} after {self._max_retries} retries: {url}")
|
||||||
|
delay = _retry_after_seconds(resp) or min(2.0**attempt, 30.0)
|
||||||
|
logger.warning("SEC %d for %s — backoff %.1fs (attempt %d)", code, url, delay, attempt)
|
||||||
|
await asyncio.sleep(delay)
|
||||||
|
continue
|
||||||
|
if code >= 400:
|
||||||
|
raise SecError(f"SEC {code} for {url}")
|
||||||
|
return resp
|
||||||
|
|
||||||
|
async def get_json(self, url: str) -> Any:
|
||||||
|
return (await self._get(url)).json()
|
||||||
|
|
||||||
|
async def get_text(self, url: str) -> str:
|
||||||
|
return (await self._get(url)).text
|
||||||
|
|
||||||
|
# -- domain fetchers ---------------------------------------------------
|
||||||
|
|
||||||
|
async def company_tickers(self) -> dict[str, int]:
|
||||||
|
"""Map normalised ticker -> CIK (int). Multi-class tickers share a CIK."""
|
||||||
|
data = await self.get_json(f"{_WWW}/files/company_tickers.json")
|
||||||
|
out: dict[str, int] = {}
|
||||||
|
for row in data.values():
|
||||||
|
sym = normalise_symbol(row.get("ticker"))
|
||||||
|
if sym:
|
||||||
|
out[sym] = int(row["cik_str"])
|
||||||
|
return out
|
||||||
|
|
||||||
|
async def submissions(self, cik: int | str, *, include_history: bool = False) -> dict[str, Any]:
|
||||||
|
"""Issuer metadata + filing list.
|
||||||
|
|
||||||
|
``filings.recent`` caps at 1000; older accessions live in
|
||||||
|
``filings.files[]`` shards. Only ``include_history=True`` (the one-time
|
||||||
|
full backfill) fetches those shards — SIC refresh and incremental runs
|
||||||
|
use the recent list alone and make no extra requests.
|
||||||
|
"""
|
||||||
|
base = await self.get_json(f"{_DATA}/submissions/CIK{cik10(cik)}.json")
|
||||||
|
filings = _rows_from_arrays(base["filings"]["recent"])
|
||||||
|
if include_history:
|
||||||
|
for shard in base["filings"].get("files") or []:
|
||||||
|
shard_data = await self.get_json(f"{_DATA}/submissions/{shard['name']}")
|
||||||
|
filings.extend(_rows_from_arrays(shard_data))
|
||||||
|
return {
|
||||||
|
"cik": int(base["cik"]),
|
||||||
|
"name": base.get("name"),
|
||||||
|
"sic": base.get("sic"),
|
||||||
|
"sic_description": base.get("sicDescription"),
|
||||||
|
"fiscal_year_end": base.get("fiscalYearEnd"),
|
||||||
|
"tickers": base.get("tickers") or [],
|
||||||
|
"filings": filings,
|
||||||
|
}
|
||||||
|
|
||||||
|
async def companyfacts(self, cik: int | str) -> dict[str, Any]:
|
||||||
|
"""Raw companyfacts JSON ({cik, entityName, facts})."""
|
||||||
|
return await self.get_json(f"{_DATA}/api/xbrl/companyfacts/CIK{cik10(cik)}.json")
|
||||||
|
|
||||||
|
async def latest_index_date(self, today: date | None = None) -> date | None:
|
||||||
|
"""The most recent published daily-index date (drives the revision). Checks
|
||||||
|
the current quarter, falling back to the previous one at a quarter boundary."""
|
||||||
|
today = today or date.today()
|
||||||
|
for year, qtr in _quarters_back(today, 2):
|
||||||
|
url = f"{_WWW}/Archives/edgar/daily-index/{year}/QTR{qtr}/index.json"
|
||||||
|
try:
|
||||||
|
idx = await self.get_json(url)
|
||||||
|
except SecNotFoundError:
|
||||||
|
continue # quarter dir absent — only 404 is "missing"
|
||||||
|
dates = [
|
||||||
|
d
|
||||||
|
for item in idx.get("directory", {}).get("item", [])
|
||||||
|
if (d := _index_file_date(item.get("name", ""))) is not None
|
||||||
|
and d <= today
|
||||||
|
]
|
||||||
|
if dates:
|
||||||
|
return max(dates)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def daily_index(self, day: date) -> list[dict[str, Any]]:
|
||||||
|
"""Parse the daily form index into 10-K/10-Q(/A) rows for all issuers.
|
||||||
|
|
||||||
|
Returns [{form, cik, accession, company}]. The caller filters to the
|
||||||
|
tracked universe. A missing index (weekend/holiday/not-yet-published)
|
||||||
|
returns [] rather than raising.
|
||||||
|
"""
|
||||||
|
qtr = (day.month - 1) // 3 + 1
|
||||||
|
url = f"{_WWW}/Archives/edgar/daily-index/{day.year}/QTR{qtr}/form.{day:%Y%m%d}.idx"
|
||||||
|
try:
|
||||||
|
text = await self.get_text(url)
|
||||||
|
except SecNotFoundError:
|
||||||
|
# Absent on a weekend is routine (SEC publishes business days only); on a
|
||||||
|
# weekday it is either a market holiday or something worth a look — a SEC
|
||||||
|
# hiccup, or a rejection page misread as absent, would otherwise let the
|
||||||
|
# importer advance past real filings silently. Log-level only, no alert:
|
||||||
|
# cheaper than carrying a holiday calendar just to stay quiet ~10 days/yr.
|
||||||
|
logger.log(
|
||||||
|
logging.INFO if day.weekday() >= 5 else logging.WARNING,
|
||||||
|
"no daily index published for %s (%s)",
|
||||||
|
day,
|
||||||
|
f"{day:%a}",
|
||||||
|
)
|
||||||
|
return [] # weekend/holiday/not-yet-published; other errors propagate
|
||||||
|
return _parse_form_index(text)
|
||||||
|
|
||||||
|
|
||||||
|
def _rows_from_arrays(arrays: dict[str, list]) -> list[dict[str, Any]]:
|
||||||
|
"""Turn SEC's parallel-array filing block into row dicts (keeping only 10-K/10-Q
|
||||||
|
family filings — the ones that carry XBRL fundamentals)."""
|
||||||
|
forms = arrays.get("form", [])
|
||||||
|
out: list[dict[str, Any]] = []
|
||||||
|
for i, form in enumerate(forms):
|
||||||
|
if form not in _FORMS_10:
|
||||||
|
continue
|
||||||
|
out.append(
|
||||||
|
{
|
||||||
|
"accession": arrays["accessionNumber"][i],
|
||||||
|
"form": form,
|
||||||
|
"report_date": arrays["reportDate"][i] or None,
|
||||||
|
"filing_date": arrays["filingDate"][i] or None,
|
||||||
|
"acceptance_datetime": arrays["acceptanceDateTime"][i] or None,
|
||||||
|
"is_xbrl": bool(arrays.get("isXBRL", [0] * len(forms))[i]),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_form_index(text: str) -> list[dict[str, Any]]:
|
||||||
|
"""Parse a daily ``form.YYYYMMDD.idx`` (fixed columns: Form / Company / CIK /
|
||||||
|
Date Filed / File Name-with-accession)."""
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
started = False
|
||||||
|
for line in text.splitlines():
|
||||||
|
if not started:
|
||||||
|
if set(line.strip()) == {"-"}: # the dashed separator row
|
||||||
|
started = True
|
||||||
|
continue
|
||||||
|
parts = line.split()
|
||||||
|
if len(parts) < 5:
|
||||||
|
continue
|
||||||
|
form = parts[0]
|
||||||
|
if form not in _FORMS_10:
|
||||||
|
continue
|
||||||
|
path = parts[-1] # edgar/data/<cik>/<accession>.txt
|
||||||
|
cik = _cik_from_path(path)
|
||||||
|
accession = _accession_from_path(path)
|
||||||
|
if cik is None or accession is None:
|
||||||
|
continue
|
||||||
|
rows.append({"form": form, "cik": cik, "accession": accession, "path": path})
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
|
def _retry_after_seconds(resp: httpx.Response) -> float | None:
|
||||||
|
"""Parse a numeric-seconds Retry-After header (SEC uses seconds), capped."""
|
||||||
|
raw = resp.headers.get("Retry-After")
|
||||||
|
if not raw:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return min(float(raw), 60.0)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _index_file_date(name: str) -> date | None:
|
||||||
|
if name.startswith("form.") and name.endswith(".idx"):
|
||||||
|
try:
|
||||||
|
return datetime.strptime(name[5:13], "%Y%m%d").date()
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _quarters_back(today: date, n: int) -> list[tuple[int, int]]:
|
||||||
|
"""(year, quarter) for `today`'s quarter and the previous n-1, newest first."""
|
||||||
|
q = (today.month - 1) // 3 + 1
|
||||||
|
out = []
|
||||||
|
y = today.year
|
||||||
|
for _ in range(n):
|
||||||
|
out.append((y, q))
|
||||||
|
q -= 1
|
||||||
|
if q == 0:
|
||||||
|
q = 4
|
||||||
|
y -= 1
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _cik_from_path(path: str) -> int | None:
|
||||||
|
segs = path.split("/")
|
||||||
|
if len(segs) >= 3 and segs[2].isdigit():
|
||||||
|
return int(segs[2])
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _accession_from_path(path: str) -> str | None:
|
||||||
|
stem = path.rsplit("/", 1)[-1]
|
||||||
|
if stem.endswith(".txt"):
|
||||||
|
stem = stem[:-4]
|
||||||
|
return stem or None
|
||||||
@@ -0,0 +1,557 @@
|
|||||||
|
"""Pure parser: SEC companyfacts -> fundamental_snapshots rows.
|
||||||
|
|
||||||
|
Turns one issuer's `companyfacts` JSON (+ its submissions filing metadata) into
|
||||||
|
per-accession snapshot rows for the filing's **primary period**, following the
|
||||||
|
A3 design (docs/dolt-sec-a3-design.md). No I/O, no DB — unit-testable against a
|
||||||
|
fixture and verifiable against a real companyfacts pull.
|
||||||
|
|
||||||
|
The load-bearing rules (design Decision 2 + review):
|
||||||
|
- Period identity comes from `end == submissions.reportDate`, never `fy/fp`
|
||||||
|
(fy/fp is the *filing's* context; comparatives inside a filing repeat it).
|
||||||
|
This applies to the stored `fiscal_year`/`fiscal_period` too: they are derived
|
||||||
|
from `reportDate` against the issuer's `fiscalYearEnd` (see `_period_identity`),
|
||||||
|
because SEC's fy/fp collide and invert often enough to break the quarter chain.
|
||||||
|
- Duration facts are stored as **cumulative YTD**: pick the fact whose span
|
||||||
|
matches the fiscal-period-to-date length (Q1≈3mo … FY≈12mo) within tolerance.
|
||||||
|
If no YTD-length fact exists, store null — never a discrete masquerading as YTD.
|
||||||
|
- Balance-sheet instants are taken at `end == reportDate`. `shares_outstanding`
|
||||||
|
is a single consolidated value: the cover-page `dei` fact (its own cover-date
|
||||||
|
`end` stored separately) if present, else `us-gaap:CommonStockSharesOutstanding`
|
||||||
|
at period end (e.g. Alphabet has no `dei` fact) — never a class sum or the
|
||||||
|
weighted-average/diluted count. Multi-class issuers report it per class, which
|
||||||
|
is dimensional and therefore absent from companyfacts entirely, so
|
||||||
|
`weighted_avg_diluted_shares` is stored alongside as an explicit fallback for
|
||||||
|
market cap — a separate column, never backfilled into `shares_outstanding`.
|
||||||
|
- Cash and debt composites are aggregate-first and mutually exclusive (each
|
||||||
|
source tag counted at most once).
|
||||||
|
|
||||||
|
`parse_snapshots` separates `skipped_filings` (no usable row produced) from
|
||||||
|
`field_issues` (a row was produced but a field is null/ambiguous) — callers must
|
||||||
|
not treat field issues as missing coverage.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import math
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime, timedelta
|
||||||
|
from typing import Any, NamedTuple
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Expected YTD span (days) per fiscal period; a duration fact must land within
|
||||||
|
# tolerance of this to count as the period's cumulative value.
|
||||||
|
_EXPECTED_YTD_DAYS = {"Q1": 91, "Q2": 182, "Q3": 273, "FY": 365}
|
||||||
|
# Period identity (see _period_identity): how far a quarter end sits before its
|
||||||
|
# fiscal-year end, and how far a fiscal-year end may drift from the nominal MMDD.
|
||||||
|
# The quarter bands are 91 days apart, so ±35 stays unambiguous even for a 4-4-5
|
||||||
|
# filer whose 16-week Q4 puts Q3 112 days out.
|
||||||
|
_QUARTER_DAYS_TO_FY_END = {"Q1": 273, "Q2": 182, "Q3": 91}
|
||||||
|
_QUARTER_TOLERANCE_DAYS = 35
|
||||||
|
_FYE_DRIFT_TOLERANCE_DAYS = 21
|
||||||
|
# Covers 52/53-week calendars *and* 4-4-5 retail ones (12/12/12/16 weeks), whose
|
||||||
|
# YTD-Q3 is 36 weeks = 251-252 days and missed a 20-day tolerance by ~2 -- so
|
||||||
|
# COST/PEP lost Q3 every year, breaking the quarter chain and nulling TTM + YoY.
|
||||||
|
# Q1 84d, Q2 168d and FY 364d were always inside. Adjacent periods stay
|
||||||
|
# unambiguous at 25 (66-116, 157-207, 248-298, 340-390).
|
||||||
|
_YTD_TOLERANCE_DAYS = 25
|
||||||
|
|
||||||
|
# us-gaap duration concepts (money), priority order; first present wins.
|
||||||
|
_DURATION_USD = {
|
||||||
|
# Order is load-bearing (first present wins) and the tail entries are
|
||||||
|
# deliberately *appended*: every issuer that already resolved keeps the same
|
||||||
|
# concept, and only issuers that resolved to nothing gain a value.
|
||||||
|
# - IncludingAssessedTax: REITs/consumer filers that tag only this variant
|
||||||
|
# (e.g. ARE, KHC) reported no revenue at all.
|
||||||
|
# - RevenuesNetOfInterestExpense: the banks' total-revenue tag. JPM/GS/WFC
|
||||||
|
# tag it in every 10-Q and `Revenues` only (if at all) in the 10-K.
|
||||||
|
"revenue": [
|
||||||
|
"RevenueFromContractWithCustomerExcludingAssessedTax",
|
||||||
|
"Revenues",
|
||||||
|
"SalesRevenueNet",
|
||||||
|
"RevenueFromContractWithCustomerIncludingAssessedTax",
|
||||||
|
"RevenuesNetOfInterestExpense",
|
||||||
|
],
|
||||||
|
"net_income": ["NetIncomeLoss"],
|
||||||
|
"operating_income": ["OperatingIncomeLoss"],
|
||||||
|
"cfo": [
|
||||||
|
"NetCashProvidedByUsedInOperatingActivities",
|
||||||
|
"NetCashProvidedByUsedInOperatingActivitiesContinuingOperations",
|
||||||
|
],
|
||||||
|
"capex": [
|
||||||
|
"PaymentsToAcquirePropertyPlantAndEquipment",
|
||||||
|
"PaymentsToAcquireProductiveAssets",
|
||||||
|
],
|
||||||
|
"depreciation_amortization": [
|
||||||
|
"DepreciationDepletionAndAmortization",
|
||||||
|
"DepreciationAmortizationAndAccretionNet",
|
||||||
|
"DepreciationAndAmortization",
|
||||||
|
],
|
||||||
|
}
|
||||||
|
# unit USD/shares. Appended (not reordered) so any issuer that already resolved
|
||||||
|
# keeps the same concept. REG tags only the continuing-operations variant on every
|
||||||
|
# filing; FCX switches by form type -- EarningsPerShareDiluted in its 10-Qs, the
|
||||||
|
# continuing-ops tag in its 10-K -- which nulled the FY row and killed Q4 + TTM.
|
||||||
|
# The basic variants are a last resort for a period that tags no diluted EPS at
|
||||||
|
# all (PPL's 2026 Q1). Basic ignores option/convert dilution so it slightly
|
||||||
|
# overstates EPS (~1.2% for PPL), but only fires when diluted is entirely absent,
|
||||||
|
# and high-dilution names always tag diluted -- so it never displaces a real one.
|
||||||
|
_EPS_CONCEPTS = [
|
||||||
|
"EarningsPerShareDiluted",
|
||||||
|
"IncomeLossFromContinuingOperationsPerDilutedShare",
|
||||||
|
"EarningsPerShareBasic",
|
||||||
|
"IncomeLossFromContinuingOperationsPerBasicShare",
|
||||||
|
]
|
||||||
|
# Weighted-average diluted share count (unit "shares"), the market-cap fallback
|
||||||
|
# for multi-class issuers whose cover-page count is dimensional and therefore
|
||||||
|
# absent from companyfacts. Always present, since EPS is computed from it.
|
||||||
|
_WEIGHTED_AVG_SHARE_CONCEPTS = [
|
||||||
|
"WeightedAverageNumberOfDilutedSharesOutstanding",
|
||||||
|
"WeightedAverageNumberOfSharesOutstandingBasicAndDiluted",
|
||||||
|
]
|
||||||
|
# us-gaap instant (balance-sheet) concepts, at end == reportDate.
|
||||||
|
_CASH = ["CashAndCashEquivalentsAtCarryingValue"]
|
||||||
|
_ST_INVESTMENTS = ["ShortTermInvestments", "MarketableSecuritiesCurrent"] # pick one
|
||||||
|
_LONG_TERM_DEBT_AGG = ["LongTermDebt"]
|
||||||
|
_LONG_TERM_DEBT_PARTS = ["LongTermDebtNoncurrent", "LongTermDebtCurrent"]
|
||||||
|
_SHORT_TERM_DEBT = ["ShortTermBorrowings", "CommercialPaper"] # pick one
|
||||||
|
|
||||||
|
|
||||||
|
class Fact(NamedTuple):
|
||||||
|
taxonomy: str
|
||||||
|
concept: str
|
||||||
|
unit: str
|
||||||
|
start: date | None # None => instant
|
||||||
|
end: date
|
||||||
|
val: float
|
||||||
|
fy: int | None
|
||||||
|
fp: str | None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class SnapshotRow:
|
||||||
|
cik: str
|
||||||
|
accession: str
|
||||||
|
form: str
|
||||||
|
filed_date: date
|
||||||
|
accepted_at: datetime
|
||||||
|
period_end: date
|
||||||
|
fiscal_year: int
|
||||||
|
fiscal_period: str
|
||||||
|
period_start: date | None = None
|
||||||
|
revenue: float | None = None
|
||||||
|
net_income: float | None = None
|
||||||
|
operating_income: float | None = None
|
||||||
|
diluted_eps: float | None = None
|
||||||
|
cfo: float | None = None
|
||||||
|
capex: float | None = None
|
||||||
|
depreciation_amortization: float | None = None
|
||||||
|
cash_and_st_investments: float | None = None
|
||||||
|
total_debt: float | None = None
|
||||||
|
shares_outstanding: float | None = None
|
||||||
|
shares_outstanding_date: date | None = None
|
||||||
|
weighted_avg_diluted_shares: float | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class FilingMeta:
|
||||||
|
report_date: date
|
||||||
|
filing_date: date
|
||||||
|
accepted_at: datetime
|
||||||
|
form: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ParseResult:
|
||||||
|
rows: list[SnapshotRow] = field(default_factory=list)
|
||||||
|
# accessions for which NO row was produced (no facts / no usable period).
|
||||||
|
skipped_filings: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
# accessions with a row but a field-level warning (e.g. ambiguous shares).
|
||||||
|
field_issues: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_snapshots(
|
||||||
|
companyfacts: dict[str, Any],
|
||||||
|
filings: dict[str, FilingMeta],
|
||||||
|
accessions: set[str],
|
||||||
|
fiscal_year_end: str | None = None,
|
||||||
|
) -> ParseResult:
|
||||||
|
"""Build snapshot rows for ``accessions`` (those with facts + filing meta).
|
||||||
|
|
||||||
|
``fiscal_year_end`` is the issuer's declared ``submissions.fiscalYearEnd``
|
||||||
|
(MMDD) and seeds period identity (see ``_period_identity``), making it
|
||||||
|
independent of SEC's unreliable fy/fp fields. It is only a hint: the issuer's
|
||||||
|
own 10-K period ends override it (see ``resolve_fiscal_year_end``). With
|
||||||
|
neither available, the old fy/fp behaviour is used.
|
||||||
|
|
||||||
|
``skipped_filings`` = no row produced (missing facts/meta or no usable period
|
||||||
|
identity); ``field_issues`` = a row was produced but a field is null/ambiguous.
|
||||||
|
Callers must not use field issues as failed-row coverage.
|
||||||
|
"""
|
||||||
|
cik = f"{int(companyfacts['cik']):010d}"
|
||||||
|
# The declared value is only a hint; the issuer's own 10-Ks are authoritative.
|
||||||
|
fiscal_year_end = resolve_fiscal_year_end(filings, fiscal_year_end)
|
||||||
|
by_accn = _index_by_accession(companyfacts)
|
||||||
|
result = ParseResult()
|
||||||
|
for accn in accessions:
|
||||||
|
meta = filings.get(accn)
|
||||||
|
facts = by_accn.get(accn)
|
||||||
|
if meta is None or not facts:
|
||||||
|
result.skipped_filings.append({"accession": accn, "reason": "no facts or filing metadata"})
|
||||||
|
continue
|
||||||
|
row, note = _parse_one(cik, accn, facts, meta, fiscal_year_end)
|
||||||
|
if row is None:
|
||||||
|
result.skipped_filings.append({"accession": accn, "reason": note or "unparseable"})
|
||||||
|
continue
|
||||||
|
result.rows.append(row)
|
||||||
|
if note:
|
||||||
|
result.field_issues.append({"accession": accn, "reason": note})
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def companyfacts_accessions(companyfacts: dict[str, Any]) -> set[str]:
|
||||||
|
"""Every accession that appears anywhere in a companyfacts payload — used by
|
||||||
|
the importer's index↔Company-Facts consistency gate."""
|
||||||
|
return set(_index_by_accession(companyfacts).keys())
|
||||||
|
|
||||||
|
|
||||||
|
def _index_by_accession(companyfacts: dict[str, Any]) -> dict[str, list[Fact]]:
|
||||||
|
"""One pass over companyfacts -> {accession: [Fact, ...]}."""
|
||||||
|
out: dict[str, list[Fact]] = {}
|
||||||
|
for taxonomy, concepts in companyfacts.get("facts", {}).items():
|
||||||
|
for concept, body in concepts.items():
|
||||||
|
for unit, facts in body.get("units", {}).items():
|
||||||
|
for f in facts:
|
||||||
|
accn = f.get("accn")
|
||||||
|
end = _d(f.get("end"))
|
||||||
|
val = f.get("val")
|
||||||
|
# Skip malformed facts so they can't be selected accidentally:
|
||||||
|
# every usable fact needs an accession, an end date, and a
|
||||||
|
# finite numeric value.
|
||||||
|
if not accn or end is None or not _finite(val):
|
||||||
|
continue
|
||||||
|
out.setdefault(accn, []).append(
|
||||||
|
Fact(
|
||||||
|
taxonomy=taxonomy,
|
||||||
|
concept=concept,
|
||||||
|
unit=unit,
|
||||||
|
start=_d(f.get("start")),
|
||||||
|
end=end,
|
||||||
|
val=val,
|
||||||
|
fy=f.get("fy"),
|
||||||
|
fp=f.get("fp"),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_one(
|
||||||
|
cik: str, accn: str, facts: list[Fact], meta: FilingMeta,
|
||||||
|
fiscal_year_end: str | None = None,
|
||||||
|
) -> tuple[SnapshotRow | None, str | None]:
|
||||||
|
"""Returns (row, note). row is None when there's no usable period identity;
|
||||||
|
note is a validation reason (row-skip reason when row is None, else a
|
||||||
|
field-level issue such as ambiguous shares)."""
|
||||||
|
fy, fp = _period_identity(meta, fiscal_year_end)
|
||||||
|
if fy is None or fp is None:
|
||||||
|
# No fiscal calendar, or a period the calendar cannot place (a transition
|
||||||
|
# period). Fall back to the filing's own context: an imperfect label still
|
||||||
|
# beats dropping the filing entirely.
|
||||||
|
fy, fp = _fiscal_context(facts, meta.report_date)
|
||||||
|
if fy is None or fp not in _EXPECTED_YTD_DAYS:
|
||||||
|
return None, "no usable period identity"
|
||||||
|
|
||||||
|
row = SnapshotRow(
|
||||||
|
cik=cik,
|
||||||
|
accession=accn,
|
||||||
|
form=meta.form,
|
||||||
|
filed_date=meta.filing_date,
|
||||||
|
accepted_at=meta.accepted_at,
|
||||||
|
period_end=meta.report_date,
|
||||||
|
fiscal_year=fy,
|
||||||
|
fiscal_period=fp,
|
||||||
|
)
|
||||||
|
|
||||||
|
# duration YTD facts (money) + EPS
|
||||||
|
for field_name, concepts in _DURATION_USD.items():
|
||||||
|
val, start = _select_ytd(facts, concepts, meta.report_date, fp, "USD")
|
||||||
|
setattr(row, field_name, val)
|
||||||
|
if field_name == "revenue" and start is not None:
|
||||||
|
row.period_start = start
|
||||||
|
eps, eps_start = _select_ytd(facts, _EPS_CONCEPTS, meta.report_date, fp, "USD/shares")
|
||||||
|
row.diluted_eps = eps
|
||||||
|
if row.period_start is None and eps_start is not None:
|
||||||
|
row.period_start = eps_start
|
||||||
|
|
||||||
|
# balance-sheet instants at reportDate
|
||||||
|
row.cash_and_st_investments = _compose_cash(facts, meta.report_date)
|
||||||
|
row.total_debt = _compose_debt(facts, meta.report_date)
|
||||||
|
shares, shares_date, ambiguous = _select_shares(facts, meta.report_date)
|
||||||
|
row.shares_outstanding = shares
|
||||||
|
row.shares_outstanding_date = shares_date
|
||||||
|
row.weighted_avg_diluted_shares = _select_weighted_avg_shares(facts, meta.report_date)
|
||||||
|
|
||||||
|
return row, ("ambiguous shares outstanding" if ambiguous else None)
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_fiscal_year_end(
|
||||||
|
filings: dict[str, FilingMeta], declared: str | None
|
||||||
|
) -> str | None:
|
||||||
|
"""The issuer's fiscal-year-end MMDD, preferring its own 10-K period ends.
|
||||||
|
|
||||||
|
``submissions.fiscalYearEnd`` is *not* reliable: Franklin Resources (BEN)
|
||||||
|
declares 1231 while every one of its 10-Ks ends 09-30. Trusting it put BEN's
|
||||||
|
fiscal Q1 (Dec) 0 days from the claimed year end — matching no quarter band —
|
||||||
|
and labelled its fiscal Q2 (Mar) as Q1, colliding two periods on one key and
|
||||||
|
destroying the quarter chain.
|
||||||
|
|
||||||
|
A 10-K's reportDate **is** the fiscal year end by definition, so it wins
|
||||||
|
whenever one is available; the declared value is only a fallback for an issuer
|
||||||
|
with no annual filing in the set. The most recent 10-K is used, so an issuer
|
||||||
|
that changed its year end is measured against its current calendar.
|
||||||
|
"""
|
||||||
|
annual = [m.report_date for m in filings.values() if m.form.startswith("10-K")]
|
||||||
|
if annual:
|
||||||
|
latest = max(annual)
|
||||||
|
return f"{latest.month:02d}{latest.day:02d}"
|
||||||
|
return declared
|
||||||
|
|
||||||
|
|
||||||
|
def _period_identity(
|
||||||
|
meta: FilingMeta, fiscal_year_end: str | None
|
||||||
|
) -> tuple[int | None, str | None]:
|
||||||
|
"""(fiscal_year, fiscal_period) from the period end and the issuer's fiscal
|
||||||
|
calendar — never from the fy/fp fields.
|
||||||
|
|
||||||
|
SEC's fy/fp describe the *filing*, and they are unreliable as period identity:
|
||||||
|
observed in production, a 10-Q labelled ``FY`` (BXP), a year ending 2025-12-31
|
||||||
|
labelled 2024 (FRT, a December filer), a year ending 2025-06-27 labelled 2027
|
||||||
|
(STX), and four different period ends all labelled 2022 Q3 (PPL). Because
|
||||||
|
readers key on (fiscal_year, fiscal_period), colliding labels silently discard
|
||||||
|
a period and inverted ones scramble the quarter chain — nulling TTM and YoY.
|
||||||
|
|
||||||
|
``period_end`` is authoritative, so identity is derived from it: the form
|
||||||
|
decides FY vs quarter, and distance to the fiscal-year end decides which
|
||||||
|
quarter. Labels need not match the issuer's own naming — a filer whose year
|
||||||
|
ends in early January (DPZ) shifts by one — they need to be unique, monotonic
|
||||||
|
and YoY-aligned, which is all the derivation asks of them. Nothing outside the
|
||||||
|
derivation reads these columns.
|
||||||
|
|
||||||
|
Known limitation: ``fiscalYearEnd`` is the issuer's *current* calendar, so a
|
||||||
|
company that has changed its fiscal year end gets its historical periods
|
||||||
|
measured against the new one. The quarter tolerance shunts most of those to
|
||||||
|
the fy/fp fallback, and a same-key collision resolves newest-wins, so the
|
||||||
|
failure mode is a degraded old year rather than a scrambled current one.
|
||||||
|
"""
|
||||||
|
fy = _fiscal_year_of(meta.report_date, fiscal_year_end)
|
||||||
|
if fy is None:
|
||||||
|
return None, None
|
||||||
|
if meta.form.startswith("10-K"):
|
||||||
|
return fy, "FY"
|
||||||
|
nominal_end = _nominal_fy_end(fy, fiscal_year_end)
|
||||||
|
if nominal_end is None:
|
||||||
|
return None, None
|
||||||
|
remaining = (nominal_end - meta.report_date).days
|
||||||
|
best = min(
|
||||||
|
_QUARTER_DAYS_TO_FY_END,
|
||||||
|
key=lambda k: abs(_QUARTER_DAYS_TO_FY_END[k] - remaining),
|
||||||
|
)
|
||||||
|
if abs(_QUARTER_DAYS_TO_FY_END[best] - remaining) > _QUARTER_TOLERANCE_DAYS:
|
||||||
|
return None, None # transition period or odd filing — let the caller fall back
|
||||||
|
return fy, best
|
||||||
|
|
||||||
|
|
||||||
|
def _nominal_fy_end(year: int, fiscal_year_end: str | None) -> date | None:
|
||||||
|
"""The issuer's nominal fiscal-year end in ``year`` from a MMDD string."""
|
||||||
|
if not fiscal_year_end or len(fiscal_year_end) != 4 or not fiscal_year_end.isdigit():
|
||||||
|
return None
|
||||||
|
month, day = int(fiscal_year_end[:2]), int(fiscal_year_end[2:])
|
||||||
|
if not 1 <= month <= 12 or not 1 <= day <= 31:
|
||||||
|
return None
|
||||||
|
while day > 28: # 52/53-week ends land on 0229/0230/0231 in some filings
|
||||||
|
try:
|
||||||
|
return date(year, month, day)
|
||||||
|
except ValueError:
|
||||||
|
day -= 1
|
||||||
|
return date(year, month, day)
|
||||||
|
|
||||||
|
|
||||||
|
def _fiscal_year_of(period_end: date, fiscal_year_end: str | None) -> int | None:
|
||||||
|
"""Which fiscal year ``period_end`` belongs to.
|
||||||
|
|
||||||
|
A 52/53-week calendar's real year end drifts around the nominal MMDD (and can
|
||||||
|
cross the calendar year), so allow drift before rolling into the next year.
|
||||||
|
"""
|
||||||
|
nominal = _nominal_fy_end(period_end.year, fiscal_year_end)
|
||||||
|
if nominal is None:
|
||||||
|
return None
|
||||||
|
return (
|
||||||
|
period_end.year
|
||||||
|
if period_end <= nominal + timedelta(days=_FYE_DRIFT_TOLERANCE_DAYS)
|
||||||
|
else period_end.year + 1
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _fiscal_context(facts: list[Fact], report_date: date) -> tuple[int | None, str | None]:
|
||||||
|
"""The filing's (fy, fp) taken as the majority context among the facts that
|
||||||
|
end at reportDate (the current-period facts, which share the filing's
|
||||||
|
context). Reject a tie so a conflicting context is never chosen arbitrarily."""
|
||||||
|
counts: dict[tuple[int, str], int] = {}
|
||||||
|
for f in facts:
|
||||||
|
if f.end == report_date and f.fy is not None and f.fp:
|
||||||
|
counts[(f.fy, f.fp)] = counts.get((f.fy, f.fp), 0) + 1
|
||||||
|
if not counts:
|
||||||
|
return None, None
|
||||||
|
ranked = sorted(counts.items(), key=lambda kv: kv[1], reverse=True)
|
||||||
|
if len(ranked) > 1 and ranked[0][1] == ranked[1][1]:
|
||||||
|
return None, None # tie → conflicting contexts, reject
|
||||||
|
return ranked[0][0]
|
||||||
|
|
||||||
|
|
||||||
|
def _select_ytd(
|
||||||
|
facts: list[Fact], concepts: list[str], report_date: date, fp: str, unit: str
|
||||||
|
) -> tuple[float | None, date | None]:
|
||||||
|
"""First present concept whose duration fact ends at reportDate and whose span
|
||||||
|
matches the fiscal-period-to-date length. Returns (val, period_start)."""
|
||||||
|
expected = _EXPECTED_YTD_DAYS[fp]
|
||||||
|
for concept in concepts:
|
||||||
|
best: Fact | None = None
|
||||||
|
best_diff: int | None = None
|
||||||
|
for f in facts:
|
||||||
|
if (
|
||||||
|
f.taxonomy != "us-gaap"
|
||||||
|
or f.concept != concept
|
||||||
|
or f.unit != unit
|
||||||
|
or f.start is None
|
||||||
|
or f.end != report_date
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
diff = abs((f.end - f.start).days - expected)
|
||||||
|
if diff <= _YTD_TOLERANCE_DAYS and (best_diff is None or diff < best_diff):
|
||||||
|
best, best_diff = f, diff
|
||||||
|
if best is not None:
|
||||||
|
return float(best.val), best.start
|
||||||
|
return None, None
|
||||||
|
|
||||||
|
|
||||||
|
def _select_instant(facts: list[Fact], concepts: list[str], report_date: date) -> float | None:
|
||||||
|
"""First present instant (balance-sheet) fact at end == reportDate, unit USD."""
|
||||||
|
for concept in concepts:
|
||||||
|
for f in facts:
|
||||||
|
if (
|
||||||
|
f.taxonomy == "us-gaap"
|
||||||
|
and f.concept == concept
|
||||||
|
and f.unit == "USD"
|
||||||
|
and f.start is None
|
||||||
|
and f.end == report_date
|
||||||
|
):
|
||||||
|
return float(f.val)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _compose_cash(facts: list[Fact], report_date: date) -> float | None:
|
||||||
|
cash = _select_instant(facts, _CASH, report_date)
|
||||||
|
st = _select_instant(facts, _ST_INVESTMENTS, report_date) # first present of the two
|
||||||
|
if cash is None and st is None:
|
||||||
|
return None
|
||||||
|
return (cash or 0.0) + (st or 0.0)
|
||||||
|
|
||||||
|
|
||||||
|
def _compose_debt(facts: list[Fact], report_date: date) -> float | None:
|
||||||
|
long_term = _select_instant(facts, _LONG_TERM_DEBT_AGG, report_date)
|
||||||
|
if long_term is None:
|
||||||
|
nc = _select_instant(facts, ["LongTermDebtNoncurrent"], report_date)
|
||||||
|
cur = _select_instant(facts, ["LongTermDebtCurrent"], report_date)
|
||||||
|
long_term = None if nc is None and cur is None else (nc or 0.0) + (cur or 0.0)
|
||||||
|
short_term = _select_instant(facts, _SHORT_TERM_DEBT, report_date)
|
||||||
|
if long_term is None and short_term is None:
|
||||||
|
return None
|
||||||
|
return (long_term or 0.0) + (short_term or 0.0)
|
||||||
|
|
||||||
|
|
||||||
|
def _select_shares(
|
||||||
|
facts: list[Fact], report_date: date
|
||||||
|
) -> tuple[float | None, date | None, bool]:
|
||||||
|
"""Issuer-wide shares outstanding as a single consolidated value (never a
|
||||||
|
class sum — companyfacts is non-dimensional — and never weighted-average/
|
||||||
|
diluted). Returns (value, shares_date, ambiguous).
|
||||||
|
|
||||||
|
1. Prefer the `dei:EntityCommonStockSharesOutstanding` cover-page instant;
|
||||||
|
its own end is the shares date (cover date != period_end).
|
||||||
|
2. Else fall back to `us-gaap:CommonStockSharesOutstanding` at period end
|
||||||
|
(e.g. Alphabet has no dei fact); shares date = reportDate.
|
||||||
|
Conflicting values within the chosen source → (None, None, True) to be
|
||||||
|
counted in validation.
|
||||||
|
"""
|
||||||
|
dei = [
|
||||||
|
f
|
||||||
|
for f in facts
|
||||||
|
if f.taxonomy == "dei"
|
||||||
|
and f.concept == "EntityCommonStockSharesOutstanding"
|
||||||
|
and f.unit == "shares"
|
||||||
|
and f.start is None
|
||||||
|
]
|
||||||
|
if dei:
|
||||||
|
if len({f.val for f in dei}) > 1:
|
||||||
|
return None, None, True
|
||||||
|
best = max(dei, key=lambda f: f.end)
|
||||||
|
return float(best.val), best.end, False
|
||||||
|
|
||||||
|
gaap = [
|
||||||
|
f
|
||||||
|
for f in facts
|
||||||
|
if f.taxonomy == "us-gaap"
|
||||||
|
and f.concept == "CommonStockSharesOutstanding"
|
||||||
|
and f.unit == "shares"
|
||||||
|
and f.start is None
|
||||||
|
and f.end == report_date
|
||||||
|
]
|
||||||
|
if gaap:
|
||||||
|
if len({f.val for f in gaap}) > 1:
|
||||||
|
return None, None, True
|
||||||
|
return float(gaap[0].val), report_date, False
|
||||||
|
|
||||||
|
return None, None, False # simply absent — not a conflict
|
||||||
|
|
||||||
|
|
||||||
|
def _select_weighted_avg_shares(facts: list[Fact], report_date: date) -> float | None:
|
||||||
|
"""The most recent quarter's weighted-average diluted share count.
|
||||||
|
|
||||||
|
Deliberately the **shortest** duration ending at reportDate, not the YTD one:
|
||||||
|
the shorter the window the closer the average sits to the current count, which
|
||||||
|
is what a market cap wants. Measured against issuers where the true
|
||||||
|
point-in-time count is available, the quarter average is within ~0.6%.
|
||||||
|
"""
|
||||||
|
best: tuple[int, float] | None = None
|
||||||
|
for concept in _WEIGHTED_AVG_SHARE_CONCEPTS:
|
||||||
|
for f in facts:
|
||||||
|
if (
|
||||||
|
f.taxonomy != "us-gaap"
|
||||||
|
or f.concept != concept
|
||||||
|
or f.unit != "shares"
|
||||||
|
or f.start is None
|
||||||
|
or f.end != report_date
|
||||||
|
or f.val <= 0
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
span = (f.end - f.start).days
|
||||||
|
if best is None or span < best[0]:
|
||||||
|
best = (span, float(f.val))
|
||||||
|
if best is not None:
|
||||||
|
return best[1] # first present concept wins, as elsewhere
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _d(value: Any) -> date | None:
|
||||||
|
if not value:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return date.fromisoformat(str(value)[:10])
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(value: Any) -> bool:
|
||||||
|
"""True for a finite numeric value (rejects None, bool, strings, NaN/inf)."""
|
||||||
|
return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value)
|
||||||
@@ -0,0 +1,991 @@
|
|||||||
|
"""SEC fundamentals importer (workstream A, phase A3).
|
||||||
|
|
||||||
|
A ``SourceImporter`` (see ``app/services/data_import.py``) that populates the
|
||||||
|
immutable ``fundamental_snapshots`` from SEC Company Facts and back-fills
|
||||||
|
``tickers.cik/sic/sic_description``. EDGAR-daily-index driven: it fetches
|
||||||
|
companyfacts only for tracked issuers that filed since the last run (full-history
|
||||||
|
backfill on the first run / for newly-added issuers). Shadow only — nothing reads
|
||||||
|
snapshots until A4.
|
||||||
|
|
||||||
|
Guardrails (design + reviews):
|
||||||
|
- ``detect_revision`` caches the resolved universe + the exact tracked index rows
|
||||||
|
and composes the revision from them; ``stage`` consumes those same cached inputs
|
||||||
|
(it does not refetch the index/universe) so promoted data always matches the
|
||||||
|
computed revision.
|
||||||
|
- Resolution is read-only in ``stage`` (proposals only); ticker writes happen in
|
||||||
|
``promote`` via ``apply_ticker_updates``.
|
||||||
|
- ``validate`` runs the **index↔Company-Facts consistency gate** before any write:
|
||||||
|
a tracked XBRL index accession missing from Company Facts fails the run (the two
|
||||||
|
are separate SEC products that can lag) so we retry rather than record a
|
||||||
|
null/partial snapshot. Non-XBRL amendments are skipped with a recorded reason.
|
||||||
|
A failure here blocks every later run (``source_max_date`` only advances on a
|
||||||
|
promoted run), so it names the offending filings in the alert and separates the
|
||||||
|
causes — ``not_in_companyfacts`` (facts lag) vs ``not_in_submissions`` (the
|
||||||
|
index row is absent from the issuer's own filing list, which no retry fixes).
|
||||||
|
- **Co-registrant recovery**, because "missing from Company Facts" is often not
|
||||||
|
missing at all: SEC files some combined parent/subsidiary filings' XBRL under
|
||||||
|
the co-registrant's CIK, so the ticker-carrying parent's own file never gets
|
||||||
|
that accession. The daily index lists every co-registrant of an accession, so
|
||||||
|
the facts are found there and re-stamped to the real filer — guarded by a
|
||||||
|
share-count continuity check so a subsidiary's standalone numbers can never be
|
||||||
|
stored as the parent's. Confirmed 2026-07-27 (NEE via FPL, DOW via Dow Chemical)
|
||||||
|
and it is not transient: an NEE filing misattributed in 2014 is still misfiled.
|
||||||
|
- **Bounded blocking.** Anything still unresolvable after ``MISSING_XBRL_RETRY_DAYS``
|
||||||
|
stops failing the whole import and enters a durable retry queue. The scheduled
|
||||||
|
importer retries queued accessions automatically, while the affected issuer is
|
||||||
|
excluded from actionable setups until its filing is recovered.
|
||||||
|
- ``promote`` inserts snapshots ``ON CONFLICT (accession) DO NOTHING`` (immutable),
|
||||||
|
reports differing existing accessions, and applies ticker updates in the same
|
||||||
|
transaction.
|
||||||
|
- ``reparse=True`` is the one exception to immutability, and it is deliberate:
|
||||||
|
it restages every accession with the current parser and **rewrites** the rows
|
||||||
|
that now reconstruct differently. Immutability protects SEC's record (one row
|
||||||
|
per accession, amendments retained) — but the stored row is *our* reconstruction,
|
||||||
|
so after a parser fix, keeping it is preserving a stale cache, not history.
|
||||||
|
Manually invoked through ``scripts/reparse_fundamentals.py``; never scheduled.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
from collections import Counter, defaultdict
|
||||||
|
from dataclasses import dataclass, field, replace
|
||||||
|
from datetime import date, datetime, timedelta, timezone
|
||||||
|
from typing import Any, Callable
|
||||||
|
|
||||||
|
from sqlalchemy import delete, select, update
|
||||||
|
|
||||||
|
from app.database import insert_for_session
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.sec_filing_gap import SecFilingGap
|
||||||
|
from app.models.system_event import SystemEvent
|
||||||
|
from app.services import fundamentals_quality_service
|
||||||
|
from app.services import sec_facts_parser as parser
|
||||||
|
from app.services import sec_universe
|
||||||
|
from app.services.data_import import STATUS_PROMOTED, ValidationResult
|
||||||
|
from app.services.sec_client import SecClient, SecError, cik10
|
||||||
|
from app.services.sec_facts_parser import FilingMeta, SnapshotRow
|
||||||
|
from app.services.sec_universe import ResolvedUniverse
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
SOURCE = "sec_facts"
|
||||||
|
_XBRL_FORMS = {"10-K", "10-Q", "10-K/A", "10-Q/A"}
|
||||||
|
# On the one-time backfill, require this fraction of tracked issuers to yield at
|
||||||
|
# least one snapshot (guards a broken fetch/parse from promoting a hollow table).
|
||||||
|
MIN_BACKFILL_COVERAGE = 0.5
|
||||||
|
# How long an index accession may stay unresolvable before the run stops failing
|
||||||
|
# on it. Genuine index↔facts lag clears within a day (a weekend stretches it to
|
||||||
|
# three); past that it is misfiled, not late, and blocking forever costs more
|
||||||
|
# than the missing filing does — see the unresolved-filing guardrail below.
|
||||||
|
MISSING_XBRL_RETRY_DAYS = 3
|
||||||
|
FILING_GAP_ESCALATE_DAYS = 14
|
||||||
|
# Share-count band a co-registrant-recovered row must land in, relative to the
|
||||||
|
# issuer's own last snapshot. Wide enough for buybacks/issuance, nowhere near
|
||||||
|
# wide enough to let a subsidiary shell's token float through (see _shares_continuous).
|
||||||
|
RECOVERY_SHARES_MIN = 0.5
|
||||||
|
RECOVERY_SHARES_MAX = 2.0
|
||||||
|
|
||||||
|
_SNAPSHOT_COLS = (
|
||||||
|
"cik", "accession", "form", "filed_date", "accepted_at", "period_start",
|
||||||
|
"period_end", "fiscal_year", "fiscal_period", "revenue", "net_income",
|
||||||
|
"operating_income", "diluted_eps", "cfo", "capex", "depreciation_amortization",
|
||||||
|
"cash_and_st_investments", "total_debt", "shares_outstanding",
|
||||||
|
"shares_outstanding_date", "weighted_avg_diluted_shares",
|
||||||
|
)
|
||||||
|
# Compare ALL source fields (every column except the accession key) to flag a
|
||||||
|
# differing existing accession — immutable, so we report, never mutate.
|
||||||
|
_COMPARE_COLS = tuple(c for c in _SNAPSHOT_COLS if c != "accession")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class StagedFundamentals:
|
||||||
|
resolved: ResolvedUniverse
|
||||||
|
sic_updates: list[tuple[int, str | None, str | None]] = field(default_factory=list)
|
||||||
|
rows: list[SnapshotRow] = field(default_factory=list)
|
||||||
|
skipped_filings: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
field_issues: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
skipped_non_xbrl: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
# Index rows we could not resolve to Company Facts, with a per-row reason
|
||||||
|
# (not_in_companyfacts | not_in_submissions | ...) — see _missing().
|
||||||
|
missing_xbrl: list[dict[str, Any]] = field(default_factory=list)
|
||||||
|
# Accessions parsed out of a co-registrant's Company Facts file.
|
||||||
|
recovered: list[dict[str, Any]] = field(default_factory=list)
|
||||||
|
invalid_payloads: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
existing_accessions: set[str] = field(default_factory=set)
|
||||||
|
# Tracked issuers whose registrant has NO XBRL 10-K/10-Q at all: they can
|
||||||
|
# never yield a snapshot, so this is a resolution problem (a ticker pointed
|
||||||
|
# at a successor shell), not missing data. See sec_universe.CIK_OVERRIDES_KEY.
|
||||||
|
no_xbrl_filings: list[dict[str, Any]] = field(default_factory=list)
|
||||||
|
discrepancies: list[dict[str, Any]] = field(default_factory=list)
|
||||||
|
backfill: bool = False
|
||||||
|
issuers_fetched: int = 0
|
||||||
|
issuers_with_rows: int = 0
|
||||||
|
|
||||||
|
|
||||||
|
def _now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
class SecFundamentalsImporter:
|
||||||
|
source = SOURCE
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
client_factory: Callable[[], SecClient] | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
reparse: bool = False,
|
||||||
|
) -> None:
|
||||||
|
self._client_factory = client_factory or (lambda: SecClient())
|
||||||
|
self.today = today or _now().date()
|
||||||
|
# Reparse: re-derive every stored accession with the CURRENT parser and
|
||||||
|
# rewrite the ones that now reconstruct differently. Snapshots are
|
||||||
|
# immutable with respect to SEC (one row per accession, amendments kept),
|
||||||
|
# but the stored row is *our reconstruction* — when a parser bug is fixed,
|
||||||
|
# leaving it stale is not immutability, it is a stale cache. Manually
|
||||||
|
# invoked via scripts/reparse_fundamentals.py; never scheduled.
|
||||||
|
self.reparse = reparse
|
||||||
|
# cached by detect_revision, consumed by stage:
|
||||||
|
self._resolved: ResolvedUniverse | None = None
|
||||||
|
self._index_rows: list[dict[str, Any]] = []
|
||||||
|
# accession -> the OTHER CIKs the daily index lists it under (co-registrants
|
||||||
|
# of a combined filing). Only populated for accessions a tracked issuer filed.
|
||||||
|
self._coregistrants: dict[str, list[int]] = {}
|
||||||
|
self._retry_rows: list[dict[str, Any]] = []
|
||||||
|
self._latest_index_date: date | None = None
|
||||||
|
self._backfill = False
|
||||||
|
|
||||||
|
# -- SourceImporter protocol -------------------------------------------
|
||||||
|
|
||||||
|
async def detect_revision(self, db) -> str | None:
|
||||||
|
async with self._client_factory() as client:
|
||||||
|
self._resolved = await sec_universe.resolve_ciks(db, client)
|
||||||
|
last_processed = await self._last_processed_index_date(db)
|
||||||
|
self._latest_index_date = await client.latest_index_date(self.today)
|
||||||
|
if self._latest_index_date is None:
|
||||||
|
raise SecError("no EDGAR daily index available")
|
||||||
|
# Reparse needs every accession restaged, not just those filed since
|
||||||
|
# the last run — the facts a fixed parser now accepts were never
|
||||||
|
# stored, so a reparse cannot be served from the database.
|
||||||
|
self._coregistrants = {}
|
||||||
|
if last_processed is None or self.reparse:
|
||||||
|
self._backfill = True
|
||||||
|
self._index_rows = []
|
||||||
|
else:
|
||||||
|
self._backfill = False
|
||||||
|
self._index_rows = await self._collect_index_rows(
|
||||||
|
client, last_processed, self._latest_index_date
|
||||||
|
)
|
||||||
|
content = sec_universe.index_content_hash(self._index_rows)
|
||||||
|
revision = sec_universe.compose_revision(
|
||||||
|
self._latest_index_date, content, self._resolved.symbol_to_cik
|
||||||
|
)
|
||||||
|
self._retry_rows = []
|
||||||
|
if not self._backfill:
|
||||||
|
self._retry_rows = await self._retry_backlog(
|
||||||
|
db,
|
||||||
|
set(self._resolved.cik_to_ticker_ids),
|
||||||
|
)
|
||||||
|
# Company Facts can change while the daily index revision stays fixed.
|
||||||
|
# Returning None deliberately bypasses the framework's no-op gate so a
|
||||||
|
# scheduled run retries every active gap.
|
||||||
|
return None if self._retry_rows else revision
|
||||||
|
|
||||||
|
async def stage(self, db) -> StagedFundamentals:
|
||||||
|
assert self._resolved is not None, "detect_revision must run first"
|
||||||
|
resolved = self._resolved
|
||||||
|
staged = StagedFundamentals(resolved=resolved, backfill=self._backfill)
|
||||||
|
|
||||||
|
cik_to_tids = resolved.cik_to_ticker_ids
|
||||||
|
# Whole index rows (not bare accessions): form + index date are what make
|
||||||
|
# an unresolvable filing diagnosable without re-walking the index by hand.
|
||||||
|
filed_by_cik: dict[int, list[dict[str, Any]]] = defaultdict(list)
|
||||||
|
for r in self._index_rows:
|
||||||
|
if r["cik"] in cik_to_tids:
|
||||||
|
filed_by_cik[r["cik"]].append(r)
|
||||||
|
|
||||||
|
# Promoted-around filings live in a small durable retry queue, including
|
||||||
|
# the one-time migration backfill. Merge them into normal incremental
|
||||||
|
# work so the scheduled importer heals them without operator action.
|
||||||
|
if not self._backfill:
|
||||||
|
seen = {
|
||||||
|
(int(cik), row["accession"])
|
||||||
|
for cik, rows in filed_by_cik.items()
|
||||||
|
for row in rows
|
||||||
|
}
|
||||||
|
for row in self._retry_rows:
|
||||||
|
cik = int(row["cik"])
|
||||||
|
key = (cik, row["accession"])
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
filed_by_cik[cik].append(row)
|
||||||
|
seen.add(key)
|
||||||
|
coregistrants = [int(value) for value in row.get("coregistrants") or []]
|
||||||
|
if coregistrants:
|
||||||
|
self._coregistrants[row["accession"]] = coregistrants
|
||||||
|
|
||||||
|
existing = await self._ciks_with_snapshots(db, set(cik_to_tids))
|
||||||
|
if self._backfill:
|
||||||
|
backfill_ciks = set(cik_to_tids)
|
||||||
|
else:
|
||||||
|
# Newly added issuers (resolved but no snapshots yet) get a full-history
|
||||||
|
# backfill; issuers that already have history are handled incrementally.
|
||||||
|
backfill_ciks = {c for c in cik_to_tids if c not in existing}
|
||||||
|
incremental_ciks = set(filed_by_cik) - backfill_ciks
|
||||||
|
|
||||||
|
# Continuity reference for co-registrant recovery, read once up front.
|
||||||
|
last_shares = await self._last_shares_outstanding(db, set(cik_to_tids))
|
||||||
|
|
||||||
|
async with self._client_factory() as client:
|
||||||
|
for cik in sorted(backfill_ciks | incremental_ciks):
|
||||||
|
is_backfill = cik in backfill_ciks
|
||||||
|
await self._stage_issuer(
|
||||||
|
client, cik, is_backfill, filed_by_cik, staged, last_shares
|
||||||
|
)
|
||||||
|
|
||||||
|
# Read-only discrepancy detection: an accession we reconstructed that is
|
||||||
|
# already stored, differing in ANY source field (immutable → report in
|
||||||
|
# validation, event on promote, never mutate). Also gives promote the
|
||||||
|
# existing set so its insert count is dialect-independent.
|
||||||
|
if staged.rows:
|
||||||
|
existing = await self._existing_by_accession(db, [r.accession for r in staged.rows])
|
||||||
|
staged.existing_accessions = set(existing)
|
||||||
|
for row in staged.rows:
|
||||||
|
old = existing.get(row.accession)
|
||||||
|
if old is not None:
|
||||||
|
fields = _diff_fields(row, old)
|
||||||
|
if fields:
|
||||||
|
staged.discrepancies.append({"accession": row.accession, "fields": fields})
|
||||||
|
return staged
|
||||||
|
|
||||||
|
async def _stage_issuer(
|
||||||
|
self, client, cik, is_backfill, filed_by_cik, staged, last_shares
|
||||||
|
) -> None:
|
||||||
|
cf = await client.companyfacts(cik)
|
||||||
|
bad = _companyfacts_structure_error(cf)
|
||||||
|
if bad is not None:
|
||||||
|
# Malformed payload (missing facts/units structure) — record separately
|
||||||
|
# and fail validation, rather than letting it degrade to skipped rows.
|
||||||
|
staged.invalid_payloads.append({"cik": cik10(cik), "reason": bad})
|
||||||
|
staged.issuers_fetched += 1
|
||||||
|
return
|
||||||
|
sub = await client.submissions(cik, include_history=is_backfill)
|
||||||
|
xbrl_meta, nonxbrl = _filing_meta(sub)
|
||||||
|
if not xbrl_meta:
|
||||||
|
staged.no_xbrl_filings.append(
|
||||||
|
{"cik": cik10(cik), "name": sub.get("name"), "tickers": sub.get("tickers")}
|
||||||
|
)
|
||||||
|
|
||||||
|
fiscal_year_end = sub.get("fiscal_year_end")
|
||||||
|
recovered_rows: list[SnapshotRow] = []
|
||||||
|
index_rows = {
|
||||||
|
row["accession"]: row for row in filed_by_cik.get(cik, [])
|
||||||
|
}
|
||||||
|
if is_backfill:
|
||||||
|
accns = set(xbrl_meta)
|
||||||
|
else:
|
||||||
|
present = parser.companyfacts_accessions(cf)
|
||||||
|
accns = set()
|
||||||
|
for index_row in filed_by_cik.get(cik, []):
|
||||||
|
accn = index_row["accession"]
|
||||||
|
if accn in nonxbrl:
|
||||||
|
staged.skipped_non_xbrl.append({"cik": cik10(cik), "accession": accn})
|
||||||
|
elif accn not in xbrl_meta:
|
||||||
|
# The daily index lists it but the issuer's own filing list does
|
||||||
|
# not (submissions lagging the index, no usable period metadata,
|
||||||
|
# or a co-registrant filing). NOT a Company-Facts lag — separate
|
||||||
|
# cause, separate fix, so it gets its own reason.
|
||||||
|
staged.missing_xbrl.append(
|
||||||
|
_missing(
|
||||||
|
cik,
|
||||||
|
index_row,
|
||||||
|
"not_in_submissions",
|
||||||
|
self.today,
|
||||||
|
self._coregistrants.get(accn),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
elif accn in present:
|
||||||
|
accns.add(accn)
|
||||||
|
else:
|
||||||
|
# Filed, XBRL, but absent from this issuer's Company Facts. Try
|
||||||
|
# the co-registrant file before treating it as missing data.
|
||||||
|
row, source_cik = await self._recover_from_coregistrant(
|
||||||
|
client, cik, accn, xbrl_meta, fiscal_year_end,
|
||||||
|
last_shares.get(cik10(cik)), staged,
|
||||||
|
)
|
||||||
|
if row is not None:
|
||||||
|
recovered_rows.append(row)
|
||||||
|
staged.recovered.append({
|
||||||
|
"cik": cik10(cik),
|
||||||
|
"accession": accn,
|
||||||
|
"source_cik": source_cik,
|
||||||
|
"form": index_row.get("form"),
|
||||||
|
})
|
||||||
|
else:
|
||||||
|
staged.missing_xbrl.append(_missing(
|
||||||
|
cik, index_row,
|
||||||
|
# Found, but it did not look like this issuer's own
|
||||||
|
# numbers — say so; it is not the same as absent.
|
||||||
|
"coregistrant_facts_rejected" if source_cik
|
||||||
|
else "not_in_companyfacts",
|
||||||
|
self.today,
|
||||||
|
self._coregistrants.get(accn),
|
||||||
|
))
|
||||||
|
|
||||||
|
# fiscalYearEnd (MMDD) is what lets the parser derive period identity from
|
||||||
|
# reportDate instead of SEC's unreliable fy/fp fields.
|
||||||
|
result = parser.parse_snapshots(cf, xbrl_meta, accns, fiscal_year_end=fiscal_year_end)
|
||||||
|
for skipped in result.skipped_filings:
|
||||||
|
index_row = index_rows.get(skipped["accession"])
|
||||||
|
if index_row is not None:
|
||||||
|
# Facts are present but our parser cannot construct a snapshot.
|
||||||
|
# A new index row keeps the normal grace period before promotion;
|
||||||
|
# a row already read from the queue retains its _retry_queue marker
|
||||||
|
# so later imports promote and retry without wedging the index.
|
||||||
|
staged.missing_xbrl.append(_missing(
|
||||||
|
cik,
|
||||||
|
index_row,
|
||||||
|
"parser_unusable",
|
||||||
|
self.today,
|
||||||
|
self._coregistrants.get(skipped["accession"]),
|
||||||
|
))
|
||||||
|
staged.rows.extend(result.rows)
|
||||||
|
staged.rows.extend(recovered_rows)
|
||||||
|
staged.skipped_filings.extend(result.skipped_filings)
|
||||||
|
staged.field_issues.extend(result.field_issues)
|
||||||
|
staged.issuers_fetched += 1
|
||||||
|
if result.rows or recovered_rows:
|
||||||
|
staged.issuers_with_rows += 1
|
||||||
|
|
||||||
|
# SIC proposal for this issuer's tickers (read-only; applied in promote).
|
||||||
|
sic = str(sub["sic"]) if sub.get("sic") else None
|
||||||
|
desc = sub.get("sic_description")
|
||||||
|
for tid in staged.resolved.cik_to_ticker_ids.get(cik, []):
|
||||||
|
staged.sic_updates.append((tid, sic, desc))
|
||||||
|
|
||||||
|
async def _recover_from_coregistrant(
|
||||||
|
self, client, cik: int, accn: str, xbrl_meta, fiscal_year_end, reference, staged,
|
||||||
|
) -> tuple[SnapshotRow | None, str | None]:
|
||||||
|
"""Look for ``accn``'s facts in a co-registrant's Company Facts file.
|
||||||
|
|
||||||
|
SEC sometimes files a combined parent/subsidiary filing's XBRL under the
|
||||||
|
co-registrant's CIK rather than the filer's — the ticker-carrying parent's
|
||||||
|
own file simply never gets that accession. Verified 2026-07-27 for NEE
|
||||||
|
(facts under Florida Power & Light) and DOW (under Dow Chemical); an NEE
|
||||||
|
filing misattributed the same way in **2014** is still misattributed, so
|
||||||
|
this does not self-correct and no amount of retrying recovers it.
|
||||||
|
|
||||||
|
Returns ``(row, source_cik)`` on success, ``(None, source_cik)`` when the
|
||||||
|
facts were found but rejected by the continuity guard, ``(None, None)``
|
||||||
|
when no co-registrant has them.
|
||||||
|
|
||||||
|
Incremental path only: the co-registrant map comes from the daily index,
|
||||||
|
which a backfill/reparse does not walk. A reparse therefore recovers a
|
||||||
|
filing only once SEC re-files it under the filer's own CIK.
|
||||||
|
"""
|
||||||
|
for co in self._coregistrants.get(accn, []):
|
||||||
|
try:
|
||||||
|
cf_co = await client.companyfacts(co)
|
||||||
|
except SecError:
|
||||||
|
continue # a co-registrant shell often has no facts file at all
|
||||||
|
if _companyfacts_structure_error(cf_co) is not None:
|
||||||
|
continue
|
||||||
|
result = parser.parse_snapshots(
|
||||||
|
cf_co, xbrl_meta, {accn}, fiscal_year_end=fiscal_year_end
|
||||||
|
)
|
||||||
|
if not result.rows:
|
||||||
|
continue
|
||||||
|
row = result.rows[0]
|
||||||
|
if not _shares_continuous(row.shares_outstanding, reference):
|
||||||
|
return None, cik10(co)
|
||||||
|
# A recovered row is the one most worth flagging, so its parser caveats
|
||||||
|
# travel with it rather than being dropped on the way out.
|
||||||
|
staged.field_issues.extend(result.field_issues)
|
||||||
|
# parse_snapshots stamps the CIK of the payload it read — re-stamp to
|
||||||
|
# the issuer that actually filed, or the row lands under the shell.
|
||||||
|
return replace(row, cik=cik10(cik)), cik10(co)
|
||||||
|
return None, None
|
||||||
|
|
||||||
|
async def validate(self, db, staged: StagedFundamentals) -> ValidationResult:
|
||||||
|
messages: list[str] = []
|
||||||
|
|
||||||
|
# Consistency gate — before any write. Only filings still inside the retry
|
||||||
|
# window block: a failure here stops every later run too (source_max_date
|
||||||
|
# advances on promotion alone), so blocking forever on a filing SEC has
|
||||||
|
# misfiled would cost far more than the one filing it withholds. Older
|
||||||
|
# ones are carried by promote() as a warning instead. The message names
|
||||||
|
# the filings: "which ones" has to be in the alert itself, not merely
|
||||||
|
# reconstructible by re-walking the index.
|
||||||
|
blocking = _within_retry_window(staged.missing_xbrl)
|
||||||
|
aged_out = _past_retry_window(staged.missing_xbrl)
|
||||||
|
if blocking:
|
||||||
|
messages.append(
|
||||||
|
f"{len(blocking)} tracked XBRL filing(s) unresolved within the "
|
||||||
|
f"{MISSING_XBRL_RETRY_DAYS}-day retry window "
|
||||||
|
f"({_reason_counts(blocking)}) — retry: {_missing_detail(blocking)}"
|
||||||
|
)
|
||||||
|
# Malformed companyfacts payloads must fail, not degrade to skipped rows.
|
||||||
|
if staged.invalid_payloads:
|
||||||
|
messages.append(
|
||||||
|
f"{len(staged.invalid_payloads)} issuer(s) returned a malformed "
|
||||||
|
"companyfacts payload (missing facts structure)"
|
||||||
|
)
|
||||||
|
|
||||||
|
accns = [r.accession for r in staged.rows]
|
||||||
|
if len(accns) != len(set(accns)):
|
||||||
|
messages.append("duplicate accession in staged snapshots")
|
||||||
|
|
||||||
|
if staged.backfill:
|
||||||
|
n_issuers = len(staged.resolved.cik_to_ticker_ids)
|
||||||
|
coverage = staged.issuers_with_rows / n_issuers if n_issuers else 0.0
|
||||||
|
if coverage < MIN_BACKFILL_COVERAGE:
|
||||||
|
messages.append(
|
||||||
|
f"backfill coverage {coverage:.0%} < {MIN_BACKFILL_COVERAGE:.0%}"
|
||||||
|
)
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
"backfill": staged.backfill,
|
||||||
|
"issuers_fetched": staged.issuers_fetched,
|
||||||
|
"issuers_with_rows": staged.issuers_with_rows,
|
||||||
|
"snapshot_rows": len(staged.rows),
|
||||||
|
"skipped_filings": len(staged.skipped_filings),
|
||||||
|
"field_issues": len(staged.field_issues),
|
||||||
|
"skipped_non_xbrl": len(staged.skipped_non_xbrl),
|
||||||
|
"no_xbrl_filings": staged.no_xbrl_filings[:50],
|
||||||
|
"no_xbrl_filings_count": len(staged.no_xbrl_filings),
|
||||||
|
"no_xbrl_ciks": sorted({
|
||||||
|
str(item["cik"])
|
||||||
|
for item in staged.no_xbrl_filings
|
||||||
|
if item.get("cik")
|
||||||
|
}),
|
||||||
|
"missing_xbrl": staged.missing_xbrl[:50],
|
||||||
|
"missing_xbrl_count": len(staged.missing_xbrl),
|
||||||
|
"missing_xbrl_blocking": len(blocking),
|
||||||
|
"recovered_from_coregistrant": staged.recovered[:50],
|
||||||
|
"recovered_count": len(staged.recovered),
|
||||||
|
# Complete compact gate input; detailed audit lists above stay capped.
|
||||||
|
"setup_blocked_ciks": sorted({
|
||||||
|
str(item["cik"])
|
||||||
|
for item in [*staged.missing_xbrl, *staged.no_xbrl_filings]
|
||||||
|
if item.get("cik")
|
||||||
|
}),
|
||||||
|
"invalid_payloads": staged.invalid_payloads,
|
||||||
|
"cik_updates": len(staged.resolved.cik_updates),
|
||||||
|
# differing existing accessions (immutable — kept, reported here)
|
||||||
|
"discrepancies": staged.discrepancies[:50],
|
||||||
|
"discrepancy_count": len(staged.discrepancies),
|
||||||
|
}
|
||||||
|
return ValidationResult(
|
||||||
|
ok=not messages,
|
||||||
|
summary=summary,
|
||||||
|
source_max_date=self._latest_index_date,
|
||||||
|
messages=messages,
|
||||||
|
# Company-Facts absence is usually publication lag, but can also be a
|
||||||
|
# permanent co-registrant misfile that the daily index did not expose.
|
||||||
|
# Defer quietly at first; the framework warns if promotions stay stale.
|
||||||
|
retryable=(
|
||||||
|
len(messages) == 1
|
||||||
|
and bool(blocking)
|
||||||
|
and all(
|
||||||
|
m.get("reason") in {"not_in_companyfacts", "parser_unusable"}
|
||||||
|
for m in blocking
|
||||||
|
)
|
||||||
|
),
|
||||||
|
deferred_alert_after_days=MISSING_XBRL_RETRY_DAYS,
|
||||||
|
deferred_alert_messages=(
|
||||||
|
[
|
||||||
|
f"{len(aged_out)} tracked SEC filing(s) remain unresolved past "
|
||||||
|
f"the {MISSING_XBRL_RETRY_DAYS}-day retry window. They will "
|
||||||
|
f"enter automatic retry and block affected symbols from setups: "
|
||||||
|
f"{_missing_detail(aged_out)}"
|
||||||
|
]
|
||||||
|
if aged_out
|
||||||
|
else []
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def promote(self, db, staged: StagedFundamentals, run_id: int) -> dict[str, int]:
|
||||||
|
inserted = 0
|
||||||
|
updated = 0
|
||||||
|
# Only accessions whose reconstruction actually changed are rewritten;
|
||||||
|
# an unchanged stored row is left completely alone.
|
||||||
|
changed = {d["accession"] for d in staged.discrepancies} if self.reparse else set()
|
||||||
|
for row in staged.rows:
|
||||||
|
if row.accession in staged.existing_accessions:
|
||||||
|
if row.accession in changed:
|
||||||
|
# Write the FULL column set (_row_values covers _SNAPSHOT_COLS)
|
||||||
|
# so a rewritten row is never half old-parse, half new-parse.
|
||||||
|
# created_at stays at the original insert; import_run_id
|
||||||
|
# attributes the rewrite.
|
||||||
|
values = _row_values(row, run_id)
|
||||||
|
values.pop("created_at", None)
|
||||||
|
await db.execute(
|
||||||
|
update(FundamentalSnapshot)
|
||||||
|
.where(FundamentalSnapshot.accession == row.accession)
|
||||||
|
.values(**values)
|
||||||
|
)
|
||||||
|
updated += 1
|
||||||
|
continue # otherwise immutable — keep the original row
|
||||||
|
stmt = insert_for_session(db, FundamentalSnapshot).values(**_row_values(row, run_id))
|
||||||
|
stmt = stmt.on_conflict_do_nothing(index_elements=["accession"]) # race belt-and-suspenders
|
||||||
|
await db.execute(stmt)
|
||||||
|
inserted += 1
|
||||||
|
|
||||||
|
# Synchronize the retry queue in the snapshot-promotion transaction.
|
||||||
|
existing_gaps = (await db.execute(select(SecFilingGap))).scalars().all()
|
||||||
|
existing_gap_accessions = {gap.accession for gap in existing_gaps}
|
||||||
|
resolved_accessions = {row.accession for row in staged.rows}
|
||||||
|
# A filing now classified non-XBRL can never yield a snapshot and is no
|
||||||
|
# longer a fundamentals completeness gap.
|
||||||
|
resolved_accessions.update(
|
||||||
|
item["accession"] for item in staged.skipped_non_xbrl
|
||||||
|
)
|
||||||
|
queue_resolved = 0
|
||||||
|
if resolved_accessions:
|
||||||
|
result = await db.execute(
|
||||||
|
delete(SecFilingGap).where(
|
||||||
|
SecFilingGap.accession.in_(resolved_accessions)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
queue_resolved = int(result.rowcount or 0)
|
||||||
|
|
||||||
|
now = _now()
|
||||||
|
tolerated = _past_retry_window(staged.missing_xbrl)
|
||||||
|
for gap in tolerated:
|
||||||
|
stmt = insert_for_session(db, SecFilingGap).values(
|
||||||
|
cik=gap["cik"],
|
||||||
|
accession=gap["accession"],
|
||||||
|
form=gap.get("form"),
|
||||||
|
index_date=gap.get("index_date"),
|
||||||
|
reason=gap["reason"],
|
||||||
|
coregistrant_ciks_json=json.dumps(gap.get("coregistrants") or []),
|
||||||
|
first_seen_at=now,
|
||||||
|
last_attempted_at=now,
|
||||||
|
)
|
||||||
|
await db.execute(
|
||||||
|
stmt.on_conflict_do_update(
|
||||||
|
index_elements=["accession"],
|
||||||
|
set_={
|
||||||
|
"cik": stmt.excluded.cik,
|
||||||
|
"form": stmt.excluded.form,
|
||||||
|
"index_date": stmt.excluded.index_date,
|
||||||
|
"reason": stmt.excluded.reason,
|
||||||
|
"coregistrant_ciks_json": stmt.excluded.coregistrant_ciks_json,
|
||||||
|
"last_attempted_at": stmt.excluded.last_attempted_at,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Remove gaps made irrelevant by a later valid 10-K/10-Q. Quality reads
|
||||||
|
# already ignore them; physical cleanup keeps the queue small.
|
||||||
|
active_ids = {gap.id for gap in await fundamentals_quality_service.active_gaps(db)}
|
||||||
|
obsolete_ids = {
|
||||||
|
gap.id for gap in existing_gaps
|
||||||
|
if gap.id not in active_ids and gap.accession not in resolved_accessions
|
||||||
|
}
|
||||||
|
if obsolete_ids:
|
||||||
|
result = await db.execute(
|
||||||
|
delete(SecFilingGap).where(SecFilingGap.id.in_(obsolete_ids))
|
||||||
|
)
|
||||||
|
queue_resolved += int(result.rowcount or 0)
|
||||||
|
|
||||||
|
newly_queued = [
|
||||||
|
gap for gap in tolerated
|
||||||
|
if gap["accession"] not in existing_gap_accessions
|
||||||
|
]
|
||||||
|
|
||||||
|
# Warn (in-transaction, so it commits atomically with the promotion) when
|
||||||
|
# any existing accession reconstructed differently — kept immutable.
|
||||||
|
if staged.discrepancies:
|
||||||
|
accns = ", ".join(d["accession"] for d in staged.discrepancies[:10])
|
||||||
|
disposition = (
|
||||||
|
f"REWRITTEN by reparse run {run_id}" if self.reparse else "kept immutable"
|
||||||
|
)
|
||||||
|
db.add(SystemEvent(
|
||||||
|
severity="warning",
|
||||||
|
source="sec_facts",
|
||||||
|
code="snapshot_reparse" if self.reparse else "snapshot_discrepancy",
|
||||||
|
message=(
|
||||||
|
f"{len(staged.discrepancies)} stored accession(s) reconstructed "
|
||||||
|
f"differently; {disposition}: {accns}"
|
||||||
|
)[:4000],
|
||||||
|
dedup_key=f"sec_facts:discrepancy:{run_id}",
|
||||||
|
created_at=_now(),
|
||||||
|
))
|
||||||
|
|
||||||
|
# Persistent current gaps get one actionable escalation rather than a
|
||||||
|
# daily warning. The nullable marker makes this durable and noise-free.
|
||||||
|
escalation_cutoff = now - timedelta(days=FILING_GAP_ESCALATE_DAYS)
|
||||||
|
aged_gaps = (
|
||||||
|
await db.execute(
|
||||||
|
select(SecFilingGap).where(
|
||||||
|
SecFilingGap.first_seen_at <= escalation_cutoff,
|
||||||
|
SecFilingGap.escalated_at.is_(None),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
if aged_gaps:
|
||||||
|
named = ", ".join(
|
||||||
|
f"{gap.cik}/{gap.accession} ({gap.reason})"
|
||||||
|
for gap in aged_gaps[:10]
|
||||||
|
)
|
||||||
|
db.add(SystemEvent(
|
||||||
|
severity="warning",
|
||||||
|
source="sec_facts",
|
||||||
|
code="filing_gap_aged",
|
||||||
|
message=(
|
||||||
|
f"{len(aged_gaps)} SEC filing gap(s) remain unresolved after "
|
||||||
|
f"{FILING_GAP_ESCALATE_DAYS} days; affected setups remain paused. "
|
||||||
|
f"Review the filing/CIK mapping or parser: {named}"
|
||||||
|
)[:4000],
|
||||||
|
dedup_key=f"sec_facts:filing_gap_aged:{run_id}",
|
||||||
|
created_at=now,
|
||||||
|
))
|
||||||
|
await db.execute(
|
||||||
|
update(SecFilingGap)
|
||||||
|
.where(SecFilingGap.id.in_([gap.id for gap in aged_gaps]))
|
||||||
|
.values(escalated_at=now)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Recovered rows are real data from an unexpected place — record where they
|
||||||
|
# came from, so a wrong recovery is auditable rather than invisible.
|
||||||
|
if staged.recovered:
|
||||||
|
named = ", ".join(
|
||||||
|
f"{r['accession']} <- CIK {r['source_cik']}" for r in staged.recovered[:10]
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"sec_facts: recovered %d filing(s) from co-registrants: %s",
|
||||||
|
len(staged.recovered),
|
||||||
|
named,
|
||||||
|
)
|
||||||
|
|
||||||
|
# One warning when a gap first enters automatic retry. Repeating it every
|
||||||
|
# day adds noise; the queue remains the durable actionable state.
|
||||||
|
if newly_queued:
|
||||||
|
symbols_by_cik: dict[str, list[str]] = defaultdict(list)
|
||||||
|
for symbol, cik in staged.resolved.symbol_to_cik.items():
|
||||||
|
symbols_by_cik[cik10(cik)].append(symbol)
|
||||||
|
named = ", ".join(
|
||||||
|
f"{'/'.join(symbols_by_cik.get(gap['cik'], [])) or gap['cik']}"
|
||||||
|
f"/{gap['accession']}"
|
||||||
|
for gap in newly_queued[:10]
|
||||||
|
)
|
||||||
|
db.add(SystemEvent(
|
||||||
|
severity="warning",
|
||||||
|
source="sec_facts",
|
||||||
|
code="unresolved_filing",
|
||||||
|
message=(
|
||||||
|
f"{len(newly_queued)} filing(s) entered automatic SEC retry. "
|
||||||
|
f"Affected symbols are blocked from new actionable setups until "
|
||||||
|
f"their filing is recovered: {named}"
|
||||||
|
)[:4000],
|
||||||
|
dedup_key=f"sec_facts:unresolved_filing:{run_id}",
|
||||||
|
created_at=_now(),
|
||||||
|
))
|
||||||
|
|
||||||
|
# A new registrant may have no XBRL filing yet. Keep it out of actionable
|
||||||
|
# setups, but log it instead of raising a recurring operator warning.
|
||||||
|
if staged.no_xbrl_filings:
|
||||||
|
named = ", ".join(
|
||||||
|
f"{e['cik']} ({e.get('name') or '?'})" for e in staged.no_xbrl_filings[:10]
|
||||||
|
)
|
||||||
|
logger.info(
|
||||||
|
"sec_facts: %d registrant(s) have no XBRL history yet: %s",
|
||||||
|
len(staged.no_xbrl_filings),
|
||||||
|
named,
|
||||||
|
)
|
||||||
|
|
||||||
|
ticker_counts = await sec_universe.apply_ticker_updates(
|
||||||
|
db, staged.resolved, staged.sic_updates
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"inserted": inserted,
|
||||||
|
"updated": updated,
|
||||||
|
"existing_unchanged": len(staged.existing_accessions) - updated,
|
||||||
|
"discrepancies": len(staged.discrepancies),
|
||||||
|
"retry_queue_added": len(newly_queued),
|
||||||
|
"retry_queue_resolved": queue_resolved,
|
||||||
|
**ticker_counts,
|
||||||
|
}
|
||||||
|
|
||||||
|
# -- helpers -----------------------------------------------------------
|
||||||
|
|
||||||
|
async def _retry_backlog(
|
||||||
|
self,
|
||||||
|
db,
|
||||||
|
tracked_ciks: set[int],
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Active typed gaps; migration 028 owns historical bootstrap."""
|
||||||
|
if not tracked_ciks:
|
||||||
|
return []
|
||||||
|
tracked = {cik10(cik) for cik in tracked_ciks}
|
||||||
|
candidates: dict[str, dict[str, Any]] = {}
|
||||||
|
|
||||||
|
queued = await fundamentals_quality_service.active_gaps(db, tracked)
|
||||||
|
for gap in queued:
|
||||||
|
try:
|
||||||
|
coregistrants = json.loads(gap.coregistrant_ciks_json or "[]")
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
coregistrants = []
|
||||||
|
candidates[gap.accession] = {
|
||||||
|
"cik": gap.cik,
|
||||||
|
"accession": gap.accession,
|
||||||
|
"form": gap.form,
|
||||||
|
"index_date": gap.index_date,
|
||||||
|
"reason": gap.reason,
|
||||||
|
"coregistrants": coregistrants,
|
||||||
|
"_retry_queue": True,
|
||||||
|
}
|
||||||
|
|
||||||
|
if not candidates:
|
||||||
|
return []
|
||||||
|
resolved = set(
|
||||||
|
(
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalSnapshot.accession).where(
|
||||||
|
FundamentalSnapshot.accession.in_(list(candidates))
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
)
|
||||||
|
return [
|
||||||
|
item
|
||||||
|
for accession, item in candidates.items()
|
||||||
|
if accession not in resolved
|
||||||
|
]
|
||||||
|
|
||||||
|
async def _last_processed_index_date(self, db) -> date | None:
|
||||||
|
return (
|
||||||
|
await db.execute(
|
||||||
|
select(DataImportRun.source_max_date)
|
||||||
|
.where(DataImportRun.source == SOURCE, DataImportRun.status == STATUS_PROMOTED)
|
||||||
|
.order_by(DataImportRun.id.desc())
|
||||||
|
.limit(1)
|
||||||
|
)
|
||||||
|
).scalar_one_or_none()
|
||||||
|
|
||||||
|
async def _collect_index_rows(
|
||||||
|
self, client: SecClient, last_processed: date, latest: date
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
# Walk EVERY unprocessed date. No cap — dropping the older part of a long
|
||||||
|
# outage while still advancing source_max_date would permanently lose
|
||||||
|
# those filings. A large gap is one-time cost, not silent data loss.
|
||||||
|
tracked = set(self._resolved.cik_to_ticker_ids) if self._resolved else set()
|
||||||
|
gap = (latest - last_processed).days
|
||||||
|
if gap > 60:
|
||||||
|
logger.warning("sec_facts: %d-day index gap since %s; walking all", gap, last_processed)
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
day = last_processed + timedelta(days=1)
|
||||||
|
while day <= latest:
|
||||||
|
# Group the whole day first: a combined filing is listed once per
|
||||||
|
# co-registrant CIK, and those sibling CIKs are the only pointer to
|
||||||
|
# where SEC may have put the XBRL (see _recover_from_coregistrant).
|
||||||
|
by_accession: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||||
|
for r in await client.daily_index(day):
|
||||||
|
if r["form"] in _XBRL_FORMS:
|
||||||
|
by_accession[r["accession"]].append(r)
|
||||||
|
for accession, group in by_accession.items():
|
||||||
|
filers = {r["cik"] for r in group}
|
||||||
|
tracked_filers = filers & tracked
|
||||||
|
if not tracked_filers:
|
||||||
|
continue
|
||||||
|
siblings = sorted(filers - tracked_filers)
|
||||||
|
if siblings:
|
||||||
|
self._coregistrants[accession] = siblings
|
||||||
|
for r in group:
|
||||||
|
if r["cik"] in tracked_filers:
|
||||||
|
r["index_date"] = day # not hashed (revision uses cik/accession)
|
||||||
|
rows.append(r)
|
||||||
|
day += timedelta(days=1)
|
||||||
|
return rows
|
||||||
|
|
||||||
|
async def _ciks_with_snapshots(self, db, ciks: set[int]) -> set[int]:
|
||||||
|
if not ciks:
|
||||||
|
return set()
|
||||||
|
cik_strs = [cik10(c) for c in ciks]
|
||||||
|
found = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalSnapshot.cik)
|
||||||
|
.where(FundamentalSnapshot.cik.in_(cik_strs))
|
||||||
|
.distinct()
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
return {int(c) for c in found}
|
||||||
|
|
||||||
|
async def _last_shares_outstanding(self, db, ciks: set[int]) -> dict[str, float]:
|
||||||
|
"""Latest known shares outstanding per tracked issuer — the continuity
|
||||||
|
reference co-registrant recovery is checked against."""
|
||||||
|
if not ciks:
|
||||||
|
return {}
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalSnapshot.cik, FundamentalSnapshot.shares_outstanding)
|
||||||
|
.where(
|
||||||
|
FundamentalSnapshot.cik.in_([cik10(c) for c in ciks]),
|
||||||
|
FundamentalSnapshot.shares_outstanding.is_not(None),
|
||||||
|
)
|
||||||
|
# Last write per cik wins, so the sort must be total: an amendment
|
||||||
|
# and its original share a period_end, and an undefined tie there
|
||||||
|
# would make recovery non-deterministic across runs and dialects.
|
||||||
|
.order_by(
|
||||||
|
FundamentalSnapshot.period_end,
|
||||||
|
FundamentalSnapshot.filed_date,
|
||||||
|
FundamentalSnapshot.accession,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
return {cik: float(shares) for cik, shares in rows}
|
||||||
|
|
||||||
|
async def _existing_by_accession(self, db, accessions: list[str]) -> dict[str, FundamentalSnapshot]:
|
||||||
|
if not accessions:
|
||||||
|
return {}
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalSnapshot).where(FundamentalSnapshot.accession.in_(accessions))
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
return {r.accession: r for r in rows}
|
||||||
|
|
||||||
|
|
||||||
|
def _companyfacts_structure_error(cf: Any) -> str | None:
|
||||||
|
"""None if the payload is structurally sound, else a reason string. Checks the
|
||||||
|
top-level ``facts`` mapping AND that every concept carries a ``units`` mapping —
|
||||||
|
a missing/non-dict units would silently drop that concept's facts otherwise."""
|
||||||
|
if not isinstance(cf, dict) or not isinstance(cf.get("facts"), dict):
|
||||||
|
return "missing facts structure"
|
||||||
|
for concepts in cf["facts"].values():
|
||||||
|
if not isinstance(concepts, dict):
|
||||||
|
return "malformed taxonomy structure"
|
||||||
|
for body in concepts.values():
|
||||||
|
if not isinstance(body, dict) or not isinstance(body.get("units"), dict):
|
||||||
|
return "missing units structure"
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _missing(
|
||||||
|
cik: int,
|
||||||
|
row: dict[str, Any],
|
||||||
|
reason: str,
|
||||||
|
today: date,
|
||||||
|
coregistrants: list[int] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""One unresolvable index row, carrying everything needed to look the filing
|
||||||
|
up by hand (EDGAR accession + the index date it was seen on) and to decide
|
||||||
|
whether it is still young enough to be worth blocking on."""
|
||||||
|
index_date = row.get("index_date")
|
||||||
|
age_days = (
|
||||||
|
(today - index_date).days if isinstance(index_date, date) else 0
|
||||||
|
)
|
||||||
|
if row.get("_retry_queue"):
|
||||||
|
age_days = max(age_days, MISSING_XBRL_RETRY_DAYS + 1)
|
||||||
|
return {
|
||||||
|
"cik": cik10(cik),
|
||||||
|
"accession": row["accession"],
|
||||||
|
"form": row.get("form"),
|
||||||
|
"index_date": index_date,
|
||||||
|
# A newly observed row without a date blocks safely. A durable queue row
|
||||||
|
# has already passed the bounded window and is forced aged-out above.
|
||||||
|
"age_days": age_days,
|
||||||
|
"reason": reason,
|
||||||
|
"coregistrants": list(coregistrants or []),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _within_retry_window(missing: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||||
|
return [m for m in missing if m.get("age_days", 0) <= MISSING_XBRL_RETRY_DAYS]
|
||||||
|
|
||||||
|
|
||||||
|
def _past_retry_window(missing: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||||
|
return [m for m in missing if m.get("age_days", 0) > MISSING_XBRL_RETRY_DAYS]
|
||||||
|
|
||||||
|
|
||||||
|
def _shares_continuous(shares: float | None, reference: float | None) -> bool:
|
||||||
|
"""Does a co-registrant-recovered share count look like this issuer's own?
|
||||||
|
|
||||||
|
The failure worth preventing is storing a subsidiary's standalone facts as the
|
||||||
|
parent's. A co-registrant shell holds a token float — Florida Power & Light
|
||||||
|
against NextEra's 2.09bn shares — so any sane band separates them while still
|
||||||
|
tolerating buybacks and issuance. With no history to compare against (a newly
|
||||||
|
tracked issuer) or no share count at all, recovery is refused, not guessed.
|
||||||
|
"""
|
||||||
|
if not shares or not reference:
|
||||||
|
return False
|
||||||
|
return RECOVERY_SHARES_MIN <= shares / reference <= RECOVERY_SHARES_MAX
|
||||||
|
|
||||||
|
|
||||||
|
def _reason_counts(missing: list[dict[str, Any]]) -> str:
|
||||||
|
counts = Counter(m["reason"] for m in missing)
|
||||||
|
return ", ".join(f"{reason}={n}" for reason, n in sorted(counts.items()))
|
||||||
|
|
||||||
|
|
||||||
|
def _missing_detail(missing: list[dict[str, Any]], limit: int = 10) -> str:
|
||||||
|
detail = ", ".join(
|
||||||
|
f"{m['cik']}/{m['accession']} {m.get('form') or '?'} "
|
||||||
|
f"[{m.get('index_date') or '?'}] {m['reason']}"
|
||||||
|
for m in missing[:limit]
|
||||||
|
)
|
||||||
|
if len(missing) > limit:
|
||||||
|
detail += f", +{len(missing) - limit} more"
|
||||||
|
return detail
|
||||||
|
|
||||||
|
|
||||||
|
def _filing_meta(sub: dict[str, Any]) -> tuple[dict[str, FilingMeta], set[str]]:
|
||||||
|
"""(xbrl_meta, nonxbrl_accessions) from a submissions payload. xbrl_meta only
|
||||||
|
includes 10-K/10-Q(/A) filings that are XBRL and have full period metadata."""
|
||||||
|
xbrl: dict[str, FilingMeta] = {}
|
||||||
|
nonxbrl: set[str] = set()
|
||||||
|
for f in sub.get("filings", []):
|
||||||
|
if f["form"] not in _XBRL_FORMS:
|
||||||
|
continue
|
||||||
|
if not f.get("is_xbrl"):
|
||||||
|
nonxbrl.add(f["accession"])
|
||||||
|
continue
|
||||||
|
if not (f.get("report_date") and f.get("filing_date") and f.get("acceptance_datetime")):
|
||||||
|
continue
|
||||||
|
xbrl[f["accession"]] = FilingMeta(
|
||||||
|
report_date=date.fromisoformat(f["report_date"]),
|
||||||
|
filing_date=date.fromisoformat(f["filing_date"]),
|
||||||
|
accepted_at=_parse_dt(f["acceptance_datetime"]),
|
||||||
|
form=f["form"],
|
||||||
|
)
|
||||||
|
return xbrl, nonxbrl
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_dt(value: str) -> datetime:
|
||||||
|
return datetime.fromisoformat(value.replace("Z", "+00:00"))
|
||||||
|
|
||||||
|
|
||||||
|
def _row_values(row: SnapshotRow, run_id: int) -> dict[str, Any]:
|
||||||
|
values = {col: getattr(row, col) for col in _SNAPSHOT_COLS}
|
||||||
|
values["import_run_id"] = run_id
|
||||||
|
values["created_at"] = _now()
|
||||||
|
return values
|
||||||
|
|
||||||
|
|
||||||
|
def _diff_fields(row: SnapshotRow, old: FundamentalSnapshot) -> list[str]:
|
||||||
|
"""Source fields where a re-parsed row differs from the stored row."""
|
||||||
|
return [
|
||||||
|
col for col in _COMPARE_COLS
|
||||||
|
if not _same_value(getattr(row, col), getattr(old, col))
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _same_value(parsed: Any, stored: Any) -> bool:
|
||||||
|
"""Compare a freshly parsed value against its stored round-trip.
|
||||||
|
|
||||||
|
Datetimes need care: every timestamp here is UTC by construction, but
|
||||||
|
``DateTime(timezone=True)`` only preserves tzinfo on Postgres — SQLite hands
|
||||||
|
back a naive value. Comparing representations would report an unchanged row
|
||||||
|
as differing, which would both spam the discrepancy warning and make a
|
||||||
|
reparse rewrite every row it touched. Compare instants instead.
|
||||||
|
"""
|
||||||
|
if isinstance(parsed, datetime) and isinstance(stored, datetime):
|
||||||
|
return _as_utc(parsed) == _as_utc(stored)
|
||||||
|
return parsed == stored
|
||||||
|
|
||||||
|
|
||||||
|
def _as_utc(value: datetime) -> datetime:
|
||||||
|
return value if value.tzinfo is not None else value.replace(tzinfo=timezone.utc)
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
"""Tracked-universe CIK/SIC resolution and the SEC importer's composite revision.
|
||||||
|
|
||||||
|
Resolves the app's tracked tickers to SEC issuers (CIK) and prepares
|
||||||
|
``tickers.cik/sic/sic_description`` back-fills. Also builds the **universe
|
||||||
|
fingerprint** in the importer's composite revision, so adding a ticker changes
|
||||||
|
the revision and forces a run instead of being ``no_op``'d away or starved
|
||||||
|
waiting for its issuer to file (A3 design, Decision 1 review fix).
|
||||||
|
|
||||||
|
**Transaction contract:** resolution is read-only — `resolve_ciks` and
|
||||||
|
`fetch_sic_updates` compute *proposed* updates and mutate nothing. They run in
|
||||||
|
the importer's `stage` (which must not write, or a failed validation would leak
|
||||||
|
changes on the framework's failure commit). The proposals are applied only in
|
||||||
|
`promote`, via `apply_ticker_updates`, atomically with the snapshot inserts.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from typing import Iterable
|
||||||
|
|
||||||
|
from sqlalchemy import select, update
|
||||||
|
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import settings_store
|
||||||
|
from app.services.earnings_alignment import normalise_symbol
|
||||||
|
from app.services.sec_client import SecClient
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# JSON {symbol: cik} pinning a ticker to a specific registrant, overriding
|
||||||
|
# company_tickers.json. Needed when SEC maps a ticker to a successor entity that
|
||||||
|
# has not filed: XOM points at CIK 2115436 "ExxonMobil Holdings Corp" (zero XBRL
|
||||||
|
# filings) while every 10-K/10-Q — including one filed 2026-05-04 — is still under
|
||||||
|
# CIK 34088. Which registrant is the real filer is a judgement about a corporate
|
||||||
|
# event, so it is pinned explicitly rather than guessed. The importer's
|
||||||
|
# `no_xbrl_filings` warning is what tells you a pin is needed.
|
||||||
|
CIK_OVERRIDES_KEY = "sec_cik_overrides"
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ResolvedUniverse:
|
||||||
|
"""Read-only result of CIK resolution. `cik_updates` are proposed writes
|
||||||
|
(ticker_id → new cik string) applied later in promote."""
|
||||||
|
|
||||||
|
symbol_to_cik: dict[str, int] = field(default_factory=dict)
|
||||||
|
cik_to_ticker_ids: dict[int, list[int]] = field(default_factory=dict)
|
||||||
|
cik_updates: list[tuple[int, str]] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
async def resolve_ciks(db, client: SecClient) -> ResolvedUniverse:
|
||||||
|
"""Resolve tracked tickers to CIKs via company_tickers.json. **Read-only** —
|
||||||
|
returns the mapping + proposed `tickers.cik` writes; mutates nothing."""
|
||||||
|
ticker_to_cik = await client.company_tickers()
|
||||||
|
overrides = await cik_overrides(db)
|
||||||
|
rows = (await db.execute(select(Ticker.id, Ticker.symbol, Ticker.cik))).all()
|
||||||
|
|
||||||
|
result = ResolvedUniverse()
|
||||||
|
for tid, symbol, current_cik in rows:
|
||||||
|
if not symbol:
|
||||||
|
continue
|
||||||
|
sym = normalise_symbol(symbol)
|
||||||
|
cik = overrides.get(sym) or ticker_to_cik.get(sym)
|
||||||
|
if cik is None:
|
||||||
|
continue # ADRs / non-SEC issuers — snapshots simply absent
|
||||||
|
result.symbol_to_cik[sym] = cik
|
||||||
|
result.cik_to_ticker_ids.setdefault(cik, []).append(tid)
|
||||||
|
if current_cik != f"{cik:010d}":
|
||||||
|
result.cik_updates.append((tid, f"{cik:010d}"))
|
||||||
|
logger.info(
|
||||||
|
"resolve_ciks: %d resolved, %d proposed cik updates",
|
||||||
|
len(result.symbol_to_cik),
|
||||||
|
len(result.cik_updates),
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
async def cik_overrides(db) -> dict[str, int]:
|
||||||
|
"""Manual ``{symbol: cik}`` pins from ``SystemSetting[CIK_OVERRIDES_KEY]``.
|
||||||
|
|
||||||
|
A malformed setting must never take the importer down, so anything unparseable
|
||||||
|
is logged and ignored — the run then falls back to company_tickers.json.
|
||||||
|
"""
|
||||||
|
raw = await settings_store.get_value(db, CIK_OVERRIDES_KEY)
|
||||||
|
if not raw:
|
||||||
|
return {}
|
||||||
|
try:
|
||||||
|
loaded = json.loads(raw)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
logger.warning("%s is not valid JSON — ignoring CIK overrides", CIK_OVERRIDES_KEY)
|
||||||
|
return {}
|
||||||
|
if not isinstance(loaded, dict):
|
||||||
|
logger.warning("%s must be a {symbol: cik} object — ignoring", CIK_OVERRIDES_KEY)
|
||||||
|
return {}
|
||||||
|
out: dict[str, int] = {}
|
||||||
|
for symbol, cik in loaded.items():
|
||||||
|
try:
|
||||||
|
out[normalise_symbol(str(symbol))] = int(cik)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
logger.warning("%s: bad entry %r -> %r — ignoring", CIK_OVERRIDES_KEY, symbol, cik)
|
||||||
|
if out:
|
||||||
|
logger.info("resolve_ciks: %d CIK override(s) applied: %s", len(out), sorted(out))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def fetch_sic_updates(
|
||||||
|
client: SecClient, cik_to_ticker_ids: dict[int, Iterable[int]]
|
||||||
|
) -> list[tuple[int, str | None, str | None]]:
|
||||||
|
"""Fetch SIC for each CIK (recent-only submissions, no history shards) and
|
||||||
|
return proposed `(ticker_id, sic, sic_description)` writes. **Read-only** — no
|
||||||
|
DB mutation. Callers pass only the CIKs that need it (e.g. missing a SIC)."""
|
||||||
|
updates: list[tuple[int, str | None, str | None]] = []
|
||||||
|
for cik, ticker_ids in cik_to_ticker_ids.items():
|
||||||
|
sub = await client.submissions(cik, include_history=False)
|
||||||
|
sic = str(sub["sic"]) if sub.get("sic") else None
|
||||||
|
desc = sub.get("sic_description")
|
||||||
|
for tid in ticker_ids:
|
||||||
|
updates.append((tid, sic, desc))
|
||||||
|
return updates
|
||||||
|
|
||||||
|
|
||||||
|
async def apply_ticker_updates(
|
||||||
|
db,
|
||||||
|
resolved: ResolvedUniverse,
|
||||||
|
sic_updates: list[tuple[int, str | None, str | None]] | None = None,
|
||||||
|
) -> dict[str, int]:
|
||||||
|
"""Apply the proposed cik / sic writes. **The only writer** — call inside
|
||||||
|
promote so it commits atomically with the snapshot inserts."""
|
||||||
|
for tid, cik in resolved.cik_updates:
|
||||||
|
await db.execute(update(Ticker).where(Ticker.id == tid).values(cik=cik))
|
||||||
|
for tid, sic, desc in sic_updates or []:
|
||||||
|
await db.execute(
|
||||||
|
update(Ticker).where(Ticker.id == tid).values(sic=sic, sic_description=desc)
|
||||||
|
)
|
||||||
|
return {"cik_updates": len(resolved.cik_updates), "sic_updates": len(sic_updates or [])}
|
||||||
|
|
||||||
|
|
||||||
|
def universe_fingerprint(symbol_to_cik: dict[str, int]) -> str:
|
||||||
|
"""Stable short hash of the tracked symbol->CIK set. Changes whenever a ticker
|
||||||
|
is added/removed or its CIK mapping changes."""
|
||||||
|
canonical = ";".join(f"{sym}:{cik}" for sym, cik in sorted(symbol_to_cik.items()))
|
||||||
|
return hashlib.blake2b(canonical.encode("utf-8"), digest_size=12).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def index_content_hash(index_rows: Iterable[dict]) -> str:
|
||||||
|
"""Order-independent hash of the tracked index accessions consumed this run."""
|
||||||
|
keys = sorted(f"{r['cik']}/{r['accession']}" for r in index_rows)
|
||||||
|
return hashlib.blake2b("|".join(keys).encode("utf-8"), digest_size=12).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def compose_revision(index_date, content_hash: str, symbol_to_cik: dict[str, int]) -> str:
|
||||||
|
"""Composite revision = processed index date + index-content hash + universe
|
||||||
|
fingerprint. Equal across runs ⇒ nothing new ⇒ no_op. Rejects a missing index
|
||||||
|
date rather than emitting a `None:...` revision that could false-match."""
|
||||||
|
if index_date is None:
|
||||||
|
raise ValueError("compose_revision requires a non-null index date")
|
||||||
|
return f"{index_date}:{content_hash}:{universe_fingerprint(symbol_to_cik)}"
|
||||||
@@ -0,0 +1,363 @@
|
|||||||
|
"""Shadow book — the validated strategy, traded automatically.
|
||||||
|
|
||||||
|
The discretionary paper book only ever contains trades the user chose to take,
|
||||||
|
inside a ~20 minute window, on days they were available. The backtest that
|
||||||
|
validated this strategy does none of that: it takes the top-ranked qualified
|
||||||
|
setups up to capacity, every session, with no human involved. That difference
|
||||||
|
makes the manual book unusable as out-of-sample evidence — it measures the
|
||||||
|
strategy *plus* discretion and availability.
|
||||||
|
|
||||||
|
The shadow book closes that gap. It mirrors ``_simulate_portfolio``'s selection
|
||||||
|
rule exactly and shares the manual book's exit policy, so the only difference
|
||||||
|
between the two books is *which* qualified setups get taken.
|
||||||
|
|
||||||
|
Parity is the load-bearing property here. Selection ordering comes from the
|
||||||
|
stored ``strategy_rank`` the scanner already wrote (the same 80/20
|
||||||
|
momentum/vol blend the backtest ranks on) rather than being recomputed, so the
|
||||||
|
two cannot drift apart.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
|
||||||
|
from sqlalchemy import select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.paper_trade import PaperTrade
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.models.trade_setup import TradeSetup
|
||||||
|
from app.models.user import User
|
||||||
|
from app.services import settings_store
|
||||||
|
from app.services.qualification import setup_qualifies
|
||||||
|
from app.services.trade_policy import SHADOW_BOOK, get_reentry_gate_locks
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
KEY_ENABLED = "shadow_book_enabled"
|
||||||
|
KEY_CAPACITY = "shadow_book_capacity"
|
||||||
|
KEY_RISK_PCT = "shadow_book_risk_pct"
|
||||||
|
KEY_START_EQUITY = "shadow_book_start_equity"
|
||||||
|
|
||||||
|
# Matches the validated configuration: 10-position book, 1% fixed-fractional
|
||||||
|
# risk. Start equity is only a sizing base — comparisons are drawn in percent
|
||||||
|
# and R-multiples, never in raw currency.
|
||||||
|
DEFAULT_CAPACITY = 10
|
||||||
|
DEFAULT_RISK_PCT = 1.0
|
||||||
|
DEFAULT_START_EQUITY = 100_000.0
|
||||||
|
|
||||||
|
# Mirrors ``_simulate_portfolio``'s SIM_NOTIONAL_CAP: no single position may
|
||||||
|
# exceed this fraction of equity, and the book never uses margin. Without the
|
||||||
|
# cap, a setup with a tight stop turns 1% risk into a position several times
|
||||||
|
# equity — a leveraged trade the validated strategy would never have taken.
|
||||||
|
NOTIONAL_CAP = 0.20
|
||||||
|
|
||||||
|
# If the last successful scan completed longer ago than this, no scan ran in the
|
||||||
|
# current pipeline pass (scans are daily, ~24h apart), so there is nothing fresh
|
||||||
|
# to trade. Comfortably longer than a scan's own duration, far shorter than the
|
||||||
|
# gap between scans.
|
||||||
|
MAX_SCAN_AGE = timedelta(hours=6)
|
||||||
|
|
||||||
|
|
||||||
|
async def get_config(db: AsyncSession) -> dict:
|
||||||
|
"""Shadow book sizing/capacity config, falling back to validated defaults."""
|
||||||
|
raw = await settings_store.get_map(
|
||||||
|
db, [KEY_CAPACITY, KEY_RISK_PCT, KEY_START_EQUITY]
|
||||||
|
)
|
||||||
|
|
||||||
|
def _num(key: str, default: float, *, minimum: float, maximum: float) -> float:
|
||||||
|
try:
|
||||||
|
value = float(raw.get(key) or default)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return default
|
||||||
|
return max(minimum, min(maximum, value))
|
||||||
|
|
||||||
|
return {
|
||||||
|
"capacity": int(_num(KEY_CAPACITY, DEFAULT_CAPACITY, minimum=1, maximum=100)),
|
||||||
|
"risk_pct": _num(KEY_RISK_PCT, DEFAULT_RISK_PCT, minimum=0.05, maximum=10.0),
|
||||||
|
"start_equity": _num(
|
||||||
|
KEY_START_EQUITY, DEFAULT_START_EQUITY, minimum=1000.0, maximum=1e9
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def is_enabled(db: AsyncSession) -> bool:
|
||||||
|
"""Shadow book writes trades to the live book, so it is opt-in."""
|
||||||
|
value = await settings_store.get_value(db, KEY_ENABLED, "false")
|
||||||
|
return str(value).strip().lower() in {"1", "true", "yes", "on"}
|
||||||
|
|
||||||
|
|
||||||
|
async def equity_and_cash(
|
||||||
|
db: AsyncSession, start_equity: float, positions: list[PaperTrade]
|
||||||
|
) -> tuple[float, float]:
|
||||||
|
"""Marked equity and free cash, matching ``_simulate_portfolio``.
|
||||||
|
|
||||||
|
The simulator sizes from *marked* equity — cash plus open positions at their
|
||||||
|
latest close — and spends from cash, so a book that is fully invested cannot
|
||||||
|
keep buying. Sizing from realized P&L alone would drift away from the
|
||||||
|
backtest as soon as positions were held across a scan.
|
||||||
|
"""
|
||||||
|
from app.services.paper_trade_service import _latest_closes
|
||||||
|
|
||||||
|
result = await db.execute(
|
||||||
|
select(PaperTrade).where(
|
||||||
|
PaperTrade.book == SHADOW_BOOK,
|
||||||
|
PaperTrade.status == "closed",
|
||||||
|
PaperTrade.close_price.is_not(None),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
realized = 0.0
|
||||||
|
for trade in result.scalars():
|
||||||
|
per_share = (
|
||||||
|
trade.close_price - trade.entry_price
|
||||||
|
if trade.direction == "long"
|
||||||
|
else trade.entry_price - trade.close_price
|
||||||
|
)
|
||||||
|
realized += per_share * trade.shares
|
||||||
|
|
||||||
|
open_cost = sum(p.entry_price * p.shares for p in positions)
|
||||||
|
marks = await _latest_closes(db, {p.ticker_id for p in positions})
|
||||||
|
open_value = sum(
|
||||||
|
(marks.get(p.ticker_id) or p.entry_price) * p.shares for p in positions
|
||||||
|
)
|
||||||
|
|
||||||
|
cash = start_equity + realized - open_cost
|
||||||
|
return cash + open_value, cash
|
||||||
|
|
||||||
|
|
||||||
|
def position_shares(
|
||||||
|
equity: float,
|
||||||
|
risk_pct: float,
|
||||||
|
entry: float,
|
||||||
|
stop: float,
|
||||||
|
*,
|
||||||
|
cash_available: float | None = None,
|
||||||
|
) -> float:
|
||||||
|
"""Shares to buy, sized exactly as ``_simulate_portfolio`` sizes them.
|
||||||
|
|
||||||
|
Fixed-fractional risk first, then the two caps the simulator applies: no
|
||||||
|
position may exceed ``NOTIONAL_CAP`` of equity, and the book cannot spend
|
||||||
|
cash it does not have. Dropping either cap lets a tight stop produce a
|
||||||
|
leveraged position and breaks compounding parity with the backtest.
|
||||||
|
"""
|
||||||
|
risk_per_share = abs(entry - stop)
|
||||||
|
if risk_per_share <= 0 or equity <= 0 or entry <= 0:
|
||||||
|
return 0.0
|
||||||
|
|
||||||
|
shares = (equity * risk_pct / 100.0) / risk_per_share
|
||||||
|
shares = min(shares, (equity * NOTIONAL_CAP) / entry)
|
||||||
|
if cash_available is not None:
|
||||||
|
shares = min(shares, max(0.0, cash_available) / entry)
|
||||||
|
# Dust guard, as in the simulator: sub-$1 positions are noise, not trades.
|
||||||
|
return shares if shares * entry >= 1.0 else 0.0
|
||||||
|
|
||||||
|
|
||||||
|
async def _open_positions(db: AsyncSession) -> list[PaperTrade]:
|
||||||
|
result = await db.execute(
|
||||||
|
select(PaperTrade).where(
|
||||||
|
PaperTrade.book == SHADOW_BOOK, PaperTrade.status == "open"
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return list(result.scalars().all())
|
||||||
|
|
||||||
|
|
||||||
|
async def _shadow_user_id(db: AsyncSession) -> int | None:
|
||||||
|
"""Shadow trades are not owned by a person; attach them to the first user."""
|
||||||
|
result = await db.execute(select(User.id).order_by(User.id.asc()).limit(1))
|
||||||
|
row = result.first()
|
||||||
|
return int(row[0]) if row else None
|
||||||
|
|
||||||
|
|
||||||
|
async def _scan_run_to_trade(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
now: datetime,
|
||||||
|
expected_run_id: str | None = None,
|
||||||
|
) -> str | None:
|
||||||
|
"""The run id whose setups the shadow book may act on, or None.
|
||||||
|
|
||||||
|
* ``expected_run_id`` set (pipeline step): the stored run id must match it
|
||||||
|
exactly. This is the airtight guarantee — a scan that was disabled or
|
||||||
|
failed in *this* pipeline never stamped this id, and a concurrent manual
|
||||||
|
scan (a separate APScheduler job, not serialised against the pipeline)
|
||||||
|
stamps its own id even when it finishes last, so neither can be mistaken
|
||||||
|
for the pipeline's own scan. Timestamp order alone cannot tell them apart.
|
||||||
|
* ``expected_run_id`` None (direct Admin trigger): fall back to the freshness
|
||||||
|
window on the last scan's own id. There is no pipeline scan to bind to, so
|
||||||
|
acting on a recent scan is the operator's explicit choice.
|
||||||
|
|
||||||
|
Setups are then selected by ``scan_run_id`` equal to the returned id, so a
|
||||||
|
concurrent scan's rows in the same time window are excluded by identity.
|
||||||
|
"""
|
||||||
|
from app.services import rr_scanner_service as rr
|
||||||
|
|
||||||
|
completed = _parse_dt(
|
||||||
|
await settings_store.get_value(db, rr.KEY_LAST_SCAN_COMPLETED)
|
||||||
|
)
|
||||||
|
run_id = await settings_store.get_value(db, rr.KEY_LAST_SCAN_RUN_ID)
|
||||||
|
if completed is None or not run_id:
|
||||||
|
return None
|
||||||
|
if expected_run_id is not None:
|
||||||
|
return run_id if run_id == expected_run_id else None
|
||||||
|
if now - completed > MAX_SCAN_AGE:
|
||||||
|
return None
|
||||||
|
return run_id
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_dt(raw: str | None) -> datetime | None:
|
||||||
|
if not raw:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return datetime.fromisoformat(raw)
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
async def _todays_qualified_setups(
|
||||||
|
db: AsyncSession,
|
||||||
|
config: dict,
|
||||||
|
*,
|
||||||
|
now: datetime,
|
||||||
|
expected_run_id: str | None = None,
|
||||||
|
) -> list[TradeSetup]:
|
||||||
|
"""Long-only qualified setups from the scan we may act on, best rank first.
|
||||||
|
|
||||||
|
Order matters here, and matches the review's requirement:
|
||||||
|
|
||||||
|
1. Take only rows the matched scan produced (``scan_run_id == run id``). A
|
||||||
|
previous run, or a manual scan overlapping in time, carries a different
|
||||||
|
id and is excluded by identity — not by a time window it could write into.
|
||||||
|
2. Keep long only. The validated strategy is long-only, but the gate permits
|
||||||
|
shorts when ``min_momentum_percentile`` is 0 (a legal admin setting), and
|
||||||
|
the cash accounting assumes longs — so this is enforced here, not left to
|
||||||
|
the gate.
|
||||||
|
3. Deduplicate to the latest row per ticker *before* qualifying, so a newer
|
||||||
|
unqualified row correctly suppresses an older qualified one rather than
|
||||||
|
the reverse.
|
||||||
|
4. Qualify, then rank by ``strategy_rank`` (unranked sort last).
|
||||||
|
"""
|
||||||
|
run_id = await _scan_run_to_trade(db, now=now, expected_run_id=expected_run_id)
|
||||||
|
if run_id is None:
|
||||||
|
return []
|
||||||
|
|
||||||
|
result = await db.execute(
|
||||||
|
select(TradeSetup).where(TradeSetup.scan_run_id == run_id)
|
||||||
|
)
|
||||||
|
rows = [s for s in result.scalars() if (s.direction or "long") == "long"]
|
||||||
|
|
||||||
|
latest: dict[int, TradeSetup] = {}
|
||||||
|
for setup in rows:
|
||||||
|
held = latest.get(setup.ticker_id)
|
||||||
|
if held is None or (setup.detected_at, setup.id) > (
|
||||||
|
held.detected_at,
|
||||||
|
held.id,
|
||||||
|
):
|
||||||
|
latest[setup.ticker_id] = setup
|
||||||
|
|
||||||
|
qualified = [s for s in latest.values() if setup_qualifies(s, config)]
|
||||||
|
return sorted(
|
||||||
|
qualified,
|
||||||
|
key=lambda s: (
|
||||||
|
s.strategy_rank if s.strategy_rank is not None else float("-inf")
|
||||||
|
),
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def open_shadow_positions(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
activation_config: dict,
|
||||||
|
opened_at: datetime | None = None,
|
||||||
|
expected_run_id: str | None = None,
|
||||||
|
) -> dict:
|
||||||
|
"""Fill free capacity with the top-ranked qualified setups.
|
||||||
|
|
||||||
|
Mirrors the backtest: rank the qualified cross-section, walk it top-down,
|
||||||
|
skip anything already held or locked out by post-stop gate-reset, and stop
|
||||||
|
at capacity. Returns a summary for the job log.
|
||||||
|
|
||||||
|
``expected_run_id`` binds this run to the scan that stamped that exact id
|
||||||
|
(the pipeline's own scan), so a scan that failed in this pipeline — or a
|
||||||
|
concurrent manual scan that finished last — cannot substitute for it. See
|
||||||
|
``_scan_run_to_trade``.
|
||||||
|
"""
|
||||||
|
summary = {
|
||||||
|
"opened": 0,
|
||||||
|
"skipped_held": 0,
|
||||||
|
"skipped_locked": 0,
|
||||||
|
"skipped_no_cash": 0,
|
||||||
|
"symbols": [],
|
||||||
|
}
|
||||||
|
config = await get_config(db)
|
||||||
|
|
||||||
|
positions = await _open_positions(db)
|
||||||
|
held = {p.ticker_id for p in positions}
|
||||||
|
free_slots = config["capacity"] - len(positions)
|
||||||
|
if free_slots <= 0:
|
||||||
|
return summary
|
||||||
|
|
||||||
|
user_id = await _shadow_user_id(db)
|
||||||
|
if user_id is None:
|
||||||
|
logger.warning("shadow book skipped: no user to attach trades to")
|
||||||
|
return summary
|
||||||
|
|
||||||
|
locks = await get_reentry_gate_locks(db, book=SHADOW_BOOK)
|
||||||
|
equity, cash = await equity_and_cash(db, config["start_equity"], positions)
|
||||||
|
timestamp = opened_at or datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
candidates = await _todays_qualified_setups(
|
||||||
|
db, activation_config, now=timestamp, expected_run_id=expected_run_id
|
||||||
|
)
|
||||||
|
for setup in candidates:
|
||||||
|
if free_slots <= 0:
|
||||||
|
break
|
||||||
|
if setup.ticker_id in held:
|
||||||
|
summary["skipped_held"] += 1
|
||||||
|
continue
|
||||||
|
if setup.ticker_id in locks:
|
||||||
|
summary["skipped_locked"] += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
entry = float(setup.entry_price or 0.0)
|
||||||
|
stop = float(setup.stop_loss or 0.0)
|
||||||
|
shares = position_shares(
|
||||||
|
equity, config["risk_pct"], entry, stop, cash_available=cash
|
||||||
|
)
|
||||||
|
if shares <= 0:
|
||||||
|
summary["skipped_no_cash"] += 1
|
||||||
|
continue
|
||||||
|
cash -= shares * entry
|
||||||
|
|
||||||
|
db.add(
|
||||||
|
PaperTrade(
|
||||||
|
user_id=user_id,
|
||||||
|
ticker_id=setup.ticker_id,
|
||||||
|
direction=setup.direction,
|
||||||
|
entry_price=entry,
|
||||||
|
shares=shares,
|
||||||
|
stop_loss=stop,
|
||||||
|
target=float(setup.target or 0.0),
|
||||||
|
status="open",
|
||||||
|
opened_at=timestamp,
|
||||||
|
fill_mode="near_close",
|
||||||
|
book=SHADOW_BOOK,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
held.add(setup.ticker_id)
|
||||||
|
free_slots -= 1
|
||||||
|
summary["opened"] += 1
|
||||||
|
summary["symbols"].append(setup.ticker_id)
|
||||||
|
|
||||||
|
if summary["opened"]:
|
||||||
|
await db.commit()
|
||||||
|
return summary
|
||||||
|
|
||||||
|
|
||||||
|
async def symbols_for(db: AsyncSession, ticker_ids: list[int]) -> list[str]:
|
||||||
|
"""Resolve ticker ids to symbols for logging."""
|
||||||
|
if not ticker_ids:
|
||||||
|
return []
|
||||||
|
result = await db.execute(select(Ticker.symbol).where(Ticker.id.in_(ticker_ids)))
|
||||||
|
return [row[0] for row in result.all()]
|
||||||
+586
-77
@@ -1,12 +1,15 @@
|
|||||||
"""S/R Detector service.
|
"""S/R Detector service.
|
||||||
|
|
||||||
Detects support/resistance levels from Volume Profile (HVN/LVN) and
|
Detects support/resistance levels from Volume Profile (POC/VA/HVN peaks)
|
||||||
Pivot Points (swing highs/lows), assigns strength scores, merges nearby
|
and Pivot Points (prominent swing highs/lows), plus light psychological
|
||||||
levels, tags as support/resistance, and persists to DB.
|
round numbers. Scores by rejection-weighted recent touches, merges nearby
|
||||||
|
levels with ATR-adaptive tolerance, tags support/resistance, caps count,
|
||||||
|
and persists to DB.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
|
||||||
from sqlalchemy import delete, select
|
from sqlalchemy import delete, select
|
||||||
@@ -17,12 +20,43 @@ from app.models.sr_level import SRLevel
|
|||||||
from app.models.ticker import Ticker
|
from app.models.ticker import Ticker
|
||||||
from app.services.indicator_service import (
|
from app.services.indicator_service import (
|
||||||
_extract_ohlcv,
|
_extract_ohlcv,
|
||||||
|
compute_atr,
|
||||||
compute_pivot_points,
|
compute_pivot_points,
|
||||||
compute_volume_profile,
|
compute_volume_profile,
|
||||||
)
|
)
|
||||||
from app.services.price_service import query_ohlcv
|
from app.services.price_service import query_ohlcv
|
||||||
|
|
||||||
DEFAULT_TOLERANCE = 0.005 # 0.5%
|
# ---------------------------------------------------------------------------
|
||||||
|
# Tunable constants (keep detection pure / deterministic)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
DEFAULT_TOLERANCE = 0.005 # fallback when ATR unavailable; also API legacy default
|
||||||
|
|
||||||
|
VP_LOOKBACK = 252
|
||||||
|
TOUCH_LOOKBACK = 252
|
||||||
|
PIVOT_LOOKBACK = 504
|
||||||
|
PIVOT_PROMINENCE_ATR = 0.75
|
||||||
|
PIVOT_PROMINENCE_PCT = 0.006
|
||||||
|
|
||||||
|
MERGE_TOL_ATR_MULT = 0.35
|
||||||
|
MERGE_TOL_MIN = 0.004 # 0.4%
|
||||||
|
MERGE_TOL_MAX = 0.015 # 1.5%
|
||||||
|
|
||||||
|
MAX_LEVELS = 16
|
||||||
|
STRENGTH_HALF_LIFE = 60 # bars
|
||||||
|
# Raw respect score is soft-mapped to 0–100 (see _raw_to_strength).
|
||||||
|
STRENGTH_SCALE = 8.0
|
||||||
|
STRENGTH_SOFT_K = 35.0 # higher → slower approach to 100
|
||||||
|
|
||||||
|
ROUND_NUMBER_RANGE = 0.15 # ±15% of spot
|
||||||
|
ROUND_NUMBER_MAX = 8
|
||||||
|
|
||||||
|
# Base strength seed before touch scoring (method priors)
|
||||||
|
_METHOD_BASE_STRENGTH = {
|
||||||
|
"volume_profile": 12,
|
||||||
|
"pivot_point": 8,
|
||||||
|
"round_number": 4,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
async def _get_ticker(db: AsyncSession, symbol: str) -> Ticker:
|
async def _get_ticker(db: AsyncSession, symbol: str) -> Ticker:
|
||||||
@@ -35,36 +69,235 @@ async def _get_ticker(db: AsyncSession, symbol: str) -> Ticker:
|
|||||||
return ticker
|
return ticker
|
||||||
|
|
||||||
|
|
||||||
def _count_price_touches(
|
def _slice_tail(
|
||||||
|
highs: list[float],
|
||||||
|
lows: list[float],
|
||||||
|
closes: list[float],
|
||||||
|
volumes: list[int],
|
||||||
|
lookback: int,
|
||||||
|
) -> tuple[list[float], list[float], list[float], list[int]]:
|
||||||
|
"""Return the last *lookback* bars (or all if shorter)."""
|
||||||
|
n = len(closes)
|
||||||
|
if lookback <= 0 or n <= lookback:
|
||||||
|
return highs, lows, closes, volumes
|
||||||
|
start = n - lookback
|
||||||
|
return highs[start:], lows[start:], closes[start:], volumes[start:]
|
||||||
|
|
||||||
|
|
||||||
|
def _atr_pct(
|
||||||
|
highs: list[float],
|
||||||
|
lows: list[float],
|
||||||
|
closes: list[float],
|
||||||
|
) -> float | None:
|
||||||
|
"""ATR as a fraction of last close, or None if insufficient data."""
|
||||||
|
try:
|
||||||
|
result = compute_atr(highs, lows, closes)
|
||||||
|
except ValidationError:
|
||||||
|
return None
|
||||||
|
atr = result["atr"]
|
||||||
|
last = closes[-1]
|
||||||
|
if last == 0:
|
||||||
|
return None
|
||||||
|
return atr / last
|
||||||
|
|
||||||
|
|
||||||
|
def _merge_tolerance(
|
||||||
|
highs: list[float],
|
||||||
|
lows: list[float],
|
||||||
|
closes: list[float],
|
||||||
|
tolerance: float | None,
|
||||||
|
) -> float:
|
||||||
|
"""Resolve merge tolerance: explicit value or ATR-adaptive clamp."""
|
||||||
|
if tolerance is not None:
|
||||||
|
return tolerance
|
||||||
|
atr_frac = _atr_pct(highs, lows, closes)
|
||||||
|
if atr_frac is None:
|
||||||
|
return DEFAULT_TOLERANCE
|
||||||
|
return max(MERGE_TOL_MIN, min(MERGE_TOL_MAX, MERGE_TOL_ATR_MULT * atr_frac))
|
||||||
|
|
||||||
|
|
||||||
|
def _bar_respect_weight(
|
||||||
|
price_level: float,
|
||||||
|
high: float,
|
||||||
|
low: float,
|
||||||
|
close: float,
|
||||||
|
prev_close: float | None,
|
||||||
|
tolerance: float,
|
||||||
|
) -> float:
|
||||||
|
"""Weight for how much a bar *respects* a level (not mere occupancy).
|
||||||
|
|
||||||
|
Only bars whose high/low **probes near the level** and closes away from
|
||||||
|
that extreme count as rejections. Full-range pass-throughs score near zero.
|
||||||
|
"""
|
||||||
|
tol = price_level * tolerance if price_level != 0 else tolerance
|
||||||
|
if tol <= 0:
|
||||||
|
tol = abs(price_level) * DEFAULT_TOLERANCE if price_level else DEFAULT_TOLERANCE
|
||||||
|
# Tight probe band: ~0.4% of price (capped), not 2× merge tolerance
|
||||||
|
band = min(max(abs(price_level) * 0.004, tol * 0.35), abs(price_level) * 0.008)
|
||||||
|
if band <= 0:
|
||||||
|
band = abs(price_level) * 0.004 if price_level else 0.01
|
||||||
|
|
||||||
|
if high + band < price_level or low - band > price_level:
|
||||||
|
return 0.0
|
||||||
|
|
||||||
|
bar_range = high - low
|
||||||
|
# Support test: low probes near level, close recovers above
|
||||||
|
support_test = abs(low - price_level) <= band and close > price_level
|
||||||
|
if support_test and bar_range > 0:
|
||||||
|
support_test = (close - low) >= 0.25 * bar_range
|
||||||
|
# Resistance test: high probes near level, close rejects below
|
||||||
|
resist_test = abs(high - price_level) <= band and close < price_level
|
||||||
|
if resist_test and bar_range > 0:
|
||||||
|
resist_test = (high - close) >= 0.25 * bar_range
|
||||||
|
|
||||||
|
if support_test or resist_test:
|
||||||
|
return 1.0
|
||||||
|
|
||||||
|
# Clear directional pass-through — barely counts
|
||||||
|
if (
|
||||||
|
prev_close is not None
|
||||||
|
and (prev_close - price_level) * (close - price_level) < 0
|
||||||
|
and low < price_level - tol
|
||||||
|
and high > price_level + tol
|
||||||
|
):
|
||||||
|
return 0.1
|
||||||
|
|
||||||
|
return 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def _raw_to_strength(raw: float) -> int:
|
||||||
|
"""Map unbounded raw score to 0–100 with soft saturation (no hard pin)."""
|
||||||
|
if raw <= 0:
|
||||||
|
return 0
|
||||||
|
# 1 - e^(-raw/k): raw=k → ~63, 2k → ~86, 3k → ~95
|
||||||
|
return max(0, min(100, int(round(100.0 * (1.0 - math.exp(-raw / STRENGTH_SOFT_K))))))
|
||||||
|
|
||||||
|
|
||||||
|
def _respect_evidence(
|
||||||
price_level: float,
|
price_level: float,
|
||||||
highs: list[float],
|
highs: list[float],
|
||||||
lows: list[float],
|
lows: list[float],
|
||||||
closes: list[float],
|
closes: list[float],
|
||||||
tolerance: float = DEFAULT_TOLERANCE,
|
tolerance: float = DEFAULT_TOLERANCE,
|
||||||
) -> int:
|
base: int = 0,
|
||||||
"""Count how many bars touched/respected a price level within tolerance."""
|
half_life: float = STRENGTH_HALF_LIFE,
|
||||||
count = 0
|
lookback: int = TOUCH_LOOKBACK,
|
||||||
tol = price_level * tolerance if price_level != 0 else tolerance
|
cooldown: int = 3,
|
||||||
for i in range(len(closes)):
|
) -> dict[str, float | int | None]:
|
||||||
# A bar "touches" the level if the level is within the bar's range
|
"""Return rejection evidence and its soft-mapped strength.
|
||||||
# (within tolerance)
|
|
||||||
if lows[i] - tol <= price_level <= highs[i] + tol:
|
|
||||||
count += 1
|
|
||||||
return count
|
|
||||||
|
|
||||||
|
*cooldown* bars after a full rejection are ignored so multi-day chop at a
|
||||||
def _strength_from_touches(touches: int, total_bars: int) -> int:
|
level counts as one test cluster, not N identical rejections.
|
||||||
"""Convert touch count to a 0-100 strength score.
|
|
||||||
|
|
||||||
More touches relative to total bars = higher strength.
|
|
||||||
Cap at 100.
|
|
||||||
"""
|
"""
|
||||||
if total_bars == 0:
|
n = len(closes)
|
||||||
return 0
|
if n == 0:
|
||||||
# Scale: each touch contributes proportionally, with a multiplier
|
return {
|
||||||
# so that a level touched ~20% of bars gets score ~100
|
"strength": _raw_to_strength(float(base)),
|
||||||
raw = (touches / total_bars) * 500.0
|
"rejection_count": 0,
|
||||||
return max(0, min(100, int(round(raw))))
|
"last_rejection_age": None,
|
||||||
|
"weighted_respects": 0.0,
|
||||||
|
}
|
||||||
|
|
||||||
|
start = max(0, n - lookback) if lookback > 0 else 0
|
||||||
|
weighted = 0.0
|
||||||
|
rejection_count = 0
|
||||||
|
last_rejection_age: int | None = None
|
||||||
|
next_ok = start
|
||||||
|
for i in range(start, n):
|
||||||
|
age = n - 1 - i
|
||||||
|
decay = 0.5 ** (age / half_life) if half_life > 0 else 1.0
|
||||||
|
prev = closes[i - 1] if i > 0 else None
|
||||||
|
w = _bar_respect_weight(
|
||||||
|
price_level, highs[i], lows[i], closes[i], prev, tolerance
|
||||||
|
)
|
||||||
|
if w >= 0.9:
|
||||||
|
if i < next_ok:
|
||||||
|
continue
|
||||||
|
weighted += decay * w
|
||||||
|
rejection_count += 1
|
||||||
|
if last_rejection_age is None or age < last_rejection_age:
|
||||||
|
last_rejection_age = age
|
||||||
|
next_ok = i + max(cooldown, 1)
|
||||||
|
elif w > 0:
|
||||||
|
weighted += decay * w
|
||||||
|
|
||||||
|
raw = float(base) + weighted * STRENGTH_SCALE
|
||||||
|
return {
|
||||||
|
"strength": _raw_to_strength(raw),
|
||||||
|
"rejection_count": rejection_count,
|
||||||
|
"last_rejection_age": last_rejection_age,
|
||||||
|
"weighted_respects": round(weighted, 6),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _strength_from_respects(
|
||||||
|
price_level: float,
|
||||||
|
highs: list[float],
|
||||||
|
lows: list[float],
|
||||||
|
closes: list[float],
|
||||||
|
tolerance: float = DEFAULT_TOLERANCE,
|
||||||
|
base: int = 0,
|
||||||
|
half_life: float = STRENGTH_HALF_LIFE,
|
||||||
|
lookback: int = TOUCH_LOOKBACK,
|
||||||
|
cooldown: int = 3,
|
||||||
|
) -> int:
|
||||||
|
"""Compatibility wrapper returning only the evidence-derived strength."""
|
||||||
|
return int(_respect_evidence(
|
||||||
|
price_level,
|
||||||
|
highs,
|
||||||
|
lows,
|
||||||
|
closes,
|
||||||
|
tolerance,
|
||||||
|
base,
|
||||||
|
half_life,
|
||||||
|
lookback,
|
||||||
|
cooldown,
|
||||||
|
)["strength"])
|
||||||
|
|
||||||
|
|
||||||
|
def _round_number_candidates(
|
||||||
|
current_price: float,
|
||||||
|
range_pct: float = ROUND_NUMBER_RANGE,
|
||||||
|
max_count: int = ROUND_NUMBER_MAX,
|
||||||
|
) -> list[float]:
|
||||||
|
"""Psychological round levels near spot (cheap order-magnet candidates)."""
|
||||||
|
if current_price <= 0:
|
||||||
|
return []
|
||||||
|
|
||||||
|
if current_price < 5:
|
||||||
|
steps = [0.5, 1.0]
|
||||||
|
elif current_price < 20:
|
||||||
|
steps = [1.0, 5.0]
|
||||||
|
elif current_price < 100:
|
||||||
|
steps = [5.0, 10.0, 25.0]
|
||||||
|
elif current_price < 500:
|
||||||
|
steps = [10.0, 25.0, 50.0, 100.0]
|
||||||
|
else:
|
||||||
|
steps = [25.0, 50.0, 100.0, 250.0]
|
||||||
|
|
||||||
|
lo = current_price * (1.0 - range_pct)
|
||||||
|
hi = current_price * (1.0 + range_pct)
|
||||||
|
found: set[float] = set()
|
||||||
|
|
||||||
|
for step in steps:
|
||||||
|
if step <= 0:
|
||||||
|
continue
|
||||||
|
# Start at first multiple at or below lo
|
||||||
|
k = math.floor(lo / step)
|
||||||
|
while True:
|
||||||
|
level = round(k * step, 4)
|
||||||
|
if level > hi + step:
|
||||||
|
break
|
||||||
|
if lo <= level <= hi and level > 0:
|
||||||
|
# Skip levels that are essentially current price
|
||||||
|
if abs(level - current_price) / current_price > 0.001:
|
||||||
|
found.add(level)
|
||||||
|
k += 1
|
||||||
|
if k > 1_000_000: # safety
|
||||||
|
break
|
||||||
|
|
||||||
|
ordered = sorted(found, key=lambda p: abs(p - current_price))
|
||||||
|
return ordered[:max_count]
|
||||||
|
|
||||||
|
|
||||||
def _extract_candidate_levels(
|
def _extract_candidate_levels(
|
||||||
@@ -73,55 +306,187 @@ def _extract_candidate_levels(
|
|||||||
closes: list[float],
|
closes: list[float],
|
||||||
volumes: list[int],
|
volumes: list[int],
|
||||||
) -> list[tuple[float, str]]:
|
) -> list[tuple[float, str]]:
|
||||||
"""Extract candidate S/R levels from Volume Profile and Pivot Points.
|
"""Extract candidate S/R levels from VP nodes, prominent pivots, rounds.
|
||||||
|
|
||||||
Returns list of (price_level, detection_method) tuples.
|
Returns list of (price_level, detection_method) tuples.
|
||||||
"""
|
"""
|
||||||
candidates: list[tuple[float, str]] = []
|
candidates: list[tuple[float, str]] = []
|
||||||
|
if not closes:
|
||||||
|
return candidates
|
||||||
|
|
||||||
# Volume Profile: HVN and LVN as candidate levels
|
current_price = closes[-1]
|
||||||
|
|
||||||
|
# --- Volume profile on recent window ---
|
||||||
|
vp_h, vp_l, vp_c, vp_v = _slice_tail(
|
||||||
|
highs, lows, closes, volumes, VP_LOOKBACK
|
||||||
|
)
|
||||||
try:
|
try:
|
||||||
vp = compute_volume_profile(highs, lows, closes, volumes)
|
vp = compute_volume_profile(vp_h, vp_l, vp_c, vp_v)
|
||||||
|
# Structural VP levels: POC, value-area edges, local HVN peaks.
|
||||||
|
# LVN intentionally omitted (rejection voids ≠ support/resistance lines).
|
||||||
|
for key in ("poc", "value_area_low", "value_area_high"):
|
||||||
|
price = vp.get(key)
|
||||||
|
if price is not None and price > 0:
|
||||||
|
candidates.append((float(price), "volume_profile"))
|
||||||
for price in vp.get("hvn", []):
|
for price in vp.get("hvn", []):
|
||||||
candidates.append((price, "volume_profile"))
|
candidates.append((float(price), "volume_profile"))
|
||||||
for price in vp.get("lvn", []):
|
|
||||||
candidates.append((price, "volume_profile"))
|
|
||||||
except ValidationError:
|
except ValidationError:
|
||||||
pass # Not enough data for volume profile
|
pass
|
||||||
|
|
||||||
|
# --- Prominent pivots on pivot lookback ---
|
||||||
|
p_h, p_l, p_c, _ = _slice_tail(highs, lows, closes, volumes, PIVOT_LOOKBACK)
|
||||||
|
atr_frac = _atr_pct(p_h, p_l, p_c)
|
||||||
|
last = p_c[-1] if p_c else current_price
|
||||||
|
if atr_frac is not None and last > 0:
|
||||||
|
prominence = max(PIVOT_PROMINENCE_ATR * atr_frac * last, PIVOT_PROMINENCE_PCT * last)
|
||||||
|
else:
|
||||||
|
prominence = PIVOT_PROMINENCE_PCT * last if last > 0 else None
|
||||||
|
|
||||||
# Pivot Points: swing highs and lows
|
|
||||||
try:
|
try:
|
||||||
pp = compute_pivot_points(highs, lows, closes)
|
pp = compute_pivot_points(p_h, p_l, p_c, min_prominence=prominence)
|
||||||
for price in pp.get("swing_highs", []):
|
for price in pp.get("swing_highs", []):
|
||||||
candidates.append((price, "pivot_point"))
|
candidates.append((float(price), "pivot_point"))
|
||||||
for price in pp.get("swing_lows", []):
|
for price in pp.get("swing_lows", []):
|
||||||
candidates.append((price, "pivot_point"))
|
candidates.append((float(price), "pivot_point"))
|
||||||
except ValidationError:
|
except ValidationError:
|
||||||
pass # Not enough data for pivot points
|
pass
|
||||||
|
|
||||||
|
# --- Psychological round numbers near spot ---
|
||||||
|
for price in _round_number_candidates(current_price):
|
||||||
|
candidates.append((price, "round_number"))
|
||||||
|
|
||||||
return candidates
|
return candidates
|
||||||
|
|
||||||
|
|
||||||
|
def _gate_target_range_centers(
|
||||||
|
highs: list[float],
|
||||||
|
lows: list[float],
|
||||||
|
closes: list[float],
|
||||||
|
num_bins: int = 20,
|
||||||
|
) -> list[float]:
|
||||||
|
"""Return the evenly spaced price proposals used by the production GTL."""
|
||||||
|
if len(closes) < 20:
|
||||||
|
raise ValidationError(
|
||||||
|
f"Range grid requires at least 20 bars, got {len(closes)}"
|
||||||
|
)
|
||||||
|
price_min = min(lows)
|
||||||
|
price_max = max(highs)
|
||||||
|
if price_max == price_min:
|
||||||
|
price_max = price_min + 1.0
|
||||||
|
bin_width = (price_max - price_min) / num_bins
|
||||||
|
return [
|
||||||
|
round(price_min + (i + 0.5) * bin_width, 4)
|
||||||
|
for i in range(num_bins)
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def detect_gate_target_ladder(
|
||||||
|
highs: list[float],
|
||||||
|
lows: list[float],
|
||||||
|
closes: list[float],
|
||||||
|
tolerance: float = DEFAULT_TOLERANCE,
|
||||||
|
) -> list[dict]:
|
||||||
|
"""Build the scanner's internal, volume-free target proposal ladder.
|
||||||
|
|
||||||
|
This is intentionally not human-facing support/resistance. It builds the
|
||||||
|
production gate's broad 20-bin range grid, adds unfiltered pivots, scores
|
||||||
|
historical price traffic, and merges nearby proposals. The returned levels
|
||||||
|
are transient and must not be persisted as chart S/R.
|
||||||
|
"""
|
||||||
|
if not closes:
|
||||||
|
return []
|
||||||
|
|
||||||
|
candidates: list[tuple[float, str]] = []
|
||||||
|
try:
|
||||||
|
candidates.extend(
|
||||||
|
(float(price), "range_grid")
|
||||||
|
for price in _gate_target_range_centers(highs, lows, closes)
|
||||||
|
)
|
||||||
|
except ValidationError:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
pivots = compute_pivot_points(highs, lows, closes)
|
||||||
|
candidates.extend(
|
||||||
|
(float(price), "pivot_point")
|
||||||
|
for price in pivots.get("swing_highs", []) + pivots.get("swing_lows", [])
|
||||||
|
)
|
||||||
|
except ValidationError:
|
||||||
|
pass
|
||||||
|
if not candidates:
|
||||||
|
return []
|
||||||
|
|
||||||
|
total_bars = len(closes)
|
||||||
|
raw: list[dict] = []
|
||||||
|
for price, method in candidates:
|
||||||
|
tol = price * tolerance if price != 0 else tolerance
|
||||||
|
touches = sum(
|
||||||
|
1 for low, high in zip(lows, highs, strict=False)
|
||||||
|
if low - tol <= price <= high + tol
|
||||||
|
)
|
||||||
|
strength = max(0, min(100, int(round((touches / total_bars) * 500.0))))
|
||||||
|
raw.append({
|
||||||
|
"price_level": price,
|
||||||
|
"strength": strength,
|
||||||
|
"detection_method": method,
|
||||||
|
"type": "",
|
||||||
|
"sources": [method],
|
||||||
|
"rejection_count": touches,
|
||||||
|
"last_rejection_age": None,
|
||||||
|
"weighted_respects": float(touches),
|
||||||
|
})
|
||||||
|
|
||||||
|
merged: list[dict] = []
|
||||||
|
for level in sorted(raw, key=lambda row: row["price_level"]):
|
||||||
|
if not merged:
|
||||||
|
merged.append(dict(level))
|
||||||
|
continue
|
||||||
|
last = merged[-1]
|
||||||
|
ref = last["price_level"]
|
||||||
|
tol = ref * tolerance if ref != 0 else tolerance
|
||||||
|
if abs(level["price_level"] - ref) > tol:
|
||||||
|
merged.append(dict(level))
|
||||||
|
continue
|
||||||
|
last["price_level"] = round(
|
||||||
|
(last["price_level"] + level["price_level"]) / 2.0, 4
|
||||||
|
)
|
||||||
|
last["strength"] = min(100, last["strength"] + level["strength"])
|
||||||
|
sources = set(last.get("sources") or [last["detection_method"]])
|
||||||
|
sources |= set(level.get("sources") or [level["detection_method"]])
|
||||||
|
last["sources"] = sorted(sources)
|
||||||
|
last["detection_method"] = (
|
||||||
|
next(iter(sources)) if len(sources) == 1 else "merged"
|
||||||
|
)
|
||||||
|
last["rejection_count"] = max(
|
||||||
|
int(last.get("rejection_count", 0)),
|
||||||
|
int(level.get("rejection_count", 0)),
|
||||||
|
)
|
||||||
|
|
||||||
|
_tag_levels(merged, closes[-1])
|
||||||
|
merged.sort(key=lambda row: row["strength"], reverse=True)
|
||||||
|
return merged
|
||||||
|
|
||||||
|
|
||||||
def _merge_levels(
|
def _merge_levels(
|
||||||
levels: list[dict],
|
levels: list[dict],
|
||||||
tolerance: float = DEFAULT_TOLERANCE,
|
tolerance: float = DEFAULT_TOLERANCE,
|
||||||
) -> list[dict]:
|
) -> list[dict]:
|
||||||
"""Merge levels within tolerance into consolidated levels.
|
"""Merge levels within tolerance into consolidated levels.
|
||||||
|
|
||||||
Levels from different methods within tolerance are merged.
|
Strength combines via max + partial min (avoids instant saturation) with
|
||||||
Merged levels combine strength scores (capped at 100) and get
|
a confluence bonus when detection methods differ. Price is strength-weighted.
|
||||||
detection_method = "merged".
|
|
||||||
"""
|
"""
|
||||||
if not levels:
|
if not levels:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
# Sort by price
|
|
||||||
sorted_levels = sorted(levels, key=lambda x: x["price_level"])
|
sorted_levels = sorted(levels, key=lambda x: x["price_level"])
|
||||||
merged: list[dict] = []
|
merged: list[dict] = []
|
||||||
|
|
||||||
for level in sorted_levels:
|
for level in sorted_levels:
|
||||||
if not merged:
|
if not merged:
|
||||||
merged.append(dict(level))
|
entry = dict(level)
|
||||||
|
sources = level.get("sources") or [level["detection_method"]]
|
||||||
|
entry["sources"] = sorted(set(sources))
|
||||||
|
merged.append(entry)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
last = merged[-1]
|
last = merged[-1]
|
||||||
@@ -129,19 +494,52 @@ def _merge_levels(
|
|||||||
tol = ref_price * tolerance if ref_price != 0 else tolerance
|
tol = ref_price * tolerance if ref_price != 0 else tolerance
|
||||||
|
|
||||||
if abs(level["price_level"] - ref_price) <= tol:
|
if abs(level["price_level"] - ref_price) <= tol:
|
||||||
# Merge: average price, combine strength, mark as merged
|
s1 = last["strength"]
|
||||||
combined_strength = min(100, last["strength"] + level["strength"])
|
s2 = level["strength"]
|
||||||
avg_price = (last["price_level"] + level["price_level"]) / 2.0
|
# Soft combine — avoid merge math pinning everything at 100
|
||||||
method = (
|
combined = int(round(0.85 * max(s1, s2) + 0.15 * min(s1, s2)))
|
||||||
"merged"
|
sources = set(last.get("sources") or [last["detection_method"]])
|
||||||
if last["detection_method"] != level["detection_method"]
|
sources |= set(level.get("sources") or [level["detection_method"]])
|
||||||
else last["detection_method"]
|
if len(sources) > 1:
|
||||||
)
|
combined = min(100, combined + 5)
|
||||||
|
else:
|
||||||
|
combined = min(100, combined)
|
||||||
|
|
||||||
|
w1, w2 = max(s1, 1), max(s2, 1)
|
||||||
|
avg_price = (last["price_level"] * w1 + level["price_level"] * w2) / (w1 + w2)
|
||||||
|
|
||||||
|
if len(sources) == 1:
|
||||||
|
method = next(iter(sources))
|
||||||
|
else:
|
||||||
|
method = "merged"
|
||||||
|
|
||||||
last["price_level"] = round(avg_price, 4)
|
last["price_level"] = round(avg_price, 4)
|
||||||
last["strength"] = combined_strength
|
last["strength"] = combined
|
||||||
last["detection_method"] = method
|
last["detection_method"] = method
|
||||||
|
last["sources"] = sorted(sources)
|
||||||
|
# Nearby candidates often describe the same price reaction, so do
|
||||||
|
# not add their rejection counts and double-count one market event.
|
||||||
|
last["rejection_count"] = max(
|
||||||
|
int(last.get("rejection_count", 0)),
|
||||||
|
int(level.get("rejection_count", 0)),
|
||||||
|
)
|
||||||
|
ages = [
|
||||||
|
age for age in (
|
||||||
|
last.get("last_rejection_age"),
|
||||||
|
level.get("last_rejection_age"),
|
||||||
|
)
|
||||||
|
if age is not None
|
||||||
|
]
|
||||||
|
last["last_rejection_age"] = min(ages) if ages else None
|
||||||
|
last["weighted_respects"] = max(
|
||||||
|
float(last.get("weighted_respects", 0.0)),
|
||||||
|
float(level.get("weighted_respects", 0.0)),
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
merged.append(dict(level))
|
entry = dict(level)
|
||||||
|
sources = level.get("sources") or [level["detection_method"]]
|
||||||
|
entry["sources"] = sorted(set(sources))
|
||||||
|
merged.append(entry)
|
||||||
|
|
||||||
return merged
|
return merged
|
||||||
|
|
||||||
@@ -159,14 +557,66 @@ def _tag_levels(
|
|||||||
return levels
|
return levels
|
||||||
|
|
||||||
|
|
||||||
|
def _cap_levels(
|
||||||
|
levels: list[dict],
|
||||||
|
max_levels: int = MAX_LEVELS,
|
||||||
|
) -> list[dict]:
|
||||||
|
"""Keep up to *max_levels* levels, interleaving support/resistance by strength."""
|
||||||
|
if max_levels <= 0 or len(levels) <= max_levels:
|
||||||
|
return levels
|
||||||
|
|
||||||
|
support = sorted(
|
||||||
|
[lvl for lvl in levels if lvl.get("type") == "support"],
|
||||||
|
key=lambda x: x["strength"],
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
resistance = sorted(
|
||||||
|
[lvl for lvl in levels if lvl.get("type") != "support"],
|
||||||
|
key=lambda x: x["strength"],
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
selected: list[dict] = []
|
||||||
|
si, ri = 0, 0
|
||||||
|
pick_support = True
|
||||||
|
while len(selected) < max_levels and (si < len(support) or ri < len(resistance)):
|
||||||
|
if pick_support:
|
||||||
|
if si < len(support):
|
||||||
|
selected.append(support[si])
|
||||||
|
si += 1
|
||||||
|
elif ri < len(resistance):
|
||||||
|
selected.append(resistance[ri])
|
||||||
|
ri += 1
|
||||||
|
else:
|
||||||
|
if ri < len(resistance):
|
||||||
|
selected.append(resistance[ri])
|
||||||
|
ri += 1
|
||||||
|
elif si < len(support):
|
||||||
|
selected.append(support[si])
|
||||||
|
si += 1
|
||||||
|
pick_support = not pick_support
|
||||||
|
|
||||||
|
selected.sort(key=lambda x: x["strength"], reverse=True)
|
||||||
|
return selected
|
||||||
|
|
||||||
|
|
||||||
def detect_sr_levels(
|
def detect_sr_levels(
|
||||||
highs: list[float],
|
highs: list[float],
|
||||||
lows: list[float],
|
lows: list[float],
|
||||||
closes: list[float],
|
closes: list[float],
|
||||||
volumes: list[int],
|
volumes: list[int],
|
||||||
tolerance: float = DEFAULT_TOLERANCE,
|
tolerance: float | None = None,
|
||||||
|
max_levels: int = MAX_LEVELS,
|
||||||
) -> list[dict]:
|
) -> list[dict]:
|
||||||
"""Detect, score, merge, and tag S/R levels from OHLCV data.
|
"""Detect, score, merge, tag, and cap S/R levels from OHLCV data.
|
||||||
|
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
tolerance:
|
||||||
|
Relative merge tolerance. ``None`` (default) uses ATR-adaptive
|
||||||
|
tolerance clamped to [0.4%, 1.5%]. Pass an explicit fraction to override.
|
||||||
|
max_levels:
|
||||||
|
Hard cap after merge (balanced support/resistance). 0 = no cap.
|
||||||
|
|
||||||
Returns list of dicts with keys: price_level, type, strength,
|
Returns list of dicts with keys: price_level, type, strength,
|
||||||
detection_method — sorted by strength descending.
|
detection_method — sorted by strength descending.
|
||||||
@@ -178,37 +628,42 @@ def detect_sr_levels(
|
|||||||
if not candidates:
|
if not candidates:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
total_bars = len(closes)
|
|
||||||
current_price = closes[-1]
|
current_price = closes[-1]
|
||||||
|
merge_tol = _merge_tolerance(highs, lows, closes, tolerance)
|
||||||
|
# Touch tolerance for strength: use merge tol (same price scale)
|
||||||
|
touch_tol = merge_tol
|
||||||
|
|
||||||
# Build level dicts with strength scores
|
# Score each candidate on recent rejection-weighted touches
|
||||||
raw_levels: list[dict] = []
|
raw_levels: list[dict] = []
|
||||||
for price, method in candidates:
|
for price, method in candidates:
|
||||||
touches = _count_price_touches(price, highs, lows, closes, tolerance)
|
base = _METHOD_BASE_STRENGTH.get(method, 0)
|
||||||
strength = _strength_from_touches(touches, total_bars)
|
evidence = _respect_evidence(
|
||||||
|
price, highs, lows, closes, touch_tol, base=base
|
||||||
|
)
|
||||||
raw_levels.append({
|
raw_levels.append({
|
||||||
"price_level": price,
|
"price_level": price,
|
||||||
"strength": strength,
|
"strength": int(evidence["strength"]),
|
||||||
"detection_method": method,
|
"detection_method": method,
|
||||||
"type": "", # will be tagged after merge
|
"type": "",
|
||||||
|
"sources": [method],
|
||||||
|
"rejection_count": int(evidence["rejection_count"]),
|
||||||
|
"last_rejection_age": evidence["last_rejection_age"],
|
||||||
|
"weighted_respects": float(evidence["weighted_respects"]),
|
||||||
})
|
})
|
||||||
|
|
||||||
# Merge nearby levels
|
merged = _merge_levels(raw_levels, merge_tol)
|
||||||
merged = _merge_levels(raw_levels, tolerance)
|
|
||||||
|
|
||||||
# Tag as support/resistance
|
|
||||||
tagged = _tag_levels(merged, current_price)
|
tagged = _tag_levels(merged, current_price)
|
||||||
|
capped = _cap_levels(tagged, max_levels=max_levels)
|
||||||
|
capped.sort(key=lambda x: x["strength"], reverse=True)
|
||||||
|
return capped
|
||||||
|
|
||||||
# Sort by strength descending
|
|
||||||
tagged.sort(key=lambda x: x["strength"], reverse=True)
|
|
||||||
|
|
||||||
return tagged
|
|
||||||
|
|
||||||
def cluster_sr_zones(
|
def cluster_sr_zones(
|
||||||
levels: list[dict],
|
levels: list[dict],
|
||||||
current_price: float,
|
current_price: float,
|
||||||
tolerance: float = 0.02,
|
tolerance: float = 0.02,
|
||||||
max_zones: int | None = None,
|
max_zones: int | None = None,
|
||||||
|
strength_mode: str = "sum",
|
||||||
) -> list[dict]:
|
) -> list[dict]:
|
||||||
"""Cluster nearby S/R levels into zones.
|
"""Cluster nearby S/R levels into zones.
|
||||||
|
|
||||||
@@ -263,8 +718,26 @@ def cluster_sr_zones(
|
|||||||
low = min(prices)
|
low = min(prices)
|
||||||
high = max(prices)
|
high = max(prices)
|
||||||
midpoint = (low + high) / 2.0
|
midpoint = (low + high) / 2.0
|
||||||
strength = min(100, sum(lvl["strength"] for lvl in cluster))
|
if strength_mode == "soft":
|
||||||
|
strongest = max(int(lvl["strength"]) for lvl in cluster)
|
||||||
|
all_sources = {
|
||||||
|
source
|
||||||
|
for lvl in cluster
|
||||||
|
for source in (lvl.get("sources") or [lvl.get("detection_method", "unknown")])
|
||||||
|
}
|
||||||
|
strength = min(100, strongest + (5 if len(all_sources) > 1 else 0))
|
||||||
|
elif strength_mode == "sum":
|
||||||
|
strength = min(100, sum(int(lvl["strength"]) for lvl in cluster))
|
||||||
|
all_sources = {
|
||||||
|
source
|
||||||
|
for lvl in cluster
|
||||||
|
for source in (lvl.get("sources") or [lvl.get("detection_method", "unknown")])
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
raise ValueError(f"Unsupported S/R zone strength mode: {strength_mode}")
|
||||||
level_count = len(cluster)
|
level_count = len(cluster)
|
||||||
|
rejection_count = max(int(lvl.get("rejection_count", 0)) for lvl in cluster)
|
||||||
|
ages = [lvl.get("last_rejection_age") for lvl in cluster if lvl.get("last_rejection_age") is not None]
|
||||||
|
|
||||||
# 4. Tag zone type
|
# 4. Tag zone type
|
||||||
zone_type = "support" if midpoint < current_price else "resistance"
|
zone_type = "support" if midpoint < current_price else "resistance"
|
||||||
@@ -276,6 +749,9 @@ def cluster_sr_zones(
|
|||||||
"strength": strength,
|
"strength": strength,
|
||||||
"type": zone_type,
|
"type": zone_type,
|
||||||
"level_count": level_count,
|
"level_count": level_count,
|
||||||
|
"sources": sorted(all_sources),
|
||||||
|
"rejection_count": rejection_count,
|
||||||
|
"last_rejection_age": min(ages) if ages else None,
|
||||||
})
|
})
|
||||||
|
|
||||||
# 5. Split into support and resistance pools, each sorted by strength desc
|
# 5. Split into support and resistance pools, each sorted by strength desc
|
||||||
@@ -319,11 +795,10 @@ def cluster_sr_zones(
|
|||||||
return selected
|
return selected
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
async def recalculate_sr_levels(
|
async def recalculate_sr_levels(
|
||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
symbol: str,
|
symbol: str,
|
||||||
tolerance: float = DEFAULT_TOLERANCE,
|
tolerance: float | None = None,
|
||||||
) -> list[SRLevel]:
|
) -> list[SRLevel]:
|
||||||
"""Recalculate S/R levels for a ticker and persist to DB.
|
"""Recalculate S/R levels for a ticker and persist to DB.
|
||||||
|
|
||||||
@@ -380,10 +855,44 @@ async def recalculate_sr_levels(
|
|||||||
async def get_sr_levels(
|
async def get_sr_levels(
|
||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
symbol: str,
|
symbol: str,
|
||||||
tolerance: float = DEFAULT_TOLERANCE,
|
tolerance: float | None = None,
|
||||||
) -> list[SRLevel]:
|
) -> list[SRLevel]:
|
||||||
"""Get S/R levels for a ticker, recalculating on every request (MVP).
|
"""Return Structural S/R for a ticker, strength descending.
|
||||||
|
|
||||||
Returns levels sorted by strength descending.
|
Default (``tolerance is None``): read persisted levels only — no rewrite.
|
||||||
|
Pipeline/ingestion call ``recalculate_sr_levels`` after OHLCV changes.
|
||||||
|
|
||||||
|
When ``tolerance`` is set: build a **transient** detect view with that merge
|
||||||
|
tolerance and do not persist it (custom merge for API clients). Transient
|
||||||
|
rows use negative ids so they cannot be confused with stored levels.
|
||||||
"""
|
"""
|
||||||
return await recalculate_sr_levels(db, symbol, tolerance)
|
if tolerance is None:
|
||||||
|
ticker = await _get_ticker(db, symbol)
|
||||||
|
result = await db.execute(
|
||||||
|
select(SRLevel)
|
||||||
|
.where(SRLevel.ticker_id == ticker.id)
|
||||||
|
.order_by(SRLevel.strength.desc())
|
||||||
|
)
|
||||||
|
return list(result.scalars().all())
|
||||||
|
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
ticker = await _get_ticker(db, symbol)
|
||||||
|
records = await query_ohlcv(db, symbol)
|
||||||
|
if not records:
|
||||||
|
return []
|
||||||
|
_, highs, lows, closes, volumes = _extract_ohlcv(records)
|
||||||
|
detected = detect_sr_levels(highs, lows, closes, volumes, tolerance)
|
||||||
|
now = datetime.utcnow()
|
||||||
|
# Ephemeral objects with the SRLevel attribute shape the router expects.
|
||||||
|
return [
|
||||||
|
SimpleNamespace( # type: ignore[return-value]
|
||||||
|
id=-(i + 1),
|
||||||
|
price_level=lvl["price_level"],
|
||||||
|
type=lvl["type"],
|
||||||
|
strength=lvl["strength"],
|
||||||
|
detection_method=lvl["detection_method"],
|
||||||
|
created_at=now,
|
||||||
|
)
|
||||||
|
for i, lvl in enumerate(detected)
|
||||||
|
]
|
||||||
|
|||||||
@@ -0,0 +1,191 @@
|
|||||||
|
"""Persist and query operational system events (warnings / errors)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
|
||||||
|
from sqlalchemy import select, update
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.system_event import SystemEvent
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
DEFAULT_LOOKBACK_DAYS = 7
|
||||||
|
DEFAULT_DEDUP_HOURS = 24
|
||||||
|
SEVERITIES = frozenset({"warning", "error"})
|
||||||
|
|
||||||
|
|
||||||
|
async def log_event(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
severity: str,
|
||||||
|
source: str,
|
||||||
|
code: str,
|
||||||
|
message: str,
|
||||||
|
symbol: str | None = None,
|
||||||
|
dedup_key: str | None = None,
|
||||||
|
dedup_hours: int = DEFAULT_DEDUP_HOURS,
|
||||||
|
) -> SystemEvent | None:
|
||||||
|
"""Insert a system event, optionally de-duplicating recent identical keys.
|
||||||
|
|
||||||
|
Returns the new row, or None when a recent dedup_key already exists.
|
||||||
|
Commits the session.
|
||||||
|
"""
|
||||||
|
severity = (severity or "").strip().lower()
|
||||||
|
if severity not in SEVERITIES:
|
||||||
|
severity = "warning"
|
||||||
|
source = (source or "system")[:64]
|
||||||
|
code = (code or "unknown")[:64]
|
||||||
|
message = (message or "").strip() or code
|
||||||
|
symbol = symbol.strip().upper()[:20] if symbol else None
|
||||||
|
dedup_key = dedup_key[:200] if dedup_key else None
|
||||||
|
|
||||||
|
if dedup_key:
|
||||||
|
cutoff = datetime.now(timezone.utc) - timedelta(hours=max(1, dedup_hours))
|
||||||
|
existing = await db.execute(
|
||||||
|
select(SystemEvent.id)
|
||||||
|
.where(
|
||||||
|
SystemEvent.dedup_key == dedup_key,
|
||||||
|
SystemEvent.created_at >= cutoff,
|
||||||
|
)
|
||||||
|
.limit(1)
|
||||||
|
)
|
||||||
|
if existing.scalar_one_or_none() is not None:
|
||||||
|
return None
|
||||||
|
|
||||||
|
row = SystemEvent(
|
||||||
|
severity=severity,
|
||||||
|
source=source,
|
||||||
|
code=code,
|
||||||
|
message=message[:4000],
|
||||||
|
symbol=symbol,
|
||||||
|
dedup_key=dedup_key,
|
||||||
|
created_at=datetime.now(timezone.utc),
|
||||||
|
)
|
||||||
|
db.add(row)
|
||||||
|
await db.commit()
|
||||||
|
await db.refresh(row)
|
||||||
|
return row
|
||||||
|
|
||||||
|
|
||||||
|
async def log_event_standalone(
|
||||||
|
*,
|
||||||
|
severity: str,
|
||||||
|
source: str,
|
||||||
|
code: str,
|
||||||
|
message: str,
|
||||||
|
symbol: str | None = None,
|
||||||
|
dedup_key: str | None = None,
|
||||||
|
) -> None:
|
||||||
|
"""Open a short-lived session and log an event (for scheduler / fire-and-forget)."""
|
||||||
|
try:
|
||||||
|
from app.database import async_session_factory
|
||||||
|
|
||||||
|
async with async_session_factory() as db:
|
||||||
|
await log_event(
|
||||||
|
db,
|
||||||
|
severity=severity,
|
||||||
|
source=source,
|
||||||
|
code=code,
|
||||||
|
message=message,
|
||||||
|
symbol=symbol,
|
||||||
|
dedup_key=dedup_key,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Failed to persist system event %s/%s", source, code)
|
||||||
|
|
||||||
|
|
||||||
|
async def list_events(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
days: int = DEFAULT_LOOKBACK_DAYS,
|
||||||
|
severity: str | None = None,
|
||||||
|
unacknowledged_only: bool = False,
|
||||||
|
limit: int = 200,
|
||||||
|
) -> list[SystemEvent]:
|
||||||
|
days = max(1, min(int(days), 30))
|
||||||
|
cutoff = datetime.now(timezone.utc) - timedelta(days=days)
|
||||||
|
stmt = (
|
||||||
|
select(SystemEvent)
|
||||||
|
.where(SystemEvent.created_at >= cutoff)
|
||||||
|
.order_by(SystemEvent.created_at.desc())
|
||||||
|
.limit(max(1, min(limit, 500)))
|
||||||
|
)
|
||||||
|
if severity in SEVERITIES:
|
||||||
|
stmt = stmt.where(SystemEvent.severity == severity)
|
||||||
|
if unacknowledged_only:
|
||||||
|
stmt = stmt.where(SystemEvent.acknowledged_at.is_(None))
|
||||||
|
result = await db.execute(stmt)
|
||||||
|
return list(result.scalars().all())
|
||||||
|
|
||||||
|
|
||||||
|
async def summary(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
days: int = DEFAULT_LOOKBACK_DAYS,
|
||||||
|
) -> dict:
|
||||||
|
"""Counts for badge + admin header."""
|
||||||
|
days = max(1, min(int(days), 30))
|
||||||
|
cutoff = datetime.now(timezone.utc) - timedelta(days=days)
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(SystemEvent.severity, SystemEvent.acknowledged_at).where(
|
||||||
|
SystemEvent.created_at >= cutoff
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).all()
|
||||||
|
total = len(rows)
|
||||||
|
unacked = 0
|
||||||
|
errors = 0
|
||||||
|
warnings = 0
|
||||||
|
for severity, acknowledged_at in rows:
|
||||||
|
if acknowledged_at is not None:
|
||||||
|
continue
|
||||||
|
unacked += 1
|
||||||
|
if severity == "error":
|
||||||
|
errors += 1
|
||||||
|
elif severity == "warning":
|
||||||
|
warnings += 1
|
||||||
|
return {
|
||||||
|
"days": days,
|
||||||
|
"total": total,
|
||||||
|
"unacknowledged": unacked,
|
||||||
|
"unacknowledged_errors": errors,
|
||||||
|
"unacknowledged_warnings": warnings,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def acknowledge_all(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
days: int = DEFAULT_LOOKBACK_DAYS,
|
||||||
|
) -> int:
|
||||||
|
"""Mark unacknowledged events in the lookback window as dismissed. Returns count."""
|
||||||
|
days = max(1, min(int(days), 30))
|
||||||
|
cutoff = datetime.now(timezone.utc) - timedelta(days=days)
|
||||||
|
now = datetime.now(timezone.utc)
|
||||||
|
result = await db.execute(
|
||||||
|
update(SystemEvent)
|
||||||
|
.where(
|
||||||
|
SystemEvent.created_at >= cutoff,
|
||||||
|
SystemEvent.acknowledged_at.is_(None),
|
||||||
|
)
|
||||||
|
.values(acknowledged_at=now)
|
||||||
|
)
|
||||||
|
await db.commit()
|
||||||
|
return int(result.rowcount or 0)
|
||||||
|
|
||||||
|
|
||||||
|
def event_to_dict(row: SystemEvent) -> dict:
|
||||||
|
return {
|
||||||
|
"id": row.id,
|
||||||
|
"severity": row.severity,
|
||||||
|
"source": row.source,
|
||||||
|
"code": row.code,
|
||||||
|
"message": row.message,
|
||||||
|
"symbol": row.symbol,
|
||||||
|
"created_at": row.created_at.isoformat() if row.created_at else None,
|
||||||
|
"acknowledged_at": row.acknowledged_at.isoformat() if row.acknowledged_at else None,
|
||||||
|
}
|
||||||
@@ -6,6 +6,7 @@ well-known universes (S&P 500, NASDAQ-100, NASDAQ All).
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -54,6 +55,43 @@ if not _CA_BUNDLE or not Path(_CA_BUNDLE).exists():
|
|||||||
else:
|
else:
|
||||||
_CA_BUNDLE_PATH = _CA_BUNDLE
|
_CA_BUNDLE_PATH = _CA_BUNDLE
|
||||||
|
|
||||||
|
# Wikipedia often returns 403 to non-browser UAs; use a normal browser-like
|
||||||
|
# identity for constituent scrapes (no cookies/login).
|
||||||
|
_HTTP_HEADERS = {
|
||||||
|
"User-Agent": (
|
||||||
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||||
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||||
|
"Chrome/126.0.0.0 Safari/537.36"
|
||||||
|
),
|
||||||
|
"Accept": "text/html,application/xhtml+xml;q=0.9,*/*;q=0.8",
|
||||||
|
"Accept-Language": "en-US,en;q=0.9",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Modern Wikipedia S&P/Nasdaq tables use exchange templates (NyseSymbol /
|
||||||
|
# NasdaqSymbol) rather than a plain <td><a>SYMBOL</a></td>. Prefer quote URLs
|
||||||
|
# and template params; keep the legacy cell pattern as a last resort.
|
||||||
|
_WIKI_SYMBOL_PATTERNS: tuple[re.Pattern[str], ...] = (
|
||||||
|
re.compile(r"nyse\.com/quote/XNYS:([A-Za-z0-9.-]{1,10})", re.IGNORECASE),
|
||||||
|
re.compile(
|
||||||
|
r"nasdaq\.com/market-activity/stocks/([A-Za-z0-9.-]{1,10})",
|
||||||
|
re.IGNORECASE,
|
||||||
|
),
|
||||||
|
# {{NyseSymbol|BNY}} / {{NasdaqSymbol|AAPL}} rendered data-mw params
|
||||||
|
re.compile(
|
||||||
|
r'"target":\{"wt":"(?:Nyse|Nasdaq)Symbol"[^}]*\}.*"wt":"([A-Z][A-Z0-9.-]{0,9})"',
|
||||||
|
re.IGNORECASE,
|
||||||
|
),
|
||||||
|
re.compile(r"<td>\s*<a[^>]*>([A-Z.]{1,10})</a>\s*</td>", re.IGNORECASE),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_wiki_symbols(html: str) -> list[str]:
|
||||||
|
"""Pull ticker symbols out of a Wikipedia constituents page."""
|
||||||
|
found: list[str] = []
|
||||||
|
for pattern in _WIKI_SYMBOL_PATTERNS:
|
||||||
|
found.extend(pattern.findall(html))
|
||||||
|
return found
|
||||||
|
|
||||||
|
|
||||||
def _validate_universe(universe: str) -> str:
|
def _validate_universe(universe: str) -> str:
|
||||||
normalised = universe.strip().lower()
|
normalised = universe.strip().lower()
|
||||||
@@ -185,20 +223,19 @@ async def _fetch_universe_symbols_from_fmp(universe: str) -> list[str]:
|
|||||||
raise ProviderError(f"Failed to fetch universe symbols from FMP for '{universe}'")
|
raise ProviderError(f"Failed to fetch universe symbols from FMP for '{universe}'")
|
||||||
|
|
||||||
|
|
||||||
async def _fetch_html_symbols(
|
async def _fetch_wiki_constituent_symbols(
|
||||||
client: httpx.AsyncClient,
|
client: httpx.AsyncClient,
|
||||||
url: str,
|
url: str,
|
||||||
pattern: str,
|
|
||||||
) -> tuple[list[str], str | None]:
|
) -> tuple[list[str], str | None]:
|
||||||
try:
|
try:
|
||||||
response = await client.get(url)
|
response = await client.get(url, headers=_HTTP_HEADERS)
|
||||||
except httpx.HTTPError as exc:
|
except httpx.HTTPError as exc:
|
||||||
return [], f"{url}: network error ({type(exc).__name__}: {exc})"
|
return [], f"{url}: network error ({type(exc).__name__}: {exc})"
|
||||||
|
|
||||||
if response.status_code != 200:
|
if response.status_code != 200:
|
||||||
return [], f"{url}: HTTP {response.status_code}"
|
return [], f"{url}: HTTP {response.status_code}"
|
||||||
|
|
||||||
matches = re.findall(pattern, response.text, flags=re.IGNORECASE)
|
matches = _extract_wiki_symbols(response.text)
|
||||||
if not matches:
|
if not matches:
|
||||||
return [], f"{url}: no symbols parsed"
|
return [], f"{url}: no symbols parsed"
|
||||||
return list(matches), None
|
return list(matches), None
|
||||||
@@ -209,7 +246,7 @@ async def _fetch_nasdaq_trader_symbols(
|
|||||||
) -> tuple[list[str], str | None]:
|
) -> tuple[list[str], str | None]:
|
||||||
url = "https://www.nasdaqtrader.com/dynamic/SymDir/nasdaqlisted.txt"
|
url = "https://www.nasdaqtrader.com/dynamic/SymDir/nasdaqlisted.txt"
|
||||||
try:
|
try:
|
||||||
response = await client.get(url)
|
response = await client.get(url, headers=_HTTP_HEADERS)
|
||||||
except httpx.HTTPError as exc:
|
except httpx.HTTPError as exc:
|
||||||
return [], f"{url}: network error ({type(exc).__name__}: {exc})"
|
return [], f"{url}: network error ({type(exc).__name__}: {exc})"
|
||||||
|
|
||||||
@@ -239,18 +276,17 @@ async def _fetch_universe_symbols_from_public(universe: str) -> tuple[list[str],
|
|||||||
|
|
||||||
sp500_url = "https://en.wikipedia.org/wiki/List_of_S%26P_500_companies"
|
sp500_url = "https://en.wikipedia.org/wiki/List_of_S%26P_500_companies"
|
||||||
nasdaq100_url = "https://en.wikipedia.org/wiki/Nasdaq-100"
|
nasdaq100_url = "https://en.wikipedia.org/wiki/Nasdaq-100"
|
||||||
wiki_symbol_pattern = r"<td>\s*<a[^>]*>([A-Z.]{1,10})</a>\s*</td>"
|
|
||||||
|
|
||||||
async with httpx.AsyncClient(timeout=30.0, verify=_CA_BUNDLE_PATH) as client:
|
async with httpx.AsyncClient(timeout=30.0, verify=_CA_BUNDLE_PATH) as client:
|
||||||
if universe == "sp500":
|
if universe == "sp500":
|
||||||
symbols, error = await _fetch_html_symbols(client, sp500_url, wiki_symbol_pattern)
|
symbols, error = await _fetch_wiki_constituent_symbols(client, sp500_url)
|
||||||
if error:
|
if error:
|
||||||
failures.append(error)
|
failures.append(error)
|
||||||
else:
|
else:
|
||||||
return symbols, failures, "wikipedia_sp500"
|
return symbols, failures, "wikipedia_sp500"
|
||||||
|
|
||||||
if universe == "nasdaq100":
|
if universe == "nasdaq100":
|
||||||
symbols, error = await _fetch_html_symbols(client, nasdaq100_url, wiki_symbol_pattern)
|
symbols, error = await _fetch_wiki_constituent_symbols(client, nasdaq100_url)
|
||||||
if error:
|
if error:
|
||||||
failures.append(error)
|
failures.append(error)
|
||||||
else:
|
else:
|
||||||
@@ -307,7 +343,10 @@ async def _write_cached_symbols(
|
|||||||
await db.commit()
|
await db.commit()
|
||||||
|
|
||||||
|
|
||||||
async def fetch_universe_symbols(db: AsyncSession, universe: str) -> list[str]:
|
async def fetch_universe_symbols(
|
||||||
|
db: AsyncSession,
|
||||||
|
universe: str,
|
||||||
|
) -> tuple[list[str], str]:
|
||||||
"""Fetch and normalise symbols for a supported universe with fallbacks.
|
"""Fetch and normalise symbols for a supported universe with fallbacks.
|
||||||
|
|
||||||
Fallback order:
|
Fallback order:
|
||||||
@@ -315,6 +354,10 @@ async def fetch_universe_symbols(db: AsyncSession, universe: str) -> list[str]:
|
|||||||
2) FMP endpoints (if available)
|
2) FMP endpoints (if available)
|
||||||
3) Cached snapshot in SystemSetting
|
3) Cached snapshot in SystemSetting
|
||||||
4) Built-in seed symbols
|
4) Built-in seed symbols
|
||||||
|
|
||||||
|
Returns ``(symbols, source_label)`` so bootstrap UI can show where the
|
||||||
|
list came from (important when Wikipedia/FMP fail and a stale cache still
|
||||||
|
lists BK instead of BNY).
|
||||||
"""
|
"""
|
||||||
normalised_universe = _validate_universe(universe)
|
normalised_universe = _validate_universe(universe)
|
||||||
failures: list[str] = []
|
failures: list[str] = []
|
||||||
@@ -324,14 +367,14 @@ async def fetch_universe_symbols(db: AsyncSession, universe: str) -> list[str]:
|
|||||||
cleaned_public = _normalise_symbols(public_symbols)
|
cleaned_public = _normalise_symbols(public_symbols)
|
||||||
if cleaned_public:
|
if cleaned_public:
|
||||||
await _write_cached_symbols(db, normalised_universe, cleaned_public, public_source or "public")
|
await _write_cached_symbols(db, normalised_universe, cleaned_public, public_source or "public")
|
||||||
return cleaned_public
|
return cleaned_public, public_source or "public"
|
||||||
|
|
||||||
try:
|
try:
|
||||||
fmp_symbols = await _fetch_universe_symbols_from_fmp(normalised_universe)
|
fmp_symbols = await _fetch_universe_symbols_from_fmp(normalised_universe)
|
||||||
cleaned_fmp = _normalise_symbols(fmp_symbols)
|
cleaned_fmp = _normalise_symbols(fmp_symbols)
|
||||||
if cleaned_fmp:
|
if cleaned_fmp:
|
||||||
await _write_cached_symbols(db, normalised_universe, cleaned_fmp, "fmp")
|
await _write_cached_symbols(db, normalised_universe, cleaned_fmp, "fmp")
|
||||||
return cleaned_fmp
|
return cleaned_fmp, "fmp"
|
||||||
except (ProviderError, ValidationError) as exc:
|
except (ProviderError, ValidationError) as exc:
|
||||||
failures.append(str(exc))
|
failures.append(str(exc))
|
||||||
|
|
||||||
@@ -342,7 +385,7 @@ async def fetch_universe_symbols(db: AsyncSession, universe: str) -> list[str]:
|
|||||||
normalised_universe,
|
normalised_universe,
|
||||||
"; ".join(failures[:3]),
|
"; ".join(failures[:3]),
|
||||||
)
|
)
|
||||||
return cached_symbols
|
return cached_symbols, "cache"
|
||||||
|
|
||||||
seed_symbols = _normalise_symbols(_SEED_UNIVERSES.get(normalised_universe, []))
|
seed_symbols = _normalise_symbols(_SEED_UNIVERSES.get(normalised_universe, []))
|
||||||
if seed_symbols:
|
if seed_symbols:
|
||||||
@@ -351,12 +394,61 @@ async def fetch_universe_symbols(db: AsyncSession, universe: str) -> list[str]:
|
|||||||
normalised_universe,
|
normalised_universe,
|
||||||
"; ".join(failures[:3]),
|
"; ".join(failures[:3]),
|
||||||
)
|
)
|
||||||
return seed_symbols
|
return seed_symbols, "seed"
|
||||||
|
|
||||||
reason = "; ".join(failures[:6]) if failures else "no provider returned symbols"
|
reason = "; ".join(failures[:6]) if failures else "no provider returned symbols"
|
||||||
raise ProviderError(f"Universe '{normalised_universe}' returned no valid symbols. Attempts: {reason}")
|
raise ProviderError(f"Universe '{normalised_universe}' returned no valid symbols. Attempts: {reason}")
|
||||||
|
|
||||||
|
|
||||||
|
async def _fetch_alpaca_asset_names() -> dict[str, str]:
|
||||||
|
"""One Alpaca Trading-API call → {internal_symbol: company_name} for all US
|
||||||
|
equities. Tries paper and live endpoints so it works with either key type."""
|
||||||
|
if not settings.alpaca_api_key or not settings.alpaca_api_secret:
|
||||||
|
raise ValidationError("Alpaca API credentials are required to backfill names")
|
||||||
|
|
||||||
|
from alpaca.trading.client import TradingClient
|
||||||
|
from alpaca.trading.enums import AssetClass, AssetStatus
|
||||||
|
from alpaca.trading.requests import GetAssetsRequest
|
||||||
|
|
||||||
|
req = GetAssetsRequest(status=AssetStatus.ACTIVE, asset_class=AssetClass.US_EQUITY)
|
||||||
|
last_err: Exception | None = None
|
||||||
|
for paper in (True, False):
|
||||||
|
try:
|
||||||
|
client = TradingClient(settings.alpaca_api_key, settings.alpaca_api_secret, paper=paper)
|
||||||
|
assets = await asyncio.to_thread(client.get_all_assets, req)
|
||||||
|
names: dict[str, str] = {}
|
||||||
|
for asset in assets:
|
||||||
|
sym = getattr(asset, "symbol", None)
|
||||||
|
nm = getattr(asset, "name", None)
|
||||||
|
if sym and nm:
|
||||||
|
names[sym.replace(".", "-").upper()] = nm # BRK.B → BRK-B
|
||||||
|
if names:
|
||||||
|
return names
|
||||||
|
except Exception as exc: # noqa: BLE001 — try the other endpoint
|
||||||
|
last_err = exc
|
||||||
|
|
||||||
|
raise ProviderError(f"Failed to fetch asset names from Alpaca: {last_err}")
|
||||||
|
|
||||||
|
|
||||||
|
async def backfill_ticker_names(db: AsyncSession, *, only_missing: bool = True) -> dict[str, int]:
|
||||||
|
"""Fill Ticker.name from Alpaca in a single request for the whole universe."""
|
||||||
|
result = await db.execute(select(Ticker))
|
||||||
|
tickers = list(result.scalars().all())
|
||||||
|
targets = [t for t in tickers if not t.name] if only_missing else tickers
|
||||||
|
if not targets:
|
||||||
|
return {"updated": 0, "checked": 0, "unmatched": 0}
|
||||||
|
|
||||||
|
names = await _fetch_alpaca_asset_names()
|
||||||
|
updated = 0
|
||||||
|
for ticker in targets:
|
||||||
|
nm = names.get(ticker.symbol.upper())
|
||||||
|
if nm and nm != ticker.name:
|
||||||
|
ticker.name = nm[:120]
|
||||||
|
updated += 1
|
||||||
|
await db.commit()
|
||||||
|
return {"updated": updated, "checked": len(targets), "unmatched": len(targets) - updated}
|
||||||
|
|
||||||
|
|
||||||
async def bootstrap_universe(
|
async def bootstrap_universe(
|
||||||
db: AsyncSession,
|
db: AsyncSession,
|
||||||
universe: str,
|
universe: str,
|
||||||
@@ -368,7 +460,7 @@ async def bootstrap_universe(
|
|||||||
Returns summary counts for added/existing/deleted symbols.
|
Returns summary counts for added/existing/deleted symbols.
|
||||||
"""
|
"""
|
||||||
normalised_universe = _validate_universe(universe)
|
normalised_universe = _validate_universe(universe)
|
||||||
symbols = await fetch_universe_symbols(db, normalised_universe)
|
symbols, source = await fetch_universe_symbols(db, normalised_universe)
|
||||||
|
|
||||||
existing_rows = await db.execute(select(Ticker.symbol))
|
existing_rows = await db.execute(select(Ticker.symbol))
|
||||||
existing_symbols = set(existing_rows.scalars().all())
|
existing_symbols = set(existing_rows.scalars().all())
|
||||||
@@ -387,10 +479,19 @@ async def bootstrap_universe(
|
|||||||
|
|
||||||
await db.commit()
|
await db.commit()
|
||||||
|
|
||||||
|
# Best-effort: fill company names for any tickers still missing one. Never let
|
||||||
|
# a name-fetch hiccup fail the bootstrap itself.
|
||||||
|
try:
|
||||||
|
await backfill_ticker_names(db, only_missing=True)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
logger.warning("Ticker name backfill failed during bootstrap", exc_info=True)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"universe": normalised_universe,
|
"universe": normalised_universe,
|
||||||
|
"source": source,
|
||||||
"total_universe_symbols": len(symbols),
|
"total_universe_symbols": len(symbols),
|
||||||
"added": len(symbols_to_add),
|
"added": len(symbols_to_add),
|
||||||
"already_tracked": len(target_symbols & existing_symbols),
|
"already_tracked": len(target_symbols & existing_symbols),
|
||||||
"deleted": deleted_count,
|
"deleted": deleted_count,
|
||||||
|
"added_symbols": symbols_to_add[:50],
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,132 @@
|
|||||||
|
"""Shared live trading-policy state and availability checks."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from collections.abc import Iterable
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
from zoneinfo import ZoneInfo
|
||||||
|
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.paper_trade import PaperTrade
|
||||||
|
|
||||||
|
# Gate-reset "day" boundary matches US cash equities session calendar, not UTC.
|
||||||
|
_REENTRY_DAY_TZ = ZoneInfo("America/New_York")
|
||||||
|
|
||||||
|
|
||||||
|
def _ny_trading_date(moment: datetime) -> date:
|
||||||
|
"""Calendar date in America/New_York for a gate-reset observation."""
|
||||||
|
if moment.tzinfo is None:
|
||||||
|
moment = moment.replace(tzinfo=timezone.utc)
|
||||||
|
return moment.astimezone(_REENTRY_DAY_TZ).date()
|
||||||
|
|
||||||
|
|
||||||
|
MANUAL_BOOK = "manual"
|
||||||
|
SHADOW_BOOK = "shadow"
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_initial_stop_trades(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
closed_before: datetime | None = None,
|
||||||
|
book: str = MANUAL_BOOK,
|
||||||
|
) -> dict[int, PaperTrade]:
|
||||||
|
"""Return a ticker's latest closed trade only when it was an initial stop.
|
||||||
|
|
||||||
|
Scoped to one ``book``: the discretionary and shadow books diverge as soon
|
||||||
|
as their entries differ, so each must see only its own stop history when
|
||||||
|
deciding whether a ticker is locked out of re-entry.
|
||||||
|
"""
|
||||||
|
ranked_stmt = (
|
||||||
|
select(
|
||||||
|
PaperTrade.id.label("trade_id"),
|
||||||
|
func.row_number()
|
||||||
|
.over(
|
||||||
|
partition_by=PaperTrade.ticker_id,
|
||||||
|
order_by=(PaperTrade.closed_at.desc(), PaperTrade.id.desc()),
|
||||||
|
)
|
||||||
|
.label("recency"),
|
||||||
|
)
|
||||||
|
.where(
|
||||||
|
PaperTrade.status == "closed",
|
||||||
|
PaperTrade.closed_at.is_not(None),
|
||||||
|
PaperTrade.book == book,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if closed_before is not None:
|
||||||
|
ranked_stmt = ranked_stmt.where(PaperTrade.closed_at <= closed_before)
|
||||||
|
ranked = ranked_stmt.subquery()
|
||||||
|
stmt = (
|
||||||
|
select(PaperTrade)
|
||||||
|
.join(ranked, ranked.c.trade_id == PaperTrade.id)
|
||||||
|
.where(
|
||||||
|
ranked.c.recency == 1,
|
||||||
|
PaperTrade.close_reason == "stop",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
result = await db.execute(stmt)
|
||||||
|
return {trade.ticker_id: trade for trade in result.scalars()}
|
||||||
|
|
||||||
|
|
||||||
|
async def get_reentry_gate_locks(
|
||||||
|
db: AsyncSession, *, book: str = MANUAL_BOOK
|
||||||
|
) -> dict[int, datetime]:
|
||||||
|
"""Return tickers still waiting for a post-stop gate failure.
|
||||||
|
|
||||||
|
A later qualified setup is actionable only after the daily scanner has
|
||||||
|
observed an unqualified evaluation after the latest initial-stop exit and
|
||||||
|
then a fresh qualification. The returned timestamp is the stop time and is
|
||||||
|
useful for diagnostics; callers normally only need the keys.
|
||||||
|
"""
|
||||||
|
latest = await _latest_initial_stop_trades(db, book=book)
|
||||||
|
return {
|
||||||
|
ticker_id: trade.closed_at
|
||||||
|
for ticker_id, trade in latest.items()
|
||||||
|
if trade.reentry_gate_requalified_at is None and trade.closed_at is not None
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
async def observe_reentry_gate_transitions(
|
||||||
|
db: AsyncSession,
|
||||||
|
*,
|
||||||
|
evaluated_ticker_ids: Iterable[int],
|
||||||
|
qualified_ticker_ids: Iterable[int],
|
||||||
|
observed_at: datetime | None = None,
|
||||||
|
book: str = MANUAL_BOOK,
|
||||||
|
) -> set[int]:
|
||||||
|
"""Persist gate-failure and later requalification observations.
|
||||||
|
|
||||||
|
Only tickers whose scan completed successfully belong in
|
||||||
|
``evaluated_ticker_ids``. This prevents a scanner exception from being
|
||||||
|
mistaken for a real gate exit. The caller owns the transaction; this helper
|
||||||
|
flushes so the new state is immediately visible in that transaction.
|
||||||
|
"""
|
||||||
|
evaluated = {int(ticker_id) for ticker_id in evaluated_ticker_ids}
|
||||||
|
if not evaluated:
|
||||||
|
return set()
|
||||||
|
qualified = {int(ticker_id) for ticker_id in qualified_ticker_ids}
|
||||||
|
timestamp = observed_at or datetime.now(timezone.utc)
|
||||||
|
latest = await _latest_initial_stop_trades(db, closed_before=timestamp, book=book)
|
||||||
|
updated: set[int] = set()
|
||||||
|
for ticker_id in evaluated:
|
||||||
|
trade = latest.get(ticker_id)
|
||||||
|
if trade is None or trade.reentry_gate_requalified_at is not None:
|
||||||
|
continue
|
||||||
|
if trade.reentry_gate_failed_at is None:
|
||||||
|
if ticker_id not in qualified:
|
||||||
|
trade.reentry_gate_failed_at = timestamp
|
||||||
|
updated.add(ticker_id)
|
||||||
|
elif ticker_id in qualified:
|
||||||
|
# Study semantics: requalify only on a *subsequent* daily observation.
|
||||||
|
# Same America/New_York calendar day as the failure does not unlock,
|
||||||
|
# even if multiple full-universe scans run (manual + near-close).
|
||||||
|
if _ny_trading_date(trade.reentry_gate_failed_at) < _ny_trading_date(
|
||||||
|
timestamp
|
||||||
|
):
|
||||||
|
trade.reentry_gate_requalified_at = timestamp
|
||||||
|
updated.add(ticker_id)
|
||||||
|
|
||||||
|
if updated:
|
||||||
|
await db.flush()
|
||||||
|
return updated
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user