Docs/dolt plan clarifications #1
@@ -27,6 +27,31 @@ FINNHUB_API_KEY=
|
|||||||
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
||||||
ALPHA_VANTAGE_API_KEY=
|
ALPHA_VANTAGE_API_KEY=
|
||||||
|
|
||||||
|
# Dolt bulk data — local clone of post-no-preference/earnings (workstream A).
|
||||||
|
# DOLT_BINARY: path to the dolt CLI (set the full path in dev if it's not on PATH,
|
||||||
|
# e.g. Windows: C:\Program Files\Dolt\bin\dolt.exe). DOLT_DATA_DIR holds the
|
||||||
|
# clones; in PRODUCTION it MUST be outside the deploy tree (deploy is
|
||||||
|
# rsync --delete) — e.g. /var/lib/signal-platform/dolt. The earnings clone lives
|
||||||
|
# at <DOLT_DATA_DIR>/<DOLT_EARNINGS_SUBDIR>. Production setup is automated by
|
||||||
|
# deploy/provision_fundamentals.sh; see docs/fundamentals-deployment.md.
|
||||||
|
DOLT_BINARY=dolt
|
||||||
|
DOLT_DATA_DIR=dolt-data
|
||||||
|
DOLT_EARNINGS_SUBDIR=earnings
|
||||||
|
# Free-space floor checked before a pull (clone is ~1.7 GB and grows). 5 GB is a
|
||||||
|
# safe production default; lower only on a space-constrained dev box.
|
||||||
|
DOLT_MIN_FREE_DISK_GB=5.0
|
||||||
|
# Hard timeout (s) on each dolt subprocess so a hung pull/sql can't pin the
|
||||||
|
# import connection + advisory lock.
|
||||||
|
DOLT_COMMAND_TIMEOUT_SECONDS=600.0
|
||||||
|
|
||||||
|
# SEC EDGAR (fundamentals, workstream A). SEC fair-access REQUIRES an identifying
|
||||||
|
# User-Agent with a REAL contact email — set it, or requests get 403'd. Stay well
|
||||||
|
# under 10 req/s (spacing below).
|
||||||
|
SEC_USER_AGENT=signal-platform/1.0 (contact: you@example.com)
|
||||||
|
SEC_REQUEST_SPACING_SECONDS=0.2
|
||||||
|
SEC_MAX_RETRIES=4
|
||||||
|
SEC_REQUEST_TIMEOUT_SECONDS=30.0
|
||||||
|
|
||||||
# Regime Monitor — FRED (VIX + HY credit spreads). Free key: https://fred.stlouisfed.org/docs/api/api_key.html
|
# Regime Monitor — FRED (VIX + HY credit spreads). Free key: https://fred.stlouisfed.org/docs/api/api_key.html
|
||||||
# Optional: without it the volatility (V1) and credit (C1) pillars show as n/a.
|
# Optional: without it the volatility (V1) and credit (C1) pillars show as n/a.
|
||||||
FRED_API_KEY=
|
FRED_API_KEY=
|
||||||
|
|||||||
@@ -39,6 +39,10 @@ alembic/versions/__pycache__/
|
|||||||
# Generated SSL bundle
|
# Generated SSL bundle
|
||||||
combined-ca-bundle.pem
|
combined-ca-bundle.pem
|
||||||
|
|
||||||
|
# Dolt local dev clones. Production keeps clones in DOLT_DATA_DIR OUTSIDE the
|
||||||
|
# repo tree (deploy is rsync --delete of the tree); this dir is dev-only.
|
||||||
|
dolt-data/
|
||||||
|
|
||||||
# Local research artifacts
|
# Local research artifacts
|
||||||
# Backtest reports in reports/ are tracked: they are the evidence behind the
|
# Backtest reports in reports/ are tracked: they are the evidence behind the
|
||||||
# production baseline in the README. The snapshot DBs they run against are not.
|
# production baseline in the README. The snapshot DBs they run against are not.
|
||||||
|
|||||||
@@ -0,0 +1,24 @@
|
|||||||
|
Third-party data attribution
|
||||||
|
============================
|
||||||
|
|
||||||
|
Earnings calendar and EPS history
|
||||||
|
---------------------------------
|
||||||
|
This application ingests the earnings calendar and EPS surprise history from the
|
||||||
|
public DoltHub repository:
|
||||||
|
|
||||||
|
post-no-preference/earnings
|
||||||
|
https://www.dolthub.com/repositories/post-no-preference/earnings
|
||||||
|
|
||||||
|
Licensed under Creative Commons Attribution-ShareAlike 4.0 International
|
||||||
|
(CC BY-SA 4.0): https://creativecommons.org/licenses/by-sa/4.0/
|
||||||
|
|
||||||
|
Use in this project: private, internal ingestion only. The data is normalized
|
||||||
|
into PostgreSQL (`earnings_events`) — the announcement calendar is aligned to the
|
||||||
|
EPS history via a minimum-cost monotonic pairing, symbols are normalized, and the
|
||||||
|
session field is mapped to bmo/amc/unknown. No public API, bulk export, or
|
||||||
|
redistribution of the data is provided. This attribution and the upstream license
|
||||||
|
are preserved per the CC BY-SA 4.0 terms. Re-review licensing before any public
|
||||||
|
or commercial access.
|
||||||
|
|
||||||
|
The post-no-preference/stocks repository (workstream B) is not used at this time
|
||||||
|
and would be reviewed separately.
|
||||||
@@ -0,0 +1,145 @@
|
|||||||
|
"""Dolt/SEC fundamentals schema — workstream A
|
||||||
|
|
||||||
|
Revision ID: 026
|
||||||
|
Revises: 025
|
||||||
|
Create Date: 2026-07-21 00:00:00.000000
|
||||||
|
|
||||||
|
Foundational schema for the Dolt bulk-data integration (workstream A): the
|
||||||
|
batch import-run audit table, the SEC-sourced immutable fundamental snapshots
|
||||||
|
(CIK-keyed, one row per accession), the Dolt earnings calendar/history, and the
|
||||||
|
SEC issuer identity columns on ``tickers``. No data is populated here — the
|
||||||
|
importers land in a later phase. ``fundamental_data`` is left untouched; its
|
||||||
|
cutover is gated separately (phase A5). ``data_import_runs`` is created first
|
||||||
|
because the other two tables carry an ``import_run_id`` FK to it.
|
||||||
|
"""
|
||||||
|
from typing import Sequence, Union
|
||||||
|
|
||||||
|
from alembic import op
|
||||||
|
import sqlalchemy as sa
|
||||||
|
|
||||||
|
|
||||||
|
revision: str = "026"
|
||||||
|
down_revision: Union[str, None] = "025"
|
||||||
|
branch_labels: Union[str, Sequence[str], None] = None
|
||||||
|
depends_on: Union[str, Sequence[str], None] = None
|
||||||
|
|
||||||
|
|
||||||
|
def upgrade() -> None:
|
||||||
|
op.create_table(
|
||||||
|
"data_import_runs",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column("source", sa.String(length=32), nullable=False),
|
||||||
|
sa.Column("revision", sa.String(length=64), nullable=True),
|
||||||
|
sa.Column("status", sa.String(length=16), nullable=False),
|
||||||
|
sa.Column("source_max_date", sa.Date(), nullable=True),
|
||||||
|
sa.Column("row_counts_json", sa.Text(), nullable=True),
|
||||||
|
sa.Column("validation_json", sa.Text(), nullable=True),
|
||||||
|
sa.Column("started_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True),
|
||||||
|
sa.Column("error_details", sa.Text(), nullable=True),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_data_import_runs_source_started", "data_import_runs", ["source", "started_at"]
|
||||||
|
)
|
||||||
|
|
||||||
|
op.create_table(
|
||||||
|
"fundamental_snapshots",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column("cik", sa.String(length=10), nullable=False),
|
||||||
|
sa.Column("accession", sa.String(length=25), nullable=False),
|
||||||
|
sa.Column("form", sa.String(length=12), nullable=False),
|
||||||
|
sa.Column("filed_date", sa.Date(), nullable=False),
|
||||||
|
sa.Column("accepted_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.Column("period_start", sa.Date(), nullable=True),
|
||||||
|
sa.Column("period_end", sa.Date(), nullable=False),
|
||||||
|
sa.Column("fiscal_year", sa.Integer(), nullable=False),
|
||||||
|
sa.Column("fiscal_period", sa.String(length=4), nullable=False),
|
||||||
|
# duration facts — cumulative YTD/FY
|
||||||
|
sa.Column("revenue", sa.Float(), nullable=True),
|
||||||
|
sa.Column("net_income", sa.Float(), nullable=True),
|
||||||
|
sa.Column("operating_income", sa.Float(), nullable=True),
|
||||||
|
sa.Column("diluted_eps", sa.Float(), nullable=True),
|
||||||
|
sa.Column("cfo", sa.Float(), nullable=True),
|
||||||
|
sa.Column("capex", sa.Float(), nullable=True),
|
||||||
|
sa.Column("depreciation_amortization", sa.Float(), nullable=True),
|
||||||
|
# balance-sheet facts — period-end
|
||||||
|
sa.Column("cash_and_st_investments", sa.Float(), nullable=True),
|
||||||
|
sa.Column("total_debt", sa.Float(), nullable=True),
|
||||||
|
sa.Column("shares_outstanding", sa.Float(), nullable=True),
|
||||||
|
sa.Column("shares_outstanding_date", sa.Date(), nullable=True),
|
||||||
|
sa.Column(
|
||||||
|
"import_run_id",
|
||||||
|
sa.Integer(),
|
||||||
|
sa.ForeignKey("data_import_runs.id", ondelete="SET NULL"),
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.UniqueConstraint("accession", name="uq_fundamental_snapshots_accession"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period",
|
||||||
|
"fundamental_snapshots",
|
||||||
|
["cik", "fiscal_year", "fiscal_period"],
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period_end",
|
||||||
|
"fundamental_snapshots",
|
||||||
|
["cik", "period_end"],
|
||||||
|
)
|
||||||
|
|
||||||
|
op.create_table(
|
||||||
|
"earnings_events",
|
||||||
|
sa.Column("id", sa.Integer(), primary_key=True),
|
||||||
|
sa.Column(
|
||||||
|
"ticker_id",
|
||||||
|
sa.Integer(),
|
||||||
|
sa.ForeignKey("tickers.id", ondelete="CASCADE"),
|
||||||
|
nullable=False,
|
||||||
|
),
|
||||||
|
sa.Column("announce_date", sa.Date(), nullable=False),
|
||||||
|
sa.Column("session", sa.String(length=10), nullable=False),
|
||||||
|
sa.Column("period_end", sa.Date(), nullable=True),
|
||||||
|
sa.Column("eps_estimate", sa.Float(), nullable=True),
|
||||||
|
sa.Column("eps_actual", sa.Float(), nullable=True),
|
||||||
|
sa.Column("source", sa.String(length=32), nullable=False),
|
||||||
|
sa.Column(
|
||||||
|
"import_run_id",
|
||||||
|
sa.Integer(),
|
||||||
|
sa.ForeignKey("data_import_runs.id", ondelete="SET NULL"),
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
sa.Column("created_at", sa.DateTime(timezone=True), nullable=False),
|
||||||
|
sa.UniqueConstraint("ticker_id", "announce_date", name="uq_earnings_ticker_announce"),
|
||||||
|
)
|
||||||
|
op.create_index(
|
||||||
|
"ix_earnings_events_announce_date", "earnings_events", ["announce_date"]
|
||||||
|
)
|
||||||
|
|
||||||
|
# SEC issuer identity on tickers (nullable; the only ticker<->issuer join point).
|
||||||
|
op.add_column("tickers", sa.Column("cik", sa.String(length=10), nullable=True))
|
||||||
|
op.add_column("tickers", sa.Column("sic", sa.String(length=4), nullable=True))
|
||||||
|
op.add_column(
|
||||||
|
"tickers", sa.Column("sic_description", sa.String(length=160), nullable=True)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downgrade() -> None:
|
||||||
|
op.drop_column("tickers", "sic_description")
|
||||||
|
op.drop_column("tickers", "sic")
|
||||||
|
op.drop_column("tickers", "cik")
|
||||||
|
|
||||||
|
op.drop_index("ix_earnings_events_announce_date", table_name="earnings_events")
|
||||||
|
op.drop_table("earnings_events")
|
||||||
|
|
||||||
|
op.drop_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period_end", table_name="fundamental_snapshots"
|
||||||
|
)
|
||||||
|
op.drop_index(
|
||||||
|
"ix_fundamental_snapshots_cik_period", table_name="fundamental_snapshots"
|
||||||
|
)
|
||||||
|
op.drop_table("fundamental_snapshots")
|
||||||
|
|
||||||
|
op.drop_index(
|
||||||
|
"ix_data_import_runs_source_started", table_name="data_import_runs"
|
||||||
|
)
|
||||||
|
op.drop_table("data_import_runs")
|
||||||
@@ -37,6 +37,30 @@ class Settings(BaseSettings):
|
|||||||
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
# Fundamentals Provider — Alpha Vantage (optional fallback)
|
||||||
alpha_vantage_api_key: str = ""
|
alpha_vantage_api_key: str = ""
|
||||||
|
|
||||||
|
# Dolt bulk-data — local clone of post-no-preference/earnings (workstream A).
|
||||||
|
# dolt_binary: full path when not on PATH (dev/Windows install). dolt_data_dir
|
||||||
|
# holds the clones; in production it MUST be outside the deploy tree (deploy is
|
||||||
|
# rsync --delete) — set DOLT_DATA_DIR to a persistent path. The earnings clone
|
||||||
|
# lives at <dolt_data_dir>/<dolt_earnings_subdir>.
|
||||||
|
dolt_binary: str = "dolt"
|
||||||
|
dolt_data_dir: str = "dolt-data"
|
||||||
|
dolt_earnings_subdir: str = "earnings"
|
||||||
|
# Headroom above the ~1.7 GB earnings clone (grows with pulls); 5 GB is a safe
|
||||||
|
# production floor — override lower only in a space-constrained dev box.
|
||||||
|
dolt_min_free_disk_gb: float = 5.0
|
||||||
|
# Bound every dolt subprocess so a hung pull/sql can't pin the import's
|
||||||
|
# connection + advisory lock indefinitely.
|
||||||
|
dolt_command_timeout_seconds: float = 600.0
|
||||||
|
|
||||||
|
# SEC EDGAR (workstream A, fundamentals). Fair-access policy REQUIRES an
|
||||||
|
# identifying User-Agent with a contact email — set a real one. Stay well
|
||||||
|
# under 10 req/s (spacing below); 403 means the UA/pattern is wrong → the
|
||||||
|
# client alerts and stops rather than retry-looping.
|
||||||
|
sec_user_agent: str = "signal-platform/1.0 (contact: set-a-real-email@example.com)"
|
||||||
|
sec_request_spacing_seconds: float = 0.2
|
||||||
|
sec_max_retries: int = 4
|
||||||
|
sec_request_timeout_seconds: float = 30.0
|
||||||
|
|
||||||
# Regime Monitor — FRED (VIX level + HY credit spreads). Optional: without it
|
# Regime Monitor — FRED (VIX level + HY credit spreads). Optional: without it
|
||||||
# the volatility (P5) and credit-spread (F2) signals are reported as n/a.
|
# the volatility (P5) and credit-spread (F2) signals are reported as n/a.
|
||||||
fred_api_key: str = ""
|
fred_api_key: str = ""
|
||||||
|
|||||||
@@ -3,6 +3,9 @@ from app.models.ohlcv import OHLCVRecord
|
|||||||
from app.models.user import User
|
from app.models.user import User
|
||||||
from app.models.sentiment import SentimentScore
|
from app.models.sentiment import SentimentScore
|
||||||
from app.models.fundamental import FundamentalData
|
from app.models.fundamental import FundamentalData
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
from app.models.score import DimensionScore, CompositeScore
|
from app.models.score import DimensionScore, CompositeScore
|
||||||
from app.models.sr_level import SRLevel
|
from app.models.sr_level import SRLevel
|
||||||
from app.models.trade_setup import TradeSetup
|
from app.models.trade_setup import TradeSetup
|
||||||
@@ -21,6 +24,9 @@ __all__ = [
|
|||||||
"User",
|
"User",
|
||||||
"SentimentScore",
|
"SentimentScore",
|
||||||
"FundamentalData",
|
"FundamentalData",
|
||||||
|
"FundamentalSnapshot",
|
||||||
|
"EarningsEvent",
|
||||||
|
"DataImportRun",
|
||||||
"DimensionScore",
|
"DimensionScore",
|
||||||
"CompositeScore",
|
"CompositeScore",
|
||||||
"SRLevel",
|
"SRLevel",
|
||||||
|
|||||||
@@ -0,0 +1,42 @@
|
|||||||
|
from datetime import date, datetime
|
||||||
|
|
||||||
|
from sqlalchemy import Date, DateTime, Index, String, Text
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class DataImportRun(Base):
|
||||||
|
"""One row per bulk-import attempt (SEC facts / Dolt earnings / Dolt stocks).
|
||||||
|
|
||||||
|
Lean audit record for the batch import framework: every attempt is logged,
|
||||||
|
whether it promoted, was a ``no_op`` (unchanged revision), or ``failed``.
|
||||||
|
``row_counts`` and ``validation`` hold JSON strings (repo convention — see
|
||||||
|
``fundamental_data.unavailable_fields_json``), not JSONB; the validation
|
||||||
|
blob carries reconciliation/discrepancy summaries so no separate conflicts
|
||||||
|
table is needed. One run per source at a time is enforced at write time by a
|
||||||
|
Postgres advisory lock keyed by ``source``.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "data_import_runs"
|
||||||
|
__table_args__ = (
|
||||||
|
Index("ix_data_import_runs_source_started", "source", "started_at"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
# sec_facts | dolt_earnings | dolt_stocks
|
||||||
|
source: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||||
|
# Dolt commit hash, or SEC archive SHA-256. Null until known.
|
||||||
|
revision: Mapped[str | None] = mapped_column(String(64), nullable=True)
|
||||||
|
# running | validated | promoted | no_op | failed
|
||||||
|
status: Mapped[str] = mapped_column(String(16), nullable=False)
|
||||||
|
source_max_date: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
row_counts_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
validation_json: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
|
started_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
|
)
|
||||||
|
completed_at: Mapped[datetime | None] = mapped_column(
|
||||||
|
DateTime(timezone=True), nullable=True
|
||||||
|
)
|
||||||
|
error_details: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||||
@@ -0,0 +1,42 @@
|
|||||||
|
from datetime import date, datetime
|
||||||
|
|
||||||
|
from sqlalchemy import Date, DateTime, Float, ForeignKey, Index, String, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column, relationship
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsEvent(Base):
|
||||||
|
"""Earnings calendar + surprise history, sourced from the DoltHub earnings repo.
|
||||||
|
|
||||||
|
Forward rows (``announce_date`` > today) are the calendar; past rows are
|
||||||
|
results. Rescheduling is handled in the importer's promotion transaction:
|
||||||
|
this source's future-dated rows are deleted and re-inserted from the new
|
||||||
|
snapshot so moved/cancelled dates never linger; past rows are never deleted.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "earnings_events"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("ticker_id", "announce_date", name="uq_earnings_ticker_announce"),
|
||||||
|
Index("ix_earnings_events_announce_date", "announce_date"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
ticker_id: Mapped[int] = mapped_column(
|
||||||
|
ForeignKey("tickers.id", ondelete="CASCADE"), nullable=False
|
||||||
|
)
|
||||||
|
announce_date: Mapped[date] = mapped_column(Date, nullable=False)
|
||||||
|
# bmo | amc | unknown (source coverage is partial)
|
||||||
|
session: Mapped[str] = mapped_column(String(10), nullable=False, default="unknown")
|
||||||
|
period_end: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
eps_estimate: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
eps_actual: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
source: Mapped[str] = mapped_column(String(32), nullable=False)
|
||||||
|
import_run_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True
|
||||||
|
)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
|
)
|
||||||
|
|
||||||
|
ticker = relationship("Ticker", back_populates="earnings_events")
|
||||||
@@ -0,0 +1,79 @@
|
|||||||
|
from datetime import date, datetime
|
||||||
|
|
||||||
|
from sqlalchemy import Date, DateTime, Float, ForeignKey, Index, String, UniqueConstraint
|
||||||
|
from sqlalchemy.orm import Mapped, mapped_column
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
|
||||||
|
|
||||||
|
class FundamentalSnapshot(Base):
|
||||||
|
"""CIK-keyed, one immutable row per SEC accession.
|
||||||
|
|
||||||
|
Keyed by issuer (CIK), not ticker — multi-class issuers (GOOG/GOOGL) share
|
||||||
|
one CIK and one set of fundamentals; the ``tickers.cik`` column is the only
|
||||||
|
join point. Amendments are retained: every accession is a distinct immutable
|
||||||
|
row, and readers pick the newest valid ``accepted_at`` per
|
||||||
|
(cik, fiscal_year, fiscal_period) at read time — no flags, no mutation.
|
||||||
|
|
||||||
|
**Facts are stored as the filing reports them, never as derived quarters.**
|
||||||
|
Duration facts (revenue, net_income, operating_income, diluted_eps, cfo,
|
||||||
|
capex, depreciation_amortization) hold the filing's normalized **cumulative
|
||||||
|
YTD/FY** value over (period_start -> period_end). Balance-sheet facts
|
||||||
|
(cash_and_st_investments, total_debt, shares_outstanding) are **period-end**
|
||||||
|
values. ``shares_outstanding`` is a single consolidated point-in-time count —
|
||||||
|
the ``dei:EntityCommonStockSharesOutstanding`` cover-page fact, or
|
||||||
|
``us-gaap:CommonStockSharesOutstanding`` at period end when no dei fact exists
|
||||||
|
(e.g. Alphabet). It is never a class sum (companyfacts is non-dimensional) nor
|
||||||
|
the weighted-average diluted count, since both consumers (estimated market cap,
|
||||||
|
YoY dilution read) want a point-in-time value. Discrete quarters (10-Q YTD
|
||||||
|
deltas, Q4 = FY - Q1..Q3), TTM, YoY and
|
||||||
|
the quarter tape are all derived at read time — so non-calendar fiscal years
|
||||||
|
resolve correctly and a later amendment never leaves a stale frozen quarter.
|
||||||
|
"""
|
||||||
|
|
||||||
|
__tablename__ = "fundamental_snapshots"
|
||||||
|
__table_args__ = (
|
||||||
|
UniqueConstraint("accession", name="uq_fundamental_snapshots_accession"),
|
||||||
|
Index("ix_fundamental_snapshots_cik_period", "cik", "fiscal_year", "fiscal_period"),
|
||||||
|
Index("ix_fundamental_snapshots_cik_period_end", "cik", "period_end"),
|
||||||
|
)
|
||||||
|
|
||||||
|
id: Mapped[int] = mapped_column(primary_key=True)
|
||||||
|
cik: Mapped[str] = mapped_column(String(10), nullable=False)
|
||||||
|
accession: Mapped[str] = mapped_column(String(25), nullable=False)
|
||||||
|
form: Mapped[str] = mapped_column(String(12), nullable=False) # 10-Q, 10-K, 10-K/A ...
|
||||||
|
filed_date: Mapped[date] = mapped_column(Date, nullable=False)
|
||||||
|
# Kept although PIT enforcement is deferred (one timestamp now vs painful retrofit).
|
||||||
|
accepted_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
||||||
|
|
||||||
|
# Period identity — required to align non-calendar fiscal years and to derive
|
||||||
|
# discrete quarters from cumulative facts.
|
||||||
|
period_start: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
period_end: Mapped[date] = mapped_column(Date, nullable=False)
|
||||||
|
fiscal_year: Mapped[int] = mapped_column(nullable=False)
|
||||||
|
fiscal_period: Mapped[str] = mapped_column(String(4), nullable=False) # Q1|Q2|Q3|Q4|FY
|
||||||
|
|
||||||
|
# Duration facts — cumulative YTD/FY over (period_start -> period_end).
|
||||||
|
revenue: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
net_income: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
operating_income: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
diluted_eps: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
cfo: Mapped[float | None] = mapped_column(Float, nullable=True) # cash flow from operations
|
||||||
|
capex: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
depreciation_amortization: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
|
||||||
|
# Balance-sheet facts — period-end values.
|
||||||
|
cash_and_st_investments: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
total_debt: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
shares_outstanding: Mapped[float | None] = mapped_column(Float, nullable=True)
|
||||||
|
# The cover-page share count (dei:EntityCommonStockSharesOutstanding) is
|
||||||
|
# reported "as of" its own date, which can differ from period_end — store it
|
||||||
|
# so market cap uses the right point-in-time count.
|
||||||
|
shares_outstanding_date: Mapped[date | None] = mapped_column(Date, nullable=True)
|
||||||
|
|
||||||
|
import_run_id: Mapped[int | None] = mapped_column(
|
||||||
|
ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True
|
||||||
|
)
|
||||||
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
|
)
|
||||||
@@ -14,6 +14,13 @@ class Ticker(Base):
|
|||||||
# Company name (e.g. "Biogen Inc."); backfilled from Alpaca, nullable for
|
# Company name (e.g. "Biogen Inc."); backfilled from Alpaca, nullable for
|
||||||
# symbols Alpaca doesn't know.
|
# symbols Alpaca doesn't know.
|
||||||
name: Mapped[str | None] = mapped_column(String(120), nullable=True)
|
name: Mapped[str | None] = mapped_column(String(120), nullable=True)
|
||||||
|
# SEC issuer identity, refreshed by the SEC fundamentals import from
|
||||||
|
# company_tickers.json / submissions. The only ticker<->issuer join point;
|
||||||
|
# multi-class tickers (GOOG/GOOGL) share these values. Nullable: not every
|
||||||
|
# symbol resolves to a CIK (e.g. ADRs, foreign issuers not in SEC data).
|
||||||
|
cik: Mapped[str | None] = mapped_column(String(10), nullable=True)
|
||||||
|
sic: Mapped[str | None] = mapped_column(String(4), nullable=True)
|
||||||
|
sic_description: Mapped[str | None] = mapped_column(String(160), nullable=True)
|
||||||
created_at: Mapped[datetime] = mapped_column(
|
created_at: Mapped[datetime] = mapped_column(
|
||||||
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
DateTime(timezone=True), default=datetime.utcnow, nullable=False
|
||||||
)
|
)
|
||||||
@@ -28,3 +35,4 @@ class Ticker(Base):
|
|||||||
trade_setups = relationship("TradeSetup", back_populates="ticker", cascade="all, delete-orphan")
|
trade_setups = relationship("TradeSetup", back_populates="ticker", cascade="all, delete-orphan")
|
||||||
watchlist_entries = relationship("WatchlistEntry", back_populates="ticker", cascade="all, delete-orphan")
|
watchlist_entries = relationship("WatchlistEntry", back_populates="ticker", cascade="all, delete-orphan")
|
||||||
ingestion_progress = relationship("IngestionProgress", back_populates="ticker", cascade="all, delete-orphan", uselist=False)
|
ingestion_progress = relationship("IngestionProgress", back_populates="ticker", cascade="all, delete-orphan", uselist=False)
|
||||||
|
earnings_events = relationship("EarningsEvent", back_populates="ticker", cascade="all, delete-orphan")
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ from app.dependencies import get_db, require_access
|
|||||||
from app.schemas.common import APIEnvelope
|
from app.schemas.common import APIEnvelope
|
||||||
from app.schemas.fundamental import FundamentalResponse
|
from app.schemas.fundamental import FundamentalResponse
|
||||||
from app.services.fundamental_service import get_fundamental
|
from app.services.fundamental_service import get_fundamental
|
||||||
|
from app.services.fundamentals_api_service import build_fundamentals_v1
|
||||||
|
|
||||||
router = APIRouter(tags=["fundamentals"])
|
router = APIRouter(tags=["fundamentals"])
|
||||||
|
|
||||||
@@ -30,14 +31,13 @@ async def read_fundamentals(
|
|||||||
_user=Depends(require_access),
|
_user=Depends(require_access),
|
||||||
db: AsyncSession = Depends(get_db),
|
db: AsyncSession = Depends(get_db),
|
||||||
) -> APIEnvelope:
|
) -> APIEnvelope:
|
||||||
"""Get latest fundamental data for a symbol."""
|
"""Get latest fundamental data for a symbol (legacy fields + additive v1)."""
|
||||||
record = await get_fundamental(db, symbol)
|
record = await get_fundamental(db, symbol)
|
||||||
|
v1 = await build_fundamentals_v1(db, symbol)
|
||||||
|
|
||||||
if record is None:
|
legacy: dict = {}
|
||||||
data = FundamentalResponse(symbol=symbol.strip().upper())
|
if record is not None:
|
||||||
else:
|
legacy = dict(
|
||||||
data = FundamentalResponse(
|
|
||||||
symbol=symbol.strip().upper(),
|
|
||||||
pe_ratio=record.pe_ratio,
|
pe_ratio=record.pe_ratio,
|
||||||
revenue_growth=record.revenue_growth,
|
revenue_growth=record.revenue_growth,
|
||||||
earnings_surprise=record.earnings_surprise,
|
earnings_surprise=record.earnings_surprise,
|
||||||
@@ -47,4 +47,5 @@ async def read_fundamentals(
|
|||||||
unavailable_fields=_parse_unavailable_fields(record.unavailable_fields_json),
|
unavailable_fields=_parse_unavailable_fields(record.unavailable_fields_json),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
data = FundamentalResponse(symbol=symbol.strip().upper(), **legacy, **v1)
|
||||||
return APIEnvelope(status="success", data=data.model_dump())
|
return APIEnvelope(status="success", data=data.model_dump())
|
||||||
|
|||||||
@@ -41,6 +41,9 @@ from app.services import (
|
|||||||
settings_store,
|
settings_store,
|
||||||
shadow_book_service,
|
shadow_book_service,
|
||||||
)
|
)
|
||||||
|
from app.services.data_import import STATUS_FAILED, SourceImporter, run_import
|
||||||
|
from app.services.dolt_earnings_importer import DoltEarningsImporter
|
||||||
|
from app.services.sec_fundamentals_importer import SecFundamentalsImporter
|
||||||
from app.services.alert_service import dispatch_alerts
|
from app.services.alert_service import dispatch_alerts
|
||||||
from app.services.backtest_service import (
|
from app.services.backtest_service import (
|
||||||
BACKTEST_TARGET_MODELS,
|
BACKTEST_TARGET_MODELS,
|
||||||
@@ -93,6 +96,8 @@ _JOB_NAMES = [
|
|||||||
"data_backfill",
|
"data_backfill",
|
||||||
"sentiment_collector",
|
"sentiment_collector",
|
||||||
"fundamental_collector",
|
"fundamental_collector",
|
||||||
|
"dolt_earnings_import",
|
||||||
|
"sec_fundamentals_import",
|
||||||
"rr_scanner",
|
"rr_scanner",
|
||||||
"ticker_universe_sync",
|
"ticker_universe_sync",
|
||||||
"alerts",
|
"alerts",
|
||||||
@@ -912,6 +917,70 @@ async def collect_fundamentals() -> None:
|
|||||||
_runtime_finish(job_name, "error", processed=processed, total=total, message=str(exc))
|
_runtime_finish(job_name, "error", processed=processed, total=total, message=str(exc))
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Jobs: shadow fundamentals sources
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
async def _run_shadow_import(job_name: str, importer: SourceImporter) -> None:
|
||||||
|
"""Run one source importer and surface its audit result in Admin → Jobs."""
|
||||||
|
_log_event(logging.INFO, "job_start", job=job_name)
|
||||||
|
_runtime_start(job_name, total=1)
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with async_session_factory() as db:
|
||||||
|
if not await _is_job_enabled(db, job_name):
|
||||||
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="disabled")
|
||||||
|
_runtime_finish(job_name, "skipped", processed=0, total=1, message="Disabled")
|
||||||
|
return
|
||||||
|
|
||||||
|
run = await run_import(importer)
|
||||||
|
if run is None:
|
||||||
|
message = "Another import for this source is already running"
|
||||||
|
_log_event(logging.INFO, "job_skipped", job=job_name, reason="source_locked")
|
||||||
|
_runtime_finish(job_name, "skipped", processed=0, total=1, message=message)
|
||||||
|
return
|
||||||
|
|
||||||
|
revision = f" · {run.revision[:12]}" if run.revision else ""
|
||||||
|
message = f"{run.status}{revision}"
|
||||||
|
if run.status == STATUS_FAILED:
|
||||||
|
message = run.error_details or message
|
||||||
|
_log_event(logging.ERROR, "job_error", job=job_name, message=message)
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=message)
|
||||||
|
return
|
||||||
|
|
||||||
|
_log_event(
|
||||||
|
logging.INFO,
|
||||||
|
"job_complete",
|
||||||
|
job=job_name,
|
||||||
|
import_status=run.status,
|
||||||
|
revision=run.revision,
|
||||||
|
)
|
||||||
|
_runtime_finish(job_name, "completed", processed=1, total=1, message=message)
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message="Cancelled")
|
||||||
|
raise
|
||||||
|
except Exception as exc:
|
||||||
|
_log_event(
|
||||||
|
logging.ERROR,
|
||||||
|
"job_error",
|
||||||
|
job=job_name,
|
||||||
|
error_type=type(exc).__name__,
|
||||||
|
message=str(exc),
|
||||||
|
)
|
||||||
|
_runtime_finish(job_name, "error", processed=0, total=1, message=str(exc))
|
||||||
|
|
||||||
|
|
||||||
|
async def run_dolt_earnings_import() -> None:
|
||||||
|
"""Pull and import the Dolt earnings calendar/results feed in shadow."""
|
||||||
|
await _run_shadow_import("dolt_earnings_import", DoltEarningsImporter())
|
||||||
|
|
||||||
|
|
||||||
|
async def run_sec_fundamentals_import() -> None:
|
||||||
|
"""Import tracked-universe SEC facts in shadow."""
|
||||||
|
await _run_shadow_import("sec_fundamentals_import", SecFundamentalsImporter())
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Job: R:R Scanner
|
# Job: R:R Scanner
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -1452,6 +1521,9 @@ SCHEDULE_DEFAULTS: dict[str, str] = {
|
|||||||
"schedule_timezone": "America/New_York",
|
"schedule_timezone": "America/New_York",
|
||||||
# Morning data/display refresh (no qualifying R:R scan).
|
# Morning data/display refresh (no qualifying R:R scan).
|
||||||
"schedule_daily_pipeline_cron": "0 2 * * *",
|
"schedule_daily_pipeline_cron": "0 2 * * *",
|
||||||
|
# Shadow source imports. They never write legacy fundamental_data before A5.
|
||||||
|
"schedule_dolt_earnings_cron": "30 2 * * *",
|
||||||
|
"schedule_sec_fundamentals_cron": "0 4 * * *",
|
||||||
# Fetch in-progress bars → scan → Telegram (manual MOC window).
|
# Fetch in-progress bars → scan → Telegram (manual MOC window).
|
||||||
"schedule_near_close_pipeline_cron": "30 15 * * mon-fri",
|
"schedule_near_close_pipeline_cron": "30 15 * * mon-fri",
|
||||||
# Fetch final bars → outcome eval (must not run on the partial near-close bar).
|
# Fetch final bars → outcome eval (must not run on the partial near-close bar).
|
||||||
@@ -1465,6 +1537,8 @@ SCHEDULE_DEFAULTS: dict[str, str] = {
|
|||||||
# job id -> schedule setting key
|
# job id -> schedule setting key
|
||||||
_CRON_JOBS: dict[str, str] = {
|
_CRON_JOBS: dict[str, str] = {
|
||||||
"daily_pipeline": "schedule_daily_pipeline_cron",
|
"daily_pipeline": "schedule_daily_pipeline_cron",
|
||||||
|
"dolt_earnings_import": "schedule_dolt_earnings_cron",
|
||||||
|
"sec_fundamentals_import": "schedule_sec_fundamentals_cron",
|
||||||
"near_close_pipeline": "schedule_near_close_pipeline_cron",
|
"near_close_pipeline": "schedule_near_close_pipeline_cron",
|
||||||
"after_close_pipeline": "schedule_after_close_pipeline_cron",
|
"after_close_pipeline": "schedule_after_close_pipeline_cron",
|
||||||
"intraday_pipeline": "schedule_intraday_pipeline_cron",
|
"intraday_pipeline": "schedule_intraday_pipeline_cron",
|
||||||
@@ -1549,6 +1623,28 @@ def configure_scheduler(schedule_config: dict[str, str] | None = None) -> None:
|
|||||||
_cron_trigger(cfg["schedule_daily_pipeline_cron"], tz, "schedule_daily_pipeline_cron"),
|
_cron_trigger(cfg["schedule_daily_pipeline_cron"], tz, "schedule_daily_pipeline_cron"),
|
||||||
id="daily_pipeline", name="Morning Pipeline", replace_existing=True,
|
id="daily_pipeline", name="Morning Pipeline", replace_existing=True,
|
||||||
)
|
)
|
||||||
|
scheduler.add_job(
|
||||||
|
run_dolt_earnings_import,
|
||||||
|
_cron_trigger(
|
||||||
|
cfg["schedule_dolt_earnings_cron"],
|
||||||
|
tz,
|
||||||
|
"schedule_dolt_earnings_cron",
|
||||||
|
),
|
||||||
|
id="dolt_earnings_import",
|
||||||
|
name="Dolt Earnings Import (shadow)",
|
||||||
|
replace_existing=True,
|
||||||
|
)
|
||||||
|
scheduler.add_job(
|
||||||
|
run_sec_fundamentals_import,
|
||||||
|
_cron_trigger(
|
||||||
|
cfg["schedule_sec_fundamentals_cron"],
|
||||||
|
tz,
|
||||||
|
"schedule_sec_fundamentals_cron",
|
||||||
|
),
|
||||||
|
id="sec_fundamentals_import",
|
||||||
|
name="SEC Fundamentals Import (shadow)",
|
||||||
|
replace_existing=True,
|
||||||
|
)
|
||||||
scheduler.add_job(
|
scheduler.add_job(
|
||||||
run_near_close_pipeline,
|
run_near_close_pipeline,
|
||||||
_cron_trigger(
|
_cron_trigger(
|
||||||
@@ -1622,6 +1718,8 @@ def configure_scheduler(schedule_config: dict[str, str] | None = None) -> None:
|
|||||||
"cron": cfg["schedule_daily_pipeline_cron"],
|
"cron": cfg["schedule_daily_pipeline_cron"],
|
||||||
"steps": [name for name, _ in _DAILY_PIPELINE_STEPS],
|
"steps": [name for name, _ in _DAILY_PIPELINE_STEPS],
|
||||||
},
|
},
|
||||||
|
dolt_earnings_import={"cron": cfg["schedule_dolt_earnings_cron"]},
|
||||||
|
sec_fundamentals_import={"cron": cfg["schedule_sec_fundamentals_cron"]},
|
||||||
near_close_pipeline={
|
near_close_pipeline={
|
||||||
"cron": cfg["schedule_near_close_pipeline_cron"],
|
"cron": cfg["schedule_near_close_pipeline_cron"],
|
||||||
"steps": [name for name, _ in _NEAR_CLOSE_PIPELINE_STEPS],
|
"steps": [name for name, _ in _NEAR_CLOSE_PIPELINE_STEPS],
|
||||||
|
|||||||
@@ -7,8 +7,74 @@ from datetime import date, datetime
|
|||||||
from pydantic import BaseModel
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
|
||||||
|
class MetricIndustry(BaseModel):
|
||||||
|
label: str
|
||||||
|
median: float
|
||||||
|
favorable_percentile: int # 0-100, polarity-aware (higher = more favorable)
|
||||||
|
peer_count: int
|
||||||
|
|
||||||
|
|
||||||
|
class MetricHistoryPoint(BaseModel):
|
||||||
|
period_end: str # YYYY-MM-DD
|
||||||
|
value: float | None
|
||||||
|
|
||||||
|
|
||||||
|
class MetricItem(BaseModel):
|
||||||
|
key: str
|
||||||
|
value: float | None = None
|
||||||
|
history: list[MetricHistoryPoint] = []
|
||||||
|
industry: MetricIndustry | None = None
|
||||||
|
period_end: str | None = None
|
||||||
|
filed_date: str | None = None
|
||||||
|
source: str = "sec"
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsNext(BaseModel):
|
||||||
|
date: str
|
||||||
|
session: str
|
||||||
|
days_until: int
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsRecent(BaseModel):
|
||||||
|
announce_date: str
|
||||||
|
period_end: str | None = None
|
||||||
|
eps_estimate: float | None = None
|
||||||
|
eps_actual: float | None = None
|
||||||
|
surprise_pct: float | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class EarningsObject(BaseModel):
|
||||||
|
next: EarningsNext | None = None
|
||||||
|
recent: list[EarningsRecent] = []
|
||||||
|
|
||||||
|
|
||||||
|
class Valuation(BaseModel):
|
||||||
|
pe: float | None = None
|
||||||
|
fcf_yield: float | None = None
|
||||||
|
market_cap_est: float | None = None
|
||||||
|
pe_industry: MetricIndustry | None = None
|
||||||
|
fcf_yield_industry: MetricIndustry | None = None
|
||||||
|
price_date: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class FundamentalsReads(BaseModel):
|
||||||
|
"""Deterministic text outputs, separate from the numeric metrics.
|
||||||
|
|
||||||
|
``by_key`` is a fixed map over every metric key plus ``pe`` and ``fcf_yield``,
|
||||||
|
each a read string or null. ``header`` is null when there is no read at all."""
|
||||||
|
|
||||||
|
header: str | None = None
|
||||||
|
by_key: dict[str, str | None] = {}
|
||||||
|
|
||||||
|
|
||||||
class FundamentalResponse(BaseModel):
|
class FundamentalResponse(BaseModel):
|
||||||
"""Envelope-ready fundamental data response."""
|
"""Envelope-ready fundamental data response.
|
||||||
|
|
||||||
|
Legacy fields are preserved unchanged (they come from ``fundamental_data`` /
|
||||||
|
the legacy providers). The additive v1 objects — earnings, metrics, valuation,
|
||||||
|
reads — are SEC/Dolt-derived and independent; a null legacy field is never
|
||||||
|
mapped onto the new SEC metrics and vice-versa.
|
||||||
|
"""
|
||||||
|
|
||||||
symbol: str
|
symbol: str
|
||||||
pe_ratio: float | None = None
|
pe_ratio: float | None = None
|
||||||
@@ -18,3 +84,9 @@ class FundamentalResponse(BaseModel):
|
|||||||
next_earnings_date: date | None = None
|
next_earnings_date: date | None = None
|
||||||
fetched_at: datetime | None = None
|
fetched_at: datetime | None = None
|
||||||
unavailable_fields: dict[str, str] = {}
|
unavailable_fields: dict[str, str] = {}
|
||||||
|
|
||||||
|
# --- additive v1 (always present; empty/null when unavailable) ---
|
||||||
|
earnings: EarningsObject | None = None
|
||||||
|
metrics: list[MetricItem] | None = None
|
||||||
|
valuation: Valuation | None = None
|
||||||
|
reads: FundamentalsReads | None = None
|
||||||
|
|||||||
@@ -612,6 +612,8 @@ VALID_JOB_NAMES = {
|
|||||||
"benchmark_collector",
|
"benchmark_collector",
|
||||||
"sentiment_collector",
|
"sentiment_collector",
|
||||||
"fundamental_collector",
|
"fundamental_collector",
|
||||||
|
"dolt_earnings_import",
|
||||||
|
"sec_fundamentals_import",
|
||||||
"rr_scanner",
|
"rr_scanner",
|
||||||
"ticker_universe_sync",
|
"ticker_universe_sync",
|
||||||
"outcome_evaluator",
|
"outcome_evaluator",
|
||||||
@@ -633,6 +635,8 @@ JOB_LABELS = {
|
|||||||
"benchmark_collector": "Benchmark Collector",
|
"benchmark_collector": "Benchmark Collector",
|
||||||
"sentiment_collector": "Sentiment Collector",
|
"sentiment_collector": "Sentiment Collector",
|
||||||
"fundamental_collector": "Fundamental Collector",
|
"fundamental_collector": "Fundamental Collector",
|
||||||
|
"dolt_earnings_import": "Dolt Earnings Import (shadow)",
|
||||||
|
"sec_fundamentals_import": "SEC Fundamentals Import (shadow)",
|
||||||
"rr_scanner": "R:R Scanner",
|
"rr_scanner": "R:R Scanner",
|
||||||
"ticker_universe_sync": "Ticker Universe Sync",
|
"ticker_universe_sync": "Ticker Universe Sync",
|
||||||
"outcome_evaluator": "Outcome Evaluator",
|
"outcome_evaluator": "Outcome Evaluator",
|
||||||
|
|||||||
@@ -0,0 +1,270 @@
|
|||||||
|
"""Source-agnostic batch import framework (Dolt/SEC bulk data → PostgreSQL).
|
||||||
|
|
||||||
|
Every bulk importer (SEC facts, Dolt earnings, later Dolt stocks) plugs into
|
||||||
|
``run_import`` and gets, for free, the plan's non-negotiables:
|
||||||
|
|
||||||
|
- **One run per source at a time** — a Postgres *session-level* advisory lock
|
||||||
|
keyed by source. It is held on a single pinned connection for the whole run,
|
||||||
|
so it survives the intermediate commits (the ``running`` row, then the
|
||||||
|
promotion) and only releases at the end. No-op on non-Postgres (tests).
|
||||||
|
- **Idempotent per revision** — the cheap ``detect_revision`` probe is compared
|
||||||
|
against the last *promoted* run; an unchanged revision records a ``no_op``
|
||||||
|
with **zero row changes** (no expensive fetch, no writes).
|
||||||
|
- **Staging then atomic promotion** — the importer stages into an in-memory
|
||||||
|
object (no physical staging tables), validation reads it, and only a passing
|
||||||
|
run calls ``promote`` whose writes commit together with the run-row flip to
|
||||||
|
``promoted`` in a single transaction.
|
||||||
|
- **Failure is inert** — a failed validation or a mid-run exception marks the
|
||||||
|
run ``failed``, alerts via the system-events path, and leaves the live tables
|
||||||
|
exactly as they were (nothing is written before ``promote``).
|
||||||
|
|
||||||
|
Every attempt — promoted, no_op, or failed — is recorded in ``data_import_runs``.
|
||||||
|
KISS: no conflicts table (summaries go in ``validation_json``), no revision
|
||||||
|
table (idempotency queries the last run), no aggregate tables.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
from typing import Any, Protocol, runtime_checkable
|
||||||
|
|
||||||
|
from sqlalchemy import select, text
|
||||||
|
from sqlalchemy.engine import Engine # noqa: F401 (typing only)
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncEngine, AsyncSession
|
||||||
|
|
||||||
|
from app.database import engine as app_engine
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
|
from app.services import system_event_service
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# data_import_runs.status values
|
||||||
|
STATUS_RUNNING = "running"
|
||||||
|
STATUS_VALIDATED = "validated"
|
||||||
|
STATUS_PROMOTED = "promoted"
|
||||||
|
STATUS_NO_OP = "no_op"
|
||||||
|
STATUS_FAILED = "failed"
|
||||||
|
|
||||||
|
_MAX_ERROR_LEN = 4000
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ValidationResult:
|
||||||
|
"""Outcome of an importer's validation gates.
|
||||||
|
|
||||||
|
``summary`` is serialized into ``validation_json`` (reconciliation /
|
||||||
|
discrepancy details live here — no separate conflicts table). ``validate``
|
||||||
|
MUST be read-only: it reads the staged object and, if needed, live tables
|
||||||
|
for comparison, but writes nothing — that invariant is what makes a failed
|
||||||
|
run leave the dataset untouched.
|
||||||
|
"""
|
||||||
|
|
||||||
|
ok: bool
|
||||||
|
summary: dict[str, Any] = field(default_factory=dict)
|
||||||
|
source_max_date: date | None = None
|
||||||
|
messages: list[str] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
@runtime_checkable
|
||||||
|
class SourceImporter(Protocol):
|
||||||
|
"""Interface a concrete bulk importer implements. All methods receive the
|
||||||
|
session bound to the lock-holding connection; ``stage`` and ``validate``
|
||||||
|
never write to live tables, only ``promote`` does."""
|
||||||
|
|
||||||
|
source: str # sec_facts | dolt_earnings | dolt_stocks
|
||||||
|
|
||||||
|
async def detect_revision(self, db: AsyncSession) -> str | None:
|
||||||
|
"""Cheap probe of the source revision (Dolt commit / SEC archive SHA).
|
||||||
|
|
||||||
|
Returns the revision id, or None when it can't be determined cheaply
|
||||||
|
(in which case idempotency is skipped and the run always stages)."""
|
||||||
|
...
|
||||||
|
|
||||||
|
async def stage(self, db: AsyncSession) -> Any:
|
||||||
|
"""Download/parse into an in-memory staged representation. No writes to
|
||||||
|
live tables."""
|
||||||
|
...
|
||||||
|
|
||||||
|
async def validate(self, db: AsyncSession, staged: Any) -> ValidationResult:
|
||||||
|
"""Run the source's validation gates against ``staged``. Read-only."""
|
||||||
|
...
|
||||||
|
|
||||||
|
async def promote(self, db: AsyncSession, staged: Any, run_id: int) -> dict[str, int]:
|
||||||
|
"""Apply ``staged`` to the live tables. Called inside the promotion
|
||||||
|
transaction; the caller commits. ``run_id`` is the current
|
||||||
|
``data_import_runs.id`` so written rows can be stamped with their
|
||||||
|
``import_run_id``. Returns row-count deltas."""
|
||||||
|
...
|
||||||
|
|
||||||
|
|
||||||
|
def _advisory_key(source: str) -> int:
|
||||||
|
"""Deterministic signed 64-bit key for a source's advisory lock."""
|
||||||
|
digest = hashlib.blake2b(source.encode("utf-8"), digest_size=8).digest()
|
||||||
|
return int.from_bytes(digest, "big", signed=True)
|
||||||
|
|
||||||
|
|
||||||
|
async def _last_promoted_revision(db: AsyncSession, source: str) -> str | None:
|
||||||
|
"""Revision of the most recent *promoted* run for ``source`` (the revision
|
||||||
|
currently loaded), or None if none has promoted yet."""
|
||||||
|
row = await db.execute(
|
||||||
|
select(DataImportRun.revision)
|
||||||
|
.where(
|
||||||
|
DataImportRun.source == source,
|
||||||
|
DataImportRun.status == STATUS_PROMOTED,
|
||||||
|
)
|
||||||
|
.order_by(DataImportRun.id.desc())
|
||||||
|
.limit(1)
|
||||||
|
)
|
||||||
|
return row.scalar_one_or_none()
|
||||||
|
|
||||||
|
|
||||||
|
def _now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
async def _alert(db: AsyncSession, source: str, code: str, messages: list[str]) -> None:
|
||||||
|
try:
|
||||||
|
await system_event_service.log_event(
|
||||||
|
db,
|
||||||
|
severity="error",
|
||||||
|
source="data_import",
|
||||||
|
code=f"{source}_{code}",
|
||||||
|
message=(("; ".join(messages)) or code)[:_MAX_ERROR_LEN],
|
||||||
|
dedup_key=f"data_import:{source}:{code}",
|
||||||
|
)
|
||||||
|
except Exception: # noqa: BLE001 — alerting must never mask the real outcome
|
||||||
|
logger.exception("Failed to emit data_import alert %s/%s", source, code)
|
||||||
|
|
||||||
|
|
||||||
|
async def run_import(
|
||||||
|
importer: SourceImporter,
|
||||||
|
*,
|
||||||
|
engine: AsyncEngine | None = None,
|
||||||
|
) -> DataImportRun | None:
|
||||||
|
"""Run one import for ``importer``.
|
||||||
|
|
||||||
|
Returns the recorded ``DataImportRun`` (promoted / no_op / failed), or None
|
||||||
|
when the per-source advisory lock is already held (another run is active).
|
||||||
|
"""
|
||||||
|
engine = engine or app_engine
|
||||||
|
source = importer.source
|
||||||
|
is_pg = engine.dialect.name == "postgresql"
|
||||||
|
key = _advisory_key(source)
|
||||||
|
|
||||||
|
async with engine.connect() as conn:
|
||||||
|
# Bind the session to this one connection so the session-level advisory
|
||||||
|
# lock persists across our commits. expire_on_commit must be set here —
|
||||||
|
# the app factory's setting doesn't carry to a directly-built session.
|
||||||
|
session = AsyncSession(bind=conn, expire_on_commit=False)
|
||||||
|
try:
|
||||||
|
if is_pg:
|
||||||
|
got = (
|
||||||
|
await session.execute(
|
||||||
|
text("SELECT pg_try_advisory_lock(:k)"), {"k": key}
|
||||||
|
)
|
||||||
|
).scalar()
|
||||||
|
await session.commit()
|
||||||
|
if not got:
|
||||||
|
logger.info("data_import %s: lock held, skipping", source)
|
||||||
|
return None
|
||||||
|
|
||||||
|
# Record the attempt FIRST — before the external revision probe, the
|
||||||
|
# most likely failure — so anything below is recorded and alerted and
|
||||||
|
# never escapes unrecorded. Revision is filled in once detected.
|
||||||
|
run = DataImportRun(
|
||||||
|
source=source,
|
||||||
|
status=STATUS_RUNNING,
|
||||||
|
started_at=_now(),
|
||||||
|
)
|
||||||
|
session.add(run)
|
||||||
|
await session.commit()
|
||||||
|
await session.refresh(run)
|
||||||
|
|
||||||
|
try:
|
||||||
|
revision = await importer.detect_revision(session)
|
||||||
|
run.revision = revision
|
||||||
|
last_rev = await _last_promoted_revision(session, source)
|
||||||
|
if revision is not None and revision == last_rev:
|
||||||
|
run.status = STATUS_NO_OP
|
||||||
|
run.completed_at = _now()
|
||||||
|
await session.commit()
|
||||||
|
logger.info("data_import %s: no_op (revision %s)", source, revision)
|
||||||
|
return run
|
||||||
|
|
||||||
|
staged = await importer.stage(session)
|
||||||
|
result = await importer.validate(session, staged)
|
||||||
|
run.source_max_date = result.source_max_date
|
||||||
|
run.validation_json = json.dumps(result.summary, default=str)
|
||||||
|
|
||||||
|
if not result.ok:
|
||||||
|
run.status = STATUS_FAILED
|
||||||
|
run.error_details = ("; ".join(result.messages))[:_MAX_ERROR_LEN]
|
||||||
|
run.completed_at = _now()
|
||||||
|
await session.commit()
|
||||||
|
await _alert(session, source, "validation_failed", result.messages)
|
||||||
|
logger.warning(
|
||||||
|
"data_import %s: validation failed: %s",
|
||||||
|
source,
|
||||||
|
result.messages,
|
||||||
|
)
|
||||||
|
return run
|
||||||
|
|
||||||
|
# Promotion: importer writes + run-row flip in one transaction.
|
||||||
|
row_counts = await importer.promote(session, staged, run.id)
|
||||||
|
run.status = STATUS_PROMOTED
|
||||||
|
run.row_counts_json = json.dumps(row_counts, default=str)
|
||||||
|
run.completed_at = _now()
|
||||||
|
await session.commit()
|
||||||
|
await session.refresh(run)
|
||||||
|
logger.info(
|
||||||
|
"data_import %s: promoted (revision %s, rows %s)",
|
||||||
|
source,
|
||||||
|
revision,
|
||||||
|
row_counts,
|
||||||
|
)
|
||||||
|
return run
|
||||||
|
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
# Deploy / scheduler shutdown: best-effort mark failed so no
|
||||||
|
# ``running`` row lingers, then let the cancellation propagate —
|
||||||
|
# never swallow it.
|
||||||
|
try:
|
||||||
|
await session.rollback()
|
||||||
|
run.status = STATUS_FAILED
|
||||||
|
run.error_details = "cancelled"
|
||||||
|
run.completed_at = _now()
|
||||||
|
await session.commit()
|
||||||
|
except BaseException: # noqa: BLE001 — best-effort during teardown
|
||||||
|
logger.warning(
|
||||||
|
"data_import %s: could not record cancellation", source
|
||||||
|
)
|
||||||
|
raise
|
||||||
|
|
||||||
|
except Exception as exc: # noqa: BLE001 — record + alert, don't crash the job
|
||||||
|
await session.rollback()
|
||||||
|
run.status = STATUS_FAILED
|
||||||
|
run.error_details = repr(exc)[:_MAX_ERROR_LEN]
|
||||||
|
run.completed_at = _now()
|
||||||
|
try:
|
||||||
|
await session.commit()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
logger.exception("data_import %s: failed to record failure", source)
|
||||||
|
await _alert(session, source, "import_error", [repr(exc)])
|
||||||
|
logger.exception("data_import %s: import error", source)
|
||||||
|
return run
|
||||||
|
|
||||||
|
finally:
|
||||||
|
if is_pg:
|
||||||
|
try:
|
||||||
|
await session.execute(
|
||||||
|
text("SELECT pg_advisory_unlock(:k)"), {"k": key}
|
||||||
|
)
|
||||||
|
await session.commit()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
logger.exception("data_import %s: failed to release lock", source)
|
||||||
|
await session.close()
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
"""Minimal async client for a local Dolt clone.
|
||||||
|
|
||||||
|
The application never runs a long-lived Dolt sql-server; it shells out to the
|
||||||
|
`dolt` CLI against a persistent clone and reads results as CSV. Every call goes
|
||||||
|
through ``asyncio.create_subprocess_exec`` because the scheduler shares one event
|
||||||
|
loop with the API (`app/scheduler.py:73`) — a blocking `subprocess.run` here
|
||||||
|
would stall request handling.
|
||||||
|
|
||||||
|
Production keeps the clone in ``DOLT_DATA_DIR`` outside the deploy tree; the
|
||||||
|
binary path and data dir are configured (see ``app/config.py``). Read via
|
||||||
|
``dolt sql -r csv``; refresh with ``pull`` and record the resulting commit hash
|
||||||
|
as the import revision.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import csv
|
||||||
|
import io
|
||||||
|
import logging
|
||||||
|
import shutil
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Default subprocess timeout. A hung `dolt pull`/`sql` would otherwise pin the
|
||||||
|
# import's connection and its advisory lock indefinitely, so every call is
|
||||||
|
# bounded; callers may override per operation.
|
||||||
|
DEFAULT_TIMEOUT = 600.0
|
||||||
|
|
||||||
|
|
||||||
|
class DoltError(RuntimeError):
|
||||||
|
"""A dolt subprocess failed, timed out, or exited non-zero."""
|
||||||
|
|
||||||
|
|
||||||
|
async def _run(
|
||||||
|
binary: str, args: list[str], *, cwd: Path, timeout: float = DEFAULT_TIMEOUT
|
||||||
|
) -> str:
|
||||||
|
proc = await asyncio.create_subprocess_exec(
|
||||||
|
binary,
|
||||||
|
*args,
|
||||||
|
cwd=str(cwd),
|
||||||
|
stdout=asyncio.subprocess.PIPE,
|
||||||
|
stderr=asyncio.subprocess.PIPE,
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
stdout, stderr = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
proc.kill()
|
||||||
|
try:
|
||||||
|
await proc.wait()
|
||||||
|
except ProcessLookupError:
|
||||||
|
pass
|
||||||
|
raise DoltError(f"dolt {args[0] if args else ''} timed out after {timeout:.0f}s")
|
||||||
|
if proc.returncode != 0:
|
||||||
|
raise DoltError(
|
||||||
|
f"dolt {' '.join(args)} failed ({proc.returncode}): "
|
||||||
|
f"{stderr.decode('utf-8', 'replace').strip()[:500]}"
|
||||||
|
)
|
||||||
|
return stdout.decode("utf-8", "replace")
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_free_disk(path: Path, min_free_gb: float) -> None:
|
||||||
|
"""Raise if free space at ``path`` is below the threshold (checked before a
|
||||||
|
pull that could grow the clone). Uses the nearest existing ancestor so it
|
||||||
|
works before the clone dir exists."""
|
||||||
|
probe = path
|
||||||
|
while not probe.exists() and probe.parent != probe:
|
||||||
|
probe = probe.parent
|
||||||
|
free_gb = shutil.disk_usage(probe).free / (1024**3)
|
||||||
|
if free_gb < min_free_gb:
|
||||||
|
raise DoltError(
|
||||||
|
f"insufficient disk for dolt at {path}: {free_gb:.1f} GB free "
|
||||||
|
f"< {min_free_gb:.1f} GB required"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def pull(repo_dir: Path, *, binary: str, timeout: float = DEFAULT_TIMEOUT) -> None:
|
||||||
|
"""`dolt pull` the persistent clone to the latest upstream revision."""
|
||||||
|
await _run(binary, ["pull"], cwd=repo_dir, timeout=timeout)
|
||||||
|
|
||||||
|
|
||||||
|
async def current_commit(
|
||||||
|
repo_dir: Path, *, binary: str, timeout: float = DEFAULT_TIMEOUT
|
||||||
|
) -> str:
|
||||||
|
"""The HEAD commit hash of the clone — used as the import revision.
|
||||||
|
|
||||||
|
Uses ``DOLT_HASHOF('HEAD')`` (which formally identifies HEAD) rather than
|
||||||
|
ordering ``dolt_log`` by timestamp."""
|
||||||
|
rows = await query_csv(
|
||||||
|
repo_dir, "SELECT DOLT_HASHOF('HEAD') AS commit_hash", binary=binary, timeout=timeout
|
||||||
|
)
|
||||||
|
if not rows or not rows[0].get("commit_hash"):
|
||||||
|
raise DoltError("could not read HEAD commit hash")
|
||||||
|
return rows[0]["commit_hash"]
|
||||||
|
|
||||||
|
|
||||||
|
async def query_csv(
|
||||||
|
repo_dir: Path, sql: str, *, binary: str, timeout: float = DEFAULT_TIMEOUT
|
||||||
|
) -> list[dict[str, str]]:
|
||||||
|
"""Run a read query and parse the CSV result into a list of dict rows."""
|
||||||
|
out = await _run(binary, ["sql", "-q", sql, "-r", "csv"], cwd=repo_dir, timeout=timeout)
|
||||||
|
if not out.strip():
|
||||||
|
return []
|
||||||
|
return list(csv.DictReader(io.StringIO(out)))
|
||||||
@@ -0,0 +1,357 @@
|
|||||||
|
"""Production importer for the DoltHub post-no-preference/earnings calendar.
|
||||||
|
|
||||||
|
A ``SourceImporter`` (see ``app/services/data_import.py``) that pulls the local
|
||||||
|
Dolt clone, aligns the announcement calendar to the EPS history with the pure DP
|
||||||
|
in ``earnings_alignment`` (reused from the research script, not extending it),
|
||||||
|
and writes ``earnings_events`` for the tracked universe.
|
||||||
|
|
||||||
|
Shadow by construction: nothing reads ``earnings_events`` until the API/panel
|
||||||
|
lands (A4), so writing it does not touch production behavior.
|
||||||
|
|
||||||
|
**Promotion is destructive** — future-dated rows for this source are deleted and
|
||||||
|
re-inserted every run so reschedules/cancellations never linger. The forward
|
||||||
|
calendar is the project's acceptance gate, so ``validate`` is fail-closed: it
|
||||||
|
blocks promotion when the staged future set is empty or has collapsed relative
|
||||||
|
to what's already loaded.
|
||||||
|
|
||||||
|
Attribution: the earnings data is CC BY-SA 4.0 from post-no-preference/earnings.
|
||||||
|
See the repo ``NOTICE``. Internal use only — no redistribution.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from sqlalchemy import case, delete, func, select
|
||||||
|
|
||||||
|
from app.config import settings
|
||||||
|
from app.database import insert_for_session
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import dolt_client, earnings_alignment
|
||||||
|
from app.services.data_import import ValidationResult
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
SOURCE = "dolt_earnings"
|
||||||
|
|
||||||
|
# Earliest announcement date to import (matches the research backfill window).
|
||||||
|
WINDOW_START = date(2020, 1, 22)
|
||||||
|
# Alignment tolerances (research defaults): an announcement may lead its period
|
||||||
|
# end by up to 14 days or lag it by up to 90.
|
||||||
|
MAX_LAG_DAYS = 90
|
||||||
|
MAX_LEAD_DAYS = 14
|
||||||
|
# Fail promotion if the staged forward calendar drops below this fraction of the
|
||||||
|
# currently-loaded forward calendar (guards the destructive re-insert against a
|
||||||
|
# partial parse / symbol-mapping regression).
|
||||||
|
MIN_FUTURE_RATIO = 0.5
|
||||||
|
# Initial-load gates (when nothing is loaded yet — the ratio gate has no baseline).
|
||||||
|
# The source publishes a forward calendar; require a real horizon, not one stray
|
||||||
|
# future row. 21 days is a conservative floor under the ~35d horizon observed on
|
||||||
|
# the live clone.
|
||||||
|
MIN_FORWARD_HORIZON_DAYS = 21
|
||||||
|
# ...and require the symbol join to reach most of the tracked universe, so a
|
||||||
|
# broken/normalization-dropped join can't seed a hollow calendar.
|
||||||
|
MIN_INITIAL_COVERAGE = 0.5
|
||||||
|
|
||||||
|
_CAL_SQL = (
|
||||||
|
"SELECT act_symbol, `date`, `when` FROM earnings_calendar "
|
||||||
|
f"WHERE `date` >= '{WINDOW_START.isoformat()}'"
|
||||||
|
)
|
||||||
|
_HIST_SQL = (
|
||||||
|
"SELECT act_symbol, period_end_date, reported, estimate FROM eps_history "
|
||||||
|
f"WHERE period_end_date >= '{(WINDOW_START.replace(year=WINDOW_START.year - 1)).isoformat()}'"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class StagedEarnings:
|
||||||
|
rows: list[dict[str, Any]]
|
||||||
|
stats: dict[str, Any] = field(default_factory=dict)
|
||||||
|
future_count: int = 0
|
||||||
|
max_announce_date: date | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
class DoltEarningsImporter:
|
||||||
|
source = SOURCE
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
repo_dir: Path | str | None = None,
|
||||||
|
binary: str | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
do_pull: bool = True,
|
||||||
|
dolt: Any = dolt_client,
|
||||||
|
) -> None:
|
||||||
|
self.repo_dir = Path(
|
||||||
|
repo_dir
|
||||||
|
or (Path(settings.dolt_data_dir) / settings.dolt_earnings_subdir)
|
||||||
|
)
|
||||||
|
self.binary = binary or settings.dolt_binary
|
||||||
|
self.today = today or _now().date()
|
||||||
|
self.do_pull = do_pull
|
||||||
|
self._dolt = dolt # injectable for tests
|
||||||
|
|
||||||
|
# -- SourceImporter protocol -------------------------------------------
|
||||||
|
|
||||||
|
async def detect_revision(self, db) -> str | None:
|
||||||
|
timeout = settings.dolt_command_timeout_seconds
|
||||||
|
if self.do_pull:
|
||||||
|
dolt_client.ensure_free_disk(self.repo_dir, settings.dolt_min_free_disk_gb)
|
||||||
|
await self._dolt.pull(self.repo_dir, binary=self.binary, timeout=timeout)
|
||||||
|
return await self._dolt.current_commit(
|
||||||
|
self.repo_dir, binary=self.binary, timeout=timeout
|
||||||
|
)
|
||||||
|
|
||||||
|
async def stage(self, db) -> StagedEarnings:
|
||||||
|
universe = await self._load_universe(db) # {normalised symbol: ticker_id}
|
||||||
|
|
||||||
|
timeout = settings.dolt_command_timeout_seconds
|
||||||
|
cal_raw = await self._dolt.query_csv(
|
||||||
|
self.repo_dir, _CAL_SQL, binary=self.binary, timeout=timeout
|
||||||
|
)
|
||||||
|
hist_raw = await self._dolt.query_csv(
|
||||||
|
self.repo_dir, _HIST_SQL, binary=self.binary, timeout=timeout
|
||||||
|
)
|
||||||
|
_require_columns(cal_raw, {"act_symbol", "date", "when"}, "earnings_calendar")
|
||||||
|
_require_columns(
|
||||||
|
hist_raw, {"act_symbol", "period_end_date", "reported", "estimate"}, "eps_history"
|
||||||
|
)
|
||||||
|
|
||||||
|
cal_parsed = _parse_calendar(cal_raw, universe)
|
||||||
|
hist_parsed = _parse_history(hist_raw, universe)
|
||||||
|
calendar, cal_stats = earnings_alignment.dedup_calendar(cal_parsed)
|
||||||
|
history, hist_stats = earnings_alignment.dedup_history(hist_parsed)
|
||||||
|
|
||||||
|
period_lower = WINDOW_START.replace(year=WINDOW_START.year - 1)
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
matched = unmatched = 0
|
||||||
|
for symbol, events in calendar.items():
|
||||||
|
ticker_id = universe[symbol]
|
||||||
|
periods = [
|
||||||
|
p for p in history.get(symbol, []) if p["period_end_date"] >= period_lower
|
||||||
|
]
|
||||||
|
matches, unmatched_events, _ = earnings_alignment.align_symbol(
|
||||||
|
events, periods, max_lag_days=MAX_LAG_DAYS, max_lead_days=MAX_LEAD_DAYS
|
||||||
|
)
|
||||||
|
matched += len(matches)
|
||||||
|
unmatched += len(unmatched_events)
|
||||||
|
matched_by_event = {e: p for e, p in matches}
|
||||||
|
for e_idx, event in enumerate(events):
|
||||||
|
p_idx = matched_by_event.get(e_idx)
|
||||||
|
period = periods[p_idx] if p_idx is not None else None
|
||||||
|
rows.append(
|
||||||
|
{
|
||||||
|
"ticker_id": ticker_id,
|
||||||
|
"symbol": symbol,
|
||||||
|
"announce_date": event["announce_date"],
|
||||||
|
"session": event["session"],
|
||||||
|
"period_end": period["period_end_date"] if period else None,
|
||||||
|
"eps_estimate": period["eps_estimate"] if period else None,
|
||||||
|
"eps_actual": period["eps_actual"] if period else None,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
future_rows = [r for r in rows if r["announce_date"] > self.today]
|
||||||
|
tickers_with_future = {r["ticker_id"] for r in future_rows}
|
||||||
|
stats = {
|
||||||
|
"calendar": cal_stats,
|
||||||
|
"eps_history": hist_stats,
|
||||||
|
"universe_size": len(universe),
|
||||||
|
"symbols_with_calendar": len(calendar),
|
||||||
|
"matched_events": matched,
|
||||||
|
"unmatched_events": unmatched,
|
||||||
|
"tracked_tickers_with_future_date": len(tickers_with_future),
|
||||||
|
}
|
||||||
|
return StagedEarnings(
|
||||||
|
rows=rows,
|
||||||
|
stats=stats,
|
||||||
|
future_count=len(future_rows),
|
||||||
|
max_announce_date=max((r["announce_date"] for r in rows), default=None),
|
||||||
|
)
|
||||||
|
|
||||||
|
async def validate(self, db, staged: StagedEarnings) -> ValidationResult:
|
||||||
|
# Promote deletes+reinserts the forward calendar, so this gate is
|
||||||
|
# fail-closed. The forward calendar is the project's acceptance gate.
|
||||||
|
messages: list[str] = []
|
||||||
|
current_future = await self._current_future_count(db)
|
||||||
|
universe_size = int(staged.stats.get("universe_size", 0) or 0)
|
||||||
|
coverage = (
|
||||||
|
staged.stats.get("symbols_with_calendar", 0) / universe_size
|
||||||
|
if universe_size
|
||||||
|
else 0.0
|
||||||
|
)
|
||||||
|
horizon_days = (
|
||||||
|
(staged.max_announce_date - self.today).days if staged.max_announce_date else 0
|
||||||
|
)
|
||||||
|
|
||||||
|
if staged.future_count == 0:
|
||||||
|
messages.append("no future-dated earnings rows staged")
|
||||||
|
elif current_future == 0:
|
||||||
|
# Initial load: no baseline for the ratio gate, so require a real
|
||||||
|
# forward horizon and broad universe coverage instead of one stray row.
|
||||||
|
if horizon_days < MIN_FORWARD_HORIZON_DAYS:
|
||||||
|
messages.append(
|
||||||
|
f"forward horizon only {horizon_days}d < {MIN_FORWARD_HORIZON_DAYS}d "
|
||||||
|
"on initial load"
|
||||||
|
)
|
||||||
|
if coverage < MIN_INITIAL_COVERAGE:
|
||||||
|
messages.append(
|
||||||
|
f"initial universe coverage {coverage:.0%} "
|
||||||
|
f"< {MIN_INITIAL_COVERAGE:.0%} — symbol join likely broken"
|
||||||
|
)
|
||||||
|
elif staged.future_count < current_future * MIN_FUTURE_RATIO:
|
||||||
|
messages.append(
|
||||||
|
f"forward calendar collapsed: staged {staged.future_count} future rows "
|
||||||
|
f"< {MIN_FUTURE_RATIO:.0%} of current {current_future}"
|
||||||
|
)
|
||||||
|
|
||||||
|
keys = [(r["ticker_id"], r["announce_date"]) for r in staged.rows]
|
||||||
|
if len(keys) != len(set(keys)):
|
||||||
|
messages.append("duplicate (ticker_id, announce_date) in staged set")
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
**staged.stats,
|
||||||
|
"staged_rows": len(staged.rows),
|
||||||
|
"future_rows": staged.future_count,
|
||||||
|
"current_future_rows": current_future,
|
||||||
|
"forward_horizon_days": horizon_days,
|
||||||
|
"universe_coverage": round(coverage, 3),
|
||||||
|
}
|
||||||
|
return ValidationResult(
|
||||||
|
ok=not messages,
|
||||||
|
summary=summary,
|
||||||
|
source_max_date=staged.max_announce_date,
|
||||||
|
messages=messages,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def promote(self, db, staged: StagedEarnings, run_id: int) -> dict[str, int]:
|
||||||
|
# Rescheduling: drop this source's future rows, then upsert the staged
|
||||||
|
# set. Past rows (results) are never deleted; moved/cancelled future
|
||||||
|
# dates simply don't reappear.
|
||||||
|
deleted = (
|
||||||
|
await db.execute(
|
||||||
|
delete(EarningsEvent).where(
|
||||||
|
EarningsEvent.source == SOURCE,
|
||||||
|
EarningsEvent.announce_date > self.today,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).rowcount or 0
|
||||||
|
|
||||||
|
now = _now()
|
||||||
|
for r in staged.rows:
|
||||||
|
stmt = insert_for_session(db, EarningsEvent).values(
|
||||||
|
ticker_id=r["ticker_id"],
|
||||||
|
announce_date=r["announce_date"],
|
||||||
|
session=r["session"],
|
||||||
|
period_end=r["period_end"],
|
||||||
|
eps_estimate=r["eps_estimate"],
|
||||||
|
eps_actual=r["eps_actual"],
|
||||||
|
source=SOURCE,
|
||||||
|
import_run_id=run_id,
|
||||||
|
created_at=now,
|
||||||
|
)
|
||||||
|
# Preserve a non-null prior EPS/period-end if a re-pairing comes back
|
||||||
|
# null; prefer a known session over 'unknown'.
|
||||||
|
stmt = stmt.on_conflict_do_update(
|
||||||
|
index_elements=["ticker_id", "announce_date"],
|
||||||
|
set_={
|
||||||
|
"session": case(
|
||||||
|
(stmt.excluded.session != "unknown", stmt.excluded.session),
|
||||||
|
else_=EarningsEvent.session,
|
||||||
|
),
|
||||||
|
"period_end": func.coalesce(
|
||||||
|
stmt.excluded.period_end, EarningsEvent.period_end
|
||||||
|
),
|
||||||
|
"eps_estimate": func.coalesce(
|
||||||
|
stmt.excluded.eps_estimate, EarningsEvent.eps_estimate
|
||||||
|
),
|
||||||
|
"eps_actual": func.coalesce(
|
||||||
|
stmt.excluded.eps_actual, EarningsEvent.eps_actual
|
||||||
|
),
|
||||||
|
"source": stmt.excluded.source,
|
||||||
|
"import_run_id": stmt.excluded.import_run_id,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
await db.execute(stmt)
|
||||||
|
|
||||||
|
return {"deleted_future": int(deleted), "upserted": len(staged.rows)}
|
||||||
|
|
||||||
|
# -- helpers -----------------------------------------------------------
|
||||||
|
|
||||||
|
async def _load_universe(self, db) -> dict[str, int]:
|
||||||
|
rows = (await db.execute(select(Ticker.id, Ticker.symbol))).all()
|
||||||
|
return {
|
||||||
|
earnings_alignment.normalise_symbol(symbol): tid
|
||||||
|
for tid, symbol in rows
|
||||||
|
if symbol
|
||||||
|
}
|
||||||
|
|
||||||
|
async def _current_future_count(self, db) -> int:
|
||||||
|
return (
|
||||||
|
await db.execute(
|
||||||
|
select(func.count())
|
||||||
|
.select_from(EarningsEvent)
|
||||||
|
.where(
|
||||||
|
EarningsEvent.source == SOURCE,
|
||||||
|
EarningsEvent.announce_date > self.today,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
).scalar_one()
|
||||||
|
|
||||||
|
|
||||||
|
def _require_columns(rows: list[dict[str, str]], required: set[str], table: str) -> None:
|
||||||
|
"""Upstream schema-change gate: a missing column stops the run (→ failed)."""
|
||||||
|
if not rows:
|
||||||
|
return
|
||||||
|
present = set(rows[0].keys())
|
||||||
|
missing = required - present
|
||||||
|
if missing:
|
||||||
|
raise ValueError(f"{table}: upstream schema change, missing columns {sorted(missing)}")
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_calendar(raw: list[dict[str, str]], universe: dict[str, int]) -> list[dict[str, Any]]:
|
||||||
|
out: list[dict[str, Any]] = []
|
||||||
|
for row in raw:
|
||||||
|
symbol = earnings_alignment.normalise_symbol(row.get("act_symbol"))
|
||||||
|
raw_date = str(row.get("date") or "")[:10]
|
||||||
|
if symbol not in universe or not raw_date:
|
||||||
|
continue
|
||||||
|
announce_date = date.fromisoformat(raw_date)
|
||||||
|
if announce_date < WINDOW_START:
|
||||||
|
continue
|
||||||
|
out.append(
|
||||||
|
{
|
||||||
|
"symbol": symbol,
|
||||||
|
"announce_date": announce_date,
|
||||||
|
"session": earnings_alignment.normalise_session(row.get("when")),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_history(raw: list[dict[str, str]], universe: dict[str, int]) -> list[dict[str, Any]]:
|
||||||
|
out: list[dict[str, Any]] = []
|
||||||
|
for row in raw:
|
||||||
|
symbol = earnings_alignment.normalise_symbol(row.get("act_symbol"))
|
||||||
|
raw_date = str(row.get("period_end_date") or "")[:10]
|
||||||
|
if symbol not in universe or not raw_date:
|
||||||
|
continue
|
||||||
|
out.append(
|
||||||
|
{
|
||||||
|
"symbol": symbol,
|
||||||
|
"period_end_date": date.fromisoformat(raw_date),
|
||||||
|
"eps_actual": earnings_alignment.safe_number(row.get("reported")),
|
||||||
|
"eps_estimate": earnings_alignment.safe_number(row.get("estimate")),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return out
|
||||||
@@ -0,0 +1,207 @@
|
|||||||
|
"""Pure calendar<->EPS-history alignment for the DoltHub earnings source.
|
||||||
|
|
||||||
|
The earnings repo keeps the announcement calendar (`earnings_calendar`) and the
|
||||||
|
reported/estimate EPS history (`eps_history`) in separate tables with no shared
|
||||||
|
key — the calendar has announce dates, the history has period-end dates. This
|
||||||
|
module reproduces the research importer's **minimum-cost monotonic alignment**
|
||||||
|
(`scripts/import_dolthub_earnings.py`) as pure, DB-free, unit-testable functions
|
||||||
|
so the production importer can reuse it without extending that one-off script.
|
||||||
|
|
||||||
|
Constants and cost function are kept identical to the research script; the DP is
|
||||||
|
what pairs each announcement with the quarter it reported, tolerating gaps on
|
||||||
|
either side. Do not tune these without re-validating surprise-history pairing.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
from collections import defaultdict
|
||||||
|
from datetime import date
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
# Alignment costs — identical to scripts/import_dolthub_earnings.py.
|
||||||
|
SKIP_EVENT_COST = 45.0
|
||||||
|
SKIP_PERIOD_COST = 45.0
|
||||||
|
_TYPICAL_ANNOUNCE_LAG_DAYS = 30 # announcements land ~a month after period end
|
||||||
|
_MISSING_SESSION_PENALTY = 3.0
|
||||||
|
|
||||||
|
# Session normalization → the three values the schema/API promise.
|
||||||
|
_SESSION_ALIASES = {
|
||||||
|
"before market open": "bmo",
|
||||||
|
"before open": "bmo",
|
||||||
|
"bmo": "bmo",
|
||||||
|
"after market close": "amc",
|
||||||
|
"after close": "amc",
|
||||||
|
"amc": "amc",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def normalise_symbol(value: Any) -> str:
|
||||||
|
"""Upper-case, trim, and map dots to dashes so the DoltHub `act_symbol`
|
||||||
|
(`BF.B`) and the app's `tickers.symbol` join after the same normalization."""
|
||||||
|
return str(value or "").strip().upper().replace(".", "-")
|
||||||
|
|
||||||
|
|
||||||
|
def normalise_session(value: Any) -> str:
|
||||||
|
"""Map the source `when` text to bmo | amc | unknown. Anything not clearly a
|
||||||
|
pre-open or post-close session (including 'during market hours' and blanks)
|
||||||
|
collapses to 'unknown' — the schema/API only promise those three."""
|
||||||
|
cleaned = str(value or "").strip().lower().replace("_", " ").replace("-", " ")
|
||||||
|
return _SESSION_ALIASES.get(cleaned, "unknown")
|
||||||
|
|
||||||
|
|
||||||
|
def safe_number(value: Any) -> float | None:
|
||||||
|
if value is None or str(value).strip() == "":
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
result = float(value)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
return result if math.isfinite(result) else None
|
||||||
|
|
||||||
|
|
||||||
|
def dedup_calendar(
|
||||||
|
rows: list[dict[str, Any]],
|
||||||
|
) -> tuple[dict[str, list[dict[str, Any]]], dict[str, int]]:
|
||||||
|
"""Collapse to one row per (symbol, announce_date), preferring a known
|
||||||
|
session over 'unknown'. Rows must be pre-parsed:
|
||||||
|
{symbol, announce_date: date, session}. Returns {symbol: [events sorted by
|
||||||
|
date]} and dedup stats."""
|
||||||
|
by_key: dict[tuple[str, date], dict[str, Any]] = {}
|
||||||
|
duplicate_rows = 0
|
||||||
|
restated_rows = 0
|
||||||
|
for row in rows:
|
||||||
|
key = (row["symbol"], row["announce_date"])
|
||||||
|
previous = by_key.get(key)
|
||||||
|
if previous is None:
|
||||||
|
by_key[key] = row
|
||||||
|
continue
|
||||||
|
duplicate_rows += 1
|
||||||
|
prev_known = previous["session"] != "unknown"
|
||||||
|
new_known = row["session"] != "unknown"
|
||||||
|
if prev_known and new_known and previous["session"] != row["session"]:
|
||||||
|
restated_rows += 1
|
||||||
|
# Prefer a row that carries a known session.
|
||||||
|
if new_known:
|
||||||
|
by_key[key] = row
|
||||||
|
grouped: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||||
|
for row in by_key.values():
|
||||||
|
grouped[row["symbol"]].append(row)
|
||||||
|
for events in grouped.values():
|
||||||
|
events.sort(key=lambda item: item["announce_date"])
|
||||||
|
return grouped, {
|
||||||
|
"deduped_rows": len(by_key),
|
||||||
|
"duplicate_rows": duplicate_rows,
|
||||||
|
"restated_rows": restated_rows,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def dedup_history(
|
||||||
|
rows: list[dict[str, Any]],
|
||||||
|
) -> tuple[dict[str, list[dict[str, Any]]], dict[str, int]]:
|
||||||
|
"""Collapse to one row per (symbol, period_end_date), preferring the row with
|
||||||
|
more non-null EPS fields. Rows must be pre-parsed:
|
||||||
|
{symbol, period_end_date: date, eps_actual, eps_estimate}."""
|
||||||
|
fields = ("eps_actual", "eps_estimate")
|
||||||
|
by_key: dict[tuple[str, date], dict[str, Any]] = {}
|
||||||
|
duplicate_rows = 0
|
||||||
|
restated_rows = 0
|
||||||
|
for row in rows:
|
||||||
|
key = (row["symbol"], row["period_end_date"])
|
||||||
|
previous = by_key.get(key)
|
||||||
|
if previous is None:
|
||||||
|
by_key[key] = row
|
||||||
|
continue
|
||||||
|
duplicate_rows += 1
|
||||||
|
if any(
|
||||||
|
previous.get(f) is not None
|
||||||
|
and row.get(f) is not None
|
||||||
|
and previous[f] != row[f]
|
||||||
|
for f in fields
|
||||||
|
):
|
||||||
|
restated_rows += 1
|
||||||
|
prev_score = sum(previous.get(f) is not None for f in fields)
|
||||||
|
new_score = sum(row.get(f) is not None for f in fields)
|
||||||
|
if new_score >= prev_score:
|
||||||
|
by_key[key] = row
|
||||||
|
grouped: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||||
|
for row in by_key.values():
|
||||||
|
grouped[row["symbol"]].append(row)
|
||||||
|
for periods in grouped.values():
|
||||||
|
periods.sort(key=lambda item: item["period_end_date"])
|
||||||
|
return grouped, {
|
||||||
|
"deduped_rows": len(by_key),
|
||||||
|
"duplicate_rows": duplicate_rows,
|
||||||
|
"restated_rows": restated_rows,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def match_cost(event: dict[str, Any], period: dict[str, Any]) -> float:
|
||||||
|
delta = (event["announce_date"] - period["period_end_date"]).days
|
||||||
|
penalty = _MISSING_SESSION_PENALTY if event.get("session") == "unknown" else 0.0
|
||||||
|
return float(abs(delta - _TYPICAL_ANNOUNCE_LAG_DAYS)) + penalty
|
||||||
|
|
||||||
|
|
||||||
|
def align_symbol(
|
||||||
|
events: list[dict[str, Any]],
|
||||||
|
periods: list[dict[str, Any]],
|
||||||
|
*,
|
||||||
|
max_lag_days: int,
|
||||||
|
max_lead_days: int,
|
||||||
|
) -> tuple[list[tuple[int, int]], list[int], list[int]]:
|
||||||
|
"""Minimum-cost monotonic calendar-to-period alignment for one symbol.
|
||||||
|
|
||||||
|
Both lists must be sorted ascending (by announce_date / period_end_date). A
|
||||||
|
match is allowed only when ``-max_lead_days <= announce_date - period_end <=
|
||||||
|
max_lag_days``. Returns (matches, unmatched_event_indices,
|
||||||
|
unmatched_period_indices).
|
||||||
|
"""
|
||||||
|
n_events = len(events)
|
||||||
|
n_periods = len(periods)
|
||||||
|
scores = [[0.0] * (n_periods + 1) for _ in range(n_events + 1)]
|
||||||
|
choices = [[""] * (n_periods + 1) for _ in range(n_events + 1)]
|
||||||
|
for e in range(n_events - 1, -1, -1):
|
||||||
|
scores[e][n_periods] = scores[e + 1][n_periods] + SKIP_EVENT_COST
|
||||||
|
choices[e][n_periods] = "event"
|
||||||
|
for p in range(n_periods - 1, -1, -1):
|
||||||
|
scores[n_events][p] = scores[n_events][p + 1] + SKIP_PERIOD_COST
|
||||||
|
choices[n_events][p] = "period"
|
||||||
|
|
||||||
|
for e in range(n_events - 1, -1, -1):
|
||||||
|
for p in range(n_periods - 1, -1, -1):
|
||||||
|
options = [
|
||||||
|
(scores[e + 1][p] + SKIP_EVENT_COST, 2, "event"),
|
||||||
|
(scores[e][p + 1] + SKIP_PERIOD_COST, 1, "period"),
|
||||||
|
]
|
||||||
|
delta = (events[e]["announce_date"] - periods[p]["period_end_date"]).days
|
||||||
|
if -max_lead_days <= delta <= max_lag_days:
|
||||||
|
options.append(
|
||||||
|
(scores[e + 1][p + 1] + match_cost(events[e], periods[p]), 0, "match")
|
||||||
|
)
|
||||||
|
score, _, choice = min(options)
|
||||||
|
scores[e][p] = score
|
||||||
|
choices[e][p] = choice
|
||||||
|
|
||||||
|
matches: list[tuple[int, int]] = []
|
||||||
|
unmatched_events: list[int] = []
|
||||||
|
unmatched_periods: list[int] = []
|
||||||
|
e = p = 0
|
||||||
|
while e < n_events or p < n_periods:
|
||||||
|
if e >= n_events:
|
||||||
|
unmatched_periods.extend(range(p, n_periods))
|
||||||
|
break
|
||||||
|
if p >= n_periods:
|
||||||
|
unmatched_events.extend(range(e, n_events))
|
||||||
|
break
|
||||||
|
choice = choices[e][p]
|
||||||
|
if choice == "match":
|
||||||
|
matches.append((e, p))
|
||||||
|
e += 1
|
||||||
|
p += 1
|
||||||
|
elif choice == "period":
|
||||||
|
unmatched_periods.append(p)
|
||||||
|
p += 1
|
||||||
|
else:
|
||||||
|
unmatched_events.append(e)
|
||||||
|
e += 1
|
||||||
|
return matches, unmatched_events, unmatched_periods
|
||||||
@@ -0,0 +1,331 @@
|
|||||||
|
"""Assemble the additive fundamentals API v1 objects (earnings, metrics,
|
||||||
|
valuation, reads) from SEC snapshots + Dolt earnings + the latest price.
|
||||||
|
|
||||||
|
Strictly additive: the router merges these into the existing FundamentalResponse
|
||||||
|
without touching legacy fields. Valuation ratios are computed at REQUEST TIME from
|
||||||
|
the stored snapshots + the latest ohlcv close (no stored valuation). Peer stats are
|
||||||
|
batched and CIK-deduplicated; invalid valuation inputs are guarded to null.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
from collections import defaultdict
|
||||||
|
from datetime import date, datetime
|
||||||
|
from typing import Any
|
||||||
|
from zoneinfo import ZoneInfo
|
||||||
|
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||||||
|
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.ohlcv import OHLCVRecord
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import fundamentals_derivation as deriv
|
||||||
|
from app.services import fundamentals_peers as peers
|
||||||
|
from app.services import fundamentals_reads as reads
|
||||||
|
|
||||||
|
# The fixed metric row set — every key always present, value null when unavailable.
|
||||||
|
METRIC_KEYS = (
|
||||||
|
"revenue_growth_yoy", "eps_growth_yoy", "operating_margin", "fcf_margin",
|
||||||
|
"net_debt", "net_debt_to_ebitda", "share_count_change_yoy",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def build_fundamentals_v1(db: AsyncSession, symbol: str, *, today: date | None = None) -> dict[str, Any]:
|
||||||
|
today = today or _ny_today()
|
||||||
|
ticker = await _ticker_by_symbol(db, symbol)
|
||||||
|
|
||||||
|
earnings = await _build_earnings(db, ticker.id, today) if ticker else _empty_earnings()
|
||||||
|
if ticker is None or not ticker.cik:
|
||||||
|
# No SEC identity: metrics present but null, valuation null, empty reads.
|
||||||
|
return {"earnings": earnings, "metrics": _empty_metrics(), "valuation": None,
|
||||||
|
"reads": _empty_reads()}
|
||||||
|
|
||||||
|
subject_cik = ticker.cik
|
||||||
|
derived = deriv.derive((await _snapshots_for(db, [subject_cik])).get(subject_cik, []))
|
||||||
|
|
||||||
|
two = peers.two_digit_sic(ticker.sic)
|
||||||
|
peer_derived: dict[str, deriv.DerivedFundamentals] = {}
|
||||||
|
peer_price_by_cik: dict[str, tuple[float, date] | None] = {}
|
||||||
|
if two:
|
||||||
|
# Subject's representative is the REQUESTED ticker (so its price is used for
|
||||||
|
# the subject in the peer set); other issuers pick a deterministic-by-symbol rep.
|
||||||
|
group = await _peer_group(db, two, subject_cik, ticker.id)
|
||||||
|
peer_snaps = await _snapshots_for(db, list(group))
|
||||||
|
peer_derived = {cik: deriv.derive(rows) for cik, rows in peer_snaps.items()}
|
||||||
|
closes = await _latest_closes(db, set(group.values()))
|
||||||
|
peer_price_by_cik = {cik: closes.get(tid) for cik, tid in group.items()}
|
||||||
|
|
||||||
|
subject_price = await _latest_close(db, ticker.id)
|
||||||
|
metrics = _build_metrics(derived, peer_derived, two)
|
||||||
|
valuation = _build_valuation(derived, subject_price, peer_derived, peer_price_by_cik, two)
|
||||||
|
reads_obj = _build_reads(metrics, valuation)
|
||||||
|
return {"earnings": earnings, "metrics": metrics, "valuation": valuation, "reads": reads_obj}
|
||||||
|
|
||||||
|
|
||||||
|
# -- earnings ----------------------------------------------------------------
|
||||||
|
|
||||||
|
async def _build_earnings(db, ticker_id: int, today: date) -> dict[str, Any]:
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(EarningsEvent).where(EarningsEvent.ticker_id == ticker_id)
|
||||||
|
)).scalars().all()
|
||||||
|
# Same-day earnings are UPCOMING (days_until 0); recent is strictly earlier.
|
||||||
|
upcoming = sorted((e for e in rows if e.announce_date >= today), key=lambda e: e.announce_date)
|
||||||
|
past = sorted((e for e in rows if e.announce_date < today), key=lambda e: e.announce_date, reverse=True)
|
||||||
|
|
||||||
|
nxt = None
|
||||||
|
if upcoming:
|
||||||
|
e = upcoming[0]
|
||||||
|
nxt = {"date": e.announce_date.isoformat(), "session": e.session,
|
||||||
|
"days_until": (e.announce_date - today).days}
|
||||||
|
recent = [{
|
||||||
|
"announce_date": e.announce_date.isoformat(),
|
||||||
|
"period_end": _iso(e.period_end),
|
||||||
|
"eps_estimate": e.eps_estimate,
|
||||||
|
"eps_actual": e.eps_actual,
|
||||||
|
"surprise_pct": _surprise_pct(e.eps_estimate, e.eps_actual),
|
||||||
|
} for e in past[:4]]
|
||||||
|
return {"next": nxt, "recent": recent}
|
||||||
|
|
||||||
|
|
||||||
|
def _surprise_pct(estimate, actual):
|
||||||
|
if estimate is None or actual is None or estimate == 0:
|
||||||
|
return None
|
||||||
|
return round((actual - estimate) / abs(estimate) * 100.0, 2)
|
||||||
|
|
||||||
|
|
||||||
|
# -- metrics -----------------------------------------------------------------
|
||||||
|
|
||||||
|
def _build_metrics(derived, peer_derived, two: str | None) -> list[dict[str, Any]]:
|
||||||
|
out = []
|
||||||
|
for key in METRIC_KEYS:
|
||||||
|
series = derived.metrics.get(key)
|
||||||
|
value = series.value if series else None
|
||||||
|
history = [{"period_end": _iso(p.period_end), "value": p.value} for p in (series.history if series else [])]
|
||||||
|
industry = None
|
||||||
|
if two and peer_derived and key in peers.HIGHER_IS_BETTER:
|
||||||
|
group_values = [
|
||||||
|
(pd.metrics.get(key).value if pd.metrics.get(key) else None)
|
||||||
|
for pd in peer_derived.values()
|
||||||
|
]
|
||||||
|
stat = peers.peer_stat_for(key, value, group_values)
|
||||||
|
if stat:
|
||||||
|
industry = {"label": f"SIC {two} peers", "median": round(stat.median, 4),
|
||||||
|
"favorable_percentile": stat.favorable_percentile, "peer_count": stat.peer_count}
|
||||||
|
out.append({
|
||||||
|
"key": key,
|
||||||
|
"value": value,
|
||||||
|
"history": history,
|
||||||
|
"industry": industry,
|
||||||
|
"period_end": _iso(series.period_end) if series else None,
|
||||||
|
"filed_date": _iso(series.filed_date) if series else None,
|
||||||
|
"source": "sec",
|
||||||
|
})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
# -- valuation (request-time) ------------------------------------------------
|
||||||
|
|
||||||
|
def _build_valuation(derived, subject_price, peer_derived, peer_price_by_cik, two) -> dict[str, Any] | None:
|
||||||
|
if derived.latest_period_end is None:
|
||||||
|
return None # no snapshots yet
|
||||||
|
price = subject_price[0] if subject_price else None
|
||||||
|
price_date = subject_price[1] if subject_price else None
|
||||||
|
if not _finite(price) or price <= 0:
|
||||||
|
return None # no usable price -> valuation null (approved contract)
|
||||||
|
|
||||||
|
pe = _pe(price, derived.ttm_diluted_eps)
|
||||||
|
market_cap = _market_cap(price, derived.shares_outstanding)
|
||||||
|
fcf_yield = _fcf_yield(derived.ttm_fcf, market_cap)
|
||||||
|
|
||||||
|
pe_industry = fcf_yield_industry = None
|
||||||
|
if two and peer_derived:
|
||||||
|
pe_values = [_pe(_p(peer_price_by_cik.get(cik)), pd.ttm_diluted_eps) for cik, pd in peer_derived.items()]
|
||||||
|
fy_values = [
|
||||||
|
_fcf_yield(pd.ttm_fcf, _market_cap(_p(peer_price_by_cik.get(cik)), pd.shares_outstanding))
|
||||||
|
for cik, pd in peer_derived.items()
|
||||||
|
]
|
||||||
|
pe_industry = _industry("pe", pe, pe_values, two)
|
||||||
|
fcf_yield_industry = _industry("fcf_yield", fcf_yield, fy_values, two)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"pe": _round(pe, 2),
|
||||||
|
"fcf_yield": _round(fcf_yield, 2),
|
||||||
|
"market_cap_est": _round(market_cap, 0),
|
||||||
|
"pe_industry": pe_industry,
|
||||||
|
"fcf_yield_industry": fcf_yield_industry,
|
||||||
|
"price_date": _iso(price_date),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _pe(price, ttm_eps):
|
||||||
|
if not _finite(price) or price <= 0 or not _finite(ttm_eps) or ttm_eps <= 0:
|
||||||
|
return None
|
||||||
|
return price / ttm_eps
|
||||||
|
|
||||||
|
|
||||||
|
def _market_cap(price, shares):
|
||||||
|
if not _finite(price) or price <= 0 or not _finite(shares) or shares <= 0:
|
||||||
|
return None
|
||||||
|
return price * shares
|
||||||
|
|
||||||
|
|
||||||
|
def _fcf_yield(ttm_fcf, market_cap):
|
||||||
|
if not _finite(ttm_fcf) or not _finite(market_cap) or market_cap <= 0:
|
||||||
|
return None
|
||||||
|
return ttm_fcf / market_cap * 100.0
|
||||||
|
|
||||||
|
|
||||||
|
def _industry(key, subject, group_values, two):
|
||||||
|
stat = peers.peer_stat_for(key, subject, group_values)
|
||||||
|
if stat is None:
|
||||||
|
return None
|
||||||
|
return {"label": f"SIC {two} peers", "median": round(stat.median, 4),
|
||||||
|
"favorable_percentile": stat.favorable_percentile, "peer_count": stat.peer_count}
|
||||||
|
|
||||||
|
|
||||||
|
# -- reads -------------------------------------------------------------------
|
||||||
|
|
||||||
|
_READ_KEYS = METRIC_KEYS + ("pe", "fcf_yield")
|
||||||
|
|
||||||
|
|
||||||
|
def _build_reads(metrics: list[dict], valuation: dict | None) -> dict[str, Any]:
|
||||||
|
by_metric = {m["key"]: m for m in metrics}
|
||||||
|
|
||||||
|
def hist(key):
|
||||||
|
return [_Pt(p["value"]) for p in by_metric.get(key, {}).get("history", [])]
|
||||||
|
|
||||||
|
growth = reads.growth_read(hist("revenue_growth_yoy"))
|
||||||
|
eps_growth = reads.growth_read(hist("eps_growth_yoy"))
|
||||||
|
op_margin = reads.margin_read(hist("operating_margin"))
|
||||||
|
fcf_margin = reads.margin_read(hist("fcf_margin"))
|
||||||
|
share = reads.share_count_read(by_metric.get("share_count_change_yoy", {}).get("value"))
|
||||||
|
leverage = reads.peer_read("net_debt_to_ebitda", _pct(by_metric.get("net_debt_to_ebitda", {}).get("industry")))
|
||||||
|
pe_read = reads.peer_read("pe", _pct(valuation.get("pe_industry"))) if valuation else None
|
||||||
|
fcf_yield_read = reads.peer_read("fcf_yield", _pct(valuation.get("fcf_yield_industry"))) if valuation else None
|
||||||
|
|
||||||
|
# Fixed by_key map over every metric + pe + fcf_yield (null where unavailable).
|
||||||
|
by_key: dict[str, str | None] = {k: None for k in _READ_KEYS}
|
||||||
|
by_key.update({
|
||||||
|
"revenue_growth_yoy": growth,
|
||||||
|
"eps_growth_yoy": eps_growth,
|
||||||
|
"operating_margin": op_margin,
|
||||||
|
"fcf_margin": fcf_margin,
|
||||||
|
"share_count_change_yoy": share,
|
||||||
|
"net_debt_to_ebitda": leverage,
|
||||||
|
"pe": pe_read,
|
||||||
|
"fcf_yield": fcf_yield_read,
|
||||||
|
})
|
||||||
|
header = reads.header_sentence(growth, op_margin, pe_read or fcf_yield_read) or None
|
||||||
|
return {"header": header, "by_key": by_key}
|
||||||
|
|
||||||
|
|
||||||
|
def _empty_reads() -> dict[str, Any]:
|
||||||
|
return {"header": None, "by_key": {k: None for k in _READ_KEYS}}
|
||||||
|
|
||||||
|
|
||||||
|
class _Pt:
|
||||||
|
__slots__ = ("value",)
|
||||||
|
|
||||||
|
def __init__(self, value):
|
||||||
|
self.value = value
|
||||||
|
|
||||||
|
|
||||||
|
def _pct(industry: dict | None):
|
||||||
|
return industry.get("favorable_percentile") if industry else None
|
||||||
|
|
||||||
|
|
||||||
|
# -- queries -----------------------------------------------------------------
|
||||||
|
|
||||||
|
async def _ticker_by_symbol(db, symbol: str) -> Ticker | None:
|
||||||
|
return (await db.execute(
|
||||||
|
select(Ticker).where(Ticker.symbol == symbol.strip().upper())
|
||||||
|
)).scalar_one_or_none()
|
||||||
|
|
||||||
|
|
||||||
|
async def _snapshots_for(db, ciks) -> dict[str, list]:
|
||||||
|
out: dict[str, list] = defaultdict(list)
|
||||||
|
if not ciks:
|
||||||
|
return out
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(FundamentalSnapshot).where(FundamentalSnapshot.cik.in_(list(ciks)))
|
||||||
|
)).scalars().all()
|
||||||
|
for r in rows:
|
||||||
|
out[r.cik].append(r)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _peer_group(db, two: str, subject_cik: str, subject_tid: int) -> dict[str, int]:
|
||||||
|
"""{cik: representative ticker_id} for tracked issuers in the 2-digit SIC group,
|
||||||
|
CIK-deduplicated. Each issuer's representative is its lexicographically-smallest
|
||||||
|
symbol (deterministic), EXCEPT the subject issuer, which uses the requested
|
||||||
|
ticker — so a multi-class subject (GOOGL) is priced by the requested class, not
|
||||||
|
an arbitrary sibling (GOOG)."""
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(Ticker.cik, Ticker.id, Ticker.symbol)
|
||||||
|
.where(Ticker.cik.is_not(None), func.substr(Ticker.sic, 1, 2) == two)
|
||||||
|
)).all()
|
||||||
|
rep: dict[str, tuple[int, str]] = {}
|
||||||
|
for cik, tid, sym in rows:
|
||||||
|
key = sym or ""
|
||||||
|
if cik not in rep or key < rep[cik][1]:
|
||||||
|
rep[cik] = (tid, key)
|
||||||
|
group = {cik: tid for cik, (tid, _) in rep.items()}
|
||||||
|
if subject_cik in group:
|
||||||
|
group[subject_cik] = subject_tid # requested ticker prices the subject
|
||||||
|
return group
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_closes(db, ticker_ids: set[int]) -> dict[int, tuple[float, date]]:
|
||||||
|
if not ticker_ids:
|
||||||
|
return {}
|
||||||
|
latest = (
|
||||||
|
select(OHLCVRecord.ticker_id, func.max(OHLCVRecord.date).label("d"))
|
||||||
|
.where(OHLCVRecord.ticker_id.in_(list(ticker_ids)))
|
||||||
|
.group_by(OHLCVRecord.ticker_id)
|
||||||
|
.subquery()
|
||||||
|
)
|
||||||
|
rows = (await db.execute(
|
||||||
|
select(OHLCVRecord.ticker_id, OHLCVRecord.close, OHLCVRecord.date).join(
|
||||||
|
latest, (OHLCVRecord.ticker_id == latest.c.ticker_id) & (OHLCVRecord.date == latest.c.d)
|
||||||
|
)
|
||||||
|
)).all()
|
||||||
|
return {tid: (close, d) for tid, close, d in rows}
|
||||||
|
|
||||||
|
|
||||||
|
async def _latest_close(db, ticker_id: int) -> tuple[float, date] | None:
|
||||||
|
return (await _latest_closes(db, {ticker_id})).get(ticker_id)
|
||||||
|
|
||||||
|
|
||||||
|
# -- helpers -----------------------------------------------------------------
|
||||||
|
|
||||||
|
def _empty_metrics() -> list[dict[str, Any]]:
|
||||||
|
return [{"key": k, "value": None, "history": [], "industry": None,
|
||||||
|
"period_end": None, "filed_date": None, "source": "sec"} for k in METRIC_KEYS]
|
||||||
|
|
||||||
|
|
||||||
|
def _empty_earnings() -> dict[str, Any]:
|
||||||
|
return {"next": None, "recent": []}
|
||||||
|
|
||||||
|
|
||||||
|
def _p(price_tuple):
|
||||||
|
return price_tuple[0] if price_tuple else None
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(v) -> bool:
|
||||||
|
return isinstance(v, (int, float)) and not isinstance(v, bool) and math.isfinite(v)
|
||||||
|
|
||||||
|
|
||||||
|
def _round(v, ndigits):
|
||||||
|
return round(v, ndigits) if _finite(v) else None
|
||||||
|
|
||||||
|
|
||||||
|
def _iso(d) -> str | None:
|
||||||
|
return d.isoformat() if d else None
|
||||||
|
|
||||||
|
|
||||||
|
def _ny_today() -> date:
|
||||||
|
"""Today's New York calendar date — the market's day, not the server's."""
|
||||||
|
return datetime.now(ZoneInfo("America/New_York")).date()
|
||||||
@@ -0,0 +1,269 @@
|
|||||||
|
"""Pure read-time derivation of fundamental metrics from stored snapshots.
|
||||||
|
|
||||||
|
`fundamental_snapshots` stores one immutable row per accession with **cumulative
|
||||||
|
YTD** duration facts and period-end balance-sheet instants (A3). This module
|
||||||
|
derives everything the UI/API shows — discrete quarters, Q4, TTM, YoY growth,
|
||||||
|
margins, leverage, dilution, and the quarter tape — at read time, per the plan's
|
||||||
|
schema decision. No I/O, no DB: it takes an issuer's snapshot rows (ORM rows or
|
||||||
|
any objects with the same attributes) and returns structured metrics.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- **Amendment selection:** for each (fiscal_year, fiscal_period), the row with
|
||||||
|
the newest `accepted_at` wins.
|
||||||
|
- **Discrete quarter** = YTD(Qn) − YTD(Qn−1); Q1 = YTD(Q1); **Q4 = YTD(FY) −
|
||||||
|
YTD(Q3)**. Any missing period → the derived value is null, never partial.
|
||||||
|
- **TTM** = sum of the trailing four discrete quarters ending at a period.
|
||||||
|
- Units follow app convention: percentages are percentage points (21.0 = 21%),
|
||||||
|
net-debt/EBITDA is a multiple, net debt is dollars.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date
|
||||||
|
from typing import Any, Iterable
|
||||||
|
|
||||||
|
_FP_TO_Q = {"Q1": 1, "Q2": 2, "Q3": 3, "FY": 4}
|
||||||
|
_Q_TO_FP = {1: "Q1", 2: "Q2", 3: "Q3", 4: "FY"}
|
||||||
|
_PREV_FP = {"Q2": "Q1", "Q3": "Q2", "FY": "Q3"}
|
||||||
|
TAPE_LEN = 4 # quarter-tape length
|
||||||
|
|
||||||
|
# Duration (flow) fields differenced from YTD into discrete quarters + summed to TTM.
|
||||||
|
_FLOW_FIELDS = (
|
||||||
|
"revenue", "net_income", "operating_income", "diluted_eps", "cfo", "capex",
|
||||||
|
"depreciation_amortization",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class MetricPoint:
|
||||||
|
period_end: date
|
||||||
|
value: float | None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class MetricSeries:
|
||||||
|
value: float | None = None
|
||||||
|
history: list[MetricPoint] = field(default_factory=list) # oldest -> newest, <= TAPE_LEN
|
||||||
|
period_end: date | None = None
|
||||||
|
filed_date: date | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class DerivedFundamentals:
|
||||||
|
metrics: dict[str, MetricSeries] = field(default_factory=dict)
|
||||||
|
# request-time valuation inputs (ratios are computed in the API with price)
|
||||||
|
ttm_diluted_eps: float | None = None
|
||||||
|
ttm_fcf: float | None = None
|
||||||
|
shares_outstanding: float | None = None
|
||||||
|
latest_period_end: date | None = None
|
||||||
|
latest_filed_date: date | None = None
|
||||||
|
|
||||||
|
|
||||||
|
def _prev_q(fy: int, q: int) -> tuple[int, int]:
|
||||||
|
return (fy, q - 1) if q > 1 else (fy - 1, 4)
|
||||||
|
|
||||||
|
|
||||||
|
def derive(snapshots: Iterable[Any]) -> DerivedFundamentals:
|
||||||
|
selected = _select_latest_per_period(snapshots)
|
||||||
|
result = DerivedFundamentals()
|
||||||
|
if not selected:
|
||||||
|
return result
|
||||||
|
|
||||||
|
# Discrete quarter values per flow field: {field: {(fy, q): value}}.
|
||||||
|
discrete = {f: _discrete_quarters(selected, f) for f in _FLOW_FIELDS}
|
||||||
|
quarters = _ordered_quarters(selected) # chronological (fy, q) with a row
|
||||||
|
latest = quarters[-1]
|
||||||
|
latest_row = selected[(latest[0], _Q_TO_FP[latest[1]])]
|
||||||
|
|
||||||
|
result.latest_period_end = latest_row.period_end
|
||||||
|
result.latest_filed_date = latest_row.filed_date
|
||||||
|
result.shares_outstanding = getattr(latest_row, "shares_outstanding", None)
|
||||||
|
result.ttm_diluted_eps = _ttm(discrete["diluted_eps"], *latest)
|
||||||
|
ttm_cfo = _ttm(discrete["cfo"], *latest)
|
||||||
|
ttm_capex = _ttm(discrete["capex"], *latest)
|
||||||
|
result.ttm_fcf = None if ttm_cfo is None or ttm_capex is None else ttm_cfo - ttm_capex
|
||||||
|
|
||||||
|
# tape = the CONSECUTIVE run of up to TAPE_LEN quarters ending at the latest,
|
||||||
|
# stopping at a gap — so trend text never compares non-adjacent periods.
|
||||||
|
tape = _consecutive_suffix(quarters, TAPE_LEN)
|
||||||
|
result.metrics = {
|
||||||
|
"revenue_growth_yoy": _yoy_growth_series(discrete["revenue"], selected, tape),
|
||||||
|
"eps_growth_yoy": _yoy_growth_series(discrete["diluted_eps"], selected, tape),
|
||||||
|
"operating_margin": _margin_series(discrete["operating_income"], discrete["revenue"], selected, tape),
|
||||||
|
"fcf_margin": _fcf_margin_series(discrete, selected, tape),
|
||||||
|
"net_debt": _instant_series(selected, tape, _net_debt),
|
||||||
|
"net_debt_to_ebitda": _leverage_series(selected, discrete, tape),
|
||||||
|
"share_count_change_yoy": _share_change_series(selected, tape),
|
||||||
|
}
|
||||||
|
for series in result.metrics.values():
|
||||||
|
series.period_end = latest_row.period_end
|
||||||
|
series.filed_date = latest_row.filed_date
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
# -- period selection --------------------------------------------------------
|
||||||
|
|
||||||
|
def _select_latest_per_period(snapshots: Iterable[Any]) -> dict[tuple[int, str], Any]:
|
||||||
|
best: dict[tuple[int, str], Any] = {}
|
||||||
|
for row in snapshots:
|
||||||
|
fp = getattr(row, "fiscal_period", None)
|
||||||
|
fy = getattr(row, "fiscal_year", None)
|
||||||
|
if fp not in _FP_TO_Q or fy is None:
|
||||||
|
continue
|
||||||
|
key = (fy, fp)
|
||||||
|
cur = best.get(key)
|
||||||
|
if cur is None or _accepted(row) > _accepted(cur):
|
||||||
|
best[key] = row
|
||||||
|
return best
|
||||||
|
|
||||||
|
|
||||||
|
def _accepted(row: Any):
|
||||||
|
return getattr(row, "accepted_at", None) or getattr(row, "filed_date", None)
|
||||||
|
|
||||||
|
|
||||||
|
def _ordered_quarters(selected: dict[tuple[int, str], Any]) -> list[tuple[int, int]]:
|
||||||
|
return sorted((fy, _FP_TO_Q[fp]) for (fy, fp) in selected)
|
||||||
|
|
||||||
|
|
||||||
|
def _consecutive_suffix(quarters: list[tuple[int, int]], n: int) -> list[tuple[int, int]]:
|
||||||
|
"""The run of up to n quarters ending at the latest, walking back only through
|
||||||
|
adjacent periods (stop at the first gap). Returned oldest -> newest."""
|
||||||
|
if not quarters:
|
||||||
|
return []
|
||||||
|
present = set(quarters)
|
||||||
|
run = [quarters[-1]]
|
||||||
|
cur = quarters[-1]
|
||||||
|
while len(run) < n:
|
||||||
|
prev = _prev_q(*cur)
|
||||||
|
if prev not in present:
|
||||||
|
break
|
||||||
|
run.append(prev)
|
||||||
|
cur = prev
|
||||||
|
run.reverse()
|
||||||
|
return run
|
||||||
|
|
||||||
|
|
||||||
|
# -- discrete + TTM ----------------------------------------------------------
|
||||||
|
|
||||||
|
def _discrete_quarters(selected: dict[tuple[int, str], Any], field_name: str) -> dict[tuple[int, int], float]:
|
||||||
|
out: dict[tuple[int, int], float] = {}
|
||||||
|
for (fy, fp), row in selected.items():
|
||||||
|
val = _discrete_value(selected, fy, fp, field_name)
|
||||||
|
if val is not None:
|
||||||
|
out[(fy, _FP_TO_Q[fp])] = val
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _discrete_value(selected, fy: int, fp: str, field_name: str) -> float | None:
|
||||||
|
cur = getattr(selected[(fy, fp)], field_name, None)
|
||||||
|
if cur is None:
|
||||||
|
return None
|
||||||
|
if fp == "Q1":
|
||||||
|
return cur
|
||||||
|
prev = selected.get((fy, _PREV_FP[fp]))
|
||||||
|
prev_val = getattr(prev, field_name, None) if prev is not None else None
|
||||||
|
if prev_val is None:
|
||||||
|
return None
|
||||||
|
return cur - prev_val
|
||||||
|
|
||||||
|
|
||||||
|
def _ttm(dq: dict[tuple[int, int], float], fy: int, q: int) -> float | None:
|
||||||
|
keys = [(fy, q)]
|
||||||
|
k = (fy, q)
|
||||||
|
for _ in range(3):
|
||||||
|
k = _prev_q(*k)
|
||||||
|
keys.append(k)
|
||||||
|
vals = [dq.get(kk) for kk in keys]
|
||||||
|
if any(v is None for v in vals):
|
||||||
|
return None
|
||||||
|
return sum(vals)
|
||||||
|
|
||||||
|
|
||||||
|
def _pct_change(cur: float | None, prior: float | None) -> float | None:
|
||||||
|
# A non-positive prior makes a YoY % meaningless (e.g. loss->profit), so null it.
|
||||||
|
if cur is None or prior is None or prior <= 0:
|
||||||
|
return None
|
||||||
|
return (cur / prior - 1.0) * 100.0
|
||||||
|
|
||||||
|
|
||||||
|
# -- per-metric series (value at latest + tape history) ----------------------
|
||||||
|
|
||||||
|
def _period_end(selected, fy: int, q: int) -> date | None:
|
||||||
|
row = selected.get((fy, _Q_TO_FP[q]))
|
||||||
|
return row.period_end if row is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _yoy_growth_series(dq, selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
cur, prior = _ttm(dq, fy, q), _ttm(dq, fy - 1, q)
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), _pct_change(cur, prior)))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _margin_series(num_dq, den_dq, selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
num, den = _ttm(num_dq, fy, q), _ttm(den_dq, fy, q)
|
||||||
|
val = None if num is None or not den else num / den * 100.0
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), val))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _fcf_margin_series(discrete, selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
cfo, capex, rev = _ttm(discrete["cfo"], fy, q), _ttm(discrete["capex"], fy, q), _ttm(discrete["revenue"], fy, q)
|
||||||
|
val = None if cfo is None or capex is None or not rev else (cfo - capex) / rev * 100.0
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), val))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _instant_series(selected, tape, fn) -> MetricSeries:
|
||||||
|
pts = [MetricPoint(_period_end(selected, fy, q), fn(selected.get((fy, _Q_TO_FP[q])))) for (fy, q) in tape]
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _leverage_series(selected, discrete, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
row = selected.get((fy, _Q_TO_FP[q]))
|
||||||
|
nd = _net_debt(row)
|
||||||
|
op, da = _ttm(discrete["operating_income"], fy, q), _ttm(discrete["depreciation_amortization"], fy, q)
|
||||||
|
ebitda = None if op is None or da is None else op + da
|
||||||
|
# Null when EBITDA <= 0: a negative denominator would flip polarity and a
|
||||||
|
# "lower is better" read would rank a distressed issuer as favorable.
|
||||||
|
val = None if nd is None or ebitda is None or ebitda <= 0 else nd / ebitda
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), val))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _share_change_series(selected, tape) -> MetricSeries:
|
||||||
|
pts = []
|
||||||
|
for (fy, q) in tape:
|
||||||
|
cur = _shares(selected.get((fy, _Q_TO_FP[q])))
|
||||||
|
prior = _shares(selected.get((fy - 1, _Q_TO_FP[q])))
|
||||||
|
pts.append(MetricPoint(_period_end(selected, fy, q), _pct_change(cur, prior)))
|
||||||
|
return _series(pts)
|
||||||
|
|
||||||
|
|
||||||
|
def _net_debt(row: Any) -> float | None:
|
||||||
|
if row is None:
|
||||||
|
return None
|
||||||
|
cash = getattr(row, "cash_and_st_investments", None)
|
||||||
|
debt = getattr(row, "total_debt", None)
|
||||||
|
# Require BOTH components — treating a missing side as zero would produce a
|
||||||
|
# partial, misleading value.
|
||||||
|
if cash is None or debt is None:
|
||||||
|
return None
|
||||||
|
return debt - cash # positive = net debt
|
||||||
|
|
||||||
|
|
||||||
|
def _shares(row: Any) -> float | None:
|
||||||
|
return getattr(row, "shares_outstanding", None) if row is not None else None
|
||||||
|
|
||||||
|
|
||||||
|
def _series(points: list[MetricPoint]) -> MetricSeries:
|
||||||
|
value = points[-1].value if points else None
|
||||||
|
return MetricSeries(value=value, history=points)
|
||||||
@@ -0,0 +1,107 @@
|
|||||||
|
"""Pure peer comparison for fundamentals (read-time).
|
||||||
|
|
||||||
|
Peers are tracked-universe issuers sharing the **first two SIC digits**,
|
||||||
|
deduplicated by CIK (GOOG/GOOGL are one issuer, one observation). This module is
|
||||||
|
the pure statistics core: given a subject value and the peer group's values for a
|
||||||
|
metric, it returns median + polarity-aware favorable percentile + peer_count, or
|
||||||
|
None when there are fewer than the minimum valid peers (the caller then omits the
|
||||||
|
industry object entirely rather than show a misleading comparison).
|
||||||
|
|
||||||
|
Grouping (which issuers share a 2-digit SIC, CIK-dedup) is the API's job; this
|
||||||
|
module only does the math. **Absolute net_debt is size-dependent and must not get
|
||||||
|
a peer percentile** — leverage is compared via net_debt_to_ebitda.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
import statistics
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
MIN_PEERS = 5
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(v: Any) -> bool:
|
||||||
|
"""True for a finite number — excludes None, bool, NaN, ±inf (plan: null/invalid)."""
|
||||||
|
return isinstance(v, (int, float)) and not isinstance(v, bool) and math.isfinite(v)
|
||||||
|
|
||||||
|
# Metric -> is a higher value more favorable? (Peer-eligible metrics only;
|
||||||
|
# absolute net_debt is intentionally absent — size-dependent.)
|
||||||
|
HIGHER_IS_BETTER: dict[str, bool] = {
|
||||||
|
"revenue_growth_yoy": True,
|
||||||
|
"eps_growth_yoy": True,
|
||||||
|
"operating_margin": True,
|
||||||
|
"fcf_margin": True,
|
||||||
|
"fcf_yield": True,
|
||||||
|
"net_debt_to_ebitda": False, # lower leverage is better
|
||||||
|
"pe": False, # cheaper is better
|
||||||
|
"share_count_change_yoy": False, # dilution is bad
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class PeerStat:
|
||||||
|
median: float
|
||||||
|
favorable_percentile: int # 0-100, polarity-aware (higher = more favorable)
|
||||||
|
peer_count: int # valid issuers in the group
|
||||||
|
|
||||||
|
|
||||||
|
def peer_stat(
|
||||||
|
subject: float | None,
|
||||||
|
group_values: list[float | None],
|
||||||
|
*,
|
||||||
|
higher_is_better: bool,
|
||||||
|
min_peers: int = MIN_PEERS,
|
||||||
|
) -> PeerStat | None:
|
||||||
|
"""Median + favorable percentile for ``subject`` within its group.
|
||||||
|
|
||||||
|
``group_values`` is every issuer's value for the metric (including the
|
||||||
|
subject), CIK-deduplicated by the caller. Null/invalid (non-finite) values are
|
||||||
|
excluded. Returns None when fewer than ``min_peers`` valid values exist, or
|
||||||
|
the subject is null/invalid.
|
||||||
|
|
||||||
|
The percentile is a **tie-aware rank against the other issuers** —
|
||||||
|
``(worse + 0.5·tied) / (peers − 1)`` — so a whole group of equal values maps
|
||||||
|
to 50, not 100, and the median maps to 50.
|
||||||
|
"""
|
||||||
|
valid = [v for v in group_values if _finite(v)]
|
||||||
|
if not _finite(subject) or len(valid) < min_peers:
|
||||||
|
return None
|
||||||
|
median = statistics.median(valid)
|
||||||
|
|
||||||
|
others = valid.copy()
|
||||||
|
try:
|
||||||
|
others.remove(subject) # rank the subject against the OTHER issuers
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
denom = len(others)
|
||||||
|
if denom == 0:
|
||||||
|
return None
|
||||||
|
if higher_is_better:
|
||||||
|
worse = sum(1 for v in others if v < subject)
|
||||||
|
else:
|
||||||
|
worse = sum(1 for v in others if v > subject)
|
||||||
|
tied = sum(1 for v in others if v == subject)
|
||||||
|
percentile = round((worse + 0.5 * tied) / denom * 100)
|
||||||
|
return PeerStat(median=median, favorable_percentile=percentile, peer_count=len(valid))
|
||||||
|
|
||||||
|
|
||||||
|
def peer_stat_for(
|
||||||
|
metric_key: str, subject: float | None, group_values: list[float | None], **kwargs
|
||||||
|
) -> PeerStat | None:
|
||||||
|
"""Convenience wrapper that looks up polarity by metric key. Returns None for
|
||||||
|
metrics not eligible for peer comparison (e.g. absolute net_debt)."""
|
||||||
|
if metric_key not in HIGHER_IS_BETTER:
|
||||||
|
return None
|
||||||
|
return peer_stat(
|
||||||
|
subject, group_values, higher_is_better=HIGHER_IS_BETTER[metric_key], **kwargs
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def two_digit_sic(sic: str | None) -> str | None:
|
||||||
|
"""The 2-digit SIC prefix used for grouping, or None if unusable."""
|
||||||
|
if not sic:
|
||||||
|
return None
|
||||||
|
digits = str(sic).strip()
|
||||||
|
return digits[:2] if len(digits) >= 2 and digits[:2].isdigit() else None
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
"""Deterministic text 'reads' for the fundamentals panel (pure, one rule set).
|
||||||
|
|
||||||
|
The tape reads and the header sentence use identical outputs — no LLM, no new
|
||||||
|
composite score. Thresholds are tunable named constants, not scattered literals
|
||||||
|
(plan: ±2pp growth, ±1pp margins, ±1% dilution, 60/40 peer bands, ≥3 periods).
|
||||||
|
|
||||||
|
Consumers pass metric series (value + dated history, from
|
||||||
|
``fundamentals_derivation``) and peer percentiles; these functions return short
|
||||||
|
strings or None (render "—", no read).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from statistics import mean
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
MIN_PERIODS = 3
|
||||||
|
GROWTH_ACCEL_PP = 2.0
|
||||||
|
MARGIN_MOVE_PP = 1.0
|
||||||
|
SHARE_DILUTION_PCT = 1.0
|
||||||
|
PEER_FAVORABLE = 60
|
||||||
|
PEER_ADVERSE = 40
|
||||||
|
|
||||||
|
|
||||||
|
def _latest_run(history: list[Any]) -> list[float]:
|
||||||
|
"""The consecutive non-null values ending at the latest point (oldest->newest).
|
||||||
|
A null latest, or an internal gap, truncates the run — so a read never reflects
|
||||||
|
a period whose displayed value is n/a."""
|
||||||
|
run: list[float] = []
|
||||||
|
for p in reversed(history):
|
||||||
|
if p.value is None:
|
||||||
|
break
|
||||||
|
run.append(p.value)
|
||||||
|
run.reverse()
|
||||||
|
return run
|
||||||
|
|
||||||
|
|
||||||
|
def growth_read(history: list[Any]) -> str | None:
|
||||||
|
"""Change in a YoY-growth series: latest − prior. Needs >= 3 consecutive
|
||||||
|
non-null values ending at the latest point."""
|
||||||
|
vals = _latest_run(history)
|
||||||
|
if len(vals) < MIN_PERIODS:
|
||||||
|
return None
|
||||||
|
delta = vals[-1] - vals[-2]
|
||||||
|
if delta >= GROWTH_ACCEL_PP:
|
||||||
|
return "accelerating"
|
||||||
|
if delta <= -GROWTH_ACCEL_PP:
|
||||||
|
return "decelerating"
|
||||||
|
return "steady"
|
||||||
|
|
||||||
|
|
||||||
|
def margin_read(history: list[Any]) -> str | None:
|
||||||
|
"""Latest margin vs the mean of prior periods (pp). Needs >= 3 consecutive
|
||||||
|
non-null values ending at the latest point."""
|
||||||
|
vals = _latest_run(history)
|
||||||
|
if len(vals) < MIN_PERIODS:
|
||||||
|
return None
|
||||||
|
delta = vals[-1] - mean(vals[:-1])
|
||||||
|
if delta >= MARGIN_MOVE_PP:
|
||||||
|
return "improving"
|
||||||
|
if delta <= -MARGIN_MOVE_PP:
|
||||||
|
return "deteriorating"
|
||||||
|
return "stable"
|
||||||
|
|
||||||
|
|
||||||
|
def share_count_read(value: float | None) -> str | None:
|
||||||
|
"""Share-count YoY %: >+1% dilution, <-1% buying back, else flat."""
|
||||||
|
if value is None:
|
||||||
|
return None
|
||||||
|
if value > SHARE_DILUTION_PCT:
|
||||||
|
return f"{value:.1f}% dilution"
|
||||||
|
if value < -SHARE_DILUTION_PCT:
|
||||||
|
return "buying back"
|
||||||
|
return "flat"
|
||||||
|
|
||||||
|
|
||||||
|
def peer_read(metric_key: str, favorable_percentile: int | None) -> str | None:
|
||||||
|
"""Peer-relative read for a metric, polarity already baked into the
|
||||||
|
percentile (higher = more favorable)."""
|
||||||
|
if favorable_percentile is None:
|
||||||
|
return None
|
||||||
|
if favorable_percentile >= PEER_FAVORABLE:
|
||||||
|
return _FAVORABLE.get(metric_key, "above peers")
|
||||||
|
if favorable_percentile <= PEER_ADVERSE:
|
||||||
|
return _ADVERSE.get(metric_key, "below peers")
|
||||||
|
return "in line"
|
||||||
|
|
||||||
|
|
||||||
|
_FAVORABLE = {
|
||||||
|
"pe": "attractively valued",
|
||||||
|
"fcf_yield": "above peers",
|
||||||
|
"net_debt_to_ebitda": "conservative leverage",
|
||||||
|
}
|
||||||
|
_ADVERSE = {
|
||||||
|
"pe": "priced above peers",
|
||||||
|
"fcf_yield": "below peers",
|
||||||
|
"net_debt_to_ebitda": "elevated leverage",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def header_sentence(
|
||||||
|
growth: str | None, margin: str | None, valuation: str | None
|
||||||
|
) -> str:
|
||||||
|
"""Join the growth / margin / peer-valuation reads with ' · ', omitting
|
||||||
|
segments with no read. Segment sources are fixed by the caller (growth =
|
||||||
|
revenue-growth read, margin = operating-margin read, valuation = P/E peer
|
||||||
|
read falling back to FCF yield)."""
|
||||||
|
parts = []
|
||||||
|
if growth:
|
||||||
|
parts.append(f"growth {growth}")
|
||||||
|
if margin:
|
||||||
|
parts.append(f"margins {margin}")
|
||||||
|
if valuation:
|
||||||
|
parts.append(f"valuation {valuation}")
|
||||||
|
return " · ".join(parts)
|
||||||
@@ -0,0 +1,351 @@
|
|||||||
|
"""Async SEC EDGAR client for the fundamentals importer (workstream A).
|
||||||
|
|
||||||
|
All access is batch (never at request time). This wraps the three SEC products
|
||||||
|
the A3 design uses — `company_tickers.json`, `submissions/`, `companyfacts/`, and
|
||||||
|
the daily filing index — behind one client that honors SEC's fair-access policy:
|
||||||
|
|
||||||
|
- an identifying ``User-Agent`` with a contact email on every request (config);
|
||||||
|
- request spacing well under the 10 req/s limit;
|
||||||
|
- exponential backoff + retry on 429;
|
||||||
|
- **403 → alert and stop** (raise ``SecForbiddenError``), never a retry-loop — a
|
||||||
|
403 means the UA or request pattern is wrong and retrying won't fix it.
|
||||||
|
|
||||||
|
Parsing lives here (index fixed-width, submissions pagination); DB writes and the
|
||||||
|
snapshot mapping live in the importer. No conditional GETs — the companyfacts
|
||||||
|
endpoint exposes no ETag/Last-Modified (verified), which is why the importer is
|
||||||
|
daily-index driven rather than polling archives.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
from datetime import date, datetime
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from app.config import settings
|
||||||
|
from app.exceptions import ProviderError
|
||||||
|
from app.services.earnings_alignment import normalise_symbol
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
_WWW = "https://www.sec.gov"
|
||||||
|
_DATA = "https://data.sec.gov"
|
||||||
|
|
||||||
|
# Resolve CA bundle for explicit httpx verify (matches app/providers/fmp.py).
|
||||||
|
_CA = os.environ.get("SSL_CERT_FILE", "")
|
||||||
|
_CA_VERIFY: str | bool = _CA if _CA and Path(_CA).exists() else True
|
||||||
|
|
||||||
|
_FORMS_10 = frozenset({"10-K", "10-Q", "10-K/A", "10-Q/A"})
|
||||||
|
|
||||||
|
|
||||||
|
class SecError(ProviderError):
|
||||||
|
"""SEC request failed (403, exhausted 429/5xx, timeout, transport, parse)."""
|
||||||
|
|
||||||
|
|
||||||
|
class SecForbiddenError(SecError):
|
||||||
|
"""SEC returned 403 — User-Agent/pattern rejected. Alert and stop."""
|
||||||
|
|
||||||
|
|
||||||
|
class SecNotFoundError(SecError):
|
||||||
|
"""SEC returned 404 — the resource does not exist (e.g. no index for a day).
|
||||||
|
|
||||||
|
The *only* error a caller may treat as 'missing' — every other SecError
|
||||||
|
(403, exhausted retries, 5xx, timeout) must propagate so a fetch failure is
|
||||||
|
never mistaken for an empty result."""
|
||||||
|
|
||||||
|
|
||||||
|
def _looks_like_contact_email(ua: str) -> bool:
|
||||||
|
if "example.com" in ua.lower() or "set-a-real-email" in ua.lower():
|
||||||
|
return False
|
||||||
|
return re.search(r"[^@\s]+@[^@\s]+\.[^@\s]+", ua) is not None
|
||||||
|
|
||||||
|
|
||||||
|
# SEC asks callers to stay well under 10 req/s; enforce a floor on real clients.
|
||||||
|
_MIN_PROD_SPACING = 0.11
|
||||||
|
|
||||||
|
|
||||||
|
def cik10(cik: int | str) -> str:
|
||||||
|
"""Zero-pad a CIK to the 10-digit form SEC URLs use (320193 -> 0000320193)."""
|
||||||
|
return str(int(cik)).zfill(10)
|
||||||
|
|
||||||
|
|
||||||
|
class SecClient:
|
||||||
|
"""Fair-access SEC HTTP client. Use as ``async with SecClient() as c:``."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
user_agent: str | None = None,
|
||||||
|
spacing_seconds: float | None = None,
|
||||||
|
max_retries: int | None = None,
|
||||||
|
timeout: float | None = None,
|
||||||
|
transport: httpx.AsyncBaseTransport | None = None,
|
||||||
|
) -> None:
|
||||||
|
self._ua = user_agent or settings.sec_user_agent
|
||||||
|
self._spacing = (
|
||||||
|
spacing_seconds if spacing_seconds is not None else settings.sec_request_spacing_seconds
|
||||||
|
)
|
||||||
|
self._max_retries = (
|
||||||
|
max_retries if max_retries is not None else settings.sec_max_retries
|
||||||
|
)
|
||||||
|
self._timeout = timeout if timeout is not None else settings.sec_request_timeout_seconds
|
||||||
|
self._transport = transport # injectable for tests
|
||||||
|
self._client: httpx.AsyncClient | None = None
|
||||||
|
self._lock = asyncio.Lock()
|
||||||
|
self._last_request = 0.0
|
||||||
|
|
||||||
|
def _validate_fair_access(self) -> None:
|
||||||
|
"""On a real (non-mocked) client, enforce SEC fair-access preconditions
|
||||||
|
so we can't accidentally hammer SEC or get 403'd: a genuine contact-email
|
||||||
|
User-Agent and a spacing floor. Mock transports skip this (tests use 0)."""
|
||||||
|
if not _looks_like_contact_email(self._ua):
|
||||||
|
raise SecError(
|
||||||
|
"sec_user_agent must contain a real contact email (got "
|
||||||
|
f"{self._ua!r}) — SEC fair-access requires it"
|
||||||
|
)
|
||||||
|
if self._spacing < _MIN_PROD_SPACING:
|
||||||
|
raise SecError(
|
||||||
|
f"sec_request_spacing_seconds {self._spacing} is below the "
|
||||||
|
f"{_MIN_PROD_SPACING}s fair-access floor"
|
||||||
|
)
|
||||||
|
|
||||||
|
async def __aenter__(self) -> "SecClient":
|
||||||
|
if self._transport is None:
|
||||||
|
self._validate_fair_access()
|
||||||
|
self._client = httpx.AsyncClient(
|
||||||
|
headers={"User-Agent": self._ua, "Accept-Encoding": "gzip, deflate"},
|
||||||
|
timeout=self._timeout,
|
||||||
|
verify=_CA_VERIFY,
|
||||||
|
transport=self._transport,
|
||||||
|
)
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(self, *exc) -> None:
|
||||||
|
if self._client is not None:
|
||||||
|
await self._client.aclose()
|
||||||
|
self._client = None
|
||||||
|
|
||||||
|
async def _throttle(self) -> None:
|
||||||
|
async with self._lock:
|
||||||
|
now = asyncio.get_event_loop().time()
|
||||||
|
wait = self._spacing - (now - self._last_request)
|
||||||
|
if wait > 0:
|
||||||
|
await asyncio.sleep(wait)
|
||||||
|
self._last_request = asyncio.get_event_loop().time()
|
||||||
|
|
||||||
|
async def _get(self, url: str) -> httpx.Response:
|
||||||
|
assert self._client is not None, "use `async with SecClient()`"
|
||||||
|
attempt = 0
|
||||||
|
while True:
|
||||||
|
await self._throttle()
|
||||||
|
try:
|
||||||
|
resp = await self._client.get(url)
|
||||||
|
except (httpx.TimeoutException, httpx.TransportError) as exc:
|
||||||
|
attempt += 1
|
||||||
|
if attempt > self._max_retries:
|
||||||
|
raise SecError(f"SEC network error for {url}: {exc}") from exc
|
||||||
|
await asyncio.sleep(min(2.0**attempt, 30.0))
|
||||||
|
continue
|
||||||
|
|
||||||
|
code = resp.status_code
|
||||||
|
if code == 403:
|
||||||
|
raise SecForbiddenError(
|
||||||
|
f"SEC 403 for {url} — User-Agent/pattern rejected; set a real "
|
||||||
|
"sec_user_agent contact email"
|
||||||
|
)
|
||||||
|
if code == 404:
|
||||||
|
raise SecNotFoundError(f"SEC 404 for {url}")
|
||||||
|
# 429 and 5xx are transient — retry with backoff, honoring Retry-After.
|
||||||
|
if code == 429 or 500 <= code < 600:
|
||||||
|
attempt += 1
|
||||||
|
if attempt > self._max_retries:
|
||||||
|
raise SecError(f"SEC {code} after {self._max_retries} retries: {url}")
|
||||||
|
delay = _retry_after_seconds(resp) or min(2.0**attempt, 30.0)
|
||||||
|
logger.warning("SEC %d for %s — backoff %.1fs (attempt %d)", code, url, delay, attempt)
|
||||||
|
await asyncio.sleep(delay)
|
||||||
|
continue
|
||||||
|
if code >= 400:
|
||||||
|
raise SecError(f"SEC {code} for {url}")
|
||||||
|
return resp
|
||||||
|
|
||||||
|
async def get_json(self, url: str) -> Any:
|
||||||
|
return (await self._get(url)).json()
|
||||||
|
|
||||||
|
async def get_text(self, url: str) -> str:
|
||||||
|
return (await self._get(url)).text
|
||||||
|
|
||||||
|
# -- domain fetchers ---------------------------------------------------
|
||||||
|
|
||||||
|
async def company_tickers(self) -> dict[str, int]:
|
||||||
|
"""Map normalised ticker -> CIK (int). Multi-class tickers share a CIK."""
|
||||||
|
data = await self.get_json(f"{_WWW}/files/company_tickers.json")
|
||||||
|
out: dict[str, int] = {}
|
||||||
|
for row in data.values():
|
||||||
|
sym = normalise_symbol(row.get("ticker"))
|
||||||
|
if sym:
|
||||||
|
out[sym] = int(row["cik_str"])
|
||||||
|
return out
|
||||||
|
|
||||||
|
async def submissions(self, cik: int | str, *, include_history: bool = False) -> dict[str, Any]:
|
||||||
|
"""Issuer metadata + filing list.
|
||||||
|
|
||||||
|
``filings.recent`` caps at 1000; older accessions live in
|
||||||
|
``filings.files[]`` shards. Only ``include_history=True`` (the one-time
|
||||||
|
full backfill) fetches those shards — SIC refresh and incremental runs
|
||||||
|
use the recent list alone and make no extra requests.
|
||||||
|
"""
|
||||||
|
base = await self.get_json(f"{_DATA}/submissions/CIK{cik10(cik)}.json")
|
||||||
|
filings = _rows_from_arrays(base["filings"]["recent"])
|
||||||
|
if include_history:
|
||||||
|
for shard in base["filings"].get("files") or []:
|
||||||
|
shard_data = await self.get_json(f"{_DATA}/submissions/{shard['name']}")
|
||||||
|
filings.extend(_rows_from_arrays(shard_data))
|
||||||
|
return {
|
||||||
|
"cik": int(base["cik"]),
|
||||||
|
"name": base.get("name"),
|
||||||
|
"sic": base.get("sic"),
|
||||||
|
"sic_description": base.get("sicDescription"),
|
||||||
|
"fiscal_year_end": base.get("fiscalYearEnd"),
|
||||||
|
"tickers": base.get("tickers") or [],
|
||||||
|
"filings": filings,
|
||||||
|
}
|
||||||
|
|
||||||
|
async def companyfacts(self, cik: int | str) -> dict[str, Any]:
|
||||||
|
"""Raw companyfacts JSON ({cik, entityName, facts})."""
|
||||||
|
return await self.get_json(f"{_DATA}/api/xbrl/companyfacts/CIK{cik10(cik)}.json")
|
||||||
|
|
||||||
|
async def latest_index_date(self, today: date | None = None) -> date | None:
|
||||||
|
"""The most recent published daily-index date (drives the revision). Checks
|
||||||
|
the current quarter, falling back to the previous one at a quarter boundary."""
|
||||||
|
today = today or date.today()
|
||||||
|
for year, qtr in _quarters_back(today, 2):
|
||||||
|
url = f"{_WWW}/Archives/edgar/daily-index/{year}/QTR{qtr}/index.json"
|
||||||
|
try:
|
||||||
|
idx = await self.get_json(url)
|
||||||
|
except SecNotFoundError:
|
||||||
|
continue # quarter dir absent — only 404 is "missing"
|
||||||
|
dates = [
|
||||||
|
d
|
||||||
|
for item in idx.get("directory", {}).get("item", [])
|
||||||
|
if (d := _index_file_date(item.get("name", ""))) is not None
|
||||||
|
and d <= today
|
||||||
|
]
|
||||||
|
if dates:
|
||||||
|
return max(dates)
|
||||||
|
return None
|
||||||
|
|
||||||
|
async def daily_index(self, day: date) -> list[dict[str, Any]]:
|
||||||
|
"""Parse the daily form index into 10-K/10-Q(/A) rows for all issuers.
|
||||||
|
|
||||||
|
Returns [{form, cik, accession, company}]. The caller filters to the
|
||||||
|
tracked universe. A missing index (weekend/holiday/not-yet-published)
|
||||||
|
returns [] rather than raising.
|
||||||
|
"""
|
||||||
|
qtr = (day.month - 1) // 3 + 1
|
||||||
|
url = f"{_WWW}/Archives/edgar/daily-index/{day.year}/QTR{qtr}/form.{day:%Y%m%d}.idx"
|
||||||
|
try:
|
||||||
|
text = await self.get_text(url)
|
||||||
|
except SecNotFoundError:
|
||||||
|
logger.info("no daily index for %s (404)", day)
|
||||||
|
return [] # weekend/holiday/not-yet-published; other errors propagate
|
||||||
|
return _parse_form_index(text)
|
||||||
|
|
||||||
|
|
||||||
|
def _rows_from_arrays(arrays: dict[str, list]) -> list[dict[str, Any]]:
|
||||||
|
"""Turn SEC's parallel-array filing block into row dicts (keeping only 10-K/10-Q
|
||||||
|
family filings — the ones that carry XBRL fundamentals)."""
|
||||||
|
forms = arrays.get("form", [])
|
||||||
|
out: list[dict[str, Any]] = []
|
||||||
|
for i, form in enumerate(forms):
|
||||||
|
if form not in _FORMS_10:
|
||||||
|
continue
|
||||||
|
out.append(
|
||||||
|
{
|
||||||
|
"accession": arrays["accessionNumber"][i],
|
||||||
|
"form": form,
|
||||||
|
"report_date": arrays["reportDate"][i] or None,
|
||||||
|
"filing_date": arrays["filingDate"][i] or None,
|
||||||
|
"acceptance_datetime": arrays["acceptanceDateTime"][i] or None,
|
||||||
|
"is_xbrl": bool(arrays.get("isXBRL", [0] * len(forms))[i]),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_form_index(text: str) -> list[dict[str, Any]]:
|
||||||
|
"""Parse a daily ``form.YYYYMMDD.idx`` (fixed columns: Form / Company / CIK /
|
||||||
|
Date Filed / File Name-with-accession)."""
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
started = False
|
||||||
|
for line in text.splitlines():
|
||||||
|
if not started:
|
||||||
|
if set(line.strip()) == {"-"}: # the dashed separator row
|
||||||
|
started = True
|
||||||
|
continue
|
||||||
|
parts = line.split()
|
||||||
|
if len(parts) < 5:
|
||||||
|
continue
|
||||||
|
form = parts[0]
|
||||||
|
if form not in _FORMS_10:
|
||||||
|
continue
|
||||||
|
path = parts[-1] # edgar/data/<cik>/<accession>.txt
|
||||||
|
cik = _cik_from_path(path)
|
||||||
|
accession = _accession_from_path(path)
|
||||||
|
if cik is None or accession is None:
|
||||||
|
continue
|
||||||
|
rows.append({"form": form, "cik": cik, "accession": accession, "path": path})
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
|
def _retry_after_seconds(resp: httpx.Response) -> float | None:
|
||||||
|
"""Parse a numeric-seconds Retry-After header (SEC uses seconds), capped."""
|
||||||
|
raw = resp.headers.get("Retry-After")
|
||||||
|
if not raw:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return min(float(raw), 60.0)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _index_file_date(name: str) -> date | None:
|
||||||
|
if name.startswith("form.") and name.endswith(".idx"):
|
||||||
|
try:
|
||||||
|
return datetime.strptime(name[5:13], "%Y%m%d").date()
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _quarters_back(today: date, n: int) -> list[tuple[int, int]]:
|
||||||
|
"""(year, quarter) for `today`'s quarter and the previous n-1, newest first."""
|
||||||
|
q = (today.month - 1) // 3 + 1
|
||||||
|
out = []
|
||||||
|
y = today.year
|
||||||
|
for _ in range(n):
|
||||||
|
out.append((y, q))
|
||||||
|
q -= 1
|
||||||
|
if q == 0:
|
||||||
|
q = 4
|
||||||
|
y -= 1
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _cik_from_path(path: str) -> int | None:
|
||||||
|
segs = path.split("/")
|
||||||
|
if len(segs) >= 3 and segs[2].isdigit():
|
||||||
|
return int(segs[2])
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _accession_from_path(path: str) -> str | None:
|
||||||
|
stem = path.rsplit("/", 1)[-1]
|
||||||
|
if stem.endswith(".txt"):
|
||||||
|
stem = stem[:-4]
|
||||||
|
return stem or None
|
||||||
@@ -0,0 +1,366 @@
|
|||||||
|
"""Pure parser: SEC companyfacts -> fundamental_snapshots rows.
|
||||||
|
|
||||||
|
Turns one issuer's `companyfacts` JSON (+ its submissions filing metadata) into
|
||||||
|
per-accession snapshot rows for the filing's **primary period**, following the
|
||||||
|
A3 design (docs/dolt-sec-a3-design.md). No I/O, no DB — unit-testable against a
|
||||||
|
fixture and verifiable against a real companyfacts pull.
|
||||||
|
|
||||||
|
The load-bearing rules (design Decision 2 + review):
|
||||||
|
- Period identity comes from `end == submissions.reportDate`, never `fy/fp`
|
||||||
|
(fy/fp is the *filing's* context; comparatives inside a filing repeat it).
|
||||||
|
- Duration facts are stored as **cumulative YTD**: pick the fact whose span
|
||||||
|
matches the fiscal-period-to-date length (Q1≈3mo … FY≈12mo) within tolerance.
|
||||||
|
If no YTD-length fact exists, store null — never a discrete masquerading as YTD.
|
||||||
|
- Balance-sheet instants are taken at `end == reportDate`. `shares_outstanding`
|
||||||
|
is a single consolidated value: the cover-page `dei` fact (its own cover-date
|
||||||
|
`end` stored separately) if present, else `us-gaap:CommonStockSharesOutstanding`
|
||||||
|
at period end (e.g. Alphabet has no `dei` fact) — never a class sum or the
|
||||||
|
weighted-average/diluted count.
|
||||||
|
- Cash and debt composites are aggregate-first and mutually exclusive (each
|
||||||
|
source tag counted at most once).
|
||||||
|
|
||||||
|
`parse_snapshots` separates `skipped_filings` (no usable row produced) from
|
||||||
|
`field_issues` (a row was produced but a field is null/ambiguous) — callers must
|
||||||
|
not treat field issues as missing coverage.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import math
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime
|
||||||
|
from typing import Any, NamedTuple
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Expected YTD span (days) per fiscal period; a duration fact must land within
|
||||||
|
# tolerance of this to count as the period's cumulative value.
|
||||||
|
_EXPECTED_YTD_DAYS = {"Q1": 91, "Q2": 182, "Q3": 273, "FY": 365}
|
||||||
|
_YTD_TOLERANCE_DAYS = 20 # covers 52/53-week fiscal calendars
|
||||||
|
|
||||||
|
# us-gaap duration concepts (money), priority order; first present wins.
|
||||||
|
_DURATION_USD = {
|
||||||
|
"revenue": [
|
||||||
|
"RevenueFromContractWithCustomerExcludingAssessedTax",
|
||||||
|
"Revenues",
|
||||||
|
"SalesRevenueNet",
|
||||||
|
],
|
||||||
|
"net_income": ["NetIncomeLoss"],
|
||||||
|
"operating_income": ["OperatingIncomeLoss"],
|
||||||
|
"cfo": [
|
||||||
|
"NetCashProvidedByUsedInOperatingActivities",
|
||||||
|
"NetCashProvidedByUsedInOperatingActivitiesContinuingOperations",
|
||||||
|
],
|
||||||
|
"capex": [
|
||||||
|
"PaymentsToAcquirePropertyPlantAndEquipment",
|
||||||
|
"PaymentsToAcquireProductiveAssets",
|
||||||
|
],
|
||||||
|
"depreciation_amortization": [
|
||||||
|
"DepreciationDepletionAndAmortization",
|
||||||
|
"DepreciationAmortizationAndAccretionNet",
|
||||||
|
"DepreciationAndAmortization",
|
||||||
|
],
|
||||||
|
}
|
||||||
|
_EPS_CONCEPTS = ["EarningsPerShareDiluted"] # unit USD/shares
|
||||||
|
# us-gaap instant (balance-sheet) concepts, at end == reportDate.
|
||||||
|
_CASH = ["CashAndCashEquivalentsAtCarryingValue"]
|
||||||
|
_ST_INVESTMENTS = ["ShortTermInvestments", "MarketableSecuritiesCurrent"] # pick one
|
||||||
|
_LONG_TERM_DEBT_AGG = ["LongTermDebt"]
|
||||||
|
_LONG_TERM_DEBT_PARTS = ["LongTermDebtNoncurrent", "LongTermDebtCurrent"]
|
||||||
|
_SHORT_TERM_DEBT = ["ShortTermBorrowings", "CommercialPaper"] # pick one
|
||||||
|
|
||||||
|
|
||||||
|
class Fact(NamedTuple):
|
||||||
|
taxonomy: str
|
||||||
|
concept: str
|
||||||
|
unit: str
|
||||||
|
start: date | None # None => instant
|
||||||
|
end: date
|
||||||
|
val: float
|
||||||
|
fy: int | None
|
||||||
|
fp: str | None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class SnapshotRow:
|
||||||
|
cik: str
|
||||||
|
accession: str
|
||||||
|
form: str
|
||||||
|
filed_date: date
|
||||||
|
accepted_at: datetime
|
||||||
|
period_end: date
|
||||||
|
fiscal_year: int
|
||||||
|
fiscal_period: str
|
||||||
|
period_start: date | None = None
|
||||||
|
revenue: float | None = None
|
||||||
|
net_income: float | None = None
|
||||||
|
operating_income: float | None = None
|
||||||
|
diluted_eps: float | None = None
|
||||||
|
cfo: float | None = None
|
||||||
|
capex: float | None = None
|
||||||
|
depreciation_amortization: float | None = None
|
||||||
|
cash_and_st_investments: float | None = None
|
||||||
|
total_debt: float | None = None
|
||||||
|
shares_outstanding: float | None = None
|
||||||
|
shares_outstanding_date: date | None = None
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class FilingMeta:
|
||||||
|
report_date: date
|
||||||
|
filing_date: date
|
||||||
|
accepted_at: datetime
|
||||||
|
form: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ParseResult:
|
||||||
|
rows: list[SnapshotRow] = field(default_factory=list)
|
||||||
|
# accessions for which NO row was produced (no facts / no usable period).
|
||||||
|
skipped_filings: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
# accessions with a row but a field-level warning (e.g. ambiguous shares).
|
||||||
|
field_issues: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_snapshots(
|
||||||
|
companyfacts: dict[str, Any],
|
||||||
|
filings: dict[str, FilingMeta],
|
||||||
|
accessions: set[str],
|
||||||
|
) -> ParseResult:
|
||||||
|
"""Build snapshot rows for ``accessions`` (those with facts + filing meta).
|
||||||
|
|
||||||
|
``skipped_filings`` = no row produced (missing facts/meta or no usable period
|
||||||
|
identity); ``field_issues`` = a row was produced but a field is null/ambiguous.
|
||||||
|
Callers must not use field issues as failed-row coverage.
|
||||||
|
"""
|
||||||
|
cik = f"{int(companyfacts['cik']):010d}"
|
||||||
|
by_accn = _index_by_accession(companyfacts)
|
||||||
|
result = ParseResult()
|
||||||
|
for accn in accessions:
|
||||||
|
meta = filings.get(accn)
|
||||||
|
facts = by_accn.get(accn)
|
||||||
|
if meta is None or not facts:
|
||||||
|
result.skipped_filings.append({"accession": accn, "reason": "no facts or filing metadata"})
|
||||||
|
continue
|
||||||
|
row, note = _parse_one(cik, accn, facts, meta)
|
||||||
|
if row is None:
|
||||||
|
result.skipped_filings.append({"accession": accn, "reason": note or "unparseable"})
|
||||||
|
continue
|
||||||
|
result.rows.append(row)
|
||||||
|
if note:
|
||||||
|
result.field_issues.append({"accession": accn, "reason": note})
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def companyfacts_accessions(companyfacts: dict[str, Any]) -> set[str]:
|
||||||
|
"""Every accession that appears anywhere in a companyfacts payload — used by
|
||||||
|
the importer's index↔Company-Facts consistency gate."""
|
||||||
|
return set(_index_by_accession(companyfacts).keys())
|
||||||
|
|
||||||
|
|
||||||
|
def _index_by_accession(companyfacts: dict[str, Any]) -> dict[str, list[Fact]]:
|
||||||
|
"""One pass over companyfacts -> {accession: [Fact, ...]}."""
|
||||||
|
out: dict[str, list[Fact]] = {}
|
||||||
|
for taxonomy, concepts in companyfacts.get("facts", {}).items():
|
||||||
|
for concept, body in concepts.items():
|
||||||
|
for unit, facts in body.get("units", {}).items():
|
||||||
|
for f in facts:
|
||||||
|
accn = f.get("accn")
|
||||||
|
end = _d(f.get("end"))
|
||||||
|
val = f.get("val")
|
||||||
|
# Skip malformed facts so they can't be selected accidentally:
|
||||||
|
# every usable fact needs an accession, an end date, and a
|
||||||
|
# finite numeric value.
|
||||||
|
if not accn or end is None or not _finite(val):
|
||||||
|
continue
|
||||||
|
out.setdefault(accn, []).append(
|
||||||
|
Fact(
|
||||||
|
taxonomy=taxonomy,
|
||||||
|
concept=concept,
|
||||||
|
unit=unit,
|
||||||
|
start=_d(f.get("start")),
|
||||||
|
end=end,
|
||||||
|
val=val,
|
||||||
|
fy=f.get("fy"),
|
||||||
|
fp=f.get("fp"),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_one(
|
||||||
|
cik: str, accn: str, facts: list[Fact], meta: FilingMeta
|
||||||
|
) -> tuple[SnapshotRow | None, str | None]:
|
||||||
|
"""Returns (row, note). row is None when there's no usable period identity;
|
||||||
|
note is a validation reason (row-skip reason when row is None, else a
|
||||||
|
field-level issue such as ambiguous shares)."""
|
||||||
|
fy, fp = _fiscal_context(facts, meta.report_date)
|
||||||
|
if fy is None or fp not in _EXPECTED_YTD_DAYS:
|
||||||
|
return None, "no usable period identity"
|
||||||
|
|
||||||
|
row = SnapshotRow(
|
||||||
|
cik=cik,
|
||||||
|
accession=accn,
|
||||||
|
form=meta.form,
|
||||||
|
filed_date=meta.filing_date,
|
||||||
|
accepted_at=meta.accepted_at,
|
||||||
|
period_end=meta.report_date,
|
||||||
|
fiscal_year=fy,
|
||||||
|
fiscal_period=fp,
|
||||||
|
)
|
||||||
|
|
||||||
|
# duration YTD facts (money) + EPS
|
||||||
|
for field_name, concepts in _DURATION_USD.items():
|
||||||
|
val, start = _select_ytd(facts, concepts, meta.report_date, fp, "USD")
|
||||||
|
setattr(row, field_name, val)
|
||||||
|
if field_name == "revenue" and start is not None:
|
||||||
|
row.period_start = start
|
||||||
|
eps, eps_start = _select_ytd(facts, _EPS_CONCEPTS, meta.report_date, fp, "USD/shares")
|
||||||
|
row.diluted_eps = eps
|
||||||
|
if row.period_start is None and eps_start is not None:
|
||||||
|
row.period_start = eps_start
|
||||||
|
|
||||||
|
# balance-sheet instants at reportDate
|
||||||
|
row.cash_and_st_investments = _compose_cash(facts, meta.report_date)
|
||||||
|
row.total_debt = _compose_debt(facts, meta.report_date)
|
||||||
|
shares, shares_date, ambiguous = _select_shares(facts, meta.report_date)
|
||||||
|
row.shares_outstanding = shares
|
||||||
|
row.shares_outstanding_date = shares_date
|
||||||
|
return row, ("ambiguous shares outstanding" if ambiguous else None)
|
||||||
|
|
||||||
|
|
||||||
|
def _fiscal_context(facts: list[Fact], report_date: date) -> tuple[int | None, str | None]:
|
||||||
|
"""The filing's (fy, fp) taken as the majority context among the facts that
|
||||||
|
end at reportDate (the current-period facts, which share the filing's
|
||||||
|
context). Reject a tie so a conflicting context is never chosen arbitrarily."""
|
||||||
|
counts: dict[tuple[int, str], int] = {}
|
||||||
|
for f in facts:
|
||||||
|
if f.end == report_date and f.fy is not None and f.fp:
|
||||||
|
counts[(f.fy, f.fp)] = counts.get((f.fy, f.fp), 0) + 1
|
||||||
|
if not counts:
|
||||||
|
return None, None
|
||||||
|
ranked = sorted(counts.items(), key=lambda kv: kv[1], reverse=True)
|
||||||
|
if len(ranked) > 1 and ranked[0][1] == ranked[1][1]:
|
||||||
|
return None, None # tie → conflicting contexts, reject
|
||||||
|
return ranked[0][0]
|
||||||
|
|
||||||
|
|
||||||
|
def _select_ytd(
|
||||||
|
facts: list[Fact], concepts: list[str], report_date: date, fp: str, unit: str
|
||||||
|
) -> tuple[float | None, date | None]:
|
||||||
|
"""First present concept whose duration fact ends at reportDate and whose span
|
||||||
|
matches the fiscal-period-to-date length. Returns (val, period_start)."""
|
||||||
|
expected = _EXPECTED_YTD_DAYS[fp]
|
||||||
|
for concept in concepts:
|
||||||
|
best: Fact | None = None
|
||||||
|
best_diff: int | None = None
|
||||||
|
for f in facts:
|
||||||
|
if (
|
||||||
|
f.taxonomy != "us-gaap"
|
||||||
|
or f.concept != concept
|
||||||
|
or f.unit != unit
|
||||||
|
or f.start is None
|
||||||
|
or f.end != report_date
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
diff = abs((f.end - f.start).days - expected)
|
||||||
|
if diff <= _YTD_TOLERANCE_DAYS and (best_diff is None or diff < best_diff):
|
||||||
|
best, best_diff = f, diff
|
||||||
|
if best is not None:
|
||||||
|
return float(best.val), best.start
|
||||||
|
return None, None
|
||||||
|
|
||||||
|
|
||||||
|
def _select_instant(facts: list[Fact], concepts: list[str], report_date: date) -> float | None:
|
||||||
|
"""First present instant (balance-sheet) fact at end == reportDate, unit USD."""
|
||||||
|
for concept in concepts:
|
||||||
|
for f in facts:
|
||||||
|
if (
|
||||||
|
f.taxonomy == "us-gaap"
|
||||||
|
and f.concept == concept
|
||||||
|
and f.unit == "USD"
|
||||||
|
and f.start is None
|
||||||
|
and f.end == report_date
|
||||||
|
):
|
||||||
|
return float(f.val)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _compose_cash(facts: list[Fact], report_date: date) -> float | None:
|
||||||
|
cash = _select_instant(facts, _CASH, report_date)
|
||||||
|
st = _select_instant(facts, _ST_INVESTMENTS, report_date) # first present of the two
|
||||||
|
if cash is None and st is None:
|
||||||
|
return None
|
||||||
|
return (cash or 0.0) + (st or 0.0)
|
||||||
|
|
||||||
|
|
||||||
|
def _compose_debt(facts: list[Fact], report_date: date) -> float | None:
|
||||||
|
long_term = _select_instant(facts, _LONG_TERM_DEBT_AGG, report_date)
|
||||||
|
if long_term is None:
|
||||||
|
nc = _select_instant(facts, ["LongTermDebtNoncurrent"], report_date)
|
||||||
|
cur = _select_instant(facts, ["LongTermDebtCurrent"], report_date)
|
||||||
|
long_term = None if nc is None and cur is None else (nc or 0.0) + (cur or 0.0)
|
||||||
|
short_term = _select_instant(facts, _SHORT_TERM_DEBT, report_date)
|
||||||
|
if long_term is None and short_term is None:
|
||||||
|
return None
|
||||||
|
return (long_term or 0.0) + (short_term or 0.0)
|
||||||
|
|
||||||
|
|
||||||
|
def _select_shares(
|
||||||
|
facts: list[Fact], report_date: date
|
||||||
|
) -> tuple[float | None, date | None, bool]:
|
||||||
|
"""Issuer-wide shares outstanding as a single consolidated value (never a
|
||||||
|
class sum — companyfacts is non-dimensional — and never weighted-average/
|
||||||
|
diluted). Returns (value, shares_date, ambiguous).
|
||||||
|
|
||||||
|
1. Prefer the `dei:EntityCommonStockSharesOutstanding` cover-page instant;
|
||||||
|
its own end is the shares date (cover date != period_end).
|
||||||
|
2. Else fall back to `us-gaap:CommonStockSharesOutstanding` at period end
|
||||||
|
(e.g. Alphabet has no dei fact); shares date = reportDate.
|
||||||
|
Conflicting values within the chosen source → (None, None, True) to be
|
||||||
|
counted in validation.
|
||||||
|
"""
|
||||||
|
dei = [
|
||||||
|
f
|
||||||
|
for f in facts
|
||||||
|
if f.taxonomy == "dei"
|
||||||
|
and f.concept == "EntityCommonStockSharesOutstanding"
|
||||||
|
and f.unit == "shares"
|
||||||
|
and f.start is None
|
||||||
|
]
|
||||||
|
if dei:
|
||||||
|
if len({f.val for f in dei}) > 1:
|
||||||
|
return None, None, True
|
||||||
|
best = max(dei, key=lambda f: f.end)
|
||||||
|
return float(best.val), best.end, False
|
||||||
|
|
||||||
|
gaap = [
|
||||||
|
f
|
||||||
|
for f in facts
|
||||||
|
if f.taxonomy == "us-gaap"
|
||||||
|
and f.concept == "CommonStockSharesOutstanding"
|
||||||
|
and f.unit == "shares"
|
||||||
|
and f.start is None
|
||||||
|
and f.end == report_date
|
||||||
|
]
|
||||||
|
if gaap:
|
||||||
|
if len({f.val for f in gaap}) > 1:
|
||||||
|
return None, None, True
|
||||||
|
return float(gaap[0].val), report_date, False
|
||||||
|
|
||||||
|
return None, None, False # simply absent — not a conflict
|
||||||
|
|
||||||
|
|
||||||
|
def _d(value: Any) -> date | None:
|
||||||
|
if not value:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
return date.fromisoformat(str(value)[:10])
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _finite(value: Any) -> bool:
|
||||||
|
"""True for a finite numeric value (rejects None, bool, strings, NaN/inf)."""
|
||||||
|
return isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(value)
|
||||||
@@ -0,0 +1,399 @@
|
|||||||
|
"""SEC fundamentals importer (workstream A, phase A3).
|
||||||
|
|
||||||
|
A ``SourceImporter`` (see ``app/services/data_import.py``) that populates the
|
||||||
|
immutable ``fundamental_snapshots`` from SEC Company Facts and back-fills
|
||||||
|
``tickers.cik/sic/sic_description``. EDGAR-daily-index driven: it fetches
|
||||||
|
companyfacts only for tracked issuers that filed since the last run (full-history
|
||||||
|
backfill on the first run / for newly-added issuers). Shadow only — nothing reads
|
||||||
|
snapshots until A4.
|
||||||
|
|
||||||
|
Guardrails (design + reviews):
|
||||||
|
- ``detect_revision`` caches the resolved universe + the exact tracked index rows
|
||||||
|
and composes the revision from them; ``stage`` consumes those same cached inputs
|
||||||
|
(it does not refetch the index/universe) so promoted data always matches the
|
||||||
|
computed revision.
|
||||||
|
- Resolution is read-only in ``stage`` (proposals only); ticker writes happen in
|
||||||
|
``promote`` via ``apply_ticker_updates``.
|
||||||
|
- ``validate`` runs the **index↔Company-Facts consistency gate** before any write:
|
||||||
|
a tracked XBRL index accession missing from Company Facts fails the run (the two
|
||||||
|
are separate SEC products that can lag) so we retry rather than record a
|
||||||
|
null/partial snapshot. Non-XBRL amendments are skipped with a recorded reason.
|
||||||
|
- ``promote`` inserts snapshots ``ON CONFLICT (accession) DO NOTHING`` (immutable),
|
||||||
|
reports differing existing accessions, and applies ticker updates in the same
|
||||||
|
transaction.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from collections import defaultdict
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import date, datetime, timedelta, timezone
|
||||||
|
from typing import Any, Callable
|
||||||
|
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
|
||||||
|
from app.database import insert_for_session
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.system_event import SystemEvent
|
||||||
|
from app.services import sec_facts_parser as parser
|
||||||
|
from app.services import sec_universe
|
||||||
|
from app.services.data_import import STATUS_PROMOTED, ValidationResult
|
||||||
|
from app.services.sec_client import SecClient, SecError, cik10
|
||||||
|
from app.services.sec_facts_parser import FilingMeta, SnapshotRow
|
||||||
|
from app.services.sec_universe import ResolvedUniverse
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
SOURCE = "sec_facts"
|
||||||
|
_XBRL_FORMS = {"10-K", "10-Q", "10-K/A", "10-Q/A"}
|
||||||
|
# On the one-time backfill, require this fraction of tracked issuers to yield at
|
||||||
|
# least one snapshot (guards a broken fetch/parse from promoting a hollow table).
|
||||||
|
MIN_BACKFILL_COVERAGE = 0.5
|
||||||
|
|
||||||
|
_SNAPSHOT_COLS = (
|
||||||
|
"cik", "accession", "form", "filed_date", "accepted_at", "period_start",
|
||||||
|
"period_end", "fiscal_year", "fiscal_period", "revenue", "net_income",
|
||||||
|
"operating_income", "diluted_eps", "cfo", "capex", "depreciation_amortization",
|
||||||
|
"cash_and_st_investments", "total_debt", "shares_outstanding",
|
||||||
|
"shares_outstanding_date",
|
||||||
|
)
|
||||||
|
# Compare ALL source fields (every column except the accession key) to flag a
|
||||||
|
# differing existing accession — immutable, so we report, never mutate.
|
||||||
|
_COMPARE_COLS = tuple(c for c in _SNAPSHOT_COLS if c != "accession")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class StagedFundamentals:
|
||||||
|
resolved: ResolvedUniverse
|
||||||
|
sic_updates: list[tuple[int, str | None, str | None]] = field(default_factory=list)
|
||||||
|
rows: list[SnapshotRow] = field(default_factory=list)
|
||||||
|
skipped_filings: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
field_issues: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
skipped_non_xbrl: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
missing_xbrl: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
invalid_payloads: list[dict[str, str]] = field(default_factory=list)
|
||||||
|
existing_accessions: set[str] = field(default_factory=set)
|
||||||
|
discrepancies: list[dict[str, Any]] = field(default_factory=list)
|
||||||
|
backfill: bool = False
|
||||||
|
issuers_fetched: int = 0
|
||||||
|
issuers_with_rows: int = 0
|
||||||
|
|
||||||
|
|
||||||
|
def _now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
class SecFundamentalsImporter:
|
||||||
|
source = SOURCE
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
client_factory: Callable[[], SecClient] | None = None,
|
||||||
|
today: date | None = None,
|
||||||
|
) -> None:
|
||||||
|
self._client_factory = client_factory or (lambda: SecClient())
|
||||||
|
self.today = today or _now().date()
|
||||||
|
# cached by detect_revision, consumed by stage:
|
||||||
|
self._resolved: ResolvedUniverse | None = None
|
||||||
|
self._index_rows: list[dict[str, Any]] = []
|
||||||
|
self._latest_index_date: date | None = None
|
||||||
|
self._backfill = False
|
||||||
|
|
||||||
|
# -- SourceImporter protocol -------------------------------------------
|
||||||
|
|
||||||
|
async def detect_revision(self, db) -> str | None:
|
||||||
|
async with self._client_factory() as client:
|
||||||
|
self._resolved = await sec_universe.resolve_ciks(db, client)
|
||||||
|
last_processed = await self._last_processed_index_date(db)
|
||||||
|
self._latest_index_date = await client.latest_index_date(self.today)
|
||||||
|
if self._latest_index_date is None:
|
||||||
|
raise SecError("no EDGAR daily index available")
|
||||||
|
if last_processed is None:
|
||||||
|
self._backfill = True
|
||||||
|
self._index_rows = []
|
||||||
|
else:
|
||||||
|
self._backfill = False
|
||||||
|
self._index_rows = await self._collect_index_rows(
|
||||||
|
client, last_processed, self._latest_index_date
|
||||||
|
)
|
||||||
|
content = sec_universe.index_content_hash(self._index_rows)
|
||||||
|
return sec_universe.compose_revision(
|
||||||
|
self._latest_index_date, content, self._resolved.symbol_to_cik
|
||||||
|
)
|
||||||
|
|
||||||
|
async def stage(self, db) -> StagedFundamentals:
|
||||||
|
assert self._resolved is not None, "detect_revision must run first"
|
||||||
|
resolved = self._resolved
|
||||||
|
staged = StagedFundamentals(resolved=resolved, backfill=self._backfill)
|
||||||
|
|
||||||
|
cik_to_tids = resolved.cik_to_ticker_ids
|
||||||
|
filed_by_cik: dict[int, list[str]] = defaultdict(list)
|
||||||
|
for r in self._index_rows:
|
||||||
|
if r["cik"] in cik_to_tids:
|
||||||
|
filed_by_cik[r["cik"]].append(r["accession"])
|
||||||
|
|
||||||
|
existing = await self._ciks_with_snapshots(db, set(cik_to_tids))
|
||||||
|
if self._backfill:
|
||||||
|
backfill_ciks = set(cik_to_tids)
|
||||||
|
else:
|
||||||
|
# Newly added issuers (resolved but no snapshots yet) get a full-history
|
||||||
|
# backfill; issuers that already have history are handled incrementally.
|
||||||
|
backfill_ciks = {c for c in cik_to_tids if c not in existing}
|
||||||
|
incremental_ciks = set(filed_by_cik) - backfill_ciks
|
||||||
|
|
||||||
|
async with self._client_factory() as client:
|
||||||
|
for cik in sorted(backfill_ciks | incremental_ciks):
|
||||||
|
is_backfill = cik in backfill_ciks
|
||||||
|
await self._stage_issuer(client, cik, is_backfill, filed_by_cik, staged)
|
||||||
|
|
||||||
|
# Read-only discrepancy detection: an accession we reconstructed that is
|
||||||
|
# already stored, differing in ANY source field (immutable → report in
|
||||||
|
# validation, event on promote, never mutate). Also gives promote the
|
||||||
|
# existing set so its insert count is dialect-independent.
|
||||||
|
if staged.rows:
|
||||||
|
existing = await self._existing_by_accession(db, [r.accession for r in staged.rows])
|
||||||
|
staged.existing_accessions = set(existing)
|
||||||
|
for row in staged.rows:
|
||||||
|
old = existing.get(row.accession)
|
||||||
|
if old is not None:
|
||||||
|
fields = _diff_fields(row, old)
|
||||||
|
if fields:
|
||||||
|
staged.discrepancies.append({"accession": row.accession, "fields": fields})
|
||||||
|
return staged
|
||||||
|
|
||||||
|
async def _stage_issuer(self, client, cik, is_backfill, filed_by_cik, staged) -> None:
|
||||||
|
cf = await client.companyfacts(cik)
|
||||||
|
bad = _companyfacts_structure_error(cf)
|
||||||
|
if bad is not None:
|
||||||
|
# Malformed payload (missing facts/units structure) — record separately
|
||||||
|
# and fail validation, rather than letting it degrade to skipped rows.
|
||||||
|
staged.invalid_payloads.append({"cik": cik10(cik), "reason": bad})
|
||||||
|
staged.issuers_fetched += 1
|
||||||
|
return
|
||||||
|
sub = await client.submissions(cik, include_history=is_backfill)
|
||||||
|
xbrl_meta, nonxbrl = _filing_meta(sub)
|
||||||
|
|
||||||
|
if is_backfill:
|
||||||
|
accns = set(xbrl_meta)
|
||||||
|
else:
|
||||||
|
present = parser.companyfacts_accessions(cf)
|
||||||
|
accns = set()
|
||||||
|
for accn in filed_by_cik.get(cik, []):
|
||||||
|
if accn in nonxbrl:
|
||||||
|
staged.skipped_non_xbrl.append({"cik": cik10(cik), "accession": accn})
|
||||||
|
elif accn in xbrl_meta and accn in present:
|
||||||
|
accns.add(accn)
|
||||||
|
else:
|
||||||
|
# XBRL (or unknown) filing not yet in Company Facts → the products
|
||||||
|
# have lagged; fail+retry rather than record nothing for it.
|
||||||
|
staged.missing_xbrl.append({"cik": cik10(cik), "accession": accn})
|
||||||
|
|
||||||
|
result = parser.parse_snapshots(cf, xbrl_meta, accns)
|
||||||
|
staged.rows.extend(result.rows)
|
||||||
|
staged.skipped_filings.extend(result.skipped_filings)
|
||||||
|
staged.field_issues.extend(result.field_issues)
|
||||||
|
staged.issuers_fetched += 1
|
||||||
|
if result.rows:
|
||||||
|
staged.issuers_with_rows += 1
|
||||||
|
|
||||||
|
# SIC proposal for this issuer's tickers (read-only; applied in promote).
|
||||||
|
sic = str(sub["sic"]) if sub.get("sic") else None
|
||||||
|
desc = sub.get("sic_description")
|
||||||
|
for tid in staged.resolved.cik_to_ticker_ids.get(cik, []):
|
||||||
|
staged.sic_updates.append((tid, sic, desc))
|
||||||
|
|
||||||
|
async def validate(self, db, staged: StagedFundamentals) -> ValidationResult:
|
||||||
|
messages: list[str] = []
|
||||||
|
|
||||||
|
# Consistency gate — before any write.
|
||||||
|
if staged.missing_xbrl:
|
||||||
|
messages.append(
|
||||||
|
f"{len(staged.missing_xbrl)} tracked XBRL filing(s) not yet in "
|
||||||
|
"Company Facts (index/facts lag) — retry"
|
||||||
|
)
|
||||||
|
# Malformed companyfacts payloads must fail, not degrade to skipped rows.
|
||||||
|
if staged.invalid_payloads:
|
||||||
|
messages.append(
|
||||||
|
f"{len(staged.invalid_payloads)} issuer(s) returned a malformed "
|
||||||
|
"companyfacts payload (missing facts structure)"
|
||||||
|
)
|
||||||
|
|
||||||
|
accns = [r.accession for r in staged.rows]
|
||||||
|
if len(accns) != len(set(accns)):
|
||||||
|
messages.append("duplicate accession in staged snapshots")
|
||||||
|
|
||||||
|
if staged.backfill:
|
||||||
|
n_issuers = len(staged.resolved.cik_to_ticker_ids)
|
||||||
|
coverage = staged.issuers_with_rows / n_issuers if n_issuers else 0.0
|
||||||
|
if coverage < MIN_BACKFILL_COVERAGE:
|
||||||
|
messages.append(
|
||||||
|
f"backfill coverage {coverage:.0%} < {MIN_BACKFILL_COVERAGE:.0%}"
|
||||||
|
)
|
||||||
|
|
||||||
|
summary = {
|
||||||
|
"backfill": staged.backfill,
|
||||||
|
"issuers_fetched": staged.issuers_fetched,
|
||||||
|
"issuers_with_rows": staged.issuers_with_rows,
|
||||||
|
"snapshot_rows": len(staged.rows),
|
||||||
|
"skipped_filings": len(staged.skipped_filings),
|
||||||
|
"field_issues": len(staged.field_issues),
|
||||||
|
"skipped_non_xbrl": len(staged.skipped_non_xbrl),
|
||||||
|
"missing_xbrl": len(staged.missing_xbrl),
|
||||||
|
"invalid_payloads": staged.invalid_payloads,
|
||||||
|
"cik_updates": len(staged.resolved.cik_updates),
|
||||||
|
# differing existing accessions (immutable — kept, reported here)
|
||||||
|
"discrepancies": staged.discrepancies[:50],
|
||||||
|
"discrepancy_count": len(staged.discrepancies),
|
||||||
|
}
|
||||||
|
return ValidationResult(
|
||||||
|
ok=not messages,
|
||||||
|
summary=summary,
|
||||||
|
source_max_date=self._latest_index_date,
|
||||||
|
messages=messages,
|
||||||
|
)
|
||||||
|
|
||||||
|
async def promote(self, db, staged: StagedFundamentals, run_id: int) -> dict[str, int]:
|
||||||
|
inserted = 0
|
||||||
|
for row in staged.rows:
|
||||||
|
if row.accession in staged.existing_accessions:
|
||||||
|
continue # immutable — keep the original row
|
||||||
|
stmt = insert_for_session(db, FundamentalSnapshot).values(**_row_values(row, run_id))
|
||||||
|
stmt = stmt.on_conflict_do_nothing(index_elements=["accession"]) # race belt-and-suspenders
|
||||||
|
await db.execute(stmt)
|
||||||
|
inserted += 1
|
||||||
|
|
||||||
|
# Warn (in-transaction, so it commits atomically with the promotion) when
|
||||||
|
# any existing accession reconstructed differently — kept immutable.
|
||||||
|
if staged.discrepancies:
|
||||||
|
accns = ", ".join(d["accession"] for d in staged.discrepancies[:10])
|
||||||
|
db.add(SystemEvent(
|
||||||
|
severity="warning",
|
||||||
|
source="sec_facts",
|
||||||
|
code="snapshot_discrepancy",
|
||||||
|
message=(
|
||||||
|
f"{len(staged.discrepancies)} stored accession(s) reconstructed "
|
||||||
|
f"differently; kept immutable: {accns}"
|
||||||
|
)[:4000],
|
||||||
|
dedup_key=f"sec_facts:discrepancy:{run_id}",
|
||||||
|
created_at=_now(),
|
||||||
|
))
|
||||||
|
|
||||||
|
ticker_counts = await sec_universe.apply_ticker_updates(
|
||||||
|
db, staged.resolved, staged.sic_updates
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"inserted": inserted,
|
||||||
|
"existing_unchanged": len(staged.existing_accessions),
|
||||||
|
"discrepancies": len(staged.discrepancies),
|
||||||
|
**ticker_counts,
|
||||||
|
}
|
||||||
|
|
||||||
|
# -- helpers -----------------------------------------------------------
|
||||||
|
|
||||||
|
async def _last_processed_index_date(self, db) -> date | None:
|
||||||
|
return (
|
||||||
|
await db.execute(
|
||||||
|
select(DataImportRun.source_max_date)
|
||||||
|
.where(DataImportRun.source == SOURCE, DataImportRun.status == STATUS_PROMOTED)
|
||||||
|
.order_by(DataImportRun.id.desc())
|
||||||
|
.limit(1)
|
||||||
|
)
|
||||||
|
).scalar_one_or_none()
|
||||||
|
|
||||||
|
async def _collect_index_rows(
|
||||||
|
self, client: SecClient, last_processed: date, latest: date
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
# Walk EVERY unprocessed date. No cap — dropping the older part of a long
|
||||||
|
# outage while still advancing source_max_date would permanently lose
|
||||||
|
# those filings. A large gap is one-time cost, not silent data loss.
|
||||||
|
tracked = set(self._resolved.cik_to_ticker_ids) if self._resolved else set()
|
||||||
|
gap = (latest - last_processed).days
|
||||||
|
if gap > 60:
|
||||||
|
logger.warning("sec_facts: %d-day index gap since %s; walking all", gap, last_processed)
|
||||||
|
rows: list[dict[str, Any]] = []
|
||||||
|
day = last_processed + timedelta(days=1)
|
||||||
|
while day <= latest:
|
||||||
|
for r in await client.daily_index(day):
|
||||||
|
if r["form"] in _XBRL_FORMS and r["cik"] in tracked:
|
||||||
|
rows.append(r)
|
||||||
|
day += timedelta(days=1)
|
||||||
|
return rows
|
||||||
|
|
||||||
|
async def _ciks_with_snapshots(self, db, ciks: set[int]) -> set[int]:
|
||||||
|
if not ciks:
|
||||||
|
return set()
|
||||||
|
cik_strs = [cik10(c) for c in ciks]
|
||||||
|
found = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalSnapshot.cik)
|
||||||
|
.where(FundamentalSnapshot.cik.in_(cik_strs))
|
||||||
|
.distinct()
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
return {int(c) for c in found}
|
||||||
|
|
||||||
|
async def _existing_by_accession(self, db, accessions: list[str]) -> dict[str, FundamentalSnapshot]:
|
||||||
|
if not accessions:
|
||||||
|
return {}
|
||||||
|
rows = (
|
||||||
|
await db.execute(
|
||||||
|
select(FundamentalSnapshot).where(FundamentalSnapshot.accession.in_(accessions))
|
||||||
|
)
|
||||||
|
).scalars().all()
|
||||||
|
return {r.accession: r for r in rows}
|
||||||
|
|
||||||
|
|
||||||
|
def _companyfacts_structure_error(cf: Any) -> str | None:
|
||||||
|
"""None if the payload is structurally sound, else a reason string. Checks the
|
||||||
|
top-level ``facts`` mapping AND that every concept carries a ``units`` mapping —
|
||||||
|
a missing/non-dict units would silently drop that concept's facts otherwise."""
|
||||||
|
if not isinstance(cf, dict) or not isinstance(cf.get("facts"), dict):
|
||||||
|
return "missing facts structure"
|
||||||
|
for concepts in cf["facts"].values():
|
||||||
|
if not isinstance(concepts, dict):
|
||||||
|
return "malformed taxonomy structure"
|
||||||
|
for body in concepts.values():
|
||||||
|
if not isinstance(body, dict) or not isinstance(body.get("units"), dict):
|
||||||
|
return "missing units structure"
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _filing_meta(sub: dict[str, Any]) -> tuple[dict[str, FilingMeta], set[str]]:
|
||||||
|
"""(xbrl_meta, nonxbrl_accessions) from a submissions payload. xbrl_meta only
|
||||||
|
includes 10-K/10-Q(/A) filings that are XBRL and have full period metadata."""
|
||||||
|
xbrl: dict[str, FilingMeta] = {}
|
||||||
|
nonxbrl: set[str] = set()
|
||||||
|
for f in sub.get("filings", []):
|
||||||
|
if f["form"] not in _XBRL_FORMS:
|
||||||
|
continue
|
||||||
|
if not f.get("is_xbrl"):
|
||||||
|
nonxbrl.add(f["accession"])
|
||||||
|
continue
|
||||||
|
if not (f.get("report_date") and f.get("filing_date") and f.get("acceptance_datetime")):
|
||||||
|
continue
|
||||||
|
xbrl[f["accession"]] = FilingMeta(
|
||||||
|
report_date=date.fromisoformat(f["report_date"]),
|
||||||
|
filing_date=date.fromisoformat(f["filing_date"]),
|
||||||
|
accepted_at=_parse_dt(f["acceptance_datetime"]),
|
||||||
|
form=f["form"],
|
||||||
|
)
|
||||||
|
return xbrl, nonxbrl
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_dt(value: str) -> datetime:
|
||||||
|
return datetime.fromisoformat(value.replace("Z", "+00:00"))
|
||||||
|
|
||||||
|
|
||||||
|
def _row_values(row: SnapshotRow, run_id: int) -> dict[str, Any]:
|
||||||
|
values = {col: getattr(row, col) for col in _SNAPSHOT_COLS}
|
||||||
|
values["import_run_id"] = run_id
|
||||||
|
values["created_at"] = _now()
|
||||||
|
return values
|
||||||
|
|
||||||
|
|
||||||
|
def _diff_fields(row: SnapshotRow, old: FundamentalSnapshot) -> list[str]:
|
||||||
|
"""Source fields where a re-parsed row differs from the stored (immutable) row."""
|
||||||
|
return [col for col in _COMPARE_COLS if getattr(row, col) != getattr(old, col)]
|
||||||
@@ -0,0 +1,119 @@
|
|||||||
|
"""Tracked-universe CIK/SIC resolution and the SEC importer's composite revision.
|
||||||
|
|
||||||
|
Resolves the app's tracked tickers to SEC issuers (CIK) and prepares
|
||||||
|
``tickers.cik/sic/sic_description`` back-fills. Also builds the **universe
|
||||||
|
fingerprint** in the importer's composite revision, so adding a ticker changes
|
||||||
|
the revision and forces a run instead of being ``no_op``'d away or starved
|
||||||
|
waiting for its issuer to file (A3 design, Decision 1 review fix).
|
||||||
|
|
||||||
|
**Transaction contract:** resolution is read-only — `resolve_ciks` and
|
||||||
|
`fetch_sic_updates` compute *proposed* updates and mutate nothing. They run in
|
||||||
|
the importer's `stage` (which must not write, or a failed validation would leak
|
||||||
|
changes on the framework's failure commit). The proposals are applied only in
|
||||||
|
`promote`, via `apply_ticker_updates`, atomically with the snapshot inserts.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import logging
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from typing import Iterable
|
||||||
|
|
||||||
|
from sqlalchemy import select, update
|
||||||
|
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services.earnings_alignment import normalise_symbol
|
||||||
|
from app.services.sec_client import SecClient
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ResolvedUniverse:
|
||||||
|
"""Read-only result of CIK resolution. `cik_updates` are proposed writes
|
||||||
|
(ticker_id → new cik string) applied later in promote."""
|
||||||
|
|
||||||
|
symbol_to_cik: dict[str, int] = field(default_factory=dict)
|
||||||
|
cik_to_ticker_ids: dict[int, list[int]] = field(default_factory=dict)
|
||||||
|
cik_updates: list[tuple[int, str]] = field(default_factory=list)
|
||||||
|
|
||||||
|
|
||||||
|
async def resolve_ciks(db, client: SecClient) -> ResolvedUniverse:
|
||||||
|
"""Resolve tracked tickers to CIKs via company_tickers.json. **Read-only** —
|
||||||
|
returns the mapping + proposed `tickers.cik` writes; mutates nothing."""
|
||||||
|
ticker_to_cik = await client.company_tickers()
|
||||||
|
rows = (await db.execute(select(Ticker.id, Ticker.symbol, Ticker.cik))).all()
|
||||||
|
|
||||||
|
result = ResolvedUniverse()
|
||||||
|
for tid, symbol, current_cik in rows:
|
||||||
|
if not symbol:
|
||||||
|
continue
|
||||||
|
sym = normalise_symbol(symbol)
|
||||||
|
cik = ticker_to_cik.get(sym)
|
||||||
|
if cik is None:
|
||||||
|
continue # ADRs / non-SEC issuers — snapshots simply absent
|
||||||
|
result.symbol_to_cik[sym] = cik
|
||||||
|
result.cik_to_ticker_ids.setdefault(cik, []).append(tid)
|
||||||
|
if current_cik != f"{cik:010d}":
|
||||||
|
result.cik_updates.append((tid, f"{cik:010d}"))
|
||||||
|
logger.info(
|
||||||
|
"resolve_ciks: %d resolved, %d proposed cik updates",
|
||||||
|
len(result.symbol_to_cik),
|
||||||
|
len(result.cik_updates),
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
async def fetch_sic_updates(
|
||||||
|
client: SecClient, cik_to_ticker_ids: dict[int, Iterable[int]]
|
||||||
|
) -> list[tuple[int, str | None, str | None]]:
|
||||||
|
"""Fetch SIC for each CIK (recent-only submissions, no history shards) and
|
||||||
|
return proposed `(ticker_id, sic, sic_description)` writes. **Read-only** — no
|
||||||
|
DB mutation. Callers pass only the CIKs that need it (e.g. missing a SIC)."""
|
||||||
|
updates: list[tuple[int, str | None, str | None]] = []
|
||||||
|
for cik, ticker_ids in cik_to_ticker_ids.items():
|
||||||
|
sub = await client.submissions(cik, include_history=False)
|
||||||
|
sic = str(sub["sic"]) if sub.get("sic") else None
|
||||||
|
desc = sub.get("sic_description")
|
||||||
|
for tid in ticker_ids:
|
||||||
|
updates.append((tid, sic, desc))
|
||||||
|
return updates
|
||||||
|
|
||||||
|
|
||||||
|
async def apply_ticker_updates(
|
||||||
|
db,
|
||||||
|
resolved: ResolvedUniverse,
|
||||||
|
sic_updates: list[tuple[int, str | None, str | None]] | None = None,
|
||||||
|
) -> dict[str, int]:
|
||||||
|
"""Apply the proposed cik / sic writes. **The only writer** — call inside
|
||||||
|
promote so it commits atomically with the snapshot inserts."""
|
||||||
|
for tid, cik in resolved.cik_updates:
|
||||||
|
await db.execute(update(Ticker).where(Ticker.id == tid).values(cik=cik))
|
||||||
|
for tid, sic, desc in sic_updates or []:
|
||||||
|
await db.execute(
|
||||||
|
update(Ticker).where(Ticker.id == tid).values(sic=sic, sic_description=desc)
|
||||||
|
)
|
||||||
|
return {"cik_updates": len(resolved.cik_updates), "sic_updates": len(sic_updates or [])}
|
||||||
|
|
||||||
|
|
||||||
|
def universe_fingerprint(symbol_to_cik: dict[str, int]) -> str:
|
||||||
|
"""Stable short hash of the tracked symbol->CIK set. Changes whenever a ticker
|
||||||
|
is added/removed or its CIK mapping changes."""
|
||||||
|
canonical = ";".join(f"{sym}:{cik}" for sym, cik in sorted(symbol_to_cik.items()))
|
||||||
|
return hashlib.blake2b(canonical.encode("utf-8"), digest_size=12).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def index_content_hash(index_rows: Iterable[dict]) -> str:
|
||||||
|
"""Order-independent hash of the tracked index accessions consumed this run."""
|
||||||
|
keys = sorted(f"{r['cik']}/{r['accession']}" for r in index_rows)
|
||||||
|
return hashlib.blake2b("|".join(keys).encode("utf-8"), digest_size=12).hexdigest()
|
||||||
|
|
||||||
|
|
||||||
|
def compose_revision(index_date, content_hash: str, symbol_to_cik: dict[str, int]) -> str:
|
||||||
|
"""Composite revision = processed index date + index-content hash + universe
|
||||||
|
fingerprint. Equal across runs ⇒ nothing new ⇒ no_op. Rejects a missing index
|
||||||
|
date rather than emitting a `None:...` revision that could false-match."""
|
||||||
|
if index_date is None:
|
||||||
|
raise ValueError("compose_revision requires a non-null index date")
|
||||||
|
return f"{index_date}:{content_hash}:{universe_fingerprint(symbol_to_cik)}"
|
||||||
Executable
+100
@@ -0,0 +1,100 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# One-time production provisioning for the shadow fundamentals sources.
|
||||||
|
# Run as root to install; run with --check as the deploy user for a read-only
|
||||||
|
# preflight. Version upgrades are intentional code changes, never "latest".
|
||||||
|
DOLT_VERSION="2.2.0"
|
||||||
|
DOLT_BINARY="${DOLT_BINARY:-/usr/local/bin/dolt}"
|
||||||
|
DOLT_DATA_DIR="${DOLT_DATA_DIR:-/var/lib/signal-platform/dolt}"
|
||||||
|
DOLT_EARNINGS_SUBDIR="${DOLT_EARNINGS_SUBDIR:-earnings}"
|
||||||
|
APP_USER="${APP_USER:-deploy}"
|
||||||
|
APP_GROUP="${APP_GROUP:-deploy}"
|
||||||
|
ENV_FILE="${ENV_FILE:-/opt/signalplatform/.env}"
|
||||||
|
MIN_FREE_GB="${DOLT_MIN_FREE_DISK_GB:-5}"
|
||||||
|
EARNINGS_DIR="${DOLT_DATA_DIR}/${DOLT_EARNINGS_SUBDIR}"
|
||||||
|
|
||||||
|
fail() {
|
||||||
|
echo "ERROR: $*" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
|
||||||
|
version_ok() {
|
||||||
|
local output
|
||||||
|
output="$("$DOLT_BINARY" version 2>/dev/null || true)"
|
||||||
|
grep -Eq "(^|[[:space:]])v?${DOLT_VERSION}([[:space:]]|$)" <<<"$output"
|
||||||
|
}
|
||||||
|
|
||||||
|
check_free_space() {
|
||||||
|
local available_kb
|
||||||
|
available_kb="$(df -Pk "$DOLT_DATA_DIR" | awk 'NR == 2 {print $4}')"
|
||||||
|
[[ "$available_kb" =~ ^[0-9]+$ ]] || fail "could not read free space for $DOLT_DATA_DIR"
|
||||||
|
if ! awk -v available="$available_kb" -v minimum_gb="$MIN_FREE_GB" \
|
||||||
|
'BEGIN { exit !(available >= minimum_gb * 1024 * 1024) }'; then
|
||||||
|
fail "$DOLT_DATA_DIR has less than ${MIN_FREE_GB} GB free"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
check_env() {
|
||||||
|
[[ -f "$ENV_FILE" ]] || fail "missing environment file: $ENV_FILE"
|
||||||
|
grep -Fqx "DOLT_BINARY=$DOLT_BINARY" "$ENV_FILE" \
|
||||||
|
|| fail "set DOLT_BINARY=$DOLT_BINARY in $ENV_FILE"
|
||||||
|
grep -Fqx "DOLT_DATA_DIR=$DOLT_DATA_DIR" "$ENV_FILE" \
|
||||||
|
|| fail "set DOLT_DATA_DIR=$DOLT_DATA_DIR in $ENV_FILE"
|
||||||
|
grep -Fqx "DOLT_EARNINGS_SUBDIR=$DOLT_EARNINGS_SUBDIR" "$ENV_FILE" \
|
||||||
|
|| fail "set DOLT_EARNINGS_SUBDIR=$DOLT_EARNINGS_SUBDIR in $ENV_FILE"
|
||||||
|
grep -Eq '^SEC_USER_AGENT=.*@.*' "$ENV_FILE" \
|
||||||
|
|| fail "SEC_USER_AGENT in $ENV_FILE must contain a real contact email"
|
||||||
|
}
|
||||||
|
|
||||||
|
check_all() {
|
||||||
|
id "$APP_USER" >/dev/null 2>&1 || fail "missing service user: $APP_USER"
|
||||||
|
[[ -x "$DOLT_BINARY" ]] || fail "missing Dolt binary: $DOLT_BINARY"
|
||||||
|
version_ok || fail "expected Dolt $DOLT_VERSION at $DOLT_BINARY"
|
||||||
|
[[ -d "$EARNINGS_DIR/.dolt" ]] \
|
||||||
|
|| fail "missing earnings clone: $EARNINGS_DIR"
|
||||||
|
if [[ "$(id -un)" == "$APP_USER" ]]; then
|
||||||
|
[[ -r "$EARNINGS_DIR/.dolt" ]] \
|
||||||
|
|| fail "earnings clone is not readable by $APP_USER"
|
||||||
|
elif command -v runuser >/dev/null 2>&1; then
|
||||||
|
runuser -u "$APP_USER" -- test -r "$EARNINGS_DIR/.dolt" \
|
||||||
|
|| fail "earnings clone is not readable by $APP_USER"
|
||||||
|
else
|
||||||
|
fail "run --check as $APP_USER (or install runuser)"
|
||||||
|
fi
|
||||||
|
check_free_space
|
||||||
|
check_env
|
||||||
|
echo "OK: Dolt $DOLT_VERSION and earnings clone are provisioned"
|
||||||
|
}
|
||||||
|
|
||||||
|
if [[ "${1:-}" == "--check" ]]; then
|
||||||
|
check_all
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
[[ "$EUID" -eq 0 ]] || fail "run provisioning as root (or use --check)"
|
||||||
|
command -v curl >/dev/null 2>&1 || fail "curl is required"
|
||||||
|
command -v runuser >/dev/null 2>&1 || fail "runuser is required"
|
||||||
|
id "$APP_USER" >/dev/null 2>&1 || fail "missing service user: $APP_USER"
|
||||||
|
|
||||||
|
if ! version_ok; then
|
||||||
|
installer="$(mktemp)"
|
||||||
|
trap 'rm -f "$installer"' EXIT
|
||||||
|
curl -fsSL \
|
||||||
|
"https://github.com/dolthub/dolt/releases/download/v${DOLT_VERSION}/install.sh" \
|
||||||
|
-o "$installer"
|
||||||
|
bash "$installer"
|
||||||
|
fi
|
||||||
|
version_ok || fail "Dolt $DOLT_VERSION installation failed"
|
||||||
|
|
||||||
|
install -d -o "$APP_USER" -g "$APP_GROUP" -m 0750 "$DOLT_DATA_DIR"
|
||||||
|
check_free_space
|
||||||
|
|
||||||
|
if [[ ! -d "$EARNINGS_DIR/.dolt" ]]; then
|
||||||
|
[[ ! -e "$EARNINGS_DIR" ]] \
|
||||||
|
|| fail "$EARNINGS_DIR exists but is not a Dolt clone"
|
||||||
|
runuser -u "$APP_USER" -- \
|
||||||
|
"$DOLT_BINARY" clone post-no-preference/earnings "$EARNINGS_DIR"
|
||||||
|
fi
|
||||||
|
|
||||||
|
check_all
|
||||||
@@ -0,0 +1,491 @@
|
|||||||
|
# Dolt bulk-data integration — implementation plan
|
||||||
|
|
||||||
|
Status: approved 2026-07-21, revised through four review rounds; direction: KISS
|
||||||
|
backend, UI value first. Hand-off document for the implementing agent;
|
||||||
|
self-contained.
|
||||||
|
|
||||||
|
## Objective
|
||||||
|
|
||||||
|
Replace the free-tier fundamentals APIs (FMP, Finnhub, Alpha Vantage) with bulk
|
||||||
|
data: SEC Company Facts for fundamentals, the DoltHub earnings repo for the
|
||||||
|
earnings calendar/history, and — later, independently — the DoltHub stocks repo for
|
||||||
|
historical OHLCV. PostgreSQL stays the production system of record.
|
||||||
|
|
||||||
|
**Delivery order: two independent workstreams.**
|
||||||
|
|
||||||
|
- **Workstream A (build first):** SEC fundamentals + Dolt earnings + API v1 +
|
||||||
|
FundamentalsPanel + decommission FMP/Finnhub/Alpha Vantage. Valuation uses the
|
||||||
|
existing Alpaca closes already in `ohlcv_records`. This alone achieves the goal
|
||||||
|
(killing the quota-limited APIs) and delivers all the UI value.
|
||||||
|
- **Workstream B (later, optional until needed):** replace historical OHLCV with
|
||||||
|
the Dolt stocks repo. The most complex machinery (4.7 GB clone, split
|
||||||
|
adjustment, source-bar table, reconciliation) lives here and blocks nothing in A.
|
||||||
|
|
||||||
|
**Guiding principle: KISS.** Plain daily importers with staging and atomic
|
||||||
|
promotion — no forensic replay, no permanent archive store, no conflict tables, no
|
||||||
|
aggregate tables. Engineering budget goes into the UI (quarter trends, peer
|
||||||
|
comparison). Deferred until a concrete need: exact source replay, point-in-time
|
||||||
|
backtest enforcement, fundamental metrics in scoring.
|
||||||
|
|
||||||
|
**Non-negotiables**
|
||||||
|
|
||||||
|
- The application never queries Dolt/DoltHub or SEC at request time. All access is
|
||||||
|
batch import → PostgreSQL. If a sync fails or the source is unchanged, production
|
||||||
|
continues on the last successfully imported data.
|
||||||
|
- Do not replace PostgreSQL with Dolt/Doltgres. Never commit to the upstream clones.
|
||||||
|
- No owned SEC Dolt repo: SEC JSON is normalized straight into PostgreSQL.
|
||||||
|
- Scoring **code** is unchanged, but swapping the data source changes production
|
||||||
|
behavior: `app/services/scoring_service.py` (~line 450) scores pe_ratio /
|
||||||
|
revenue_growth / earnings_surprise from `fundamental_data`, so new definitions
|
||||||
|
change rankings even with identical code. Cutover of `fundamental_data`
|
||||||
|
population requires the **score-parity gate** (phase A5) — never silently. All
|
||||||
|
*new* metrics are display-only.
|
||||||
|
- Intraday (10:00–15:00), near-close (15:30) and after-close (16:45) pipelines stay
|
||||||
|
on Alpaca unchanged.
|
||||||
|
|
||||||
|
## Data sources
|
||||||
|
|
||||||
|
1. **SEC Company Facts + submissions bulk files** (free, no key; costs bandwidth,
|
||||||
|
CPU and disk — optimize accordingly) — XBRL facts per **issuer (CIK), not per
|
||||||
|
ticker**. Tickers resolve to a CIK via SEC `company_tickers.json` (multi-class
|
||||||
|
issuers like GOOGL/GOOG share one CIK and one set of fundamentals). Submissions
|
||||||
|
also supply the SIC code (peer grouping) and `acceptanceDateTime`. Handle unit
|
||||||
|
variants and fiscal-period alignment (derive Q4 = FY − Q1..Q3 where needed).
|
||||||
|
**Amendments:** retain every accession immutably; readers select the newest
|
||||||
|
valid `accepted_at` snapshot per reporting period at read time. No flags, no
|
||||||
|
mutation.
|
||||||
|
2. **`post-no-preference/earnings`** (DoltHub) — announcement date, BMO/AMC session
|
||||||
|
(partial), period end, EPS estimate/actual, surprise history. Small clone.
|
||||||
|
`scripts/import_dolthub_earnings.py` is a research/SQLite importer — reuse its
|
||||||
|
normalization/alignment logic (calendar↔EPS-history monotonic alignment, SUE
|
||||||
|
scaling) but write a production PostgreSQL importer; do not extend the script.
|
||||||
|
3. **`post-no-preference/stocks`** (DoltHub, **workstream B**) — daily raw OHLCV
|
||||||
|
(unadjusted), symbol metadata, splits, dividends. Publishes ~01:30 ET the
|
||||||
|
following calendar day. Clone is ~4.7 GB.
|
||||||
|
|
||||||
|
**Licensing (phase A0) — DECIDED 2026-07-22:** `post-no-preference/earnings` is
|
||||||
|
**approved for private/internal ingestion under CC BY-SA 4.0**. Conditions the A2
|
||||||
|
importer must honor: preserve the upstream license, attribution, and transformation
|
||||||
|
notes (retain a CC BY-SA 4.0 reference + attribution to `post-no-preference/earnings`
|
||||||
|
and a note of the transformations applied — e.g. in a repo `NOTICE`/attribution file
|
||||||
|
and the importer module); **no public API, bulk export, or redistribution** of the
|
||||||
|
data; re-review licensing before any public or commercial access. The
|
||||||
|
`post-no-preference/stocks` repo (workstream B) is **not** covered here and will be
|
||||||
|
reviewed separately if B begins.
|
||||||
|
|
||||||
|
## Schema
|
||||||
|
|
||||||
|
**Migration 026 (workstream A)** — current head: `025_trade_setup_scan_run_id`:
|
||||||
|
|
||||||
|
- `data_import_runs` — lean: id, source (`sec_facts` | `dolt_earnings` |
|
||||||
|
`dolt_stocks`), revision (Dolt commit hash, or SEC archive SHA-256), status
|
||||||
|
(`running`/`validated`/`promoted`/`no_op`/`failed`), source_max_date, row_counts
|
||||||
|
JSON, validation JSON (includes reconciliation/discrepancy summaries — no
|
||||||
|
separate conflicts table; details go to structured logs), started_at,
|
||||||
|
completed_at, error_details. One run per source at a time (Postgres advisory
|
||||||
|
lock keyed by source).
|
||||||
|
- `earnings_events` — ticker_id, announce_date, session (`bmo`/`amc`/`unknown`),
|
||||||
|
period_end, eps_estimate, eps_actual, source, import_run_id. Unique
|
||||||
|
(ticker_id, announce_date). **Rescheduling:** within each promotion transaction,
|
||||||
|
delete this source's future-dated rows (announce_date > today) and re-insert
|
||||||
|
from the new snapshot, so moved or cancelled dates never linger. Past rows
|
||||||
|
(results) are never deleted.
|
||||||
|
- `tickers` — add nullable `cik`, `sic`, `sic_description` (from SEC submissions /
|
||||||
|
`company_tickers.json`; refreshed by the SEC import; multi-class tickers share
|
||||||
|
values). The only ticker↔issuer join point.
|
||||||
|
- `fundamental_snapshots` — **CIK-keyed, one immutable row per accession**: cik,
|
||||||
|
accession (unique), form, filed_date, **accepted_at** (kept although PIT
|
||||||
|
enforcement is deferred — one timestamp now vs painful retrofit later),
|
||||||
|
**period_start, period_end, fiscal_year, fiscal_period** (the filing's own
|
||||||
|
`dei`/`us-gaap` period identity — required to align non-calendar fiscal years and
|
||||||
|
to derive discrete quarters from cumulative facts), and the **price-independent
|
||||||
|
raw facts** so metrics are recomputable. **Store facts as the filing reports
|
||||||
|
them, not as derived quarters:** duration facts (revenue, net income, diluted EPS,
|
||||||
|
CFO, capex, EBITDA inputs) retain the filing's normalized **cumulative YTD/FY**
|
||||||
|
value for the (period_start → period_end) span; balance-sheet facts (cash+ST
|
||||||
|
investments, total debt, shares outstanding) are **period-end** values.
|
||||||
|
``shares_outstanding`` is a point-in-time count
|
||||||
|
(``dei:EntityCommonStockSharesOutstanding``), not the weighted-average diluted
|
||||||
|
share count — both consumers (est. market cap, YoY dilution) want a
|
||||||
|
point-in-time value. **Nothing
|
||||||
|
derived is frozen into a row:** discrete quarters (10-Q YTD deltas, Q4 = FY −
|
||||||
|
Q1..Q3), TTM, YoY and the four-period metric histories are all computed **at read time** by
|
||||||
|
picking the newest valid accepted_at snapshot for *each* required period — so
|
||||||
|
non-calendar fiscal years resolve correctly and a later amendment to a prior
|
||||||
|
quarter is reflected automatically without ever storing a stale derived quarter.
|
||||||
|
Readers pick the newest valid accepted_at per period; history powers the UI
|
||||||
|
reference comparisons and deterministic reads.
|
||||||
|
- Keep `fundamental_data` (`app/models/fundamental.py`) as the latest-value compat
|
||||||
|
cache, repopulated by the daily SEC job — but only after the phase-A5 parity
|
||||||
|
gate.
|
||||||
|
|
||||||
|
**Migration 027 (workstream B, written when B starts):**
|
||||||
|
|
||||||
|
- `ohlcv_source_bars` — source-truth bar table, required because `ohlcv_records`
|
||||||
|
allows one row per (ticker_id, date) (`app/models/ohlcv.py:12`) and Alpaca
|
||||||
|
ingestion upserts it in place (`app/services/price_service.py:82`) — Dolt and
|
||||||
|
Alpaca bars cannot coexist there. Holds **Dolt raw (unadjusted) bars only** —
|
||||||
|
Alpaca bars are already split-adjusted at the provider (`app/providers/alpaca.py:77`
|
||||||
|
requests `Adjustment.SPLIT`) and live exclusively in `ohlcv_records`. Columns:
|
||||||
|
source (`dolt`), adjustment (`raw` — explicit), ticker_id, date, OHLCV,
|
||||||
|
import_run_id; unique (source, ticker_id, date). Changed bars are counted in the
|
||||||
|
run's validation JSON and logged before overwrite.
|
||||||
|
- `corporate_actions` — ticker_id, type (`split`/`dividend`), ex_date,
|
||||||
|
ratio/amount, source, import_run_id. Unique (ticker_id, type, ex_date).
|
||||||
|
- `ohlcv_records` — add nullable `import_run_id` FK and `source` text (default
|
||||||
|
`'alpaca'`).
|
||||||
|
|
||||||
|
## Import framework
|
||||||
|
|
||||||
|
Every importer: idempotent per revision (same Dolt commit / archive checksum →
|
||||||
|
`no_op`, zero row changes); stage into a representation outside the live tables
|
||||||
|
first (in-memory for the small workstream-A sources; a file/table handle is fine
|
||||||
|
if workstream B ever needs it); promotion in one transaction;
|
||||||
|
safe to retry; a failed or unchanged run leaves the current dataset untouched.
|
||||||
|
Record every attempt in `data_import_runs`.
|
||||||
|
|
||||||
|
**SEC access requirements (operational safeguards, per SEC fair-access policy):**
|
||||||
|
send an identifying `User-Agent` with a contact email on every request; stay far
|
||||||
|
below the 10 req/s limit (the bulk endpoints need only a handful of requests per
|
||||||
|
run); exponential backoff on 429; a 403 means the User-Agent or request pattern is
|
||||||
|
wrong — alert and stop, never retry-loop. See SEC developer resources
|
||||||
|
(https://www.sec.gov/about/developer-resources).
|
||||||
|
|
||||||
|
**Reproducibility scope (deliberately limited):** the normalized snapshots in
|
||||||
|
PostgreSQL *are* the durable record. Keep only the last ~2 SEC archives on disk for
|
||||||
|
debugging. Byte-level replay of old runs is out of scope until a concrete need.
|
||||||
|
|
||||||
|
Dolt access: `dolt pull` on the persistent clone, record the resulting commit hash,
|
||||||
|
read via `dolt sql -r csv` (no long-running sql-server). **The scheduler shares one
|
||||||
|
event loop with the API** (`app/scheduler.py:73`) — run dolt/unzip/download
|
||||||
|
subprocesses via `asyncio.create_subprocess_exec` (or an executor), never blocking
|
||||||
|
calls. Check free disk space before pulling; alert and skip if below threshold.
|
||||||
|
|
||||||
|
**Deployment constraints:** deploy is `rsync --delete` of the repo tree
|
||||||
|
(`.gitea/workflows/deploy.yml:127`), so clones and archives must live **outside the
|
||||||
|
deployment path** — an env-configured persistent directory (e.g.
|
||||||
|
`DOLT_DATA_DIR=/var/lib/signal-platform/dolt`). The dolt binary is a new prod
|
||||||
|
runtime dependency: install it once with the version-pinned provisioner in
|
||||||
|
`deploy/provision_fundamentals.sh`; operational steps are in
|
||||||
|
`docs/fundamentals-deployment.md`. The clone is reproducible from DoltHub; the
|
||||||
|
normalized PostgreSQL rows remain part of the normal database backup.
|
||||||
|
|
||||||
|
**Validation gates (block promotion, raise an alert via the existing system-events
|
||||||
|
path):** source freshness as expected; tracked-universe coverage; no duplicate
|
||||||
|
business keys; fundamental units/periods consistent; row-count deltas within
|
||||||
|
reason; upstream schema change stops promotion. Workstream B adds: OHLC sanity
|
||||||
|
(high ≥ open/close/low, low ≤ open/close/high, volume ≥ 0); no unexplained split
|
||||||
|
discontinuities.
|
||||||
|
|
||||||
|
**Split adjustment (workstream B):** `ohlcv_source_bars` + `corporate_actions` are
|
||||||
|
the source of truth; canonical `ohlcv_records` is *generated* from them to match
|
||||||
|
Alpaca `Adjustment.SPLIT`, selecting only adjustment = `raw` rows as input so
|
||||||
|
adjustment is applied exactly once. A newly published split rewrites the symbol's
|
||||||
|
entire adjusted history — treat whole-symbol rewrites as a normal import event
|
||||||
|
(exempt that symbol from the row-count-delta gate for that run) and stamp rows with
|
||||||
|
the import_run_id (a backtest↔prod parity guard exists; changed history changes
|
||||||
|
backtests).
|
||||||
|
|
||||||
|
## Scheduling (`app/scheduler.py` — `SCHEDULE_DEFAULTS` / `_CRON_JOBS`, ~line 1451)
|
||||||
|
|
||||||
|
Follow the existing pattern: cron strings in SystemSettings via
|
||||||
|
`app/services/settings_store.py`, day-of-week as names never numbers, logging via
|
||||||
|
`_log_event`.
|
||||||
|
|
||||||
|
Workstream A:
|
||||||
|
|
||||||
|
- Dolt earnings import: daily ~02:30 ET (with the future-row replacement above).
|
||||||
|
- **SEC fundamentals job: daily ~04:00 ET.** One job, three steps:
|
||||||
|
(a) detect a composite revision from the latest EDGAR daily-index date, the
|
||||||
|
exact tracked index rows, and the tracked-universe fingerprint — an unchanged
|
||||||
|
revision is a `no_op` before Company Facts are fetched;
|
||||||
|
(b) when changed, fetch and parse Company Facts only for tracked-universe CIKs
|
||||||
|
that filed, plus full available history for the first run or a newly added
|
||||||
|
issuer, through validation→atomic promotion. Universe resolution and the exact
|
||||||
|
index inputs are cached during revision detection and reused during staging;
|
||||||
|
CIKs are resolved from `company_tickers.json` without writes until promotion;
|
||||||
|
(c) **always, locally, and only after production activation** (the phase-A5
|
||||||
|
parity approval): refresh the legacy `fundamental_data` fields and mark affected
|
||||||
|
cached fundamental scores stale. **Sources differ per field** — do not assume all
|
||||||
|
five come from SEC: `pe_ratio` and `market_cap` from the newest valid snapshots ×
|
||||||
|
latest PostgreSQL close, each with its own formula — `pe_ratio` = latest close /
|
||||||
|
TTM diluted EPS; `market_cap` = issuer-wide shares outstanding × latest close;
|
||||||
|
`revenue_growth` from the snapshots alone; `earnings_surprise` and
|
||||||
|
`next_earnings_date` from `earnings_events` (the Dolt earnings feed — these two do
|
||||||
|
not exist in SEC facts). Before activation the job imports snapshots only
|
||||||
|
(shadow). Step (c) must run identically when SEC is unreachable — prices move
|
||||||
|
daily even when filings don't, and the earnings-derived fields already live in
|
||||||
|
PostgreSQL.
|
||||||
|
**The new API valuation object is not stored anywhere** — it is computed at
|
||||||
|
request time (below). No valuation cache or table exists.
|
||||||
|
|
||||||
|
Workstream B:
|
||||||
|
|
||||||
|
- Dolt OHLCV+splits pull/import: `0 2 * * tue-sat` ET. If source_max_date is not
|
||||||
|
fresh, retry hourly until ~06:00, then give up quietly. After a successful
|
||||||
|
import, reconcile the previous session's Dolt-derived bars against the Alpaca
|
||||||
|
bars; summary into the run's validation JSON, details to logs.
|
||||||
|
- Move `schedule_daily_pipeline_cron` (morning refresh) from `0 2 * * *` to
|
||||||
|
`0 3 * * *` (only needed once the 02:00 slot is taken by the OHLCV pull).
|
||||||
|
**Late Dolt publication is a non-event:** the canonical scan runs at 15:30 on
|
||||||
|
Alpaca, so the morning pipeline runs normally even when the import hasn't
|
||||||
|
landed — no gating, no defensive coupling.
|
||||||
|
|
||||||
|
## Metrics catalog (curated — TTM basis)
|
||||||
|
|
||||||
|
Snapshots store **price-independent per-period facts** (the "snapshot" column below
|
||||||
|
means *derived from stored snapshots, assembled across periods at read time* — see
|
||||||
|
Schema — not frozen at import); price-dependent ratios are never frozen into
|
||||||
|
snapshots and have **no storage location at all**: the API computes
|
||||||
|
them at request time from the stored snapshots + the latest `ohlcv_records` close
|
||||||
|
(both already in PostgreSQL, so this works identically when SEC is unreachable).
|
||||||
|
The only stored price-dependent values are the legacy `fundamental_data` fields
|
||||||
|
that scoring already reads, refreshed daily by step (c) after activation.
|
||||||
|
|
||||||
|
| Metric | Definition | Where computed |
|
||||||
|
|---|---|---|
|
||||||
|
| Revenue growth YoY | TTM revenue vs prior TTM | snapshot |
|
||||||
|
| EPS growth YoY | TTM diluted EPS vs prior TTM | snapshot |
|
||||||
|
| Operating margin + 4q trend | TTM operating income / revenue | snapshot |
|
||||||
|
| FCF margin | (TTM CFO − capex) / revenue | snapshot |
|
||||||
|
| Net debt | total debt − (cash + ST investments); positive = net debt | snapshot |
|
||||||
|
| Net debt / EBITDA | net debt / TTM EBITDA | snapshot |
|
||||||
|
| Share count Δ YoY | shares outstanding vs year ago | snapshot |
|
||||||
|
| Trailing P/E | price / TTM diluted EPS | request time |
|
||||||
|
| FCF yield | TTM FCF / est. market cap | request time |
|
||||||
|
| Est. market cap | issuer-wide shares outstanding × ticker price | request time |
|
||||||
|
| Earnings surprise history | last 4+ from `earnings_events` | query |
|
||||||
|
|
||||||
|
**Market cap is an estimate** (issuer-wide shares outstanding × one ticker's price —
|
||||||
|
approximate for multi-class issuers). Share count comes from a single consolidated
|
||||||
|
value, not a class sum: prefer the one `dei:EntityCommonStockSharesOutstanding`
|
||||||
|
cover-page fact; if absent (e.g. Alphabet) fall back to
|
||||||
|
`us-gaap:CommonStockSharesOutstanding` at period end. companyfacts is
|
||||||
|
non-dimensional, so class-specific facts can't be summed reliably — never do that,
|
||||||
|
and never substitute weighted-average/diluted shares; if conflicting values remain,
|
||||||
|
store null. Label it "est." in the UI and round aggressively rather than withholding
|
||||||
|
it; false precision is the failure mode, not the approximation.
|
||||||
|
|
||||||
|
**Units follow existing app conventions:** percentages are percentage points
|
||||||
|
(21.0 = 21%), P/E and net-debt/EBITDA are multiples, market cap and net debt are
|
||||||
|
dollars.
|
||||||
|
|
||||||
|
Deliberately **excluded**: ROIC (invested-capital/NOPAT normalization too noisy),
|
||||||
|
gross margin (COGS tagging too inconsistent), any new composite score.
|
||||||
|
|
||||||
|
## Peer comparison (read-time only)
|
||||||
|
|
||||||
|
- Peer group = tracked-universe issuers sharing the **first two SIC digits**,
|
||||||
|
**deduplicated by CIK** — GOOG and GOOGL are one issuer, one observation, in
|
||||||
|
medians, percentiles and peer_count.
|
||||||
|
- Computed at read time from current snapshots — no aggregate tables until
|
||||||
|
performance demonstrates a need.
|
||||||
|
- Medians exclude null/invalid values. **Fewer than 5 valid peer issuers → omit
|
||||||
|
the peer result entirely** rather than showing a misleading universe comparison.
|
||||||
|
- Percentile direction respects metric polarity (higher-is-better for FCF yield,
|
||||||
|
lower-is-better for P/E and leverage).
|
||||||
|
|
||||||
|
## API contract (additive v1)
|
||||||
|
|
||||||
|
Every existing top-level field is preserved unchanged (name, type, position) —
|
||||||
|
backend and frontend ship independently, no breaking interval. New objects, exact
|
||||||
|
names and types:
|
||||||
|
|
||||||
|
```jsonc
|
||||||
|
{
|
||||||
|
// ...all existing legacy fields, unchanged...
|
||||||
|
"earnings": {
|
||||||
|
"next": {"date": "YYYY-MM-DD", "session": "bmo|amc|unknown", "days_until": 12} | null,
|
||||||
|
"recent": [ // newest first, max 4, may be empty
|
||||||
|
{"announce_date": "YYYY-MM-DD", "period_end": "YYYY-MM-DD|null",
|
||||||
|
"eps_estimate": 1.02|null, "eps_actual": 1.10|null, "surprise_pct": 7.8|null}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"metrics": [ // fixed row set — every key always present, value null when unavailable
|
||||||
|
{
|
||||||
|
"key": "revenue_growth_yoy", // revenue_growth_yoy | eps_growth_yoy | operating_margin |
|
||||||
|
// fcf_margin | net_debt | net_debt_to_ebitda | share_count_change_yoy
|
||||||
|
"value": 18.0, // number | null — pp / multiples / dollars per units above
|
||||||
|
"history": [ // oldest→newest, max 4 points, [] when unavailable
|
||||||
|
{"period_end": "YYYY-MM-DD", "value": 8.0}
|
||||||
|
],
|
||||||
|
"industry": { // object | null — null when < 5 valid peer issuers (CIK-deduped)
|
||||||
|
"label": "SIC 73 peers", // truthful 2-digit group label — grouping IS 2-digit,
|
||||||
|
// so no 4-digit description like "Prepackaged Software"
|
||||||
|
"median": 11.0,
|
||||||
|
"favorable_percentile": 82, // 0-100, polarity-aware (higher = more favorable)
|
||||||
|
"peer_count": 12 // issuers, not tickers
|
||||||
|
},
|
||||||
|
"period_end": "YYYY-MM-DD|null",
|
||||||
|
"filed_date": "YYYY-MM-DD|null",
|
||||||
|
"source": "sec|dolt|legacy_api"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"valuation": { // object | null (null until SEC snapshots exist, phase A3); same industry sub-object rules
|
||||||
|
// computed at REQUEST TIME from stored snapshots + latest PostgreSQL close —
|
||||||
|
// no valuation cache or table; unaffected by SEC availability
|
||||||
|
"pe": 29.2|null, "fcf_yield": 3.8|null,
|
||||||
|
"market_cap_est": 1.2e9|null, // estimated — UI labels "est."
|
||||||
|
"pe_industry": {...}|null, "fcf_yield_industry": {...}|null,
|
||||||
|
"price_date": "YYYY-MM-DD" // close used for the ratios
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Null/freshness semantics: absent data is `null` with the row still present (the UI
|
||||||
|
shows "n/a", never hides rows); every metric carries its own source, period and
|
||||||
|
filing date — no panel-wide source label. The objects may serve partial data during
|
||||||
|
rollout (e.g. `earnings` live, `metrics` still `legacy_api`); the shape never
|
||||||
|
changes.
|
||||||
|
|
||||||
|
## UI — `frontend/src/components/ticker/FundamentalsPanel.tsx`
|
||||||
|
|
||||||
|
One distinctive visual device — the **Reference Rails** — in an otherwise restrained
|
||||||
|
panel. Preserve the app's dark glass styling and numeric typography.
|
||||||
|
|
||||||
|
```
|
||||||
|
Fundamentals
|
||||||
|
Growth accelerating · margins improving · valuation priced above peers
|
||||||
|
Next earnings Aug 3 · AMC EPS surprises ▂ ▅ ▃ ▆
|
||||||
|
|
||||||
|
Operating trend less favorable ← ref → more favorable
|
||||||
|
Revenue growth 18%
|
||||||
|
───────────────│━━━━● +3pp vs prior · accelerating
|
||||||
|
Share count YoY −1.7%
|
||||||
|
───────────────│━━━━● buying back
|
||||||
|
|
||||||
|
Valuation & balance less favorable ← median → more favorable
|
||||||
|
P/E 29.2×
|
||||||
|
────●━━━━━━━━━━│──── priced above peers · median 23.5× · 12 peers
|
||||||
|
```
|
||||||
|
|
||||||
|
- Growth and margins: horizontal rails compare the latest value with the prior
|
||||||
|
quarter or prior-period average; share-count YoY compares with zero. The rail
|
||||||
|
is normalized so right is always more favorable, including buybacks.
|
||||||
|
- P/E, FCF yield and leverage: horizontal favorable-percentile rails with a
|
||||||
|
peer-median marker. No decorative rail when `industry` is null (< 5 peers).
|
||||||
|
- Every row keeps the exact value and one deterministic comparison caption;
|
||||||
|
missing values render `n/a`, and insufficient peers render `peers n/a`.
|
||||||
|
- Earnings: four bars around a shared zero baseline — cyan beats, coral misses, gray
|
||||||
|
unavailable — plus next date and BMO/AMC session countdown.
|
||||||
|
- Accessibility: color is always paired with text; neutral/ambiguous stays gray;
|
||||||
|
rails and earnings bars expose complete ARIA descriptions.
|
||||||
|
- Remove the hard-coded "FMP" source label; surface filing and price-date
|
||||||
|
provenance in the footer.
|
||||||
|
|
||||||
|
**Deterministic reads — one shared rule set.** Implement as a single function with
|
||||||
|
named constants; the metric reads and the header sentence use identical outputs. No
|
||||||
|
LLM, no new composite score. Defaults (tunable constants, not scattered literals):
|
||||||
|
|
||||||
|
- A series read requires ≥ 3 periods; otherwise show "—" and no read.
|
||||||
|
- Growth metrics (pp): latest − prior ≥ +2.0 → "accelerating";
|
||||||
|
≤ −2.0 → "decelerating"; else "steady".
|
||||||
|
- Margins (latest vs mean of prior periods, pp): ≥ +1.0 → "improving";
|
||||||
|
≤ −1.0 → "deteriorating"; else "stable" (phrased "above/below own average"
|
||||||
|
where the layout calls for it).
|
||||||
|
- Share count YoY: > +1.0% → "N% dilution"; < −1.0% → "buying back"; else "flat".
|
||||||
|
- Peer-relative: favorable_percentile ≥ 60 → favorable ("above peers");
|
||||||
|
≤ 40 → adverse ("priced above peers" for P/E, "elevated leverage" for
|
||||||
|
net-debt/EBITDA); else "in line".
|
||||||
|
- Header sentence: join the growth read, margin read and peer-relative valuation
|
||||||
|
read with " · ", omitting segments that have no read (e.g. "Growth accelerating
|
||||||
|
· margins stable · valuation above industry median"). **Segment sources are
|
||||||
|
fixed:** growth = revenue growth read; margins = operating margin read;
|
||||||
|
valuation = P/E peer-relative read, falling back to FCF yield when P/E is null.
|
||||||
|
This keeps the header unambiguous when sibling metrics (EPS vs revenue growth,
|
||||||
|
P/E vs FCF yield) point in different directions.
|
||||||
|
|
||||||
|
## Decommissioning (end of workstream A)
|
||||||
|
|
||||||
|
Remove completely: FMP (`app/providers/fmp.py`), Finnhub + Alpha Vantage
|
||||||
|
(`app/providers/fundamentals_chain.py`), their config keys (`app/config.py`), and
|
||||||
|
their wiring in `app/scheduler.py`, `app/routers/ingestion.py`,
|
||||||
|
`app/services/ticker_universe_service.py`. Retain: Alpaca (prices), FRED,
|
||||||
|
sentiment provider, Telegram. Note: decommissioning does **not** depend on
|
||||||
|
workstream B — Alpaca remains the price source throughout.
|
||||||
|
|
||||||
|
## Rollout
|
||||||
|
|
||||||
|
**Workstream A:**
|
||||||
|
|
||||||
|
- A0. License review **DONE** (earnings approved for private/internal use under
|
||||||
|
CC BY-SA 4.0, no redistribution — see Licensing above). The Dolt version,
|
||||||
|
persistent `DOLT_DATA_DIR`, clone, and production checks are captured in
|
||||||
|
`deploy/provision_fundamentals.sh` and `docs/fundamentals-deployment.md`.
|
||||||
|
- A1. Migration 026, import-run framework.
|
||||||
|
- A2. Earnings ingestion in shadow (writes `earnings_events`, prod untouched);
|
||||||
|
verify forward-calendar coverage and rescheduling behavior.
|
||||||
|
- A3. SEC daily job in shadow (writes `fundamental_snapshots`). **Primary technical
|
||||||
|
risk here: Q4 derivation and fiscal-period alignment** — non-calendar fiscal years,
|
||||||
|
restatements/amendments, and XBRL unit/dimension variants; budget accordingly.
|
||||||
|
- A4. API v1 + FundamentalsPanel + peer comparison — served from snapshots and
|
||||||
|
earnings_events, independent of the scoring cutover (the additive API supports
|
||||||
|
partial data). UI value ships before anything touches scoring inputs.
|
||||||
|
- A5. **Score-parity gate** → `fundamental_data` cutover: compute candidate
|
||||||
|
pe_ratio/revenue_growth/earnings_surprise from SEC/Dolt side by side with the
|
||||||
|
API values across the tracked universe, report per-field deltas and resulting
|
||||||
|
fundamental-score/ranking changes, require explicit approval. Definition
|
||||||
|
changes (e.g. TTM vs provider convention) called out, not averaged away.
|
||||||
|
- A6. Remove FMP/Finnhub/Alpha Vantage; keep monitoring + manual fallback.
|
||||||
|
|
||||||
|
**Workstream B (independent, start when wanted):**
|
||||||
|
|
||||||
|
- B0. Stocks clone (~4.7 GB) provisioned; migration 027.
|
||||||
|
- B1. OHLCV + split adjustment in shadow (writes `ohlcv_source_bars` only; Alpaca
|
||||||
|
keeps owning `ohlcv_records`); historical backfill.
|
||||||
|
- B2. Reconciliation window (≥ 2 weeks) vs Alpaca; review validation summaries.
|
||||||
|
- B3. Promote Dolt as historical OHLCV source (canonical rebuilt from raw source
|
||||||
|
bars + splits); morning pipeline → 03:00.
|
||||||
|
|
||||||
|
## Test plan
|
||||||
|
|
||||||
|
- Daily SEC job: changed revision imports; unchanged conditional-HTTP check is a
|
||||||
|
`no_op` with zero downloads and zero row changes; validation failure leaves
|
||||||
|
production untouched; source unavailable still runs the local
|
||||||
|
`fundamental_data` refresh (step c, post-activation); before activation the job
|
||||||
|
never writes `fundamental_data`.
|
||||||
|
- Valuation endpoint returns identical values with SEC reachable and unreachable
|
||||||
|
(pure PostgreSQL computation); no valuation rows exist in any table.
|
||||||
|
- Multi-class tickers resolve to the same CIK snapshots; peer medians and
|
||||||
|
peer_count are CIK-deduplicated (GOOG+GOOGL = one observation).
|
||||||
|
- Earnings rescheduling: a moved future date replaces the old row atomically; a
|
||||||
|
cancelled date disappears; historical results are never touched.
|
||||||
|
- Amendment selection: for a period with multiple accessions, the newest valid
|
||||||
|
accepted_at wins at read time; older rows remain unchanged.
|
||||||
|
- History arrays are chronological, ≤ 4 points.
|
||||||
|
- Percentage-point units stay compatible with existing formatters and scoring
|
||||||
|
inputs.
|
||||||
|
- Deterministic reads: threshold boundary cases (exactly +2.0pp, exactly 60th
|
||||||
|
percentile) resolve per the stated rules; header uses identical outputs and
|
||||||
|
falls back from P/E to FCF yield for the valuation segment when P/E is null.
|
||||||
|
- Peer comparison disappears below 5 peer issuers; favorable-percentile direction
|
||||||
|
correct for both polarities.
|
||||||
|
- Workstream B: split-adjusted OHLCV matches Alpaca on representative normal /
|
||||||
|
split / reverse-split symbols.
|
||||||
|
- UI states: positive, adverse, neutral, insufficient history, insufficient
|
||||||
|
peers; mobile layout; non-color accessibility.
|
||||||
|
- Unit, integration, scheduler and frontend suites pass.
|
||||||
|
|
||||||
|
## Acceptance criteria
|
||||||
|
|
||||||
|
- App works normally with Dolt/DoltHub/SEC unreachable.
|
||||||
|
- Re-running the same revision: zero duplicate or changed rows.
|
||||||
|
- **Upcoming earnings dates present and timely for the tracked universe** — the
|
||||||
|
forward calendar is the hardest thing to replace and gates decommissioning.
|
||||||
|
- Coverage meets the tracked-universe target; scheduler runs cleanly with
|
||||||
|
FMP/Finnhub/AV keys removed from the environment.
|
||||||
|
- Score-parity diff reviewed and approved before `fundamental_data` cutover.
|
||||||
|
- Scheduled imports never block the API event loop.
|
||||||
|
|
||||||
|
## Deferred (explicitly, until a concrete need appears)
|
||||||
|
|
||||||
|
- Workstream B itself is deferred relative to A and blocks nothing in A.
|
||||||
|
- Exact byte-level source replay of historical imports; permanent archive store.
|
||||||
|
- Point-in-time backtest enforcement (`accepted_at` is stored now; derivation and
|
||||||
|
backtest visibility rules are built only when fundamentals enter
|
||||||
|
scoring/backtesting).
|
||||||
|
- Fundamental metrics in the score; sector-relative scoring.
|
||||||
|
- Aggregate/rollup tables for peer statistics.
|
||||||
|
- Any valuation cache or table (request-time computation from snapshots + latest
|
||||||
|
close suffices).
|
||||||
|
- A dedicated conflicts table (validation JSON + logs suffice).
|
||||||
@@ -0,0 +1,247 @@
|
|||||||
|
# A3 design — SEC fundamentals importer
|
||||||
|
|
||||||
|
Status: **design APPROVED 2026-07-22 — three decisions signed off (daily-index
|
||||||
|
fetch, primary-period-only snapshots, full-history backfill) + four review
|
||||||
|
correctness fixes folded in (composite revision incl. universe fingerprint;
|
||||||
|
submissions pagination shards for full history; index↔Company-Facts consistency
|
||||||
|
gate; insert-only immutability with discrepancy reporting; deterministic cash/debt
|
||||||
|
composition). Ready to implement.**
|
||||||
|
Companion to `docs/dolt-integration-plan.md` (workstream A, phase A3). Grounded in
|
||||||
|
live SEC data probes (Apple CIK 0000320193, company_tickers, submissions, daily-index).
|
||||||
|
|
||||||
|
## Objective (unchanged from the plan)
|
||||||
|
|
||||||
|
Populate `fundamental_snapshots` (CIK-keyed, one immutable row per accession)
|
||||||
|
and `tickers.cik/sic/sic_description` from SEC data, as a `SourceImporter`
|
||||||
|
plugging into the A1 framework. Shadow only (A3): nothing reads snapshots until
|
||||||
|
A4; `fundamental_data` is untouched until the A5 parity gate. All new metrics
|
||||||
|
are display-only.
|
||||||
|
|
||||||
|
## What the SEC data actually looks like (probed, not assumed)
|
||||||
|
|
||||||
|
`data.sec.gov/api/xbrl/companyfacts/CIK##########.json` — one JSON per **issuer
|
||||||
|
(CIK)** aggregating every period across every filing. Shape:
|
||||||
|
`facts.us-gaap.<Concept>.units.<unit>[] = {start, end, val, fy, fp, form, filed, accn, frame}`.
|
||||||
|
|
||||||
|
Ground-truth findings that drive the design:
|
||||||
|
|
||||||
|
1. **`fp` is only `Q1|Q2|Q3|FY` — there is no `Q4`.** Q4 must be derived.
|
||||||
|
2. **`fy`/`fp` are the *filing's* fiscal context, not each fact's period.** Proven:
|
||||||
|
Apple's FY2019 10-K carries a discrete Q3-FY2018 revenue fact
|
||||||
|
(`start 2018-07-01, end 2018-09-29, val 62.9B`) tagged `fp=FY` — it's a
|
||||||
|
comparative. **Period identity lives in `(start, end)` + the filing's
|
||||||
|
`reportDate`, never in `fp/fy`.** Selecting values by `fp` would silently mix
|
||||||
|
comparatives into the wrong period.
|
||||||
|
3. SEC provides **both** discrete 3-month facts **and** YTD-cumulative facts
|
||||||
|
(Apple Q2 FY26: YTD `254,940` over 6mo *and* discrete `111,184` over 3mo;
|
||||||
|
`143,756 + 111,184 = 254,940`). This confirms the stored-YTD schema: store
|
||||||
|
cumulative YTD per filing, derive discretes/Q4/TTM at read time.
|
||||||
|
4. Instant facts (`dei:EntityCommonStockSharesOutstanding`) end on the **cover
|
||||||
|
date** (2026-04-17), which differs from `period_end` (2026-03-28) → the
|
||||||
|
`shares_outstanding_date` column added in migration 026.
|
||||||
|
5. **No conditional-GET support:** the companyfacts endpoint returns no `ETag`
|
||||||
|
and no `Last-Modified`. AAPL's file is 3.75 MB. So ~505 unconditional fetches
|
||||||
|
≈ 0.5–1.5 GB *per run* — the plan's "conditional HTTP no-op" is impossible on
|
||||||
|
this endpoint. This is the fact that decides the fetch strategy (below).
|
||||||
|
6. `submissions/CIK##########.json` supplies `sic`, `sicDescription`,
|
||||||
|
`fiscalYearEnd` (e.g. `0926`), and per-accession `reportDate` +
|
||||||
|
`acceptanceDateTime` — the keys for period selection and `accepted_at`.
|
||||||
|
7. `company_tickers.json` uses **dash** tickers (`BRK-B`, `BRK-A`) and maps
|
||||||
|
`GOOGL`/`GOOG` to the **same** `cik_str` (1652044). The ticker→CIK join reuses
|
||||||
|
the earnings importer's `normalise_symbol` (dot→dash), so both sides match.
|
||||||
|
|
||||||
|
## Decision 1 (APPROVED) — fetch strategy: EDGAR daily-index driven
|
||||||
|
|
||||||
|
**Plan said** bulk `companyfacts.zip` + ETag no-op. **Reality:** the data.sec.gov
|
||||||
|
endpoints expose no validators, and the bulk zip is multi-GB and changes ~daily
|
||||||
|
(all of EDGAR), so ETag would rarely match → near-daily multi-GB download to get
|
||||||
|
505 issuers. Per-CIK *conditional* fetch is impossible (finding 5). Per-CIK
|
||||||
|
*unconditional* is 0.5–1.5 GB every night.
|
||||||
|
|
||||||
|
**Recommended:** drive off the **EDGAR daily-index** (`daily-index/YYYY/QTRn/
|
||||||
|
form.YYYYMMDD.idx` — fixed-width Form/Company/CIK/Date/accession, ~3300 rows/day,
|
||||||
|
confirmed). Each run:
|
||||||
|
|
||||||
|
- `detect_revision` → a **composite revision**, not just the date:
|
||||||
|
`latest-index-date` + a hash of the index content processed this run + a
|
||||||
|
**fingerprint of the tracked symbol→CIK set**. The CIK fingerprint is essential:
|
||||||
|
a newly added ticker changes the revision and forces a run, so a new ticker is
|
||||||
|
never `no_op`'d away or starved waiting for its issuer to file. Equal composite
|
||||||
|
revision → `no_op`.
|
||||||
|
- **No backfill sentinel.** The *absence of a prior promoted run* is what triggers
|
||||||
|
the initial full-history backfill; `source_max_date` records the processed index
|
||||||
|
date each run.
|
||||||
|
- `stage` → for each index date since the last processed one, parse the form
|
||||||
|
index, keep rows where `form ∈ {10-K, 10-Q, 10-K/A, 10-Q/A}` **and** CIK ∈
|
||||||
|
tracked set, then fetch `companyfacts/CIK.json` for **only those few issuers**
|
||||||
|
and extract their newly-reported period(s). Most nights this is a handful of
|
||||||
|
issuers → near-zero transfer, respectful of SEC fair-access.
|
||||||
|
- **First run (backfill)**: no prior promoted run → fetch companyfacts for all
|
||||||
|
tracked CIKs once (~1 GB one-time) and seed **full** history. Full history needs
|
||||||
|
the paginated submissions shards — see "CIK resolution" below.
|
||||||
|
|
||||||
|
Why this over the alternatives: transfer scales with *filings*, not with all of
|
||||||
|
EDGAR or with the universe size every night; it restores the revision/no_op
|
||||||
|
model; and it's the lightest load on SEC. Cost: daily-index parsing + date
|
||||||
|
bookkeeping (store last-processed index date in `data_import_runs` /
|
||||||
|
settings). **This deviates from the plan's "bulk zip" — requesting sign-off.**
|
||||||
|
|
||||||
|
## Decision 2 (APPROVED) — snapshot mapping: primary-period, YTD, immutable
|
||||||
|
|
||||||
|
One `fundamental_snapshots` row per accession, representing the filing's
|
||||||
|
**primary current period only** (not its comparatives):
|
||||||
|
|
||||||
|
- **Select the primary period by `end == submissions.reportDate[accn]`** (finding
|
||||||
|
2), *not* by `fp/fy`. `fiscal_period` label comes from the filing's own `fp`
|
||||||
|
(a 10-Q's own `fp` matches its current quarter; a 10-K → `FY`);
|
||||||
|
`fiscal_year`/`period_start`/`period_end` from the selected facts + submissions.
|
||||||
|
- **Duration facts → cumulative YTD.** For each concept, pick the duration fact
|
||||||
|
with `accn == thisFiling`, `end == reportDate`, and `start ≈ fiscal-year start`
|
||||||
|
(derived from `fiscalYearEnd`), sanity-checked by span length (Q1≈3mo, Q2≈6mo,
|
||||||
|
Q3≈9mo, FY≈12mo). **If the YTD fact is absent, store null — never a discrete
|
||||||
|
masquerading as cumulative** (that would poison read-time differencing).
|
||||||
|
- **Balance-sheet instants → at `end == reportDate`.** `shares_outstanding` is
|
||||||
|
the exception: prefer the `dei:EntityCommonStockSharesOutstanding` cover-page
|
||||||
|
fact and store *its own* `end` in `shares_outstanding_date` (cover date ≠
|
||||||
|
period_end); when there is no dei fact (e.g. Alphabet) fall back to
|
||||||
|
`us-gaap:CommonStockSharesOutstanding` at `reportDate`. A single consolidated
|
||||||
|
value — never a class sum (companyfacts is non-dimensional) nor
|
||||||
|
weighted-average/diluted; conflicting values → null.
|
||||||
|
- **Amendments:** a real `10-K/A` / `10-Q/A` is a new accession → a new immutable
|
||||||
|
row for the same `(cik, fy, fp)`; readers pick the newest valid `accepted_at`.
|
||||||
|
- **Out of scope (stated, not silent):** restatements that appear *only* as
|
||||||
|
comparatives inside a later normal filing are **not** captured — only a real
|
||||||
|
amendment updates a prior period. This narrows the plan's "newest accepted_at
|
||||||
|
per period" to amendment-driven updates; a deliberate KISS boundary.
|
||||||
|
- **Immutable means insert-only, not upsert.** `promote` **inserts** new accession
|
||||||
|
rows with `ON CONFLICT (accession) DO NOTHING`. An accession never mutates: if a
|
||||||
|
re-fetch reconstructs *different* values for an accession already stored, that is
|
||||||
|
a **discrepancy to report** (into `validation_json` + a system event), never a
|
||||||
|
silent overwrite, and the original `import_run_id` is never replaced. (Ordinary
|
||||||
|
updates arrive as a *new* amendment accession, which is a new row.)
|
||||||
|
|
||||||
|
## Read-time derivation (constrains the importer; built in A4)
|
||||||
|
|
||||||
|
From the per-accession YTD rows, all at read time (newest `accepted_at` per
|
||||||
|
period), following the schema decision already in the plan:
|
||||||
|
|
||||||
|
- discrete quarter = YTD(Qn) − YTD(Qn−1); **Q4 = FY − YTD(Q3)**.
|
||||||
|
- TTM = sum of the trailing four discrete quarters (e.g. TTM@Q2 = FY(prev) +
|
||||||
|
YTD(Q2) − YTD(Q2 prev year)).
|
||||||
|
- YoY = period vs same period a year earlier.
|
||||||
|
- **Hard rule the importer must enable: any missing period in a run → the derived
|
||||||
|
value is `null`, never a partial number.** So the importer must aim for
|
||||||
|
complete consecutive quarter runs per issuer and report gaps.
|
||||||
|
|
||||||
|
## Metric tag catalog (prioritized us-gaap tags + fallbacks)
|
||||||
|
|
||||||
|
Tagging is inconsistent across issuers (the plan's known risk). Each metric
|
||||||
|
resolves through an ordered tag list; first present wins; unit-checked.
|
||||||
|
|
||||||
|
| Snapshot field | Primary tag | Fallbacks | Unit |
|
||||||
|
|---|---|---|---|
|
||||||
|
| revenue | `RevenueFromContractWithCustomerExcludingAssessedTax` | `Revenues`, `SalesRevenueNet` | USD |
|
||||||
|
| net_income | `NetIncomeLoss` | — | USD |
|
||||||
|
| operating_income | `OperatingIncomeLoss` | — | USD |
|
||||||
|
| diluted_eps | `EarningsPerShareDiluted` | — | USD/shares |
|
||||||
|
| cfo | `NetCashProvidedByUsedInOperatingActivities` | `...ContinuingOperations` | USD |
|
||||||
|
| capex | `PaymentsToAcquirePropertyPlantAndEquipment` | `PaymentsToAcquireProductiveAssets` | USD |
|
||||||
|
| depreciation_amortization | `DepreciationDepletionAndAmortization` | `DepreciationAmortizationAndAccretionNet`, `DepreciationAndAmortization` | USD |
|
||||||
|
| cash_and_st_investments | see composition rule | — | USD |
|
||||||
|
| total_debt | see composition rule | — | USD |
|
||||||
|
| shares_outstanding | `dei:EntityCommonStockSharesOutstanding` | — | shares |
|
||||||
|
|
||||||
|
**Composite fields — deterministic, aggregate-first, no double counting.** Each
|
||||||
|
source tag contributes at most once:
|
||||||
|
|
||||||
|
- `cash_and_st_investments` = `CashAndCashEquivalentsAtCarryingValue`
|
||||||
|
**+ short-term investments**, where ST investments = the **first present** of
|
||||||
|
[`ShortTermInvestments`, `MarketableSecuritiesCurrent`] — never both summed.
|
||||||
|
- `total_debt` = **long-term component + short-term component**, where
|
||||||
|
- long-term = first present of [`LongTermDebt` (the aggregate, already includes
|
||||||
|
current + noncurrent portions), **else** (`LongTermDebtNoncurrent` +
|
||||||
|
`LongTermDebtCurrent`)];
|
||||||
|
- short-term borrowings = first present of [`ShortTermBorrowings`,
|
||||||
|
`CommercialPaper`] (0 if neither).
|
||||||
|
So the long-term aggregate and its components are mutually exclusive, and CP vs
|
||||||
|
short-term-borrowings is a single pick — nothing is counted twice.
|
||||||
|
|
||||||
|
EBITDA (for net-debt/EBITDA) is derived at read time = operating_income + D&A.
|
||||||
|
Concepts absent for an issuer → that field is null (display-only; no synthesis).
|
||||||
|
The exact tag lists live as named constants, tunable without touching logic.
|
||||||
|
|
||||||
|
## Fiscal-period identity
|
||||||
|
|
||||||
|
`fiscalYearEnd` (MMDD from submissions) anchors the fiscal-year start for YTD
|
||||||
|
span checks and Q4 derivation. Non-calendar fiscal years (Apple's Sept) are
|
||||||
|
handled because we key on `(start, end)` + `reportDate`, not calendar quarters.
|
||||||
|
`fiscal_year`/`fiscal_period` are stored from the filing's own `fy`/`fp` for its
|
||||||
|
primary period (safe — a filing's own context is correct for its current period).
|
||||||
|
|
||||||
|
## CIK resolution & tickers backfill
|
||||||
|
|
||||||
|
- From `company_tickers.json`: `normalise_symbol(ticker) → cik_str`. Set
|
||||||
|
`tickers.cik` for each tracked ticker (multi-class share one CIK).
|
||||||
|
- From `submissions/CIK.json`: `sic`, `sicDescription`, `fiscalYearEnd` →
|
||||||
|
`tickers.sic/sic_description` (+ fiscal anchor for YTD/Q4).
|
||||||
|
- **Submissions is paginated — full history needs the shards.** `filings.recent`
|
||||||
|
holds only the **latest 1000** filings (verified: Apple `recent` = 1000). Older
|
||||||
|
accessions live in `filings.files[]` = `[{name, filingFrom, filingTo,
|
||||||
|
filingCount}]` (e.g. `CIK0000320193-submissions-001.json`, 1236 filings
|
||||||
|
1994–2015), each a bare object with the **same parallel arrays** including
|
||||||
|
`reportDate`, `acceptanceDateTime`, and `isXBRL`. The full-history backfill must
|
||||||
|
**follow every `filings.files[].name`** to obtain period identity + `accepted_at`
|
||||||
|
+ `isXBRL` for pre-1000 accessions. Incremental runs only need `recent`.
|
||||||
|
- Refreshed by the SEC job; a newly added ticker self-resolves on its next run
|
||||||
|
(the CIK fingerprint in the revision forces that run) — until then its snapshots
|
||||||
|
are absent → metrics null, per the plan.
|
||||||
|
|
||||||
|
## SourceImporter mapping (source = `sec_facts`)
|
||||||
|
|
||||||
|
- `detect_revision` → latest daily-index date (or `backfill` sentinel on first run).
|
||||||
|
- `stage` → resolve tracked CIKs; (incremental) parse indices since last date →
|
||||||
|
tracked filers → fetch their companyfacts → build per-accession snapshot rows;
|
||||||
|
(backfill) fetch all tracked companyfacts. In-memory staged set (KISS, per A1).
|
||||||
|
- `validate` (fail-closed) → tracked-universe **coverage floor** (issuers with ≥1
|
||||||
|
snapshot); **unit/period sanity** (YTD spans within tolerance; EPS in USD/shares);
|
||||||
|
**no duplicate accession**; **filings skipped for missing period identity are
|
||||||
|
counted in `validation_json`** (carry-forward from A1 review); an unexpected
|
||||||
|
companyfacts shape (missing `facts`/`units`) stops promotion.
|
||||||
|
- **Index↔Company-Facts consistency gate (the daily index and Company Facts are
|
||||||
|
separate SEC products that can lag each other):** for every tracked index
|
||||||
|
accession marked `isXBRL`, confirm that accession actually appears in the fetched
|
||||||
|
companyfacts before promotion. If any is missing → **fail the run and retry
|
||||||
|
later** — do **not** advance the revision and do **not** record an
|
||||||
|
incomplete/null snapshot for it. Non-XBRL amendments are skipped with a recorded
|
||||||
|
reason in `validation_json`. (The framework only stores the revision on a
|
||||||
|
promoted run, so a failed consistency check naturally leaves the revision behind
|
||||||
|
for retry.)
|
||||||
|
- `promote` → **insert** snapshot rows (`ON CONFLICT (accession) DO NOTHING`;
|
||||||
|
immutable — see Decision 2), stamped `import_run_id`; refresh
|
||||||
|
`tickers.cik/sic/sic_description`. A re-fetch that reconstructs different values
|
||||||
|
for an existing accession is reported as a discrepancy, never a silent mutation.
|
||||||
|
Non-destructive (append-only accessions) — no future-row deletion like earnings.
|
||||||
|
|
||||||
|
## SEC fair-access (operational, per the plan's non-negotiable)
|
||||||
|
|
||||||
|
Identifying `User-Agent` with contact email on every request; well under 10 req/s
|
||||||
|
with spacing; exponential backoff on 429; **403 → alert and stop, never
|
||||||
|
retry-loop**. New config: `sec_user_agent`, `sec_request_spacing_seconds`,
|
||||||
|
`sec_max_retries`. Keep only the last ~2 fetched artifacts on disk for debugging
|
||||||
|
(reproducibility is the normalized Postgres rows, per the plan).
|
||||||
|
|
||||||
|
## Explicitly out of scope for A3
|
||||||
|
|
||||||
|
- `fundamental_data` cutover (A5 parity gate) — snapshots only in A3.
|
||||||
|
- The read-time derivation, API object, and panel (A4).
|
||||||
|
- Comparative-only restatements (Decision 2).
|
||||||
|
- Point-in-time backtest enforcement (`accepted_at` stored, not yet enforced).
|
||||||
|
|
||||||
|
## Decisions (signed off 2026-07-22)
|
||||||
|
|
||||||
|
1. **Fetch** — EDGAR daily-index driven (Decision 1). Approved deviation from the
|
||||||
|
plan's bulk zip.
|
||||||
|
2. **Snapshot mapping** — primary-period-only per accession; comparative-only
|
||||||
|
restatements out of scope (Decision 2). Approved.
|
||||||
|
3. **Backfill depth** — seed **full** available history per issuer on first run
|
||||||
|
(cheap to store; powers the quarter tape / multi-year YoY). Approved.
|
||||||
@@ -0,0 +1,133 @@
|
|||||||
|
# Fundamentals production deployment
|
||||||
|
|
||||||
|
This is the one-time production setup for the Dolt earnings and SEC fundamentals
|
||||||
|
imports. Both imports remain shadow inputs until the separate A5 scoring-cutover
|
||||||
|
approval. Do not add OS cron entries: the application scheduler owns both jobs.
|
||||||
|
|
||||||
|
## What the deployment adds
|
||||||
|
|
||||||
|
- `Dolt Earnings Import (shadow)` runs daily at 02:30 America/New_York.
|
||||||
|
- `SEC Fundamentals Import (shadow)` runs daily at 04:00 America/New_York.
|
||||||
|
- Both jobs are visible, toggleable, and manually triggerable in Admin → Jobs.
|
||||||
|
- Cron expressions are editable in Admin → Schedule.
|
||||||
|
- Every attempt is recorded in `data_import_runs`; failures also create a system
|
||||||
|
event. A failed validation does not promote partial data.
|
||||||
|
|
||||||
|
The systemd service uses one application worker. The import framework also holds
|
||||||
|
a PostgreSQL advisory lock per source, so an overlapping manual/scheduled run is
|
||||||
|
skipped safely.
|
||||||
|
|
||||||
|
## Prerequisites
|
||||||
|
|
||||||
|
The production `.env` at `/opt/signalplatform/.env` must contain:
|
||||||
|
|
||||||
|
```dotenv
|
||||||
|
DOLT_BINARY=/usr/local/bin/dolt
|
||||||
|
DOLT_DATA_DIR=/var/lib/signal-platform/dolt
|
||||||
|
DOLT_EARNINGS_SUBDIR=earnings
|
||||||
|
DOLT_MIN_FREE_DISK_GB=5.0
|
||||||
|
SEC_USER_AGENT=signal-platform/1.0 (contact: real-address@example.com)
|
||||||
|
SEC_REQUEST_SPACING_SECONDS=0.2
|
||||||
|
```
|
||||||
|
|
||||||
|
Use a real monitored contact address. Keep at least 5 GB free at the Dolt data
|
||||||
|
path; 8–10 GB gives comfortable growth headroom. The data directory must stay
|
||||||
|
outside `/opt/signalplatform`, because deployments use `rsync --delete` there.
|
||||||
|
|
||||||
|
## One-time provisioning
|
||||||
|
|
||||||
|
First deploy the commit containing this bundle to production. Then SSH to the
|
||||||
|
server and run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd /opt/signalplatform
|
||||||
|
sudo bash ./deploy/provision_fundamentals.sh
|
||||||
|
sudo -u deploy bash ./deploy/provision_fundamentals.sh --check
|
||||||
|
sudo systemctl restart signalplatform.service
|
||||||
|
curl -fsS http://127.0.0.1:8998/api/v1/health
|
||||||
|
```
|
||||||
|
|
||||||
|
The provisioner is idempotent. It installs the pinned Dolt version, creates the
|
||||||
|
persistent directory as `deploy:deploy`, clones
|
||||||
|
`post-no-preference/earnings`, verifies free space and `.env`, and refuses an
|
||||||
|
unexpected Dolt version. It does not modify PostgreSQL or start an import.
|
||||||
|
|
||||||
|
Do not replace the pinned version with `latest`. A future Dolt upgrade should be
|
||||||
|
a reviewed change to `DOLT_VERSION`, followed by the same provision/check flow.
|
||||||
|
|
||||||
|
## First-run verification
|
||||||
|
|
||||||
|
In Admin → Jobs, wait until no other job is running, then:
|
||||||
|
|
||||||
|
1. Trigger **Dolt Earnings Import (shadow)**. Expect `completed` with import
|
||||||
|
status `promoted`; a repeat without an upstream change should report `no_op`.
|
||||||
|
2. Trigger **SEC Fundamentals Import (shadow)**. The first run performs the
|
||||||
|
tracked-universe history backfill and can take materially longer than a daily
|
||||||
|
incremental run. Expect `completed` with import status `promoted`.
|
||||||
|
3. Check Admin → System Events. There should be no new import error.
|
||||||
|
4. Confirm the next-run times correspond to 02:30 and 04:00 New York time.
|
||||||
|
5. Open several ticker pages and confirm the fundamentals panel has populated
|
||||||
|
data and still handles partial/missing issuers cleanly.
|
||||||
|
|
||||||
|
Optional database verification:
|
||||||
|
|
||||||
|
```sql
|
||||||
|
SELECT source, status, revision, source_max_date, started_at, completed_at,
|
||||||
|
validation_json
|
||||||
|
FROM data_import_runs
|
||||||
|
WHERE source IN ('dolt_earnings', 'sec_facts')
|
||||||
|
ORDER BY id DESC
|
||||||
|
LIMIT 10;
|
||||||
|
|
||||||
|
SELECT count(*) FROM earnings_events WHERE source = 'dolt_earnings';
|
||||||
|
SELECT count(*), count(DISTINCT cik) FROM fundamental_snapshots;
|
||||||
|
```
|
||||||
|
|
||||||
|
During the longer first SEC run, execute the following in a second SSH session.
|
||||||
|
It opens an independent database connection and attempts the same source lock:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd /opt/signalplatform
|
||||||
|
sudo -u deploy .venv/bin/python - <<'PY'
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
from sqlalchemy import text
|
||||||
|
|
||||||
|
from app.database import engine
|
||||||
|
from app.services.data_import import _advisory_key
|
||||||
|
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
key = _advisory_key("sec_facts")
|
||||||
|
async with engine.connect() as connection:
|
||||||
|
acquired = await connection.scalar(
|
||||||
|
text("SELECT pg_try_advisory_lock(:key)"), {"key": key}
|
||||||
|
)
|
||||||
|
print("UNEXPECTED: lock acquired" if acquired else "OK: source lock is busy")
|
||||||
|
if acquired:
|
||||||
|
await connection.execute(
|
||||||
|
text("SELECT pg_advisory_unlock(:key)"), {"key": key}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
asyncio.run(main())
|
||||||
|
PY
|
||||||
|
```
|
||||||
|
|
||||||
|
Expect `OK: source lock is busy`. This is the remaining live-PostgreSQL
|
||||||
|
mutual-exclusion check; SQLite unit tests cannot exercise PostgreSQL advisory
|
||||||
|
locks. A second Admin trigger should independently report the job as busy.
|
||||||
|
|
||||||
|
## Failure and rollback
|
||||||
|
|
||||||
|
- Disable the failing shadow job in Admin → Jobs. This stops scheduled imports
|
||||||
|
without changing existing data or the legacy scoring path.
|
||||||
|
- Inspect the job runtime, latest `data_import_runs.validation_json`, service
|
||||||
|
logs, and Admin → System Events before retrying.
|
||||||
|
- Re-run `sudo -u deploy bash ./deploy/provision_fundamentals.sh --check` for
|
||||||
|
binary, clone, permission, disk, or environment failures.
|
||||||
|
- The Dolt clone is a reproducible cache and does not need a bespoke backup.
|
||||||
|
PostgreSQL (including `earnings_events`, `fundamental_snapshots`, and import
|
||||||
|
audit rows) must remain covered by the normal production database backup.
|
||||||
|
- Do not proceed to A5 while either shadow feed is unhealthy or the parity gate
|
||||||
|
has not received explicit approval.
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
<!doctype html>
|
||||||
|
<html lang="en" class="dark">
|
||||||
|
<head>
|
||||||
|
<meta charset="UTF-8" />
|
||||||
|
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||||
|
<title>FundamentalsPanel harness</title>
|
||||||
|
<link rel="preconnect" href="https://fonts.googleapis.com" />
|
||||||
|
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin />
|
||||||
|
<link
|
||||||
|
href="https://fonts.googleapis.com/css2?family=Space+Grotesk:wght@400;500;600;700&family=Instrument+Sans:wght@400;500;600;700&family=IBM+Plex+Mono:wght@400;500;600&display=swap"
|
||||||
|
rel="stylesheet"
|
||||||
|
/>
|
||||||
|
</head>
|
||||||
|
<body class="bg-[#0a0b11] text-gray-100 font-sans">
|
||||||
|
<div id="root"></div>
|
||||||
|
<script type="module" src="/src/dev/harness.tsx"></script>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
@@ -6,10 +6,12 @@ import { SkeletonTable } from '../ui/Skeleton';
|
|||||||
const DEFAULTS: ScheduleConfig = {
|
const DEFAULTS: ScheduleConfig = {
|
||||||
schedule_timezone: 'America/New_York',
|
schedule_timezone: 'America/New_York',
|
||||||
schedule_daily_pipeline_cron: '0 2 * * *',
|
schedule_daily_pipeline_cron: '0 2 * * *',
|
||||||
schedule_near_close_pipeline_cron: '30 15 * * 1-5',
|
schedule_dolt_earnings_cron: '30 2 * * *',
|
||||||
schedule_after_close_pipeline_cron: '45 16 * * 1-5',
|
schedule_sec_fundamentals_cron: '0 4 * * *',
|
||||||
schedule_intraday_pipeline_cron: '0 10-15 * * 1-5',
|
schedule_near_close_pipeline_cron: '30 15 * * mon-fri',
|
||||||
schedule_fundamentals_cron: '0 1 * * 1',
|
schedule_after_close_pipeline_cron: '45 16 * * mon-fri',
|
||||||
|
schedule_intraday_pipeline_cron: '0 10-15 * * mon-fri',
|
||||||
|
schedule_fundamentals_cron: '0 1 * * mon',
|
||||||
};
|
};
|
||||||
|
|
||||||
const FIELDS: { key: keyof ScheduleConfig; label: string; hint: string; mono?: boolean }[] = [
|
const FIELDS: { key: keyof ScheduleConfig; label: string; hint: string; mono?: boolean }[] = [
|
||||||
@@ -24,6 +26,18 @@ const FIELDS: { key: keyof ScheduleConfig; label: string; hint: string; mono?: b
|
|||||||
hint: 'OHLCV → benchmark → sentiment → regime → alerts (no R:R scan). Default 02:00 ET so regime-quadrant changes hit Telegram in the morning.',
|
hint: 'OHLCV → benchmark → sentiment → regime → alerts (no R:R scan). Default 02:00 ET so regime-quadrant changes hit Telegram in the morning.',
|
||||||
mono: true,
|
mono: true,
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
key: 'schedule_dolt_earnings_cron',
|
||||||
|
label: 'Dolt earnings (shadow)',
|
||||||
|
hint: 'Pull and import earnings dates/results daily at 02:30 ET. Live scoring remains untouched before A5.',
|
||||||
|
mono: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: 'schedule_sec_fundamentals_cron',
|
||||||
|
label: 'SEC fundamentals (shadow)',
|
||||||
|
hint: 'Import tracked-universe SEC facts daily at 04:00 ET. Unchanged revisions become no-op runs.',
|
||||||
|
mono: true,
|
||||||
|
},
|
||||||
{
|
{
|
||||||
key: 'schedule_near_close_pipeline_cron',
|
key: 'schedule_near_close_pipeline_cron',
|
||||||
label: 'Near-close pipeline (scan + alert)',
|
label: 'Near-close pipeline (scan + alert)',
|
||||||
@@ -44,8 +58,8 @@ const FIELDS: { key: keyof ScheduleConfig; label: string; hint: string; mono?: b
|
|||||||
},
|
},
|
||||||
{
|
{
|
||||||
key: 'schedule_fundamentals_cron',
|
key: 'schedule_fundamentals_cron',
|
||||||
label: 'Fundamentals (weekly)',
|
label: 'Legacy fundamentals (weekly)',
|
||||||
hint: 'Slow, rate-limited. Default early Monday ET.',
|
hint: 'Existing provider chain retained until the A5 parity approval and A6 removal.',
|
||||||
mono: true,
|
mono: true,
|
||||||
},
|
},
|
||||||
];
|
];
|
||||||
|
|||||||
@@ -1,157 +1,397 @@
|
|||||||
import { useMemo, useState } from 'react';
|
import { useMemo, type ReactNode } from 'react';
|
||||||
import { formatPercent, formatLargeNumber } from '../../lib/format';
|
import type {
|
||||||
import {
|
EarningsRecent,
|
||||||
fundamentalScore,
|
FundamentalResponse,
|
||||||
metricStatus,
|
MetricItem,
|
||||||
overallFundamentalStatus,
|
MetricIndustry,
|
||||||
} from '../../lib/fundamentals';
|
} from '../../lib/types';
|
||||||
import type { FundamentalResponse } from '../../lib/types';
|
|
||||||
|
|
||||||
interface FundamentalsPanelProps {
|
interface FundamentalsPanelProps {
|
||||||
data: FundamentalResponse;
|
data: FundamentalResponse;
|
||||||
}
|
}
|
||||||
|
|
||||||
const FIELD_LABELS: Record<string, string> = {
|
/** Favorable / neutral / adverse — always paired with the read text. */
|
||||||
pe_ratio: 'P/E Ratio',
|
type Tone = 'good' | 'flat' | 'bad';
|
||||||
revenue_growth: 'Revenue Growth',
|
|
||||||
earnings_surprise: 'Earnings Surprise',
|
|
||||||
market_cap: 'Market Cap',
|
|
||||||
};
|
|
||||||
|
|
||||||
type MetricKey = 'pe_ratio' | 'revenue_growth' | 'earnings_surprise' | 'market_cap';
|
// Horizon tokens.
|
||||||
|
const HZ = {
|
||||||
|
text: '#EDEEF3',
|
||||||
|
muted: '#9AA0B0',
|
||||||
|
track: '#5D6373',
|
||||||
|
fav: '#6EC9DB', // cyan
|
||||||
|
adv: '#EF9182', // coral
|
||||||
|
};
|
||||||
|
const toneColor = (t: Tone) => (t === 'good' ? HZ.fav : t === 'bad' ? HZ.adv : HZ.muted);
|
||||||
|
|
||||||
|
const POSITIVE_READS = new Set([
|
||||||
|
'accelerating', 'improving', 'above peers', 'above own average', 'buying back',
|
||||||
|
'attractively valued', 'conservative leverage',
|
||||||
|
]);
|
||||||
|
const NEGATIVE_READS = new Set([
|
||||||
|
'decelerating', 'deteriorating', 'priced above peers', 'elevated leverage', 'below peers',
|
||||||
|
]);
|
||||||
|
function readTone(read: string | null | undefined): Tone {
|
||||||
|
if (!read) return 'flat';
|
||||||
|
if (POSITIVE_READS.has(read)) return 'good';
|
||||||
|
if (NEGATIVE_READS.has(read)) return 'bad';
|
||||||
|
if (read.includes('dilution')) return 'bad';
|
||||||
|
return 'flat';
|
||||||
|
}
|
||||||
|
|
||||||
|
function pct(v: number | null | undefined): string {
|
||||||
|
if (v == null || !Number.isFinite(v)) return 'n/a';
|
||||||
|
return `${Math.round(v * 10) / 10}%`;
|
||||||
|
}
|
||||||
|
function mult(v: number | null | undefined): string {
|
||||||
|
if (v == null || !Number.isFinite(v)) return 'n/a';
|
||||||
|
return `${v.toFixed(1)}×`;
|
||||||
|
}
|
||||||
|
function money(v: number | null | undefined): string {
|
||||||
|
if (v == null || !Number.isFinite(v)) return 'n/a';
|
||||||
|
const abs = Math.abs(v);
|
||||||
|
if (abs >= 1e12) return `$${(v / 1e12).toFixed(1)}T`;
|
||||||
|
if (abs >= 1e9) return `$${(v / 1e9).toFixed(1)}B`;
|
||||||
|
if (abs >= 1e6) return `$${(v / 1e6).toFixed(1)}M`;
|
||||||
|
return `$${v.toFixed(0)}`;
|
||||||
|
}
|
||||||
|
function signedPp(v: number): string {
|
||||||
|
return `${v >= 0 ? '+' : ''}${Math.round(v * 10) / 10}pp`;
|
||||||
|
}
|
||||||
|
function capitalize(s: string): string {
|
||||||
|
return s.length ? s[0].toUpperCase() + s.slice(1) : s;
|
||||||
|
}
|
||||||
|
function finiteOrNull(v: number | null | undefined): number | null {
|
||||||
|
return v != null && Number.isFinite(v) ? v : null;
|
||||||
|
}
|
||||||
|
function latestHistory(metric: MetricItem | undefined): number[] {
|
||||||
|
const run: number[] = [];
|
||||||
|
const history = metric?.history ?? [];
|
||||||
|
for (let i = history.length - 1; i >= 0; i -= 1) {
|
||||||
|
const value = finiteOrNull(history[i].value);
|
||||||
|
if (value == null) break;
|
||||||
|
run.unshift(value);
|
||||||
|
}
|
||||||
|
return run;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Parse YYYY-MM-DD as a LOCAL calendar date (no UTC off-by-one west of UTC). */
|
||||||
|
function parseLocalDate(s: string): Date {
|
||||||
|
const [y, m, d] = s.split('-').map(Number);
|
||||||
|
return new Date(y, (m ?? 1) - 1, d ?? 1);
|
||||||
|
}
|
||||||
|
function shortDate(s: string): string {
|
||||||
|
return parseLocalDate(s).toLocaleDateString(undefined, { month: 'short', day: 'numeric' });
|
||||||
|
}
|
||||||
|
|
||||||
export function FundamentalsPanel({ data }: FundamentalsPanelProps) {
|
export function FundamentalsPanel({ data }: FundamentalsPanelProps) {
|
||||||
const [expanded, setExpanded] = useState<boolean>(false);
|
const metrics = useMemo(
|
||||||
|
() => Object.fromEntries((data.metrics ?? []).map((m) => [m.key, m])),
|
||||||
|
[data.metrics],
|
||||||
|
) as Record<string, MetricItem | undefined>;
|
||||||
|
const reads = data.reads?.by_key ?? {};
|
||||||
|
const val = data.valuation;
|
||||||
|
const earnings = data.earnings;
|
||||||
|
const provenance = (data.metrics ?? []).find((m) => m.period_end) ?? null;
|
||||||
|
|
||||||
const score = useMemo(
|
const hasAny = (data.metrics ?? []).some((m) => m.value != null) || !!val || !!earnings?.next
|
||||||
() =>
|
|| (earnings?.recent?.length ?? 0) > 0;
|
||||||
fundamentalScore({
|
|
||||||
pe_ratio: data.pe_ratio,
|
|
||||||
revenue_growth: data.revenue_growth,
|
|
||||||
earnings_surprise: data.earnings_surprise,
|
|
||||||
}),
|
|
||||||
[data.pe_ratio, data.revenue_growth, data.earnings_surprise],
|
|
||||||
);
|
|
||||||
const overall = overallFundamentalStatus(score);
|
|
||||||
|
|
||||||
const items: {
|
const trendRows: { key: string; label: string; kind: 'growth' | 'margin' | 'share' }[] = [
|
||||||
key: MetricKey;
|
{ key: 'revenue_growth_yoy', label: 'Revenue growth', kind: 'growth' },
|
||||||
label: string;
|
{ key: 'eps_growth_yoy', label: 'EPS growth', kind: 'growth' },
|
||||||
value: number | null;
|
{ key: 'operating_margin', label: 'Operating margin', kind: 'margin' },
|
||||||
format: (v: number) => string;
|
{ key: 'fcf_margin', label: 'FCF margin', kind: 'margin' },
|
||||||
}[] = [
|
{ key: 'share_count_change_yoy', label: 'Share count YoY', kind: 'share' },
|
||||||
{ key: 'pe_ratio', label: 'P/E Ratio', value: data.pe_ratio, format: (v) => v.toFixed(2) },
|
|
||||||
{ key: 'revenue_growth', label: 'Revenue Growth', value: data.revenue_growth, format: formatPercent },
|
|
||||||
{ key: 'earnings_surprise', label: 'Earnings Surprise', value: data.earnings_surprise, format: formatPercent },
|
|
||||||
{ key: 'market_cap', label: 'Market Cap', value: data.market_cap, format: formatLargeNumber },
|
|
||||||
];
|
];
|
||||||
|
|
||||||
const unavailableEntries = Object.entries(data.unavailable_fields ?? {});
|
const valueRows: {
|
||||||
|
label: string; value: number | null; industry: MetricIndustry | null;
|
||||||
|
readKey: string; fmt: (v: number | null) => string;
|
||||||
|
}[] = [
|
||||||
|
{ label: 'Net debt / EBITDA', value: metrics.net_debt_to_ebitda?.value ?? null,
|
||||||
|
industry: metrics.net_debt_to_ebitda?.industry ?? null, readKey: 'net_debt_to_ebitda', fmt: mult },
|
||||||
|
{ label: 'P/E', value: val?.pe ?? null, industry: val?.pe_industry ?? null, readKey: 'pe', fmt: mult },
|
||||||
|
{ label: 'FCF yield', value: val?.fcf_yield ?? null, industry: val?.fcf_yield_industry ?? null,
|
||||||
|
readKey: 'fcf_yield', fmt: pct },
|
||||||
|
];
|
||||||
|
|
||||||
return (
|
return (
|
||||||
<div className="glass p-5">
|
<section className="glass p-5" aria-label="Fundamentals">
|
||||||
<div className="mb-3 flex items-baseline justify-between gap-2">
|
<h3 className="text-[10px] font-medium uppercase tracking-widest" style={{ color: HZ.muted }}>Fundamentals</h3>
|
||||||
<h3 className="text-xs font-medium uppercase tracking-widest text-gray-500">Fundamentals</h3>
|
{data.reads?.header ? (
|
||||||
{score != null && (
|
<p className="mt-0.5 text-[15px] leading-snug" style={{ color: HZ.text }}>
|
||||||
<span className="num text-[10px] text-gray-600" title="Equal average of available P/E, growth, and surprise (need 2+)">
|
{capitalize(data.reads.header)}
|
||||||
score {score.toFixed(0)}
|
</p>
|
||||||
</span>
|
) : !hasAny ? (
|
||||||
)}
|
<p className="mt-1 text-[15px] leading-snug" style={{ color: HZ.muted }}>
|
||||||
</div>
|
No fundamentals reported yet.
|
||||||
|
</p>
|
||||||
|
) : null}
|
||||||
|
|
||||||
<p className={`text-sm font-semibold ${overall.tone}`}>{overall.text}</p>
|
{hasAny && (
|
||||||
|
<>
|
||||||
|
<EarningsStrip earnings={earnings} />
|
||||||
|
|
||||||
<div className="mt-3 space-y-3 text-sm">
|
<div className="mt-4 grid gap-x-8 gap-y-5 sm:grid-cols-2">
|
||||||
{items.map((item) => {
|
<div>
|
||||||
const reason = data.unavailable_fields?.[item.key];
|
<SectionHead label="Operating trend" axis="less favorable ← ref → more favorable" />
|
||||||
const status = item.value !== null ? metricStatus(item.key, item.value) : null;
|
<div className="mt-2.5 space-y-3.5">
|
||||||
let display: React.ReactNode;
|
{trendRows.map((r) => (
|
||||||
let valueClass = 'text-gray-200';
|
<TrendRow key={r.key} label={r.label} kind={r.kind}
|
||||||
|
metric={metrics[r.key]} read={reads[r.key]} />
|
||||||
if (item.value !== null) {
|
))}
|
||||||
display = item.format(item.value);
|
|
||||||
} else if (reason) {
|
|
||||||
display = reason;
|
|
||||||
valueClass = 'text-amber-400';
|
|
||||||
} else {
|
|
||||||
display = '—';
|
|
||||||
}
|
|
||||||
|
|
||||||
return (
|
|
||||||
<div key={item.key} className="flex items-start justify-between gap-3">
|
|
||||||
<span className="text-gray-400">{item.label}</span>
|
|
||||||
<div className="min-w-0 text-right">
|
|
||||||
<div className={`num ${valueClass}`}>{display}</div>
|
|
||||||
{status && (
|
|
||||||
<div className={`mt-0.5 text-[11.5px] font-medium ${status.tone}`}>{status.text}</div>
|
|
||||||
)}
|
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
);
|
<div>
|
||||||
})}
|
<SectionHead label="Valuation & balance" axis="less favorable ← median → more favorable" />
|
||||||
</div>
|
<div className="mt-2.5 space-y-3.5">
|
||||||
|
{valueRows.map((r) => (
|
||||||
<p className="mt-4 text-[11px] leading-relaxed text-gray-500">
|
<ValueRow key={r.label} {...r} read={reads[r.readKey]} />
|
||||||
Score = average of available P/E, revenue growth, and earnings surprise (need 2+).
|
))}
|
||||||
{' '}P/E: lower scores higher (≈15 best, ≈45 worst).
|
|
||||||
{' '}Growth / surprise: 0% is neutral; stronger positives lift the score.
|
|
||||||
{' '}Market cap is size context only — not scored.
|
|
||||||
</p>
|
|
||||||
|
|
||||||
<button
|
|
||||||
type="button"
|
|
||||||
onClick={() => setExpanded((prev) => !prev)}
|
|
||||||
className="mt-3 flex w-full items-center justify-center gap-1 text-xs text-gray-500 transition-colors hover:text-gray-300"
|
|
||||||
aria-expanded={expanded}
|
|
||||||
aria-label={expanded ? 'Collapse details' : 'Expand details'}
|
|
||||||
>
|
|
||||||
<svg
|
|
||||||
className={`h-4 w-4 transition-transform ${expanded ? 'rotate-180' : ''}`}
|
|
||||||
fill="none"
|
|
||||||
viewBox="0 0 24 24"
|
|
||||||
stroke="currentColor"
|
|
||||||
strokeWidth={2}
|
|
||||||
>
|
|
||||||
<path strokeLinecap="round" strokeLinejoin="round" d="M19 9l-7 7-7-7" />
|
|
||||||
</svg>
|
|
||||||
</button>
|
|
||||||
|
|
||||||
{expanded && (
|
|
||||||
<div className="mt-3 space-y-3 border-t border-white/10 pt-3">
|
|
||||||
<div className="space-y-1 text-sm">
|
|
||||||
<div className="flex justify-between">
|
|
||||||
<span className="text-gray-500">Data Source</span>
|
|
||||||
<span className="text-gray-300">FMP</span>
|
|
||||||
</div>
|
|
||||||
{data.fetched_at && (
|
|
||||||
<div className="flex justify-between">
|
|
||||||
<span className="text-gray-500">Fetched</span>
|
|
||||||
<span className="text-gray-300">{new Date(data.fetched_at).toLocaleString()}</span>
|
|
||||||
</div>
|
</div>
|
||||||
)}
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
{unavailableEntries.length > 0 && (
|
<Provenance provenance={provenance} priceDate={val?.price_date ?? null}
|
||||||
<div>
|
marketCap={val?.market_cap_est ?? null} />
|
||||||
<span className="text-xs font-medium uppercase tracking-widest text-gray-500">Unavailable Fields</span>
|
</>
|
||||||
<ul className="mt-1 space-y-1">
|
|
||||||
{unavailableEntries.map(([field, reason]) => (
|
|
||||||
<li key={field} className="flex justify-between text-sm">
|
|
||||||
<span className="text-gray-400">{FIELD_LABELS[field] ?? field}</span>
|
|
||||||
<span className="text-amber-400">{reason}</span>
|
|
||||||
</li>
|
|
||||||
))}
|
|
||||||
</ul>
|
|
||||||
</div>
|
|
||||||
)}
|
|
||||||
</div>
|
|
||||||
)}
|
)}
|
||||||
|
</section>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
{!expanded && data.fetched_at && (
|
function SectionHead({ label, axis }: { label: string; axis: string }) {
|
||||||
<p className="mt-2 text-xs text-gray-500">
|
return (
|
||||||
Updated {new Date(data.fetched_at).toLocaleDateString()}
|
<div className="flex flex-wrap items-baseline justify-between gap-x-2">
|
||||||
</p>
|
<span className="text-[10px] font-medium uppercase tracking-widest" style={{ color: HZ.muted }}>{label}</span>
|
||||||
)}
|
<span className="text-[11px]" style={{ color: HZ.track }}>{axis}</span>
|
||||||
</div>
|
</div>
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function Bullet({ label, value, rail, comparison }: {
|
||||||
|
label: string; value: ReactNode; rail: ReactNode | null; comparison: ReactNode;
|
||||||
|
}) {
|
||||||
|
return (
|
||||||
|
<div>
|
||||||
|
<div className="flex items-baseline justify-between gap-2">
|
||||||
|
<span className="truncate text-sm" style={{ color: HZ.muted }}>{label}</span>
|
||||||
|
<span className="num text-[15px]" style={{ color: HZ.text }}>{value}</span>
|
||||||
|
</div>
|
||||||
|
{rail && <div className="mt-1.5">{rail}</div>}
|
||||||
|
<div className={rail ? 'mt-1 text-[11.5px] leading-snug' : 'mt-1.5 text-[11.5px] leading-snug'}>
|
||||||
|
{comparison}
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- operating-trend row (delta vs reference, favorable = right) ------------
|
||||||
|
|
||||||
|
function TrendRow({ label, kind, metric, read }: {
|
||||||
|
label: string; kind: 'growth' | 'margin' | 'share';
|
||||||
|
metric: MetricItem | undefined; read: string | null | undefined;
|
||||||
|
}) {
|
||||||
|
const tone = readTone(read);
|
||||||
|
const value = finiteOrNull(metric?.value);
|
||||||
|
const history = latestHistory(metric);
|
||||||
|
|
||||||
|
let ref: number | null = null;
|
||||||
|
let refWord = '';
|
||||||
|
let halfRange = 8;
|
||||||
|
let neutral = 2;
|
||||||
|
let favSign = 1; // +1: higher is favorable; -1: lower is favorable
|
||||||
|
if (kind === 'growth') {
|
||||||
|
ref = history.length >= 2 ? history[history.length - 2] : null;
|
||||||
|
refWord = 'prior'; halfRange = 8; neutral = 2; favSign = 1;
|
||||||
|
} else if (kind === 'margin') {
|
||||||
|
const prior = history.slice(0, -1);
|
||||||
|
ref = prior.length ? prior.reduce((a, b) => a + b, 0) / prior.length : null;
|
||||||
|
refWord = 'avg'; halfRange = 4; neutral = 1; favSign = 1;
|
||||||
|
} else {
|
||||||
|
ref = 0; refWord = ''; halfRange = 5; neutral = 1; favSign = -1; // buyback (negative) is favorable
|
||||||
|
}
|
||||||
|
const delta = value != null && ref != null ? value - ref : null;
|
||||||
|
|
||||||
|
const comparison = value == null ? (
|
||||||
|
<span style={{ color: HZ.track }}>n/a</span>
|
||||||
|
) : delta == null ? (
|
||||||
|
<span style={{ color: HZ.track }}>history n/a</span>
|
||||||
|
) : (
|
||||||
|
<span style={{ color: toneColor(tone) }}>
|
||||||
|
{kind !== 'share' && (
|
||||||
|
<span className="num">{signedPp(delta)} vs {refWord} · </span>
|
||||||
|
)}
|
||||||
|
{read ?? '—'}
|
||||||
|
</span>
|
||||||
|
);
|
||||||
|
|
||||||
|
const rail = delta != null
|
||||||
|
? <DeltaRail favOffset={favSign * delta} halfRange={halfRange} neutral={neutral} tone={tone}
|
||||||
|
ariaLabel={`${label} ${pct(value)}, ${delta != null ? `${signedPp(delta)} vs reference` : ''}, ${read ?? 'no read'}`} />
|
||||||
|
: null;
|
||||||
|
|
||||||
|
return <Bullet label={label} value={pct(value)} rail={rail} comparison={comparison} />;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Comparison rail centered on a reference line (not a progress bar). favOffset > 0
|
||||||
|
* is favorable and moves the dot RIGHT for every metric. */
|
||||||
|
function DeltaRail({ favOffset, halfRange, neutral, tone, ariaLabel }: {
|
||||||
|
favOffset: number; halfRange: number; neutral: number; tone: Tone; ariaLabel: string;
|
||||||
|
}) {
|
||||||
|
const clamped = Math.max(-halfRange, Math.min(halfRange, favOffset));
|
||||||
|
const pos = 50 + (clamped / halfRange) * 50;
|
||||||
|
const barLeft = Math.min(50, pos);
|
||||||
|
const barWidth = Math.abs(pos - 50);
|
||||||
|
const bandHalf = (neutral / halfRange) * 50;
|
||||||
|
return (
|
||||||
|
<span className="relative block h-1.5 min-w-[7rem] flex-1 rounded-full" role="img" aria-label={ariaLabel}
|
||||||
|
style={{ background: 'rgba(93,99,115,0.22)' }}>
|
||||||
|
<span className="absolute inset-y-0 rounded-full" aria-hidden
|
||||||
|
style={{ left: `${50 - bandHalf}%`, width: `${2 * bandHalf}%`, background: 'rgba(93,99,115,0.4)' }} />
|
||||||
|
<span className="absolute inset-y-[-2px] w-px" aria-hidden style={{ left: '50%', background: 'rgba(237,238,243,0.55)' }} />
|
||||||
|
<span className="absolute inset-y-0 rounded-full" aria-hidden style={{ left: `${barLeft}%`, width: `${barWidth}%`, background: toneColor(tone) }} />
|
||||||
|
<Dot pos={pos} tone={tone} />
|
||||||
|
</span>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- valuation/balance row (favorable percentile vs median) ----------------
|
||||||
|
|
||||||
|
function ValueRow({ label, value, industry, read, fmt }: {
|
||||||
|
label: string; value: number | null; industry: MetricIndustry | null;
|
||||||
|
read: string | null | undefined; fmt: (v: number | null) => string;
|
||||||
|
}) {
|
||||||
|
const tone = readTone(read);
|
||||||
|
const safeValue = finiteOrNull(value);
|
||||||
|
const rail = safeValue == null || !industry
|
||||||
|
? null
|
||||||
|
: <PercentileRail percentile={industry.favorable_percentile} tone={tone}
|
||||||
|
ariaLabel={`${label}: ${industry.favorable_percentile}th favorable percentile vs ${industry.label}, median ${fmt(industry.median)}, ${industry.peer_count} peers`} />;
|
||||||
|
const comparison = safeValue == null ? (
|
||||||
|
<span style={{ color: HZ.track }}>n/a</span>
|
||||||
|
) : industry ? (
|
||||||
|
<span style={{ color: toneColor(tone) }}>
|
||||||
|
{read ?? 'in line'}<span style={{ color: HZ.muted }}> · median {fmt(industry.median)} · {industry.peer_count} peers</span>
|
||||||
|
</span>
|
||||||
|
) : (
|
||||||
|
<span style={{ color: HZ.track }}>peers n/a</span>
|
||||||
|
);
|
||||||
|
return <Bullet label={label} value={fmt(safeValue)} rail={rail} comparison={comparison} />;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** 0-100 favorable-percentile rail with the peer median fixed at 50; right = more favorable. */
|
||||||
|
function PercentileRail({ percentile, tone, ariaLabel }: {
|
||||||
|
percentile: number; tone: Tone; ariaLabel: string;
|
||||||
|
}) {
|
||||||
|
const p = Math.max(0, Math.min(100, percentile));
|
||||||
|
const barLeft = Math.min(50, p);
|
||||||
|
const barWidth = Math.abs(p - 50);
|
||||||
|
return (
|
||||||
|
<span className="relative block h-1.5 min-w-[7rem] flex-1 rounded-full" role="img" aria-label={ariaLabel}
|
||||||
|
style={{ background: 'rgba(93,99,115,0.22)' }}>
|
||||||
|
<span className="absolute inset-y-[-2px] w-px" aria-hidden style={{ left: '50%', background: 'rgba(237,238,243,0.55)' }} />
|
||||||
|
<span className="absolute inset-y-0 rounded-full" aria-hidden style={{ left: `${barLeft}%`, width: `${barWidth}%`, background: toneColor(tone) }} />
|
||||||
|
<Dot pos={p} tone={tone} />
|
||||||
|
</span>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
function Dot({ pos, tone }: { pos: number; tone: Tone }) {
|
||||||
|
return (
|
||||||
|
<span className="absolute top-1/2 h-2.5 w-2.5 -translate-x-1/2 -translate-y-1/2 rounded-full" aria-hidden
|
||||||
|
style={{ left: `${pos}%`, background: toneColor(tone), boxShadow: '0 0 0 2px #11131C' }} />
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- earnings + provenance -------------------------------------------------
|
||||||
|
|
||||||
|
function Provenance({ provenance, priceDate, marketCap }: {
|
||||||
|
provenance: MetricItem | null; priceDate: string | null; marketCap: number | null;
|
||||||
|
}) {
|
||||||
|
if (!provenance?.period_end && !priceDate) return null;
|
||||||
|
return (
|
||||||
|
<p className="mt-4 border-t border-white/10 pt-2 text-[11px] leading-relaxed" style={{ color: HZ.track }}>
|
||||||
|
{provenance?.period_end && (
|
||||||
|
<>SEC filings · latest {shortDate(provenance.period_end)}
|
||||||
|
{provenance.filed_date && <> (filed {shortDate(provenance.filed_date)})</>}</>
|
||||||
|
)}
|
||||||
|
{priceDate && (
|
||||||
|
<>{provenance?.period_end ? ' · ' : ''}Valuation at {shortDate(priceDate)} close · market cap {money(marketCap)} est.</>
|
||||||
|
)}
|
||||||
|
</p>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
function EarningsStrip({ earnings }: { earnings: FundamentalResponse['earnings'] }) {
|
||||||
|
const next = earnings?.next;
|
||||||
|
const recent = earnings?.recent ?? [];
|
||||||
|
const when = next ? (next.days_until === 0 ? 'today' : `${next.days_until}d`) : null;
|
||||||
|
return (
|
||||||
|
<div className="mt-2 flex flex-wrap items-center justify-between gap-x-4 gap-y-1.5">
|
||||||
|
<span className="text-sm" style={{ color: HZ.text }}>
|
||||||
|
<span style={{ color: HZ.muted }}>Next earnings </span>
|
||||||
|
{next ? (
|
||||||
|
<>
|
||||||
|
{shortDate(next.date)}
|
||||||
|
{' · '}
|
||||||
|
<span className="uppercase">{next.session === 'unknown' ? 'TBD' : next.session}</span>
|
||||||
|
<span style={{ color: HZ.muted }}> · {when}</span>
|
||||||
|
</>
|
||||||
|
) : (
|
||||||
|
<span style={{ color: HZ.muted }}>no date</span>
|
||||||
|
)}
|
||||||
|
</span>
|
||||||
|
{recent.length > 0 && <SurpriseSpark recent={recent} />}
|
||||||
|
</div>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Four tiny diverging bars around a zero baseline: beat above (cyan), miss below
|
||||||
|
* (coral), height ~ |surprise %|. Reads as a beat/miss history at a glance. */
|
||||||
|
function SurpriseSpark({ recent }: { recent: EarningsRecent[] }) {
|
||||||
|
const ordered = recent.slice().reverse();
|
||||||
|
const description = ordered.map((e) => {
|
||||||
|
const surprise = e.surprise_pct;
|
||||||
|
const amount = surprise != null ? ` ${surprise > 0 ? '+' : ''}${surprise}%` : '';
|
||||||
|
return `${e.announce_date} ${surpriseLabel(e)}${amount}`;
|
||||||
|
}).join(', ');
|
||||||
|
return (
|
||||||
|
<span className="flex items-center gap-2">
|
||||||
|
<span className="text-[10px] uppercase tracking-widest" style={{ color: HZ.track }}>
|
||||||
|
EPS surprises
|
||||||
|
</span>
|
||||||
|
<span
|
||||||
|
className="relative flex h-7 items-center gap-1.5 px-0.5"
|
||||||
|
role="img"
|
||||||
|
tabIndex={0}
|
||||||
|
title={description}
|
||||||
|
aria-label={`Recent EPS surprises, oldest to newest: ${description}`}
|
||||||
|
>
|
||||||
|
<span className="absolute inset-x-0 top-1/2 h-px" aria-hidden style={{ background: 'rgba(93,99,115,0.5)' }} />
|
||||||
|
{ordered.map((e, i) => <SurpriseBar key={i} e={e} />)}
|
||||||
|
</span>
|
||||||
|
</span>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
function surpriseLabel(e: EarningsRecent): string {
|
||||||
|
const beat = e.eps_actual != null && e.eps_estimate != null ? e.eps_actual - e.eps_estimate : null;
|
||||||
|
return beat == null ? 'n/a' : beat > 0 ? 'beat' : beat < 0 ? 'miss' : 'in line';
|
||||||
|
}
|
||||||
|
|
||||||
|
function SurpriseBar({ e }: { e: EarningsRecent }) {
|
||||||
|
const beat = e.eps_actual != null && e.eps_estimate != null ? e.eps_actual - e.eps_estimate : null;
|
||||||
|
const tone: Tone = beat == null ? 'flat' : beat > 0 ? 'good' : beat < 0 ? 'bad' : 'flat';
|
||||||
|
const s = e.surprise_pct;
|
||||||
|
const mag = s != null ? Math.min(Math.abs(s), 15) / 15 : 0; // cap at 15%
|
||||||
|
const h = beat == null ? 2 : 3 + mag * 9; // px
|
||||||
|
const up = (beat ?? 0) >= 0;
|
||||||
|
return (
|
||||||
|
<span className="relative z-[1] block h-7 w-2"
|
||||||
|
title={`${e.announce_date}: ${surpriseLabel(e)}${s != null ? ` ${s > 0 ? '+' : ''}${s}%` : ''}`}>
|
||||||
|
<span className="absolute inset-x-0 rounded-[1px]" aria-hidden
|
||||||
|
style={{ height: h, background: toneColor(tone), ...(up ? { bottom: '50%' } : { top: '50%' }) }} />
|
||||||
|
</span>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,131 @@
|
|||||||
|
/* Dev-only visual harness for FundamentalsPanel. Served at /harness.html by
|
||||||
|
* `vite`. Not imported by the app. Renders the three key states so desktop and
|
||||||
|
* mobile can be eyeballed with representative fixtures. */
|
||||||
|
import { createRoot } from 'react-dom/client';
|
||||||
|
import '../styles/globals.css';
|
||||||
|
import { FundamentalsPanel } from '../components/ticker/FundamentalsPanel';
|
||||||
|
import type { FundamentalResponse, MetricItem } from '../lib/types';
|
||||||
|
|
||||||
|
function h(period: string, value: number | null) {
|
||||||
|
return { period_end: period, value };
|
||||||
|
}
|
||||||
|
const P = ['2025-06-30', '2025-09-30', '2025-12-31', '2026-03-28'];
|
||||||
|
|
||||||
|
function dateFromToday(days: number): string {
|
||||||
|
const date = new Date();
|
||||||
|
date.setHours(12, 0, 0, 0);
|
||||||
|
date.setDate(date.getDate() + days);
|
||||||
|
return [
|
||||||
|
date.getFullYear(),
|
||||||
|
String(date.getMonth() + 1).padStart(2, '0'),
|
||||||
|
String(date.getDate()).padStart(2, '0'),
|
||||||
|
].join('-');
|
||||||
|
}
|
||||||
|
|
||||||
|
function metric(key: string, value: number | null, hist: (number | null)[],
|
||||||
|
industry: MetricItem['industry'] = null): MetricItem {
|
||||||
|
return {
|
||||||
|
key: key as MetricItem['key'], value,
|
||||||
|
history: hist.map((v, i) => h(P[i], v)),
|
||||||
|
industry, period_end: '2026-03-28', filed_date: '2026-05-01', source: 'sec',
|
||||||
|
};
|
||||||
|
}
|
||||||
|
const ind = (median: number, favorable_percentile: number) =>
|
||||||
|
({ label: 'SIC 35 peers', median, favorable_percentile, peer_count: 12 });
|
||||||
|
|
||||||
|
const legacy = {
|
||||||
|
pe_ratio: null, revenue_growth: null, earnings_surprise: null, market_cap: null,
|
||||||
|
next_earnings_date: null, fetched_at: null, unavailable_fields: {},
|
||||||
|
};
|
||||||
|
|
||||||
|
const full: FundamentalResponse = {
|
||||||
|
symbol: 'AAPL', ...legacy,
|
||||||
|
earnings: {
|
||||||
|
next: { date: dateFromToday(12), session: 'amc', days_until: 12 },
|
||||||
|
recent: [
|
||||||
|
{ announce_date: '2025-08-01', period_end: '2025-06-30', eps_estimate: 1.4, eps_actual: 1.6, surprise_pct: 14.3 },
|
||||||
|
{ announce_date: '2025-11-01', period_end: '2025-09-30', eps_estimate: 1.7, eps_actual: 1.9, surprise_pct: 11.8 },
|
||||||
|
{ announce_date: '2026-02-01', period_end: '2025-12-31', eps_estimate: 2.6, eps_actual: 2.4, surprise_pct: -7.7 },
|
||||||
|
{ announce_date: '2026-05-01', period_end: '2026-03-28', eps_estimate: 1.5, eps_actual: 1.65, surprise_pct: 10.0 },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
metrics: [
|
||||||
|
metric('revenue_growth_yoy', 18, [8, 11, 15, 18], ind(11, 82)),
|
||||||
|
metric('eps_growth_yoy', 24, [10, 18, 22, 24], ind(15, 70)),
|
||||||
|
metric('operating_margin', 32, [30, 31, 31, 32], ind(22, 88)),
|
||||||
|
metric('fcf_margin', 28, [24, 25, 27, 28], ind(18, 80)),
|
||||||
|
metric('net_debt', 16.2e9, [46e9, 44e9, 24e9, 16.2e9], null),
|
||||||
|
metric('net_debt_to_ebitda', 1.4, [1.9, 1.7, 1.5, 1.4], ind(2.1, 68)),
|
||||||
|
metric('share_count_change_yoy', -1.7, [-2.4, -2.2, -2.3, -1.7], null),
|
||||||
|
],
|
||||||
|
valuation: {
|
||||||
|
pe: 29.2, fcf_yield: 3.8, market_cap_est: 3.2e12,
|
||||||
|
pe_industry: ind(23.5, 30), fcf_yield_industry: ind(3.1, 70), price_date: '2026-05-01',
|
||||||
|
},
|
||||||
|
reads: {
|
||||||
|
header: 'growth accelerating · margins improving · valuation priced above peers',
|
||||||
|
by_key: {
|
||||||
|
revenue_growth_yoy: 'accelerating', eps_growth_yoy: 'accelerating',
|
||||||
|
operating_margin: 'improving', fcf_margin: 'improving',
|
||||||
|
share_count_change_yoy: 'buying back', net_debt_to_ebitda: 'conservative leverage',
|
||||||
|
pe: 'priced above peers', fcf_yield: 'above peers', net_debt: null,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
const partial: FundamentalResponse = {
|
||||||
|
symbol: 'NEWCO', ...legacy,
|
||||||
|
earnings: { next: { date: dateFromToday(0), session: 'unknown', days_until: 0 }, recent: [] },
|
||||||
|
metrics: [
|
||||||
|
metric('revenue_growth_yoy', 12, [null, 8, 10, 12], null),
|
||||||
|
metric('eps_growth_yoy', null, [null, null, null, null], null),
|
||||||
|
metric('operating_margin', 25, [24, 24, 25, 25], null),
|
||||||
|
metric('fcf_margin', null, [null, null, null, null], null),
|
||||||
|
metric('net_debt', null, [], null),
|
||||||
|
metric('net_debt_to_ebitda', 1.9, [1.7, 1.8, 1.9, 1.9], null),
|
||||||
|
metric('share_count_change_yoy', 2.1, [1.8, 2.0, 2.0, 2.1], null),
|
||||||
|
],
|
||||||
|
valuation: {
|
||||||
|
pe: 15.2, fcf_yield: null, market_cap_est: 5.4e8,
|
||||||
|
pe_industry: null, fcf_yield_industry: null, price_date: '2026-05-01',
|
||||||
|
},
|
||||||
|
reads: {
|
||||||
|
header: 'growth steady · margins stable',
|
||||||
|
by_key: {
|
||||||
|
revenue_growth_yoy: 'steady', operating_margin: 'stable',
|
||||||
|
share_count_change_yoy: '2.1% dilution', net_debt_to_ebitda: null,
|
||||||
|
pe: null, fcf_yield: null, eps_growth_yoy: null, fcf_margin: null, net_debt: null,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
const empty: FundamentalResponse = {
|
||||||
|
symbol: 'ADR', ...legacy,
|
||||||
|
earnings: { next: null, recent: [] },
|
||||||
|
metrics: [
|
||||||
|
'revenue_growth_yoy', 'eps_growth_yoy', 'operating_margin', 'fcf_margin',
|
||||||
|
'net_debt', 'net_debt_to_ebitda', 'share_count_change_yoy',
|
||||||
|
].map((k) => metric(k, null, [])),
|
||||||
|
valuation: null,
|
||||||
|
reads: { header: null, by_key: {} },
|
||||||
|
};
|
||||||
|
|
||||||
|
function Case({ title, data }: { title: string; data: FundamentalResponse }) {
|
||||||
|
return (
|
||||||
|
<div>
|
||||||
|
<div className="mb-1.5 text-[11px] uppercase tracking-widest text-gray-500">{title}</div>
|
||||||
|
<FundamentalsPanel data={data} />
|
||||||
|
</div>
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
createRoot(document.getElementById('root')!).render(
|
||||||
|
<div className="mx-auto max-w-3xl space-y-8 p-6">
|
||||||
|
<p className="text-[11px] uppercase tracking-widest text-gray-500">
|
||||||
|
Desktop width (~768px, two columns). Resize the browser to ~390px to check mobile (single column).
|
||||||
|
</p>
|
||||||
|
<Case title="Full" data={full} />
|
||||||
|
<Case title="Partial · insufficient peers" data={partial} />
|
||||||
|
<Case title="Empty" data={empty} />
|
||||||
|
</div>,
|
||||||
|
);
|
||||||
@@ -191,6 +191,8 @@ export interface ActivationConfig {
|
|||||||
export interface ScheduleConfig {
|
export interface ScheduleConfig {
|
||||||
schedule_timezone: string;
|
schedule_timezone: string;
|
||||||
schedule_daily_pipeline_cron: string;
|
schedule_daily_pipeline_cron: string;
|
||||||
|
schedule_dolt_earnings_cron: string;
|
||||||
|
schedule_sec_fundamentals_cron: string;
|
||||||
schedule_near_close_pipeline_cron: string;
|
schedule_near_close_pipeline_cron: string;
|
||||||
schedule_after_close_pipeline_cron: string;
|
schedule_after_close_pipeline_cron: string;
|
||||||
schedule_intraday_pipeline_cron: string;
|
schedule_intraday_pipeline_cron: string;
|
||||||
@@ -703,8 +705,74 @@ export interface SentimentResponse {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Fundamentals
|
// Fundamentals
|
||||||
|
export interface MetricIndustry {
|
||||||
|
label: string;
|
||||||
|
median: number;
|
||||||
|
favorable_percentile: number; // 0-100, polarity-aware (higher = more favorable)
|
||||||
|
peer_count: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface MetricHistoryPoint {
|
||||||
|
period_end: string | null; // YYYY-MM-DD
|
||||||
|
value: number | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export type MetricKey =
|
||||||
|
| 'revenue_growth_yoy'
|
||||||
|
| 'eps_growth_yoy'
|
||||||
|
| 'operating_margin'
|
||||||
|
| 'fcf_margin'
|
||||||
|
| 'net_debt'
|
||||||
|
| 'net_debt_to_ebitda'
|
||||||
|
| 'share_count_change_yoy';
|
||||||
|
|
||||||
|
export interface MetricItem {
|
||||||
|
key: MetricKey;
|
||||||
|
value: number | null;
|
||||||
|
history: MetricHistoryPoint[];
|
||||||
|
industry: MetricIndustry | null;
|
||||||
|
period_end: string | null;
|
||||||
|
filed_date: string | null;
|
||||||
|
source: string; // 'sec' | 'legacy_api'
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface EarningsNext {
|
||||||
|
date: string;
|
||||||
|
session: string; // bmo | amc | unknown
|
||||||
|
days_until: number;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface EarningsRecent {
|
||||||
|
announce_date: string;
|
||||||
|
period_end: string | null;
|
||||||
|
eps_estimate: number | null;
|
||||||
|
eps_actual: number | null;
|
||||||
|
surprise_pct: number | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface EarningsObject {
|
||||||
|
next: EarningsNext | null;
|
||||||
|
recent: EarningsRecent[];
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface Valuation {
|
||||||
|
pe: number | null;
|
||||||
|
fcf_yield: number | null;
|
||||||
|
market_cap_est: number | null;
|
||||||
|
pe_industry: MetricIndustry | null;
|
||||||
|
fcf_yield_industry: MetricIndustry | null;
|
||||||
|
price_date: string | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface FundamentalsReads {
|
||||||
|
header: string | null;
|
||||||
|
// fixed map over every metric key plus 'pe' and 'fcf_yield'; null when unavailable
|
||||||
|
by_key: Record<string, string | null>;
|
||||||
|
}
|
||||||
|
|
||||||
export interface FundamentalResponse {
|
export interface FundamentalResponse {
|
||||||
symbol: string;
|
symbol: string;
|
||||||
|
// legacy fields (unchanged)
|
||||||
pe_ratio: number | null;
|
pe_ratio: number | null;
|
||||||
revenue_growth: number | null;
|
revenue_growth: number | null;
|
||||||
earnings_surprise: number | null;
|
earnings_surprise: number | null;
|
||||||
@@ -712,6 +780,11 @@ export interface FundamentalResponse {
|
|||||||
next_earnings_date: string | null;
|
next_earnings_date: string | null;
|
||||||
fetched_at: string | null;
|
fetched_at: string | null;
|
||||||
unavailable_fields: Record<string, string>;
|
unavailable_fields: Record<string, string>;
|
||||||
|
// additive v1 (SEC/Dolt-derived) — present, with null/empty when unavailable
|
||||||
|
earnings: EarningsObject | null;
|
||||||
|
metrics: MetricItem[] | null;
|
||||||
|
valuation: Valuation | null;
|
||||||
|
reads: FundamentalsReads | null;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Indicators
|
// Indicators
|
||||||
|
|||||||
@@ -0,0 +1,241 @@
|
|||||||
|
"""Orchestration tests for the source-agnostic import framework.
|
||||||
|
|
||||||
|
These drive ``run_import`` with a fake importer to prove the framework's
|
||||||
|
guarantees: idempotent no_op on an unchanged revision, atomic promotion on a
|
||||||
|
new revision, and — the load-bearing one — a failed validation or a mid-run
|
||||||
|
exception leaves the live tables untouched.
|
||||||
|
|
||||||
|
The advisory-lock branch is a no-op on SQLite, so lock mutual-exclusion has NO
|
||||||
|
coverage here (PG-verify-pending); only the deterministic key derivation is
|
||||||
|
unit-tested. Per the SQLite StaticPool caveat we never share a connection: each
|
||||||
|
test uses its own temp-file engine and seeds/asserts with short-lived sessions
|
||||||
|
sequenced around the ``run_import`` call.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
import app.models # noqa: F401 register models on Base.metadata
|
||||||
|
from app.models.data_import_run import DataImportRun
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.system_event import SystemEvent
|
||||||
|
from app.services.data_import import (
|
||||||
|
STATUS_FAILED,
|
||||||
|
STATUS_NO_OP,
|
||||||
|
STATUS_PROMOTED,
|
||||||
|
ValidationResult,
|
||||||
|
_advisory_key,
|
||||||
|
run_import,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
async def engine():
|
||||||
|
"""A dedicated temp-file SQLite engine (independent connections, unlike the
|
||||||
|
shared in-memory test engine) so ``run_import`` can pin its own connection."""
|
||||||
|
fd, path = tempfile.mkstemp(suffix=".db")
|
||||||
|
os.close(fd)
|
||||||
|
eng = create_async_engine(f"sqlite+aiosqlite:///{path}")
|
||||||
|
async with eng.begin() as conn:
|
||||||
|
await conn.run_sync(Base.metadata.create_all)
|
||||||
|
try:
|
||||||
|
yield eng
|
||||||
|
finally:
|
||||||
|
await eng.dispose()
|
||||||
|
try:
|
||||||
|
os.unlink(path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _factory(engine) -> async_sessionmaker[AsyncSession]:
|
||||||
|
return async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
|
||||||
|
|
||||||
|
class FakeImporter:
|
||||||
|
"""Writes ``n_rows`` fundamental_snapshots on promote (CIK-keyed, no ticker
|
||||||
|
FK) so the live-table effect is easy to count."""
|
||||||
|
|
||||||
|
source = "sec_facts"
|
||||||
|
|
||||||
|
def __init__(self, revision, *, ok=True, n_rows=3, raise_in="none"):
|
||||||
|
self.revision = revision
|
||||||
|
self.ok = ok
|
||||||
|
self.n_rows = n_rows
|
||||||
|
self.raise_in = raise_in
|
||||||
|
self.staged_called = False
|
||||||
|
self.promoted = False
|
||||||
|
|
||||||
|
async def detect_revision(self, db):
|
||||||
|
if self.raise_in == "detect":
|
||||||
|
raise RuntimeError("boom-detect")
|
||||||
|
return self.revision
|
||||||
|
|
||||||
|
async def stage(self, db):
|
||||||
|
self.staged_called = True
|
||||||
|
if self.raise_in == "stage":
|
||||||
|
raise RuntimeError("boom-stage")
|
||||||
|
if self.raise_in == "cancel":
|
||||||
|
raise asyncio.CancelledError()
|
||||||
|
return list(range(self.n_rows))
|
||||||
|
|
||||||
|
async def validate(self, db, staged):
|
||||||
|
return ValidationResult(
|
||||||
|
ok=self.ok,
|
||||||
|
summary={"staged_rows": len(staged)},
|
||||||
|
source_max_date=date(2026, 7, 21),
|
||||||
|
messages=[] if self.ok else ["coverage below threshold"],
|
||||||
|
)
|
||||||
|
|
||||||
|
async def promote(self, db, staged, run_id):
|
||||||
|
if self.raise_in == "promote":
|
||||||
|
# write one row THEN raise, to prove rollback undoes partial writes
|
||||||
|
db.add(_snapshot(self.revision, 999))
|
||||||
|
raise RuntimeError("boom-promote")
|
||||||
|
for i in staged:
|
||||||
|
db.add(_snapshot(self.revision, i, run_id=run_id))
|
||||||
|
self.promoted = True
|
||||||
|
self.promoted_run_id = run_id
|
||||||
|
return {"fundamental_snapshots": len(staged)}
|
||||||
|
|
||||||
|
|
||||||
|
def _snapshot(revision: str, i: int, run_id: int | None = None) -> FundamentalSnapshot:
|
||||||
|
return FundamentalSnapshot(
|
||||||
|
cik=f"{i:010d}",
|
||||||
|
accession=f"{revision}-{i:06d}",
|
||||||
|
form="10-Q",
|
||||||
|
filed_date=date(2026, 7, 1),
|
||||||
|
accepted_at=datetime(2026, 7, 1, tzinfo=timezone.utc),
|
||||||
|
period_end=date(2026, 6, 30),
|
||||||
|
fiscal_year=2026,
|
||||||
|
fiscal_period="Q2",
|
||||||
|
revenue=1000.0 + i,
|
||||||
|
import_run_id=run_id,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def _count(factory, model) -> int:
|
||||||
|
async with factory() as s:
|
||||||
|
return (await s.execute(select(func.count()).select_from(model))).scalar_one()
|
||||||
|
|
||||||
|
|
||||||
|
async def _runs(factory) -> list[DataImportRun]:
|
||||||
|
async with factory() as s:
|
||||||
|
return list(
|
||||||
|
(await s.execute(select(DataImportRun).order_by(DataImportRun.id))).scalars()
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_advisory_key_deterministic_and_distinct():
|
||||||
|
assert _advisory_key("sec_facts") == _advisory_key("sec_facts")
|
||||||
|
assert _advisory_key("sec_facts") != _advisory_key("dolt_earnings")
|
||||||
|
for src in ("sec_facts", "dolt_earnings", "dolt_stocks"):
|
||||||
|
k = _advisory_key(src)
|
||||||
|
assert -(2**63) <= k < 2**63 # fits Postgres bigint
|
||||||
|
|
||||||
|
|
||||||
|
async def test_promote_writes_and_records_run(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
run = await run_import(FakeImporter("rev1", n_rows=4), engine=engine)
|
||||||
|
|
||||||
|
assert run is not None and run.status == STATUS_PROMOTED
|
||||||
|
assert run.revision == "rev1"
|
||||||
|
assert run.row_counts_json is not None and "fundamental_snapshots" in run.row_counts_json
|
||||||
|
assert run.source_max_date == date(2026, 7, 21)
|
||||||
|
assert run.completed_at is not None
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 4
|
||||||
|
# rows stamped with the run id
|
||||||
|
async with factory() as s:
|
||||||
|
stamped = (
|
||||||
|
await s.execute(select(FundamentalSnapshot.import_run_id))
|
||||||
|
).scalars().all()
|
||||||
|
assert stamped and all(rid == run.id for rid in stamped)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_no_op_on_repeated_revision(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await run_import(FakeImporter("rev1", n_rows=4), engine=engine)
|
||||||
|
|
||||||
|
second = FakeImporter("rev1", n_rows=4)
|
||||||
|
run = await run_import(second, engine=engine)
|
||||||
|
|
||||||
|
assert run is not None and run.status == STATUS_NO_OP
|
||||||
|
assert second.staged_called is False # never fetched
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 4 # unchanged
|
||||||
|
|
||||||
|
runs = await _runs(factory)
|
||||||
|
assert [r.status for r in runs] == [STATUS_PROMOTED, STATUS_NO_OP]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_new_revision_after_promote_stages_again(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await run_import(FakeImporter("rev1", n_rows=2), engine=engine)
|
||||||
|
run = await run_import(FakeImporter("rev2", n_rows=3), engine=engine)
|
||||||
|
|
||||||
|
assert run is not None and run.status == STATUS_PROMOTED
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 5 # 2 + 3
|
||||||
|
|
||||||
|
|
||||||
|
async def test_failed_validation_leaves_data_untouched(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await run_import(FakeImporter("rev1", n_rows=3), engine=engine) # baseline
|
||||||
|
|
||||||
|
run = await run_import(FakeImporter("rev2", ok=False, n_rows=5), engine=engine)
|
||||||
|
|
||||||
|
assert run is not None and run.status == STATUS_FAILED
|
||||||
|
assert "coverage" in (run.error_details or "")
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 3 # untouched
|
||||||
|
assert await _count(factory, SystemEvent) == 1 # alerted
|
||||||
|
|
||||||
|
|
||||||
|
async def test_exception_in_promote_rolls_back(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await run_import(FakeImporter("rev1", n_rows=3), engine=engine) # baseline
|
||||||
|
|
||||||
|
run = await run_import(FakeImporter("rev2", raise_in="promote"), engine=engine)
|
||||||
|
|
||||||
|
assert run is not None and run.status == STATUS_FAILED
|
||||||
|
assert "boom-promote" in (run.error_details or "")
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 3 # partial write rolled back
|
||||||
|
assert await _count(factory, SystemEvent) == 1
|
||||||
|
|
||||||
|
|
||||||
|
async def test_detection_failure_records_and_alerts(engine):
|
||||||
|
"""The external revision probe is the most likely failure — it must produce a
|
||||||
|
recorded failed run + alert, not an unrecorded escaping exception."""
|
||||||
|
factory = _factory(engine)
|
||||||
|
imp = FakeImporter("rev1", raise_in="detect")
|
||||||
|
|
||||||
|
run = await run_import(imp, engine=engine)
|
||||||
|
|
||||||
|
assert run is not None and run.status == STATUS_FAILED
|
||||||
|
assert "boom-detect" in (run.error_details or "")
|
||||||
|
assert imp.staged_called is False
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 0
|
||||||
|
assert await _count(factory, SystemEvent) == 1 # alerted
|
||||||
|
runs = await _runs(factory)
|
||||||
|
assert len(runs) == 1 and runs[0].status == STATUS_FAILED # attempt recorded
|
||||||
|
|
||||||
|
|
||||||
|
async def test_cancellation_marks_failed_and_reraises(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
|
||||||
|
with pytest.raises(asyncio.CancelledError):
|
||||||
|
await run_import(FakeImporter("rev1", raise_in="cancel"), engine=engine)
|
||||||
|
|
||||||
|
runs = await _runs(factory)
|
||||||
|
assert len(runs) == 1 and runs[0].status == STATUS_FAILED # no lingering running row
|
||||||
|
assert runs[0].error_details == "cancelled"
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 0
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
"""Tests for the async dolt subprocess wrapper's failure handling.
|
||||||
|
|
||||||
|
Uses the Python interpreter as a stand-in subprocess (cross-platform, no dolt
|
||||||
|
needed) to prove a non-zero exit and a hung command both raise DoltError — the
|
||||||
|
latter is what stops a hung pull from pinning the import connection + advisory
|
||||||
|
lock forever.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from app.services import dolt_client
|
||||||
|
from app.services.dolt_client import DoltError
|
||||||
|
|
||||||
|
|
||||||
|
async def test_run_raises_on_nonzero_exit():
|
||||||
|
with pytest.raises(DoltError) as exc:
|
||||||
|
await dolt_client._run(
|
||||||
|
sys.executable, ["-c", "import sys; sys.exit(3)"], cwd=Path.cwd(), timeout=30
|
||||||
|
)
|
||||||
|
assert "3" in str(exc.value)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_run_times_out_and_kills():
|
||||||
|
start = time.monotonic()
|
||||||
|
with pytest.raises(DoltError) as exc:
|
||||||
|
await dolt_client._run(
|
||||||
|
sys.executable,
|
||||||
|
["-c", "import time; time.sleep(30)"],
|
||||||
|
cwd=Path.cwd(),
|
||||||
|
timeout=0.5,
|
||||||
|
)
|
||||||
|
elapsed = time.monotonic() - start
|
||||||
|
assert "timed out" in str(exc.value)
|
||||||
|
assert elapsed < 10 # killed promptly, not waited out
|
||||||
|
|
||||||
|
|
||||||
|
async def test_run_returns_stdout_on_success():
|
||||||
|
out = await dolt_client._run(
|
||||||
|
sys.executable, ["-c", "print('hello')"], cwd=Path.cwd(), timeout=30
|
||||||
|
)
|
||||||
|
assert out.strip() == "hello"
|
||||||
@@ -0,0 +1,264 @@
|
|||||||
|
"""Integration tests for the DoltHub earnings importer, driven through the real
|
||||||
|
import framework with a fake dolt client (no subprocess, no clone).
|
||||||
|
|
||||||
|
Covers the load-bearing behaviors: symbol-normalized join to the tracked
|
||||||
|
universe, calendar<->history pairing (matched → EPS, unmatched → null), the
|
||||||
|
destructive-but-safe reschedule/cancel promotion, past rows never deleted, and
|
||||||
|
the fail-closed forward-calendar gates that guard the destructive promote.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
import tempfile
|
||||||
|
from datetime import date
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from sqlalchemy import select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
import app.models # noqa: F401
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services.data_import import STATUS_FAILED, STATUS_PROMOTED, run_import
|
||||||
|
from app.services.dolt_earnings_importer import DoltEarningsImporter
|
||||||
|
|
||||||
|
TODAY = date(2026, 7, 22)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
async def engine():
|
||||||
|
fd, path = tempfile.mkstemp(suffix=".db")
|
||||||
|
os.close(fd)
|
||||||
|
eng = create_async_engine(f"sqlite+aiosqlite:///{path}")
|
||||||
|
async with eng.begin() as conn:
|
||||||
|
await conn.run_sync(Base.metadata.create_all)
|
||||||
|
try:
|
||||||
|
yield eng
|
||||||
|
finally:
|
||||||
|
await eng.dispose()
|
||||||
|
try:
|
||||||
|
os.unlink(path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _factory(engine):
|
||||||
|
return async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
|
||||||
|
|
||||||
|
class FakeDolt:
|
||||||
|
def __init__(self, calendar, history, commit="c1"):
|
||||||
|
self.calendar = calendar
|
||||||
|
self.history = history
|
||||||
|
self.commit = commit
|
||||||
|
self.pulled = False
|
||||||
|
|
||||||
|
async def pull(self, repo_dir, *, binary, timeout=None):
|
||||||
|
self.pulled = True
|
||||||
|
|
||||||
|
async def current_commit(self, repo_dir, *, binary, timeout=None):
|
||||||
|
return self.commit
|
||||||
|
|
||||||
|
async def query_csv(self, repo_dir, sql, *, binary, timeout=None):
|
||||||
|
if "earnings_calendar" in sql:
|
||||||
|
return self.calendar
|
||||||
|
if "eps_history" in sql:
|
||||||
|
return self.history
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def _cal(symbol, d, when="After market close"):
|
||||||
|
return {"act_symbol": symbol, "date": d, "when": when}
|
||||||
|
|
||||||
|
|
||||||
|
def _hist(symbol, pe, reported, estimate):
|
||||||
|
return {"act_symbol": symbol, "period_end_date": pe, "reported": str(reported), "estimate": str(estimate)}
|
||||||
|
|
||||||
|
|
||||||
|
def _importer(fake, commit=None):
|
||||||
|
if commit:
|
||||||
|
fake.commit = commit
|
||||||
|
return DoltEarningsImporter(
|
||||||
|
repo_dir="unused", binary="unused", today=TODAY, do_pull=False, dolt=fake
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def _seed_tickers(factory, symbols):
|
||||||
|
async with factory() as s:
|
||||||
|
for sym in symbols:
|
||||||
|
s.add(Ticker(symbol=sym))
|
||||||
|
await s.commit()
|
||||||
|
async with factory() as s:
|
||||||
|
return {sym: tid for tid, sym in (await s.execute(select(Ticker.id, Ticker.symbol))).all()}
|
||||||
|
|
||||||
|
|
||||||
|
async def _events(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
rows = (
|
||||||
|
await s.execute(select(EarningsEvent).order_by(EarningsEvent.announce_date))
|
||||||
|
).scalars().all()
|
||||||
|
return list(rows)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
async def test_stage_and_promote_basic(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed_tickers(factory, ["AAPL"])
|
||||||
|
fake = FakeDolt(
|
||||||
|
calendar=[_cal("AAPL", "2026-05-01"), _cal("AAPL", "2026-08-20")],
|
||||||
|
history=[_hist("AAPL", "2026-03-31", 1.5, 1.4)], # only the reported quarter
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(fake), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_PROMOTED
|
||||||
|
assert run.source_max_date == date(2026, 8, 20)
|
||||||
|
|
||||||
|
events = await _events(factory)
|
||||||
|
assert len(events) == 2
|
||||||
|
past = next(e for e in events if e.announce_date == date(2026, 5, 1))
|
||||||
|
future = next(e for e in events if e.announce_date == date(2026, 8, 20))
|
||||||
|
# past announcement paired to the reported quarter
|
||||||
|
assert past.eps_actual == 1.5 and past.eps_estimate == 1.4
|
||||||
|
assert past.period_end == date(2026, 3, 31) and past.session == "amc"
|
||||||
|
assert past.import_run_id == run.id
|
||||||
|
# future announcement has no results yet → null EPS/period, session kept
|
||||||
|
assert future.eps_actual is None and future.period_end is None
|
||||||
|
assert future.session == "amc"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_symbol_normalisation_join(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
ids = await _seed_tickers(factory, ["AAPL", "BRK.B"])
|
||||||
|
fake = FakeDolt(
|
||||||
|
calendar=[_cal("AAPL", "2026-08-20"), _cal("BRK.B", "2026-08-25")], # dotted source symbol
|
||||||
|
history=[],
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(fake), engine=engine)
|
||||||
|
assert run.status == STATUS_PROMOTED
|
||||||
|
|
||||||
|
events = await _events(factory)
|
||||||
|
mapped = {e.ticker_id for e in events}
|
||||||
|
assert mapped == {ids["AAPL"], ids["BRK.B"]} # dotted BRK.B joined via normalization
|
||||||
|
|
||||||
|
|
||||||
|
async def test_reschedule_moves_future_row(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed_tickers(factory, ["AAPL"])
|
||||||
|
|
||||||
|
fake1 = FakeDolt(calendar=[_cal("AAPL", "2026-08-20")], history=[], commit="c1")
|
||||||
|
await run_import(_importer(fake1), engine=engine)
|
||||||
|
|
||||||
|
fake2 = FakeDolt(calendar=[_cal("AAPL", "2026-08-27")], history=[], commit="c2") # moved
|
||||||
|
run2 = await run_import(_importer(fake2), engine=engine)
|
||||||
|
assert run2.status == STATUS_PROMOTED
|
||||||
|
|
||||||
|
dates = {e.announce_date for e in await _events(factory)}
|
||||||
|
assert dates == {date(2026, 8, 27)} # old future date gone, new one present
|
||||||
|
|
||||||
|
|
||||||
|
async def test_cancellation_removes_future_row(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed_tickers(factory, ["AAPL"])
|
||||||
|
|
||||||
|
fake1 = FakeDolt(
|
||||||
|
calendar=[_cal("AAPL", "2026-08-01"), _cal("AAPL", "2026-08-15")], history=[], commit="c1"
|
||||||
|
)
|
||||||
|
await run_import(_importer(fake1), engine=engine)
|
||||||
|
|
||||||
|
fake2 = FakeDolt(calendar=[_cal("AAPL", "2026-08-15")], history=[], commit="c2") # 08-01 cancelled
|
||||||
|
await run_import(_importer(fake2), engine=engine)
|
||||||
|
|
||||||
|
dates = {e.announce_date for e in await _events(factory)}
|
||||||
|
assert dates == {date(2026, 8, 15)}
|
||||||
|
|
||||||
|
|
||||||
|
async def test_past_row_never_deleted(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed_tickers(factory, ["AAPL"])
|
||||||
|
|
||||||
|
fake1 = FakeDolt(
|
||||||
|
calendar=[_cal("AAPL", "2026-05-01"), _cal("AAPL", "2026-08-20")], history=[], commit="c1"
|
||||||
|
)
|
||||||
|
await run_import(_importer(fake1), engine=engine)
|
||||||
|
|
||||||
|
# Second import's calendar omits the past date but keeps a future one.
|
||||||
|
fake2 = FakeDolt(calendar=[_cal("AAPL", "2026-08-20")], history=[], commit="c2")
|
||||||
|
await run_import(_importer(fake2), engine=engine)
|
||||||
|
|
||||||
|
dates = {e.announce_date for e in await _events(factory)}
|
||||||
|
assert date(2026, 5, 1) in dates # past result survived
|
||||||
|
assert date(2026, 8, 20) in dates
|
||||||
|
|
||||||
|
|
||||||
|
async def test_validate_fails_when_no_future(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed_tickers(factory, ["AAPL"])
|
||||||
|
fake = FakeDolt(calendar=[_cal("AAPL", "2026-05-01")], history=[]) # only past
|
||||||
|
|
||||||
|
run = await run_import(_importer(fake), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_FAILED
|
||||||
|
assert "future" in (run.error_details or "")
|
||||||
|
assert len(await _events(factory)) == 0 # nothing promoted
|
||||||
|
|
||||||
|
|
||||||
|
async def test_validate_fails_on_forward_collapse(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed_tickers(factory, ["AAPL", "MSFT", "NVDA", "AMZN"])
|
||||||
|
|
||||||
|
fake1 = FakeDolt(
|
||||||
|
calendar=[
|
||||||
|
_cal("AAPL", "2026-08-20"),
|
||||||
|
_cal("MSFT", "2026-08-21"),
|
||||||
|
_cal("NVDA", "2026-08-22"),
|
||||||
|
_cal("AMZN", "2026-08-23"),
|
||||||
|
],
|
||||||
|
history=[],
|
||||||
|
commit="c1",
|
||||||
|
)
|
||||||
|
await run_import(_importer(fake1), engine=engine)
|
||||||
|
assert len([e for e in await _events(factory) if e.announce_date > TODAY]) == 4
|
||||||
|
|
||||||
|
# Only one future row now → 1 < 50% of 4 → fail-closed, no destructive wipe.
|
||||||
|
fake2 = FakeDolt(calendar=[_cal("AAPL", "2026-08-20")], history=[], commit="c2")
|
||||||
|
run2 = await run_import(_importer(fake2), engine=engine)
|
||||||
|
|
||||||
|
assert run2.status == STATUS_FAILED
|
||||||
|
assert "collapsed" in (run2.error_details or "")
|
||||||
|
assert len([e for e in await _events(factory) if e.announce_date > TODAY]) == 4 # preserved
|
||||||
|
|
||||||
|
|
||||||
|
# --- Real-clone smoke test: exercises the actual dolt subprocess + parse + align
|
||||||
|
# against the local clone. Skips in CI / anywhere the binary or clone is absent.
|
||||||
|
|
||||||
|
_DOLT_BIN = os.environ.get("DOLT_BINARY") or shutil.which("dolt") or r"C:\Program Files\Dolt\bin\dolt.exe"
|
||||||
|
_CLONE_DIR = Path("dolt-data/earnings")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(
|
||||||
|
not (Path(_DOLT_BIN).exists() and _CLONE_DIR.exists()),
|
||||||
|
reason="real dolt binary / earnings clone not available",
|
||||||
|
)
|
||||||
|
async def test_real_clone_smoke(engine):
|
||||||
|
from app.services import dolt_client
|
||||||
|
|
||||||
|
factory = _factory(engine)
|
||||||
|
# A few tickers spanning near + further-out reporters so the initial-load
|
||||||
|
# forward-horizon gate (>= 21d) is satisfied on the fixed clone.
|
||||||
|
await _seed_tickers(factory, ["AAPL", "MSFT", "NVDA", "JPM", "BRK.B"])
|
||||||
|
imp = DoltEarningsImporter(
|
||||||
|
repo_dir=_CLONE_DIR, binary=_DOLT_BIN, today=date.today(), do_pull=False, dolt=dolt_client
|
||||||
|
)
|
||||||
|
run = await run_import(imp, engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_PROMOTED
|
||||||
|
events = await _events(factory)
|
||||||
|
assert events, "no earnings parsed from the real clone"
|
||||||
|
assert any(e.announce_date > date.today() for e in events), "no forward calendar"
|
||||||
|
assert any(e.eps_actual is not None for e in events), "no calendar<->history pairing"
|
||||||
@@ -0,0 +1,104 @@
|
|||||||
|
"""Unit tests for the pure calendar<->EPS-history alignment.
|
||||||
|
|
||||||
|
Anchored on the research script's exact constants (SKIP costs = 45, typical lag
|
||||||
|
= 30, session penalty = 3, windows 90/14) so a silently changed constant fails
|
||||||
|
here rather than quietly corrupting surprise-history pairing.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from datetime import date
|
||||||
|
|
||||||
|
from app.services import earnings_alignment as ea
|
||||||
|
|
||||||
|
|
||||||
|
def test_normalise_symbol():
|
||||||
|
assert ea.normalise_symbol("bf.b ") == "BF-B"
|
||||||
|
assert ea.normalise_symbol(" aapl") == "AAPL"
|
||||||
|
assert ea.normalise_symbol(None) == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_normalise_session():
|
||||||
|
assert ea.normalise_session("Before market open") == "bmo"
|
||||||
|
assert ea.normalise_session("After market close") == "amc"
|
||||||
|
assert ea.normalise_session("During market hours") == "unknown"
|
||||||
|
assert ea.normalise_session(None) == "unknown"
|
||||||
|
assert ea.normalise_session("") == "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def test_safe_number():
|
||||||
|
assert ea.safe_number("1.5") == 1.5
|
||||||
|
assert ea.safe_number("") is None
|
||||||
|
assert ea.safe_number("not-a-number") is None
|
||||||
|
assert ea.safe_number("nan") is None # non-finite rejected
|
||||||
|
|
||||||
|
|
||||||
|
def test_constants_pinned():
|
||||||
|
assert ea.SKIP_EVENT_COST == 45.0
|
||||||
|
assert ea.SKIP_PERIOD_COST == 45.0
|
||||||
|
assert ea._TYPICAL_ANNOUNCE_LAG_DAYS == 30
|
||||||
|
assert ea._MISSING_SESSION_PENALTY == 3.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_match_cost_uses_pinned_lag_and_penalty():
|
||||||
|
period = {"period_end_date": date(2026, 3, 31)}
|
||||||
|
# delta == 30 (typical lag) → base cost 0; known session → no penalty
|
||||||
|
e_known = {"announce_date": date(2026, 4, 30), "session": "amc"}
|
||||||
|
assert ea.match_cost(e_known, period) == 0.0
|
||||||
|
# unknown session adds the penalty
|
||||||
|
e_unknown = {"announce_date": date(2026, 4, 30), "session": "unknown"}
|
||||||
|
assert ea.match_cost(e_unknown, period) == 3.0
|
||||||
|
# delta 45 → |45-30| == 15
|
||||||
|
e_far = {"announce_date": date(2026, 5, 15), "session": "amc"}
|
||||||
|
assert ea.match_cost(e_far, period) == 15.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_dedup_calendar_prefers_known_session():
|
||||||
|
rows = [
|
||||||
|
{"symbol": "AAPL", "announce_date": date(2026, 5, 1), "session": "unknown"},
|
||||||
|
{"symbol": "AAPL", "announce_date": date(2026, 5, 1), "session": "amc"},
|
||||||
|
]
|
||||||
|
grouped, stats = ea.dedup_calendar(rows)
|
||||||
|
assert stats["duplicate_rows"] == 1
|
||||||
|
assert grouped["AAPL"][0]["session"] == "amc"
|
||||||
|
|
||||||
|
|
||||||
|
def test_dedup_history_prefers_fuller_row():
|
||||||
|
rows = [
|
||||||
|
{"symbol": "AAPL", "period_end_date": date(2026, 3, 31), "eps_actual": 2.0, "eps_estimate": None},
|
||||||
|
{"symbol": "AAPL", "period_end_date": date(2026, 3, 31), "eps_actual": 2.0, "eps_estimate": 1.9},
|
||||||
|
]
|
||||||
|
grouped, stats = ea.dedup_history(rows)
|
||||||
|
assert stats["duplicate_rows"] == 1
|
||||||
|
kept = grouped["AAPL"][0]
|
||||||
|
assert kept["eps_actual"] == 2.0 and kept["eps_estimate"] == 1.9
|
||||||
|
|
||||||
|
|
||||||
|
def _events(*days):
|
||||||
|
return [{"announce_date": d, "session": "amc"} for d in days]
|
||||||
|
|
||||||
|
|
||||||
|
def _periods(*days):
|
||||||
|
return [{"period_end_date": d, "eps_actual": 1.0, "eps_estimate": 0.9} for d in days]
|
||||||
|
|
||||||
|
|
||||||
|
def test_align_matches_monotonic_pairs():
|
||||||
|
# two announcements ~30d after two quarter ends
|
||||||
|
events = _events(date(2026, 4, 30), date(2026, 7, 30))
|
||||||
|
periods = _periods(date(2026, 3, 31), date(2026, 6, 30))
|
||||||
|
matches, um_events, um_periods = ea.align_symbol(
|
||||||
|
events, periods, max_lag_days=90, max_lead_days=14
|
||||||
|
)
|
||||||
|
assert matches == [(0, 0), (1, 1)]
|
||||||
|
assert um_events == [] and um_periods == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_align_leaves_out_of_window_unmatched():
|
||||||
|
# announcement 200 days after the only period end → outside the 90d window
|
||||||
|
events = _events(date(2026, 10, 17))
|
||||||
|
periods = _periods(date(2026, 3, 31))
|
||||||
|
matches, um_events, um_periods = ea.align_symbol(
|
||||||
|
events, periods, max_lag_days=90, max_lead_days=14
|
||||||
|
)
|
||||||
|
assert matches == []
|
||||||
|
assert um_events == [0] and um_periods == [0]
|
||||||
@@ -0,0 +1,252 @@
|
|||||||
|
"""Integration tests for the additive fundamentals API v1 assembly."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
import app.models # noqa: F401
|
||||||
|
from app.models.earnings_event import EarningsEvent
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.ohlcv import OHLCVRecord
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.schemas.fundamental import FundamentalResponse
|
||||||
|
from app.services.fundamentals_api_service import METRIC_KEYS, build_fundamentals_v1
|
||||||
|
|
||||||
|
UTC = timezone.utc
|
||||||
|
TODAY = date(2026, 10, 15)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
async def factory():
|
||||||
|
fd, path = tempfile.mkstemp(suffix=".db")
|
||||||
|
os.close(fd)
|
||||||
|
eng = create_async_engine(f"sqlite+aiosqlite:///{path}")
|
||||||
|
async with eng.begin() as conn:
|
||||||
|
await conn.run_sync(Base.metadata.create_all)
|
||||||
|
try:
|
||||||
|
yield async_sessionmaker(eng, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
finally:
|
||||||
|
await eng.dispose()
|
||||||
|
try:
|
||||||
|
os.unlink(path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
_MONTHS = [3, 6, 9, 12]
|
||||||
|
_FP = ["Q1", "Q2", "Q3", "FY"]
|
||||||
|
|
||||||
|
|
||||||
|
async def _seed_issuer(s, symbol, cik, sic, rev_base, price, *, eps_base=1.0, snapshots=True):
|
||||||
|
t = Ticker(symbol=symbol, cik=cik, sic=sic)
|
||||||
|
s.add(t)
|
||||||
|
await s.flush()
|
||||||
|
if snapshots:
|
||||||
|
# three fiscal years so YoY growth reads have a >=3 consecutive run
|
||||||
|
for fy, mult in [(2024, 0.9), (2025, 1.0), (2026, 1.1)]:
|
||||||
|
shares = {2024: 1050, 2025: 1000, 2026: 950}[fy] # steady buyback
|
||||||
|
rev = [rev_base * mult * x for x in (1.0, 1.05, 1.1, 1.15)]
|
||||||
|
eps = [eps_base * mult * x for x in (1.0, 1.05, 1.1, 1.15)]
|
||||||
|
for i, fp in enumerate(_FP):
|
||||||
|
pe = date(fy, _MONTHS[i], 28)
|
||||||
|
s.add(FundamentalSnapshot(
|
||||||
|
cik=cik, accession=f"{cik}-{fy}-{fp}", form="10-K" if fp == "FY" else "10-Q",
|
||||||
|
filed_date=pe, accepted_at=datetime(fy, _MONTHS[i], 28, tzinfo=UTC),
|
||||||
|
period_end=pe, fiscal_year=fy, fiscal_period=fp,
|
||||||
|
revenue=sum(rev[: i + 1]), operating_income=sum(rev[: i + 1]) * 0.2,
|
||||||
|
diluted_eps=sum(eps[: i + 1]), cfo=sum(rev[: i + 1]) * 0.25,
|
||||||
|
capex=sum(rev[: i + 1]) * 0.05, depreciation_amortization=sum(rev[: i + 1]) * 0.05,
|
||||||
|
cash_and_st_investments=40, total_debt=100, shares_outstanding=shares))
|
||||||
|
s.add(OHLCVRecord(ticker_id=t.id, date=date(2026, 10, 1), open=price, high=price, low=price, close=price, volume=1000))
|
||||||
|
return t.id
|
||||||
|
|
||||||
|
|
||||||
|
async def _seed_group(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
aapl = await _seed_issuer(s, "AAPL", "0000000001", "3571", rev_base=1000, price=200, eps_base=2.0)
|
||||||
|
for i in range(5): # 5 peers in SIC 35xx so the group has >= 5 valid issuers
|
||||||
|
await _seed_issuer(s, f"PEER{i}", f"000000010{i}", "3572", rev_base=500 + i * 100, price=50 + i * 10)
|
||||||
|
# AAPL earnings: one upcoming, one past with a surprise
|
||||||
|
s.add(EarningsEvent(ticker_id=aapl, announce_date=date(2026, 11, 1), session="amc", source="dolt_earnings"))
|
||||||
|
s.add(EarningsEvent(ticker_id=aapl, announce_date=date(2026, 8, 1), session="amc",
|
||||||
|
period_end=date(2026, 6, 30), eps_estimate=2.0, eps_actual=2.2, source="dolt_earnings"))
|
||||||
|
await s.commit()
|
||||||
|
return aapl
|
||||||
|
|
||||||
|
|
||||||
|
async def test_full_assembly(factory):
|
||||||
|
await _seed_group(factory)
|
||||||
|
async with factory() as s:
|
||||||
|
v1 = await build_fundamentals_v1(s, "AAPL", today=TODAY)
|
||||||
|
|
||||||
|
# earnings
|
||||||
|
assert v1["earnings"]["next"] == {"date": "2026-11-01", "session": "amc", "days_until": 17}
|
||||||
|
recent = v1["earnings"]["recent"]
|
||||||
|
assert recent and recent[0]["surprise_pct"] == pytest.approx(10.0)
|
||||||
|
|
||||||
|
# metrics — fixed key set, all present
|
||||||
|
assert [m["key"] for m in v1["metrics"]] == list(METRIC_KEYS)
|
||||||
|
by_key = {m["key"]: m for m in v1["metrics"]}
|
||||||
|
assert by_key["revenue_growth_yoy"]["value"] is not None
|
||||||
|
assert by_key["revenue_growth_yoy"]["source"] == "sec"
|
||||||
|
assert len(by_key["operating_margin"]["history"]) >= 3
|
||||||
|
# peer industry present for eligible metric (6 issuers), absent for size-dependent net_debt
|
||||||
|
assert by_key["operating_margin"]["industry"] is not None
|
||||||
|
assert by_key["operating_margin"]["industry"]["peer_count"] == 6
|
||||||
|
assert by_key["operating_margin"]["industry"]["label"] == "SIC 35 peers"
|
||||||
|
assert by_key["net_debt"]["industry"] is None
|
||||||
|
|
||||||
|
# valuation computed at request time
|
||||||
|
val = v1["valuation"]
|
||||||
|
assert val["pe"] is not None and val["market_cap_est"] is not None
|
||||||
|
assert val["price_date"] == "2026-10-01"
|
||||||
|
assert val["pe_industry"] is not None
|
||||||
|
|
||||||
|
# reads: header string + fixed by_key map (every metric + pe + fcf_yield)
|
||||||
|
assert v1["reads"]["header"]
|
||||||
|
assert set(v1["reads"]["by_key"]) == set(METRIC_KEYS) | {"pe", "fcf_yield"}
|
||||||
|
data = FundamentalResponse(symbol="AAPL", pe_ratio=12.3, **v1) # legacy + v1 additive
|
||||||
|
dumped = data.model_dump()
|
||||||
|
assert dumped["pe_ratio"] == 12.3 # legacy preserved untouched
|
||||||
|
assert dumped["metrics"][0]["key"] == "revenue_growth_yoy"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_same_day_earnings_is_next_with_zero_days(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
t = Ticker(symbol="TDY", cik=None)
|
||||||
|
s.add(t)
|
||||||
|
await s.flush()
|
||||||
|
s.add(EarningsEvent(ticker_id=t.id, announce_date=TODAY, session="bmo", source="dolt_earnings"))
|
||||||
|
s.add(EarningsEvent(ticker_id=t.id, announce_date=date(2026, 9, 1), session="amc",
|
||||||
|
eps_estimate=1.0, eps_actual=1.1, source="dolt_earnings"))
|
||||||
|
await s.commit()
|
||||||
|
async with factory() as s:
|
||||||
|
v1 = await build_fundamentals_v1(s, "TDY", today=TODAY)
|
||||||
|
assert v1["earnings"]["next"] == {"date": TODAY.isoformat(), "session": "bmo", "days_until": 0}
|
||||||
|
# the same-day event is upcoming, not in recent
|
||||||
|
assert all(r["announce_date"] != TODAY.isoformat() for r in v1["earnings"]["recent"])
|
||||||
|
|
||||||
|
|
||||||
|
async def test_eps_growth_read_is_populated(factory):
|
||||||
|
await _seed_group(factory)
|
||||||
|
async with factory() as s:
|
||||||
|
v1 = await build_fundamentals_v1(s, "AAPL", today=TODAY)
|
||||||
|
assert v1["reads"]["by_key"]["eps_growth_yoy"] is not None # EPS read now computed
|
||||||
|
|
||||||
|
|
||||||
|
async def test_non_positive_price_guards_valuation(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
t = Ticker(symbol="ZERO", cik="0000000055", sic="3571")
|
||||||
|
s.add(t)
|
||||||
|
await s.flush()
|
||||||
|
s.add(FundamentalSnapshot(cik="0000000055", accession="z", form="10-K", filed_date=date(2026, 1, 1),
|
||||||
|
accepted_at=datetime(2026, 1, 1, tzinfo=UTC), period_end=date(2025, 12, 31),
|
||||||
|
fiscal_year=2025, fiscal_period="FY", diluted_eps=5.0, shares_outstanding=1000))
|
||||||
|
s.add(OHLCVRecord(ticker_id=t.id, date=date(2026, 1, 2), open=0, high=0, low=0, close=0, volume=1))
|
||||||
|
await s.commit()
|
||||||
|
async with factory() as s:
|
||||||
|
v1 = await build_fundamentals_v1(s, "ZERO", today=TODAY)
|
||||||
|
assert v1["valuation"] is None # close of 0 is not a usable price
|
||||||
|
|
||||||
|
|
||||||
|
async def test_no_cik_ticker_yields_null_metrics(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
s.add(Ticker(symbol="ADR", cik=None)) # no SEC identity
|
||||||
|
await s.commit()
|
||||||
|
async with factory() as s:
|
||||||
|
v1 = await build_fundamentals_v1(s, "ADR", today=TODAY)
|
||||||
|
assert v1["valuation"] is None
|
||||||
|
assert all(m["value"] is None and m["industry"] is None for m in v1["metrics"])
|
||||||
|
assert v1["reads"]["header"] is None
|
||||||
|
assert v1["reads"]["by_key"] == {k: None for k in list(METRIC_KEYS) + ["pe", "fcf_yield"]}
|
||||||
|
|
||||||
|
|
||||||
|
async def test_industry_omitted_below_five_peers(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
await _seed_issuer(s, "SOLO", "0000000009", "9999", rev_base=1000, price=100, eps_base=2.0)
|
||||||
|
await s.commit()
|
||||||
|
async with factory() as s:
|
||||||
|
v1 = await build_fundamentals_v1(s, "SOLO", today=TODAY)
|
||||||
|
# only 1 issuer in the group -> below MIN_PEERS -> every industry omitted
|
||||||
|
assert all(m["industry"] is None for m in v1["metrics"])
|
||||||
|
assert v1["valuation"]["pe_industry"] is None
|
||||||
|
# but the subject's own valuation still computes
|
||||||
|
assert v1["valuation"]["pe"] is not None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_valuation_guarded_without_price(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
t = Ticker(symbol="NOPX", cik="0000000077", sic="3571")
|
||||||
|
s.add(t)
|
||||||
|
await s.flush()
|
||||||
|
# snapshots but NO ohlcv close
|
||||||
|
s.add(FundamentalSnapshot(cik="0000000077", accession="a", form="10-K", filed_date=date(2026, 1, 1),
|
||||||
|
accepted_at=datetime(2026, 1, 1, tzinfo=UTC), period_end=date(2025, 12, 31),
|
||||||
|
fiscal_year=2025, fiscal_period="FY", diluted_eps=5.0, shares_outstanding=1000))
|
||||||
|
await s.commit()
|
||||||
|
async with factory() as s:
|
||||||
|
v1 = await build_fundamentals_v1(s, "NOPX", today=TODAY)
|
||||||
|
# no usable price -> valuation is null under the approved contract
|
||||||
|
assert v1["valuation"] is None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_multiclass_subject_priced_by_requested_ticker(factory):
|
||||||
|
cik = "0001652044"
|
||||||
|
async with factory() as s:
|
||||||
|
# GOOGL and GOOG share one CIK/snapshots but trade at different prices
|
||||||
|
await _seed_issuer(s, "GOOGL", cik, "7372", rev_base=1000, price=200, eps_base=2.0)
|
||||||
|
# add a second class sharing the CIK: same snapshots exist; just its own ticker+price
|
||||||
|
goog = Ticker(symbol="GOOG", cik=cik, sic="7372")
|
||||||
|
s.add(goog)
|
||||||
|
await s.flush()
|
||||||
|
s.add(OHLCVRecord(ticker_id=goog.id, date=date(2026, 10, 1), open=100, high=100, low=100, close=100, volume=1))
|
||||||
|
for i in range(4): # peers so the group has >= 5 valid issuers
|
||||||
|
await _seed_issuer(s, f"P{i}", f"000000020{i}", "7373", rev_base=600 + i * 50, price=40 + i * 5)
|
||||||
|
await s.commit()
|
||||||
|
|
||||||
|
async with factory() as s:
|
||||||
|
googl = await build_fundamentals_v1(s, "GOOGL", today=TODAY)
|
||||||
|
goog_v = await build_fundamentals_v1(s, "GOOG", today=TODAY)
|
||||||
|
|
||||||
|
# subject P/E uses the REQUESTED class's price (200 vs 100), not an arbitrary sibling
|
||||||
|
assert googl["valuation"]["pe"] == pytest.approx(goog_v["valuation"]["pe"] * 2, rel=1e-6)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_endpoint_merges_legacy_and_v1(client, db_session):
|
||||||
|
from datetime import timezone as _tz
|
||||||
|
|
||||||
|
from app.dependencies import require_access
|
||||||
|
from app.main import app
|
||||||
|
from app.models.fundamental import FundamentalData
|
||||||
|
|
||||||
|
app.dependency_overrides[require_access] = lambda: None
|
||||||
|
try:
|
||||||
|
t = Ticker(symbol="AAPL", cik="0000000001", sic="3571")
|
||||||
|
db_session.add(t)
|
||||||
|
await db_session.flush()
|
||||||
|
db_session.add(FundamentalData(ticker_id=t.id, pe_ratio=12.3, revenue_growth=5.0,
|
||||||
|
fetched_at=datetime(2026, 1, 1, tzinfo=_tz.utc)))
|
||||||
|
db_session.add(FundamentalSnapshot(cik="0000000001", accession="a", form="10-K",
|
||||||
|
filed_date=date(2026, 1, 1), accepted_at=datetime(2026, 1, 1, tzinfo=_tz.utc),
|
||||||
|
period_end=date(2025, 12, 31), fiscal_year=2025, fiscal_period="FY",
|
||||||
|
diluted_eps=5.0, shares_outstanding=1000))
|
||||||
|
db_session.add(OHLCVRecord(ticker_id=t.id, date=date(2026, 1, 2), open=100, high=100, low=100, close=100, volume=1))
|
||||||
|
await db_session.flush()
|
||||||
|
|
||||||
|
resp = await client.get("/api/v1/fundamentals/AAPL")
|
||||||
|
assert resp.status_code == 200
|
||||||
|
data = resp.json()["data"]
|
||||||
|
assert data["pe_ratio"] == 12.3 # legacy preserved
|
||||||
|
assert data["revenue_growth"] == 5.0
|
||||||
|
assert len(data["metrics"]) == 7 # additive v1
|
||||||
|
assert data["earnings"] is not None
|
||||||
|
assert "by_key" in data["reads"]
|
||||||
|
assert data["valuation"]["price_date"] == "2026-01-02"
|
||||||
|
finally:
|
||||||
|
app.dependency_overrides.pop(require_access, None)
|
||||||
@@ -0,0 +1,174 @@
|
|||||||
|
"""Tests for pure read-time derivation of fundamentals from YTD snapshots."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from app.services import fundamentals_derivation as fd
|
||||||
|
|
||||||
|
UTC = timezone.utc
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Snap:
|
||||||
|
fiscal_year: int
|
||||||
|
fiscal_period: str
|
||||||
|
period_end: date
|
||||||
|
filed_date: date
|
||||||
|
accepted_at: datetime
|
||||||
|
revenue: float | None = None
|
||||||
|
net_income: float | None = None
|
||||||
|
operating_income: float | None = None
|
||||||
|
diluted_eps: float | None = None
|
||||||
|
cfo: float | None = None
|
||||||
|
capex: float | None = None
|
||||||
|
depreciation_amortization: float | None = None
|
||||||
|
cash_and_st_investments: float | None = None
|
||||||
|
total_debt: float | None = None
|
||||||
|
shares_outstanding: float | None = None
|
||||||
|
|
||||||
|
|
||||||
|
_FP = ["Q1", "Q2", "Q3", "FY"]
|
||||||
|
_ENDS = { # period_end per (fy, quarter index 0..3)
|
||||||
|
2025: [date(2024, 12, 31), date(2025, 3, 31), date(2025, 6, 30), date(2025, 9, 30)],
|
||||||
|
2026: [date(2025, 12, 31), date(2026, 3, 31), date(2026, 6, 30), date(2026, 9, 30)],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _year(fy, discretes: dict[str, list[float]], instants: dict[str, list] | None = None):
|
||||||
|
"""Build 4 snapshot rows (Q1,Q2,Q3,FY) with YTD-cumulative flow fields from the
|
||||||
|
given per-quarter discrete values; instants set as-is per quarter."""
|
||||||
|
rows = []
|
||||||
|
for i, fp in enumerate(_FP):
|
||||||
|
r = Snap(fy, fp, _ENDS[fy][i], _ENDS[fy][i], datetime(fy, 1 + i, 1, tzinfo=UTC))
|
||||||
|
for fname, ds in discretes.items():
|
||||||
|
setattr(r, fname, round(sum(ds[: i + 1]), 4)) # cumulative YTD
|
||||||
|
for fname, vals in (instants or {}).items():
|
||||||
|
setattr(r, fname, vals[i])
|
||||||
|
rows.append(r)
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
|
def _two_years():
|
||||||
|
rev25 = [100, 110, 120, 130]
|
||||||
|
rev26 = [110, 121, 132, 143] # +10% each quarter YoY
|
||||||
|
rows = _year(2025, {
|
||||||
|
"revenue": rev25,
|
||||||
|
"operating_income": [x * 0.2 for x in rev25],
|
||||||
|
"diluted_eps": [1.0, 1.1, 1.2, 1.3],
|
||||||
|
"cfo": [x * 0.25 for x in rev25],
|
||||||
|
"capex": [x * 0.05 for x in rev25],
|
||||||
|
"depreciation_amortization": [x * 0.05 for x in rev25],
|
||||||
|
}, instants={"shares_outstanding": [1000, 1000, 1000, 1000], "cash_and_st_investments": [40] * 4, "total_debt": [140] * 4})
|
||||||
|
rows += _year(2026, {
|
||||||
|
"revenue": rev26,
|
||||||
|
"operating_income": [x * 0.2 for x in rev26],
|
||||||
|
"diluted_eps": [1.1, 1.21, 1.32, 1.43],
|
||||||
|
"cfo": [x * 0.25 for x in rev26],
|
||||||
|
"capex": [x * 0.05 for x in rev26],
|
||||||
|
"depreciation_amortization": [x * 0.05 for x in rev26],
|
||||||
|
}, instants={"shares_outstanding": [900, 900, 900, 900], "cash_and_st_investments": [50] * 4, "total_debt": [150] * 4})
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
|
def test_revenue_growth_yoy_and_q4_derivation():
|
||||||
|
d = fd.derive(_two_years())
|
||||||
|
# TTM revenue FY2026 = 110+121+132+143 = 506; FY2025 = 460 -> +10%
|
||||||
|
assert d.metrics["revenue_growth_yoy"].value == pytest.approx(10.0, abs=1e-6)
|
||||||
|
# latest period is FY2026
|
||||||
|
assert d.latest_period_end == date(2026, 9, 30)
|
||||||
|
# tape has 4 points, newest last, each carrying a period_end
|
||||||
|
hist = d.metrics["revenue_growth_yoy"].history
|
||||||
|
assert len(hist) == 4 and hist[-1].period_end == date(2026, 9, 30)
|
||||||
|
|
||||||
|
|
||||||
|
def test_operating_and_fcf_margin():
|
||||||
|
d = fd.derive(_two_years())
|
||||||
|
assert d.metrics["operating_margin"].value == pytest.approx(20.0, abs=1e-6)
|
||||||
|
# FCF margin = (TTM cfo - TTM capex)/TTM rev = (0.25 - 0.05) = 20%
|
||||||
|
assert d.metrics["fcf_margin"].value == pytest.approx(20.0, abs=1e-6)
|
||||||
|
|
||||||
|
|
||||||
|
def test_net_debt_leverage_and_share_dilution():
|
||||||
|
d = fd.derive(_two_years())
|
||||||
|
# net debt = total_debt - cash = 150 - 50 = 100 (latest instant)
|
||||||
|
assert d.metrics["net_debt"].value == pytest.approx(100.0)
|
||||||
|
# EBITDA TTM = TTM operating_income + TTM D&A; net_debt/ebitda
|
||||||
|
op_ttm = 506 * 0.2 # 101.2
|
||||||
|
da_ttm = 506 * 0.05 # 25.3
|
||||||
|
assert d.metrics["net_debt_to_ebitda"].value == pytest.approx(100.0 / (op_ttm + da_ttm), rel=1e-6)
|
||||||
|
# shares 900 vs 1000 a year earlier -> -10% (buyback)
|
||||||
|
assert d.metrics["share_count_change_yoy"].value == pytest.approx(-10.0, abs=1e-6)
|
||||||
|
|
||||||
|
|
||||||
|
def test_valuation_inputs():
|
||||||
|
d = fd.derive(_two_years())
|
||||||
|
# TTM diluted EPS FY2026 = 1.1+1.21+1.32+1.43 = 5.06
|
||||||
|
assert d.ttm_diluted_eps == pytest.approx(5.06, abs=1e-6)
|
||||||
|
# TTM FCF = TTM cfo - TTM capex = 506*0.25 - 506*0.05 = 101.2
|
||||||
|
assert d.ttm_fcf == pytest.approx(506 * 0.20, abs=1e-6)
|
||||||
|
assert d.shares_outstanding == 900
|
||||||
|
|
||||||
|
|
||||||
|
def test_missing_period_yields_null_never_partial():
|
||||||
|
rows = _two_years()
|
||||||
|
# drop FY2026 Q3 -> discrete Q3 and Q4 (needs YTD Q3) become underivable,
|
||||||
|
# so TTM at FY2026 is null -> revenue growth null (not a partial sum)
|
||||||
|
rows = [r for r in rows if not (r.fiscal_year == 2026 and r.fiscal_period == "Q3")]
|
||||||
|
d = fd.derive(rows)
|
||||||
|
assert d.metrics["revenue_growth_yoy"].value is None
|
||||||
|
assert d.ttm_diluted_eps is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_net_debt_requires_both_components():
|
||||||
|
rows = _two_years()
|
||||||
|
for r in rows: # drop debt on the latest year -> can't form net debt
|
||||||
|
if r.fiscal_year == 2026:
|
||||||
|
r.total_debt = None
|
||||||
|
d = fd.derive(rows)
|
||||||
|
assert d.metrics["net_debt"].value is None
|
||||||
|
assert d.metrics["net_debt_to_ebitda"].value is None # net debt null -> leverage null
|
||||||
|
|
||||||
|
|
||||||
|
def test_leverage_null_when_ebitda_nonpositive():
|
||||||
|
rows = _two_years()
|
||||||
|
for r in rows: # negative operating income -> TTM EBITDA <= 0
|
||||||
|
r.operating_income = -abs(r.revenue)
|
||||||
|
r.depreciation_amortization = 1
|
||||||
|
d = fd.derive(rows)
|
||||||
|
assert d.metrics["net_debt"].value == pytest.approx(100.0) # net debt still valid
|
||||||
|
assert d.metrics["net_debt_to_ebitda"].value is None # but leverage nulled
|
||||||
|
|
||||||
|
|
||||||
|
def test_tape_stops_at_a_gap():
|
||||||
|
rows = [r for r in _two_years() if not (r.fiscal_year == 2026 and r.fiscal_period == "Q1")]
|
||||||
|
d = fd.derive(rows)
|
||||||
|
hist = d.metrics["operating_margin"].history
|
||||||
|
# consecutive suffix ending at FY2026: Q2, Q3, FY (not compressed across the Q1 gap)
|
||||||
|
assert [p.period_end for p in hist] == [date(2026, 3, 31), date(2026, 6, 30), date(2026, 9, 30)]
|
||||||
|
|
||||||
|
|
||||||
|
def test_yoy_growth_null_when_prior_nonpositive():
|
||||||
|
rows = _two_years()
|
||||||
|
for r in rows: # prior-year TTM EPS becomes negative
|
||||||
|
if r.fiscal_year == 2025:
|
||||||
|
r.diluted_eps = -abs(r.diluted_eps)
|
||||||
|
d = fd.derive(rows)
|
||||||
|
assert d.metrics["eps_growth_yoy"].value is None # loss->profit is not a %
|
||||||
|
|
||||||
|
|
||||||
|
def test_amendment_selection_newest_accepted_wins():
|
||||||
|
rows = _two_years()
|
||||||
|
# an amendment to FY2026 FY restates revenue YTD higher, accepted later
|
||||||
|
amended = Snap(2026, "FY", date(2026, 9, 30), date(2026, 11, 1),
|
||||||
|
datetime(2027, 1, 1, tzinfo=UTC), revenue=999999,
|
||||||
|
operating_income=100, diluted_eps=1.43, cfo=100, capex=10,
|
||||||
|
depreciation_amortization=25, shares_outstanding=900,
|
||||||
|
cash_and_st_investments=50, total_debt=150)
|
||||||
|
d = fd.derive(rows + [amended])
|
||||||
|
# Q4 revenue discrete now uses the amended YTD(FY)=999999 minus YTD(Q3)=363
|
||||||
|
# so TTM/growth reflects the amendment, proving newest accepted_at won.
|
||||||
|
assert d.metrics["revenue_growth_yoy"].value != pytest.approx(10.0, abs=1e-6)
|
||||||
@@ -0,0 +1,58 @@
|
|||||||
|
"""Tests for pure peer statistics."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import math
|
||||||
|
|
||||||
|
from app.services import fundamentals_peers as pr
|
||||||
|
|
||||||
|
|
||||||
|
def test_median_ranks_at_50_tie_aware():
|
||||||
|
s = pr.peer_stat(3, [1, 2, 3, 4, 5], higher_is_better=True)
|
||||||
|
assert s.median == 3
|
||||||
|
assert s.favorable_percentile == 50 # tie-aware rank of the median
|
||||||
|
assert s.peer_count == 5
|
||||||
|
|
||||||
|
|
||||||
|
def test_all_equal_peers_rank_at_50():
|
||||||
|
s = pr.peer_stat(3, [3, 3, 3, 3, 3], higher_is_better=True)
|
||||||
|
assert s.favorable_percentile == 50 # not 100 — ties don't get full credit
|
||||||
|
|
||||||
|
|
||||||
|
def test_peer_stat_lower_is_better_flips_direction():
|
||||||
|
assert pr.peer_stat(3, [1, 2, 3, 4, 5], higher_is_better=False).favorable_percentile == 50
|
||||||
|
|
||||||
|
|
||||||
|
def test_peer_stat_unique_top_and_bottom():
|
||||||
|
assert pr.peer_stat(5, [1, 2, 3, 4, 5], higher_is_better=True).favorable_percentile == 100
|
||||||
|
assert pr.peer_stat(1, [1, 2, 3, 4, 5], higher_is_better=True).favorable_percentile == 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_peer_stat_requires_min_valid_peers():
|
||||||
|
assert pr.peer_stat(3, [1, 2, 3, None], higher_is_better=True) is None # 3 valid < 5
|
||||||
|
assert pr.peer_stat(None, [1, 2, 3, 4, 5], higher_is_better=True) is None # null subject
|
||||||
|
|
||||||
|
|
||||||
|
def test_peer_stat_excludes_null_and_non_finite():
|
||||||
|
s = pr.peer_stat(3, [1, 2, 3, 4, 5, None, math.nan, math.inf, -math.inf], higher_is_better=True)
|
||||||
|
assert s.peer_count == 5 # nulls + NaN/inf dropped
|
||||||
|
# a non-finite subject is invalid
|
||||||
|
assert pr.peer_stat(math.nan, [1, 2, 3, 4, 5], higher_is_better=True) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_net_debt_is_not_peer_eligible():
|
||||||
|
assert pr.peer_stat_for("net_debt", 100, [10, 20, 30, 40, 50]) is None # size-dependent
|
||||||
|
assert pr.peer_stat_for("net_debt_to_ebitda", 1.0, [1, 2, 3, 4, 5]) is not None
|
||||||
|
|
||||||
|
|
||||||
|
def test_peer_stat_for_uses_polarity():
|
||||||
|
# pe is lower-is-better: a low pe beats most peers
|
||||||
|
s = pr.peer_stat_for("pe", 10, [10, 20, 30, 40, 50])
|
||||||
|
assert s.favorable_percentile == 100
|
||||||
|
|
||||||
|
|
||||||
|
def test_two_digit_sic():
|
||||||
|
assert pr.two_digit_sic("7372") == "73"
|
||||||
|
assert pr.two_digit_sic("3571") == "35"
|
||||||
|
assert pr.two_digit_sic(None) is None
|
||||||
|
assert pr.two_digit_sic("x") is None
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
"""Tests for deterministic text reads, incl. threshold boundaries."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
|
from app.services import fundamentals_reads as rd
|
||||||
|
|
||||||
|
|
||||||
|
def _hist(*values):
|
||||||
|
return [SimpleNamespace(value=v, period_end=None) for v in values]
|
||||||
|
|
||||||
|
|
||||||
|
def test_growth_read_boundaries():
|
||||||
|
assert rd.growth_read(_hist(5, 6, 8)) == "accelerating" # +2.0 exactly (>=)
|
||||||
|
assert rd.growth_read(_hist(5, 6, 7.9)) == "steady" # +1.9 < 2.0
|
||||||
|
assert rd.growth_read(_hist(10, 9, 7)) == "decelerating" # -2.0 exactly
|
||||||
|
assert rd.growth_read(_hist(5, 6)) is None # < 3 periods
|
||||||
|
|
||||||
|
|
||||||
|
def test_reads_use_latest_nonnull_suffix():
|
||||||
|
# latest displayed value is n/a -> no read (never reflect a null latest)
|
||||||
|
assert rd.growth_read(_hist(5, 6, 8, None)) is None
|
||||||
|
assert rd.margin_read(_hist(19, 20, 22, None)) is None
|
||||||
|
# an internal gap truncates the run -> fewer than 3 consecutive -> no read
|
||||||
|
assert rd.growth_read(_hist(5, 6, None, 8)) is None
|
||||||
|
# a clean 3-run after an older gap still reads
|
||||||
|
assert rd.growth_read(_hist(None, 5, 6, 8)) == "accelerating"
|
||||||
|
|
||||||
|
|
||||||
|
def test_margin_read_vs_mean_of_prior():
|
||||||
|
# prior mean = (19+20)/2 = 19.5; latest 21 -> +1.5 -> improving
|
||||||
|
assert rd.margin_read(_hist(19, 20, 21)) == "improving"
|
||||||
|
# latest exactly +1.0 over prior mean -> improving
|
||||||
|
assert rd.margin_read(_hist(20, 20, 21)) == "improving"
|
||||||
|
# within band
|
||||||
|
assert rd.margin_read(_hist(20, 20, 20.5)) == "stable"
|
||||||
|
assert rd.margin_read(_hist(21, 20)) is None # < 3 periods
|
||||||
|
|
||||||
|
|
||||||
|
def test_share_count_read():
|
||||||
|
assert rd.share_count_read(1.8) == "1.8% dilution"
|
||||||
|
assert rd.share_count_read(-2.0) == "buying back"
|
||||||
|
assert rd.share_count_read(1.0) == "flat" # boundary: not > 1.0
|
||||||
|
assert rd.share_count_read(None) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_peer_read_bands_and_polarity_phrasing():
|
||||||
|
assert rd.peer_read("operating_margin", 60) == "above peers" # boundary favorable
|
||||||
|
assert rd.peer_read("operating_margin", 40) == "below peers" # boundary adverse
|
||||||
|
assert rd.peer_read("operating_margin", 50) == "in line"
|
||||||
|
assert rd.peer_read("pe", 65) == "attractively valued"
|
||||||
|
assert rd.peer_read("pe", 30) == "priced above peers"
|
||||||
|
assert rd.peer_read("net_debt_to_ebitda", 20) == "elevated leverage"
|
||||||
|
assert rd.peer_read("pe", None) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_header_sentence_omits_missing_segments():
|
||||||
|
assert rd.header_sentence("accelerating", "stable", "priced above peers") == (
|
||||||
|
"growth accelerating · margins stable · valuation priced above peers"
|
||||||
|
)
|
||||||
|
assert rd.header_sentence(None, "improving", None) == "margins improving"
|
||||||
|
assert rd.header_sentence(None, None, None) == ""
|
||||||
@@ -80,6 +80,28 @@ class TestTradingDayCrons:
|
|||||||
)
|
)
|
||||||
assert fire.strftime("%a") == "Mon"
|
assert fire.strftime("%a") == "Mon"
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("key", "hour", "minute"),
|
||||||
|
(
|
||||||
|
("schedule_dolt_earnings_cron", 2, 30),
|
||||||
|
("schedule_sec_fundamentals_cron", 4, 0),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
def test_shadow_imports_run_daily_at_expected_et_time(
|
||||||
|
self, key: str, hour: int, minute: int
|
||||||
|
):
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from apscheduler.triggers.cron import CronTrigger
|
||||||
|
|
||||||
|
trigger = CronTrigger.from_crontab(
|
||||||
|
SCHEDULE_DEFAULTS[key], timezone=SCHEDULE_DEFAULTS["schedule_timezone"]
|
||||||
|
)
|
||||||
|
fire = trigger.get_next_fire_time(
|
||||||
|
None, datetime(2026, 7, 19, tzinfo=trigger.timezone)
|
||||||
|
)
|
||||||
|
assert (fire.hour, fire.minute) == (hour, minute)
|
||||||
|
|
||||||
|
|
||||||
class TestScheduleConfig:
|
class TestScheduleConfig:
|
||||||
async def test_defaults_when_unset(self, session: AsyncSession):
|
async def test_defaults_when_unset(self, session: AsyncSession):
|
||||||
|
|||||||
@@ -1,5 +1,7 @@
|
|||||||
"""Unit tests for app.scheduler module."""
|
"""Unit tests for app.scheduler module."""
|
||||||
|
|
||||||
|
from types import SimpleNamespace
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from app.scheduler import (
|
from app.scheduler import (
|
||||||
@@ -8,7 +10,9 @@ from app.scheduler import (
|
|||||||
_parse_frequency,
|
_parse_frequency,
|
||||||
_resume_tickers,
|
_resume_tickers,
|
||||||
_last_successful,
|
_last_successful,
|
||||||
|
_run_shadow_import,
|
||||||
configure_scheduler,
|
configure_scheduler,
|
||||||
|
get_job_runtime_snapshot,
|
||||||
queue_backtest_options,
|
queue_backtest_options,
|
||||||
queue_backtest_target_model,
|
queue_backtest_target_model,
|
||||||
scheduler,
|
scheduler,
|
||||||
@@ -106,6 +110,8 @@ class TestConfigureScheduler:
|
|||||||
"benchmark_collector",
|
"benchmark_collector",
|
||||||
"sentiment_collector",
|
"sentiment_collector",
|
||||||
"fundamental_collector",
|
"fundamental_collector",
|
||||||
|
"dolt_earnings_import",
|
||||||
|
"sec_fundamentals_import",
|
||||||
"rr_scanner",
|
"rr_scanner",
|
||||||
"shadow_book",
|
"shadow_book",
|
||||||
"ticker_universe_sync",
|
"ticker_universe_sync",
|
||||||
@@ -137,6 +143,8 @@ class TestConfigureScheduler:
|
|||||||
"data_collector",
|
"data_collector",
|
||||||
"data_backfill",
|
"data_backfill",
|
||||||
"fundamental_collector",
|
"fundamental_collector",
|
||||||
|
"dolt_earnings_import",
|
||||||
|
"sec_fundamentals_import",
|
||||||
"market_regime",
|
"market_regime",
|
||||||
"near_close_pipeline",
|
"near_close_pipeline",
|
||||||
"regime_monitor",
|
"regime_monitor",
|
||||||
@@ -147,3 +155,91 @@ class TestConfigureScheduler:
|
|||||||
"shadow_book",
|
"shadow_book",
|
||||||
"ticker_universe_sync",
|
"ticker_universe_sync",
|
||||||
])
|
])
|
||||||
|
|
||||||
|
|
||||||
|
class _SessionContext:
|
||||||
|
async def __aenter__(self):
|
||||||
|
return object()
|
||||||
|
|
||||||
|
async def __aexit__(self, *exc):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
class TestShadowImportJobs:
|
||||||
|
@staticmethod
|
||||||
|
def _session_factory():
|
||||||
|
return _SessionContext()
|
||||||
|
|
||||||
|
async def test_promoted_run_surfaces_completion(self, monkeypatch):
|
||||||
|
async def enabled(db, job_name):
|
||||||
|
return True
|
||||||
|
|
||||||
|
async def imported(importer):
|
||||||
|
return SimpleNamespace(
|
||||||
|
status="promoted", revision="abcdef1234567890", error_details=None
|
||||||
|
)
|
||||||
|
|
||||||
|
monkeypatch.setattr("app.scheduler.async_session_factory", self._session_factory)
|
||||||
|
monkeypatch.setattr("app.scheduler._is_job_enabled", enabled)
|
||||||
|
monkeypatch.setattr("app.scheduler.run_import", imported)
|
||||||
|
|
||||||
|
await _run_shadow_import("dolt_earnings_import", object())
|
||||||
|
|
||||||
|
runtime = get_job_runtime_snapshot("dolt_earnings_import")
|
||||||
|
assert runtime["status"] == "completed"
|
||||||
|
assert runtime["processed"] == 1
|
||||||
|
assert runtime["message"] == "promoted · abcdef123456"
|
||||||
|
|
||||||
|
async def test_failed_run_surfaces_error(self, monkeypatch):
|
||||||
|
async def enabled(db, job_name):
|
||||||
|
return True
|
||||||
|
|
||||||
|
async def imported(importer):
|
||||||
|
return SimpleNamespace(
|
||||||
|
status="failed", revision=None, error_details="validation failed"
|
||||||
|
)
|
||||||
|
|
||||||
|
monkeypatch.setattr("app.scheduler.async_session_factory", self._session_factory)
|
||||||
|
monkeypatch.setattr("app.scheduler._is_job_enabled", enabled)
|
||||||
|
monkeypatch.setattr("app.scheduler.run_import", imported)
|
||||||
|
|
||||||
|
await _run_shadow_import("sec_fundamentals_import", object())
|
||||||
|
|
||||||
|
runtime = get_job_runtime_snapshot("sec_fundamentals_import")
|
||||||
|
assert runtime["status"] == "error"
|
||||||
|
assert runtime["processed"] == 0
|
||||||
|
assert runtime["message"] == "validation failed"
|
||||||
|
|
||||||
|
async def test_source_lock_surfaces_skipped(self, monkeypatch):
|
||||||
|
async def enabled(db, job_name):
|
||||||
|
return True
|
||||||
|
|
||||||
|
async def imported(importer):
|
||||||
|
return None
|
||||||
|
|
||||||
|
monkeypatch.setattr("app.scheduler.async_session_factory", self._session_factory)
|
||||||
|
monkeypatch.setattr("app.scheduler._is_job_enabled", enabled)
|
||||||
|
monkeypatch.setattr("app.scheduler.run_import", imported)
|
||||||
|
|
||||||
|
await _run_shadow_import("dolt_earnings_import", object())
|
||||||
|
|
||||||
|
runtime = get_job_runtime_snapshot("dolt_earnings_import")
|
||||||
|
assert runtime["status"] == "skipped"
|
||||||
|
assert "already running" in runtime["message"]
|
||||||
|
|
||||||
|
async def test_disabled_job_never_runs_importer(self, monkeypatch):
|
||||||
|
async def disabled(db, job_name):
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def should_not_run(importer):
|
||||||
|
raise AssertionError("disabled job ran importer")
|
||||||
|
|
||||||
|
monkeypatch.setattr("app.scheduler.async_session_factory", self._session_factory)
|
||||||
|
monkeypatch.setattr("app.scheduler._is_job_enabled", disabled)
|
||||||
|
monkeypatch.setattr("app.scheduler.run_import", should_not_run)
|
||||||
|
|
||||||
|
await _run_shadow_import("sec_fundamentals_import", object())
|
||||||
|
|
||||||
|
runtime = get_job_runtime_snapshot("sec_fundamentals_import")
|
||||||
|
assert runtime["status"] == "skipped"
|
||||||
|
assert runtime["message"] == "Disabled"
|
||||||
|
|||||||
@@ -0,0 +1,225 @@
|
|||||||
|
"""Tests for the SEC client's parsing and fair-access behavior, via a mocked
|
||||||
|
httpx transport (no network)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from datetime import date
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from app.services import sec_client as sc
|
||||||
|
from app.services.sec_client import SecClient, SecError, SecForbiddenError
|
||||||
|
|
||||||
|
COMPANY_TICKERS = {
|
||||||
|
"0": {"cik_str": 320193, "ticker": "AAPL", "title": "Apple Inc."},
|
||||||
|
"1": {"cik_str": 1652044, "ticker": "GOOGL", "title": "Alphabet"},
|
||||||
|
"2": {"cik_str": 1652044, "ticker": "GOOG", "title": "Alphabet"},
|
||||||
|
"3": {"cik_str": 1067983, "ticker": "BRK-B", "title": "Berkshire"},
|
||||||
|
}
|
||||||
|
|
||||||
|
SUBMISSIONS_BASE = {
|
||||||
|
"cik": 320193,
|
||||||
|
"name": "Apple Inc.",
|
||||||
|
"sic": "3571",
|
||||||
|
"sicDescription": "Electronic Computers",
|
||||||
|
"fiscalYearEnd": "0926",
|
||||||
|
"tickers": ["AAPL"],
|
||||||
|
"filings": {
|
||||||
|
"recent": {
|
||||||
|
"accessionNumber": ["0000320193-26-000013", "0000320193-26-000006", "0000320193-26-000099"],
|
||||||
|
"form": ["10-Q", "10-K", "8-K"],
|
||||||
|
"reportDate": ["2026-03-28", "2025-09-27", "2026-04-01"],
|
||||||
|
"filingDate": ["2026-05-01", "2026-01-30", "2026-04-02"],
|
||||||
|
"acceptanceDateTime": ["2026-05-01T10:01:00.000Z", "2025-10-31T10:01:26.000Z", "2026-04-02T09:00:00.000Z"],
|
||||||
|
"isXBRL": [1, 1, 0],
|
||||||
|
},
|
||||||
|
"files": [{"name": "CIK0000320193-submissions-001.json", "filingFrom": "1994-01-26", "filingTo": "2015-05-27"}],
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
SUBMISSIONS_SHARD = {
|
||||||
|
"accessionNumber": ["0000320193-94-000002"],
|
||||||
|
"form": ["10-Q"],
|
||||||
|
"reportDate": ["1993-12-31"],
|
||||||
|
"filingDate": ["1994-01-26"],
|
||||||
|
"acceptanceDateTime": ["1994-01-26T05:00:00.000Z"],
|
||||||
|
"isXBRL": [0],
|
||||||
|
}
|
||||||
|
|
||||||
|
FORM_IDX = """Description: Daily Index of EDGAR Dissemination Feed by Form Type
|
||||||
|
|
||||||
|
Form Type Company Name CIK Date Filed File Name
|
||||||
|
-------------------------------------------------------------------------------
|
||||||
|
10-K/A Starfighters Space, Inc. 1947016 20260721 edgar/data/1947016/0001062993-26-003746.txt
|
||||||
|
10-Q 3M CO 66740 20260721 edgar/data/66740/0000066740-26-000246.txt
|
||||||
|
8-K Some Corp 12345 20260721 edgar/data/12345/0000012345-26-000001.txt
|
||||||
|
10-Q CALIX, INC 1406666 20260721 edgar/data/1406666/0001406666-26-000034.txt
|
||||||
|
"""
|
||||||
|
|
||||||
|
DIR_JSON = {"directory": {"item": [
|
||||||
|
{"name": "form.20260720.idx"}, {"name": "form.20260721.idx"}, {"name": "company.20260721.idx"},
|
||||||
|
]}}
|
||||||
|
|
||||||
|
|
||||||
|
def _handler(request: httpx.Request) -> httpx.Response:
|
||||||
|
url = str(request.url)
|
||||||
|
if url.endswith("company_tickers.json"):
|
||||||
|
return httpx.Response(200, json=COMPANY_TICKERS)
|
||||||
|
if url.endswith("submissions/CIK0000320193.json"):
|
||||||
|
return httpx.Response(200, json=SUBMISSIONS_BASE)
|
||||||
|
if url.endswith("CIK0000320193-submissions-001.json"):
|
||||||
|
return httpx.Response(200, json=SUBMISSIONS_SHARD)
|
||||||
|
if url.endswith("form.20260721.idx"):
|
||||||
|
return httpx.Response(200, text=FORM_IDX)
|
||||||
|
if url.endswith("QTR3/index.json"):
|
||||||
|
return httpx.Response(200, json=DIR_JSON)
|
||||||
|
return httpx.Response(404)
|
||||||
|
|
||||||
|
|
||||||
|
def _client(**kw):
|
||||||
|
return SecClient(transport=httpx.MockTransport(_handler), spacing_seconds=0, **kw)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_company_tickers_normalised_and_multiclass():
|
||||||
|
async with _client() as c:
|
||||||
|
m = await c.company_tickers()
|
||||||
|
assert m["AAPL"] == 320193
|
||||||
|
assert m["GOOGL"] == m["GOOG"] == 1652044 # multi-class share one CIK
|
||||||
|
assert m["BRK-B"] == 1067983 # dash form
|
||||||
|
|
||||||
|
|
||||||
|
async def test_submissions_history_merges_shards_and_filters_forms():
|
||||||
|
async with _client() as c:
|
||||||
|
sub = await c.submissions(320193, include_history=True)
|
||||||
|
assert sub["sic"] == "3571" and sub["fiscal_year_end"] == "0926"
|
||||||
|
accns = {f["accession"] for f in sub["filings"]}
|
||||||
|
# 10-Q + 10-K from recent, 10-Q from the shard; the 8-K is filtered out
|
||||||
|
assert accns == {"0000320193-26-000013", "0000320193-26-000006", "0000320193-94-000002"}
|
||||||
|
older = next(f for f in sub["filings"] if f["accession"] == "0000320193-94-000002")
|
||||||
|
assert older["report_date"] == "1993-12-31" and older["is_xbrl"] is False
|
||||||
|
|
||||||
|
|
||||||
|
async def test_submissions_recent_only_skips_shard_requests():
|
||||||
|
seen = []
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
seen.append(str(request.url))
|
||||||
|
return _handler(request)
|
||||||
|
|
||||||
|
async with SecClient(transport=httpx.MockTransport(handler), spacing_seconds=0) as c:
|
||||||
|
sub = await c.submissions(320193) # include_history defaults False
|
||||||
|
assert not any("submissions-001" in u for u in seen) # no shard fetch
|
||||||
|
accns = {f["accession"] for f in sub["filings"]}
|
||||||
|
assert accns == {"0000320193-26-000013", "0000320193-26-000006"} # recent only
|
||||||
|
|
||||||
|
|
||||||
|
async def test_daily_index_parses_10kq_rows():
|
||||||
|
async with _client() as c:
|
||||||
|
rows = await c.daily_index(date(2026, 7, 21))
|
||||||
|
forms = {r["form"] for r in rows}
|
||||||
|
assert forms == {"10-K/A", "10-Q"} # 8-K excluded
|
||||||
|
mmm = next(r for r in rows if r["cik"] == 66740)
|
||||||
|
assert mmm["accession"] == "0000066740-26-000246"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_latest_index_date():
|
||||||
|
async with _client() as c:
|
||||||
|
d = await c.latest_index_date(today=date(2026, 7, 22))
|
||||||
|
assert d == date(2026, 7, 21)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_403_raises_forbidden_no_retry():
|
||||||
|
calls = {"n": 0}
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
calls["n"] += 1
|
||||||
|
return httpx.Response(403)
|
||||||
|
|
||||||
|
async with SecClient(transport=httpx.MockTransport(handler), spacing_seconds=0) as c:
|
||||||
|
with pytest.raises(SecForbiddenError):
|
||||||
|
await c.get_json("https://data.sec.gov/x")
|
||||||
|
assert calls["n"] == 1 # alert and stop, never retry-loop
|
||||||
|
|
||||||
|
|
||||||
|
async def test_429_retries_then_succeeds(monkeypatch):
|
||||||
|
async def _instant(_):
|
||||||
|
return None
|
||||||
|
|
||||||
|
monkeypatch.setattr(sc.asyncio, "sleep", _instant) # no real backoff wait
|
||||||
|
calls = {"n": 0}
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
calls["n"] += 1
|
||||||
|
if calls["n"] < 3:
|
||||||
|
return httpx.Response(429)
|
||||||
|
return httpx.Response(200, json={"ok": True})
|
||||||
|
|
||||||
|
async with SecClient(transport=httpx.MockTransport(handler), spacing_seconds=0, max_retries=5) as c:
|
||||||
|
data = await c.get_json("https://data.sec.gov/y")
|
||||||
|
assert data == {"ok": True} and calls["n"] == 3
|
||||||
|
|
||||||
|
|
||||||
|
async def test_429_gives_up_after_max_retries(monkeypatch):
|
||||||
|
async def _instant(_):
|
||||||
|
return None
|
||||||
|
|
||||||
|
monkeypatch.setattr(sc.asyncio, "sleep", _instant)
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
return httpx.Response(429)
|
||||||
|
|
||||||
|
async with SecClient(transport=httpx.MockTransport(handler), spacing_seconds=0, max_retries=2) as c:
|
||||||
|
with pytest.raises(SecError):
|
||||||
|
await c.get_json("https://data.sec.gov/z")
|
||||||
|
|
||||||
|
|
||||||
|
def _status_client(status):
|
||||||
|
def handler(request):
|
||||||
|
return httpx.Response(status)
|
||||||
|
# max_retries=0 so 5xx/429 raise immediately (no retry sleeps)
|
||||||
|
return SecClient(transport=httpx.MockTransport(handler), spacing_seconds=0, max_retries=0)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_index_methods_propagate_403():
|
||||||
|
# A 403 must NOT be mistaken for "no index".
|
||||||
|
async with _status_client(403) as c:
|
||||||
|
with pytest.raises(sc.SecForbiddenError):
|
||||||
|
await c.daily_index(date(2026, 7, 21))
|
||||||
|
async with _status_client(403) as c:
|
||||||
|
with pytest.raises(sc.SecForbiddenError):
|
||||||
|
await c.latest_index_date(today=date(2026, 7, 22))
|
||||||
|
|
||||||
|
|
||||||
|
async def test_index_methods_propagate_500():
|
||||||
|
async with _status_client(500) as c:
|
||||||
|
with pytest.raises(SecError):
|
||||||
|
await c.daily_index(date(2026, 7, 21))
|
||||||
|
async with _status_client(500) as c:
|
||||||
|
with pytest.raises(SecError):
|
||||||
|
await c.latest_index_date(today=date(2026, 7, 22))
|
||||||
|
|
||||||
|
|
||||||
|
async def test_only_404_is_treated_as_missing():
|
||||||
|
async with _status_client(404) as c:
|
||||||
|
assert await c.daily_index(date(2026, 7, 21)) == []
|
||||||
|
assert await c.latest_index_date(today=date(2026, 7, 22)) is None
|
||||||
|
|
||||||
|
|
||||||
|
async def test_fair_access_validation_on_real_client():
|
||||||
|
# Placeholder email rejected.
|
||||||
|
with pytest.raises(SecError):
|
||||||
|
async with SecClient(user_agent="signal-platform (contact: you@example.com)"):
|
||||||
|
pass
|
||||||
|
# Non-email UA rejected.
|
||||||
|
with pytest.raises(SecError):
|
||||||
|
async with SecClient(user_agent="signal-platform"):
|
||||||
|
pass
|
||||||
|
# Valid UA but unsafe production spacing rejected.
|
||||||
|
with pytest.raises(SecError):
|
||||||
|
async with SecClient(user_agent="signal-platform real@person.io", spacing_seconds=0.0):
|
||||||
|
pass
|
||||||
|
# Valid UA + safe spacing opens fine.
|
||||||
|
async with SecClient(user_agent="signal-platform real@person.io", spacing_seconds=0.2):
|
||||||
|
pass
|
||||||
@@ -0,0 +1,272 @@
|
|||||||
|
"""Tests for the companyfacts -> snapshot parser, on a realistic Apple-shaped
|
||||||
|
fixture (the structure verified by live probe)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from app.services.sec_facts_parser import (
|
||||||
|
Fact,
|
||||||
|
FilingMeta,
|
||||||
|
_compose_cash,
|
||||||
|
_compose_debt,
|
||||||
|
_fiscal_context,
|
||||||
|
_select_shares,
|
||||||
|
parse_snapshots,
|
||||||
|
)
|
||||||
|
|
||||||
|
UTC = timezone.utc
|
||||||
|
|
||||||
|
|
||||||
|
def _dur(start, end, val, accn, fy=2026, fp="Q2"):
|
||||||
|
return {"start": start, "end": end, "val": val, "fy": fy, "fp": fp, "accn": accn, "form": "10-Q"}
|
||||||
|
|
||||||
|
|
||||||
|
def _inst(end, val, accn, fy=2026, fp="Q2"):
|
||||||
|
return {"end": end, "val": val, "fy": fy, "fp": fp, "accn": accn, "form": "10-Q"}
|
||||||
|
|
||||||
|
|
||||||
|
COMPANYFACTS = {
|
||||||
|
"cik": 320193,
|
||||||
|
"facts": {
|
||||||
|
"us-gaap": {
|
||||||
|
"RevenueFromContractWithCustomerExcludingAssessedTax": {"units": {"USD": [
|
||||||
|
_dur("2025-09-28", "2025-12-27", 143756, "A", fp="Q1"), # Q1 discrete == YTD
|
||||||
|
_dur("2025-09-28", "2026-03-28", 254940, "B"), # Q2 YTD (181d) <- want this
|
||||||
|
_dur("2025-12-28", "2026-03-28", 111184, "B"), # Q2 discrete (90d)
|
||||||
|
]}},
|
||||||
|
"NetIncomeLoss": {"units": {"USD": [_dur("2025-09-28", "2026-03-28", 40000, "B")]}},
|
||||||
|
# only a discrete-length fact for Q2 -> must be null, not the discrete
|
||||||
|
"OperatingIncomeLoss": {"units": {"USD": [_dur("2025-12-28", "2026-03-28", 30000, "B")]}},
|
||||||
|
"EarningsPerShareDiluted": {"units": {"USD/shares": [_dur("2025-09-28", "2026-03-28", 2.55, "B")]}},
|
||||||
|
"CashAndCashEquivalentsAtCarryingValue": {"units": {"USD": [_inst("2026-03-28", 30000, "B")]}},
|
||||||
|
"MarketableSecuritiesCurrent": {"units": {"USD": [_inst("2026-03-28", 20000, "B")]}},
|
||||||
|
"LongTermDebtNoncurrent": {"units": {"USD": [_inst("2026-03-28", 80000, "B")]}},
|
||||||
|
"LongTermDebtCurrent": {"units": {"USD": [_inst("2026-03-28", 10000, "B")]}},
|
||||||
|
},
|
||||||
|
"dei": {
|
||||||
|
"EntityCommonStockSharesOutstanding": {"units": {"shares": [
|
||||||
|
{"end": "2026-04-17", "val": 14687356000, "fy": 2026, "fp": "Q2", "accn": "B", "form": "10-Q"},
|
||||||
|
]}},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
FILINGS = {
|
||||||
|
"A": FilingMeta(date(2025, 12, 27), date(2026, 1, 30), datetime(2026, 1, 30, 11, 1, tzinfo=UTC), "10-Q"),
|
||||||
|
"B": FilingMeta(date(2026, 3, 28), date(2026, 5, 1), datetime(2026, 5, 1, 10, 1, tzinfo=UTC), "10-Q"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _by_accn(rows):
|
||||||
|
return {r.accession: r for r in rows}
|
||||||
|
|
||||||
|
|
||||||
|
def test_parses_ytd_not_discrete_and_cover_date_shares():
|
||||||
|
res = parse_snapshots(COMPANYFACTS, FILINGS, {"A", "B"})
|
||||||
|
assert not res.skipped_filings and not res.field_issues
|
||||||
|
b = _by_accn(res.rows)["B"]
|
||||||
|
|
||||||
|
assert (b.cik, b.fiscal_year, b.fiscal_period) == ("0000320193", 2026, "Q2")
|
||||||
|
assert b.period_end == date(2026, 3, 28)
|
||||||
|
assert b.period_start == date(2025, 9, 28) # YTD start (fiscal-year start)
|
||||||
|
assert b.revenue == 254940 # the 6-month YTD, NOT the 111184 discrete
|
||||||
|
assert b.net_income == 40000
|
||||||
|
assert b.operating_income is None # only a discrete-length fact existed -> null
|
||||||
|
assert b.diluted_eps == 2.55
|
||||||
|
# cash + first-present ST investment (MarketableSecuritiesCurrent), each once
|
||||||
|
assert b.cash_and_st_investments == 50000
|
||||||
|
# long-term parts summed (no aggregate, no short-term)
|
||||||
|
assert b.total_debt == 90000
|
||||||
|
assert b.shares_outstanding == 14687356000
|
||||||
|
assert b.shares_outstanding_date == date(2026, 4, 17) # cover date != period_end
|
||||||
|
assert b.filed_date == date(2026, 5, 1)
|
||||||
|
assert b.accepted_at == datetime(2026, 5, 1, 10, 1, tzinfo=UTC)
|
||||||
|
|
||||||
|
|
||||||
|
def test_q1_discrete_is_the_ytd():
|
||||||
|
res = parse_snapshots(COMPANYFACTS, FILINGS, {"A"})
|
||||||
|
a = _by_accn(res.rows)["A"]
|
||||||
|
assert a.fiscal_period == "Q1"
|
||||||
|
assert a.revenue == 143756 # Q1 YTD == Q1 discrete
|
||||||
|
assert a.period_start == date(2025, 9, 28)
|
||||||
|
|
||||||
|
|
||||||
|
def test_skips_filing_without_usable_period():
|
||||||
|
cf = {
|
||||||
|
"cik": 320193,
|
||||||
|
"facts": {"us-gaap": {"NetIncomeLoss": {"units": {"USD": [
|
||||||
|
{"start": "2025-09-28", "end": "2026-03-28", "val": 1, "fy": 2026, "fp": "H1", "accn": "X", "form": "10-Q"},
|
||||||
|
]}}}},
|
||||||
|
}
|
||||||
|
filings = {"X": FilingMeta(date(2026, 3, 28), date(2026, 5, 1), datetime(2026, 5, 1, tzinfo=UTC), "10-Q")}
|
||||||
|
res = parse_snapshots(cf, filings, {"X"})
|
||||||
|
assert res.rows == []
|
||||||
|
assert res.skipped_filings == [{"accession": "X", "reason": "no usable period identity"}]
|
||||||
|
|
||||||
|
|
||||||
|
def test_missing_accession_is_skipped():
|
||||||
|
res = parse_snapshots(COMPANYFACTS, FILINGS, {"NOPE"})
|
||||||
|
assert res.rows == []
|
||||||
|
assert res.skipped_filings == [{"accession": "NOPE", "reason": "no facts or filing metadata"}]
|
||||||
|
|
||||||
|
|
||||||
|
def test_debt_prefers_aggregate_over_parts():
|
||||||
|
rd = date(2026, 3, 28)
|
||||||
|
facts = [
|
||||||
|
Fact("us-gaap", "LongTermDebt", "USD", None, rd, 95000, 2026, "Q2"),
|
||||||
|
Fact("us-gaap", "LongTermDebtNoncurrent", "USD", None, rd, 80000, 2026, "Q2"),
|
||||||
|
Fact("us-gaap", "LongTermDebtCurrent", "USD", None, rd, 10000, 2026, "Q2"),
|
||||||
|
Fact("us-gaap", "CommercialPaper", "USD", None, rd, 5000, 2026, "Q2"),
|
||||||
|
]
|
||||||
|
# aggregate (95000) used, parts ignored; + one short-term pick (5000)
|
||||||
|
assert _compose_debt(facts, rd) == 100000
|
||||||
|
|
||||||
|
|
||||||
|
def test_cash_picks_one_st_investment_source():
|
||||||
|
rd = date(2026, 3, 28)
|
||||||
|
facts = [
|
||||||
|
Fact("us-gaap", "CashAndCashEquivalentsAtCarryingValue", "USD", None, rd, 30000, 2026, "Q2"),
|
||||||
|
Fact("us-gaap", "ShortTermInvestments", "USD", None, rd, 15000, 2026, "Q2"),
|
||||||
|
Fact("us-gaap", "MarketableSecuritiesCurrent", "USD", None, rd, 20000, 2026, "Q2"),
|
||||||
|
]
|
||||||
|
# ShortTermInvestments is first in priority -> 30000 + 15000 (not both ST tags)
|
||||||
|
assert _compose_cash(facts, rd) == 45000
|
||||||
|
|
||||||
|
|
||||||
|
RD = date(2026, 3, 28)
|
||||||
|
|
||||||
|
|
||||||
|
def _dei(end, val):
|
||||||
|
return Fact("dei", "EntityCommonStockSharesOutstanding", "shares", None, end, val, 2026, "Q2")
|
||||||
|
|
||||||
|
|
||||||
|
def _gaap_shares(end, val):
|
||||||
|
return Fact("us-gaap", "CommonStockSharesOutstanding", "shares", None, end, val, 2026, "Q2")
|
||||||
|
|
||||||
|
|
||||||
|
def test_shares_prefers_dei_cover_page():
|
||||||
|
facts = [_dei(date(2026, 4, 17), 100), _gaap_shares(RD, 999)]
|
||||||
|
assert _select_shares(facts, RD) == (100.0, date(2026, 4, 17), False)
|
||||||
|
|
||||||
|
|
||||||
|
def test_shares_falls_back_to_usgaap_at_report_date():
|
||||||
|
# Alphabet case: no dei fact; us-gaap current + a prior comparative.
|
||||||
|
facts = [_gaap_shares(date(2025, 12, 31), 888), _gaap_shares(RD, 12116)]
|
||||||
|
assert _select_shares(facts, RD) == (12116.0, RD, False) # comparative excluded
|
||||||
|
|
||||||
|
|
||||||
|
def test_shares_conflict_returns_null_ambiguous():
|
||||||
|
facts = [_dei(RD, 100), _dei(RD, 200)] # two differing consolidated values
|
||||||
|
assert _select_shares(facts, RD) == (None, None, True)
|
||||||
|
|
||||||
|
|
||||||
|
def test_shares_never_uses_weighted_average():
|
||||||
|
facts = [Fact("us-gaap", "WeightedAverageNumberOfDilutedSharesOutstanding", "shares", None, RD, 5, 2026, "Q2")]
|
||||||
|
assert _select_shares(facts, RD) == (None, None, False) # not a shares source
|
||||||
|
|
||||||
|
|
||||||
|
def test_conflicting_fiscal_context_is_rejected():
|
||||||
|
facts = [
|
||||||
|
Fact("us-gaap", "Revenues", "USD", date(2025, 9, 28), RD, 1, 2026, "Q2"),
|
||||||
|
Fact("us-gaap", "NetIncomeLoss", "USD", date(2025, 9, 28), RD, 2, 2025, "Q3"),
|
||||||
|
] # 1-1 tie between two contexts at reportDate
|
||||||
|
assert _fiscal_context(facts, RD) == (None, None)
|
||||||
|
# a clear majority wins
|
||||||
|
facts.append(Fact("us-gaap", "OperatingIncomeLoss", "USD", date(2025, 9, 28), RD, 3, 2026, "Q2"))
|
||||||
|
assert _fiscal_context(facts, RD) == (2026, "Q2")
|
||||||
|
|
||||||
|
|
||||||
|
def test_conflicting_context_skips_row():
|
||||||
|
cf = {"cik": 1, "facts": {"us-gaap": {
|
||||||
|
"Revenues": {"units": {"USD": [_dur("2025-09-28", "2026-03-28", 1, "B", fp="Q2")]}},
|
||||||
|
"NetIncomeLoss": {"units": {"USD": [_dur("2025-09-28", "2026-03-28", 2, "B", fy=2025, fp="Q3")]}},
|
||||||
|
}}}
|
||||||
|
filings = {"B": FilingMeta(RD, date(2026, 5, 1), datetime(2026, 5, 1, tzinfo=UTC), "10-Q")}
|
||||||
|
res = parse_snapshots(cf, filings, {"B"})
|
||||||
|
assert res.rows == []
|
||||||
|
assert res.skipped_filings == [{"accession": "B", "reason": "no usable period identity"}]
|
||||||
|
|
||||||
|
|
||||||
|
def test_foreign_taxonomy_and_malformed_facts_ignored():
|
||||||
|
cf = {"cik": 1, "facts": {
|
||||||
|
"us-gaap": {
|
||||||
|
"Revenues": {"units": {"USD": [_dur("2025-09-28", "2026-03-28", 500, "B")]}},
|
||||||
|
"NetIncomeLoss": {"units": {"USD": [
|
||||||
|
{"start": "2025-09-28", "end": "2026-03-28", "val": None, "fy": 2026, "fp": "Q2", "accn": "B"},
|
||||||
|
]}},
|
||||||
|
"OperatingIncomeLoss": {"units": {"USD": [
|
||||||
|
{"start": "2025-09-28", "end": "2026-03-28", "val": float("nan"), "fy": 2026, "fp": "Q2", "accn": "B"},
|
||||||
|
]}},
|
||||||
|
},
|
||||||
|
"acme": {"RevenueFromContractWithCustomerExcludingAssessedTax": {"units": {"USD": [
|
||||||
|
_dur("2025-09-28", "2026-03-28", 99999, "B"), # custom taxonomy — must be ignored
|
||||||
|
]}}},
|
||||||
|
}}
|
||||||
|
filings = {"B": FilingMeta(RD, date(2026, 5, 1), datetime(2026, 5, 1, tzinfo=UTC), "10-Q")}
|
||||||
|
res = parse_snapshots(cf, filings, {"B"})
|
||||||
|
assert res.rows[0].revenue == 500 # us-gaap Revenues, not the acme concept
|
||||||
|
assert res.rows[0].net_income is None # val None ignored
|
||||||
|
assert res.rows[0].operating_income is None # NaN ignored
|
||||||
|
|
||||||
|
|
||||||
|
def test_ambiguous_shares_produces_row_plus_note():
|
||||||
|
cf = {"cik": 1, "facts": {
|
||||||
|
"us-gaap": {"Revenues": {"units": {"USD": [_dur("2025-09-28", "2026-03-28", 500, "B")]}}},
|
||||||
|
"dei": {"EntityCommonStockSharesOutstanding": {"units": {"shares": [
|
||||||
|
{"end": "2026-04-17", "val": 100, "fy": 2026, "fp": "Q2", "accn": "B"},
|
||||||
|
{"end": "2026-04-17", "val": 200, "fy": 2026, "fp": "Q2", "accn": "B"},
|
||||||
|
]}}},
|
||||||
|
}}
|
||||||
|
filings = {"B": FilingMeta(RD, date(2026, 5, 1), datetime(2026, 5, 1, tzinfo=UTC), "10-Q")}
|
||||||
|
res = parse_snapshots(cf, filings, {"B"})
|
||||||
|
assert len(res.rows) == 1 and res.rows[0].shares_outstanding is None # row kept, shares null
|
||||||
|
assert res.field_issues == [{"accession": "B", "reason": "ambiguous shares outstanding"}]
|
||||||
|
assert res.skipped_filings == [] # a field issue is NOT a skipped filing
|
||||||
|
|
||||||
|
|
||||||
|
# Opt-in live check against real Apple companyfacts. Skips unless SEC_LIVE=1 and a
|
||||||
|
# real SEC_USER_AGENT are set (network + fair-access contact email).
|
||||||
|
@pytest.mark.skipif(
|
||||||
|
not (os.environ.get("SEC_LIVE") and os.environ.get("SEC_USER_AGENT")),
|
||||||
|
reason="set SEC_LIVE=1 + SEC_USER_AGENT to run the live SEC parser check",
|
||||||
|
)
|
||||||
|
async def test_live_apple_parse_invariants():
|
||||||
|
from app.services.sec_client import SecClient
|
||||||
|
|
||||||
|
def _dt(s):
|
||||||
|
return datetime.fromisoformat(s.replace("Z", "+00:00")) if s else None
|
||||||
|
|
||||||
|
async with SecClient(user_agent=os.environ["SEC_USER_AGENT"]) as c:
|
||||||
|
cf = await c.companyfacts(320193)
|
||||||
|
sub = await c.submissions(320193, include_history=False)
|
||||||
|
|
||||||
|
filings = {
|
||||||
|
f["accession"]: FilingMeta(
|
||||||
|
date.fromisoformat(f["report_date"]),
|
||||||
|
date.fromisoformat(f["filing_date"]),
|
||||||
|
_dt(f["acceptance_datetime"]),
|
||||||
|
f["form"],
|
||||||
|
)
|
||||||
|
for f in sub["filings"]
|
||||||
|
if f["report_date"] and f["filing_date"] and f["acceptance_datetime"]
|
||||||
|
}
|
||||||
|
rows = parse_snapshots(cf, filings, set(filings)).rows
|
||||||
|
assert len(rows) > 20
|
||||||
|
assert all(r.period_end and r.fiscal_year and r.fiscal_period for r in rows)
|
||||||
|
# YTD revenue is non-decreasing within a fiscal year
|
||||||
|
by_fy: dict[int, list] = {}
|
||||||
|
for r in rows:
|
||||||
|
if r.revenue is not None:
|
||||||
|
by_fy.setdefault(r.fiscal_year, []).append((r.fiscal_period, r.revenue))
|
||||||
|
order = {"Q1": 1, "Q2": 2, "Q3": 3, "FY": 4}
|
||||||
|
for fy, series in by_fy.items():
|
||||||
|
series.sort(key=lambda x: order[x[0]])
|
||||||
|
vals = [v for _, v in series]
|
||||||
|
assert vals == sorted(vals), f"YTD revenue not monotonic in FY{fy}: {series}"
|
||||||
|
# shares cover-date differs from period_end
|
||||||
|
latest = max(rows, key=lambda r: r.period_end)
|
||||||
|
assert latest.shares_outstanding_date != latest.period_end
|
||||||
@@ -0,0 +1,413 @@
|
|||||||
|
"""Integration tests for the SEC fundamentals importer, driven through the real
|
||||||
|
import framework with a fake SEC client (no network)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
from datetime import date, datetime, timezone
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from sqlalchemy import func, select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
import app.models # noqa: F401
|
||||||
|
from app.models.fundamental_snapshot import FundamentalSnapshot
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services.data_import import STATUS_FAILED, STATUS_PROMOTED, run_import
|
||||||
|
from app.services.sec_fundamentals_importer import SecFundamentalsImporter
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
async def engine():
|
||||||
|
fd, path = tempfile.mkstemp(suffix=".db")
|
||||||
|
os.close(fd)
|
||||||
|
eng = create_async_engine(f"sqlite+aiosqlite:///{path}")
|
||||||
|
async with eng.begin() as conn:
|
||||||
|
await conn.run_sync(Base.metadata.create_all)
|
||||||
|
try:
|
||||||
|
yield eng
|
||||||
|
finally:
|
||||||
|
await eng.dispose()
|
||||||
|
try:
|
||||||
|
os.unlink(path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _factory(engine):
|
||||||
|
return async_sessionmaker(engine, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
|
||||||
|
|
||||||
|
# --- fixture SEC data (AAPL, cik 320193) -----------------------------------
|
||||||
|
|
||||||
|
def _rev(start, end, val, fy, fp, accn):
|
||||||
|
return {"start": start, "end": end, "val": val, "fy": fy, "fp": fp, "accn": accn, "form": "10-K"}
|
||||||
|
|
||||||
|
|
||||||
|
def _shares(end, val, accn, fy, fp):
|
||||||
|
return {"end": end, "val": val, "fy": fy, "fp": fp, "accn": accn, "form": "10-K"}
|
||||||
|
|
||||||
|
|
||||||
|
def _companyfacts(rev_facts, share_facts, cik=320193):
|
||||||
|
return {
|
||||||
|
"cik": cik,
|
||||||
|
"facts": {
|
||||||
|
"us-gaap": {"RevenueFromContractWithCustomerExcludingAssessedTax": {"units": {"USD": rev_facts}}},
|
||||||
|
"dei": {"EntityCommonStockSharesOutstanding": {"units": {"shares": share_facts}}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _filing(accn, form, report, filed, accepted, is_xbrl=True):
|
||||||
|
return {"accession": accn, "form": form, "report_date": report, "filing_date": filed,
|
||||||
|
"acceptance_datetime": accepted, "is_xbrl": is_xbrl}
|
||||||
|
|
||||||
|
|
||||||
|
CF_K = _rev("2024-09-29", "2025-09-27", 416161, 2025, "FY", "K")
|
||||||
|
CF_Q1 = _rev("2025-09-28", "2025-12-27", 143756, 2026, "Q1", "Q")
|
||||||
|
SH_K = _shares("2025-10-17", 14776, "K", 2025, "FY")
|
||||||
|
SH_Q1 = _shares("2026-01-16", 14681, "Q", 2026, "Q1")
|
||||||
|
|
||||||
|
SUB_FILINGS = [
|
||||||
|
_filing("K", "10-K", "2025-09-27", "2025-10-31", "2025-10-31T10:01:26.000Z"),
|
||||||
|
_filing("Q", "10-Q", "2025-12-27", "2026-01-30", "2026-01-30T11:01:00.000Z"),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _submissions(filings):
|
||||||
|
return {"cik": 320193, "sic": "3571", "sic_description": "Electronic Computers",
|
||||||
|
"fiscal_year_end": "0926", "tickers": ["AAPL"], "filings": filings}
|
||||||
|
|
||||||
|
|
||||||
|
class FakeSecClient:
|
||||||
|
def __init__(self, *, tickers, companyfacts, submissions, latest_index, daily=None):
|
||||||
|
self._tickers = tickers
|
||||||
|
self._cf = companyfacts
|
||||||
|
self._sub = submissions
|
||||||
|
self._latest = latest_index
|
||||||
|
self._daily = daily or {}
|
||||||
|
|
||||||
|
async def __aenter__(self):
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(self, *a):
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def company_tickers(self):
|
||||||
|
return dict(self._tickers)
|
||||||
|
|
||||||
|
async def latest_index_date(self, today=None):
|
||||||
|
return self._latest
|
||||||
|
|
||||||
|
async def daily_index(self, day):
|
||||||
|
return list(self._daily.get(day, []))
|
||||||
|
|
||||||
|
async def companyfacts(self, cik):
|
||||||
|
return self._cf[int(cik)]
|
||||||
|
|
||||||
|
async def submissions(self, cik, *, include_history=False):
|
||||||
|
return self._sub[int(cik)]
|
||||||
|
|
||||||
|
|
||||||
|
def _importer(client, today=date(2026, 2, 1)):
|
||||||
|
return SecFundamentalsImporter(client_factory=lambda: client, today=today)
|
||||||
|
|
||||||
|
|
||||||
|
async def _seed(factory, symbols):
|
||||||
|
async with factory() as s:
|
||||||
|
for sym in symbols:
|
||||||
|
s.add(Ticker(symbol=sym))
|
||||||
|
await s.commit()
|
||||||
|
|
||||||
|
|
||||||
|
async def _count(factory, model):
|
||||||
|
async with factory() as s:
|
||||||
|
return (await s.execute(select(func.count()).select_from(model))).scalar_one()
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
async def test_backfill_inserts_snapshots_and_ticker_meta(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL"])
|
||||||
|
client = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS)},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(client), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_PROMOTED
|
||||||
|
assert run.source_max_date == date(2026, 1, 31)
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 2
|
||||||
|
|
||||||
|
async with factory() as s:
|
||||||
|
t = (await s.execute(select(Ticker))).scalar_one()
|
||||||
|
assert t.cik == "0000320193" and t.sic == "3571"
|
||||||
|
snaps = (await s.execute(select(FundamentalSnapshot))).scalars().all()
|
||||||
|
assert {x.fiscal_period for x in snaps} == {"FY", "Q1"}
|
||||||
|
assert all(x.import_run_id == run.id for x in snaps)
|
||||||
|
fy = next(x for x in snaps if x.fiscal_period == "FY")
|
||||||
|
assert fy.revenue == 416161 and fy.shares_outstanding == 14776
|
||||||
|
|
||||||
|
|
||||||
|
async def test_incremental_adds_only_new_filing(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL"])
|
||||||
|
backfill = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS)},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
await run_import(_importer(backfill), engine=engine)
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 2
|
||||||
|
|
||||||
|
# A new Q2 10-Q appears in the daily index and Company Facts.
|
||||||
|
cf_q2 = _rev("2025-09-28", "2026-03-28", 254940, 2026, "Q2", "Q2A")
|
||||||
|
sh_q2 = _shares("2026-04-17", 14687, "Q2A", 2026, "Q2")
|
||||||
|
incr = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1, cf_q2], [SH_K, SH_Q1, sh_q2])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS + [
|
||||||
|
_filing("Q2A", "10-Q", "2026-03-28", "2026-05-01", "2026-05-01T10:01:00.000Z")])},
|
||||||
|
latest_index=date(2026, 5, 2),
|
||||||
|
daily={date(2026, 5, 1): [{"form": "10-Q", "cik": 320193, "accession": "Q2A"}]},
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(incr, today=date(2026, 5, 3)), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_PROMOTED
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 3 # only Q2A added
|
||||||
|
async with factory() as s:
|
||||||
|
q2 = (await s.execute(
|
||||||
|
select(FundamentalSnapshot).where(FundamentalSnapshot.accession == "Q2A")
|
||||||
|
)).scalar_one()
|
||||||
|
assert q2.fiscal_period == "Q2" and q2.revenue == 254940
|
||||||
|
|
||||||
|
|
||||||
|
async def test_consistency_gate_fails_when_facts_lag_index(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL"])
|
||||||
|
backfill = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS)},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
await run_import(_importer(backfill), engine=engine)
|
||||||
|
|
||||||
|
# Index + submissions list an XBRL filing "GHOST" that Company Facts lacks.
|
||||||
|
incr = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])}, # no GHOST
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS + [
|
||||||
|
_filing("GHOST", "10-Q", "2026-03-28", "2026-05-01", "2026-05-01T10:01:00.000Z")])},
|
||||||
|
latest_index=date(2026, 5, 2),
|
||||||
|
daily={date(2026, 5, 1): [{"form": "10-Q", "cik": 320193, "accession": "GHOST"}]},
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(incr, today=date(2026, 5, 3)), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_FAILED
|
||||||
|
assert "Company Facts" in (run.error_details or "")
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 2 # nothing new written
|
||||||
|
|
||||||
|
|
||||||
|
async def test_non_xbrl_amendment_skipped_not_failed(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL"])
|
||||||
|
backfill = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS)},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
await run_import(_importer(backfill), engine=engine)
|
||||||
|
|
||||||
|
incr = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS + [
|
||||||
|
_filing("AMD", "10-K/A", "2025-09-27", "2026-05-01", "2026-05-01T10:01:00.000Z", is_xbrl=False)])},
|
||||||
|
latest_index=date(2026, 5, 2),
|
||||||
|
daily={date(2026, 5, 1): [{"form": "10-K/A", "cik": 320193, "accession": "AMD"}]},
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(incr, today=date(2026, 5, 3)), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_PROMOTED # non-XBRL amendment is skipped, not a failure
|
||||||
|
assert "skipped_non_xbrl" in (run.validation_json or "")
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 2
|
||||||
|
|
||||||
|
|
||||||
|
async def test_failed_backfill_leaves_tickers_unwritten(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL", "MSFT", "NVDA"]) # 3 resolve, only AAPL yields rows
|
||||||
|
client = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193, "MSFT": 789019, "NVDA": 1045810},
|
||||||
|
companyfacts={
|
||||||
|
320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1]),
|
||||||
|
789019: {"cik": 789019, "facts": {}}, # no facts -> no rows
|
||||||
|
1045810: {"cik": 1045810, "facts": {}},
|
||||||
|
},
|
||||||
|
submissions={
|
||||||
|
320193: _submissions(SUB_FILINGS),
|
||||||
|
789019: {"cik": 789019, "sic": "7372", "sic_description": "x", "filings": []},
|
||||||
|
1045810: {"cik": 1045810, "sic": "3674", "sic_description": "y", "filings": []},
|
||||||
|
},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(client), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_FAILED # coverage 1/3 < 50%
|
||||||
|
assert "coverage" in (run.error_details or "")
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 0
|
||||||
|
# read-only resolution: no ticker cik/sic written on a failed run
|
||||||
|
async with factory() as s:
|
||||||
|
assert all(t.cik is None and t.sic is None for t in (await s.execute(select(Ticker))).scalars())
|
||||||
|
|
||||||
|
|
||||||
|
async def test_index_gap_over_45_days_loses_no_filings(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL"])
|
||||||
|
backfill = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS)},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
await run_import(_importer(backfill), engine=engine)
|
||||||
|
|
||||||
|
# 74-day gap; the filing sits in the OLD part (>45d before latest).
|
||||||
|
cf_q2 = _rev("2025-09-28", "2026-03-28", 254940, 2026, "Q2", "OLD")
|
||||||
|
sh_q2 = _shares("2026-04-17", 14687, "OLD", 2026, "Q2")
|
||||||
|
incr = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1, cf_q2], [SH_K, SH_Q1, sh_q2])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS + [
|
||||||
|
_filing("OLD", "10-Q", "2026-03-28", "2026-02-10", "2026-02-10T10:01:00.000Z")])},
|
||||||
|
latest_index=date(2026, 4, 15),
|
||||||
|
daily={date(2026, 2, 10): [{"form": "10-Q", "cik": 320193, "accession": "OLD"}]},
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(incr, today=date(2026, 4, 16)), engine=engine)
|
||||||
|
assert run.status == STATUS_PROMOTED
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 3 # the old-gap filing was NOT lost
|
||||||
|
|
||||||
|
|
||||||
|
async def test_newly_added_issuer_backfills_without_filing(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL"])
|
||||||
|
backfill = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])},
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS)},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
await run_import(_importer(backfill), engine=engine)
|
||||||
|
|
||||||
|
# MSFT added to the universe later; it did NOT file (not in the daily index).
|
||||||
|
await _seed(factory, ["MSFT"])
|
||||||
|
msft_rev = _rev("2024-07-01", "2025-06-30", 270000, 2025, "FY", "M")
|
||||||
|
msft_sh = _shares("2025-07-15", 7400, "M", 2025, "FY")
|
||||||
|
incr = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193, "MSFT": 789019},
|
||||||
|
companyfacts={
|
||||||
|
320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1]),
|
||||||
|
789019: _companyfacts([msft_rev], [msft_sh], cik=789019),
|
||||||
|
},
|
||||||
|
submissions={
|
||||||
|
320193: _submissions(SUB_FILINGS),
|
||||||
|
789019: {"cik": 789019, "sic": "7372", "sic_description": "Prepackaged Software",
|
||||||
|
"filings": [_filing("M", "10-K", "2025-06-30", "2025-07-30", "2025-07-30T10:00:00.000Z")]},
|
||||||
|
},
|
||||||
|
# SAME index date as the prior run and no filing: only the universe
|
||||||
|
# fingerprint (MSFT added) changes the revision, so this proves the
|
||||||
|
# fingerprint alone prevents starvation.
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
daily={}, # MSFT did not file
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(incr, today=date(2026, 2, 1)), engine=engine)
|
||||||
|
assert run.status == STATUS_PROMOTED
|
||||||
|
async with factory() as s:
|
||||||
|
msft = (await s.execute(
|
||||||
|
select(FundamentalSnapshot).where(FundamentalSnapshot.cik == "0000789019")
|
||||||
|
)).scalars().all()
|
||||||
|
assert len(msft) == 1 and msft[0].revenue == 270000 # full-history backfill despite no filing
|
||||||
|
|
||||||
|
|
||||||
|
async def test_malformed_companyfacts_fails_validation(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL", "MSFT"])
|
||||||
|
client = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193, "MSFT": 789019},
|
||||||
|
companyfacts={
|
||||||
|
320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1]),
|
||||||
|
789019: {"cik": 789019}, # malformed — no "facts" structure
|
||||||
|
},
|
||||||
|
submissions={
|
||||||
|
320193: _submissions(SUB_FILINGS),
|
||||||
|
789019: {"cik": 789019, "sic": "7372", "sic_description": "x", "filings": []},
|
||||||
|
},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(client), engine=engine)
|
||||||
|
assert run.status == STATUS_FAILED
|
||||||
|
assert "malformed" in (run.error_details or "")
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 0 # nothing promoted
|
||||||
|
|
||||||
|
|
||||||
|
async def test_missing_units_structure_fails_validation(engine):
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL", "MSFT"])
|
||||||
|
client = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193, "MSFT": 789019},
|
||||||
|
companyfacts={
|
||||||
|
320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1]),
|
||||||
|
# facts present, but a concept is missing its units mapping
|
||||||
|
789019: {"cik": 789019, "facts": {"us-gaap": {"Revenues": {"label": "Revenues"}}}},
|
||||||
|
},
|
||||||
|
submissions={
|
||||||
|
320193: _submissions(SUB_FILINGS),
|
||||||
|
789019: {"cik": 789019, "sic": "7372", "sic_description": "x", "filings": []},
|
||||||
|
},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(client), engine=engine)
|
||||||
|
assert run.status == STATUS_FAILED
|
||||||
|
assert "malformed" in (run.error_details or "")
|
||||||
|
assert "units" in (run.validation_json or "")
|
||||||
|
assert await _count(factory, FundamentalSnapshot) == 0
|
||||||
|
|
||||||
|
|
||||||
|
async def test_discrepancy_in_shares_is_detected_and_reported(engine):
|
||||||
|
from app.models.system_event import SystemEvent
|
||||||
|
utc = timezone.utc
|
||||||
|
factory = _factory(engine)
|
||||||
|
await _seed(factory, ["AAPL"])
|
||||||
|
# Pre-store accession K matching what the parser will produce EXCEPT shares.
|
||||||
|
async with factory() as s:
|
||||||
|
s.add(FundamentalSnapshot(
|
||||||
|
cik="0000320193", accession="K", form="10-K", filed_date=date(2025, 10, 31),
|
||||||
|
accepted_at=datetime(2025, 10, 31, 10, 1, 26, tzinfo=utc), period_start=date(2024, 9, 29),
|
||||||
|
period_end=date(2025, 9, 27), fiscal_year=2025, fiscal_period="FY", revenue=416161.0,
|
||||||
|
shares_outstanding=999.0, import_run_id=1, created_at=datetime(2025, 10, 31, tzinfo=utc)))
|
||||||
|
await s.commit()
|
||||||
|
|
||||||
|
client = FakeSecClient(
|
||||||
|
tickers={"AAPL": 320193},
|
||||||
|
companyfacts={320193: _companyfacts([CF_K, CF_Q1], [SH_K, SH_Q1])}, # SH_K = 14776 != 999
|
||||||
|
submissions={320193: _submissions(SUB_FILINGS)},
|
||||||
|
latest_index=date(2026, 1, 31),
|
||||||
|
)
|
||||||
|
run = await run_import(_importer(client), engine=engine)
|
||||||
|
|
||||||
|
assert run.status == STATUS_PROMOTED # a discrepancy is reported, not a failure
|
||||||
|
assert '"discrepancy_count": 1' in (run.validation_json or "")
|
||||||
|
assert "shares_outstanding" in (run.validation_json or "")
|
||||||
|
async with factory() as s:
|
||||||
|
k = (await s.execute(select(FundamentalSnapshot).where(FundamentalSnapshot.accession == "K"))).scalar_one()
|
||||||
|
assert k.shares_outstanding == 999.0 and k.import_run_id == 1 # immutable — not overwritten
|
||||||
|
events = (await s.execute(select(SystemEvent).where(SystemEvent.code == "snapshot_discrepancy"))).scalars().all()
|
||||||
|
assert len(events) == 1 and events[0].severity == "warning"
|
||||||
@@ -0,0 +1,120 @@
|
|||||||
|
"""Tests for CIK/SIC resolution (read-only), apply-in-promote, and the
|
||||||
|
composite-revision fingerprint."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from sqlalchemy import select
|
||||||
|
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker, create_async_engine
|
||||||
|
|
||||||
|
from app.database import Base
|
||||||
|
import app.models # noqa: F401
|
||||||
|
from app.models.ticker import Ticker
|
||||||
|
from app.services import sec_universe as su
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
async def factory():
|
||||||
|
fd, path = tempfile.mkstemp(suffix=".db")
|
||||||
|
os.close(fd)
|
||||||
|
eng = create_async_engine(f"sqlite+aiosqlite:///{path}")
|
||||||
|
async with eng.begin() as conn:
|
||||||
|
await conn.run_sync(Base.metadata.create_all)
|
||||||
|
try:
|
||||||
|
yield async_sessionmaker(eng, class_=AsyncSession, expire_on_commit=False)
|
||||||
|
finally:
|
||||||
|
await eng.dispose()
|
||||||
|
try:
|
||||||
|
os.unlink(path)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class FakeSecClient:
|
||||||
|
def __init__(self, tickers, submissions=None):
|
||||||
|
self._tickers = tickers
|
||||||
|
self._submissions = submissions or {}
|
||||||
|
|
||||||
|
async def company_tickers(self):
|
||||||
|
return dict(self._tickers)
|
||||||
|
|
||||||
|
async def submissions(self, cik, *, include_history=False):
|
||||||
|
return self._submissions[int(cik)]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_resolve_ciks_is_read_only_and_proposes_updates(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
for sym in ["AAPL", "GOOGL", "GOOG", "ZZZZ"]: # ZZZZ not in SEC
|
||||||
|
s.add(Ticker(symbol=sym))
|
||||||
|
await s.commit()
|
||||||
|
|
||||||
|
client = FakeSecClient({"AAPL": 320193, "GOOGL": 1652044, "GOOG": 1652044})
|
||||||
|
async with factory() as s:
|
||||||
|
resolved = await su.resolve_ciks(s, client)
|
||||||
|
assert not s.dirty and not s.new # NOTHING mutated during resolution
|
||||||
|
await s.rollback()
|
||||||
|
|
||||||
|
assert resolved.symbol_to_cik == {"AAPL": 320193, "GOOGL": 1652044, "GOOG": 1652044}
|
||||||
|
assert len(resolved.cik_updates) == 3 # AAPL, GOOGL, GOOG (ZZZZ unresolved)
|
||||||
|
assert set(resolved.cik_to_ticker_ids) == {320193, 1652044}
|
||||||
|
|
||||||
|
# Read-only really means the DB is untouched until apply.
|
||||||
|
async with factory() as s:
|
||||||
|
ciks = {t.symbol: t.cik for t in (await s.execute(select(Ticker))).scalars()}
|
||||||
|
assert all(v is None for v in ciks.values())
|
||||||
|
|
||||||
|
|
||||||
|
async def test_apply_ticker_updates_writes_cik_and_sic(factory):
|
||||||
|
async with factory() as s:
|
||||||
|
for sym in ["GOOGL", "GOOG"]:
|
||||||
|
s.add(Ticker(symbol=sym))
|
||||||
|
await s.commit()
|
||||||
|
|
||||||
|
client = FakeSecClient(
|
||||||
|
{"GOOGL": 1652044, "GOOG": 1652044},
|
||||||
|
submissions={1652044: {"sic": "7370", "sic_description": "Services-Computer"}},
|
||||||
|
)
|
||||||
|
async with factory() as s:
|
||||||
|
resolved = await su.resolve_ciks(s, client)
|
||||||
|
sic_updates = await su.fetch_sic_updates(client, resolved.cik_to_ticker_ids)
|
||||||
|
counts = await su.apply_ticker_updates(s, resolved, sic_updates)
|
||||||
|
await s.commit()
|
||||||
|
|
||||||
|
assert counts == {"cik_updates": 2, "sic_updates": 2}
|
||||||
|
async with factory() as s:
|
||||||
|
rows = {t.symbol: (t.cik, t.sic, t.sic_description) for t in (await s.execute(select(Ticker))).scalars()}
|
||||||
|
assert rows["GOOGL"] == ("0001652044", "7370", "Services-Computer")
|
||||||
|
assert rows["GOOG"] == ("0001652044", "7370", "Services-Computer")
|
||||||
|
|
||||||
|
|
||||||
|
async def test_fetch_sic_updates_is_read_only(factory):
|
||||||
|
client = FakeSecClient({}, submissions={1: {"sic": "1", "sic_description": "x"}})
|
||||||
|
updates = await su.fetch_sic_updates(client, {1: [10, 11]})
|
||||||
|
assert updates == [(10, "1", "x"), (11, "1", "x")] # proposals only, no DB touched
|
||||||
|
|
||||||
|
|
||||||
|
def test_universe_fingerprint_changes_on_membership():
|
||||||
|
a = su.universe_fingerprint({"AAPL": 320193, "MSFT": 789019})
|
||||||
|
same = su.universe_fingerprint({"MSFT": 789019, "AAPL": 320193}) # order-independent
|
||||||
|
added = su.universe_fingerprint({"AAPL": 320193, "MSFT": 789019, "NVDA": 1045810})
|
||||||
|
remapped = su.universe_fingerprint({"AAPL": 999, "MSFT": 789019})
|
||||||
|
assert a == same
|
||||||
|
assert a != added # new ticker forces a new revision
|
||||||
|
assert a != remapped # changed CIK mapping forces a new revision
|
||||||
|
|
||||||
|
|
||||||
|
def test_compose_revision_rejects_missing_index_date():
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
su.compose_revision(None, "abc", {"AAPL": 320193})
|
||||||
|
|
||||||
|
|
||||||
|
def test_compose_revision_and_index_hash():
|
||||||
|
rows = [{"cik": 320193, "accession": "a-1"}, {"cik": 66740, "accession": "b-2"}]
|
||||||
|
h1 = su.index_content_hash(rows)
|
||||||
|
h2 = su.index_content_hash(list(reversed(rows)))
|
||||||
|
assert h1 == h2 # order-independent
|
||||||
|
rev = su.compose_revision("2026-07-21", h1, {"AAPL": 320193})
|
||||||
|
assert rev.startswith("2026-07-21:") and rev.count(":") == 2
|
||||||
Reference in New Issue
Block a user