Fix/sec fundamentals parity gaps #2

Merged
dennisthiessen merged 4 commits from fix/sec-fundamentals-parity-gaps into main 2026-07-24 10:58:17 +02:00
14 changed files with 22634 additions and 33 deletions
@@ -0,0 +1,43 @@
"""fundamental_snapshots.weighted_avg_diluted_shares — market-cap fallback
Revision ID: 027
Revises: 026
Create Date: 2026-07-24 00:00:00.000000
Multi-class issuers report the cover-page share count per share class. That is a
dimensional fact and Company Facts is non-dimensional, so it is absent entirely:
META has never tagged it, CMCSA stops in 2009, BRK-B in 2011, CHTR in 2016 (when
the Time Warner Cable deal made it multi-class). `shares_outstanding` is
therefore null for a large slice of the mega-cap universe, which silently removes
both `market_cap_est` and `fcf_yield`.
The weighted-average diluted count is always present (EPS requires it) and is
consolidated across classes. Measured against issuers where the true
point-in-time count IS available, it lands within ~0.6%: GOOGL 0.9936, MRNA
1.0045, AAPL 0.9974, MSFT 0.9978.
Stored as its own column rather than backfilled into `shares_outstanding`, so the
point-in-time column keeps its strict meaning and the fallback stays an explicit,
labelled read-time decision. Existing rows are null until a reparse.
"""
from typing import Sequence, Union
from alembic import op
import sqlalchemy as sa
revision: str = "027"
down_revision: Union[str, None] = "026"
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None
def upgrade() -> None:
op.add_column(
"fundamental_snapshots",
sa.Column("weighted_avg_diluted_shares", sa.Float(), nullable=True),
)
def downgrade() -> None:
op.drop_column("fundamental_snapshots", "weighted_avg_diluted_shares")
+10 -2
View File
@@ -12,8 +12,10 @@ class FundamentalSnapshot(Base):
Keyed by issuer (CIK), not ticker — multi-class issuers (GOOG/GOOGL) share Keyed by issuer (CIK), not ticker — multi-class issuers (GOOG/GOOGL) share
one CIK and one set of fundamentals; the ``tickers.cik`` column is the only one CIK and one set of fundamentals; the ``tickers.cik`` column is the only
join point. Amendments are retained: every accession is a distinct immutable join point. Amendments are retained: every accession is a distinct immutable
row, and readers pick the newest valid ``accepted_at`` per row, and readers resolve (cik, fiscal_year, fiscal_period) at read time by
(cik, fiscal_year, fiscal_period) at read time — no flags, no mutation. taking the newest ``accepted_at`` **per field**, falling back to the newest
accession that actually reports one — a partial amendment (a 10-K/A adding
Part III reports no financial facts) must not blank the period — no flags, no mutation.
**Facts are stored as the filing reports them, never as derived quarters.** **Facts are stored as the filing reports them, never as derived quarters.**
Duration facts (revenue, net_income, operating_income, diluted_eps, cfo, Duration facts (revenue, net_income, operating_income, diluted_eps, cfo,
@@ -70,6 +72,12 @@ class FundamentalSnapshot(Base):
# reported "as of" its own date, which can differ from period_end — store it # reported "as of" its own date, which can differ from period_end — store it
# so market cap uses the right point-in-time count. # so market cap uses the right point-in-time count.
shares_outstanding_date: Mapped[date | None] = mapped_column(Date, nullable=True) shares_outstanding_date: Mapped[date | None] = mapped_column(Date, nullable=True)
# Weighted-average diluted count for the filing's most recent quarter — the
# market-cap fallback when the cover-page count is absent, which it always is
# for multi-class issuers (per-class facts are dimensional, and companyfacts
# is not). An average is not cumulative, so unlike the duration facts above
# this is NOT a YTD value: it is the shortest-span fact ending at period_end.
weighted_avg_diluted_shares: Mapped[float | None] = mapped_column(Float, nullable=True)
import_run_id: Mapped[int | None] = mapped_column( import_run_id: Mapped[int | None] = mapped_column(
ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True
+7 -1
View File
@@ -145,11 +145,17 @@ async def run_import(
importer: SourceImporter, importer: SourceImporter,
*, *,
engine: AsyncEngine | None = None, engine: AsyncEngine | None = None,
force: bool = False,
) -> DataImportRun | None: ) -> DataImportRun | None:
"""Run one import for ``importer``. """Run one import for ``importer``.
Returns the recorded ``DataImportRun`` (promoted / no_op / failed), or None Returns the recorded ``DataImportRun`` (promoted / no_op / failed), or None
when the per-source advisory lock is already held (another run is active). when the per-source advisory lock is already held (another run is active).
``force`` runs even when the revision is unchanged. The revision tracks the
*source*, so a re-import driven by a change on our side — a parser fix that
makes stored rows stale — is a no_op under the normal gate. Manually invoked
only; scheduled jobs must leave it False so an unchanged source stays a no_op.
""" """
engine = engine or app_engine engine = engine or app_engine
source = importer.source source = importer.source
@@ -189,7 +195,7 @@ async def run_import(
revision = await importer.detect_revision(session) revision = await importer.detect_revision(session)
run.revision = revision run.revision = revision
last_rev = await _last_promoted_revision(session, source) last_rev = await _last_promoted_revision(session, source)
if revision is not None and revision == last_rev: if not force and revision is not None and revision == last_rev:
run.status = STATUS_NO_OP run.status = STATUS_NO_OP
run.completed_at = _now() run.completed_at = _now()
await session.commit() await session.commit()
+11
View File
@@ -155,6 +155,17 @@ def _build_valuation(derived, subject_price, peer_derived, peer_price_by_cik, tw
"pe": _round(pe, 2), "pe": _round(pe, 2),
"fcf_yield": _round(fcf_yield, 2), "fcf_yield": _round(fcf_yield, 2),
"market_cap_est": _round(market_cap, 0), "market_cap_est": _round(market_cap, 0),
# market_cap_est and fcf_yield both rest on the share count. When it came
# from the weighted-average diluted fallback (multi-class issuers, whose
# per-class cover-page count is absent from companyfacts), say so rather
# than presenting a period average as a point-in-time count.
"shares_estimated": bool(
market_cap is not None and derived.shares_outstanding_estimated
),
# A null P/E is ambiguous: no earnings data, or earnings we deliberately
# suppressed. Only the latter carries a caveat, so a split-contaminated
# TTM says why instead of looking like missing data.
"pe_caveat": derived.ttm_diluted_eps_caveat if pe is None else None,
"pe_industry": pe_industry, "pe_industry": pe_industry,
"fcf_yield_industry": fcf_yield_industry, "fcf_yield_industry": fcf_yield_industry,
"price_date": _iso(price_date), "price_date": _iso(price_date),
+98 -12
View File
@@ -8,8 +8,10 @@ schema decision. No I/O, no DB: it takes an issuer's snapshot rows (ORM rows or
any objects with the same attributes) and returns structured metrics. any objects with the same attributes) and returns structured metrics.
Rules: Rules:
- **Amendment selection:** for each (fiscal_year, fiscal_period), the row with - **Amendment selection:** for each (fiscal_year, fiscal_period), the newest
the newest `accepted_at` wins. `accepted_at` wins **per field**, falling back to the newest row that actually
reports one. A partial amendment (a 10-K/A adding Part III carries no financial
facts) must not blank the period.
- **Discrete quarter** = YTD(Qn) YTD(Qn1); Q1 = YTD(Q1); **Q4 = YTD(FY) - **Discrete quarter** = YTD(Qn) YTD(Qn1); Q1 = YTD(Q1); **Q4 = YTD(FY)
YTD(Q3)**. Any missing period → the derived value is null, never partial. YTD(Q3)**. Any missing period → the derived value is null, never partial.
- **TTM** = sum of the trailing four discrete quarters ending at a period. - **TTM** = sum of the trailing four discrete quarters ending at a period.
@@ -21,6 +23,7 @@ from __future__ import annotations
from dataclasses import dataclass, field from dataclasses import dataclass, field
from datetime import date from datetime import date
from types import SimpleNamespace
from typing import Any, Iterable from typing import Any, Iterable
_FP_TO_Q = {"Q1": 1, "Q2": 2, "Q3": 3, "FY": 4} _FP_TO_Q = {"Q1": 1, "Q2": 2, "Q3": 3, "FY": 4}
@@ -38,6 +41,20 @@ _FLOW_FIELDS = (
"revenue", "net_income", "operating_income", "diluted_eps", "cfo", "capex", "revenue", "net_income", "operating_income", "diluted_eps", "cfo", "capex",
"depreciation_amortization", "depreciation_amortization",
) )
# Reported facts resolved independently across a period's accessions (see
# _merge_amendments); period identity/provenance is taken from the newest one.
_MERGED_FIELDS = (
*_FLOW_FIELDS,
"cash_and_st_investments", "total_debt", "shares_outstanding",
"shares_outstanding_date", "weighted_avg_diluted_shares",
# period_start is set alongside revenue by the parser, so it follows the same
# fallback: a bare amendment reports neither and must not blank it.
"period_start",
)
_CARRIED_FIELDS = (
"fiscal_year", "fiscal_period", "period_end", "filed_date",
"accepted_at", "form", "accession", "cik",
)
@dataclass @dataclass
@@ -60,8 +77,15 @@ class DerivedFundamentals:
metrics: dict[str, MetricSeries] = field(default_factory=dict) metrics: dict[str, MetricSeries] = field(default_factory=dict)
# request-time valuation inputs (ratios are computed in the API with price) # request-time valuation inputs (ratios are computed in the API with price)
ttm_diluted_eps: float | None = None ttm_diluted_eps: float | None = None
# Set when ttm_diluted_eps was suppressed rather than simply unavailable.
ttm_diluted_eps_caveat: str | None = None
ttm_fcf: float | None = None ttm_fcf: float | None = None
shares_outstanding: float | None = None shares_outstanding: float | None = None
# True when shares_outstanding came from the weighted-average diluted count
# because the point-in-time cover-page count was absent (always so for
# multi-class issuers). Consumers must label anything derived from it as
# estimated — it is a period average, not a point-in-time count.
shares_outstanding_estimated: bool = False
latest_period_end: date | None = None latest_period_end: date | None = None
latest_filed_date: date | None = None latest_filed_date: date | None = None
@@ -85,6 +109,15 @@ def derive(snapshots: Iterable[Any]) -> DerivedFundamentals:
result.latest_period_end = latest_row.period_end result.latest_period_end = latest_row.period_end
result.latest_filed_date = latest_row.filed_date result.latest_filed_date = latest_row.filed_date
result.shares_outstanding = getattr(latest_row, "shares_outstanding", None) result.shares_outstanding = getattr(latest_row, "shares_outstanding", None)
if result.shares_outstanding is None:
# Multi-class issuers (META, CMCSA, BRK-B, CHTR, FOXA, NWSA, LEN) report
# the cover-page count per class, which is dimensional and so absent from
# companyfacts — leaving market cap and FCF yield silently unavailable for
# some of the largest names. The weighted-average diluted count is always
# present and within ~0.6% of the true count where both exist, so fall
# back to it and mark the result estimated rather than show nothing.
result.shares_outstanding = getattr(latest_row, "weighted_avg_diluted_shares", None)
result.shares_outstanding_estimated = result.shares_outstanding is not None
result.ttm_diluted_eps = _ttm(discrete["diluted_eps"], *latest) result.ttm_diluted_eps = _ttm(discrete["diluted_eps"], *latest)
ttm_cfo = _ttm(discrete["cfo"], *latest) ttm_cfo = _ttm(discrete["cfo"], *latest)
ttm_capex = _ttm(discrete["capex"], *latest) ttm_capex = _ttm(discrete["capex"], *latest)
@@ -102,7 +135,14 @@ def derive(snapshots: Iterable[Any]) -> DerivedFundamentals:
"net_debt_to_ebitda": _leverage_series(selected, discrete, tape), "net_debt_to_ebitda": _leverage_series(selected, discrete, tape),
"share_count_change_yoy": _share_change_series(selected, tape), "share_count_change_yoy": _share_change_series(selected, tape),
} }
_guard_split_sensitive_metrics(result.metrics) # TTM EPS sums four quarters of *per-share* values, so a split inside that
# window mixes pre- and post-split units — the same distortion the guard
# already catches for the series, and the one that produced BKNG's P/E of
# 1.10. Left unguarded it does not merely mislead: a nonsense-low P/E clamps
# to a perfect 100 fundamental sub-score, so it must null out like the rest.
if _guard_split_sensitive_metrics(result.metrics):
result.ttm_diluted_eps = None
result.ttm_diluted_eps_caveat = SPLIT_SENSITIVE_CAVEAT
for series in result.metrics.values(): for series in result.metrics.values():
series.period_end = latest_row.period_end series.period_end = latest_row.period_end
series.filed_date = latest_row.filed_date series.filed_date = latest_row.filed_date
@@ -112,17 +152,57 @@ def derive(snapshots: Iterable[Any]) -> DerivedFundamentals:
# -- period selection -------------------------------------------------------- # -- period selection --------------------------------------------------------
def _select_latest_per_period(snapshots: Iterable[Any]) -> dict[tuple[int, str], Any]: def _select_latest_per_period(snapshots: Iterable[Any]) -> dict[tuple[int, str], Any]:
best: dict[tuple[int, str], Any] = {} grouped: dict[tuple[int, str], list[Any]] = {}
for row in snapshots: for row in snapshots:
fp = getattr(row, "fiscal_period", None) fp = getattr(row, "fiscal_period", None)
fy = getattr(row, "fiscal_year", None) fy = getattr(row, "fiscal_year", None)
if fp not in _FP_TO_Q or fy is None: if fp not in _FP_TO_Q or fy is None:
continue continue
key = (fy, fp) grouped.setdefault((fy, fp), []).append(row)
cur = best.get(key) return {key: _merge_amendments(rows) for key, rows in grouped.items()}
if cur is None or _accepted(row) > _accepted(cur):
best[key] = row
return best def _merge_amendments(rows: list[Any]) -> Any:
"""Resolve one period from its accessions: newest wins, per field.
Amendments are frequently partial — a 10-K/A filed only to add Part III
reports no financial facts at all. Taking the newest accession wholesale
would blank every field it omits and null the period downstream (and with
it TTM and YoY, which need an unbroken quarter chain), so each field falls
back to the newest accession that actually reports it.
Only rows sharing the newest row's ``period_end`` are merged. A same-key row
covering a *different* period is a mislabelled filing, not an amendment, and
blending the two would silently mix fiscal years.
"""
if len(rows) == 1:
return rows[0]
ordered = sorted(rows, key=_amendment_order, reverse=True) # newest first
newest = ordered[0]
same_period = [
row
for row in ordered
if getattr(row, "period_end", None) == getattr(newest, "period_end", None)
]
if len(same_period) == 1:
return newest
merged = SimpleNamespace(**{name: getattr(newest, name, None) for name in _CARRIED_FIELDS})
for name in _MERGED_FIELDS:
merged_value = None
for row in same_period: # newest first
value = getattr(row, name, None)
if value is not None:
merged_value = value
break
setattr(merged, name, merged_value)
return merged
def _amendment_order(row: Any) -> tuple[bool, Any]:
# (has-timestamp, timestamp) so a row without one sorts oldest instead of
# raising when compared against a row that has one.
accepted = _accepted(row)
return (accepted is not None, accepted)
def _accepted(row: Any): def _accepted(row: Any):
@@ -255,18 +335,21 @@ def _share_change_series(selected, tape) -> MetricSeries:
return _series(pts) return _series(pts)
def _guard_split_sensitive_metrics(metrics: dict[str, MetricSeries]) -> None: def _guard_split_sensitive_metrics(metrics: dict[str, MetricSeries]) -> bool:
"""Suppress historical comparisons likely distorted by a corporate action. """Suppress historical comparisons likely distorted by a corporate action.
Company Facts has no point-in-time split factors. A large YoY share-count Company Facts has no point-in-time split factors. A large YoY share-count
move can therefore make both the point-in-time share comparison and move can therefore make both the point-in-time share comparison and
per-share EPS growth non-comparable. Keep the raw facts in snapshots, but per-share EPS growth non-comparable. Keep the raw facts in snapshots, but
expose nulls plus an explicit caveat in the user-facing derived series. expose nulls plus an explicit caveat in the user-facing derived series.
Returns True when the *latest* period is suspect, so callers can apply the
same suppression to per-share scalars derived from that window.
""" """
shares = metrics.get("share_count_change_yoy") shares = metrics.get("share_count_change_yoy")
eps = metrics.get("eps_growth_yoy") eps = metrics.get("eps_growth_yoy")
if shares is None or eps is None: if shares is None or eps is None:
return return False
suspect_periods = { suspect_periods = {
point.period_end point.period_end
@@ -275,18 +358,21 @@ def _guard_split_sensitive_metrics(metrics: dict[str, MetricSeries]) -> None:
and abs(point.value) >= SPLIT_SUSPECT_SHARE_CHANGE_PCT and abs(point.value) >= SPLIT_SUSPECT_SHARE_CHANGE_PCT
} }
if not suspect_periods: if not suspect_periods:
return return False
latest_suspect = False
for series in (shares, eps): for series in (shares, eps):
latest_guarded = bool( latest_guarded = bool(
series.history and series.history[-1].period_end in suspect_periods series.history and series.history[-1].period_end in suspect_periods
) )
latest_suspect = latest_suspect or latest_guarded
for point in series.history: for point in series.history:
if point.period_end in suspect_periods: if point.period_end in suspect_periods:
point.value = None point.value = None
series.value = series.history[-1].value if series.history else None series.value = series.history[-1].value if series.history else None
if latest_guarded: if latest_guarded:
series.caveat = SPLIT_SENSITIVE_CAVEAT series.caveat = SPLIT_SENSITIVE_CAVEAT
return latest_suspect
def _net_debt(row: Any) -> float | None: def _net_debt(row: Any) -> float | None:
+170 -6
View File
@@ -8,6 +8,9 @@ fixture and verifiable against a real companyfacts pull.
The load-bearing rules (design Decision 2 + review): The load-bearing rules (design Decision 2 + review):
- Period identity comes from `end == submissions.reportDate`, never `fy/fp` - Period identity comes from `end == submissions.reportDate`, never `fy/fp`
(fy/fp is the *filing's* context; comparatives inside a filing repeat it). (fy/fp is the *filing's* context; comparatives inside a filing repeat it).
This applies to the stored `fiscal_year`/`fiscal_period` too: they are derived
from `reportDate` against the issuer's `fiscalYearEnd` (see `_period_identity`),
because SEC's fy/fp collide and invert often enough to break the quarter chain.
- Duration facts are stored as **cumulative YTD**: pick the fact whose span - Duration facts are stored as **cumulative YTD**: pick the fact whose span
matches the fiscal-period-to-date length (Q1≈3mo … FY≈12mo) within tolerance. matches the fiscal-period-to-date length (Q1≈3mo … FY≈12mo) within tolerance.
If no YTD-length fact exists, store null — never a discrete masquerading as YTD. If no YTD-length fact exists, store null — never a discrete masquerading as YTD.
@@ -15,7 +18,10 @@ The load-bearing rules (design Decision 2 + review):
is a single consolidated value: the cover-page `dei` fact (its own cover-date is a single consolidated value: the cover-page `dei` fact (its own cover-date
`end` stored separately) if present, else `us-gaap:CommonStockSharesOutstanding` `end` stored separately) if present, else `us-gaap:CommonStockSharesOutstanding`
at period end (e.g. Alphabet has no `dei` fact) — never a class sum or the at period end (e.g. Alphabet has no `dei` fact) — never a class sum or the
weighted-average/diluted count. weighted-average/diluted count. Multi-class issuers report it per class, which
is dimensional and therefore absent from companyfacts entirely, so
`weighted_avg_diluted_shares` is stored alongside as an explicit fallback for
market cap — a separate column, never backfilled into `shares_outstanding`.
- Cash and debt composites are aggregate-first and mutually exclusive (each - Cash and debt composites are aggregate-first and mutually exclusive (each
source tag counted at most once). source tag counted at most once).
@@ -29,7 +35,7 @@ from __future__ import annotations
import logging import logging
import math import math
from dataclasses import dataclass, field from dataclasses import dataclass, field
from datetime import date, datetime from datetime import date, datetime, timedelta
from typing import Any, NamedTuple from typing import Any, NamedTuple
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -37,14 +43,35 @@ logger = logging.getLogger(__name__)
# Expected YTD span (days) per fiscal period; a duration fact must land within # Expected YTD span (days) per fiscal period; a duration fact must land within
# tolerance of this to count as the period's cumulative value. # tolerance of this to count as the period's cumulative value.
_EXPECTED_YTD_DAYS = {"Q1": 91, "Q2": 182, "Q3": 273, "FY": 365} _EXPECTED_YTD_DAYS = {"Q1": 91, "Q2": 182, "Q3": 273, "FY": 365}
_YTD_TOLERANCE_DAYS = 20 # covers 52/53-week fiscal calendars # Period identity (see _period_identity): how far a quarter end sits before its
# fiscal-year end, and how far a fiscal-year end may drift from the nominal MMDD.
# The quarter bands are 91 days apart, so ±35 stays unambiguous even for a 4-4-5
# filer whose 16-week Q4 puts Q3 112 days out.
_QUARTER_DAYS_TO_FY_END = {"Q1": 273, "Q2": 182, "Q3": 91}
_QUARTER_TOLERANCE_DAYS = 35
_FYE_DRIFT_TOLERANCE_DAYS = 21
# Covers 52/53-week calendars *and* 4-4-5 retail ones (12/12/12/16 weeks), whose
# YTD-Q3 is 36 weeks = 251-252 days and missed a 20-day tolerance by ~2 -- so
# COST/PEP lost Q3 every year, breaking the quarter chain and nulling TTM + YoY.
# Q1 84d, Q2 168d and FY 364d were always inside. Adjacent periods stay
# unambiguous at 25 (66-116, 157-207, 248-298, 340-390).
_YTD_TOLERANCE_DAYS = 25
# us-gaap duration concepts (money), priority order; first present wins. # us-gaap duration concepts (money), priority order; first present wins.
_DURATION_USD = { _DURATION_USD = {
# Order is load-bearing (first present wins) and the tail entries are
# deliberately *appended*: every issuer that already resolved keeps the same
# concept, and only issuers that resolved to nothing gain a value.
# - IncludingAssessedTax: REITs/consumer filers that tag only this variant
# (e.g. ARE, KHC) reported no revenue at all.
# - RevenuesNetOfInterestExpense: the banks' total-revenue tag. JPM/GS/WFC
# tag it in every 10-Q and `Revenues` only (if at all) in the 10-K.
"revenue": [ "revenue": [
"RevenueFromContractWithCustomerExcludingAssessedTax", "RevenueFromContractWithCustomerExcludingAssessedTax",
"Revenues", "Revenues",
"SalesRevenueNet", "SalesRevenueNet",
"RevenueFromContractWithCustomerIncludingAssessedTax",
"RevenuesNetOfInterestExpense",
], ],
"net_income": ["NetIncomeLoss"], "net_income": ["NetIncomeLoss"],
"operating_income": ["OperatingIncomeLoss"], "operating_income": ["OperatingIncomeLoss"],
@@ -62,7 +89,27 @@ _DURATION_USD = {
"DepreciationAndAmortization", "DepreciationAndAmortization",
], ],
} }
_EPS_CONCEPTS = ["EarningsPerShareDiluted"] # unit USD/shares # unit USD/shares. Appended (not reordered) so any issuer that already resolved
# keeps the same concept. REG tags only the continuing-operations variant on every
# filing; FCX switches by form type -- EarningsPerShareDiluted in its 10-Qs, the
# continuing-ops tag in its 10-K -- which nulled the FY row and killed Q4 + TTM.
# The basic variants are a last resort for a period that tags no diluted EPS at
# all (PPL's 2026 Q1). Basic ignores option/convert dilution so it slightly
# overstates EPS (~1.2% for PPL), but only fires when diluted is entirely absent,
# and high-dilution names always tag diluted -- so it never displaces a real one.
_EPS_CONCEPTS = [
"EarningsPerShareDiluted",
"IncomeLossFromContinuingOperationsPerDilutedShare",
"EarningsPerShareBasic",
"IncomeLossFromContinuingOperationsPerBasicShare",
]
# Weighted-average diluted share count (unit "shares"), the market-cap fallback
# for multi-class issuers whose cover-page count is dimensional and therefore
# absent from companyfacts. Always present, since EPS is computed from it.
_WEIGHTED_AVG_SHARE_CONCEPTS = [
"WeightedAverageNumberOfDilutedSharesOutstanding",
"WeightedAverageNumberOfSharesOutstandingBasicAndDiluted",
]
# us-gaap instant (balance-sheet) concepts, at end == reportDate. # us-gaap instant (balance-sheet) concepts, at end == reportDate.
_CASH = ["CashAndCashEquivalentsAtCarryingValue"] _CASH = ["CashAndCashEquivalentsAtCarryingValue"]
_ST_INVESTMENTS = ["ShortTermInvestments", "MarketableSecuritiesCurrent"] # pick one _ST_INVESTMENTS = ["ShortTermInvestments", "MarketableSecuritiesCurrent"] # pick one
@@ -104,6 +151,7 @@ class SnapshotRow:
total_debt: float | None = None total_debt: float | None = None
shares_outstanding: float | None = None shares_outstanding: float | None = None
shares_outstanding_date: date | None = None shares_outstanding_date: date | None = None
weighted_avg_diluted_shares: float | None = None
@dataclass @dataclass
@@ -127,9 +175,14 @@ def parse_snapshots(
companyfacts: dict[str, Any], companyfacts: dict[str, Any],
filings: dict[str, FilingMeta], filings: dict[str, FilingMeta],
accessions: set[str], accessions: set[str],
fiscal_year_end: str | None = None,
) -> ParseResult: ) -> ParseResult:
"""Build snapshot rows for ``accessions`` (those with facts + filing meta). """Build snapshot rows for ``accessions`` (those with facts + filing meta).
``fiscal_year_end`` is the issuer's ``submissions.fiscalYearEnd`` (MMDD) and
is what makes period identity independent of SEC's unreliable fy/fp fields
(see ``_period_identity``). Omitting it falls back to the old fy/fp behaviour.
``skipped_filings`` = no row produced (missing facts/meta or no usable period ``skipped_filings`` = no row produced (missing facts/meta or no usable period
identity); ``field_issues`` = a row was produced but a field is null/ambiguous. identity); ``field_issues`` = a row was produced but a field is null/ambiguous.
Callers must not use field issues as failed-row coverage. Callers must not use field issues as failed-row coverage.
@@ -143,7 +196,7 @@ def parse_snapshots(
if meta is None or not facts: if meta is None or not facts:
result.skipped_filings.append({"accession": accn, "reason": "no facts or filing metadata"}) result.skipped_filings.append({"accession": accn, "reason": "no facts or filing metadata"})
continue continue
row, note = _parse_one(cik, accn, facts, meta) row, note = _parse_one(cik, accn, facts, meta, fiscal_year_end)
if row is None: if row is None:
result.skipped_filings.append({"accession": accn, "reason": note or "unparseable"}) result.skipped_filings.append({"accession": accn, "reason": note or "unparseable"})
continue continue
@@ -190,11 +243,17 @@ def _index_by_accession(companyfacts: dict[str, Any]) -> dict[str, list[Fact]]:
def _parse_one( def _parse_one(
cik: str, accn: str, facts: list[Fact], meta: FilingMeta cik: str, accn: str, facts: list[Fact], meta: FilingMeta,
fiscal_year_end: str | None = None,
) -> tuple[SnapshotRow | None, str | None]: ) -> tuple[SnapshotRow | None, str | None]:
"""Returns (row, note). row is None when there's no usable period identity; """Returns (row, note). row is None when there's no usable period identity;
note is a validation reason (row-skip reason when row is None, else a note is a validation reason (row-skip reason when row is None, else a
field-level issue such as ambiguous shares).""" field-level issue such as ambiguous shares)."""
fy, fp = _period_identity(meta, fiscal_year_end)
if fy is None or fp is None:
# No fiscal calendar, or a period the calendar cannot place (a transition
# period). Fall back to the filing's own context: an imperfect label still
# beats dropping the filing entirely.
fy, fp = _fiscal_context(facts, meta.report_date) fy, fp = _fiscal_context(facts, meta.report_date)
if fy is None or fp not in _EXPECTED_YTD_DAYS: if fy is None or fp not in _EXPECTED_YTD_DAYS:
return None, "no usable period identity" return None, "no usable period identity"
@@ -227,9 +286,86 @@ def _parse_one(
shares, shares_date, ambiguous = _select_shares(facts, meta.report_date) shares, shares_date, ambiguous = _select_shares(facts, meta.report_date)
row.shares_outstanding = shares row.shares_outstanding = shares
row.shares_outstanding_date = shares_date row.shares_outstanding_date = shares_date
row.weighted_avg_diluted_shares = _select_weighted_avg_shares(facts, meta.report_date)
return row, ("ambiguous shares outstanding" if ambiguous else None) return row, ("ambiguous shares outstanding" if ambiguous else None)
def _period_identity(
meta: FilingMeta, fiscal_year_end: str | None
) -> tuple[int | None, str | None]:
"""(fiscal_year, fiscal_period) from the period end and the issuer's fiscal
calendar — never from the fy/fp fields.
SEC's fy/fp describe the *filing*, and they are unreliable as period identity:
observed in production, a 10-Q labelled ``FY`` (BXP), a year ending 2025-12-31
labelled 2024 (FRT, a December filer), a year ending 2025-06-27 labelled 2027
(STX), and four different period ends all labelled 2022 Q3 (PPL). Because
readers key on (fiscal_year, fiscal_period), colliding labels silently discard
a period and inverted ones scramble the quarter chain — nulling TTM and YoY.
``period_end`` is authoritative, so identity is derived from it: the form
decides FY vs quarter, and distance to the fiscal-year end decides which
quarter. Labels need not match the issuer's own naming — a filer whose year
ends in early January (DPZ) shifts by one — they need to be unique, monotonic
and YoY-aligned, which is all the derivation asks of them. Nothing outside the
derivation reads these columns.
Known limitation: ``fiscalYearEnd`` is the issuer's *current* calendar, so a
company that has changed its fiscal year end gets its historical periods
measured against the new one. The quarter tolerance shunts most of those to
the fy/fp fallback, and a same-key collision resolves newest-wins, so the
failure mode is a degraded old year rather than a scrambled current one.
"""
fy = _fiscal_year_of(meta.report_date, fiscal_year_end)
if fy is None:
return None, None
if meta.form.startswith("10-K"):
return fy, "FY"
nominal_end = _nominal_fy_end(fy, fiscal_year_end)
if nominal_end is None:
return None, None
remaining = (nominal_end - meta.report_date).days
best = min(
_QUARTER_DAYS_TO_FY_END,
key=lambda k: abs(_QUARTER_DAYS_TO_FY_END[k] - remaining),
)
if abs(_QUARTER_DAYS_TO_FY_END[best] - remaining) > _QUARTER_TOLERANCE_DAYS:
return None, None # transition period or odd filing — let the caller fall back
return fy, best
def _nominal_fy_end(year: int, fiscal_year_end: str | None) -> date | None:
"""The issuer's nominal fiscal-year end in ``year`` from a MMDD string."""
if not fiscal_year_end or len(fiscal_year_end) != 4 or not fiscal_year_end.isdigit():
return None
month, day = int(fiscal_year_end[:2]), int(fiscal_year_end[2:])
if not 1 <= month <= 12 or not 1 <= day <= 31:
return None
while day > 28: # 52/53-week ends land on 0229/0230/0231 in some filings
try:
return date(year, month, day)
except ValueError:
day -= 1
return date(year, month, day)
def _fiscal_year_of(period_end: date, fiscal_year_end: str | None) -> int | None:
"""Which fiscal year ``period_end`` belongs to.
A 52/53-week calendar's real year end drifts around the nominal MMDD (and can
cross the calendar year), so allow drift before rolling into the next year.
"""
nominal = _nominal_fy_end(period_end.year, fiscal_year_end)
if nominal is None:
return None
return (
period_end.year
if period_end <= nominal + timedelta(days=_FYE_DRIFT_TOLERANCE_DAYS)
else period_end.year + 1
)
def _fiscal_context(facts: list[Fact], report_date: date) -> tuple[int | None, str | None]: def _fiscal_context(facts: list[Fact], report_date: date) -> tuple[int | None, str | None]:
"""The filing's (fy, fp) taken as the majority context among the facts that """The filing's (fy, fp) taken as the majority context among the facts that
end at reportDate (the current-period facts, which share the filing's end at reportDate (the current-period facts, which share the filing's
@@ -352,6 +488,34 @@ def _select_shares(
return None, None, False # simply absent — not a conflict return None, None, False # simply absent — not a conflict
def _select_weighted_avg_shares(facts: list[Fact], report_date: date) -> float | None:
"""The most recent quarter's weighted-average diluted share count.
Deliberately the **shortest** duration ending at reportDate, not the YTD one:
the shorter the window the closer the average sits to the current count, which
is what a market cap wants. Measured against issuers where the true
point-in-time count is available, the quarter average is within ~0.6%.
"""
best: tuple[int, float] | None = None
for concept in _WEIGHTED_AVG_SHARE_CONCEPTS:
for f in facts:
if (
f.taxonomy != "us-gaap"
or f.concept != concept
or f.unit != "shares"
or f.start is None
or f.end != report_date
or f.val <= 0
):
continue
span = (f.end - f.start).days
if best is None or span < best[0]:
best = (span, float(f.val))
if best is not None:
return best[1] # first present concept wins, as elsewhere
return None
def _d(value: Any) -> date | None: def _d(value: Any) -> date | None:
if not value: if not value:
return None return None
+105 -10
View File
@@ -21,6 +21,12 @@ Guardrails (design + reviews):
- ``promote`` inserts snapshots ``ON CONFLICT (accession) DO NOTHING`` (immutable), - ``promote`` inserts snapshots ``ON CONFLICT (accession) DO NOTHING`` (immutable),
reports differing existing accessions, and applies ticker updates in the same reports differing existing accessions, and applies ticker updates in the same
transaction. transaction.
- ``reparse=True`` is the one exception to immutability, and it is deliberate:
it restages every accession with the current parser and **rewrites** the rows
that now reconstruct differently. Immutability protects SEC's record (one row
per accession, amendments retained) — but the stored row is *our* reconstruction,
so after a parser fix, keeping it is preserving a stale cache, not history.
Manually invoked through ``scripts/reparse_fundamentals.py``; never scheduled.
""" """
from __future__ import annotations from __future__ import annotations
@@ -31,7 +37,7 @@ from dataclasses import dataclass, field
from datetime import date, datetime, timedelta, timezone from datetime import date, datetime, timedelta, timezone
from typing import Any, Callable from typing import Any, Callable
from sqlalchemy import select from sqlalchemy import select, update
from app.database import insert_for_session from app.database import insert_for_session
from app.models.data_import_run import DataImportRun from app.models.data_import_run import DataImportRun
@@ -57,7 +63,7 @@ _SNAPSHOT_COLS = (
"period_end", "fiscal_year", "fiscal_period", "revenue", "net_income", "period_end", "fiscal_year", "fiscal_period", "revenue", "net_income",
"operating_income", "diluted_eps", "cfo", "capex", "depreciation_amortization", "operating_income", "diluted_eps", "cfo", "capex", "depreciation_amortization",
"cash_and_st_investments", "total_debt", "shares_outstanding", "cash_and_st_investments", "total_debt", "shares_outstanding",
"shares_outstanding_date", "shares_outstanding_date", "weighted_avg_diluted_shares",
) )
# Compare ALL source fields (every column except the accession key) to flag a # Compare ALL source fields (every column except the accession key) to flag a
# differing existing accession — immutable, so we report, never mutate. # differing existing accession — immutable, so we report, never mutate.
@@ -75,6 +81,10 @@ class StagedFundamentals:
missing_xbrl: list[dict[str, str]] = field(default_factory=list) missing_xbrl: list[dict[str, str]] = field(default_factory=list)
invalid_payloads: list[dict[str, str]] = field(default_factory=list) invalid_payloads: list[dict[str, str]] = field(default_factory=list)
existing_accessions: set[str] = field(default_factory=set) existing_accessions: set[str] = field(default_factory=set)
# Tracked issuers whose registrant has NO XBRL 10-K/10-Q at all: they can
# never yield a snapshot, so this is a resolution problem (a ticker pointed
# at a successor shell), not missing data. See sec_universe.CIK_OVERRIDES_KEY.
no_xbrl_filings: list[dict[str, Any]] = field(default_factory=list)
discrepancies: list[dict[str, Any]] = field(default_factory=list) discrepancies: list[dict[str, Any]] = field(default_factory=list)
backfill: bool = False backfill: bool = False
issuers_fetched: int = 0 issuers_fetched: int = 0
@@ -93,9 +103,17 @@ class SecFundamentalsImporter:
*, *,
client_factory: Callable[[], SecClient] | None = None, client_factory: Callable[[], SecClient] | None = None,
today: date | None = None, today: date | None = None,
reparse: bool = False,
) -> None: ) -> None:
self._client_factory = client_factory or (lambda: SecClient()) self._client_factory = client_factory or (lambda: SecClient())
self.today = today or _now().date() self.today = today or _now().date()
# Reparse: re-derive every stored accession with the CURRENT parser and
# rewrite the ones that now reconstruct differently. Snapshots are
# immutable with respect to SEC (one row per accession, amendments kept),
# but the stored row is *our reconstruction* — when a parser bug is fixed,
# leaving it stale is not immutability, it is a stale cache. Manually
# invoked via scripts/reparse_fundamentals.py; never scheduled.
self.reparse = reparse
# cached by detect_revision, consumed by stage: # cached by detect_revision, consumed by stage:
self._resolved: ResolvedUniverse | None = None self._resolved: ResolvedUniverse | None = None
self._index_rows: list[dict[str, Any]] = [] self._index_rows: list[dict[str, Any]] = []
@@ -111,7 +129,10 @@ class SecFundamentalsImporter:
self._latest_index_date = await client.latest_index_date(self.today) self._latest_index_date = await client.latest_index_date(self.today)
if self._latest_index_date is None: if self._latest_index_date is None:
raise SecError("no EDGAR daily index available") raise SecError("no EDGAR daily index available")
if last_processed is None: # Reparse needs every accession restaged, not just those filed since
# the last run — the facts a fixed parser now accepts were never
# stored, so a reparse cannot be served from the database.
if last_processed is None or self.reparse:
self._backfill = True self._backfill = True
self._index_rows = [] self._index_rows = []
else: else:
@@ -175,6 +196,10 @@ class SecFundamentalsImporter:
return return
sub = await client.submissions(cik, include_history=is_backfill) sub = await client.submissions(cik, include_history=is_backfill)
xbrl_meta, nonxbrl = _filing_meta(sub) xbrl_meta, nonxbrl = _filing_meta(sub)
if not xbrl_meta:
staged.no_xbrl_filings.append(
{"cik": cik10(cik), "name": sub.get("name"), "tickers": sub.get("tickers")}
)
if is_backfill: if is_backfill:
accns = set(xbrl_meta) accns = set(xbrl_meta)
@@ -191,7 +216,11 @@ class SecFundamentalsImporter:
# have lagged; fail+retry rather than record nothing for it. # have lagged; fail+retry rather than record nothing for it.
staged.missing_xbrl.append({"cik": cik10(cik), "accession": accn}) staged.missing_xbrl.append({"cik": cik10(cik), "accession": accn})
result = parser.parse_snapshots(cf, xbrl_meta, accns) # fiscalYearEnd (MMDD) is what lets the parser derive period identity from
# reportDate instead of SEC's unreliable fy/fp fields.
result = parser.parse_snapshots(
cf, xbrl_meta, accns, fiscal_year_end=sub.get("fiscal_year_end")
)
staged.rows.extend(result.rows) staged.rows.extend(result.rows)
staged.skipped_filings.extend(result.skipped_filings) staged.skipped_filings.extend(result.skipped_filings)
staged.field_issues.extend(result.field_issues) staged.field_issues.extend(result.field_issues)
@@ -241,6 +270,8 @@ class SecFundamentalsImporter:
"skipped_filings": len(staged.skipped_filings), "skipped_filings": len(staged.skipped_filings),
"field_issues": len(staged.field_issues), "field_issues": len(staged.field_issues),
"skipped_non_xbrl": len(staged.skipped_non_xbrl), "skipped_non_xbrl": len(staged.skipped_non_xbrl),
"no_xbrl_filings": staged.no_xbrl_filings[:50],
"no_xbrl_filings_count": len(staged.no_xbrl_filings),
"missing_xbrl": len(staged.missing_xbrl), "missing_xbrl": len(staged.missing_xbrl),
"invalid_payloads": staged.invalid_payloads, "invalid_payloads": staged.invalid_payloads,
"cik_updates": len(staged.resolved.cik_updates), "cik_updates": len(staged.resolved.cik_updates),
@@ -257,9 +288,26 @@ class SecFundamentalsImporter:
async def promote(self, db, staged: StagedFundamentals, run_id: int) -> dict[str, int]: async def promote(self, db, staged: StagedFundamentals, run_id: int) -> dict[str, int]:
inserted = 0 inserted = 0
updated = 0
# Only accessions whose reconstruction actually changed are rewritten;
# an unchanged stored row is left completely alone.
changed = {d["accession"] for d in staged.discrepancies} if self.reparse else set()
for row in staged.rows: for row in staged.rows:
if row.accession in staged.existing_accessions: if row.accession in staged.existing_accessions:
continue # immutable — keep the original row if row.accession in changed:
# Write the FULL column set (_row_values covers _SNAPSHOT_COLS)
# so a rewritten row is never half old-parse, half new-parse.
# created_at stays at the original insert; import_run_id
# attributes the rewrite.
values = _row_values(row, run_id)
values.pop("created_at", None)
await db.execute(
update(FundamentalSnapshot)
.where(FundamentalSnapshot.accession == row.accession)
.values(**values)
)
updated += 1
continue # otherwise immutable — keep the original row
stmt = insert_for_session(db, FundamentalSnapshot).values(**_row_values(row, run_id)) stmt = insert_for_session(db, FundamentalSnapshot).values(**_row_values(row, run_id))
stmt = stmt.on_conflict_do_nothing(index_elements=["accession"]) # race belt-and-suspenders stmt = stmt.on_conflict_do_nothing(index_elements=["accession"]) # race belt-and-suspenders
await db.execute(stmt) await db.execute(stmt)
@@ -269,24 +317,50 @@ class SecFundamentalsImporter:
# any existing accession reconstructed differently — kept immutable. # any existing accession reconstructed differently — kept immutable.
if staged.discrepancies: if staged.discrepancies:
accns = ", ".join(d["accession"] for d in staged.discrepancies[:10]) accns = ", ".join(d["accession"] for d in staged.discrepancies[:10])
disposition = (
f"REWRITTEN by reparse run {run_id}" if self.reparse else "kept immutable"
)
db.add(SystemEvent( db.add(SystemEvent(
severity="warning", severity="warning",
source="sec_facts", source="sec_facts",
code="snapshot_discrepancy", code="snapshot_reparse" if self.reparse else "snapshot_discrepancy",
message=( message=(
f"{len(staged.discrepancies)} stored accession(s) reconstructed " f"{len(staged.discrepancies)} stored accession(s) reconstructed "
f"differently; kept immutable: {accns}" f"differently; {disposition}: {accns}"
)[:4000], )[:4000],
dedup_key=f"sec_facts:discrepancy:{run_id}", dedup_key=f"sec_facts:discrepancy:{run_id}",
created_at=_now(), created_at=_now(),
)) ))
# A tracked issuer whose registrant has no XBRL filings can never produce a
# snapshot, and it is restaged on every run forever. That is a resolution
# problem, not missing data, and it is silent without this.
if staged.no_xbrl_filings:
named = ", ".join(
f"{e['cik']} ({e.get('name') or '?'})" for e in staged.no_xbrl_filings[:10]
)
db.add(SystemEvent(
severity="warning",
source="sec_facts",
code="no_xbrl_filings",
message=(
f"{len(staged.no_xbrl_filings)} tracked issuer(s) resolved to a "
f"registrant with no XBRL 10-K/10-Q. Either a successor shell "
f"(pin the real filer via the '{sec_universe.CIK_OVERRIDES_KEY}' "
f"setting) or a new registrant that has not filed its first "
f"10-K/10-Q yet, which needs nothing and clears itself: {named}"
)[:4000],
dedup_key=f"sec_facts:no_xbrl_filings:{run_id}",
created_at=_now(),
))
ticker_counts = await sec_universe.apply_ticker_updates( ticker_counts = await sec_universe.apply_ticker_updates(
db, staged.resolved, staged.sic_updates db, staged.resolved, staged.sic_updates
) )
return { return {
"inserted": inserted, "inserted": inserted,
"existing_unchanged": len(staged.existing_accessions), "updated": updated,
"existing_unchanged": len(staged.existing_accessions) - updated,
"discrepancies": len(staged.discrepancies), "discrepancies": len(staged.discrepancies),
**ticker_counts, **ticker_counts,
} }
@@ -395,5 +469,26 @@ def _row_values(row: SnapshotRow, run_id: int) -> dict[str, Any]:
def _diff_fields(row: SnapshotRow, old: FundamentalSnapshot) -> list[str]: def _diff_fields(row: SnapshotRow, old: FundamentalSnapshot) -> list[str]:
"""Source fields where a re-parsed row differs from the stored (immutable) row.""" """Source fields where a re-parsed row differs from the stored row."""
return [col for col in _COMPARE_COLS if getattr(row, col) != getattr(old, col)] return [
col for col in _COMPARE_COLS
if not _same_value(getattr(row, col), getattr(old, col))
]
def _same_value(parsed: Any, stored: Any) -> bool:
"""Compare a freshly parsed value against its stored round-trip.
Datetimes need care: every timestamp here is UTC by construction, but
``DateTime(timezone=True)`` only preserves tzinfo on Postgres — SQLite hands
back a naive value. Comparing representations would report an unchanged row
as differing, which would both spam the discrepancy warning and make a
reparse rewrite every row it touched. Compare instants instead.
"""
if isinstance(parsed, datetime) and isinstance(stored, datetime):
return _as_utc(parsed) == _as_utc(stored)
return parsed == stored
def _as_utc(value: datetime) -> datetime:
return value if value.tzinfo is not None else value.replace(tzinfo=timezone.utc)
+41 -1
View File
@@ -16,6 +16,7 @@ changes on the framework's failure commit). The proposals are applied only in
from __future__ import annotations from __future__ import annotations
import hashlib import hashlib
import json
import logging import logging
from dataclasses import dataclass, field from dataclasses import dataclass, field
from typing import Iterable from typing import Iterable
@@ -23,11 +24,21 @@ from typing import Iterable
from sqlalchemy import select, update from sqlalchemy import select, update
from app.models.ticker import Ticker from app.models.ticker import Ticker
from app.services import settings_store
from app.services.earnings_alignment import normalise_symbol from app.services.earnings_alignment import normalise_symbol
from app.services.sec_client import SecClient from app.services.sec_client import SecClient
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
# JSON {symbol: cik} pinning a ticker to a specific registrant, overriding
# company_tickers.json. Needed when SEC maps a ticker to a successor entity that
# has not filed: XOM points at CIK 2115436 "ExxonMobil Holdings Corp" (zero XBRL
# filings) while every 10-K/10-Q — including one filed 2026-05-04 — is still under
# CIK 34088. Which registrant is the real filer is a judgement about a corporate
# event, so it is pinned explicitly rather than guessed. The importer's
# `no_xbrl_filings` warning is what tells you a pin is needed.
CIK_OVERRIDES_KEY = "sec_cik_overrides"
@dataclass @dataclass
class ResolvedUniverse: class ResolvedUniverse:
@@ -43,6 +54,7 @@ async def resolve_ciks(db, client: SecClient) -> ResolvedUniverse:
"""Resolve tracked tickers to CIKs via company_tickers.json. **Read-only** — """Resolve tracked tickers to CIKs via company_tickers.json. **Read-only** —
returns the mapping + proposed `tickers.cik` writes; mutates nothing.""" returns the mapping + proposed `tickers.cik` writes; mutates nothing."""
ticker_to_cik = await client.company_tickers() ticker_to_cik = await client.company_tickers()
overrides = await cik_overrides(db)
rows = (await db.execute(select(Ticker.id, Ticker.symbol, Ticker.cik))).all() rows = (await db.execute(select(Ticker.id, Ticker.symbol, Ticker.cik))).all()
result = ResolvedUniverse() result = ResolvedUniverse()
@@ -50,7 +62,7 @@ async def resolve_ciks(db, client: SecClient) -> ResolvedUniverse:
if not symbol: if not symbol:
continue continue
sym = normalise_symbol(symbol) sym = normalise_symbol(symbol)
cik = ticker_to_cik.get(sym) cik = overrides.get(sym) or ticker_to_cik.get(sym)
if cik is None: if cik is None:
continue # ADRs / non-SEC issuers — snapshots simply absent continue # ADRs / non-SEC issuers — snapshots simply absent
result.symbol_to_cik[sym] = cik result.symbol_to_cik[sym] = cik
@@ -65,6 +77,34 @@ async def resolve_ciks(db, client: SecClient) -> ResolvedUniverse:
return result return result
async def cik_overrides(db) -> dict[str, int]:
"""Manual ``{symbol: cik}`` pins from ``SystemSetting[CIK_OVERRIDES_KEY]``.
A malformed setting must never take the importer down, so anything unparseable
is logged and ignored — the run then falls back to company_tickers.json.
"""
raw = await settings_store.get_value(db, CIK_OVERRIDES_KEY)
if not raw:
return {}
try:
loaded = json.loads(raw)
except (TypeError, ValueError):
logger.warning("%s is not valid JSON — ignoring CIK overrides", CIK_OVERRIDES_KEY)
return {}
if not isinstance(loaded, dict):
logger.warning("%s must be a {symbol: cik} object — ignoring", CIK_OVERRIDES_KEY)
return {}
out: dict[str, int] = {}
for symbol, cik in loaded.items():
try:
out[normalise_symbol(str(symbol))] = int(cik)
except (TypeError, ValueError):
logger.warning("%s: bad entry %r -> %r — ignoring", CIK_OVERRIDES_KEY, symbol, cik)
if out:
logger.info("resolve_ciks: %d CIK override(s) applied: %s", len(out), sorted(out))
return out
async def fetch_sic_updates( async def fetch_sic_updates(
client: SecClient, cik_to_ticker_ids: dict[int, Iterable[int]] client: SecClient, cik_to_ticker_ids: dict[int, Iterable[int]]
) -> list[tuple[int, str | None, str | None]]: ) -> list[tuple[int, str | None, str | None]]:
@@ -0,0 +1,855 @@
# A5 parity report — root-cause findings
Investigation of `fundamentals-parity-20260723T210658161480Z.json` (511 tickers,
generated 2026-07-23). Method: replayed the production parser
(`sec_facts_parser.parse_snapshots`) and derivation (`fundamentals_derivation.derive`)
against **live SEC companyfacts**, using the importer's own `_filing_meta` and
backfill accession set, then cross-checked prices against IBKR. No database was
available locally, so every conclusion below is reproduced from source data rather
than read out of prod.
Repro script: `scratchpad/diag.py` (`--history` replays the full backfill path).
Every claim below was verified on the named issuer. Names that were *not*
individually inspected are listed as unclassified — an earlier draft of this
document guessed their cause from fiscal-year-end dates and was wrong for most of
them, so the guessing is not repeated here.
## Verdict
Where both sides have a value the candidate data is good: P/E spearman 0.968,
revenue growth agreeing to 4 decimals for most names, score spearman 0.825. Every
defect found is a **parser/derivation bug or an identity problem** — not a data
quality problem with SEC or Dolt. The largest cluster is period identity, which is
exactly the risk A3 flagged as primary.
## 1. P/E outliers — splits corrupt TTM EPS, and the split guard doesn't cover it
`derive()` sets `result.ttm_diluted_eps` at `fundamentals_derivation.py:88` and only
calls `_guard_split_sensitive_metrics()` at line 105, which annotates `result.metrics`
(the `MetricSeries` objects). `ttm_diluted_eps` is a bare scalar and is never guarded.
`fundamentals_parity_service._pe()` consumes it directly.
The cleanest evidence that the *candidate* side is the broken one: reconcile each
P/E against the report's own price. Legacy comes out sane in both cases, candidate
does not.
**BKNG — guard fired, nobody listened.** Share count jumps 31.7M → 774.9M between the
FY2025 10-K and the 2026 Q1 10-Q (≈25:1 split). TTM EPS therefore sums three pre-split
quarters (27.31 + 84.01 + 44.18 = 155.50) plus one post-split quarter (1.36) =
**156.86** — mixed units. Live price $172.83 matches the price the report implies
exactly (1.1018 × 156.86 = 172.83), so the price is correct and current. Against that
price, legacy's P/E of 22.44 implies EPS ≈ 7.70 — a coherent post-split number, versus
the candidate's 156.86. The derivation *did* raise `"Not comparable: share count
changed at least 25%; possible split or corporate action."` on `eps_growth_yoy` and
`share_count_change_yoy` — P/E never sees it.
**KLAC — the guard cannot fire.** The split post-dates the most recent 10-Q (period end
2026-03-31), so no snapshot shows any share-count change (`share_count_change_yoy` =
1.2%). TTM EPS **35.31** is internally consistent and entirely pre-split; the price
($223.30 live, ≈218.7 in the report) is post-split. Reconciling: legacy P/E 60.21
against the report price implies EPS ≈ 3.63 ≈ 35.31/9.7 — i.e. legacy is consistent
with a ~10:1 split and correct, and the candidate is off by exactly the split factor.
(IBKR's split-adjusted `open_52w` of 89.36 corroborates 10:1.)
This is the important case: **a split after the latest filing is undetectable from
snapshots alone.** No share-count test can catch it. Reconciliation needs a corporate
actions source or a price-vs-EPS plausibility check.
**COF — not a bug, a definition difference.** Shares 383M → 639M in 2025 Q2 is the
Discover acquisition. TTM GAAP EPS is genuinely $3.92 because the merger-charge quarter
(10.19) sits in the window. Candidate P/E 51.01 is arithmetically correct on a GAAP TTM
basis; legacy's 11.61 is an adjusted/forward convention. Disclose, don't fix. Note this
single row drives the report's largest change (rank 1 → 456).
## 2. Bank revenue growth — concept-mapping gap (confirmed)
`sec_facts_parser._DURATION_USD["revenue"]` is:
```
RevenueFromContractWithCustomerExcludingAssessedTax, Revenues, SalesRevenueNet
```
Banks tag **`RevenuesNetOfInterestExpense`** in their 10-Qs:
| filer | 2026 Q1 10-Q tags present | parsed `revenue` |
|---|---|---|
| JPM | `RevenuesNetOfInterestExpense` 49,836M, `NoninterestIncome`, `InterestIncomeExpenseNet` | **null** |
| GS | `RevenuesNetOfInterestExpense` 17,227M, `InterestAndDividendIncomeOperating`, … | **null** |
| WFC | `RevenuesNetOfInterestExpense` 21,436M, … | **null** |
JPM's FY2025 10-K *also* tags `Revenues` (182,447M — identical value), so only the annual
row populates; GS never tags `Revenues` at all. Revenue growth needs five consecutive
quarterly values, so it is null for the whole cluster (JPM, GS, MS, WFC, TFC, MTB, FITB,
RF, SYF, BNY, BX, BLK, SPGI, ACGL, CBOE).
A second variant of the same gap: **ARE** and **KHC** tag
`RevenueFromContractWithCustomer**Including**AssessedTax` — also absent from the list —
so revenue is null on every row while EPS parses fine.
**Fix:** add `RevenuesNetOfInterestExpense` and the `IncludingAssessedTax` variant.
**Latent risk while you're in there:** `RevenueFromContractWithCustomerExcludingAssessedTax`
is *first* and "first present wins". For a bank that tags it, it captures only ASC-606 fee
revenue, not total revenue — a silently **understated** number rather than a null, which is
worse. DVN shows the same hazard from the other side: its 2026 Q1 tags both
`RevenueFromContractWithCustomerExcludingAssessedTax` (4,508M) and `Revenues` (3,807M),
an 18% difference decided purely by list order.
## 3. Period identity — the largest cluster, three confirmed mechanisms
### 3a. Fiscal-year label collisions (CRM, FRT, STX)
`_fiscal_context` majority-votes SEC's `fy`/`fp` fields, and `_select_latest_per_period`
keys on `(fiscal_year, fiscal_period)`. When SEC's labels disagree with the calendar, two
distinct periods collide on one key and **one is silently discarded**:
- **CRM** — two rows keyed `2025 FY`, ending 2025-01-31 and 2026-01-31.
- **FRT** — two rows keyed `2024 FY`, ending 2024-12-31 and 2025-12-31.
- **STX** — the year ending 2025-06-27 is labelled **`2027 FY`**, so it sorts *after*
`2026 Q3` (period end 2026-04-03) and is taken as the latest quarter.
The survivor's `period_end` then contradicts the fiscal ordering, Q4 derivation and the
consecutive-quarter chain break, and TTM EPS + YoY both go null.
**FRT is a calendar-year (Dec) filer**, so this is *not* limited to non-calendar fiscal
years — the earlier assumption that it was is wrong. Any filer SEC labels inconsistently
is exposed.
### 3b. Amendment selection blanks a period (DVN)
DVN has two rows for `2025 FY` (both ending 2025-12-31): the 10-K with complete financials,
and a **10-K/A carrying no financial facts at the report date** (`rev=None eps=None`).
`_select_latest_per_period` takes the newest `accepted_at`, so **the empty amendment wins**
and the FY2025 row becomes all-null, breaking the chain.
This is the most dangerous of the three: it is not exotic. Any issuer filing a 10-K/A —
including routine Part III amendments that restate nothing — silently loses that period.
The rule needs to prefer the newest accession *that actually carries the fact*, per field,
rather than the newest accession outright.
### 3c. 4-4-5 retail calendar — Q3 only, misses by ~2 days (COST, PEP)
`_EXPECTED_YTD_DAYS["Q3"] = 273` with `_YTD_TOLERANCE_DAYS = 20` accepts 253293 days. A
12/12/12/16-week filer's YTD-Q3 is 36 weeks ≈ **251252 days** — just under the floor.
Confirmed, facts present and rejected:
- COST 2026 Q3: `RevenueFromContractWithCustomerExcludingAssessedTax` span=**251d**
val=207,431M, `EarningsPerShareDiluted` span=251d val=14.01 → row stored with
`rev=None eps=None start=None`. Same for 2025 Q3 and 2024 Q3.
- PEP: every Q3 row is `rev=None eps=None`; Q1/Q2/FY all populate.
Q1 (83d vs 91±20), Q2 (167d vs 182±20) and FY (363364d vs 365±20) all pass — only Q3
fails, every year. The code comment claims the tolerance "covers 52/53-week fiscal
calendars"; it does not cover 4-4-5 ones.
Note this does **not** apply to ordinary 13-week 52/53-week filers (STX's Q3 YTD is 279d and
passes) — their failures are 3a, not this.
**Fix:** widen the Q3 tolerance to ~25 days, or derive the expected span from the filer's own
fiscal calendar rather than a fixed 91/182/273.
## 4. CIK identity (XOM)
SEC's `company_tickers.json` now maps **XOM → CIK 2115436 "ExxonMobil Holdings Corp", which
has 0 filings**. All 26 XBRL 10-K/10-Qs sit under the old CIK **34088 "EXXON MOBIL CORP"**.
XOM therefore has no snapshots at all, and nothing in the pipeline notices that a tracked
issuer resolved to a CIK with zero filings.
PSKY (5 filings) and Q (3 filings) are genuinely new registrants — expected, not a bug.
## Status of the 25 names that lose their fundamental score
Production requires ≥2 metrics (`scoring_service.py:502`), the same rule the parity harness
uses, so these genuinely drop the fundamental dimension and the composite renormalises over
the remaining four.
| cause (confirmed on the named issuer) | names |
|---|---|
| FY label collision (3a) | CRM, FRT, STX |
| 4-4-5 Q3 span (3c) | COST, PEP |
| revenue concept gap (§2) | ARE, KHC |
| amendment blanks period (3b) | DVN |
| CIK identity (§4) | XOM |
| new registrant — expected | PSKY, Q |
| **not yet classified** | AZO, BXP, CRWD, FCX, HAL, MOS, MTD, NTAP, PPL, REG, SJM, SWKS, WDAY |
13 of 25 confirmed. The unclassified 13 have not been inspected and should not be assumed to
share a cause — the confirmed set already spans five distinct mechanisms.
## Recommended order of work
1. **Amendment selection (3b)** — highest blast radius, affects any 10-K/A filer, and the
current rule is wrong in principle rather than at the margin.
2. **Revenue concept list (§2)** — add `RevenuesNetOfInterestExpense` and
`IncludingAssessedTax`; audit the ASC-606-first priority, which can understate rather
than null.
3. **Q3 YTD span tolerance (3c)** — effectively one line.
4. **XOM CIK remap (§4)** — plus a validation that flags any tracked ticker resolving to a
CIK with zero XBRL filings.
5. **Split safety for `ttm_diluted_eps` (§1)** — propagate the existing guard to the scalar,
and add a price-vs-EPS plausibility check for splits that post-date the last filing.
6. **Fiscal-period identity (3a)** — the deepest fix; consider keying period identity on
`period_end` rather than SEC's `fy`/`fp`.
Re-run the parity report after these and re-classify the remaining 13 before making a
cutover decision. The current report should not be approved as-is: its coverage gaps are
artifacts of the above, not real absences in the source data.
---
# Fixes applied (items 13)
| # | change | file | effective |
|---|---|---|---|
| 1 | amendment resolution is now **per field** — newest accession that actually reports a fact wins; only rows sharing the newest `period_end` are merged, so a mislabelled filing is never blended in | `fundamentals_derivation.py` | **read time — immediately** |
| 2 | appended `RevenueFromContractWithCustomerIncludingAssessedTax` and `RevenuesNetOfInterestExpense` to the revenue concept list | `sec_facts_parser.py` | parse time — **needs reparse** |
| 3 | YTD span tolerance 20 → 25 days, covering 4-4-5 retail calendars | `sec_facts_parser.py` | parse time — **needs reparse** |
Fix 2 is deliberately **additive**: the new tags go at the end of the priority list, so
every issuer that already resolved keeps the same concept and only issuers that resolved
to nothing gain a value. A regression test pins that ordering.
Tests: 7 added across `test_sec_facts_parser.py` and `test_fundamentals_derivation.py`.
The 5 behaviour-changing ones were confirmed to fail against the pre-fix code; the other 2
are invariance guards that pass both ways. Full unit suite: 795 passed.
## Validation against live SEC data
Re-ran the parser + derivation on live companyfacts. Every targeted name recovers, and
the recovered values independently agree with the legacy provider:
| name | cause | revenue growth before → after | legacy | TTM EPS after |
|---|---|---|---|---|
| COST | 4-4-5 Q3 | null → **9.2311** | 9.23 | 19.88 |
| PEP | 4-4-5 Q3 | null → **5.6197** | 5.62 | 7.63 |
| KHC | concept (Including) | null → **1.7457** | 1.75 | 4.85 |
| DVN | partial 10-K/A | null → **0.0956** | 1.51 | 3.59 |
| ARE | concept (Including) | null → **5.3462** | 9.53 | 6.27 |
| JPM | concept (bank) | null → **3.3388** | 108.98 | 20.89 |
| GS | concept (bank) | null → **11.1974** | 6.67 | 54.75 |
| WFC | concept (bank) | null → **4.1847** | 72.75 | 6.47 |
COST/PEP/KHC matching legacy to two decimals is strong evidence the parse is now correct.
The banks are the opposite case and worth noting for the cutover argument: legacy's JPM
109% and WFC 73% "revenue growth" are not plausible for a bank, while the SEC-derived
3.3% and 4.2% are — here the candidate is **better** than what it would replace. DVN and
ARE still differ from legacy; DVN is the `Revenues` vs ASC-606 ambiguity noted in §2 and
is the one open definition question.
Regression check on names that were already correct — IRM, KLAC, BKNG — reproduces their
previous values exactly (IRM 15.6375, KLAC 13.3895, BKNG 14.9506; TTM EPS unchanged).
Nothing that worked before moved.
### Concept consistency across the bank chains (checked, clean)
Because `Revenues` still outranks `RevenuesNetOfInterestExpense`, a filer could resolve the
FY row to one concept and its quarters to the other — which would make
`Q4 = YTD(FY) YTD(Q3)` a subtraction across two definitions, and poison every TTM window
containing it. Checked all 15 recovered banks (`scratchpad/concept_check.py`):
- **14 resolve a single concept across the whole chain** (GS, WFC, MS, TFC, MTB, FITB, RF,
SYF, BNY, BX, BLK, SPGI, ACGL, CBOE).
- **JPM is mixed but benign**: its FY2025 row tags both, at an *identical* 182,447M, so Q4
subtracts like for like. No filer showed the two tags disagreeing where both appear.
So the "candidate beats legacy for banks" claim above is safe as stated. **Residual risk:**
a future filer whose two tags differ would fail silently. Cheapest hardening is to treat
the two as one logical revenue concept rather than separate priority entries; the detector
script above turns this into a one-command check.
## Operational note — the parser fixes need a deliberate reparse
`sec_fundamentals_importer.promote()` treats snapshots as **immutable per accession**: a
re-run skips any accession already stored and records a `snapshot_discrepancy` SystemEvent
instead. So fixes 2 and 3 change nothing for rows already in the database — recovering
COST/PEP/JPM/etc. requires deleting the affected snapshot rows and re-importing, or adding
an explicit reparse path. Usefully, the discrepancy warning names exactly which stored
accessions now reconstruct differently, so a dry run over existing data will enumerate the
blast radius before anything is rewritten.
---
# Second pass — all 25 lost names now classified
Re-ran `diag.py --history` over every previously unclassified name, with fixes 13 in
place. (One name, DPZ, had been dropped from the unclassified list when this document was
rewritten; it is included here.)
## 11 of 25 recover
COST, PEP, KHC, DVN, ARE, **AZO, MOS, SJM, SWKS, HAL, DPZ** — and again the recovered
revenue growth matches the legacy provider to two decimals on every one:
| name | candidate | legacy | | name | candidate | legacy |
|---|---|---|---|---|---|---|
| AZO | 5.7405 | 5.74 | | SWKS | 2.3303 | 2.33 |
| MOS | 12.3388 | 12.34 | | HAL | 1.7201 | 1.72 |
| SJM | 3.7222 | 3.72 | | DPZ | 5.1573 | 5.16 |
Precisely: all 11 clear the ≥2-metric floor and regain a fundamental score. P/E returns for
AZO, MOS, SWKS, DPZ, COST, PEP and DVN. ARE, KHC and SJM have genuinely negative TTM EPS,
so their P/E stays null correctly. **HAL's TTM EPS is still null and the cause is not yet
established** — it scores on revenue growth + surprise. Loose end.
## 14 remain, in four causes
| cause | names | count |
|---|---|---|
| **fiscal-year label collisions (§3a)** | CRM, FRT, STX, BXP, CRWD, MTD, NTAP, WDAY, PPL | **9** |
| **EPS concept gap (new — §5 below)** | FCX, REG | 2 |
| CIK identity (§4) | XOM | 1 |
| new registrant — expected, not a bug | PSKY, Q | 2 |
The label bug is now the dominant cause by a wide margin, and it is more varied than first
described — it is not only colliding `fiscal_year` values:
- **BXP** — a *10-Q* for period end 2026-03-31 is labelled `2026 **FY**`. The **fiscal
period** is wrong, not just the year, so `_select_ytd` then measures the 90-day fact
against the 365-day FY expectation and rejects it too.
- **NTAP, WDAY, MTD, CRWD** — two different period-ends colliding on one key (the pattern
first seen on CRM/FRT).
- **PPL** — the worst observed: **four** rows keyed `2022 Q3`, with period ends 2022-09-30,
2023-03-31, 2023-06-30 and 2023-09-30.
## 5. New cause — EPS concept coverage
`_EPS_CONCEPTS = ["EarningsPerShareDiluted"]` is the only tag read. Confirmed by listing
every `USD/shares` duration concept in the relevant filings:
- **REG** tags only `IncomeLossFromContinuingOperationsPerDilutedShare`, on every filing —
EPS is null everywhere, so no TTM EPS and no P/E, ever.
- **FCX** is the nastier shape: its **10-Qs** tag `EarningsPerShareDiluted`, but its
**10-K** tags only `IncomeLossFromContinuingOperationsPerDilutedShare`. The FY row loses
EPS, so `Q4 = YTD(FY) YTD(Q3)` is undefined and TTM dies — an issuer that switches
concept *by form type* looks like partial data rather than a mapping gap.
**Fix:** append `IncomeLossFromContinuingOperationsPerDilutedShare` to `_EPS_CONCEPTS`.
Same additive shape as the revenue fix; recovers REG outright and FCX's FY row.
**Related decision, not a fix:** PPL's 2026 Q1 tags *no diluted variant at all* — only
`EarningsPerShareBasic` and `IncomeLossFromContinuingOperationsPerBasicShare`. Adding the
diluted continuing-ops tag does not help it. Falling back to basic EPS is a definition
change (basic ≠ diluted) and should be an explicit call, not a silent one.
---
# Third pass — fixes #2 and #3 applied
| # | change | file | effective |
|---|---|---|---|
| 2a | `_guard_split_sensitive_metrics()` now returns whether the *latest* period is split-suspect, and `derive()` nulls `ttm_diluted_eps` (setting `ttm_diluted_eps_caveat`) when it is | `fundamentals_derivation.py` | read time — immediately |
| 3 | appended `IncomeLossFromContinuingOperationsPerDilutedShare` to `_EPS_CONCEPTS` | `sec_facts_parser.py` | parse time — needs reparse |
5 tests added; the 3 behaviour-changing ones confirmed to fail against pre-fix code, 2 are
invariance guards. Full unit suite: **800 passed**.
## Validated on live data
| name | before | after | |
|---|---|---|---|
| FCX | TTM EPS null | **1.89** | recovered |
| REG | TTM EPS null | **2.92** | recovered |
| BKNG | TTM EPS 156.86 → P/E **1.10** | **null** + caveat | false perfect score removed |
| COF | TTM EPS 3.92 → P/E 51.01 | **null** + caveat | see side effect below |
| KLAC | TTM EPS 35.31 → P/E **6.19** | unchanged | **still wrong — 2b not fixed** |
| IRM, COST | — | unchanged | no regression |
FCX and REG regain a fundamental score (EPS + surprise clears the ≥2 floor). Their
**revenue growth is still null** — both are also blocked by the label bug (REG has a
mislabelled duplicate `2024 Q2`; FCX is missing its 2024 FY row entirely).
## Threshold decision — RESOLVED: keep 25%
Measured against the database (`scratchpad/share_change_check.sql`): **15 of 467 comparable
issuers (3.2%)** trip the ≥25% guard on their latest period.
| band | names | cause |
|---|---|---|
| ≥200% | BKNG 23.8×, ORLY 14.5×, NFLX 9.8×, NOW 5.0×, TPL 3.0× | forward splits |
| 50142% | CHTR (query artifact), **AMCR 68% (1-for-5 reverse split)**, WAT, COF | split + stock-funded M&A |
| 2547% | OMC, BG, HBAN, FITB, COHR, RKLB | stock-funded M&A, ordinary dilution |
**Keep the threshold at 25%**, for three reasons — the first of which is empirical and came
out of checking AMCR:
1. **A real split trips at only 68%.** AMCR's 1-for-5 reverse consolidation
(2,308,359,941 → 462,045,690 shares, ratio 4.996, between the Nov 2025 and Feb 2026
10-Qs) shows up as 68%. Raising the bar to 100% to spare the M&A cases would have let a
genuine split straight through. Split magnitude and M&A magnitude overlap in practice,
not just in theory.
2. **The cost is milder than first described.** Losing P/E leaves revenue growth + earnings
surprise = 2 metrics, which still clears the ≥2 floor. Affected issuers keep a
fundamental score; they lose one of three inputs.
3. **The severities are asymmetric.** A missed split yields a P/E off by 1025×, clamping to
a *perfect 100* sub-score. Over-nulling yields a missing input the scorer already handles
by renormalising.
Honest caveat: the guard is blunt — it detects that a share base moved, not how much damage
resulted. AMCR's pre-fix P/E was 28.61 against legacy's 29.47, i.e. only ~10-15% off, because
most of its YTD figures had already been restated on the post-split basis. So the guard
sometimes removes a roughly-usable number. That is the accepted price of a rule that cannot
measure the split factor.
Two data notes from the same check:
- **CHTR is a query artifact, not a guard trip.** The SQL picks the newest period *with* a
share count, while `derive()` picks the newest period and then reads shares off it. CHTR's
recent snapshots have a null `shares_outstanding`, so the query fell back to the 2016 Time
Warner merger. In the real path its change is None and the guard never fires — so the true
count is ~14. But it also means **CHTR has no recent share count, which breaks its market
cap in the API** — a separate small bug.
- **AMCR was suspected of being a `shares_outstanding` parsing bug and is not.** It is a real
corporate action, correctly detected. `abs()` in the guard already handles reverse splits.
## Side effect — COF
The guard fires on *any* ≥25% YoY share-count move, not only splits. COF's 383M → 639M jump
is the Discover acquisition, so it now nulls too and **loses the P/E of 51.01** that this
document previously called "arithmetically correct on a GAAP TTM basis".
I think nulling is right: TTM EPS sums four quarters whose per-share figures use different
weighted-average denominators, and across a 67% share change that sum is not a meaningful
per-share number regardless of whether the cause was a split or an acquisition. It follows
the formula without being a valid result.
But the cost is real and worth stating plainly: **any issuer doing a large stock-funded
acquisition loses its P/E for four quarters.** That frequency has not been measured — it
needs a count of `|share_count_change_yoy| ≥ 25%` across the universe, which needs the
database. If it turns out to be common, the alternative is a higher or split-shaped
threshold, at the cost of letting more BKNG-class errors through.
## 2b is genuinely unfixed
KLAC's split post-dates its most recent 10-Q, so no snapshot carries any share-count
evidence and no guard built on share counts can fire. Its P/E is still 6.19 — the true P/E
divided by the split factor. I did not ship a heuristic for this: the obvious one, flagging
implausibly low P/Es, would misfire on genuinely cheap names — CHTR (3.42) and CMCSA (4.30)
sit below KLAC's corrupted 6.19 in this very report. Detecting it needs an actual
corporate-actions source, or a price-vs-share-count reconciliation against an external
market-cap reference.
---
# Fourth pass — the reparse path
Snapshots are immutable per accession, so the parser fixes never reached stored rows.
`promote()` skipped them and logged a discrepancy. Reparse is the deliberate exception:
immutability protects *SEC's* record, but the stored row is **our reconstruction** — after a
parser fix, keeping it is preserving a stale cache, not preserving history.
| change | file |
|---|---|
| `run_import(..., force=True)` bypasses the unchanged-revision no-op. The revision tracks the *source*; a fix on our side leaves it unchanged, so the gate would skip the run | `data_import.py` |
| `SecFundamentalsImporter(reparse=True)` — forces full-history staging, and `promote()` rewrites the accessions whose reconstruction changed, stamping `import_run_id` | `sec_fundamentals_importer.py` |
| `scripts/reparse_fundamentals.py`**dry run by default**, `--apply` to write | new |
Unchanged rows are never touched; only accessions appearing in `staged.discrepancies` are
rewritten. The update writes the full `_SNAPSHOT_COLS` set via the same `_row_values()` the
insert uses, so a rewritten row can never be half old-parse and half new-parse. `created_at`
keeps its original value. Nothing is wired into the scheduler.
## A real bug the tests caught: false-positive discrepancies
`test_reparse_leaves_unchanged_rows_untouched` failed on first run — reparsing *identical*
data reported a change. Cause: `accepted_at` is written tz-aware UTC but
`DateTime(timezone=True)` only preserves tzinfo on Postgres; SQLite returns it naive, so
`_diff_fields` compared representations and saw a difference.
Left alone this would have made the dry-run report claim **every row needs rewriting**
exactly the misleading signal that makes a blast-radius report worthless. `_diff_fields` now
compares datetime *instants* via `_same_value()`. This also fixes a latent false positive in
the pre-existing `snapshot_discrepancy` warning, which shares the same code path.
## Verification
4 reparse tests added, driven through the real import framework with the fake SEC client.
The key one seeds the database through the **pre-fix parser** (monkeypatching
`_YTD_TOLERANCE_DAYS` back to 20 so a 4-4-5 Q3 is rejected and stored as null), then reparses
with the fixed parser and asserts the row is rewritten in place with new provenance — the
production scenario end to end. Also covered: unchanged rows keep their original
`import_run_id`; `reparse=False` still reports and refuses to mutate; `force` bypasses the
no-op. Full suite: **804 passed**.
Not verifiable here: this reads and writes production Postgres, which is unreachable from
this machine, so the SQLite harness is the limit of what could be self-tested. The UPDATE is
plain SQLAlchemy Core with no dialect-specific constructs.
## Running it
```
python scripts/reparse_fundamentals.py # dry run, writes nothing
python scripts/reparse_fundamentals.py --apply # rewrite changed rows
```
Two cautions for whoever runs it:
- **Read the dry run for *kinds* of change, not just the count.** The tolerance 20→25 change
newly accepts facts for arbitrary filers, not only the names investigated here. Sample
changed rows for issuers that were never on the list and confirm they are recovered nulls
and corrected values — not something unexpected.
- **It refetches Company Facts for every tracked issuer** under the SEC throttle, because the
facts a fixed parser now accepts were never stored. Expect a long run; the dry run pays
that cost too, so budget for two passes.
Scope: this rewrites `fundamental_snapshots` only. Those rows currently feed the fundamentals
API/UI and the parity report — scoring still reads the legacy `fundamental_data` table, and
nothing in the backtest path touches `FundamentalSnapshot`. So a reparse **cannot** move
composite scores or backtests until the A5 cutover happens. The plan's "changed history
changes backtests" caution applies to workstream B's OHLCV rewrites, not to this.
---
# Fifth pass — period identity
The parser's own stated rule was *"period identity comes from `end == reportDate`, never
`fy/fp`"* — but `_fiscal_context()` derived the stored `fiscal_year`/`fiscal_period` by
majority-voting exactly those fy/fp fields. The labelling contradicted the module's own
principle, and SEC's labels are unreliable enough to break the quarter chain.
`_period_identity()` now derives both from `period_end` against the issuer's
`submissions.fiscalYearEnd`: **the form decides FY vs quarter** (a 10-Q can no longer be
labelled FY), and **distance to the fiscal-year end decides which quarter**. The MMDD is
threaded through `parse_snapshots(..., fiscal_year_end=...)`; without it the old fy/fp path
is used unchanged, so nothing regresses for issuers lacking a calendar.
**Rejected approach:** classifying the period by fact spans. Every 10-Q carries both a YTD
*and* a discrete fact ending at reportDate, so "best span match" reads COST's Q2 (167d) as a
Q1; and taking the *longest* span mislabelled IRM's Q3 2020 10-Q as FY because that filing
carries a 12-month fact. The prototype caught this as a regression on a working name before
any code was written. Distance-to-year-end needs no facts at all and is unambiguous — the
quarter bands sit 91 days apart, so ±35 absorbs even a 4-4-5 filer's 16-week Q4.
**Labels no longer match issuer naming in one case, deliberately.** A filer whose year ends
in early January (DPZ, `fiscalYearEnd` 0102) shifts by one. That is harmless: `fiscal_year`
and `fiscal_period` appear nowhere in the API schemas or routers — they are internal keys the
derivation uses for ordering, YTD differencing and YoY pairing, and the API surfaces
`period_end`. The requirement is uniqueness, monotonicity and YoY alignment, not nomenclature.
DPZ's derived values are byte-identical before and after the shift, which is the proof.
## Prototype evidence (before implementing)
Collisions = two period ends on one key, one silently discarded. Inversions = a period
sorting before one that precedes it.
| | CRM | FRT | STX | BXP | PPL | MTD | NTAP | WDAY | CRWD | COST | PEP | IRM | DPZ | AMCR | AAPL |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| collisions before | 1 | 1 | 0 | 0 | 4 | 5 | 2 | 3 | 2 | 0 | 0 | 0 | 1 | 0 | 0 |
| collisions after | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
| inversions before | 1 | 1 | 1 | 0 | 2 | 10 | 2 | 3 | 5 | 1 | 0 | 0 | 2 | 0 | 0 |
| inversions after | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
## Validated on live data
8 of the 9 recover fully, every one matching the legacy provider to two decimals:
| name | TTM EPS | revenue growth | legacy |
|---|---|---|---|
| CRM | 8.63 | 10.9818 | 10.98 |
| FRT | 5.77 | 7.4263 | 7.43 |
| STX | 10.54 | 28.9227 | 28.92 |
| BXP | 1.99 | 1.6227 | 1.62 |
| MTD | 42.57 | 6.7785 | 6.78 |
| NTAP | 6.35 | 5.3713 | 5.37 |
| WDAY | 3.21 | 13.3165 | 13.32 |
| CRWD | 0.10 | 23.1667 | 23.17 |
**PPL is partial**: revenue growth recovers (8.3353) but TTM EPS is still null — its 2026 Q1
tags no diluted EPS variant at all, which is the open basic-vs-diluted decision, not this bug.
Note legacy claims 58.81% revenue growth for a utility; 8.34% is far more plausible.
**Two bonus recoveries**: FCX and REG had recovered EPS in the fourth pass but their revenue
growth was still blocked by label collisions. REG now reads 7.7569 against legacy's 7.76.
FCX reads 5.4378 against legacy's 24.23 — a genuine disagreement, likely the same
`Revenues` vs ASC-606 ambiguity flagged for DVN in §2, and worth resolving with that decision.
**Regression check — all byte-identical:** IRM 0.92/15.637543, COST 19.88/9.231107,
PEP 7.63/5.619741, DPZ 17.64/5.157289, AMCR null/64.834349, JPM 20.89/3.338823,
DVN 3.59/0.095648, AZO 145.39/5.740494. Nothing that worked before moved.
7 tests added at `_period_identity` covering each production shape (10-Q-labelled-FY,
December collision, January and mid-year ends, 4-4-5 quarters, the January-crossing shift,
and the no-calendar fallback). Full suite: **811 passed**.
## Reparse note
This changes `fiscal_year`/`fiscal_period` for a large share of rows — every non-December
filer, not only the broken ones. The dry-run count will be **much** larger than for the
earlier fixes, and that is expected. Read it by field: `fiscal_year`/`fiscal_period` churn is
the intended relabelling; changes to *value* columns are the recoveries.
## Where the 25 stand now
22 of 25 have a fundamental score again. Remaining: **XOM** (CIK identity, still unfixed) and
**PSKY / Q**, which are new registrants without enough filing history — correct behaviour,
not a bug.
---
# Sixth pass — CIK identity, and a much larger finding about share counts
## XOM: pinned, plus the validation that should have caught it
`company_tickers.json` maps XOM to CIK 2115436 "ExxonMobil Holdings Corp", which has **zero
XBRL filings**, while every 10-K/10-Q — including one filed 2026-05-04 — is still under CIK
34088. Which registrant is the real filer is a judgement about a corporate event, so it is
**pinned explicitly** rather than guessed:
- `sec_universe.cik_overrides()` reads a `{symbol: cik}` JSON map from
`SystemSetting['sec_cik_overrides']` and applies it ahead of `company_tickers.json`.
A malformed setting is logged and ignored, never fatal.
- **To fix XOM, set:** `sec_cik_overrides = {"XOM": 34088}`.
The more valuable half is that nothing noticed. A tracked issuer resolving to a registrant
with no XBRL filings can never produce a snapshot, and is restaged on *every* run forever.
The importer now records those in `staged.no_xbrl_filings`, reports them in the validation
summary (`no_xbrl_filings_count`), and raises a `no_xbrl_filings` SystemEvent naming the CIKs
and pointing at the override setting. It warns rather than fails — one misresolved ticker
must not block the whole import.
3 tests added. Full suite: **814 passed**.
## CHTR was not a bug, and the real problem is much bigger
I previously called this "a separate small bug". Both halves were wrong.
CHTR's `dei:EntityCommonStockSharesOutstanding` facts stop at **2016-06-30** — exactly when
the Time Warner Cable / Bright House deal closed and Charter became a multi-class issuer.
Since then the cover page reports the count **per share class**, which is dimensional, and
companyfacts is non-dimensional — so the facts are simply not in the API. Its recent filings
tag no consolidated common-share concept at all, only preferred and treasury.
This is not specific to CHTR. Of 12 issuers checked, **7 have no share count at all**:
| issuer | latest `shares_outstanding` | dei fact history |
|---|---|---|
| META | null (4/4 recent) | **never tagged** (n=0) |
| CMCSA | null (4/4 recent) | stops 2009-12-31 |
| BRK-B | null (4/4 recent) | stops 2011-04-29 |
| CHTR | null (4/4 recent) | stops 2016-06-30 |
| FOXA, NWSA, LEN | null (4/4 recent) | — |
| GOOGL / GOOG | 12,230,000,000 | works via the `us-gaap` fallback |
So **market cap is silently unavailable for a meaningful slice of the large-cap universe**,
and it is a source limitation rather than a parser defect: the two obvious workarounds are
both already-rejected design decisions — class sums are impossible (the per-class facts are
not in companyfacts at all), and the weighted-average diluted count is explicitly excluded
because market cap needs a point-in-time value.
**No code change made.** Substituting weighted-average diluted shares would silently
overturn a deliberate design decision and produce a subtly wrong market cap for exactly the
biggest, most-watched names. That is a call to make explicitly, so it is listed as a decision
below rather than quietly implemented.
---
# Seventh pass — multi-class share counts (decision taken: weighted-average fallback)
## Why this fallback, and why not the alternatives
Two candidates existed. The one **not** taken: derive the count as
`net_income ÷ diluted_eps` from columns already stored — no migration at all, and measured
accurate (GOOGL +0.48%, MRNA 0.45%, AAPL +0.19%, MSFT +0.18%). Rejected because it depends
on the derived quarter chain — the very thing these fixes have been repairing, and FOXA
already fails it — and because the two-class EPS method makes `net_income` differ from the
EPS numerator for exactly the multi-class issuers this targets.
Taken instead: store the **reported** `WeightedAverageNumberOfDilutedSharesOutstanding`.
It is the number the filer computed, needs no chain, and covers one issuer more.
| control | point-in-time | wavg diluted (latest qtr) | ratio |
|---|---|---|---|
| GOOGL | 12,230,000,000 | 12,309,000,000 | 0.9936 |
| MRNA | 396,786,259 | 395,000,000 | 1.0045 |
| AAPL | 14,687,356,000 | 14,725,873,000 | 0.9974 |
| MSFT | 7,428,434,704 | 7,445,000,000 | 0.9978 |
## Shape of the change
- **Migration 027** adds `fundamental_snapshots.weighted_avg_diluted_shares`. A separate
column, never backfilled into `shares_outstanding`, so the point-in-time column keeps its
strict meaning and the fallback stays a read-time decision.
- **Parser** stores the **shortest**-span fact ending at `period_end` (the most recent
quarter's average, closest to the current count) — deliberately not the YTD one, since an
average is not cumulative and the YTD convention does not apply.
- **Derivation** falls back only when the cover-page count is absent, and sets
`shares_outstanding_estimated`.
- **API** exposes `shares_estimated`, so `market_cap_est` and `fcf_yield` are never presented
as exact when they rest on a period average.
## Validated on live data
| issuer | shares_outstanding | estimated |
|---|---|---|
| GOOGL, AAPL, MSFT, MRNA | unchanged point-in-time values | **False** |
| META | 2,564,000,000 | True |
| CMCSA | 3,570,000,000 | True |
| CHTR | 126,849,271 | True |
| FOXA | 432,000,000 | True |
| NWSA | 555,700,000 | True |
| LEN | 240,776,000 | True |
| **BRK-B** | **still null** | False |
6 of 7 recovered, no regression on the controls. **BRK-B remains unavailable** and honestly
so: Berkshire reports per *equivalent Class A share*, dimensionally, so it has no consolidated
weighted-average fact either. Nothing in companyfacts can give it a share count.
Known caveat, accepted: for issuers using the two-class method the count is the EPS
denominator. For CHTR that is Class A only — which is also the basis on which Charter's equity
market cap is normally quoted, so it is the right number for this purpose, but it is not
"all shares of all classes".
3 tests added. Full suite: **817 passed**. Alembic single head at 027.
**Needs the reparse to land:** existing rows have `weighted_avg_diluted_shares = NULL` until
`scripts/reparse_fundamentals.py --apply` runs, so market cap stays missing for these issuers
until then.
---
# Eighth pass — revenue basis (decision: keep ASC-606, no change)
The two concepts measure different things: `RevenueFromContractWithCustomerExcludingAssessedTax`
is customer-contract revenue (an E&P's oil/gas/NGL sales), while `Revenues` is the total
income-statement line, which for commodity producers folds in mark-to-market derivative
gains/losses. That is why DVN's ASC-606 figure is *larger*: 4,508M of sales minus ~701M of
hedging losses gives the 3,807M `Revenues` line.
Measured across 21 issuers (deliberately energy-weighted, where the gap concentrates):
- Both tags present and differing >1%: **5 of 21** — DVN +18.4%, COP 14.3%, OXY +6.5%,
FCX 2.8%, PPL +1.6%. Everyone else tags one, or they are identical (COST +0.0%).
- Concept choice **flips within an issuer's chain: 0 of 21**. Whichever tag wins, the series
is internally consistent, so YoY never compares two definitions.
**Decision: keep ASC-606 first, change nothing.** Derivative gains/losses are mean-reverting
and sign-flipping; folding them into "revenue growth" turns the sub-score into a partial
hedging-P&L read for exactly the affected names. The consistency argument for switching is
empirically absent (zero flips), and changing would churn every dual-tagging issuer's stored
value — widening the reparse diff — to make ~5 names noisier.
**Correction to the fourth/fifth-pass note:** FCX's disagreement with legacy (+5.44% vs
24.23%) is **not** this ambiguity. Its two tags differ by only 2.8%, and FCX's own revenue
rose 22,703M → 25,186M YoY, so 24% is not credible — legacy is simply wrong there, and this
decision does not touch it. So the basis choice moves only DVN, COP, OXY.
The mirror hazard — an issuer where ASC-606 is only a *fragment* of revenue (a bank's fee
income) — was checked (all 15 recovered banks resolve total revenue, not a fragment). A
fragment-detection warning was prototyped and then **removed**: with no UI surface it would
only have lived in the run summary, and the case it guards against is not currently present.
Documented and closed rather than shipped as dead plumbing. If a fragment case ever appears,
it shows up as an implausibly low revenue in the next parity report.
---
# Ninth pass — basic-EPS fallback (PPL) and HAL resolved
## PPL: basic-EPS fallback (decision taken)
PPL's 2026 Q1 tags no diluted EPS variant at all, only basic — a single-filing omission
(its other quarters tag diluted), but that one missing period broke the quarter chain and
nulled TTM. `EarningsPerShareBasic` / `IncomeLossFromContinuingOperationsPerBasicShare` are
now appended to `_EPS_CONCEPTS`, last, so they only fire when no diluted variant exists.
Evidence (19-name scan): a basic fallback helps exactly **1 name (PPL)**. Basic-vs-diluted is
~0.51.2% for most, +1.2% for PPL. The one name where it genuinely diverges (TSLA +13.3%)
already tags diluted, so it never reaches the fallback. Basic is always ≥ diluted, so the
result slightly overstates EPS / understates P/E — accepted, since it fires only on an
otherwise-null period.
Validated: PPL TTM EPS null → **1.63** (≈$36 / 1.63 = 22.1 vs legacy P/E 22.43). AAPL, MSFT,
DUK, HAL unchanged — diluted still wins wherever present. 2 tests added. Full suite: **819
passed**.
## HAL: resolved, and it was never our bug
HAL's TTM EPS is now **1.81** (≈$33 / 1.81 = 18.2 vs legacy P/E 18.01) — the period-identity
and EPS-concept work already fixed it. The "unexplained null" is closed.
Its 2024 EPS values are garbage (680000, 1480000, …) because **Halliburton's own 2024 XBRL
tags `EarningsPerShareDiluted = 680000` in unit USD/shares** — a filer scale error in the
source, faithfully stored. It only poisons TTM windows that include 2024, which the current
point-in-time report does not use, so no code change: clamping EPS to "plausible" values would
risk masking real ones. Documented as a known source-data quirk.
This does surface a latent robustness point (not acted on): a single fat-fingered per-share
value poisons any TTM window it lands in. It is invisible in the current report and out of
scope here, but worth a note if historical TTM series are ever surfaced.
## All 25 lost names accounted for
| status | names |
|---|---|
| **recovered** (22) | ARE, AZO, BXP, COST, CRM, CRWD, DPZ, DVN, FCX, FRT, HAL, KHC, MOS, MTD, NTAP, PEP, PPL, REG, SJM, STX, SWKS, WDAY |
| **XOM** | fixed by the `sec_cik_overrides` pin (needs the setting applied) |
| **PSKY, Q** | new registrants without enough filing history — correct behaviour, not a bug |
## Still outstanding
Revised after the second pass, in the order I would take them:
Everything actionable without a live database is now done. What remains is one hard
data limitation and two operational steps that only run against production.
1. ~~**§3a fiscal-period identity**~~**done**, fifth pass.
2. ~~**§1 split contamination, part (a)**~~**done**, third pass.
3. ~~**§5 EPS concept gap**~~**done**, third pass.
4. ~~**§4 XOM CIK remap** + zero-filings validation~~**done**, sixth pass.
5. ~~**Reparse path**~~**done**, fourth pass.
6. ~~**Multi-class share counts**~~**done**, seventh pass (weighted-average fallback).
7. ~~**DVN/FCX revenue basis**~~**decided**, eighth pass (keep ASC-606, no change).
8. ~~**PPL basic-EPS fallback**~~**done**, ninth pass.
9. ~~**HAL null TTM EPS**~~**resolved**, ninth pass (already fixed; 2024 is a filer error).
## Review finding — two fixes on this branch silently interacted
Caught in review, not by me. `_merge_amendments` (the per-field amendment fix, first pass)
builds the merged period from `_MERGED_FIELDS` + `_CARRIED_FIELDS` alone, so a column in
neither list is **absent** from the merged row, not merely stale — and every caller reads it
with `getattr(row, name, None)`, which quietly returns `None`.
`weighted_avg_diluted_shares` (the market-cap fallback, seventh pass) was never added to
`_MERGED_FIELDS`. The failure needed both fixes to be present at once: a multi-class issuer
*and* a partial amendment on its latest period — META with a Part-III-only 10-K/A — would
silently lose market cap and FCF yield again, i.e. the seventh pass's fix undone by the
first pass's mechanism. I updated `_SNAPSHOT_COLS` in the importer when adding the column
but not `_MERGED_FIELDS` in the derivation.
Fixed, with a regression test for the specific case. The more useful addition is a guard —
`test_merge_lists_cover_every_parser_field` asserts the two lists cover every `SnapshotRow`
field, so the *next* column added fails loudly instead of losing data quietly. Verified it
would have caught this one.
Lesson worth keeping: a hand-maintained field list that reconstructs an object is a silent
data-loss footgun. `_SNAPSHOT_COLS` (importer) and `_MERGED_FIELDS` (derivation) must both
track the parser's `SnapshotRow`, and only one of them is now enforced by a test.
## Genuinely unfixable from this data
- **§1 part (b)** — a split post-dating the last filing (KLAC). No snapshot carries
share-count evidence, so no guard built on share counts can fire. Needs a corporate-actions
source or an external market-cap reconciliation.
- **BRK-B market cap** — Berkshire reports per equivalent Class A share, dimensionally, so it
has neither a cover-page count nor a weighted-average one. Nothing in companyfacts can give
it a share count.
## Operational steps (production only — cannot run from here)
- Apply the setting `sec_cik_overrides = {"XOM": 34088}`.
- Run `scripts/reparse_fundamentals.py` — dry run first, then `--apply`. This is what lands
every parser-side fix (revenue/EPS concepts, Q3 span, period identity, weighted-average
shares via migration 027) onto existing rows. Until it runs, those fixes are inert in prod.
## Standing decision, revisit only if it bites
- **COF-class share-change threshold** — kept at 25%. Revisit only if the 3.2% universe
hit-rate proves painful.
## Known source-data quirk, not acted on
- A single fat-fingered per-share value in a filer's XBRL (HAL 2024) poisons any TTM window
it lands in. Invisible in the current point-in-time report; relevant only if historical TTM
series are ever surfaced.
PSKY and Q need nothing — they are new registrants without enough filing history, which is
correct behaviour.
File diff suppressed because it is too large Load Diff
+120
View File
@@ -0,0 +1,120 @@
"""Re-derive every stored SEC snapshot with the current parser.
Snapshots are immutable per accession, so a parser fix does not reach rows that
are already stored: a normal import skips them and only logs a
``snapshot_discrepancy``. This script is the deliberate, manual exception --
it restages every accession from SEC Company Facts and rewrites the rows whose
reconstruction changed.
**Dry run by default.** Nothing is written unless ``--apply`` is passed. The dry
run stages and validates exactly as the real run does (both are read-only) and
reports the full blast radius: how many rows would change, which fields, and
per-symbol before/after samples.
Cost: a reparse cannot be served from the database -- the facts a fixed parser now
accepts were never stored -- so it refetches Company Facts for every tracked issuer
under the SEC fair-access throttle. Expect a long run and a lot of network.
Scope note: this rewrites ``fundamental_snapshots`` only. As of the A5 gate those
rows feed the fundamentals API/UI and the parity report; scoring still reads the
legacy ``fundamental_data`` table, so a reparse does not move composite scores or
backtests until the cutover happens.
Examples
--------
# dry run: report what would change, write nothing
python scripts/reparse_fundamentals.py
# dry run, showing more per-field detail
python scripts/reparse_fundamentals.py --samples 40
# actually rewrite the changed rows
python scripts/reparse_fundamentals.py --apply
"""
from __future__ import annotations
import argparse
import asyncio
import sys
from collections import Counter
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.database import async_session_factory # noqa: E402
from app.services.data_import import run_import # noqa: E402
from app.services.sec_fundamentals_importer import SecFundamentalsImporter # noqa: E402
def _parse_args() -> argparse.Namespace:
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--apply", action="store_true",
help="rewrite changed rows (default: dry run, writes nothing)")
ap.add_argument("--samples", type=int, default=20,
help="how many changed accessions to show in detail (default 20)")
return ap.parse_args()
async def _dry_run(samples: int) -> int:
importer = SecFundamentalsImporter(reparse=True)
async with async_session_factory() as db:
print("staging every tracked issuer from SEC Company Facts (this is the slow part)...")
revision = await importer.detect_revision(db)
staged = await importer.stage(db)
result = await importer.validate(db, staged)
print(f"\nrevision : {revision}")
print(f"issuers fetched : {staged.issuers_fetched}")
print(f"rows reconstructed : {len(staged.rows)}")
print(f"already stored : {len(staged.existing_accessions)}")
print(f"WOULD BE REWRITTEN : {len(staged.discrepancies)}")
print(f"new inserts : {len(staged.rows) - len(staged.existing_accessions)}")
print(f"validation ok : {result.ok}")
if not result.ok:
print(f"validation messages : {result.messages}")
if staged.discrepancies:
field_counts = Counter(f for d in staged.discrepancies for f in d["fields"])
print("\nchanged fields (accession count per field):")
for name, count in field_counts.most_common():
print(f" {name:28s} {count}")
by_accession = {r.accession: r for r in staged.rows}
print(f"\nfirst {min(samples, len(staged.discrepancies))} changed accessions:")
for d in staged.discrepancies[:samples]:
row = by_accession.get(d["accession"])
where = f"{row.cik} {row.fiscal_year} {row.fiscal_period}" if row else "?"
print(f" {d['accession']} {where:28s} {', '.join(d['fields'])}")
print(
"\nDRY RUN -- nothing was written."
"\nCheck that the changes are the *kinds* you expect (recovered nulls,"
"\ncorrected values) and sample issuers you did not anticipate before"
"\nre-running with --apply."
)
return 0 if result.ok else 1
async def _apply() -> int:
# force=True: the revision tracks SEC, which has not changed — the staleness
# is on our side, so the normal no-op gate would skip this.
run = await run_import(SecFundamentalsImporter(reparse=True), force=True)
if run is None:
print("another sec_facts import holds the lock; nothing done")
return 1
print(f"run {run.id}: status={run.status}")
print(f" revision : {run.revision}")
print(f" row_counts : {run.row_counts_json}")
if run.error_details:
print(f" error : {run.error_details}")
return 0 if run.status == "promoted" else 1
def main() -> int:
args = _parse_args()
return asyncio.run(_apply() if args.apply else _dry_run(args.samples))
if __name__ == "__main__":
raise SystemExit(main())
+143
View File
@@ -29,6 +29,7 @@ class Snap:
cash_and_st_investments: float | None = None cash_and_st_investments: float | None = None
total_debt: float | None = None total_debt: float | None = None
shares_outstanding: float | None = None shares_outstanding: float | None = None
weighted_avg_diluted_shares: float | None = None
_FP = ["Q1", "Q2", "Q3", "FY"] _FP = ["Q1", "Q2", "Q3", "FY"]
@@ -188,3 +189,145 @@ def test_amendment_selection_newest_accepted_wins():
# Q4 revenue discrete now uses the amended YTD(FY)=999999 minus YTD(Q3)=363 # Q4 revenue discrete now uses the amended YTD(FY)=999999 minus YTD(Q3)=363
# so TTM/growth reflects the amendment, proving newest accepted_at won. # so TTM/growth reflects the amendment, proving newest accepted_at won.
assert d.metrics["revenue_growth_yoy"].value != pytest.approx(10.0, abs=1e-6) assert d.metrics["revenue_growth_yoy"].value != pytest.approx(10.0, abs=1e-6)
# -- partial amendments (A5 parity findings) ---------------------------------
def test_partial_amendment_does_not_blank_the_period():
# DVN's FY2025 10-K/A carries no financial facts at the report date. Taking
# the newest accession wholesale nulled the period, and with it the quarter
# chain, TTM and YoY.
rows = _two_years()
part_iii_only = Snap(2026, "FY", date(2026, 9, 30), date(2026, 11, 1),
datetime(2027, 1, 1, tzinfo=UTC))
baseline = fd.derive(rows)
d = fd.derive(rows + [part_iii_only])
assert d.ttm_diluted_eps == pytest.approx(baseline.ttm_diluted_eps)
assert d.metrics["revenue_growth_yoy"].value == pytest.approx(
baseline.metrics["revenue_growth_yoy"].value
)
def test_amendment_restating_one_field_leaves_the_others_intact():
rows = _two_years()
revenue_only = Snap(2026, "FY", date(2026, 9, 30), date(2026, 11, 1),
datetime(2027, 1, 1, tzinfo=UTC), revenue=999999)
baseline = fd.derive(rows)
d = fd.derive(rows + [revenue_only])
assert d.metrics["revenue_growth_yoy"].value != pytest.approx(
baseline.metrics["revenue_growth_yoy"].value
)
assert d.ttm_diluted_eps == pytest.approx(baseline.ttm_diluted_eps) # fell back
def test_same_key_row_for_a_different_period_is_never_merged():
# SEC labels two different year-ends with one fiscal_year for some filers
# (FRT, CRM). That is a mislabelled filing, not an amendment -- merging the
# two would silently blend fiscal years.
rows = _two_years()
mislabelled = Snap(2026, "FY", date(2027, 9, 30), date(2027, 11, 1),
datetime(2027, 12, 1, tzinfo=UTC), revenue=999999)
selected = fd._select_latest_per_period(rows + [mislabelled])
assert selected[(2026, "FY")] is mislabelled
# -- split safety for the TTM EPS scalar (A5 parity findings) ----------------
def _split_rows():
"""Two years where the share count jumps ~25x at the latest quarter, as
BKNG's did (31.7M -> 774.9M) when its split landed mid-window."""
rows = _two_years()
for row in rows:
if (row.fiscal_year, row.fiscal_period) == (2026, "FY"):
row.shares_outstanding = 25000.0 # vs 1000 a year earlier
return rows
def test_split_suppresses_ttm_diluted_eps():
# TTM sums four quarters of per-share values; a split inside the window
# mixes units. Unguarded this produced BKNG's P/E of 1.10, which clamps to a
# *perfect* fundamental sub-score -- worse than having no value at all.
d = fd.derive(_split_rows())
assert d.ttm_diluted_eps is None
assert d.ttm_diluted_eps_caveat == fd.SPLIT_SENSITIVE_CAVEAT
def test_ttm_diluted_eps_survives_when_no_split_is_suspected():
d = fd.derive(_two_years())
assert d.ttm_diluted_eps is not None
assert d.ttm_diluted_eps_caveat is None
def test_split_guard_leaves_dollar_scalars_alone():
# Only per-share values are split-sensitive; FCF is in dollars.
baseline = fd.derive(_two_years())
d = fd.derive(_split_rows())
assert d.ttm_fcf == pytest.approx(baseline.ttm_fcf)
# -- multi-class share-count fallback (A5 parity findings) -------------------
def test_shares_fall_back_to_weighted_average_when_cover_page_count_is_absent():
# META/CMCSA/BRK-B/CHTR report the cover-page count per share class, which is
# dimensional and therefore absent from companyfacts -- silently removing
# market cap and FCF yield for some of the largest issuers.
rows = _two_years()
for row in rows:
row.shares_outstanding = None
row.weighted_avg_diluted_shares = 2_564_000_000.0
d = fd.derive(rows)
assert d.shares_outstanding == 2_564_000_000.0
assert d.shares_outstanding_estimated is True
def test_point_in_time_share_count_is_preferred_and_not_flagged():
baseline = fd.derive(_two_years()).shares_outstanding
assert baseline is not None, "fixture should carry a cover-page count"
rows = _two_years()
for row in rows:
row.weighted_avg_diluted_shares = 1.0 # must lose to the real count
d = fd.derive(rows)
assert d.shares_outstanding == baseline
assert d.shares_outstanding_estimated is False
def test_weighted_average_fallback_survives_a_partial_amendment():
# A Part-III-only 10-K/A on a multi-class issuer's latest period: the merged
# row must keep the weighted-average count, or market cap silently vanishes.
rows = _two_years()
for row in rows:
row.shares_outstanding = None
row.weighted_avg_diluted_shares = 2_564_000_000.0
part_iii_only = Snap(2026, "FY", date(2026, 9, 30), date(2026, 11, 1),
datetime(2027, 1, 1, tzinfo=UTC))
d = fd.derive(rows + [part_iii_only])
assert d.shares_outstanding == 2_564_000_000.0
assert d.shares_outstanding_estimated is True
def test_no_share_count_at_all_stays_none_and_unflagged():
rows = _two_years()
for row in rows:
row.shares_outstanding = None
d = fd.derive(rows)
assert d.shares_outstanding is None
assert d.shares_outstanding_estimated is False
def test_merge_lists_cover_every_parser_field():
"""_MERGED_FIELDS/_CARRIED_FIELDS are hand-maintained, and _merge_amendments
builds the merged row from them alone — so a parser field missing from both
is not merely stale on a merged period, it is *absent*, and callers using
getattr(row, name, None) read None. That is how weighted_avg_diluted_shares
silently lost market cap for multi-class issuers with a partial amendment.
Adding a column to SnapshotRow must fail here rather than lose data quietly.
"""
import dataclasses
from app.services.sec_facts_parser import SnapshotRow
parser_fields = {f.name for f in dataclasses.fields(SnapshotRow)}
covered = set(fd._MERGED_FIELDS) | set(fd._CARRIED_FIELDS)
assert not parser_fields - covered, (
f"parser fields not merged or carried: {sorted(parser_fields - covered)}"
)
+224
View File
@@ -270,3 +270,227 @@ async def test_live_apple_parse_invariants():
# shares cover-date differs from period_end # shares cover-date differs from period_end
latest = max(rows, key=lambda r: r.period_end) latest = max(rows, key=lambda r: r.period_end)
assert latest.shares_outstanding_date != latest.period_end assert latest.shares_outstanding_date != latest.period_end
# -- revenue concept coverage (A5 parity findings) ---------------------------
def _one_filing(concepts: dict, *, start: str, end: str, fp: str):
"""A single 10-Q whose facts are the given {concept: value} at one YTD span."""
facts = {
name: {"units": {"USD": [_dur(start, end, val, "X", fp=fp)]}}
for name, val in concepts.items()
}
companyfacts = {"cik": 19617, "facts": {"us-gaap": facts}}
filings = {
"X": FilingMeta(
date.fromisoformat(end), date(2026, 5, 1), datetime(2026, 5, 1, 10, tzinfo=UTC), "10-Q"
)
}
return parse_snapshots(companyfacts, filings, {"X"})
def test_revenue_reads_banks_total_revenue_tag():
# JPM/GS/WFC tag RevenuesNetOfInterestExpense in every 10-Q and never (or
# only annually) `Revenues` -- previously null, so revenue growth was too.
res = _one_filing(
{"RevenuesNetOfInterestExpense": 49836}, start="2026-01-01", end="2026-03-31", fp="Q1"
)
assert res.rows[0].revenue == 49836
def test_revenue_reads_including_assessed_tax_variant():
# ARE/KHC tag only the Including variant.
res = _one_filing(
{"RevenueFromContractWithCustomerIncludingAssessedTax": 671},
start="2026-01-01", end="2026-03-31", fp="Q1",
)
assert res.rows[0].revenue == 671
def test_revenue_concept_priority_is_unchanged_by_the_added_tags():
# The new entries are appended, so any issuer that already resolved keeps
# the same concept -- only issuers that resolved to nothing gain a value.
res = _one_filing(
{
"RevenueFromContractWithCustomerExcludingAssessedTax": 100,
"RevenueFromContractWithCustomerIncludingAssessedTax": 110,
"RevenuesNetOfInterestExpense": 120,
"Revenues": 130,
},
start="2026-01-01", end="2026-03-31", fp="Q1",
)
assert res.rows[0].revenue == 100
def test_four_four_five_q3_ytd_span_is_accepted():
# A 12/12/12/16-week filer's YTD-Q3 is 36 weeks = 251 days (COST 2026 Q3),
# which missed the old 20-day tolerance around 273 by ~2 and dropped Q3
# every year -- breaking the quarter chain and nulling TTM and YoY.
res = _one_filing(
{"RevenueFromContractWithCustomerExcludingAssessedTax": 207431},
start="2025-09-01", end="2026-05-10", fp="Q3",
)
assert (date(2026, 5, 10) - date(2025, 9, 1)).days == 251
assert res.rows[0].revenue == 207431
def test_eps_falls_back_to_continuing_operations_variant():
# REG tags only this variant on every filing; FCX tags it in its 10-K while
# using EarningsPerShareDiluted in its 10-Qs.
companyfacts = {
"cik": 910606,
"facts": {"us-gaap": {"IncomeLossFromContinuingOperationsPerDilutedShare": {
"units": {"USD/shares": [_dur("2026-01-01", "2026-03-31", 1.81, "X", fp="Q1")]}
}}},
}
filings = {"X": FilingMeta(date(2026, 3, 31), date(2026, 5, 1),
datetime(2026, 5, 1, 10, tzinfo=UTC), "10-Q")}
res = parse_snapshots(companyfacts, filings, {"X"})
assert res.rows[0].diluted_eps == 1.81
def test_eps_concept_priority_is_unchanged_by_the_added_tag():
companyfacts = {
"cik": 831259,
"facts": {"us-gaap": {
"EarningsPerShareDiluted": {
"units": {"USD/shares": [_dur("2026-01-01", "2026-03-31", 0.61, "X", fp="Q1")]}},
"IncomeLossFromContinuingOperationsPerDilutedShare": {
"units": {"USD/shares": [_dur("2026-01-01", "2026-03-31", 0.75, "X", fp="Q1")]}},
}},
}
filings = {"X": FilingMeta(date(2026, 3, 31), date(2026, 5, 1),
datetime(2026, 5, 1, 10, tzinfo=UTC), "10-Q")}
res = parse_snapshots(companyfacts, filings, {"X"})
assert res.rows[0].diluted_eps == 0.61
# -- period identity from the fiscal calendar, not SEC's fy/fp ---------------
from app.services.sec_facts_parser import _period_identity # noqa: E402
def _meta(end: str, form: str = "10-Q") -> FilingMeta:
d = date.fromisoformat(end)
return FilingMeta(d, d, datetime(d.year, d.month, d.day, tzinfo=UTC), form)
def test_a_10q_is_never_labelled_fy():
# BXP: a 10-Q for period end 2026-03-31 carried fy/fp saying "2026 FY", which
# collided with the real annual row and measured a 90-day fact against the
# 365-day FY expectation.
fy, fp = _period_identity(_meta("2026-03-31"), "1231")
assert (fy, fp) == (2026, "Q1")
def test_december_filer_years_do_not_collide():
# FRT: two 10-Ks, ending 2024-12-31 and 2025-12-31, both labelled "2024 FY".
assert _period_identity(_meta("2024-12-31", "10-K"), "1231") == (2024, "FY")
assert _period_identity(_meta("2025-12-31", "10-K"), "1231") == (2025, "FY")
def test_january_year_end_groups_its_quarters():
# CRM/CRWD/WDAY: the year ending 2026-01-31 and its own quarters must share a
# fiscal year, and must not collide with the year ending 2025-01-31.
assert _period_identity(_meta("2026-01-31", "10-K"), "0131") == (2026, "FY")
assert _period_identity(_meta("2025-01-31", "10-K"), "0131") == (2025, "FY")
assert _period_identity(_meta("2025-04-30"), "0131") == (2026, "Q1")
assert _period_identity(_meta("2025-07-31"), "0131") == (2026, "Q2")
assert _period_identity(_meta("2025-10-31"), "0131") == (2026, "Q3")
def test_mid_year_end_orders_correctly():
# STX: the year ending 2025-06-27 was labelled "2027 FY" and sorted after
# quarters that precede it.
assert _period_identity(_meta("2025-06-27", "10-K"), "0627") == (2025, "FY")
assert _period_identity(_meta("2025-10-03"), "0627") == (2026, "Q1")
assert _period_identity(_meta("2026-01-02"), "0627") == (2026, "Q2")
assert _period_identity(_meta("2026-04-03"), "0627") == (2026, "Q3")
def test_four_four_five_quarters_place_correctly():
# COST: a 12/12/12/16-week year leaves Q3 112 days from the year end, not 91.
assert _period_identity(_meta("2025-11-23"), "0830") == (2026, "Q1")
assert _period_identity(_meta("2026-02-15"), "0830") == (2026, "Q2")
assert _period_identity(_meta("2026-05-10"), "0830") == (2026, "Q3")
assert _period_identity(_meta("2026-08-30", "10-K"), "0830") == (2026, "FY")
def test_year_end_crossing_january_still_groups_one_year():
# DPZ (fiscalYearEnd 0102): the label shifts by one against Domino's own
# naming, which is fine -- a year and its quarters must simply agree.
year, _ = _period_identity(_meta("2025-12-28", "10-K"), "0102")
assert (year, "FY") == _period_identity(_meta("2025-12-28", "10-K"), "0102")
assert _period_identity(_meta("2025-03-23"), "0102") == (year, "Q1")
assert _period_identity(_meta("2025-06-15"), "0102") == (year, "Q2")
assert _period_identity(_meta("2025-09-07"), "0102") == (year, "Q3")
def test_missing_fiscal_calendar_falls_back_to_filing_context():
assert _period_identity(_meta("2026-03-31"), None) == (None, None)
# ...and parse_snapshots then uses the fy/fp path, preserving old behaviour.
res = parse_snapshots(COMPANYFACTS, FILINGS, {"B"})
assert (res.rows[0].fiscal_year, res.rows[0].fiscal_period) == (2026, "Q2")
def test_eps_falls_back_to_basic_only_when_no_diluted_variant_exists():
# PPL's 2026 Q1 tags no diluted EPS at all, only basic -- one missing period
# broke the quarter chain and nulled TTM.
companyfacts = {
"cik": 922224,
"facts": {"us-gaap": {"EarningsPerShareBasic": {
"units": {"USD/shares": [_dur("2026-01-01", "2026-03-31", 0.60, "X", fp="Q1")]}
}}},
}
filings = {"X": FilingMeta(date(2026, 3, 31), date(2026, 5, 1),
datetime(2026, 5, 1, 10, tzinfo=UTC), "10-Q")}
res = parse_snapshots(companyfacts, filings, {"X"}, fiscal_year_end="1231")
assert res.rows[0].diluted_eps == 0.60
def test_diluted_still_wins_over_basic_when_both_present():
companyfacts = {
"cik": 320193,
"facts": {"us-gaap": {
"EarningsPerShareDiluted": {
"units": {"USD/shares": [_dur("2026-01-01", "2026-03-31", 1.36, "X", fp="Q1")]}},
"EarningsPerShareBasic": {
"units": {"USD/shares": [_dur("2026-01-01", "2026-03-31", 1.40, "X", fp="Q1")]}},
}},
}
filings = {"X": FilingMeta(date(2026, 3, 31), date(2026, 5, 1),
datetime(2026, 5, 1, 10, tzinfo=UTC), "10-Q")}
res = parse_snapshots(companyfacts, filings, {"X"}, fiscal_year_end="1231")
assert res.rows[0].diluted_eps == 1.36
def test_weighted_average_shares_prefers_the_shortest_span():
# A 10-Q carries both the quarter's average and the YTD one. The shorter
# window sits closer to the current count, which is what market cap wants.
companyfacts = {
"cik": 1326801,
"facts": {"us-gaap": {"WeightedAverageNumberOfDilutedSharesOutstanding": {
"units": {"shares": [
_dur("2026-01-01", "2026-09-30", 2_600_000_000, "X", fp="Q3"), # YTD
_dur("2026-07-01", "2026-09-30", 2_564_000_000, "X", fp="Q3"), # quarter
]}
}}},
}
filings = {"X": FilingMeta(date(2026, 9, 30), date(2026, 11, 1),
datetime(2026, 11, 1, 10, tzinfo=UTC), "10-Q")}
res = parse_snapshots(companyfacts, filings, {"X"}, fiscal_year_end="1231")
assert res.rows[0].weighted_avg_diluted_shares == 2_564_000_000
def test_weighted_average_shares_falls_back_to_the_basic_and_diluted_concept():
companyfacts = {
"cik": 1326801,
"facts": {"us-gaap": {"WeightedAverageNumberOfSharesOutstandingBasicAndDiluted": {
"units": {"shares": [_dur("2026-07-01", "2026-09-30", 500_000, "X", fp="Q3")]}
}}},
}
filings = {"X": FilingMeta(date(2026, 9, 30), date(2026, 11, 1),
datetime(2026, 11, 1, 10, tzinfo=UTC), "10-Q")}
res = parse_snapshots(companyfacts, filings, {"X"}, fiscal_year_end="1231")
assert res.rows[0].weighted_avg_diluted_shares == 500_000
@@ -411,3 +411,181 @@ async def test_discrepancy_in_shares_is_detected_and_reported(engine):
assert k.shares_outstanding == 999.0 and k.import_run_id == 1 # immutable — not overwritten assert k.shares_outstanding == 999.0 and k.import_run_id == 1 # immutable — not overwritten
events = (await s.execute(select(SystemEvent).where(SystemEvent.code == "snapshot_discrepancy"))).scalars().all() events = (await s.execute(select(SystemEvent).where(SystemEvent.code == "snapshot_discrepancy"))).scalars().all()
assert len(events) == 1 and events[0].severity == "warning" assert len(events) == 1 and events[0].severity == "warning"
# --- reparse: rewriting rows a fixed parser reconstructs differently --------
# A 4-4-5 filer's YTD-Q3 span (36 weeks = 251 days). The old 20-day tolerance
# around 273 rejected it and stored revenue=None; 25 accepts it. Reparsing with
# the fixed parser is exactly the situation this mode exists for.
CF_Q3_445 = _rev("2025-09-01", "2026-05-10", 207431, 2026, "Q3", "Q3F")
SUB_445 = [_filing("Q3F", "10-Q", "2026-05-10", "2026-06-01", "2026-06-01T10:01:00.000Z")]
def _445_client():
return FakeSecClient(
tickers={"AAPL": 320193},
companyfacts={320193: _companyfacts([CF_Q3_445], [_shares("2026-05-15", 100, "Q3F", 2026, "Q3")])},
submissions={320193: _submissions(SUB_445)},
latest_index=date(2026, 6, 1),
)
async def _import_with_old_tolerance(engine, monkeypatch):
"""Seed the DB the way the pre-fix parser did: Q3 revenue rejected -> null."""
from app.services import sec_facts_parser
monkeypatch.setattr(sec_facts_parser, "_YTD_TOLERANCE_DAYS", 20)
run = await run_import(_importer(_445_client(), today=date(2026, 6, 2)), engine=engine)
monkeypatch.undo()
return run
async def test_reparse_rewrites_rows_the_fixed_parser_reads_differently(engine, monkeypatch):
factory = _factory(engine)
await _seed(factory, ["AAPL"])
first = await _import_with_old_tolerance(engine, monkeypatch)
async with factory() as s:
stale = (await s.execute(select(FundamentalSnapshot))).scalar_one()
assert stale.revenue is None, "precondition: the old parser stored a null"
# Reparse with the current (fixed) parser. force=True because SEC has not
# changed -- the staleness is on our side, so the revision gate would no-op.
run = await run_import(
SecFundamentalsImporter(
client_factory=lambda: _445_client(), today=date(2026, 6, 2), reparse=True
),
engine=engine,
force=True,
)
assert run.status == STATUS_PROMOTED
assert '"updated": 1' in run.row_counts_json
async with factory() as s:
fixed = (await s.execute(select(FundamentalSnapshot))).scalar_one()
assert fixed.revenue == 207431 # rewritten in place
assert fixed.accession == stale.accession
assert fixed.import_run_id == run.id # rewrite is attributable
assert fixed.import_run_id != first.id
async def test_reparse_leaves_unchanged_rows_untouched(engine):
factory = _factory(engine)
await _seed(factory, ["AAPL"])
first = await run_import(_importer(_445_client(), today=date(2026, 6, 2)), engine=engine)
run = await run_import(
SecFundamentalsImporter(
client_factory=lambda: _445_client(), today=date(2026, 6, 2), reparse=True
),
engine=engine,
force=True,
)
assert '"updated": 0' in run.row_counts_json
async with factory() as s:
row = (await s.execute(select(FundamentalSnapshot))).scalar_one()
assert row.import_run_id == first.id # provenance preserved, no needless rewrite
async def test_without_reparse_a_differing_row_stays_immutable(engine, monkeypatch):
"""The default contract is unchanged: report the discrepancy, never mutate."""
factory = _factory(engine)
await _seed(factory, ["AAPL"])
await _import_with_old_tolerance(engine, monkeypatch)
importer = SecFundamentalsImporter(
client_factory=lambda: _445_client(), today=date(2026, 6, 2), reparse=True
)
async with _factory(engine)() as db:
await importer.detect_revision(db)
staged = await importer.stage(db)
importer.reparse = False # same staged diff, default disposition
counts = await importer.promote(db, staged, run_id=999)
await db.commit()
assert staged.discrepancies, "the diff should still be detected and reported"
assert counts["updated"] == 0
async with factory() as s:
row = (await s.execute(select(FundamentalSnapshot))).scalar_one()
assert row.revenue is None # untouched
async def test_force_bypasses_the_unchanged_revision_no_op(engine):
factory = _factory(engine)
await _seed(factory, ["AAPL"])
await run_import(_importer(_445_client(), today=date(2026, 6, 2)), engine=engine)
same = _importer(_445_client(), today=date(2026, 6, 2))
assert (await run_import(same, engine=engine)).status == "no_op"
forced = SecFundamentalsImporter(
client_factory=lambda: _445_client(), today=date(2026, 6, 2), reparse=True
)
assert (await run_import(forced, engine=engine, force=True)).status == STATUS_PROMOTED
# --- CIK resolution: successor registrants with no filings -----------------
async def test_issuer_with_no_xbrl_filings_is_reported_not_silent(engine):
"""XOM resolved to CIK 2115436 'ExxonMobil Holdings Corp', which has zero
filings, so it produced no snapshots and nothing said why."""
factory = _factory(engine)
await _seed(factory, ["AAPL"])
client = FakeSecClient(
tickers={"AAPL": 320193},
companyfacts={320193: _companyfacts([CF_K], [SH_K])},
submissions={320193: {**_submissions([]), "name": "Shell Holdings Corp"}},
latest_index=date(2026, 1, 31),
)
importer = _importer(client)
async with factory() as db:
await importer.detect_revision(db)
staged = await importer.stage(db)
result = await importer.validate(db, staged)
assert result.summary["no_xbrl_filings_count"] == 1
assert staged.no_xbrl_filings[0]["cik"] == "0000320193"
assert staged.no_xbrl_filings[0]["name"] == "Shell Holdings Corp"
async def test_cik_override_pins_a_ticker_to_the_real_filer(engine):
from app.models.settings import SystemSetting
from app.services.sec_universe import CIK_OVERRIDES_KEY, resolve_ciks
factory = _factory(engine)
await _seed(factory, ["AAPL"])
async with factory() as s:
s.add(SystemSetting(key=CIK_OVERRIDES_KEY, value='{"AAPL": 34088}'))
await s.commit()
client = FakeSecClient(
tickers={"AAPL": 320193}, # SEC points at the wrong registrant
companyfacts={}, submissions={}, latest_index=date(2026, 1, 31),
)
async with factory() as db:
resolved = await resolve_ciks(db, client)
assert resolved.symbol_to_cik["AAPL"] == 34088
assert resolved.cik_updates == [(1, "0000034088")]
async def test_malformed_cik_override_is_ignored_not_fatal(engine):
from app.models.settings import SystemSetting
from app.services.sec_universe import CIK_OVERRIDES_KEY, resolve_ciks
factory = _factory(engine)
await _seed(factory, ["AAPL"])
async with factory() as s:
s.add(SystemSetting(key=CIK_OVERRIDES_KEY, value="not json at all"))
await s.commit()
client = FakeSecClient(
tickers={"AAPL": 320193}, companyfacts={}, submissions={},
latest_index=date(2026, 1, 31),
)
async with factory() as db:
resolved = await resolve_ciks(db, client)
assert resolved.symbol_to_cik["AAPL"] == 320193 # fell back to company_tickers