diff --git a/alembic/versions/026_dolt_fundamentals_schema.py b/alembic/versions/026_dolt_fundamentals_schema.py new file mode 100644 index 0000000..3e2dd8c --- /dev/null +++ b/alembic/versions/026_dolt_fundamentals_schema.py @@ -0,0 +1,144 @@ +"""Dolt/SEC fundamentals schema — workstream A + +Revision ID: 026 +Revises: 025 +Create Date: 2026-07-21 00:00:00.000000 + +Foundational schema for the Dolt bulk-data integration (workstream A): the +batch import-run audit table, the SEC-sourced immutable fundamental snapshots +(CIK-keyed, one row per accession), the Dolt earnings calendar/history, and the +SEC issuer identity columns on ``tickers``. No data is populated here — the +importers land in a later phase. ``fundamental_data`` is left untouched; its +cutover is gated separately (phase A5). ``data_import_runs`` is created first +because the other two tables carry an ``import_run_id`` FK to it. +""" +from typing import Sequence, Union + +from alembic import op +import sqlalchemy as sa + + +revision: str = "026" +down_revision: Union[str, None] = "025" +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + op.create_table( + "data_import_runs", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("source", sa.String(length=32), nullable=False), + sa.Column("revision", sa.String(length=64), nullable=True), + sa.Column("status", sa.String(length=16), nullable=False), + sa.Column("source_max_date", sa.Date(), nullable=True), + sa.Column("row_counts_json", sa.Text(), nullable=True), + sa.Column("validation_json", sa.Text(), nullable=True), + sa.Column("started_at", sa.DateTime(timezone=True), nullable=False), + sa.Column("completed_at", sa.DateTime(timezone=True), nullable=True), + sa.Column("error_details", sa.Text(), nullable=True), + ) + op.create_index( + "ix_data_import_runs_source_started", "data_import_runs", ["source", "started_at"] + ) + + op.create_table( + "fundamental_snapshots", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column("cik", sa.String(length=10), nullable=False), + sa.Column("accession", sa.String(length=25), nullable=False), + sa.Column("form", sa.String(length=12), nullable=False), + sa.Column("filed_date", sa.Date(), nullable=False), + sa.Column("accepted_at", sa.DateTime(timezone=True), nullable=False), + sa.Column("period_start", sa.Date(), nullable=True), + sa.Column("period_end", sa.Date(), nullable=False), + sa.Column("fiscal_year", sa.Integer(), nullable=False), + sa.Column("fiscal_period", sa.String(length=4), nullable=False), + # duration facts — cumulative YTD/FY + sa.Column("revenue", sa.Float(), nullable=True), + sa.Column("net_income", sa.Float(), nullable=True), + sa.Column("operating_income", sa.Float(), nullable=True), + sa.Column("diluted_eps", sa.Float(), nullable=True), + sa.Column("cfo", sa.Float(), nullable=True), + sa.Column("capex", sa.Float(), nullable=True), + sa.Column("depreciation_amortization", sa.Float(), nullable=True), + # balance-sheet facts — period-end + sa.Column("cash_and_st_investments", sa.Float(), nullable=True), + sa.Column("total_debt", sa.Float(), nullable=True), + sa.Column("shares_outstanding", sa.Float(), nullable=True), + sa.Column( + "import_run_id", + sa.Integer(), + sa.ForeignKey("data_import_runs.id", ondelete="SET NULL"), + nullable=True, + ), + sa.Column("created_at", sa.DateTime(timezone=True), nullable=False), + sa.UniqueConstraint("accession", name="uq_fundamental_snapshots_accession"), + ) + op.create_index( + "ix_fundamental_snapshots_cik_period", + "fundamental_snapshots", + ["cik", "fiscal_year", "fiscal_period"], + ) + op.create_index( + "ix_fundamental_snapshots_cik_period_end", + "fundamental_snapshots", + ["cik", "period_end"], + ) + + op.create_table( + "earnings_events", + sa.Column("id", sa.Integer(), primary_key=True), + sa.Column( + "ticker_id", + sa.Integer(), + sa.ForeignKey("tickers.id", ondelete="CASCADE"), + nullable=False, + ), + sa.Column("announce_date", sa.Date(), nullable=False), + sa.Column("session", sa.String(length=10), nullable=False), + sa.Column("period_end", sa.Date(), nullable=True), + sa.Column("eps_estimate", sa.Float(), nullable=True), + sa.Column("eps_actual", sa.Float(), nullable=True), + sa.Column("source", sa.String(length=32), nullable=False), + sa.Column( + "import_run_id", + sa.Integer(), + sa.ForeignKey("data_import_runs.id", ondelete="SET NULL"), + nullable=True, + ), + sa.Column("created_at", sa.DateTime(timezone=True), nullable=False), + sa.UniqueConstraint("ticker_id", "announce_date", name="uq_earnings_ticker_announce"), + ) + op.create_index( + "ix_earnings_events_announce_date", "earnings_events", ["announce_date"] + ) + + # SEC issuer identity on tickers (nullable; the only ticker<->issuer join point). + op.add_column("tickers", sa.Column("cik", sa.String(length=10), nullable=True)) + op.add_column("tickers", sa.Column("sic", sa.String(length=4), nullable=True)) + op.add_column( + "tickers", sa.Column("sic_description", sa.String(length=160), nullable=True) + ) + + +def downgrade() -> None: + op.drop_column("tickers", "sic_description") + op.drop_column("tickers", "sic") + op.drop_column("tickers", "cik") + + op.drop_index("ix_earnings_events_announce_date", table_name="earnings_events") + op.drop_table("earnings_events") + + op.drop_index( + "ix_fundamental_snapshots_cik_period_end", table_name="fundamental_snapshots" + ) + op.drop_index( + "ix_fundamental_snapshots_cik_period", table_name="fundamental_snapshots" + ) + op.drop_table("fundamental_snapshots") + + op.drop_index( + "ix_data_import_runs_source_started", table_name="data_import_runs" + ) + op.drop_table("data_import_runs") diff --git a/app/models/__init__.py b/app/models/__init__.py index b59a7c8..eedcf09 100644 --- a/app/models/__init__.py +++ b/app/models/__init__.py @@ -3,6 +3,9 @@ from app.models.ohlcv import OHLCVRecord from app.models.user import User from app.models.sentiment import SentimentScore from app.models.fundamental import FundamentalData +from app.models.fundamental_snapshot import FundamentalSnapshot +from app.models.earnings_event import EarningsEvent +from app.models.data_import_run import DataImportRun from app.models.score import DimensionScore, CompositeScore from app.models.sr_level import SRLevel from app.models.trade_setup import TradeSetup @@ -21,6 +24,9 @@ __all__ = [ "User", "SentimentScore", "FundamentalData", + "FundamentalSnapshot", + "EarningsEvent", + "DataImportRun", "DimensionScore", "CompositeScore", "SRLevel", diff --git a/app/models/data_import_run.py b/app/models/data_import_run.py new file mode 100644 index 0000000..92507ac --- /dev/null +++ b/app/models/data_import_run.py @@ -0,0 +1,42 @@ +from datetime import date, datetime + +from sqlalchemy import Date, DateTime, Index, String, Text +from sqlalchemy.orm import Mapped, mapped_column + +from app.database import Base + + +class DataImportRun(Base): + """One row per bulk-import attempt (SEC facts / Dolt earnings / Dolt stocks). + + Lean audit record for the batch import framework: every attempt is logged, + whether it promoted, was a ``no_op`` (unchanged revision), or ``failed``. + ``row_counts`` and ``validation`` hold JSON strings (repo convention — see + ``fundamental_data.unavailable_fields_json``), not JSONB; the validation + blob carries reconciliation/discrepancy summaries so no separate conflicts + table is needed. One run per source at a time is enforced at write time by a + Postgres advisory lock keyed by ``source``. + """ + + __tablename__ = "data_import_runs" + __table_args__ = ( + Index("ix_data_import_runs_source_started", "source", "started_at"), + ) + + id: Mapped[int] = mapped_column(primary_key=True) + # sec_facts | dolt_earnings | dolt_stocks + source: Mapped[str] = mapped_column(String(32), nullable=False) + # Dolt commit hash, or SEC archive SHA-256. Null until known. + revision: Mapped[str | None] = mapped_column(String(64), nullable=True) + # running | validated | promoted | no_op | failed + status: Mapped[str] = mapped_column(String(16), nullable=False) + source_max_date: Mapped[date | None] = mapped_column(Date, nullable=True) + row_counts_json: Mapped[str | None] = mapped_column(Text, nullable=True) + validation_json: Mapped[str | None] = mapped_column(Text, nullable=True) + started_at: Mapped[datetime] = mapped_column( + DateTime(timezone=True), default=datetime.utcnow, nullable=False + ) + completed_at: Mapped[datetime | None] = mapped_column( + DateTime(timezone=True), nullable=True + ) + error_details: Mapped[str | None] = mapped_column(Text, nullable=True) diff --git a/app/models/earnings_event.py b/app/models/earnings_event.py new file mode 100644 index 0000000..561098d --- /dev/null +++ b/app/models/earnings_event.py @@ -0,0 +1,42 @@ +from datetime import date, datetime + +from sqlalchemy import Date, DateTime, Float, ForeignKey, Index, String, UniqueConstraint +from sqlalchemy.orm import Mapped, mapped_column, relationship + +from app.database import Base + + +class EarningsEvent(Base): + """Earnings calendar + surprise history, sourced from the DoltHub earnings repo. + + Forward rows (``announce_date`` > today) are the calendar; past rows are + results. Rescheduling is handled in the importer's promotion transaction: + this source's future-dated rows are deleted and re-inserted from the new + snapshot so moved/cancelled dates never linger; past rows are never deleted. + """ + + __tablename__ = "earnings_events" + __table_args__ = ( + UniqueConstraint("ticker_id", "announce_date", name="uq_earnings_ticker_announce"), + Index("ix_earnings_events_announce_date", "announce_date"), + ) + + id: Mapped[int] = mapped_column(primary_key=True) + ticker_id: Mapped[int] = mapped_column( + ForeignKey("tickers.id", ondelete="CASCADE"), nullable=False + ) + announce_date: Mapped[date] = mapped_column(Date, nullable=False) + # bmo | amc | unknown (source coverage is partial) + session: Mapped[str] = mapped_column(String(10), nullable=False, default="unknown") + period_end: Mapped[date | None] = mapped_column(Date, nullable=True) + eps_estimate: Mapped[float | None] = mapped_column(Float, nullable=True) + eps_actual: Mapped[float | None] = mapped_column(Float, nullable=True) + source: Mapped[str] = mapped_column(String(32), nullable=False) + import_run_id: Mapped[int | None] = mapped_column( + ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True + ) + created_at: Mapped[datetime] = mapped_column( + DateTime(timezone=True), default=datetime.utcnow, nullable=False + ) + + ticker = relationship("Ticker", back_populates="earnings_events") diff --git a/app/models/fundamental_snapshot.py b/app/models/fundamental_snapshot.py new file mode 100644 index 0000000..4e85885 --- /dev/null +++ b/app/models/fundamental_snapshot.py @@ -0,0 +1,73 @@ +from datetime import date, datetime + +from sqlalchemy import Date, DateTime, Float, ForeignKey, Index, String, UniqueConstraint +from sqlalchemy.orm import Mapped, mapped_column + +from app.database import Base + + +class FundamentalSnapshot(Base): + """CIK-keyed, one immutable row per SEC accession. + + Keyed by issuer (CIK), not ticker — multi-class issuers (GOOG/GOOGL) share + one CIK and one set of fundamentals; the ``tickers.cik`` column is the only + join point. Amendments are retained: every accession is a distinct immutable + row, and readers pick the newest valid ``accepted_at`` per + (cik, fiscal_year, fiscal_period) at read time — no flags, no mutation. + + **Facts are stored as the filing reports them, never as derived quarters.** + Duration facts (revenue, net_income, operating_income, diluted_eps, cfo, + capex, depreciation_amortization) hold the filing's normalized **cumulative + YTD/FY** value over (period_start -> period_end). Balance-sheet facts + (cash_and_st_investments, total_debt, shares_outstanding) are **period-end** + values. ``shares_outstanding`` is a point-in-time count + (``dei:EntityCommonStockSharesOutstanding``, summed across share classes for + a multi-class issuer) — deliberately not the weighted-average diluted share + count, since both consumers (estimated market cap, YoY dilution read) want a + point-in-time value. Discrete quarters (10-Q YTD deltas, Q4 = FY - Q1..Q3), TTM, YoY and + the quarter tape are all derived at read time — so non-calendar fiscal years + resolve correctly and a later amendment never leaves a stale frozen quarter. + """ + + __tablename__ = "fundamental_snapshots" + __table_args__ = ( + UniqueConstraint("accession", name="uq_fundamental_snapshots_accession"), + Index("ix_fundamental_snapshots_cik_period", "cik", "fiscal_year", "fiscal_period"), + Index("ix_fundamental_snapshots_cik_period_end", "cik", "period_end"), + ) + + id: Mapped[int] = mapped_column(primary_key=True) + cik: Mapped[str] = mapped_column(String(10), nullable=False) + accession: Mapped[str] = mapped_column(String(25), nullable=False) + form: Mapped[str] = mapped_column(String(12), nullable=False) # 10-Q, 10-K, 10-K/A ... + filed_date: Mapped[date] = mapped_column(Date, nullable=False) + # Kept although PIT enforcement is deferred (one timestamp now vs painful retrofit). + accepted_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False) + + # Period identity — required to align non-calendar fiscal years and to derive + # discrete quarters from cumulative facts. + period_start: Mapped[date | None] = mapped_column(Date, nullable=True) + period_end: Mapped[date] = mapped_column(Date, nullable=False) + fiscal_year: Mapped[int] = mapped_column(nullable=False) + fiscal_period: Mapped[str] = mapped_column(String(4), nullable=False) # Q1|Q2|Q3|Q4|FY + + # Duration facts — cumulative YTD/FY over (period_start -> period_end). + revenue: Mapped[float | None] = mapped_column(Float, nullable=True) + net_income: Mapped[float | None] = mapped_column(Float, nullable=True) + operating_income: Mapped[float | None] = mapped_column(Float, nullable=True) + diluted_eps: Mapped[float | None] = mapped_column(Float, nullable=True) + cfo: Mapped[float | None] = mapped_column(Float, nullable=True) # cash flow from operations + capex: Mapped[float | None] = mapped_column(Float, nullable=True) + depreciation_amortization: Mapped[float | None] = mapped_column(Float, nullable=True) + + # Balance-sheet facts — period-end values. + cash_and_st_investments: Mapped[float | None] = mapped_column(Float, nullable=True) + total_debt: Mapped[float | None] = mapped_column(Float, nullable=True) + shares_outstanding: Mapped[float | None] = mapped_column(Float, nullable=True) + + import_run_id: Mapped[int | None] = mapped_column( + ForeignKey("data_import_runs.id", ondelete="SET NULL"), nullable=True + ) + created_at: Mapped[datetime] = mapped_column( + DateTime(timezone=True), default=datetime.utcnow, nullable=False + ) diff --git a/app/models/ticker.py b/app/models/ticker.py index 67e9be8..6f15177 100644 --- a/app/models/ticker.py +++ b/app/models/ticker.py @@ -14,6 +14,13 @@ class Ticker(Base): # Company name (e.g. "Biogen Inc."); backfilled from Alpaca, nullable for # symbols Alpaca doesn't know. name: Mapped[str | None] = mapped_column(String(120), nullable=True) + # SEC issuer identity, refreshed by the SEC fundamentals import from + # company_tickers.json / submissions. The only ticker<->issuer join point; + # multi-class tickers (GOOG/GOOGL) share these values. Nullable: not every + # symbol resolves to a CIK (e.g. ADRs, foreign issuers not in SEC data). + cik: Mapped[str | None] = mapped_column(String(10), nullable=True) + sic: Mapped[str | None] = mapped_column(String(4), nullable=True) + sic_description: Mapped[str | None] = mapped_column(String(160), nullable=True) created_at: Mapped[datetime] = mapped_column( DateTime(timezone=True), default=datetime.utcnow, nullable=False ) @@ -28,3 +35,4 @@ class Ticker(Base): trade_setups = relationship("TradeSetup", back_populates="ticker", cascade="all, delete-orphan") watchlist_entries = relationship("WatchlistEntry", back_populates="ticker", cascade="all, delete-orphan") ingestion_progress = relationship("IngestionProgress", back_populates="ticker", cascade="all, delete-orphan", uselist=False) + earnings_events = relationship("EarningsEvent", back_populates="ticker", cascade="all, delete-orphan")