feat(jobs): persist each job's last run so it survives a restart
Job outcomes lived only in scheduler._job_runtime, an in-memory dict. Every
deploy wiped it, so Admin -> Jobs could report "Active" with no indication a job
had ever run or how it ended -- which is the main thing that page is for.
New job_run_state table (migration 031): one row per job, upserted on job_name.
Deliberately not history -- system_events already grows unbounded with no
retention job, and a second append-only operational table would repeat that
debt. Adding history later is purely additive.
Written from two hooks, NOT from _runtime_finish. That looked cheapest (one
function, ~40 call sites) but unit tests invoke job coroutines directly, so it
would fire detached DB writes at the real session factory throughout the suite,
and there is no testing flag to guard on.
- An APScheduler EVENT_JOB_EXECUTED/ERROR listener covers everything the
scheduler fires, including manual triggers. Its detached task is held in a
module-level set (a bare create_task result can be collected mid-flight) and
drained in the app lifespan before engine.dispose().
- _run_pipeline persists directly, and must: pipeline steps are plain
coroutine calls that emit no scheduler events, so the listener cannot see
them. The step persist sits AFTER the except that swallows step errors --
inside it, exactly the failed runs worth seeing would be skipped. The
orchestrator persists in the finally, and the disabled early-return persists
too, or "skipped" is silently dropped.
_persist_job_run never raises: a persistence failure must not break an otherwise
successful pipeline.
The API reports this as last_run_* and leaves runtime_* meaning strictly live
in-memory state. Reusing runtime_status would have been a regression, not a
no-op: JobControls drives the status chip from it (a job that errored eight days
ago would read "Last run error" forever instead of "Active") and picks the
rate-limit banner from it (a week-old rate limit would pin the banner
permanently). Tests pin the split.
The table starts empty; each job fills its row the next time it finishes. No
backfill from system_events, which records only warning/error outcomes under a
different status vocabulary and would invent successes that never happened.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
+60
-1
@@ -33,6 +33,7 @@ from app.models.ticker import Ticker
|
||||
from app.exceptions import ProviderError
|
||||
from app.providers.alpaca import AlpacaOHLCVProvider
|
||||
from app.providers.protocol import SentimentData
|
||||
from app.services import job_run_store
|
||||
from app.services import (
|
||||
ingestion_service,
|
||||
pipeline_run,
|
||||
@@ -87,6 +88,19 @@ scheduler = AsyncIOScheduler(
|
||||
)
|
||||
|
||||
|
||||
def _on_job_finished(event: object) -> None:
|
||||
"""Persist the run, then re-pause the job if it only runs on demand.
|
||||
|
||||
Covers every job APScheduler fires itself, including manual triggers.
|
||||
Pipeline *steps* are invoked as plain coroutines and emit no events, so
|
||||
``_run_pipeline`` persists those directly.
|
||||
"""
|
||||
job_id = getattr(event, "job_id", None)
|
||||
if job_id:
|
||||
_schedule_persist(job_id)
|
||||
_repause_after_manual_run(event)
|
||||
|
||||
|
||||
def _repause_after_manual_run(event: object) -> None:
|
||||
"""Re-pause a job that only ever runs on demand, once its run finishes.
|
||||
|
||||
@@ -112,7 +126,7 @@ def _repause_after_manual_run(event: object) -> None:
|
||||
logger.debug("Could not re-pause %s after its run", job_id, exc_info=True)
|
||||
|
||||
|
||||
scheduler.add_listener(_repause_after_manual_run, EVENT_JOB_EXECUTED | EVENT_JOB_ERROR)
|
||||
scheduler.add_listener(_on_job_finished, EVENT_JOB_EXECUTED | EVENT_JOB_ERROR)
|
||||
|
||||
# Track last successful ticker per job for rate-limit resume
|
||||
_last_successful: dict[str, str | None] = {
|
||||
@@ -308,6 +322,44 @@ def _runtime_finish(
|
||||
pass
|
||||
|
||||
|
||||
async def _persist_job_run(job_name: str) -> None:
|
||||
"""Write a job's finished runtime row to the durable last-run table.
|
||||
|
||||
Never raises: a persistence failure must not break the pipeline that was
|
||||
otherwise successful. The in-memory row stays authoritative for live state.
|
||||
"""
|
||||
runtime = _job_runtime.get(job_name)
|
||||
if not runtime or runtime.get("running") or not runtime.get("finished_at"):
|
||||
return
|
||||
try:
|
||||
async with async_session_factory() as db:
|
||||
await job_run_store.record_finish(db, job_name, runtime)
|
||||
await db.commit()
|
||||
except Exception:
|
||||
logger.exception("Could not persist last-run state for %s", job_name)
|
||||
|
||||
|
||||
# Detached persists are kept referenced: a bare create_task result can be
|
||||
# garbage-collected mid-flight, and the shutdown drain needs something to await.
|
||||
_persist_tasks: set[asyncio.Task] = set()
|
||||
|
||||
|
||||
def _schedule_persist(job_name: str) -> None:
|
||||
try:
|
||||
task = asyncio.get_running_loop().create_task(_persist_job_run(job_name))
|
||||
except RuntimeError: # no loop (sync context / tests) — nothing to persist
|
||||
return
|
||||
_persist_tasks.add(task)
|
||||
task.add_done_callback(_persist_tasks.discard)
|
||||
|
||||
|
||||
async def flush_job_run_persists(timeout: float = 5.0) -> None:
|
||||
"""Await in-flight last-run writes. Called from the app's shutdown path."""
|
||||
if not _persist_tasks:
|
||||
return
|
||||
await asyncio.wait(set(_persist_tasks), timeout=timeout)
|
||||
|
||||
|
||||
def get_job_runtime_snapshot(job_name: str | None = None) -> dict[str, dict[str, object]] | dict[str, object]:
|
||||
if job_name is not None:
|
||||
return dict(_job_runtime.get(job_name, {}))
|
||||
@@ -1346,6 +1398,7 @@ async def _run_pipeline(job_name: str, steps: list[tuple[str, str]]) -> None:
|
||||
if not await _is_job_enabled(db, job_name):
|
||||
_log_event(logging.INFO, "job_skipped", job=job_name, reason="disabled")
|
||||
_runtime_finish(job_name, "skipped", processed=0, total=0, message="Disabled")
|
||||
await _persist_job_run(job_name)
|
||||
return
|
||||
|
||||
total = len(steps)
|
||||
@@ -1361,6 +1414,11 @@ async def _run_pipeline(job_name: str, steps: list[tuple[str, str]]) -> None:
|
||||
await funcs[func_name]()
|
||||
except Exception:
|
||||
logger.exception("%s step %s failed", job_name, step_name)
|
||||
# Outside the except on purpose: the step's own _runtime_finish has
|
||||
# already recorded its outcome, so persisting here captures failures
|
||||
# too. Steps are plain coroutine calls and fire no scheduler events,
|
||||
# so the listener cannot see them -- this is their only write path.
|
||||
await _persist_job_run(step_name)
|
||||
done += 1
|
||||
_runtime_finish(job_name, "completed", processed=done, total=total, message="Pipeline complete")
|
||||
_log_event(logging.INFO, "job_complete", job=job_name)
|
||||
@@ -1369,6 +1427,7 @@ async def _run_pipeline(job_name: str, steps: list[tuple[str, str]]) -> None:
|
||||
_log_event(logging.ERROR, "job_error", job=job_name, error_type=type(exc).__name__, message=str(exc))
|
||||
finally:
|
||||
pipeline_run.release(token)
|
||||
await _persist_job_run(job_name)
|
||||
|
||||
|
||||
async def run_daily_pipeline() -> None:
|
||||
|
||||
Reference in New Issue
Block a user