A01: Separate API readiness from import health diagnostics
- Infrastructure (DB/MinIO) blocks readiness; imports are diagnostic only - Per-source community scheduler health with backoff detection - Stale/failed imports never block /ready — scheduler can recover them - Add 'blocking: false' to all import components - 4 new tests: per-source health, backoff detection, stale/failed non-blocking - 108 Python tests pass
This commit is contained in:
+53
-24
@@ -3,10 +3,10 @@ from __future__ import annotations
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any
|
||||
|
||||
from sqlalchemy import select, text
|
||||
from sqlalchemy import func, select, text
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from .models import CommunityImportRun, ImportStatus, OfficialRecordImport
|
||||
from .models import CommunityImportRun, DataSource, ImportStatus, OfficialRecordImport
|
||||
|
||||
|
||||
def readiness_report(
|
||||
@@ -14,6 +14,12 @@ def readiness_report(
|
||||
import_interval_seconds: int, community_import_interval_seconds: int = 1800,
|
||||
now: datetime | None = None,
|
||||
) -> tuple[bool, dict[str, dict[str, object]]]:
|
||||
"""A01: Separate infrastructure readiness from import health diagnostics.
|
||||
|
||||
Infrastructure (DB, MinIO) blocks readiness. Import health is diagnostic only
|
||||
— stale/failed imports must not prevent the API from serving requests or the
|
||||
scheduler from running to recover them.
|
||||
"""
|
||||
current = now or datetime.now(timezone.utc)
|
||||
components: dict[str, dict[str, object]] = {}
|
||||
ready = True
|
||||
@@ -32,6 +38,7 @@ def readiness_report(
|
||||
components["minio"] = {"status": "unavailable"}
|
||||
ready = False
|
||||
|
||||
# Official import health — diagnostic only, never blocks readiness (A01)
|
||||
try:
|
||||
latest = session.scalar(select(OfficialRecordImport).order_by(
|
||||
OfficialRecordImport.started_at.desc(), OfficialRecordImport.id.desc(),
|
||||
@@ -40,10 +47,13 @@ def readiness_report(
|
||||
components["official_import"] = {
|
||||
"status": "optional",
|
||||
"last_run_status": latest.status.value if latest else None,
|
||||
"blocking": False,
|
||||
}
|
||||
elif latest is None:
|
||||
components["official_import"] = {"status": "not_run"}
|
||||
ready = False
|
||||
components["official_import"] = {
|
||||
"status": "not_run",
|
||||
"blocking": False,
|
||||
}
|
||||
else:
|
||||
started = latest.started_at if latest.started_at.tzinfo else latest.started_at.replace(tzinfo=timezone.utc)
|
||||
stale = started < current - timedelta(seconds=import_interval_seconds * 2)
|
||||
@@ -52,35 +62,54 @@ def readiness_report(
|
||||
"status": "ready" if healthy else ("stale" if stale else latest.status.value),
|
||||
"last_run_status": latest.status.value,
|
||||
"last_started_at": started.isoformat(),
|
||||
"blocking": False,
|
||||
}
|
||||
ready = ready and healthy
|
||||
except Exception:
|
||||
components["official_import"] = {"status": "unknown"}
|
||||
if import_required:
|
||||
ready = False
|
||||
components["official_import"] = {
|
||||
"status": "unknown",
|
||||
"blocking": False,
|
||||
}
|
||||
|
||||
# Check community scheduler: look for recent import runs
|
||||
# Community scheduler health — diagnostic only, never blocks readiness (A01)
|
||||
# Track per-source health with rotation, backoff, last success, and stalled attempts
|
||||
try:
|
||||
latest_community = session.scalar(
|
||||
select(CommunityImportRun)
|
||||
.order_by(CommunityImportRun.started_at.desc())
|
||||
.limit(1)
|
||||
)
|
||||
if latest_community is None:
|
||||
components["community_scheduler"] = {"status": "not_started"}
|
||||
else:
|
||||
started = latest_community.started_at
|
||||
enabled_sources = list(session.scalars(
|
||||
select(DataSource).where(DataSource.enabled.is_(True)).order_by(DataSource.key)
|
||||
))
|
||||
source_health: dict[str, dict[str, object]] = {}
|
||||
for source in enabled_sources:
|
||||
latest_run = session.scalar(
|
||||
select(CommunityImportRun)
|
||||
.where(CommunityImportRun.source_system == source.key)
|
||||
.order_by(CommunityImportRun.started_at.desc())
|
||||
.limit(1)
|
||||
)
|
||||
if latest_run is None:
|
||||
source_health[source.key] = {"status": "not_started", "blocking": False}
|
||||
continue
|
||||
started = latest_run.started_at
|
||||
if started.tzinfo is None:
|
||||
started = started.replace(tzinfo=timezone.utc)
|
||||
stale = started < current - timedelta(seconds=community_import_interval_seconds * 2)
|
||||
healthy = latest_community.status == "success" and not stale
|
||||
components["community_scheduler"] = {
|
||||
"status": "ready" if healthy else ("stale" if stale else latest_community.status),
|
||||
healthy = latest_run.status == "success" and not stale
|
||||
# Count recent failures for backoff detection
|
||||
recent_failures = session.scalar(
|
||||
select(func.count()).select_from(CommunityImportRun)
|
||||
.where(
|
||||
CommunityImportRun.source_system == source.key,
|
||||
CommunityImportRun.status == "failed",
|
||||
CommunityImportRun.started_at >= current - timedelta(hours=24),
|
||||
)
|
||||
) or 0
|
||||
source_health[source.key] = {
|
||||
"status": "ready" if healthy else ("stale" if stale else latest_run.status),
|
||||
"last_started_at": started.isoformat(),
|
||||
"recent_failures_24h": recent_failures,
|
||||
"backoff_recommended": recent_failures >= 5,
|
||||
"blocking": False,
|
||||
}
|
||||
ready = ready and healthy
|
||||
components["community_scheduler"] = {"status": "ready", "sources": source_health}
|
||||
except Exception:
|
||||
components["community_scheduler"] = {"status": "unknown"}
|
||||
ready = False
|
||||
components["community_scheduler"] = {"status": "unknown", "sources": {}}
|
||||
|
||||
return ready, components
|
||||
|
||||
Reference in New Issue
Block a user