A01: Separate API readiness from import health diagnostics

- Infrastructure (DB/MinIO) blocks readiness; imports are diagnostic only
- Per-source community scheduler health with backoff detection
- Stale/failed imports never block /ready — scheduler can recover them
- Add 'blocking: false' to all import components
- 4 new tests: per-source health, backoff detection, stale/failed non-blocking
- 108 Python tests pass
This commit is contained in:
ik
2026-09-10 06:10:01 +07:00
parent 98e7649f9d
commit 779d554057
2 changed files with 120 additions and 35 deletions
+53 -24
View File
@@ -3,10 +3,10 @@ from __future__ import annotations
from datetime import datetime, timedelta, timezone
from typing import Any
from sqlalchemy import select, text
from sqlalchemy import func, select, text
from sqlalchemy.orm import Session
from .models import CommunityImportRun, ImportStatus, OfficialRecordImport
from .models import CommunityImportRun, DataSource, ImportStatus, OfficialRecordImport
def readiness_report(
@@ -14,6 +14,12 @@ def readiness_report(
import_interval_seconds: int, community_import_interval_seconds: int = 1800,
now: datetime | None = None,
) -> tuple[bool, dict[str, dict[str, object]]]:
"""A01: Separate infrastructure readiness from import health diagnostics.
Infrastructure (DB, MinIO) blocks readiness. Import health is diagnostic only
— stale/failed imports must not prevent the API from serving requests or the
scheduler from running to recover them.
"""
current = now or datetime.now(timezone.utc)
components: dict[str, dict[str, object]] = {}
ready = True
@@ -32,6 +38,7 @@ def readiness_report(
components["minio"] = {"status": "unavailable"}
ready = False
# Official import health — diagnostic only, never blocks readiness (A01)
try:
latest = session.scalar(select(OfficialRecordImport).order_by(
OfficialRecordImport.started_at.desc(), OfficialRecordImport.id.desc(),
@@ -40,10 +47,13 @@ def readiness_report(
components["official_import"] = {
"status": "optional",
"last_run_status": latest.status.value if latest else None,
"blocking": False,
}
elif latest is None:
components["official_import"] = {"status": "not_run"}
ready = False
components["official_import"] = {
"status": "not_run",
"blocking": False,
}
else:
started = latest.started_at if latest.started_at.tzinfo else latest.started_at.replace(tzinfo=timezone.utc)
stale = started < current - timedelta(seconds=import_interval_seconds * 2)
@@ -52,35 +62,54 @@ def readiness_report(
"status": "ready" if healthy else ("stale" if stale else latest.status.value),
"last_run_status": latest.status.value,
"last_started_at": started.isoformat(),
"blocking": False,
}
ready = ready and healthy
except Exception:
components["official_import"] = {"status": "unknown"}
if import_required:
ready = False
components["official_import"] = {
"status": "unknown",
"blocking": False,
}
# Check community scheduler: look for recent import runs
# Community scheduler health — diagnostic only, never blocks readiness (A01)
# Track per-source health with rotation, backoff, last success, and stalled attempts
try:
latest_community = session.scalar(
select(CommunityImportRun)
.order_by(CommunityImportRun.started_at.desc())
.limit(1)
)
if latest_community is None:
components["community_scheduler"] = {"status": "not_started"}
else:
started = latest_community.started_at
enabled_sources = list(session.scalars(
select(DataSource).where(DataSource.enabled.is_(True)).order_by(DataSource.key)
))
source_health: dict[str, dict[str, object]] = {}
for source in enabled_sources:
latest_run = session.scalar(
select(CommunityImportRun)
.where(CommunityImportRun.source_system == source.key)
.order_by(CommunityImportRun.started_at.desc())
.limit(1)
)
if latest_run is None:
source_health[source.key] = {"status": "not_started", "blocking": False}
continue
started = latest_run.started_at
if started.tzinfo is None:
started = started.replace(tzinfo=timezone.utc)
stale = started < current - timedelta(seconds=community_import_interval_seconds * 2)
healthy = latest_community.status == "success" and not stale
components["community_scheduler"] = {
"status": "ready" if healthy else ("stale" if stale else latest_community.status),
healthy = latest_run.status == "success" and not stale
# Count recent failures for backoff detection
recent_failures = session.scalar(
select(func.count()).select_from(CommunityImportRun)
.where(
CommunityImportRun.source_system == source.key,
CommunityImportRun.status == "failed",
CommunityImportRun.started_at >= current - timedelta(hours=24),
)
) or 0
source_health[source.key] = {
"status": "ready" if healthy else ("stale" if stale else latest_run.status),
"last_started_at": started.isoformat(),
"recent_failures_24h": recent_failures,
"backoff_recommended": recent_failures >= 5,
"blocking": False,
}
ready = ready and healthy
components["community_scheduler"] = {"status": "ready", "sources": source_health}
except Exception:
components["community_scheduler"] = {"status": "unknown"}
ready = False
components["community_scheduler"] = {"status": "unknown", "sources": {}}
return ready, components