fix: restore stage-based Alpha check classification

This commit is contained in:
yuxuanhui
2026-09-12 22:34:14 +08:00
parent 45eb4c3a17
commit c18960946b
13 changed files with 253 additions and 25 deletions
@@ -0,0 +1,91 @@
"""Restore stage-based check classification, preserving raw platform snapshots."""
from datetime import datetime, timezone
import sqlalchemy as sa
from alembic import op
revision = "0019"
down_revision = "0018"
branch_labels = None
depends_on = None
def timestamp(value):
"""Read historical timestamps; missing or invalid evidence cannot prove a /check."""
try:
value = datetime.fromisoformat(value) if isinstance(value, str) else value
return value.replace(tzinfo=timezone.utc) if value.tzinfo is None else value
except (ValueError, TypeError, AttributeError):
return None
def classify(checks, checked, legacy):
"""Frozen migration rules; legacy means the pre-0019 correlation-presence rule."""
checks = checks if isinstance(checks, list) else []
checks = [c for c in checks if not (isinstance(c, dict) and c.get("name") == "REGULAR_SUBMISSION")]
valid = [c for c in checks if isinstance(c, dict)]
results = [c.get("result") for c in valid]
if not legacy:
results = [v.upper() if isinstance(v, str) else None for v in results]
failures = sum(v == "FAIL" for v in results)
if failures:
return "FAIL_1" if failures == 1 else "FAIL_2"
allowed = ("PASS",) if legacy else ("PASS", "PENDING", "WARNING")
if not checks or len(valid) != len(checks) or any(v not in allowed for v in results):
return "PENDING"
passed = any(c.get("name") == "PROD_CORRELATION" for c in valid) if legacy else checked
return "PASS" if passed else "PRE_CHECK"
def reclassify(legacy=False):
"""Recompute in batches; only a check checkpoint newer than the sync proves its stage.
A prior PASS is not evidence because the old rule inferred it from a check name.
A subsequent sync replaces the snapshot and is classified as pre-check again.
"""
alphas = sa.table("alphas", sa.column("id", sa.String()), sa.column("checks", sa.JSON()),
sa.column("synced_at", sa.DateTime(timezone=True)), sa.column("check_type", sa.String()))
jobs = sa.table("sync_jobs", sa.column("kind", sa.String()), sa.column("payload", sa.JSON()),
sa.column("checkpoint", sa.JSON()))
connection = op.get_bind()
last_id = None
while True:
query = sa.select(alphas.c.id, alphas.c.checks, alphas.c.synced_at).order_by(alphas.c.id).limit(500)
if last_id is not None:
query = query.where(alphas.c.id > last_id)
rows = connection.execute(query).mappings().all()
if not rows:
break
checked_at = {}
if not legacy:
observations = connection.execute(sa.select(jobs.c.checkpoint).where(
jobs.c.kind == "submission_check",
jobs.c.payload["alpha_ids"][0].as_string().in_([r["id"] for r in rows]),
)).scalars()
for checkpoint in observations:
if not isinstance(checkpoint, dict) or checkpoint.get("phase") != "checked":
continue
alpha_id = checkpoint.get("alpha_id")
observed = timestamp(checkpoint.get("checked_at"))
if isinstance(alpha_id, str) and observed and (
alpha_id not in checked_at or observed > checked_at[alpha_id]
):
checked_at[alpha_id] = observed
updates = []
for row in rows:
synced = timestamp(row["synced_at"])
observed = checked_at.get(row["id"])
updates.append({"snapshot_id": row["id"], "classification": classify(
row["checks"], bool(synced and observed and observed > synced), legacy)})
connection.execute(alphas.update().where(alphas.c.id == sa.bindparam("snapshot_id"))
.values(check_type=sa.bindparam("classification")), updates)
last_id = rows[-1]["id"]
def upgrade():
reclassify()
def downgrade():
reclassify(legacy=True)