fix: restore stage-based Alpha check classification

This commit is contained in:
yuxuanhui
2026-09-12 22:34:14 +08:00
parent 45eb4c3a17
commit c18960946b
13 changed files with 253 additions and 25 deletions
+11 -9
View File
@@ -7,7 +7,7 @@ from datetime import datetime
from sqlalchemy import or_, select, update
from .models import Alpha, Research, ResearchTag, SelfCorrelation, now
from .platform_checks import split_checks, submission_limits
from .platform_checks import check_result, split_checks, submission_limits
from .research.provenance import source_alpha_ids
METRIC_FIELDS = (
@@ -20,16 +20,17 @@ def failed_checks(checks):
"""Return failed Alpha check names, excluding submission limits and local correlation."""
return [
check.get("name") if isinstance(check.get("name"), str) else "未命名检查"
for check in split_checks(checks)[0] if isinstance(check, dict) and check.get("result") == "FAIL"
for check in split_checks(checks)[0] if isinstance(check, dict) and check_result(check) == "FAIL"
] if isinstance(checks, list) else []
def snapshot_columns(settings, metrics, checks):
def snapshot_columns(settings, metrics, checks, *, checked=False):
"""Derive list fields from a platform snapshot, preserving missing metrics as null.
Submission limits are excluded. Only explicit Alpha FAIL results count.
Empty, malformed and unfinished Alpha checks are
pending; all known checks passing without PROD_CORRELATION is only a pre-check.
Sync snapshots with no failures are PRE_CHECK; a completed explicit /check
with no failures is PASS. WARNING/PENDING do not count as failures, matching
the legacy workflow. Empty, malformed or unknown results remain PENDING.
No submission eligibility or activity eligibility is inferred here.
"""
settings = settings if isinstance(settings, dict) else {}
@@ -40,10 +41,10 @@ def snapshot_columns(settings, metrics, checks):
by_name = {check["name"]: check for check in valid if isinstance(check.get("name"), str)}
if failures:
check_type = "FAIL_1" if failures == 1 else "FAIL_2"
elif not checks or len(valid) != len(checks) or any(check.get("result") != "PASS" for check in valid):
elif not checks or len(valid) != len(checks) or any(check_result(check) not in ("PASS", "WARNING", "PENDING") for check in valid):
check_type = "PENDING"
else:
check_type = "PASS" if "PROD_CORRELATION" in by_name else "PRE_CHECK"
check_type = "PASS" if checked else "PRE_CHECK"
# /check values are freshest; submitted snapshots also expose a scalar in IS.
prod_correlation = number(by_name.get("PROD_CORRELATION", {}).get("value"))
if prod_correlation is None:
@@ -66,12 +67,13 @@ def snapshot_columns(settings, metrics, checks):
}
def check_summary(checks):
def check_summary(checks, *, check_type):
"""Separate cached Alpha findings from submission limits; infer no live eligibility."""
return {
"check_type": snapshot_columns({}, {}, checks)["check_type"],
"check_type": check_type,
"failed_checks": failed_checks(checks),
"submission_limits": submission_limits(checks),
"meaning": "PRE_CHECK 为同步无失败项;PASS 为主动检查完成且无失败项。PENDING/WARNING 不算失败,不代表全部检查项 PASS 或当前可提交",
}
+8 -2
View File
@@ -1,6 +1,12 @@
"""Classify platform evidence without discarding unknown checks or inferring eligibility."""
def check_result(check):
"""Normalize known upstream result casing without rewriting the raw evidence."""
value = check.get("result") if isinstance(check, dict) else None
return value.upper() if isinstance(value, str) else None
def is_submission_limit(check):
"""Recognize only the confirmed account-limit check; unknown names remain Alpha checks."""
return isinstance(check, dict) and check.get("name") == "REGULAR_SUBMISSION"
@@ -16,8 +22,8 @@ def split_checks(checks):
def submission_limits(checks):
"""Summarize the observed limit, never the account's current allowance or reset time."""
_, limits = split_checks(checks)
status = "blocked" if any(c.get("result") == "FAIL" for c in limits) else (
"not_blocked" if limits and all(c.get("result") == "PASS" for c in limits) else "unknown"
status = "blocked" if any(check_result(c) == "FAIL" for c in limits) else (
"not_blocked" if limits and all(check_result(c) == "PASS" for c in limits) else "unknown"
)
return {"status": status, "checks": limits,
"meaning": "仅反映缓存观测时的提交限制,不代表当前额度或正式提交资格"}
+1 -1
View File
@@ -222,7 +222,7 @@ class ResearchAccess:
return {"alpha_id": args.alpha_id, "snapshot": submission_fingerprint(context),
**context, "descriptions": {key: item["description"] for key, item in context["sections"].items()},
"can_check": alpha.status == "UNSUBMITTED" and await correlation_allows_check(self.db, args.alpha_id),
"checks": alpha.checks, "check_summary": check_summary(alpha.checks), "source": "local_cache", "production_submission": False,
"checks": alpha.checks, "check_summary": check_summary(alpha.checks, check_type=alpha.check_type), "source": "local_cache", "production_submission": False,
"job_id": job.id if job else None, "job_status": job.status if job else None,
"checked_at": job.checkpoint.get("checked_at") if job else None}
+2 -2
View File
@@ -207,7 +207,7 @@ def router(runner, ai):
return {
"snapshot": fingerprint(context),
"checks": alpha.checks,
"check_summary": check_summary(alpha.checks),
"check_summary": check_summary(alpha.checks, check_type=alpha.check_type),
"sections": context["sections"],
"descriptions": {key: item["description"] for key, item in context["sections"].items()},
"model": config.description_model,
@@ -355,7 +355,7 @@ async def run_check(runner, job_id, payload):
alpha.checks = sanitize(checks)
alpha.is_metrics = {**alpha.is_metrics, "checks": alpha.checks}
alpha.raw = {**alpha.raw, "is": {**(alpha.raw.get("is") or {}), "checks": alpha.checks}}
for key, value in snapshot_columns(alpha.settings, alpha.is_metrics, alpha.checks).items():
for key, value in snapshot_columns(alpha.settings, alpha.is_metrics, alpha.checks, checked=True).items():
setattr(alpha, key, value)
db.add(JobItem(job_id=job_id, alpha_id=alpha_id))
job.processed = 1
@@ -0,0 +1,91 @@
"""Restore stage-based check classification, preserving raw platform snapshots."""
from datetime import datetime, timezone
import sqlalchemy as sa
from alembic import op
revision = "0019"
down_revision = "0018"
branch_labels = None
depends_on = None
def timestamp(value):
"""Read historical timestamps; missing or invalid evidence cannot prove a /check."""
try:
value = datetime.fromisoformat(value) if isinstance(value, str) else value
return value.replace(tzinfo=timezone.utc) if value.tzinfo is None else value
except (ValueError, TypeError, AttributeError):
return None
def classify(checks, checked, legacy):
"""Frozen migration rules; legacy means the pre-0019 correlation-presence rule."""
checks = checks if isinstance(checks, list) else []
checks = [c for c in checks if not (isinstance(c, dict) and c.get("name") == "REGULAR_SUBMISSION")]
valid = [c for c in checks if isinstance(c, dict)]
results = [c.get("result") for c in valid]
if not legacy:
results = [v.upper() if isinstance(v, str) else None for v in results]
failures = sum(v == "FAIL" for v in results)
if failures:
return "FAIL_1" if failures == 1 else "FAIL_2"
allowed = ("PASS",) if legacy else ("PASS", "PENDING", "WARNING")
if not checks or len(valid) != len(checks) or any(v not in allowed for v in results):
return "PENDING"
passed = any(c.get("name") == "PROD_CORRELATION" for c in valid) if legacy else checked
return "PASS" if passed else "PRE_CHECK"
def reclassify(legacy=False):
"""Recompute in batches; only a check checkpoint newer than the sync proves its stage.
A prior PASS is not evidence because the old rule inferred it from a check name.
A subsequent sync replaces the snapshot and is classified as pre-check again.
"""
alphas = sa.table("alphas", sa.column("id", sa.String()), sa.column("checks", sa.JSON()),
sa.column("synced_at", sa.DateTime(timezone=True)), sa.column("check_type", sa.String()))
jobs = sa.table("sync_jobs", sa.column("kind", sa.String()), sa.column("payload", sa.JSON()),
sa.column("checkpoint", sa.JSON()))
connection = op.get_bind()
last_id = None
while True:
query = sa.select(alphas.c.id, alphas.c.checks, alphas.c.synced_at).order_by(alphas.c.id).limit(500)
if last_id is not None:
query = query.where(alphas.c.id > last_id)
rows = connection.execute(query).mappings().all()
if not rows:
break
checked_at = {}
if not legacy:
observations = connection.execute(sa.select(jobs.c.checkpoint).where(
jobs.c.kind == "submission_check",
jobs.c.payload["alpha_ids"][0].as_string().in_([r["id"] for r in rows]),
)).scalars()
for checkpoint in observations:
if not isinstance(checkpoint, dict) or checkpoint.get("phase") != "checked":
continue
alpha_id = checkpoint.get("alpha_id")
observed = timestamp(checkpoint.get("checked_at"))
if isinstance(alpha_id, str) and observed and (
alpha_id not in checked_at or observed > checked_at[alpha_id]
):
checked_at[alpha_id] = observed
updates = []
for row in rows:
synced = timestamp(row["synced_at"])
observed = checked_at.get(row["id"])
updates.append({"snapshot_id": row["id"], "classification": classify(
row["checks"], bool(synced and observed and observed > synced), legacy)})
connection.execute(alphas.update().where(alphas.c.id == sa.bindparam("snapshot_id"))
.values(check_type=sa.bindparam("classification")), updates)
last_id = rows[-1]["id"]
def upgrade():
reclassify()
def downgrade():
reclassify(legacy=True)
+5 -5
View File
@@ -37,10 +37,10 @@ def checks(failures):
(None, "PENDING"),
([None], "PENDING"),
([{}], "PENDING"),
([{"name": "LOW_SHARPE", "result": "WARNING"}], "PENDING"),
([{"name": "LOW_SHARPE", "result": "WARNING"}], "PRE_CHECK"),
([{"name": "LOW_SHARPE", "result": "PASS"}], "PRE_CHECK"),
([{"name": "PROD_CORRELATION", "result": "PENDING"}], "PENDING"),
(checks(0), "PASS"),
([{"name": "PROD_CORRELATION", "result": "PENDING"}], "PRE_CHECK"),
(checks(0), "PRE_CHECK"),
(checks(1), "FAIL_1"),
(checks(2), "FAIL_2"),
(checks(3), "FAIL_2"),
@@ -58,7 +58,7 @@ async def test_checks_filter_before_pagination_and_share_export_scope(app, logge
for check_type, expected in [
("FAIL_1", ["failed1"]),
("FAIL_2", ["failed2", "failed3"]),
("PASS", ["failed0"]),
("PRE_CHECK", ["failed0"]),
("PENDING", ["unknown"]),
]:
response = await logged_in.get(
@@ -218,7 +218,7 @@ def test_migration_backfills_multiple_batches_and_preserves_research(tmp_path, m
rows = db.execute(sa.select(alphas).order_by(alphas.c.id)).mappings().all()
assert len(rows) == 503
for i, row in enumerate(rows):
expected = snapshot_columns(row["settings"], row["is_metrics"], checks(i % 4))
expected = snapshot_columns(row["settings"], row["is_metrics"], checks(i % 4), checked=True)
assert {key: row[key] for key in expected} == expected
record = db.execute(sa.select(research)).mappings().one()
assert record["tags"] == ["PPAC"] and record["note"] == "keep" and record["version"] == 7
@@ -0,0 +1,64 @@
"""Historical stage recovery requires a persisted /check newer than the latest sync."""
from datetime import datetime, timezone
from pathlib import Path
import sqlalchemy as sa
from alembic import command
from alembic.config import Config
from cryptography.fernet import Fernet
def test_stage_backfill_preserves_evidence_and_uses_checkpoints(tmp_path, monkeypatch):
path = tmp_path / "stages.db"
monkeypatch.setenv("DATABASE_URL", f"sqlite+aiosqlite:///{path}")
monkeypatch.setenv("ADMIN_PASSWORD", "migration-test-only")
monkeypatch.setenv("ENCRYPTION_KEY", Fernet.generate_key().decode())
monkeypatch.setenv("WQ_EMAIL", "")
monkeypatch.setenv("WQ_PASSWORD", "")
root = Path(__file__).resolve().parents[1]
config = Config(str(root / "alembic.ini"))
config.set_main_option("script_location", str(root / "migrations"))
command.upgrade(config, "0018")
engine = sa.create_engine(f"sqlite:///{path}")
alphas = sa.Table("alphas", sa.MetaData(), autoload_with=engine)
jobs = sa.Table("sync_jobs", sa.MetaData(), autoload_with=engine)
synced = datetime(2026, 9, 12, 0, tzinfo=timezone.utc)
pending = [{"name": "PROD_CORRELATION", "result": "PENDING"},
{"name": "REGULAR_SUBMISSION", "result": "FAIL"}]
passed = [{"name": "PROD_CORRELATION", "result": "PASS"}]
patterns = [
(pending, "PENDING", "PRE_CHECK", None),
(pending, "PENDING", "PASS", {"phase": "checked", "checked_at": "2026-09-12T01:00:00+00:00"}),
(pending, "PENDING", "PRE_CHECK", {"phase": "checked", "checked_at": "2026-09-11T23:00:00+00:00"}),
(pending, "PENDING", "PRE_CHECK", {"phase": "check"}),
(passed, "PASS", "PRE_CHECK", None),
([{"name": "LOW_SHARPE", "result": "FAIL"}], "FAIL_1", "FAIL_1", None),
([{}], "PENDING", "PENDING", {"phase": "checked", "checked_at": "2026-09-12T01:00:00Z"}),
]
with engine.begin() as db:
db.execute(alphas.insert(), [
{"id": f"stage{i:04}", "hidden": False, "settings": {}, "os_metrics": {},
"is_metrics": {"checks": patterns[i % 7][0]}, "checks": patterns[i % 7][0],
"check_type": patterns[i % 7][1], "synced_at": synced,
"raw": {"is": {"checks": patterns[i % 7][0]}}}
for i in range(503)
])
db.execute(jobs.insert(), [
{"id": f"job{i}", "kind": "submission_check", "status": "completed",
"payload": {"alpha_ids": [f"stage{i:04}"]},
"checkpoint": {**patterns[i % 7][3], "alpha_id": f"stage{i:04}"},
"processed": 1, "failed": 0, "total": 1, "cancel_requested": False,
"created_at": synced, "updated_at": synced}
for i in range(503) if patterns[i % 7][3]
])
for target, expected_index in [("0019", 2), ("0018", 1), ("0019", 2)]:
(command.upgrade if target == "0019" else command.downgrade)(config, target)
with engine.connect() as db:
rows = db.execute(sa.select(alphas).order_by(alphas.c.id)).mappings().all()
assert len(rows) == 503
for i, row in enumerate(rows):
assert row["check_type"] == patterns[i % 7][expected_index]
assert row["checks"] == row["raw"]["is"]["checks"] == row["is_metrics"]["checks"] == patterns[i % 7][0]
command.check(config)
engine.dispose()
+56
View File
@@ -0,0 +1,56 @@
"""The same checks have different meanings at sync and explicit /check stages."""
import pytest
from app.alphas import snapshot_columns
@pytest.mark.parametrize("checks", [
[{"name": "LOW_SHARPE", "result": "PASS"}],
[{"name": "PROD_CORRELATION", "result": "PASS"}],
[{"name": "PROD_CORRELATION", "result": "PENDING"}],
[{"name": "MATCHES_THEMES", "result": "WARNING"}],
])
def test_stage_not_correlation_presence_decides_pass(checks):
assert snapshot_columns({}, {}, checks)["check_type"] == "PRE_CHECK"
assert snapshot_columns({}, {}, checks, checked=True)["check_type"] == "PASS"
@pytest.mark.parametrize("checked", [False, True])
@pytest.mark.parametrize("checks,expected", [
([], "PENDING"),
([None], "PENDING"),
([{}], "PENDING"),
([{"name": "UNKNOWN", "result": "OTHER"}], "PENDING"),
([{"name": "REGULAR_SUBMISSION", "result": "FAIL"}], "PENDING"),
([{"name": "LOW_SHARPE", "result": "fail"}, {"name": "REGULAR_SUBMISSION", "result": "FAIL"}], "FAIL_1"),
([{"name": "LOW_SHARPE", "result": "FAIL"}, {"name": "LOW_FITNESS", "result": "Fail"}], "FAIL_2"),
])
def test_stage_preserves_failures_and_missing_evidence(checked, checks, expected):
assert snapshot_columns({}, {}, checks, checked=checked)["check_type"] == expected
async def test_sync_check_and_resync_use_distinct_stages(app, logged_in, monkeypatch):
from app.alphas import upsert_alpha
from tests import test_submission
from tests.conftest import alpha
checks = [{"name": "LOW_SHARPE", "result": "PASS"},
{"name": "PROD_CORRELATION", "result": "PENDING"},
{"name": "MATCHES_THEMES", "result": "WARNING"},
{"name": "REGULAR_SUBMISSION", "result": "FAIL"}]
monkeypatch.setattr(test_submission, "CHECKS", checks)
await test_submission.setup(app, alpha(**{"is": {"checks": checks}}))
endpoint = "/api/v1/alphas/alpha1/submission"
assert (await logged_in.get(endpoint)).json()["check_summary"]["check_type"] == "PRE_CHECK"
response = await test_submission.enqueue(logged_in)
assert response.status_code == 202
await app.state.runner.execute(response.json()["id"])
state = (await logged_in.get(endpoint)).json()
assert state["job"]["status"] == "completed"
assert state["check_summary"]["check_type"] == "PASS"
assert state["check_summary"]["submission_limits"]["status"] == "blocked"
assert (await logged_in.get("/api/v1/alphas/alpha1")).json()["check_type"] == "PASS"
async with app.state.sessions.begin() as db:
await upsert_alpha(db, alpha(**{"is": {"checks": checks}}))
assert (await logged_in.get(endpoint)).json()["check_summary"]["check_type"] == "PRE_CHECK"
+2 -2
View File
@@ -11,7 +11,7 @@ from tests.test_submission import FIELDS, Description, setup
@pytest.mark.parametrize("kind", ["REGULAR", "SUPER"])
@pytest.mark.parametrize("result", ["PASS", "FAIL"])
@pytest.mark.parametrize("result", ["PASS", "FAIL", "PENDING", "WARNING"])
@pytest.mark.parametrize("limited", [False, True])
async def test_mcp_check_never_submits(mcp_app, kind, result, limited, monkeypatch):
from tests import test_submission
@@ -38,7 +38,7 @@ async def test_mcp_check_never_submits(mcp_app, kind, result, limited, monkeypat
assert job["status"] == "completed", job
data = await invoke(mcp_app, principal, "get_submission_check", {"alpha_id": "alpha1"})
assert data["checks"] == checks and data["checked_at"]
assert data["check_summary"]["check_type"] == ("PASS" if result == "PASS" else "FAIL_1")
assert data["check_summary"]["check_type"] == ("FAIL_1" if result == "FAIL" else "PASS")
assert data["check_summary"]["submission_limits"]["status"] == ("blocked" if limited else "unknown")
assert data["production_submission"] is False
assert platform.patches == [{s: {"description": args["descriptions"][s]} for s in sections}]
@@ -47,5 +47,6 @@ def test_limit_migration_is_reversible_and_preserves_snapshots(tmp_path, monkeyp
for i, row in enumerate(rows):
assert row["check_type"] == patterns[i % 5][position]
assert row["checks"] == row["is_metrics"]["checks"] == row["raw"]["is"]["checks"] == patterns[i % 5][0]
command.upgrade(config, "head")
command.check(config)
engine.dispose()
+3 -3
View File
@@ -8,13 +8,13 @@ LIMIT = {"name": "REGULAR_SUBMISSION", "result": "FAIL", "value": 4, "limit": 4}
@pytest.mark.parametrize("checks,expected", [
([{"name": "PROD_CORRELATION", "result": "PASS"}], "PASS"),
([{"name": "PROD_CORRELATION", "result": "PASS"}], "PRE_CHECK"),
([{"name": "LOW_SHARPE", "result": "PASS"}], "PRE_CHECK"),
([{"name": "LOW_SHARPE", "result": "FAIL"}], "FAIL_1"),
([], "PENDING"),
([None], "PENDING"),
([{"name": "UNKNOWN_CHECK", "result": "FAIL"}], "FAIL_1"),
([{"name": "PROD_CORRELATION", "result": "PENDING"}], "PENDING"),
([{"name": "PROD_CORRELATION", "result": "PENDING"}], "PRE_CHECK"),
])
def test_limit_does_not_change_alpha_verdict(checks, expected):
original = [*checks, LIMIT]
@@ -28,7 +28,7 @@ def test_all_limit_states_are_separate(result):
from app.alphas import check_summary
checks = [{"name": "PROD_CORRELATION", "result": "PASS"}, {**LIMIT, "result": result}]
summary = check_summary(checks)
summary = check_summary(checks, check_type="PASS")
assert summary["check_type"] == "PASS"
assert summary["submission_limits"]["status"] == ("not_blocked" if result == "PASS" else "unknown")