feat: implement scoped dataset catalog and template input drafts

This commit is contained in:
yuxuanhui
2026-09-08 09:22:53 +08:00
parent 404a4d8a04
commit 01d169d118
30 changed files with 3024 additions and 37 deletions
+137
View File
@@ -0,0 +1,137 @@
"""Explicit research scope and catalog contracts; unknown platform types remain strings."""
from datetime import datetime, timezone
from typing import Annotated, Literal
from pydantic import AfterValidator, BaseModel, Field, model_validator
from ..schemas import Contract
def utc_timestamp(value: datetime) -> datetime:
"""SQLite drops tzinfo; catalog source times always denote UTC instants."""
return value.replace(tzinfo=timezone.utc) if value.tzinfo is None else value
UTCTimestamp = Annotated[datetime, AfterValidator(utc_timestamp)]
# Supported research scopes, not an assertion about a connected account's permissions.
UNIVERSES = {
"USA": ["TOP3000", "TOP1000", "TOP500", "TOP200"],
"CHN": ["TOP2000"],
"EUR": ["TOP2500", "TOP1200"],
"ASI": ["TOP1000"],
"GLB": ["TOP3000"],
"JPN": ["TOP1600"],
"HKG": ["TOP800"],
}
class Scope(Contract):
instrument_type: Literal["EQUITY"] = "EQUITY"
region: str
universe: str
delay: int = Field(ge=0, le=1)
@model_validator(mode="after")
def valid_scope(self):
if self.universe not in UNIVERSES.get(self.region, []):
raise ValueError("不支持的 Region / Universe 组合")
return self
def key(self):
return f"{self.instrument_type}|{self.region}|{self.universe}|{self.delay}"
class CatalogFilters(Scope):
q: str = Field(default="", max_length=300)
category: str | None = None
subcategory: str | None = None
field_type: str | None = None
coverage_min: float | None = Field(default=None, ge=0, le=1)
sort: Literal[
"id", "name", "category", "field_count", "coverage", "user_count", "alpha_count", "field_type"
] = "name"
direction: Literal["asc", "desc"] = "asc"
limit: int = Field(default=25, ge=1, le=100)
offset: int = Field(default=0, ge=0)
class CatalogJobInput(Contract):
scope: Scope
dataset_id: str | None = Field(default=None, min_length=1, max_length=200)
class NoteInput(Contract):
note: str = Field(max_length=20000)
version: int = Field(ge=1)
class InputPreparation(Contract):
scope: Scope
dataset_id: str = Field(min_length=1, max_length=200)
collection_version: str
selection: Literal["all", "explicit"] = "all"
excluded_ids: list[str] = Field(default_factory=list, max_length=100000)
@model_validator(mode="after")
def valid_selection(self):
if self.selection == "all" and self.excluded_ids:
raise ValueError("全部字段不能同时提供排除项")
return self
class NoteOutput(BaseModel):
note: str
version: int
updated_at: UTCTimestamp
class EntryOutput(BaseModel):
id: str
name: str | None
category: str | None
subcategory: str | None
field_type: str | None
coverage: float | None
user_count: int | None
alpha_count: int | None
field_count: int | None
description: str | None
unit: str | None
synced_at: UTCTimestamp
collection_version: str | None = None
complete_count: int | None = None
research: NoteOutput | None = None
scope: Scope | None = None
dataset_id: str | None = None
class CatalogPage(BaseModel):
items: list[EntryOutput]
total: int
limit: int
offset: int
collection_version: str | None
complete_count: int | None
synced_at: UTCTimestamp | None
categories: dict[str, list[str]] = Field(default_factory=dict)
field_types: list[str] = Field(default_factory=list)
class InputOutput(BaseModel):
id: str
status: Literal["draft"] = "draft"
scope: Scope
dataset_id: str
collection_version: str
selection: str
field_ids: list[str]
field_types: dict[str, str | None]
created_at: UTCTimestamp
class CollectionOutput(BaseModel):
collection_version: str | None
field_ids: list[str]