From e1e10ce29307378802a77df3721d021a39308616 Mon Sep 17 00:00:00 2001 From: ao gong <41768719+ageorge156@users.noreply.github.com> Date: Tue, 1 Sep 2026 12:04:48 +0800 Subject: [PATCH] feat: close backtest evidence contract --- MODULE_SPEC.yaml | 7 +- README.md | 22 +- src/quant_engine/artifact.py | 670 +++++++++++++++++- src/quant_engine/governed_pipeline.py | 638 ++++++++++++++++- .../fixtures/backtest-evidence-v1.golden.json | 17 + tests/governance/test_module_spec.py | 19 +- tests/test_backtest_contracts.py | 533 ++++++++++++++ 7 files changed, 1891 insertions(+), 15 deletions(-) create mode 100644 tests/fixtures/backtest-evidence-v1.golden.json create mode 100644 tests/test_backtest_contracts.py diff --git a/MODULE_SPEC.yaml b/MODULE_SPEC.yaml index 894a099..d6af360 100644 --- a/MODULE_SPEC.yaml +++ b/MODULE_SPEC.yaml @@ -1,7 +1,7 @@ { "schema_version": 1, "module_id": "quant_engine", - "authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 2, "effective_from": "2026-09-01T00:00:00+08:00"}, + "authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 3, "effective_from": "2026-09-01T00:00:00+08:00"}, "repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"}, "bounded_context": { "domain": "quantitative-research-engine", @@ -18,6 +18,7 @@ {"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"}, {"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"}, {"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"}, + {"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact evidence without persistence or decision authority.", "status": "operational"}, {"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"} ], "data": {"owns": [ @@ -27,7 +28,9 @@ "contracts": { "provides": [ {"contract_id": "researchhub.factor-definition", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"}, - {"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"} + {"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"}, + {"contract_id": "researchhub.backtest-run-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/governed_pipeline.py"}, + {"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"} ], "consumes": [ {"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"}, diff --git a/README.md b/README.md index b35c73e..060261c 100644 --- a/README.md +++ b/README.md @@ -26,8 +26,8 @@ - `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark) - `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表 - `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排) -- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;全链路带确定性 ID,风险拒绝时禁止生成订单意图 -- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表与 manifest +- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef` +- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest` - `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计 - `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效 - `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test) @@ -241,6 +241,24 @@ identity 均保持不变。迁移只能通过 content-addressed `LegacyFactorBin `bind_legacy_factor()` 或 `project_legacy_factor()`;后者是有损投影,不表示旧 digest 与新定义 digest 等价,也不会把旧 run 静默升级为新合同。 +## 回测引用与证据合同 v1 + +`quant_engine.governed_pipeline.BacktestRunRef` 是合格回测运行身份的唯一权威。`run_id` 只由 +已验收的 Dataset Snapshot / Data Foundation / `FactorSetRef` 身份、universe、日历与公司行动 +祖先、策略、执行/成本模型、严格整数 seed、完整代码提交、环境锁、配置、时间和重放血缘决定; +它不包含任何输出摘要。重放必须绑定直接父运行、连续 attempt 和不变的 +`replay_spec_digest`,输入漂移或血缘环会失败关闭。 + +`quant_engine.artifact.BacktestEvidenceManifest` 只摘要 `ResearchRunArtifact` 已有的九张事实表。 +固定 `offline_research_v1` 映射为 `run`、`signal`、`fill`、`position_nav`、`performance`、 +`attribution`、`risk_snapshot` 与 `replay`;每张表都保留列模式摘要、行数和内容摘要,空 risk +表也必须有稳定 schema。`manifest_id` 由完整 RunRef 与输出证据决定,因此结果变化不会反向改变 +`run_id`。当前 artifact 不拥有订单或拒绝事实,所以此画像明确不声明 `order` / `rejection`。 + +旧 `BacktestRun` 只能通过 `build_legacy_backtest_evidence_manifest()` 显式映射为 +`LEGACY_EXPLORATORY`;不能隐式提升为 `CONTRACT_QUALIFIED`。所有资格均只描述离线证据闭合, +不表示投资有效、组合获批、Paper、生产或实盘就绪。 + ## 治理垂直切片 `governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。 diff --git a/src/quant_engine/artifact.py b/src/quant_engine/artifact.py index 2d2b8f8..2f2d8a4 100644 --- a/src/quant_engine/artifact.py +++ b/src/quant_engine/artifact.py @@ -11,18 +11,27 @@ from __future__ import annotations import hashlib import json import math -from collections.abc import Mapping +import re +from collections.abc import Mapping, Sequence from dataclasses import dataclass -from datetime import date, datetime -from typing import Any +from datetime import UTC, date, datetime +from enum import StrEnum +from typing import Any, Never, Self, cast import numpy as np import pandas as pd +from quant_engine.governed_pipeline import ( + BacktestContractError, + BacktestContractErrorCode, + BacktestRun, + BacktestRunRef, +) from quant_engine.research_pipeline import FactorBacktestResult from quant_engine.risk import CovarianceSnapshot, labeled_component_risk RESEARCH_ARTIFACT_SCHEMA_VERSION = "1.1.0" +BACKTEST_EVIDENCE_SCHEMA_VERSION = "1.0.0" RISK_COLUMNS = [ "run_id", @@ -41,7 +50,14 @@ RISK_COLUMNS = [ __all__ = [ "RESEARCH_ARTIFACT_SCHEMA_VERSION", + "BACKTEST_EVIDENCE_SCHEMA_VERSION", "ResearchRunArtifact", + "EvidenceQualification", + "BacktestEvidenceTable", + "BacktestEvidenceEntry", + "BacktestEvidenceManifest", + "build_backtest_evidence_manifest", + "build_legacy_backtest_evidence_manifest", "build_research_run_artifact", ] @@ -162,6 +178,184 @@ class ResearchRunArtifact: } +class EvidenceQualification(StrEnum): + """Evidence closure state; none of the states grants decision or trading authority.""" + + LEGACY_EXPLORATORY = "legacy_exploratory" + EXPLORATORY = "exploratory" + CONTRACT_QUALIFIED = "contract_qualified" + + +@dataclass(frozen=True, slots=True) +class BacktestEvidenceTable: + """Digest and row count for one existing artifact table.""" + + logical_name: str + row_count: int + schema_digest: str + content_digest: str + + def to_dict(self) -> dict[str, object]: + return { + "logical_name": self.logical_name, + "row_count": self.row_count, + "schema_digest": self.schema_digest, + "content_digest": self.content_digest, + } + + +@dataclass(frozen=True, slots=True) +class BacktestEvidenceEntry: + """Closed logical evidence category over one or more authoritative tables.""" + + category: str + tables: tuple[BacktestEvidenceTable, ...] + evidence_digest: str + reconciliation: str + + def to_dict(self) -> dict[str, object]: + return { + "category": self.category, + "tables": [table.to_dict() for table in self.tables], + "evidence_digest": self.evidence_digest, + "reconciliation": self.reconciliation, + } + + +@dataclass(frozen=True, slots=True, init=False) +class BacktestEvidenceManifest: + """Content-addressed closure over a run reference and existing artifact evidence.""" + + contract_name: str + schema_version: str + manifest_id: str + run_id: str + profile: str + artifact_available_at: str + qualification: EvidenceQualification + evidence_digest: str + evidence: tuple[BacktestEvidenceEntry, ...] + backtest_run_ref: BacktestRunRef | None + legacy_backtest_run: BacktestRun | None + + def _run_reference_dict(self) -> dict[str, object]: + if self.backtest_run_ref is not None: + return { + "kind": "backtest_run_ref", + "value": self.backtest_run_ref.to_dict(), + } + if self.legacy_backtest_run is None: + raise RuntimeError("manifest has no run reference") + return { + "kind": "legacy_backtest_run", + "value": _legacy_run_dict(self.legacy_backtest_run), + } + + def to_dict(self) -> dict[str, Any]: + return { + "contract_name": self.contract_name, + "schema_version": self.schema_version, + "manifest_id": self.manifest_id, + "run_id": self.run_id, + "profile": self.profile, + "artifact_available_at": self.artifact_available_at, + "qualification": self.qualification.value, + "run_reference": self._run_reference_dict(), + "evidence_digest": self.evidence_digest, + "evidence": [item.to_dict() for item in self.evidence], + } + + def to_json(self) -> str: + return _canonical_mapping_json(self.to_dict()) + + @classmethod + def from_dict( + cls, + value: Any, + *, + backtest_run_ref: BacktestRunRef, + artifact: ResearchRunArtifact, + ) -> Self: + if type(value) is not dict: + _manifest_fail(BacktestContractErrorCode.TYPE_ERROR, "$", "must be an object") + expected_fields = { + "contract_name", + "schema_version", + "manifest_id", + "run_id", + "profile", + "artifact_available_at", + "qualification", + "run_reference", + "evidence_digest", + "evidence", + } + missing = sorted(expected_fields - set(value)) + if missing: + _manifest_fail( + BacktestContractErrorCode.MISSING_FIELD, + f"$.{missing[0]}", + "field is required", + ) + unknown = sorted(set(value) - expected_fields) + if unknown: + _manifest_fail( + BacktestContractErrorCode.UNKNOWN_FIELD, + f"$.{unknown[0]}", + "field is not permitted", + ) + raw_evidence = value["evidence"] + if type(raw_evidence) is not list: + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.evidence", + "must be an array", + ) + seen: set[str] = set() + for index, raw_entry in enumerate(raw_evidence): + if type(raw_entry) is not dict: + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + f"$.evidence[{index}]", + "must be an object", + ) + category = raw_entry.get("category") + if type(category) is not str: + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + f"$.evidence[{index}].category", + "must be a string", + ) + if category in seen: + _manifest_fail( + BacktestContractErrorCode.INVALID_VALUE, + f"$.evidence[{index}].category", + "duplicate evidence category", + ) + seen.add(category) + try: + qualification = EvidenceQualification(value["qualification"]) + except (TypeError, ValueError) as error: + raise BacktestContractError( + BacktestContractErrorCode.INVALID_VALUE, + "$.qualification", + "unsupported evidence qualification", + ) from error + rebuilt = build_backtest_evidence_manifest( + backtest_run_ref, + artifact, + artifact_available_at=value["artifact_available_at"], + qualification=qualification, + ) + if value != rebuilt.to_dict(): + _manifest_fail( + BacktestContractErrorCode.IDENTITY_MISMATCH, + "$", + "serialized manifest does not match closed artifact evidence", + ) + return cast(Self, rebuilt) + + def _required_text(value: str, name: str, *, max_length: int | None = None) -> str: normalized = value.strip() if not normalized: @@ -222,6 +416,474 @@ def _frame_records(frame: pd.DataFrame) -> list[dict[str, object]]: ] +_MANIFEST_DIGEST = re.compile(r"^sha256:[0-9a-f]{64}$") +_OFFLINE_RESEARCH_V1: tuple[tuple[str, tuple[str, ...]], ...] = ( + ("run", ("run",)), + ("signal", ("signals",)), + ("fill", ("trades",)), + ("position_nav", ("positions", "nav")), + ("performance", ("performance",)), + ("attribution", ("attribution", "attribution_daily")), + ("risk_snapshot", ("risk",)), + ("replay", ()), +) +_ARTIFACT_TABLE_NAMES = tuple( + table_name + for category, table_names in _OFFLINE_RESEARCH_V1 + if category != "replay" + for table_name in table_names +) +_REQUIRED_TABLE_COLUMNS: Mapping[str, frozenset[str]] = { + "run": frozenset( + { + "run_id", + "data_snapshot_id", + "strategy_id", + "strategy_version", + "code_revision", + "config_hash", + "finished_at", + } + ), + "signals": frozenset({"run_id", "signal_date", "execution_date", "asset_id"}), + "trades": frozenset({"run_id", "trade_id", "signal_id", "trade_date"}), + "positions": frozenset({"run_id", "trade_date", "asset_id", "weight"}), + "nav": frozenset({"run_id", "trade_date", "nav", "pnl_pct"}), + "performance": frozenset({"run_id", "n_trades", "n_days"}), + "attribution": frozenset({"run_id", "trade_date", "asset_id", "asset_total"}), + "attribution_daily": frozenset( + {"run_id", "trade_date", "transaction_cost", "residual", "total_return"} + ), + "risk": frozenset(RISK_COLUMNS), +} + + +def _manifest_fail( + code: BacktestContractErrorCode, + path: str, + detail: str, +) -> Never: + raise BacktestContractError(code, path, detail) + + +def _manifest_digest(value: object) -> str: + encoded = _canonical_mapping_json({"value": value}).encode("utf-8") + return f"sha256:{hashlib.sha256(encoded).hexdigest()}" + + +def _manifest_required_digest(value: object, path: str) -> str: + if type(value) is not str: + _manifest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be a string") + if _MANIFEST_DIGEST.fullmatch(value) is None: + _manifest_fail( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be a lowercase sha256 digest", + ) + return value + + +def _manifest_instant(value: object, path: str) -> tuple[str, datetime]: + try: + timestamp = pd.Timestamp(value) + except (TypeError, ValueError) as error: + raise BacktestContractError( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be a valid timestamp", + ) from error + if timestamp.tzinfo is None: + _manifest_fail( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must include a timezone", + ) + normalized = timestamp.tz_convert("UTC").to_pydatetime() + return normalized.isoformat().replace("+00:00", "Z"), normalized + + +def _legacy_run_dict(run: BacktestRun) -> dict[str, object]: + created_at, _ = _manifest_instant(run.created_at, "$.legacy_backtest_run.created_at") + return { + "run_id": str(run.run_id), + "dataset_snapshot_id": str(run.dataset_snapshot_id), + "factor_version_id": str(run.factor_version_id), + "strategy_version_id": str(run.strategy_version_id), + "code_revision": str(run.code_revision), + "config_hash": str(run.config_hash), + "created_at": created_at, + } + + +def _artifact_frames(artifact: ResearchRunArtifact) -> dict[str, pd.DataFrame]: + if not isinstance(artifact, ResearchRunArtifact): + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.artifact", + "ResearchRunArtifact is required", + ) + frames: dict[str, pd.DataFrame] = {} + for table_name in _ARTIFACT_TABLE_NAMES: + try: + frame = getattr(artifact, table_name) + except (AttributeError, IndexError, KeyError, TypeError, ValueError) as error: + raise BacktestContractError( + BacktestContractErrorCode.TYPE_ERROR, + f"$.artifact.tables.{table_name}", + "artifact table is unavailable", + ) from error + if not isinstance(frame, pd.DataFrame): + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + f"$.artifact.tables.{table_name}", + "artifact table must be a DataFrame", + ) + if len(frame.columns) != len({str(column) for column in frame.columns}): + _manifest_fail( + BacktestContractErrorCode.INVALID_VALUE, + f"$.artifact.tables.{table_name}.columns", + "column names must be unique", + ) + missing = sorted(_REQUIRED_TABLE_COLUMNS[table_name] - set(frame.columns)) + if missing: + _manifest_fail( + BacktestContractErrorCode.MISSING_FIELD, + f"$.artifact.tables.{table_name}.columns.{missing[0]}", + "required column is missing", + ) + frames[table_name] = frame + return frames + + +def _validate_table_run_ids(frames: Mapping[str, pd.DataFrame], run_id: str) -> None: + for table_name, frame in frames.items(): + if frame.empty: + continue + values = frame["run_id"].tolist() + if any(type(value) is not str or value != run_id for value in values): + _manifest_fail( + BacktestContractErrorCode.IDENTITY_MISMATCH, + f"$.artifact.tables.{table_name}.run_id", + "table rows do not bind the manifest run", + ) + + +def _table_evidence( + frames: Mapping[str, pd.DataFrame], + expected_table_digests: Mapping[str, str] | None, +) -> dict[str, BacktestEvidenceTable]: + if expected_table_digests is not None and not isinstance(expected_table_digests, Mapping): + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.expected_table_digests", + "must be an object", + ) + if expected_table_digests is not None: + for key in expected_table_digests: + if type(key) is not str: + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.expected_table_digests", + "table names must be strings", + ) + summaries: dict[str, BacktestEvidenceTable] = {} + for table_name in _ARTIFACT_TABLE_NAMES: + frame = frames[table_name] + columns = [str(column) for column in frame.columns] + schema = { + "columns": columns, + "dtypes": [str(dtype) for dtype in frame.dtypes.tolist()], + } + content = {"columns": columns, "records": _frame_records(frame)} + summary = BacktestEvidenceTable( + logical_name=table_name, + row_count=len(frame), + schema_digest=_manifest_digest(schema), + content_digest=_manifest_digest(content), + ) + if expected_table_digests is not None and table_name in expected_table_digests: + expected = _manifest_required_digest( + expected_table_digests[table_name], + f"$.expected_table_digests.{table_name}", + ) + if summary.content_digest != expected: + _manifest_fail( + BacktestContractErrorCode.EVIDENCE_MISMATCH, + f"$.artifact.tables.{table_name}.content_digest", + "table content differs from the supplied evidence digest", + ) + summaries[table_name] = summary + if expected_table_digests is not None: + unknown = sorted(set(expected_table_digests) - set(summaries)) + if unknown: + _manifest_fail( + BacktestContractErrorCode.UNKNOWN_FIELD, + f"$.expected_table_digests.{unknown[0]}", + "unknown artifact table", + ) + return summaries + + +def _evidence_entries( + summaries: Mapping[str, BacktestEvidenceTable], + run_reference: Mapping[str, object], + *, + legacy: bool, +) -> tuple[BacktestEvidenceEntry, ...]: + all_tables = [summaries[name].to_dict() for name in _ARTIFACT_TABLE_NAMES] + entries: list[BacktestEvidenceEntry] = [] + for category, table_names in _OFFLINE_RESEARCH_V1: + tables = tuple(summaries[name] for name in table_names) + digest_payload: object = ( + {"run_reference": dict(run_reference), "tables": all_tables} + if category == "replay" + else {"category": category, "tables": [table.to_dict() for table in tables]} + ) + entries.append( + BacktestEvidenceEntry( + category=category, + tables=tables, + evidence_digest=_manifest_digest(digest_payload), + reconciliation=( + "legacy_incomplete" + if legacy and category == "replay" + else "legacy_source_mapped" + if legacy + else "closed" + ), + ) + ) + return tuple(entries) + + +def _run_row(frames: Mapping[str, pd.DataFrame]) -> pd.Series[Any]: + run = frames["run"] + if len(run) != 1: + _manifest_fail( + BacktestContractErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.run.row_count", + "run table must contain exactly one row", + ) + return run.iloc[0] + + +def _validate_qualified_run( + backtest_run_ref: BacktestRunRef, + frames: Mapping[str, pd.DataFrame], +) -> None: + row = _run_row(frames) + expected = { + "run_id": backtest_run_ref.run_id, + "data_snapshot_id": backtest_run_ref.dataset_snapshot_id, + "strategy_id": backtest_run_ref.strategy_id, + "strategy_version": backtest_run_ref.strategy_version, + "code_revision": backtest_run_ref.code_revision, + "config_hash": backtest_run_ref.configuration_digest.removeprefix("sha256:"), + } + for column, expected_value in expected.items(): + if type(row[column]) is not str or row[column] != expected_value: + _manifest_fail( + BacktestContractErrorCode.IDENTITY_MISMATCH, + f"$.artifact.tables.run.{column}", + "run identity differs from BacktestRunRef", + ) + + +def _validate_legacy_run( + legacy_run: BacktestRun, + frames: Mapping[str, pd.DataFrame], +) -> None: + row = _run_row(frames) + expected = { + "run_id": legacy_run.run_id, + "data_snapshot_id": legacy_run.dataset_snapshot_id, + "strategy_id": legacy_run.strategy_version_id.rsplit("@", maxsplit=1)[0], + "strategy_version": legacy_run.strategy_version_id.rsplit("@", maxsplit=1)[-1], + "code_revision": legacy_run.code_revision, + "config_hash": legacy_run.config_hash, + } + for column, expected_value in expected.items(): + if str(row[column]) != str(expected_value): + _manifest_fail( + BacktestContractErrorCode.IDENTITY_MISMATCH, + f"$.artifact.tables.run.{column}", + "artifact differs from explicit legacy run", + ) + + +def _build_evidence_manifest( + *, + run_id: str, + run_reference: dict[str, object], + artifact_available_at: object, + qualification: EvidenceQualification, + summaries: Mapping[str, BacktestEvidenceTable], + backtest_run_ref: BacktestRunRef | None, + legacy_backtest_run: BacktestRun | None, +) -> BacktestEvidenceManifest: + available_text, _ = _manifest_instant( + artifact_available_at, + "$.artifact_available_at", + ) + legacy = legacy_backtest_run is not None + evidence = _evidence_entries(summaries, run_reference, legacy=legacy) + evidence_digest = _manifest_digest([entry.to_dict() for entry in evidence]) + payload: dict[str, object] = { + "contract_name": "researchhub.backtest-evidence-manifest", + "schema_version": BACKTEST_EVIDENCE_SCHEMA_VERSION, + "run_id": run_id, + "profile": "offline_research_v1", + "artifact_available_at": available_text, + "qualification": qualification.value, + "run_reference": run_reference, + "evidence_digest": evidence_digest, + "evidence": [entry.to_dict() for entry in evidence], + } + manifest_id = ( + "rhbacktestevidencev1:sha256:" + f"{hashlib.sha256(_canonical_mapping_json(payload).encode('utf-8')).hexdigest()}" + ) + instance = object.__new__(BacktestEvidenceManifest) + values: dict[str, object] = { + **payload, + "manifest_id": manifest_id, + "qualification": qualification, + "evidence": evidence, + "backtest_run_ref": backtest_run_ref, + "legacy_backtest_run": legacy_backtest_run, + } + values.pop("run_reference") + for name, value in values.items(): + object.__setattr__(instance, name, value) + return instance + + +def build_backtest_evidence_manifest( + backtest_run_ref: BacktestRunRef, + artifact: ResearchRunArtifact, + *, + artifact_available_at: str | pd.Timestamp | datetime, + qualification: EvidenceQualification = EvidenceQualification.CONTRACT_QUALIFIED, + expected_table_digests: Mapping[str, str] | None = None, +) -> BacktestEvidenceManifest: + """Close the fixed offline profile without recomputing artifact business facts.""" + if not isinstance(backtest_run_ref, BacktestRunRef): + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.backtest_run_ref", + "complete BacktestRunRef is required", + ) + if not isinstance(qualification, EvidenceQualification): + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.qualification", + "EvidenceQualification is required", + ) + if qualification is EvidenceQualification.LEGACY_EXPLORATORY: + _manifest_fail( + BacktestContractErrorCode.READINESS_ESCALATION, + "$.qualification", + "legacy qualification requires the explicit legacy bridge", + ) + available_text, available_time = _manifest_instant( + artifact_available_at, + "$.artifact_available_at", + ) + _, computed_time = _manifest_instant( + backtest_run_ref.computed_at, + "$.backtest_run_ref.computed_at", + ) + if available_time < computed_time: + _manifest_fail( + BacktestContractErrorCode.TIME_ORDER_VIOLATION, + "$.artifact_available_at", + "artifact cannot be available before the run computation", + ) + frames = _artifact_frames(artifact) + _validate_table_run_ids(frames, backtest_run_ref.run_id) + _validate_qualified_run(backtest_run_ref, frames) + _, finished_time = _manifest_instant( + _run_row(frames)["finished_at"], + "$.artifact.tables.run.finished_at", + ) + if available_time < finished_time: + _manifest_fail( + BacktestContractErrorCode.TIME_ORDER_VIOLATION, + "$.artifact_available_at", + "artifact cannot be available before the artifact run finished", + ) + summaries = _table_evidence(frames, expected_table_digests) + run_reference: dict[str, object] = { + "kind": "backtest_run_ref", + "value": backtest_run_ref.to_dict(), + } + return _build_evidence_manifest( + run_id=backtest_run_ref.run_id, + run_reference=run_reference, + artifact_available_at=available_text, + qualification=qualification, + summaries=summaries, + backtest_run_ref=backtest_run_ref, + legacy_backtest_run=None, + ) + + +def build_legacy_backtest_evidence_manifest( + legacy_run: BacktestRun, + artifact: ResearchRunArtifact, + *, + artifact_available_at: str | pd.Timestamp | datetime, +) -> BacktestEvidenceManifest: + """Explicitly map an old run to non-qualified, lossy exploratory evidence.""" + if not isinstance(legacy_run, BacktestRun): + _manifest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.legacy_backtest_run", + "BacktestRun is required", + ) + available_text, available_time = _manifest_instant( + artifact_available_at, + "$.artifact_available_at", + ) + _, created_time = _manifest_instant( + legacy_run.created_at, + "$.legacy_backtest_run.created_at", + ) + if available_time < created_time: + _manifest_fail( + BacktestContractErrorCode.TIME_ORDER_VIOLATION, + "$.artifact_available_at", + "legacy evidence cannot predate its run", + ) + frames = _artifact_frames(artifact) + _validate_table_run_ids(frames, legacy_run.run_id) + _validate_legacy_run(legacy_run, frames) + _, finished_time = _manifest_instant( + _run_row(frames)["finished_at"], + "$.artifact.tables.run.finished_at", + ) + if available_time < finished_time: + _manifest_fail( + BacktestContractErrorCode.TIME_ORDER_VIOLATION, + "$.artifact_available_at", + "legacy evidence cannot predate artifact completion", + ) + summaries = _table_evidence(frames, None) + run_reference: dict[str, object] = { + "kind": "legacy_backtest_run", + "value": _legacy_run_dict(legacy_run), + } + return _build_evidence_manifest( + run_id=legacy_run.run_id, + run_reference=run_reference, + artifact_available_at=available_text, + qualification=EvidenceQualification.LEGACY_EXPLORATORY, + summaries=summaries, + backtest_run_ref=None, + legacy_backtest_run=legacy_run, + ) + + def _build_nav( result: FactorBacktestResult, run_id: str, @@ -511,7 +1173,7 @@ def build_research_run_artifact( raise TypeError("result must be a FactorBacktestResult") if result.nav.empty: raise ValueError("result must contain at least one research session") - normalized_run_id = _required_text(run_id, "run_id", max_length=64) + normalized_run_id = _required_text(run_id, "run_id", max_length=128) normalized_strategy_id = _required_text(strategy_id, "strategy_id") normalized_strategy_name = _required_text(strategy_name, "strategy_name") normalized_strategy_version = _required_text(strategy_version, "strategy_version") diff --git a/src/quant_engine/governed_pipeline.py b/src/quant_engine/governed_pipeline.py index 96e8433..d6cb2b0 100644 --- a/src/quant_engine/governed_pipeline.py +++ b/src/quant_engine/governed_pipeline.py @@ -11,27 +11,35 @@ import hashlib import json import math import re -from collections.abc import Mapping +from collections.abc import Mapping, Sequence from dataclasses import asdict, dataclass from datetime import UTC, datetime from enum import StrEnum from types import MappingProxyType +from typing import Any, Never, Self import pandas as pd from quant_engine.execution import ExecutionConfig from quant_engine.factor_contracts import ( ContractErrorCode, + DataFoundationEnvelope, + DatasetSnapshotEnvelope, FactorContractError, FactorDefinition, + FactorSetRef, LegacyFactorBinding, + canonical_json, + canonical_json_bytes, ) from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research _SHA256 = re.compile(r"^[0-9a-f]{64}$") +_PREFIXED_SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$") _GIT_SHA = re.compile(r"^[0-9a-f]{40}$") __all__ = [ + "BACKTEST_RUN_REF_SCHEMA_VERSION", "DatasetSnapshot", "FactorVersion", "bind_legacy_factor", @@ -39,6 +47,9 @@ __all__ = [ "StrategyStage", "StrategyVersion", "BacktestRun", + "BacktestContractErrorCode", + "BacktestContractError", + "BacktestRunRef", "PortfolioTarget", "RiskPolicy", "RiskDecisionStatus", @@ -51,6 +62,631 @@ __all__ = [ ] +BACKTEST_RUN_REF_SCHEMA_VERSION = "1.0.0" + + +class BacktestContractErrorCode(StrEnum): + """Stable rejection categories shared by the S3 public contracts.""" + + TYPE_ERROR = "type_error" + MISSING_FIELD = "missing_field" + UNKNOWN_FIELD = "unknown_field" + INVALID_FORMAT = "invalid_format" + INVALID_VALUE = "invalid_value" + IDENTITY_MISMATCH = "identity_mismatch" + QUALIFICATION_REJECTED = "qualification_rejected" + TIME_ORDER_VIOLATION = "time_order_violation" + INPUT_CLOSURE_VIOLATION = "input_closure_violation" + LINEAGE_VIOLATION = "lineage_violation" + EVIDENCE_MISMATCH = "evidence_mismatch" + READINESS_ESCALATION = "readiness_escalation" + + +class BacktestContractError(ValueError): + """Typed deterministic contract rejection with an exact JSON path.""" + + def __init__(self, code: BacktestContractErrorCode, path: str, detail: str) -> None: + self.code = code + self.path = path + self.detail = detail + super().__init__(f"{code.value} at {path}: {detail}") + + +def _backtest_fail( + code: BacktestContractErrorCode, + path: str, + detail: str, +) -> Never: + raise BacktestContractError(code, path, detail) + + +def _backtest_object(value: Any, path: str, fields: Sequence[str]) -> dict[str, Any]: + if type(value) is not dict: + _backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be an object") + required = set(fields) + missing = sorted(required - set(value)) + if missing: + _backtest_fail( + BacktestContractErrorCode.MISSING_FIELD, + f"{path}.{missing[0]}", + "field is required", + ) + unknown = sorted(set(value) - required) + if unknown: + _backtest_fail( + BacktestContractErrorCode.UNKNOWN_FIELD, + f"{path}.{unknown[0]}", + "field is not permitted", + ) + return value + + +def _backtest_text(value: Any, path: str) -> str: + if type(value) is not str: + _backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be a string") + if not value or value != value.strip(): + _backtest_fail( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be non-empty canonical text", + ) + return value + + +def _backtest_digest(value: Any, path: str) -> str: + digest = _backtest_text(value, path) + if _PREFIXED_SHA256.fullmatch(digest) is None: + _backtest_fail( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be a lowercase sha256 digest", + ) + return digest + + +def _backtest_git_revision(value: Any, path: str) -> str: + revision = _backtest_text(value, path) + if _GIT_SHA.fullmatch(revision) is None: + _backtest_fail( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be a lowercase 40-character Git commit", + ) + return revision + + +def _backtest_instant(value: Any, path: str) -> tuple[str, datetime]: + text = _backtest_text(value, path) + if not text.endswith("Z"): + _backtest_fail( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be a canonical UTC instant", + ) + try: + parsed = datetime.fromisoformat(text.removesuffix("Z") + "+00:00") + except ValueError as error: + raise BacktestContractError( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be a canonical UTC instant", + ) from error + normalized = parsed.astimezone(UTC).isoformat().replace("+00:00", "Z") + if normalized != text: + _backtest_fail( + BacktestContractErrorCode.INVALID_FORMAT, + path, + "must be a canonical UTC instant", + ) + return text, parsed + + +def _backtest_integer(value: Any, path: str, *, minimum: int = 0) -> int: + if type(value) is not int: + _backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be an integer") + if value < minimum: + _backtest_fail( + BacktestContractErrorCode.INVALID_VALUE, + path, + f"must be at least {minimum}", + ) + return value + + +def _backtest_string_tuple(value: Any, path: str) -> tuple[str, ...]: + if type(value) not in {tuple, list}: + _backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be an array") + normalized = tuple( + _backtest_text(item, f"{path}[{index}]") for index, item in enumerate(value) + ) + if len(normalized) != len(set(normalized)): + _backtest_fail( + BacktestContractErrorCode.INVALID_VALUE, + path, + "items must be unique", + ) + return tuple(sorted(normalized)) + + +def _content_address_digest(value: str, prefix: str, path: str) -> str: + if not value.startswith(prefix) or _SHA256.fullmatch(value.removeprefix(prefix)) is None: + _backtest_fail( + BacktestContractErrorCode.IDENTITY_MISMATCH, + path, + "authority identity is not content addressed", + ) + return f"sha256:{value.removeprefix(prefix)}" + + +def _digest_document(value: object) -> str: + return f"sha256:{hashlib.sha256(canonical_json_bytes(value)).hexdigest()}" + + +def _selected_revision_closure( + foundation: DataFoundationEnvelope, + factor_set: FactorSetRef, + field: str, +) -> tuple[str, ...]: + payload = foundation.to_dict() + raw_views = payload.get("standardized_views") + if type(raw_views) is not list: + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.foundation.standardized_views", + "validated Foundation views are required", + ) + selected = set(factor_set.selected_view_ref_ids) + found: set[str] = set() + seen_views: set[str] = set() + for index, raw_view in enumerate(raw_views): + if type(raw_view) is not dict: + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + f"$.foundation.standardized_views[{index}]", + "validated view object is required", + ) + view_id = raw_view.get("view_ref_id") + if view_id not in selected: + continue + seen_views.add(str(view_id)) + revisions = raw_view.get(field) + path = f"$.foundation.standardized_views[{index}].{field}" + if type(revisions) is not list: + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + path, + "revision closure is required", + ) + for revision_index, revision_id in enumerate(revisions): + found.add(_backtest_text(revision_id, f"{path}[{revision_index}]")) + if seen_views != selected: + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.factor_set.selected_view_ref_ids", + "selected FactorSet views are not Foundation members", + ) + return tuple(sorted(found)) + + +@dataclass(frozen=True, slots=True, init=False) +class BacktestRunRef: + """Immutable input/configuration/replay identity determined before output exists.""" + + contract_name: str + schema_version: str + run_id: str + dataset_snapshot_id: str + dataset_content_digest: str + dataset_manifest_digest: str + foundation_id: str + foundation_digest: str + factor_set_id: str + factor_set_digest: str + factor_output_content_digest: str + universe_digest: str + trading_calendar_revision_ids: tuple[str, ...] + trading_calendar_digest: str + corporate_action_revision_ids: tuple[str, ...] + corporate_action_digest: str + strategy_id: str + strategy_version: str + strategy_digest: str + execution_model_version: str + execution_model_digest: str + cost_model_version: str + cost_model_digest: str + random_seed: int + code_revision: str + environment_lock_digest: str + configuration_digest: str + evaluation_at: str + computed_at: str + replay_spec_digest: str + replay_parent_run_id: str | None + replay_reason: str | None + replay_attempt: int + replay_ancestor_run_ids: tuple[str, ...] + + @classmethod + def create( + cls, + *, + dataset_snapshot: DatasetSnapshotEnvelope, + foundation: DataFoundationEnvelope, + factor_set: FactorSetRef, + universe_digest: str, + trading_calendar_revision_ids: Sequence[str], + corporate_action_revision_ids: Sequence[str], + strategy_id: str, + strategy_version: str, + strategy_digest: str, + execution_model_version: str, + execution_model_digest: str, + cost_model_version: str, + cost_model_digest: str, + random_seed: int, + code_revision: str, + environment_lock_digest: str, + configuration_digest: str, + evaluation_at: str, + computed_at: str, + parent: BacktestRunRef | None = None, + replay_reason: str | None = None, + replay_attempt: int = 0, + ) -> Self: + return cls._build( + dataset_snapshot=dataset_snapshot, + foundation=foundation, + factor_set=factor_set, + universe_digest=universe_digest, + trading_calendar_revision_ids=trading_calendar_revision_ids, + corporate_action_revision_ids=corporate_action_revision_ids, + strategy_id=strategy_id, + strategy_version=strategy_version, + strategy_digest=strategy_digest, + execution_model_version=execution_model_version, + execution_model_digest=execution_model_digest, + cost_model_version=cost_model_version, + cost_model_digest=cost_model_digest, + random_seed=random_seed, + code_revision=code_revision, + environment_lock_digest=environment_lock_digest, + configuration_digest=configuration_digest, + evaluation_at=evaluation_at, + computed_at=computed_at, + parent=parent, + replay_reason=replay_reason, + replay_attempt=replay_attempt, + ) + + @classmethod + def _build( + cls, + *, + dataset_snapshot: Any, + foundation: Any, + factor_set: Any, + universe_digest: Any, + trading_calendar_revision_ids: Any, + corporate_action_revision_ids: Any, + strategy_id: Any, + strategy_version: Any, + strategy_digest: Any, + execution_model_version: Any, + execution_model_digest: Any, + cost_model_version: Any, + cost_model_digest: Any, + random_seed: Any, + code_revision: Any, + environment_lock_digest: Any, + configuration_digest: Any, + evaluation_at: Any, + computed_at: Any, + parent: BacktestRunRef | None, + replay_reason: Any, + replay_attempt: Any, + ) -> Self: + if not isinstance(dataset_snapshot, DatasetSnapshotEnvelope): + _backtest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.dataset_snapshot", + "complete validated DatasetSnapshotEnvelope required", + ) + if not isinstance(foundation, DataFoundationEnvelope): + _backtest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.foundation", + "complete validated DataFoundationEnvelope required", + ) + if not isinstance(factor_set, FactorSetRef): + _backtest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.factor_set", + "complete validated FactorSetRef required", + ) + if foundation.dataset_snapshot_id != dataset_snapshot.snapshot_id: + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.foundation.dataset_snapshot_id", + "Foundation snapshot mismatch", + ) + if ( + factor_set.dataset_snapshot_id != dataset_snapshot.snapshot_id + or factor_set.foundation_id != foundation.foundation_id + or factor_set.upstream_evidence.snapshot_content_digest + != dataset_snapshot.content_digest + or factor_set.upstream_evidence.snapshot_manifest_digest + != dataset_snapshot.manifest_digest + ): + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.factor_set", + "FactorSet upstream authority mismatch", + ) + expected_calendars = _selected_revision_closure( + foundation, + factor_set, + "trading_calendar_revision_ids", + ) + supplied_calendars = _backtest_string_tuple( + trading_calendar_revision_ids, + "$.trading_calendar_revision_ids", + ) + if not expected_calendars or supplied_calendars != expected_calendars: + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.trading_calendar_revision_ids", + "must exactly match selected FactorSet calendar ancestry", + ) + expected_actions = _selected_revision_closure( + foundation, + factor_set, + "corporate_action_revision_ids", + ) + supplied_actions = _backtest_string_tuple( + corporate_action_revision_ids, + "$.corporate_action_revision_ids", + ) + if supplied_actions != expected_actions: + _backtest_fail( + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.corporate_action_revision_ids", + "must exactly match selected FactorSet corporate-action ancestry", + ) + + normalized_evaluation, evaluation_time = _backtest_instant( + evaluation_at, + "$.evaluation_at", + ) + normalized_computed, computed_time = _backtest_instant(computed_at, "$.computed_at") + _, factor_available = _backtest_instant( + factor_set.artifact_available_at, + "$.factor_set.artifact_available_at", + ) + if computed_time < evaluation_time or computed_time < factor_available: + _backtest_fail( + BacktestContractErrorCode.TIME_ORDER_VIOLATION, + "$.computed_at", + "FactorSet availability and evaluation must not follow computation", + ) + + normalized_seed = _backtest_integer(random_seed, "$.random_seed") + replay_count = _backtest_integer(replay_attempt, "$.replay_attempt") + spec_payload: dict[str, object] = { + "dataset_snapshot_id": dataset_snapshot.snapshot_id, + "dataset_content_digest": dataset_snapshot.content_digest, + "dataset_manifest_digest": dataset_snapshot.manifest_digest, + "foundation_id": foundation.foundation_id, + "foundation_digest": _content_address_digest( + foundation.foundation_id, + "rhdfv1:sha256:", + "$.foundation.foundation_id", + ), + "factor_set_id": factor_set.factor_set_id, + "factor_set_digest": _content_address_digest( + factor_set.factor_set_id, + "rhfactorsetv1:sha256:", + "$.factor_set.factor_set_id", + ), + "factor_output_content_digest": factor_set.output_content_digest, + "universe_digest": _backtest_digest(universe_digest, "$.universe_digest"), + "trading_calendar_revision_ids": list(supplied_calendars), + "trading_calendar_digest": _digest_document(list(supplied_calendars)), + "corporate_action_revision_ids": list(supplied_actions), + "corporate_action_digest": _digest_document(list(supplied_actions)), + "strategy_id": _backtest_text(strategy_id, "$.strategy_id"), + "strategy_version": _backtest_text(strategy_version, "$.strategy_version"), + "strategy_digest": _backtest_digest(strategy_digest, "$.strategy_digest"), + "execution_model_version": _backtest_text( + execution_model_version, + "$.execution_model_version", + ), + "execution_model_digest": _backtest_digest( + execution_model_digest, + "$.execution_model_digest", + ), + "cost_model_version": _backtest_text(cost_model_version, "$.cost_model_version"), + "cost_model_digest": _backtest_digest( + cost_model_digest, + "$.cost_model_digest", + ), + "random_seed": normalized_seed, + "code_revision": _backtest_git_revision(code_revision, "$.code_revision"), + "environment_lock_digest": _backtest_digest( + environment_lock_digest, + "$.environment_lock_digest", + ), + "configuration_digest": _backtest_digest( + configuration_digest, + "$.configuration_digest", + ), + "evaluation_at": normalized_evaluation, + } + replay_spec_digest = _digest_document(spec_payload) + if parent is None: + if replay_reason is not None: + _backtest_fail( + BacktestContractErrorCode.LINEAGE_VIOLATION, + "$.replay_reason", + "root run cannot declare a replay reason", + ) + if replay_count != 0: + _backtest_fail( + BacktestContractErrorCode.LINEAGE_VIOLATION, + "$.replay_attempt", + "root run must use attempt zero", + ) + normalized_reason = None + parent_run_id = None + ancestors: tuple[str, ...] = () + else: + if not isinstance(parent, BacktestRunRef): + _backtest_fail( + BacktestContractErrorCode.TYPE_ERROR, + "$.parent", + "BacktestRunRef parent is required", + ) + normalized_reason = _backtest_text(replay_reason, "$.replay_reason") + if replay_count != parent.replay_attempt + 1: + _backtest_fail( + BacktestContractErrorCode.LINEAGE_VIOLATION, + "$.replay_attempt", + "replay attempt must follow its parent", + ) + if replay_spec_digest != parent.replay_spec_digest: + _backtest_fail( + BacktestContractErrorCode.LINEAGE_VIOLATION, + "$.replay_spec_digest", + "replay cannot claim changed deterministic inputs", + ) + _, parent_computed = _backtest_instant(parent.computed_at, "$.parent.computed_at") + if computed_time <= parent_computed: + _backtest_fail( + BacktestContractErrorCode.TIME_ORDER_VIOLATION, + "$.computed_at", + "replay computation must follow its parent", + ) + parent_run_id = parent.run_id + ancestors = (*parent.replay_ancestor_run_ids, parent.run_id) + if len(ancestors) != len(set(ancestors)): + _backtest_fail( + BacktestContractErrorCode.LINEAGE_VIOLATION, + "$.replay_ancestor_run_ids", + "replay lineage contains a cycle", + ) + + payload: dict[str, object] = { + "contract_name": "researchhub.backtest-run-ref", + "schema_version": BACKTEST_RUN_REF_SCHEMA_VERSION, + **spec_payload, + "computed_at": normalized_computed, + "replay_spec_digest": replay_spec_digest, + "replay_parent_run_id": parent_run_id, + "replay_reason": normalized_reason, + "replay_attempt": replay_count, + "replay_ancestor_run_ids": list(ancestors), + } + run_id = f"rhbacktestrunv1:sha256:{hashlib.sha256(canonical_json_bytes(payload)).hexdigest()}" + values: dict[str, object] = { + **payload, + "run_id": run_id, + "trading_calendar_revision_ids": supplied_calendars, + "corporate_action_revision_ids": supplied_actions, + "replay_ancestor_run_ids": ancestors, + } + instance = object.__new__(cls) + for name, value in values.items(): + object.__setattr__(instance, name, value) + return instance + + def to_dict(self) -> dict[str, Any]: + return { + "contract_name": self.contract_name, + "schema_version": self.schema_version, + "run_id": self.run_id, + "dataset_snapshot_id": self.dataset_snapshot_id, + "dataset_content_digest": self.dataset_content_digest, + "dataset_manifest_digest": self.dataset_manifest_digest, + "foundation_id": self.foundation_id, + "foundation_digest": self.foundation_digest, + "factor_set_id": self.factor_set_id, + "factor_set_digest": self.factor_set_digest, + "factor_output_content_digest": self.factor_output_content_digest, + "universe_digest": self.universe_digest, + "trading_calendar_revision_ids": list(self.trading_calendar_revision_ids), + "trading_calendar_digest": self.trading_calendar_digest, + "corporate_action_revision_ids": list(self.corporate_action_revision_ids), + "corporate_action_digest": self.corporate_action_digest, + "strategy_id": self.strategy_id, + "strategy_version": self.strategy_version, + "strategy_digest": self.strategy_digest, + "execution_model_version": self.execution_model_version, + "execution_model_digest": self.execution_model_digest, + "cost_model_version": self.cost_model_version, + "cost_model_digest": self.cost_model_digest, + "random_seed": self.random_seed, + "code_revision": self.code_revision, + "environment_lock_digest": self.environment_lock_digest, + "configuration_digest": self.configuration_digest, + "evaluation_at": self.evaluation_at, + "computed_at": self.computed_at, + "replay_spec_digest": self.replay_spec_digest, + "replay_parent_run_id": self.replay_parent_run_id, + "replay_reason": self.replay_reason, + "replay_attempt": self.replay_attempt, + "replay_ancestor_run_ids": list(self.replay_ancestor_run_ids), + } + + def to_json(self) -> str: + return canonical_json(self.to_dict()) + + @classmethod + def from_dict( + cls, + value: Any, + *, + dataset_snapshot: DatasetSnapshotEnvelope, + foundation: DataFoundationEnvelope, + factor_set: FactorSetRef, + parent: BacktestRunRef | None = None, + ) -> Self: + field_names = tuple(cls.__dataclass_fields__) + item = _backtest_object(value, "$", field_names) + rebuilt = cls._build( + dataset_snapshot=dataset_snapshot, + foundation=foundation, + factor_set=factor_set, + universe_digest=item["universe_digest"], + trading_calendar_revision_ids=item["trading_calendar_revision_ids"], + corporate_action_revision_ids=item["corporate_action_revision_ids"], + strategy_id=item["strategy_id"], + strategy_version=item["strategy_version"], + strategy_digest=item["strategy_digest"], + execution_model_version=item["execution_model_version"], + execution_model_digest=item["execution_model_digest"], + cost_model_version=item["cost_model_version"], + cost_model_digest=item["cost_model_digest"], + random_seed=item["random_seed"], + code_revision=item["code_revision"], + environment_lock_digest=item["environment_lock_digest"], + configuration_digest=item["configuration_digest"], + evaluation_at=item["evaluation_at"], + computed_at=item["computed_at"], + parent=parent, + replay_reason=item["replay_reason"], + replay_attempt=item["replay_attempt"], + ) + expected = rebuilt.to_dict() + for name in field_names: + if item[name] != expected[name]: + _backtest_fail( + BacktestContractErrorCode.IDENTITY_MISMATCH, + f"$.{name}", + "serialized run reference does not match admitted authorities", + ) + return rebuilt + + def _required_text(value: str, name: str) -> str: normalized = value.strip() if not normalized: diff --git a/tests/fixtures/backtest-evidence-v1.golden.json b/tests/fixtures/backtest-evidence-v1.golden.json new file mode 100644 index 0000000..c9a5d07 --- /dev/null +++ b/tests/fixtures/backtest-evidence-v1.golden.json @@ -0,0 +1,17 @@ +{ + "run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "replay_spec_digest": "sha256:20f07fcb526bc38b4ba3d71d6c3b00a8c63ad96cab0fecd869326503ab42d98b", + "manifest_id": "rhbacktestevidencev1:sha256:47e86c53e696672a8931e1cb496b6a48344a4ffbf84bbec4818ea2671fee8d42", + "evidence_digest": "sha256:907775054f4316bcb2f559a3068493210cebaa97bb90e462ba39ae0762eea98e", + "table_content_digests": { + "run": "sha256:7d947ec93f714641669cbb14bd69dd8cf30387aabb7f3a086918fc2e878ab05c", + "signals": "sha256:72dc15064cc45d7c51d2dd4b8c3a6d8c7d4155232d5d70d3c9e7697fab70ce50", + "trades": "sha256:2b8b9321e7993941ac486cf50cab5b4c6425b2571a4701f0eb02273ec9a96c53", + "positions": "sha256:b456a48fab51742b05084ca6dcaf01c03dfe5b39ad71215ada070e9b1f59f7ea", + "nav": "sha256:25649efce860b76410f87dbd36c81085887620086dc0b48b07099918f65d9c78", + "performance": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf", + "attribution": "sha256:ad0f12f668d0a2d9ae5b3636d29989ab4bafcfb95f5de77abeb52a6d9e95d366", + "attribution_daily": "sha256:d4459ad650f88871d7b1e40392033037b5818ef92fdf1ee021868929fc1455a5", + "risk": "sha256:4816dd4812b5ff2e97e74bf34ca2221bfde675b387a96e277683557f0ad7d975" + } +} diff --git a/tests/governance/test_module_spec.py b/tests/governance/test_module_spec.py index ad189b2..b98de24 100644 --- a/tests/governance/test_module_spec.py +++ b/tests/governance/test_module_spec.py @@ -16,19 +16,26 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower() for term in ("investment advice", "live order", "credentials", "source facts"): assert term in prohibited - assert spec["authority"]["revision"] == 2 + assert spec["authority"]["revision"] == 3 assert { (item["contract_id"], item["version"]) for item in spec["contracts"]["provides"] } == { ("researchhub.factor-definition", "1.0.0"), ("researchhub.factor-set-ref", "1.0.0"), + ("researchhub.backtest-run-ref", "1.0.0"), + ("researchhub.backtest-evidence-manifest", "1.0.0"), } - assert all( - item["authority"] == "quant_engine" - and item["path"] == "src/quant_engine/factor_contracts.py" - for item in spec["contracts"]["provides"] - ) + expected_paths = { + "researchhub.factor-definition": "src/quant_engine/factor_contracts.py", + "researchhub.factor-set-ref": "src/quant_engine/factor_contracts.py", + "researchhub.backtest-run-ref": "src/quant_engine/governed_pipeline.py", + "researchhub.backtest-evidence-manifest": "src/quant_engine/artifact.py", + } + assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"]) + assert { + item["contract_id"]: item["path"] for item in spec["contracts"]["provides"] + } == expected_paths assert { (item["contract_id"], item["version"]) for item in spec["contracts"]["consumes"] diff --git a/tests/test_backtest_contracts.py b/tests/test_backtest_contracts.py new file mode 100644 index 0000000..aa68c27 --- /dev/null +++ b/tests/test_backtest_contracts.py @@ -0,0 +1,533 @@ +"""Backtest run-reference and closed-evidence contract conformance.""" + +from __future__ import annotations + +import copy +import hashlib +import json +from dataclasses import replace +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +import pandas as pd +import pytest + +from quant_engine.artifact import ( + BacktestEvidenceManifest, + EvidenceQualification, + ResearchRunArtifact, + build_backtest_evidence_manifest, + build_legacy_backtest_evidence_manifest, + build_research_run_artifact, +) +from quant_engine.execution import ExecutionConfig +from quant_engine.factor_contracts import ( + ActorIdentity, + AvailabilityMode, + Causation, + DataFoundationEnvelope, + DatasetSnapshotEnvelope, + FactorInput, + FactorSetRef, + InputBinding, + OutputArtifactRef, + OutputCoverage, + OutputQuality, + OutputQualityCheck, + ProducerIdentity, + ViewAvailability, + canonical_json_bytes, + factor_definition_from_alpha158, + factor_input_schema_digest, +) +from quant_engine.governed_pipeline import ( + BacktestContractError, + BacktestContractErrorCode, + BacktestRun, + BacktestRunRef, +) +from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research + + +ROOT = Path(__file__).resolve().parents[1] +FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json" +BACKTEST_FIXTURE = ROOT / "tests" / "fixtures" / "backtest-evidence-v1.golden.json" +VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c" +VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef" +CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078" +ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275" +PARAMETERS = {"lag_sessions": 1, "top_k": 1} + + +def _sha256(value: bytes) -> str: + return f"sha256:{hashlib.sha256(value).hexdigest()}" + + +def _accepted_authorities() -> tuple[ + DatasetSnapshotEnvelope, + DataFoundationEnvelope, + FactorSetRef, +]: + fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8")) + snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"]) + foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"]) + factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume")) + definition = factor_definition_from_alpha158( + "alpha_005", + version="1.0.0", + parameters={}, + inputs=(factor_input,), + implementation_digest="sha256:" + "1" * 64, + input_schema_digest=factor_input_schema_digest((factor_input,)), + valid_from="2026-01-01T00:00:00.000000Z", + valid_until="2027-01-01T00:00:00Z", + warmup_sessions=10, + lag_sessions=1, + producer=ProducerIdentity("quant_engine", "1.0.0"), + code_revision="c" * 40, + ) + output_schema_bytes = canonical_json_bytes(fixture["output_schema"]) + output_content_bytes = canonical_json_bytes(fixture["output_content"]) + artifact_ref = OutputArtifactRef.create( + schema_digest=_sha256(output_schema_bytes), + content_digest=_sha256(output_content_bytes), + ) + factor_set = FactorSetRef.create( + definitions=(definition,), + dataset_snapshot=snapshot, + foundation=foundation, + selected_view_ref_ids=(VIEW_REF_ID,), + input_bindings=( + InputBinding(definition.definition_id, "market", VIEW_REF_ID, VIEW_SCHEMA_DIGEST), + ), + view_availability=( + ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64), + ), + output_quality=OutputQuality( + "passed", + (OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),), + ), + output_coverage=OutputCoverage( + "complete", 1, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64 + ), + output_schema_bytes=output_schema_bytes, + output_content_bytes=output_content_bytes, + output_artifact_ref=artifact_ref, + availability_mode=AvailabilityMode.AS_AVAILABLE, + evaluation_at="2026-01-03T11:00:00Z", + computed_at="2026-01-03T10:15:00Z", + artifact_available_at="2026-01-03T10:20:00Z", + producer=ProducerIdentity("quant_engine", "1.0.0"), + code_revision="c" * 40, + actor=ActorIdentity("service", "factor_worker_v1"), + correlation_id="research_run_001", + causation=Causation("foundation", foundation.foundation_id), + evidence_scope="synthetic_fixture", + decision_eligible=False, + ) + return snapshot, foundation, factor_set + + +def _config_digest(parameters: dict[str, object] | None = None) -> str: + encoded = json.dumps( + PARAMETERS if parameters is None else parameters, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode("utf-8") + return _sha256(encoded) + + +def _run_ref(**overrides: Any) -> BacktestRunRef: + snapshot, foundation, factor_set = _accepted_authorities() + arguments: dict[str, Any] = { + "dataset_snapshot": snapshot, + "foundation": foundation, + "factor_set": factor_set, + "universe_digest": "sha256:" + "5" * 64, + "trading_calendar_revision_ids": (CALENDAR_REVISION_ID,), + "corporate_action_revision_ids": (ACTION_REVISION_ID,), + "strategy_id": "alpha-top1", + "strategy_version": "1.0.0", + "strategy_digest": "sha256:" + "6" * 64, + "execution_model_version": "1.0.0", + "execution_model_digest": "sha256:" + "7" * 64, + "cost_model_version": "1.0.0", + "cost_model_digest": "sha256:" + "8" * 64, + "random_seed": 7, + "code_revision": "d" * 40, + "environment_lock_digest": "sha256:" + "9" * 64, + "configuration_digest": _config_digest(), + "evaluation_at": "2026-01-08T01:00:00Z", + "computed_at": "2026-01-08T02:00:00Z", + } + arguments.update(overrides) + return BacktestRunRef.create(**arguments) + + +def _backtest_result() -> FactorBacktestResult: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2]) + opens = pd.DataFrame( + {"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]}, + index=dates, + ) + return run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ), + ) + + +def _artifact(run_ref: BacktestRunRef, *, run_id: str | None = None) -> ResearchRunArtifact: + result = _backtest_result() + benchmark = pd.Series( + [0.0, 0.01, -0.01, 0.02], + index=result.returns.index, + name="benchmark_return", + ) + return build_research_run_artifact( + result, + run_id=run_ref.run_id if run_id is None else run_id, + strategy_id=run_ref.strategy_id, + strategy_name="Alpha Top 1", + strategy_version=run_ref.strategy_version, + engine_version="1.2.0", + code_revision=run_ref.code_revision, + data_snapshot_id=run_ref.dataset_snapshot_id, + calendar="CN-A", + timezone="Asia/Shanghai", + started_at="2026-01-08T10:00:00+08:00", + finished_at="2026-01-08T10:01:00+08:00", + parameters=PARAMETERS, + benchmark_id="000300.SH", + benchmark_returns=benchmark, + ) + + +def _assert_error( + error: pytest.ExceptionInfo[BacktestContractError], + code: BacktestContractErrorCode, + path: str, +) -> None: + assert error.value.code is code + assert error.value.path == path + + +def test_backtest_run_ref_is_deterministic_and_binds_only_opaque_authorities() -> None: + first = _run_ref() + second = _run_ref() + + assert first == second + assert first.run_id.startswith("rhbacktestrunv1:sha256:") + assert first.replay_spec_digest.startswith("sha256:") + assert first.dataset_snapshot_id.startswith("rhdsv1:sha256:") + assert first.foundation_id.startswith("rhdfv1:sha256:") + assert first.factor_set_id.startswith("rhfactorsetv1:sha256:") + assert first.trading_calendar_revision_ids == (CALENDAR_REVISION_ID,) + assert first.corporate_action_revision_ids == (ACTION_REVISION_ID,) + assert first.replay_parent_run_id is None + assert first.replay_attempt == 0 + assert first.replay_ancestor_run_ids == () + snapshot, foundation, factor_set = _accepted_authorities() + assert BacktestRunRef.from_dict( + first.to_dict(), + dataset_snapshot=snapshot, + foundation=foundation, + factor_set=factor_set, + ) == first + forbidden = ("latest", "locator", "uri", "credential", "provider", "broker") + assert not any(token in first.to_json().lower() for token in forbidden) + + +@pytest.mark.parametrize( + ("field", "value"), + [ + ("universe_digest", "sha256:" + "a" * 64), + ("strategy_digest", "sha256:" + "b" * 64), + ("execution_model_digest", "sha256:" + "c" * 64), + ("cost_model_digest", "sha256:" + "e" * 64), + ("random_seed", 8), + ("code_revision", "e" * 40), + ("environment_lock_digest", "sha256:" + "f" * 64), + ("configuration_digest", "sha256:" + "0" * 64), + ("evaluation_at", "2026-01-08T01:00:01Z"), + ("computed_at", "2026-01-08T02:00:01Z"), + ], +) +def test_every_governed_run_input_mutation_changes_run_identity( + field: str, + value: object, +) -> None: + assert _run_ref(**{field: value}).run_id != _run_ref().run_id + + +def test_backtest_run_ref_rejects_unclosed_upstream_and_unsafe_scalars() -> None: + with pytest.raises(BacktestContractError) as wrong_calendar: + _run_ref(trading_calendar_revision_ids=()) + _assert_error( + wrong_calendar, + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.trading_calendar_revision_ids", + ) + with pytest.raises(BacktestContractError) as wrong_action: + _run_ref(corporate_action_revision_ids=()) + _assert_error( + wrong_action, + BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION, + "$.corporate_action_revision_ids", + ) + with pytest.raises(BacktestContractError) as bool_seed: + _run_ref(random_seed=True) + _assert_error(bool_seed, BacktestContractErrorCode.TYPE_ERROR, "$.random_seed") + with pytest.raises(BacktestContractError) as bad_revision: + _run_ref(code_revision="abc") + _assert_error(bad_revision, BacktestContractErrorCode.INVALID_FORMAT, "$.code_revision") + with pytest.raises(BacktestContractError) as bad_digest: + _run_ref(universe_digest="5" * 64) + _assert_error(bad_digest, BacktestContractErrorCode.INVALID_FORMAT, "$.universe_digest") + with pytest.raises(BacktestContractError) as lookahead: + _run_ref(computed_at="2026-01-08T00:59:59Z") + _assert_error(lookahead, BacktestContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at") + with pytest.raises(BacktestContractError) as factor_type: + _run_ref(factor_set="rhfactorsetv1:sha256:" + "0" * 64) + _assert_error(factor_type, BacktestContractErrorCode.TYPE_ERROR, "$.factor_set") + + +def test_replay_lineage_is_acyclic_and_cannot_claim_changed_inputs() -> None: + parent = _run_ref() + replay = _run_ref( + computed_at="2026-01-08T03:00:00Z", + parent=parent, + replay_reason="deterministic_reproduction", + replay_attempt=1, + ) + + assert replay.run_id != parent.run_id + assert replay.replay_spec_digest == parent.replay_spec_digest + assert replay.replay_parent_run_id == parent.run_id + assert replay.replay_ancestor_run_ids == (parent.run_id,) + + with pytest.raises(BacktestContractError) as changed_input: + _run_ref( + universe_digest="sha256:" + "a" * 64, + computed_at="2026-01-08T03:00:00Z", + parent=parent, + replay_reason="changed_universe", + replay_attempt=1, + ) + _assert_error( + changed_input, + BacktestContractErrorCode.LINEAGE_VIOLATION, + "$.replay_spec_digest", + ) + with pytest.raises(BacktestContractError) as skipped_attempt: + _run_ref( + computed_at="2026-01-08T03:00:00Z", + parent=parent, + replay_reason="skipped_attempt", + replay_attempt=2, + ) + _assert_error( + skipped_attempt, + BacktestContractErrorCode.LINEAGE_VIOLATION, + "$.replay_attempt", + ) + + +def test_offline_research_manifest_closes_exact_existing_evidence_mapping() -> None: + run_ref = _run_ref() + artifact = _artifact(run_ref) + first = build_backtest_evidence_manifest( + run_ref, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + qualification=EvidenceQualification.CONTRACT_QUALIFIED, + ) + second = build_backtest_evidence_manifest( + run_ref, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + qualification=EvidenceQualification.CONTRACT_QUALIFIED, + ) + + assert first == second + assert first.manifest_id.startswith("rhbacktestevidencev1:sha256:") + assert first.run_id == run_ref.run_id + assert first.profile == "offline_research_v1" + assert first.qualification is EvidenceQualification.CONTRACT_QUALIFIED + mapping = { + item.category: tuple(table.logical_name for table in item.tables) + for item in first.evidence + } + assert mapping == { + "run": ("run",), + "signal": ("signals",), + "fill": ("trades",), + "position_nav": ("positions", "nav"), + "performance": ("performance",), + "attribution": ("attribution", "attribution_daily"), + "risk_snapshot": ("risk",), + "replay": (), + } + assert "order" not in mapping + assert "rejection" not in mapping + risk = next(item for item in first.evidence if item.category == "risk_snapshot") + assert risk.tables[0].row_count == 0 + assert risk.tables[0].schema_digest.startswith("sha256:") + + changed_performance = artifact.performance + changed_performance.loc[0, "n_days"] += 1 + changed_artifact = replace(artifact, _performance=changed_performance) + changed = build_backtest_evidence_manifest( + run_ref, + changed_artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + assert changed.manifest_id != first.manifest_id + assert run_ref.run_id == first.run_id == changed.run_id + + +def test_manifest_rejects_missing_mismatched_or_duplicate_evidence() -> None: + run_ref = _run_ref() + artifact = _artifact(run_ref) + with pytest.raises(BacktestContractError) as wrong_run: + build_backtest_evidence_manifest( + run_ref, + _artifact(run_ref, run_id="different-run"), + artifact_available_at="2026-01-08T02:05:00Z", + ) + _assert_error( + wrong_run, + BacktestContractErrorCode.IDENTITY_MISMATCH, + "$.artifact.tables.run.run_id", + ) + missing_signals = replace(artifact, _signals=None) # type: ignore[arg-type] + with pytest.raises(BacktestContractError) as missing_table: + build_backtest_evidence_manifest( + run_ref, + missing_signals, + artifact_available_at="2026-01-08T02:05:00Z", + ) + _assert_error( + missing_table, + BacktestContractErrorCode.TYPE_ERROR, + "$.artifact.tables.signals", + ) + with pytest.raises(BacktestContractError) as digest_mismatch: + build_backtest_evidence_manifest( + run_ref, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + expected_table_digests={"performance": "sha256:" + "0" * 64}, + ) + _assert_error( + digest_mismatch, + BacktestContractErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.performance.content_digest", + ) + manifest = build_backtest_evidence_manifest( + run_ref, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + duplicate = manifest.to_dict() + duplicate["evidence"].append(copy.deepcopy(duplicate["evidence"][0])) + with pytest.raises(BacktestContractError) as duplicate_category: + BacktestEvidenceManifest.from_dict( + duplicate, + backtest_run_ref=run_ref, + artifact=artifact, + ) + _assert_error( + duplicate_category, + BacktestContractErrorCode.INVALID_VALUE, + "$.evidence[8].category", + ) + + +def test_legacy_bridge_is_explicit_and_cannot_be_contract_qualified() -> None: + run_ref = _run_ref() + legacy_run = BacktestRun( + run_id="legacy-run-001", + dataset_snapshot_id=run_ref.dataset_snapshot_id, + factor_version_id="alpha_005@1.0.0", + strategy_version_id="alpha-top1@1.0.0", + code_revision=run_ref.code_revision, + config_hash=_config_digest().removeprefix("sha256:"), + created_at=datetime(2026, 1, 8, 2, 0, tzinfo=UTC), + ) + artifact = _artifact(run_ref, run_id=legacy_run.run_id) + manifest = build_legacy_backtest_evidence_manifest( + legacy_run, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + + assert manifest.qualification is EvidenceQualification.LEGACY_EXPLORATORY + assert manifest.run_id == legacy_run.run_id + assert manifest.backtest_run_ref is None + assert manifest.to_dict()["run_reference"]["kind"] == "legacy_backtest_run" + with pytest.raises(BacktestContractError) as implicit_promotion: + build_backtest_evidence_manifest( # type: ignore[arg-type] + legacy_run, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + qualification=EvidenceQualification.CONTRACT_QUALIFIED, + ) + _assert_error( + implicit_promotion, + BacktestContractErrorCode.TYPE_ERROR, + "$.backtest_run_ref", + ) + + +def test_golden_contract_and_architecture_boundary() -> None: + run_ref = _run_ref() + artifact = _artifact(run_ref) + manifest = build_backtest_evidence_manifest( + run_ref, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + golden = json.loads(BACKTEST_FIXTURE.read_text(encoding="utf-8")) + + table_digests = { + table.logical_name: table.content_digest + for item in manifest.evidence + for table in item.tables + } + assert golden == { + "run_id": run_ref.run_id, + "replay_spec_digest": run_ref.replay_spec_digest, + "manifest_id": manifest.manifest_id, + "evidence_digest": manifest.evidence_digest, + "table_content_digests": table_digests, + } + governed_source = (ROOT / "src" / "quant_engine" / "governed_pipeline.py").read_text( + encoding="utf-8" + ) + artifact_source = (ROOT / "src" / "quant_engine" / "artifact.py").read_text( + encoding="utf-8" + ) + assert "from quant_engine.artifact" not in governed_source + assert "BacktestRunRef" in governed_source + assert "BacktestEvidenceManifest" not in governed_source + assert "BacktestEvidenceManifest" in artifact_source + assert not (ROOT / "src" / "quant_engine" / "backtest_contracts.py").exists()