feat: close backtest evidence contract

This commit is contained in:
ao gong
2026-09-01 12:04:48 +08:00
parent 598c2b92a2
commit e1e10ce293
7 changed files with 1891 additions and 15 deletions
+5 -2
View File
@@ -1,7 +1,7 @@
{ {
"schema_version": 1, "schema_version": 1,
"module_id": "quant_engine", "module_id": "quant_engine",
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 2, "effective_from": "2026-09-01T00:00:00+08:00"}, "authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 3, "effective_from": "2026-09-01T00:00:00+08:00"},
"repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"}, "repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"},
"bounded_context": { "bounded_context": {
"domain": "quantitative-research-engine", "domain": "quantitative-research-engine",
@@ -18,6 +18,7 @@
{"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"}, {"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"},
{"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"}, {"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"},
{"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"}, {"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"},
{"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact evidence without persistence or decision authority.", "status": "operational"},
{"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"} {"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"}
], ],
"data": {"owns": [ "data": {"owns": [
@@ -27,7 +28,9 @@
"contracts": { "contracts": {
"provides": [ "provides": [
{"contract_id": "researchhub.factor-definition", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"}, {"contract_id": "researchhub.factor-definition", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
{"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"} {"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
{"contract_id": "researchhub.backtest-run-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/governed_pipeline.py"},
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"}
], ],
"consumes": [ "consumes": [
{"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"}, {"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"},
+20 -2
View File
@@ -26,8 +26,8 @@
- `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark) - `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark)
- `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表 - `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表
- `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排) - `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排)
- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;全链路带确定性 ID,风险拒绝时禁止生成订单意图 - `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef`
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表与 manifest - `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest`
- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计 - `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计
- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效 - `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效
- `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test) - `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test)
@@ -241,6 +241,24 @@ identity 均保持不变。迁移只能通过 content-addressed `LegacyFactorBin
`bind_legacy_factor()` 或 `project_legacy_factor()`;后者是有损投影,不表示旧 digest 与新定义 `bind_legacy_factor()` 或 `project_legacy_factor()`;后者是有损投影,不表示旧 digest 与新定义
digest 等价,也不会把旧 run 静默升级为新合同。 digest 等价,也不会把旧 run 静默升级为新合同。
## 回测引用与证据合同 v1
`quant_engine.governed_pipeline.BacktestRunRef` 是合格回测运行身份的唯一权威。`run_id` 只由
已验收的 Dataset Snapshot / Data Foundation / `FactorSetRef` 身份、universe、日历与公司行动
祖先、策略、执行/成本模型、严格整数 seed、完整代码提交、环境锁、配置、时间和重放血缘决定;
它不包含任何输出摘要。重放必须绑定直接父运行、连续 attempt 和不变的
`replay_spec_digest`,输入漂移或血缘环会失败关闭。
`quant_engine.artifact.BacktestEvidenceManifest` 只摘要 `ResearchRunArtifact` 已有的九张事实表。
固定 `offline_research_v1` 映射为 `run`、`signal`、`fill`、`position_nav`、`performance`、
`attribution`、`risk_snapshot` 与 `replay`;每张表都保留列模式摘要、行数和内容摘要,空 risk
表也必须有稳定 schema。`manifest_id` 由完整 RunRef 与输出证据决定,因此结果变化不会反向改变
`run_id`。当前 artifact 不拥有订单或拒绝事实,所以此画像明确不声明 `order` / `rejection`。
旧 `BacktestRun` 只能通过 `build_legacy_backtest_evidence_manifest()` 显式映射为
`LEGACY_EXPLORATORY`;不能隐式提升为 `CONTRACT_QUALIFIED`。所有资格均只描述离线证据闭合,
不表示投资有效、组合获批、Paper、生产或实盘就绪。
## 治理垂直切片 ## 治理垂直切片
`governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。 `governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。
+666 -4
View File
@@ -11,18 +11,27 @@ from __future__ import annotations
import hashlib import hashlib
import json import json
import math import math
from collections.abc import Mapping import re
from collections.abc import Mapping, Sequence
from dataclasses import dataclass from dataclasses import dataclass
from datetime import date, datetime from datetime import UTC, date, datetime
from typing import Any from enum import StrEnum
from typing import Any, Never, Self, cast
import numpy as np import numpy as np
import pandas as pd import pandas as pd
from quant_engine.governed_pipeline import (
BacktestContractError,
BacktestContractErrorCode,
BacktestRun,
BacktestRunRef,
)
from quant_engine.research_pipeline import FactorBacktestResult from quant_engine.research_pipeline import FactorBacktestResult
from quant_engine.risk import CovarianceSnapshot, labeled_component_risk from quant_engine.risk import CovarianceSnapshot, labeled_component_risk
RESEARCH_ARTIFACT_SCHEMA_VERSION = "1.1.0" RESEARCH_ARTIFACT_SCHEMA_VERSION = "1.1.0"
BACKTEST_EVIDENCE_SCHEMA_VERSION = "1.0.0"
RISK_COLUMNS = [ RISK_COLUMNS = [
"run_id", "run_id",
@@ -41,7 +50,14 @@ RISK_COLUMNS = [
__all__ = [ __all__ = [
"RESEARCH_ARTIFACT_SCHEMA_VERSION", "RESEARCH_ARTIFACT_SCHEMA_VERSION",
"BACKTEST_EVIDENCE_SCHEMA_VERSION",
"ResearchRunArtifact", "ResearchRunArtifact",
"EvidenceQualification",
"BacktestEvidenceTable",
"BacktestEvidenceEntry",
"BacktestEvidenceManifest",
"build_backtest_evidence_manifest",
"build_legacy_backtest_evidence_manifest",
"build_research_run_artifact", "build_research_run_artifact",
] ]
@@ -162,6 +178,184 @@ class ResearchRunArtifact:
} }
class EvidenceQualification(StrEnum):
"""Evidence closure state; none of the states grants decision or trading authority."""
LEGACY_EXPLORATORY = "legacy_exploratory"
EXPLORATORY = "exploratory"
CONTRACT_QUALIFIED = "contract_qualified"
@dataclass(frozen=True, slots=True)
class BacktestEvidenceTable:
"""Digest and row count for one existing artifact table."""
logical_name: str
row_count: int
schema_digest: str
content_digest: str
def to_dict(self) -> dict[str, object]:
return {
"logical_name": self.logical_name,
"row_count": self.row_count,
"schema_digest": self.schema_digest,
"content_digest": self.content_digest,
}
@dataclass(frozen=True, slots=True)
class BacktestEvidenceEntry:
"""Closed logical evidence category over one or more authoritative tables."""
category: str
tables: tuple[BacktestEvidenceTable, ...]
evidence_digest: str
reconciliation: str
def to_dict(self) -> dict[str, object]:
return {
"category": self.category,
"tables": [table.to_dict() for table in self.tables],
"evidence_digest": self.evidence_digest,
"reconciliation": self.reconciliation,
}
@dataclass(frozen=True, slots=True, init=False)
class BacktestEvidenceManifest:
"""Content-addressed closure over a run reference and existing artifact evidence."""
contract_name: str
schema_version: str
manifest_id: str
run_id: str
profile: str
artifact_available_at: str
qualification: EvidenceQualification
evidence_digest: str
evidence: tuple[BacktestEvidenceEntry, ...]
backtest_run_ref: BacktestRunRef | None
legacy_backtest_run: BacktestRun | None
def _run_reference_dict(self) -> dict[str, object]:
if self.backtest_run_ref is not None:
return {
"kind": "backtest_run_ref",
"value": self.backtest_run_ref.to_dict(),
}
if self.legacy_backtest_run is None:
raise RuntimeError("manifest has no run reference")
return {
"kind": "legacy_backtest_run",
"value": _legacy_run_dict(self.legacy_backtest_run),
}
def to_dict(self) -> dict[str, Any]:
return {
"contract_name": self.contract_name,
"schema_version": self.schema_version,
"manifest_id": self.manifest_id,
"run_id": self.run_id,
"profile": self.profile,
"artifact_available_at": self.artifact_available_at,
"qualification": self.qualification.value,
"run_reference": self._run_reference_dict(),
"evidence_digest": self.evidence_digest,
"evidence": [item.to_dict() for item in self.evidence],
}
def to_json(self) -> str:
return _canonical_mapping_json(self.to_dict())
@classmethod
def from_dict(
cls,
value: Any,
*,
backtest_run_ref: BacktestRunRef,
artifact: ResearchRunArtifact,
) -> Self:
if type(value) is not dict:
_manifest_fail(BacktestContractErrorCode.TYPE_ERROR, "$", "must be an object")
expected_fields = {
"contract_name",
"schema_version",
"manifest_id",
"run_id",
"profile",
"artifact_available_at",
"qualification",
"run_reference",
"evidence_digest",
"evidence",
}
missing = sorted(expected_fields - set(value))
if missing:
_manifest_fail(
BacktestContractErrorCode.MISSING_FIELD,
f"$.{missing[0]}",
"field is required",
)
unknown = sorted(set(value) - expected_fields)
if unknown:
_manifest_fail(
BacktestContractErrorCode.UNKNOWN_FIELD,
f"$.{unknown[0]}",
"field is not permitted",
)
raw_evidence = value["evidence"]
if type(raw_evidence) is not list:
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.evidence",
"must be an array",
)
seen: set[str] = set()
for index, raw_entry in enumerate(raw_evidence):
if type(raw_entry) is not dict:
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
f"$.evidence[{index}]",
"must be an object",
)
category = raw_entry.get("category")
if type(category) is not str:
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
f"$.evidence[{index}].category",
"must be a string",
)
if category in seen:
_manifest_fail(
BacktestContractErrorCode.INVALID_VALUE,
f"$.evidence[{index}].category",
"duplicate evidence category",
)
seen.add(category)
try:
qualification = EvidenceQualification(value["qualification"])
except (TypeError, ValueError) as error:
raise BacktestContractError(
BacktestContractErrorCode.INVALID_VALUE,
"$.qualification",
"unsupported evidence qualification",
) from error
rebuilt = build_backtest_evidence_manifest(
backtest_run_ref,
artifact,
artifact_available_at=value["artifact_available_at"],
qualification=qualification,
)
if value != rebuilt.to_dict():
_manifest_fail(
BacktestContractErrorCode.IDENTITY_MISMATCH,
"$",
"serialized manifest does not match closed artifact evidence",
)
return cast(Self, rebuilt)
def _required_text(value: str, name: str, *, max_length: int | None = None) -> str: def _required_text(value: str, name: str, *, max_length: int | None = None) -> str:
normalized = value.strip() normalized = value.strip()
if not normalized: if not normalized:
@@ -222,6 +416,474 @@ def _frame_records(frame: pd.DataFrame) -> list[dict[str, object]]:
] ]
_MANIFEST_DIGEST = re.compile(r"^sha256:[0-9a-f]{64}$")
_OFFLINE_RESEARCH_V1: tuple[tuple[str, tuple[str, ...]], ...] = (
("run", ("run",)),
("signal", ("signals",)),
("fill", ("trades",)),
("position_nav", ("positions", "nav")),
("performance", ("performance",)),
("attribution", ("attribution", "attribution_daily")),
("risk_snapshot", ("risk",)),
("replay", ()),
)
_ARTIFACT_TABLE_NAMES = tuple(
table_name
for category, table_names in _OFFLINE_RESEARCH_V1
if category != "replay"
for table_name in table_names
)
_REQUIRED_TABLE_COLUMNS: Mapping[str, frozenset[str]] = {
"run": frozenset(
{
"run_id",
"data_snapshot_id",
"strategy_id",
"strategy_version",
"code_revision",
"config_hash",
"finished_at",
}
),
"signals": frozenset({"run_id", "signal_date", "execution_date", "asset_id"}),
"trades": frozenset({"run_id", "trade_id", "signal_id", "trade_date"}),
"positions": frozenset({"run_id", "trade_date", "asset_id", "weight"}),
"nav": frozenset({"run_id", "trade_date", "nav", "pnl_pct"}),
"performance": frozenset({"run_id", "n_trades", "n_days"}),
"attribution": frozenset({"run_id", "trade_date", "asset_id", "asset_total"}),
"attribution_daily": frozenset(
{"run_id", "trade_date", "transaction_cost", "residual", "total_return"}
),
"risk": frozenset(RISK_COLUMNS),
}
def _manifest_fail(
code: BacktestContractErrorCode,
path: str,
detail: str,
) -> Never:
raise BacktestContractError(code, path, detail)
def _manifest_digest(value: object) -> str:
encoded = _canonical_mapping_json({"value": value}).encode("utf-8")
return f"sha256:{hashlib.sha256(encoded).hexdigest()}"
def _manifest_required_digest(value: object, path: str) -> str:
if type(value) is not str:
_manifest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be a string")
if _MANIFEST_DIGEST.fullmatch(value) is None:
_manifest_fail(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be a lowercase sha256 digest",
)
return value
def _manifest_instant(value: object, path: str) -> tuple[str, datetime]:
try:
timestamp = pd.Timestamp(value)
except (TypeError, ValueError) as error:
raise BacktestContractError(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be a valid timestamp",
) from error
if timestamp.tzinfo is None:
_manifest_fail(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must include a timezone",
)
normalized = timestamp.tz_convert("UTC").to_pydatetime()
return normalized.isoformat().replace("+00:00", "Z"), normalized
def _legacy_run_dict(run: BacktestRun) -> dict[str, object]:
created_at, _ = _manifest_instant(run.created_at, "$.legacy_backtest_run.created_at")
return {
"run_id": str(run.run_id),
"dataset_snapshot_id": str(run.dataset_snapshot_id),
"factor_version_id": str(run.factor_version_id),
"strategy_version_id": str(run.strategy_version_id),
"code_revision": str(run.code_revision),
"config_hash": str(run.config_hash),
"created_at": created_at,
}
def _artifact_frames(artifact: ResearchRunArtifact) -> dict[str, pd.DataFrame]:
if not isinstance(artifact, ResearchRunArtifact):
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.artifact",
"ResearchRunArtifact is required",
)
frames: dict[str, pd.DataFrame] = {}
for table_name in _ARTIFACT_TABLE_NAMES:
try:
frame = getattr(artifact, table_name)
except (AttributeError, IndexError, KeyError, TypeError, ValueError) as error:
raise BacktestContractError(
BacktestContractErrorCode.TYPE_ERROR,
f"$.artifact.tables.{table_name}",
"artifact table is unavailable",
) from error
if not isinstance(frame, pd.DataFrame):
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
f"$.artifact.tables.{table_name}",
"artifact table must be a DataFrame",
)
if len(frame.columns) != len({str(column) for column in frame.columns}):
_manifest_fail(
BacktestContractErrorCode.INVALID_VALUE,
f"$.artifact.tables.{table_name}.columns",
"column names must be unique",
)
missing = sorted(_REQUIRED_TABLE_COLUMNS[table_name] - set(frame.columns))
if missing:
_manifest_fail(
BacktestContractErrorCode.MISSING_FIELD,
f"$.artifact.tables.{table_name}.columns.{missing[0]}",
"required column is missing",
)
frames[table_name] = frame
return frames
def _validate_table_run_ids(frames: Mapping[str, pd.DataFrame], run_id: str) -> None:
for table_name, frame in frames.items():
if frame.empty:
continue
values = frame["run_id"].tolist()
if any(type(value) is not str or value != run_id for value in values):
_manifest_fail(
BacktestContractErrorCode.IDENTITY_MISMATCH,
f"$.artifact.tables.{table_name}.run_id",
"table rows do not bind the manifest run",
)
def _table_evidence(
frames: Mapping[str, pd.DataFrame],
expected_table_digests: Mapping[str, str] | None,
) -> dict[str, BacktestEvidenceTable]:
if expected_table_digests is not None and not isinstance(expected_table_digests, Mapping):
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.expected_table_digests",
"must be an object",
)
if expected_table_digests is not None:
for key in expected_table_digests:
if type(key) is not str:
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.expected_table_digests",
"table names must be strings",
)
summaries: dict[str, BacktestEvidenceTable] = {}
for table_name in _ARTIFACT_TABLE_NAMES:
frame = frames[table_name]
columns = [str(column) for column in frame.columns]
schema = {
"columns": columns,
"dtypes": [str(dtype) for dtype in frame.dtypes.tolist()],
}
content = {"columns": columns, "records": _frame_records(frame)}
summary = BacktestEvidenceTable(
logical_name=table_name,
row_count=len(frame),
schema_digest=_manifest_digest(schema),
content_digest=_manifest_digest(content),
)
if expected_table_digests is not None and table_name in expected_table_digests:
expected = _manifest_required_digest(
expected_table_digests[table_name],
f"$.expected_table_digests.{table_name}",
)
if summary.content_digest != expected:
_manifest_fail(
BacktestContractErrorCode.EVIDENCE_MISMATCH,
f"$.artifact.tables.{table_name}.content_digest",
"table content differs from the supplied evidence digest",
)
summaries[table_name] = summary
if expected_table_digests is not None:
unknown = sorted(set(expected_table_digests) - set(summaries))
if unknown:
_manifest_fail(
BacktestContractErrorCode.UNKNOWN_FIELD,
f"$.expected_table_digests.{unknown[0]}",
"unknown artifact table",
)
return summaries
def _evidence_entries(
summaries: Mapping[str, BacktestEvidenceTable],
run_reference: Mapping[str, object],
*,
legacy: bool,
) -> tuple[BacktestEvidenceEntry, ...]:
all_tables = [summaries[name].to_dict() for name in _ARTIFACT_TABLE_NAMES]
entries: list[BacktestEvidenceEntry] = []
for category, table_names in _OFFLINE_RESEARCH_V1:
tables = tuple(summaries[name] for name in table_names)
digest_payload: object = (
{"run_reference": dict(run_reference), "tables": all_tables}
if category == "replay"
else {"category": category, "tables": [table.to_dict() for table in tables]}
)
entries.append(
BacktestEvidenceEntry(
category=category,
tables=tables,
evidence_digest=_manifest_digest(digest_payload),
reconciliation=(
"legacy_incomplete"
if legacy and category == "replay"
else "legacy_source_mapped"
if legacy
else "closed"
),
)
)
return tuple(entries)
def _run_row(frames: Mapping[str, pd.DataFrame]) -> pd.Series[Any]:
run = frames["run"]
if len(run) != 1:
_manifest_fail(
BacktestContractErrorCode.EVIDENCE_MISMATCH,
"$.artifact.tables.run.row_count",
"run table must contain exactly one row",
)
return run.iloc[0]
def _validate_qualified_run(
backtest_run_ref: BacktestRunRef,
frames: Mapping[str, pd.DataFrame],
) -> None:
row = _run_row(frames)
expected = {
"run_id": backtest_run_ref.run_id,
"data_snapshot_id": backtest_run_ref.dataset_snapshot_id,
"strategy_id": backtest_run_ref.strategy_id,
"strategy_version": backtest_run_ref.strategy_version,
"code_revision": backtest_run_ref.code_revision,
"config_hash": backtest_run_ref.configuration_digest.removeprefix("sha256:"),
}
for column, expected_value in expected.items():
if type(row[column]) is not str or row[column] != expected_value:
_manifest_fail(
BacktestContractErrorCode.IDENTITY_MISMATCH,
f"$.artifact.tables.run.{column}",
"run identity differs from BacktestRunRef",
)
def _validate_legacy_run(
legacy_run: BacktestRun,
frames: Mapping[str, pd.DataFrame],
) -> None:
row = _run_row(frames)
expected = {
"run_id": legacy_run.run_id,
"data_snapshot_id": legacy_run.dataset_snapshot_id,
"strategy_id": legacy_run.strategy_version_id.rsplit("@", maxsplit=1)[0],
"strategy_version": legacy_run.strategy_version_id.rsplit("@", maxsplit=1)[-1],
"code_revision": legacy_run.code_revision,
"config_hash": legacy_run.config_hash,
}
for column, expected_value in expected.items():
if str(row[column]) != str(expected_value):
_manifest_fail(
BacktestContractErrorCode.IDENTITY_MISMATCH,
f"$.artifact.tables.run.{column}",
"artifact differs from explicit legacy run",
)
def _build_evidence_manifest(
*,
run_id: str,
run_reference: dict[str, object],
artifact_available_at: object,
qualification: EvidenceQualification,
summaries: Mapping[str, BacktestEvidenceTable],
backtest_run_ref: BacktestRunRef | None,
legacy_backtest_run: BacktestRun | None,
) -> BacktestEvidenceManifest:
available_text, _ = _manifest_instant(
artifact_available_at,
"$.artifact_available_at",
)
legacy = legacy_backtest_run is not None
evidence = _evidence_entries(summaries, run_reference, legacy=legacy)
evidence_digest = _manifest_digest([entry.to_dict() for entry in evidence])
payload: dict[str, object] = {
"contract_name": "researchhub.backtest-evidence-manifest",
"schema_version": BACKTEST_EVIDENCE_SCHEMA_VERSION,
"run_id": run_id,
"profile": "offline_research_v1",
"artifact_available_at": available_text,
"qualification": qualification.value,
"run_reference": run_reference,
"evidence_digest": evidence_digest,
"evidence": [entry.to_dict() for entry in evidence],
}
manifest_id = (
"rhbacktestevidencev1:sha256:"
f"{hashlib.sha256(_canonical_mapping_json(payload).encode('utf-8')).hexdigest()}"
)
instance = object.__new__(BacktestEvidenceManifest)
values: dict[str, object] = {
**payload,
"manifest_id": manifest_id,
"qualification": qualification,
"evidence": evidence,
"backtest_run_ref": backtest_run_ref,
"legacy_backtest_run": legacy_backtest_run,
}
values.pop("run_reference")
for name, value in values.items():
object.__setattr__(instance, name, value)
return instance
def build_backtest_evidence_manifest(
backtest_run_ref: BacktestRunRef,
artifact: ResearchRunArtifact,
*,
artifact_available_at: str | pd.Timestamp | datetime,
qualification: EvidenceQualification = EvidenceQualification.CONTRACT_QUALIFIED,
expected_table_digests: Mapping[str, str] | None = None,
) -> BacktestEvidenceManifest:
"""Close the fixed offline profile without recomputing artifact business facts."""
if not isinstance(backtest_run_ref, BacktestRunRef):
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.backtest_run_ref",
"complete BacktestRunRef is required",
)
if not isinstance(qualification, EvidenceQualification):
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.qualification",
"EvidenceQualification is required",
)
if qualification is EvidenceQualification.LEGACY_EXPLORATORY:
_manifest_fail(
BacktestContractErrorCode.READINESS_ESCALATION,
"$.qualification",
"legacy qualification requires the explicit legacy bridge",
)
available_text, available_time = _manifest_instant(
artifact_available_at,
"$.artifact_available_at",
)
_, computed_time = _manifest_instant(
backtest_run_ref.computed_at,
"$.backtest_run_ref.computed_at",
)
if available_time < computed_time:
_manifest_fail(
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
"$.artifact_available_at",
"artifact cannot be available before the run computation",
)
frames = _artifact_frames(artifact)
_validate_table_run_ids(frames, backtest_run_ref.run_id)
_validate_qualified_run(backtest_run_ref, frames)
_, finished_time = _manifest_instant(
_run_row(frames)["finished_at"],
"$.artifact.tables.run.finished_at",
)
if available_time < finished_time:
_manifest_fail(
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
"$.artifact_available_at",
"artifact cannot be available before the artifact run finished",
)
summaries = _table_evidence(frames, expected_table_digests)
run_reference: dict[str, object] = {
"kind": "backtest_run_ref",
"value": backtest_run_ref.to_dict(),
}
return _build_evidence_manifest(
run_id=backtest_run_ref.run_id,
run_reference=run_reference,
artifact_available_at=available_text,
qualification=qualification,
summaries=summaries,
backtest_run_ref=backtest_run_ref,
legacy_backtest_run=None,
)
def build_legacy_backtest_evidence_manifest(
legacy_run: BacktestRun,
artifact: ResearchRunArtifact,
*,
artifact_available_at: str | pd.Timestamp | datetime,
) -> BacktestEvidenceManifest:
"""Explicitly map an old run to non-qualified, lossy exploratory evidence."""
if not isinstance(legacy_run, BacktestRun):
_manifest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.legacy_backtest_run",
"BacktestRun is required",
)
available_text, available_time = _manifest_instant(
artifact_available_at,
"$.artifact_available_at",
)
_, created_time = _manifest_instant(
legacy_run.created_at,
"$.legacy_backtest_run.created_at",
)
if available_time < created_time:
_manifest_fail(
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
"$.artifact_available_at",
"legacy evidence cannot predate its run",
)
frames = _artifact_frames(artifact)
_validate_table_run_ids(frames, legacy_run.run_id)
_validate_legacy_run(legacy_run, frames)
_, finished_time = _manifest_instant(
_run_row(frames)["finished_at"],
"$.artifact.tables.run.finished_at",
)
if available_time < finished_time:
_manifest_fail(
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
"$.artifact_available_at",
"legacy evidence cannot predate artifact completion",
)
summaries = _table_evidence(frames, None)
run_reference: dict[str, object] = {
"kind": "legacy_backtest_run",
"value": _legacy_run_dict(legacy_run),
}
return _build_evidence_manifest(
run_id=legacy_run.run_id,
run_reference=run_reference,
artifact_available_at=available_text,
qualification=EvidenceQualification.LEGACY_EXPLORATORY,
summaries=summaries,
backtest_run_ref=None,
legacy_backtest_run=legacy_run,
)
def _build_nav( def _build_nav(
result: FactorBacktestResult, result: FactorBacktestResult,
run_id: str, run_id: str,
@@ -511,7 +1173,7 @@ def build_research_run_artifact(
raise TypeError("result must be a FactorBacktestResult") raise TypeError("result must be a FactorBacktestResult")
if result.nav.empty: if result.nav.empty:
raise ValueError("result must contain at least one research session") raise ValueError("result must contain at least one research session")
normalized_run_id = _required_text(run_id, "run_id", max_length=64) normalized_run_id = _required_text(run_id, "run_id", max_length=128)
normalized_strategy_id = _required_text(strategy_id, "strategy_id") normalized_strategy_id = _required_text(strategy_id, "strategy_id")
normalized_strategy_name = _required_text(strategy_name, "strategy_name") normalized_strategy_name = _required_text(strategy_name, "strategy_name")
normalized_strategy_version = _required_text(strategy_version, "strategy_version") normalized_strategy_version = _required_text(strategy_version, "strategy_version")
+637 -1
View File
@@ -11,27 +11,35 @@ import hashlib
import json import json
import math import math
import re import re
from collections.abc import Mapping from collections.abc import Mapping, Sequence
from dataclasses import asdict, dataclass from dataclasses import asdict, dataclass
from datetime import UTC, datetime from datetime import UTC, datetime
from enum import StrEnum from enum import StrEnum
from types import MappingProxyType from types import MappingProxyType
from typing import Any, Never, Self
import pandas as pd import pandas as pd
from quant_engine.execution import ExecutionConfig from quant_engine.execution import ExecutionConfig
from quant_engine.factor_contracts import ( from quant_engine.factor_contracts import (
ContractErrorCode, ContractErrorCode,
DataFoundationEnvelope,
DatasetSnapshotEnvelope,
FactorContractError, FactorContractError,
FactorDefinition, FactorDefinition,
FactorSetRef,
LegacyFactorBinding, LegacyFactorBinding,
canonical_json,
canonical_json_bytes,
) )
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
_SHA256 = re.compile(r"^[0-9a-f]{64}$") _SHA256 = re.compile(r"^[0-9a-f]{64}$")
_PREFIXED_SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$")
_GIT_SHA = re.compile(r"^[0-9a-f]{40}$") _GIT_SHA = re.compile(r"^[0-9a-f]{40}$")
__all__ = [ __all__ = [
"BACKTEST_RUN_REF_SCHEMA_VERSION",
"DatasetSnapshot", "DatasetSnapshot",
"FactorVersion", "FactorVersion",
"bind_legacy_factor", "bind_legacy_factor",
@@ -39,6 +47,9 @@ __all__ = [
"StrategyStage", "StrategyStage",
"StrategyVersion", "StrategyVersion",
"BacktestRun", "BacktestRun",
"BacktestContractErrorCode",
"BacktestContractError",
"BacktestRunRef",
"PortfolioTarget", "PortfolioTarget",
"RiskPolicy", "RiskPolicy",
"RiskDecisionStatus", "RiskDecisionStatus",
@@ -51,6 +62,631 @@ __all__ = [
] ]
BACKTEST_RUN_REF_SCHEMA_VERSION = "1.0.0"
class BacktestContractErrorCode(StrEnum):
"""Stable rejection categories shared by the S3 public contracts."""
TYPE_ERROR = "type_error"
MISSING_FIELD = "missing_field"
UNKNOWN_FIELD = "unknown_field"
INVALID_FORMAT = "invalid_format"
INVALID_VALUE = "invalid_value"
IDENTITY_MISMATCH = "identity_mismatch"
QUALIFICATION_REJECTED = "qualification_rejected"
TIME_ORDER_VIOLATION = "time_order_violation"
INPUT_CLOSURE_VIOLATION = "input_closure_violation"
LINEAGE_VIOLATION = "lineage_violation"
EVIDENCE_MISMATCH = "evidence_mismatch"
READINESS_ESCALATION = "readiness_escalation"
class BacktestContractError(ValueError):
"""Typed deterministic contract rejection with an exact JSON path."""
def __init__(self, code: BacktestContractErrorCode, path: str, detail: str) -> None:
self.code = code
self.path = path
self.detail = detail
super().__init__(f"{code.value} at {path}: {detail}")
def _backtest_fail(
code: BacktestContractErrorCode,
path: str,
detail: str,
) -> Never:
raise BacktestContractError(code, path, detail)
def _backtest_object(value: Any, path: str, fields: Sequence[str]) -> dict[str, Any]:
if type(value) is not dict:
_backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be an object")
required = set(fields)
missing = sorted(required - set(value))
if missing:
_backtest_fail(
BacktestContractErrorCode.MISSING_FIELD,
f"{path}.{missing[0]}",
"field is required",
)
unknown = sorted(set(value) - required)
if unknown:
_backtest_fail(
BacktestContractErrorCode.UNKNOWN_FIELD,
f"{path}.{unknown[0]}",
"field is not permitted",
)
return value
def _backtest_text(value: Any, path: str) -> str:
if type(value) is not str:
_backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be a string")
if not value or value != value.strip():
_backtest_fail(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be non-empty canonical text",
)
return value
def _backtest_digest(value: Any, path: str) -> str:
digest = _backtest_text(value, path)
if _PREFIXED_SHA256.fullmatch(digest) is None:
_backtest_fail(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be a lowercase sha256 digest",
)
return digest
def _backtest_git_revision(value: Any, path: str) -> str:
revision = _backtest_text(value, path)
if _GIT_SHA.fullmatch(revision) is None:
_backtest_fail(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be a lowercase 40-character Git commit",
)
return revision
def _backtest_instant(value: Any, path: str) -> tuple[str, datetime]:
text = _backtest_text(value, path)
if not text.endswith("Z"):
_backtest_fail(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be a canonical UTC instant",
)
try:
parsed = datetime.fromisoformat(text.removesuffix("Z") + "+00:00")
except ValueError as error:
raise BacktestContractError(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be a canonical UTC instant",
) from error
normalized = parsed.astimezone(UTC).isoformat().replace("+00:00", "Z")
if normalized != text:
_backtest_fail(
BacktestContractErrorCode.INVALID_FORMAT,
path,
"must be a canonical UTC instant",
)
return text, parsed
def _backtest_integer(value: Any, path: str, *, minimum: int = 0) -> int:
if type(value) is not int:
_backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be an integer")
if value < minimum:
_backtest_fail(
BacktestContractErrorCode.INVALID_VALUE,
path,
f"must be at least {minimum}",
)
return value
def _backtest_string_tuple(value: Any, path: str) -> tuple[str, ...]:
if type(value) not in {tuple, list}:
_backtest_fail(BacktestContractErrorCode.TYPE_ERROR, path, "must be an array")
normalized = tuple(
_backtest_text(item, f"{path}[{index}]") for index, item in enumerate(value)
)
if len(normalized) != len(set(normalized)):
_backtest_fail(
BacktestContractErrorCode.INVALID_VALUE,
path,
"items must be unique",
)
return tuple(sorted(normalized))
def _content_address_digest(value: str, prefix: str, path: str) -> str:
if not value.startswith(prefix) or _SHA256.fullmatch(value.removeprefix(prefix)) is None:
_backtest_fail(
BacktestContractErrorCode.IDENTITY_MISMATCH,
path,
"authority identity is not content addressed",
)
return f"sha256:{value.removeprefix(prefix)}"
def _digest_document(value: object) -> str:
return f"sha256:{hashlib.sha256(canonical_json_bytes(value)).hexdigest()}"
def _selected_revision_closure(
foundation: DataFoundationEnvelope,
factor_set: FactorSetRef,
field: str,
) -> tuple[str, ...]:
payload = foundation.to_dict()
raw_views = payload.get("standardized_views")
if type(raw_views) is not list:
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.foundation.standardized_views",
"validated Foundation views are required",
)
selected = set(factor_set.selected_view_ref_ids)
found: set[str] = set()
seen_views: set[str] = set()
for index, raw_view in enumerate(raw_views):
if type(raw_view) is not dict:
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
f"$.foundation.standardized_views[{index}]",
"validated view object is required",
)
view_id = raw_view.get("view_ref_id")
if view_id not in selected:
continue
seen_views.add(str(view_id))
revisions = raw_view.get(field)
path = f"$.foundation.standardized_views[{index}].{field}"
if type(revisions) is not list:
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
path,
"revision closure is required",
)
for revision_index, revision_id in enumerate(revisions):
found.add(_backtest_text(revision_id, f"{path}[{revision_index}]"))
if seen_views != selected:
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.factor_set.selected_view_ref_ids",
"selected FactorSet views are not Foundation members",
)
return tuple(sorted(found))
@dataclass(frozen=True, slots=True, init=False)
class BacktestRunRef:
"""Immutable input/configuration/replay identity determined before output exists."""
contract_name: str
schema_version: str
run_id: str
dataset_snapshot_id: str
dataset_content_digest: str
dataset_manifest_digest: str
foundation_id: str
foundation_digest: str
factor_set_id: str
factor_set_digest: str
factor_output_content_digest: str
universe_digest: str
trading_calendar_revision_ids: tuple[str, ...]
trading_calendar_digest: str
corporate_action_revision_ids: tuple[str, ...]
corporate_action_digest: str
strategy_id: str
strategy_version: str
strategy_digest: str
execution_model_version: str
execution_model_digest: str
cost_model_version: str
cost_model_digest: str
random_seed: int
code_revision: str
environment_lock_digest: str
configuration_digest: str
evaluation_at: str
computed_at: str
replay_spec_digest: str
replay_parent_run_id: str | None
replay_reason: str | None
replay_attempt: int
replay_ancestor_run_ids: tuple[str, ...]
@classmethod
def create(
cls,
*,
dataset_snapshot: DatasetSnapshotEnvelope,
foundation: DataFoundationEnvelope,
factor_set: FactorSetRef,
universe_digest: str,
trading_calendar_revision_ids: Sequence[str],
corporate_action_revision_ids: Sequence[str],
strategy_id: str,
strategy_version: str,
strategy_digest: str,
execution_model_version: str,
execution_model_digest: str,
cost_model_version: str,
cost_model_digest: str,
random_seed: int,
code_revision: str,
environment_lock_digest: str,
configuration_digest: str,
evaluation_at: str,
computed_at: str,
parent: BacktestRunRef | None = None,
replay_reason: str | None = None,
replay_attempt: int = 0,
) -> Self:
return cls._build(
dataset_snapshot=dataset_snapshot,
foundation=foundation,
factor_set=factor_set,
universe_digest=universe_digest,
trading_calendar_revision_ids=trading_calendar_revision_ids,
corporate_action_revision_ids=corporate_action_revision_ids,
strategy_id=strategy_id,
strategy_version=strategy_version,
strategy_digest=strategy_digest,
execution_model_version=execution_model_version,
execution_model_digest=execution_model_digest,
cost_model_version=cost_model_version,
cost_model_digest=cost_model_digest,
random_seed=random_seed,
code_revision=code_revision,
environment_lock_digest=environment_lock_digest,
configuration_digest=configuration_digest,
evaluation_at=evaluation_at,
computed_at=computed_at,
parent=parent,
replay_reason=replay_reason,
replay_attempt=replay_attempt,
)
@classmethod
def _build(
cls,
*,
dataset_snapshot: Any,
foundation: Any,
factor_set: Any,
universe_digest: Any,
trading_calendar_revision_ids: Any,
corporate_action_revision_ids: Any,
strategy_id: Any,
strategy_version: Any,
strategy_digest: Any,
execution_model_version: Any,
execution_model_digest: Any,
cost_model_version: Any,
cost_model_digest: Any,
random_seed: Any,
code_revision: Any,
environment_lock_digest: Any,
configuration_digest: Any,
evaluation_at: Any,
computed_at: Any,
parent: BacktestRunRef | None,
replay_reason: Any,
replay_attempt: Any,
) -> Self:
if not isinstance(dataset_snapshot, DatasetSnapshotEnvelope):
_backtest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.dataset_snapshot",
"complete validated DatasetSnapshotEnvelope required",
)
if not isinstance(foundation, DataFoundationEnvelope):
_backtest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.foundation",
"complete validated DataFoundationEnvelope required",
)
if not isinstance(factor_set, FactorSetRef):
_backtest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.factor_set",
"complete validated FactorSetRef required",
)
if foundation.dataset_snapshot_id != dataset_snapshot.snapshot_id:
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.foundation.dataset_snapshot_id",
"Foundation snapshot mismatch",
)
if (
factor_set.dataset_snapshot_id != dataset_snapshot.snapshot_id
or factor_set.foundation_id != foundation.foundation_id
or factor_set.upstream_evidence.snapshot_content_digest
!= dataset_snapshot.content_digest
or factor_set.upstream_evidence.snapshot_manifest_digest
!= dataset_snapshot.manifest_digest
):
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.factor_set",
"FactorSet upstream authority mismatch",
)
expected_calendars = _selected_revision_closure(
foundation,
factor_set,
"trading_calendar_revision_ids",
)
supplied_calendars = _backtest_string_tuple(
trading_calendar_revision_ids,
"$.trading_calendar_revision_ids",
)
if not expected_calendars or supplied_calendars != expected_calendars:
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.trading_calendar_revision_ids",
"must exactly match selected FactorSet calendar ancestry",
)
expected_actions = _selected_revision_closure(
foundation,
factor_set,
"corporate_action_revision_ids",
)
supplied_actions = _backtest_string_tuple(
corporate_action_revision_ids,
"$.corporate_action_revision_ids",
)
if supplied_actions != expected_actions:
_backtest_fail(
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.corporate_action_revision_ids",
"must exactly match selected FactorSet corporate-action ancestry",
)
normalized_evaluation, evaluation_time = _backtest_instant(
evaluation_at,
"$.evaluation_at",
)
normalized_computed, computed_time = _backtest_instant(computed_at, "$.computed_at")
_, factor_available = _backtest_instant(
factor_set.artifact_available_at,
"$.factor_set.artifact_available_at",
)
if computed_time < evaluation_time or computed_time < factor_available:
_backtest_fail(
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
"$.computed_at",
"FactorSet availability and evaluation must not follow computation",
)
normalized_seed = _backtest_integer(random_seed, "$.random_seed")
replay_count = _backtest_integer(replay_attempt, "$.replay_attempt")
spec_payload: dict[str, object] = {
"dataset_snapshot_id": dataset_snapshot.snapshot_id,
"dataset_content_digest": dataset_snapshot.content_digest,
"dataset_manifest_digest": dataset_snapshot.manifest_digest,
"foundation_id": foundation.foundation_id,
"foundation_digest": _content_address_digest(
foundation.foundation_id,
"rhdfv1:sha256:",
"$.foundation.foundation_id",
),
"factor_set_id": factor_set.factor_set_id,
"factor_set_digest": _content_address_digest(
factor_set.factor_set_id,
"rhfactorsetv1:sha256:",
"$.factor_set.factor_set_id",
),
"factor_output_content_digest": factor_set.output_content_digest,
"universe_digest": _backtest_digest(universe_digest, "$.universe_digest"),
"trading_calendar_revision_ids": list(supplied_calendars),
"trading_calendar_digest": _digest_document(list(supplied_calendars)),
"corporate_action_revision_ids": list(supplied_actions),
"corporate_action_digest": _digest_document(list(supplied_actions)),
"strategy_id": _backtest_text(strategy_id, "$.strategy_id"),
"strategy_version": _backtest_text(strategy_version, "$.strategy_version"),
"strategy_digest": _backtest_digest(strategy_digest, "$.strategy_digest"),
"execution_model_version": _backtest_text(
execution_model_version,
"$.execution_model_version",
),
"execution_model_digest": _backtest_digest(
execution_model_digest,
"$.execution_model_digest",
),
"cost_model_version": _backtest_text(cost_model_version, "$.cost_model_version"),
"cost_model_digest": _backtest_digest(
cost_model_digest,
"$.cost_model_digest",
),
"random_seed": normalized_seed,
"code_revision": _backtest_git_revision(code_revision, "$.code_revision"),
"environment_lock_digest": _backtest_digest(
environment_lock_digest,
"$.environment_lock_digest",
),
"configuration_digest": _backtest_digest(
configuration_digest,
"$.configuration_digest",
),
"evaluation_at": normalized_evaluation,
}
replay_spec_digest = _digest_document(spec_payload)
if parent is None:
if replay_reason is not None:
_backtest_fail(
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_reason",
"root run cannot declare a replay reason",
)
if replay_count != 0:
_backtest_fail(
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_attempt",
"root run must use attempt zero",
)
normalized_reason = None
parent_run_id = None
ancestors: tuple[str, ...] = ()
else:
if not isinstance(parent, BacktestRunRef):
_backtest_fail(
BacktestContractErrorCode.TYPE_ERROR,
"$.parent",
"BacktestRunRef parent is required",
)
normalized_reason = _backtest_text(replay_reason, "$.replay_reason")
if replay_count != parent.replay_attempt + 1:
_backtest_fail(
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_attempt",
"replay attempt must follow its parent",
)
if replay_spec_digest != parent.replay_spec_digest:
_backtest_fail(
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_spec_digest",
"replay cannot claim changed deterministic inputs",
)
_, parent_computed = _backtest_instant(parent.computed_at, "$.parent.computed_at")
if computed_time <= parent_computed:
_backtest_fail(
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
"$.computed_at",
"replay computation must follow its parent",
)
parent_run_id = parent.run_id
ancestors = (*parent.replay_ancestor_run_ids, parent.run_id)
if len(ancestors) != len(set(ancestors)):
_backtest_fail(
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_ancestor_run_ids",
"replay lineage contains a cycle",
)
payload: dict[str, object] = {
"contract_name": "researchhub.backtest-run-ref",
"schema_version": BACKTEST_RUN_REF_SCHEMA_VERSION,
**spec_payload,
"computed_at": normalized_computed,
"replay_spec_digest": replay_spec_digest,
"replay_parent_run_id": parent_run_id,
"replay_reason": normalized_reason,
"replay_attempt": replay_count,
"replay_ancestor_run_ids": list(ancestors),
}
run_id = f"rhbacktestrunv1:sha256:{hashlib.sha256(canonical_json_bytes(payload)).hexdigest()}"
values: dict[str, object] = {
**payload,
"run_id": run_id,
"trading_calendar_revision_ids": supplied_calendars,
"corporate_action_revision_ids": supplied_actions,
"replay_ancestor_run_ids": ancestors,
}
instance = object.__new__(cls)
for name, value in values.items():
object.__setattr__(instance, name, value)
return instance
def to_dict(self) -> dict[str, Any]:
return {
"contract_name": self.contract_name,
"schema_version": self.schema_version,
"run_id": self.run_id,
"dataset_snapshot_id": self.dataset_snapshot_id,
"dataset_content_digest": self.dataset_content_digest,
"dataset_manifest_digest": self.dataset_manifest_digest,
"foundation_id": self.foundation_id,
"foundation_digest": self.foundation_digest,
"factor_set_id": self.factor_set_id,
"factor_set_digest": self.factor_set_digest,
"factor_output_content_digest": self.factor_output_content_digest,
"universe_digest": self.universe_digest,
"trading_calendar_revision_ids": list(self.trading_calendar_revision_ids),
"trading_calendar_digest": self.trading_calendar_digest,
"corporate_action_revision_ids": list(self.corporate_action_revision_ids),
"corporate_action_digest": self.corporate_action_digest,
"strategy_id": self.strategy_id,
"strategy_version": self.strategy_version,
"strategy_digest": self.strategy_digest,
"execution_model_version": self.execution_model_version,
"execution_model_digest": self.execution_model_digest,
"cost_model_version": self.cost_model_version,
"cost_model_digest": self.cost_model_digest,
"random_seed": self.random_seed,
"code_revision": self.code_revision,
"environment_lock_digest": self.environment_lock_digest,
"configuration_digest": self.configuration_digest,
"evaluation_at": self.evaluation_at,
"computed_at": self.computed_at,
"replay_spec_digest": self.replay_spec_digest,
"replay_parent_run_id": self.replay_parent_run_id,
"replay_reason": self.replay_reason,
"replay_attempt": self.replay_attempt,
"replay_ancestor_run_ids": list(self.replay_ancestor_run_ids),
}
def to_json(self) -> str:
return canonical_json(self.to_dict())
@classmethod
def from_dict(
cls,
value: Any,
*,
dataset_snapshot: DatasetSnapshotEnvelope,
foundation: DataFoundationEnvelope,
factor_set: FactorSetRef,
parent: BacktestRunRef | None = None,
) -> Self:
field_names = tuple(cls.__dataclass_fields__)
item = _backtest_object(value, "$", field_names)
rebuilt = cls._build(
dataset_snapshot=dataset_snapshot,
foundation=foundation,
factor_set=factor_set,
universe_digest=item["universe_digest"],
trading_calendar_revision_ids=item["trading_calendar_revision_ids"],
corporate_action_revision_ids=item["corporate_action_revision_ids"],
strategy_id=item["strategy_id"],
strategy_version=item["strategy_version"],
strategy_digest=item["strategy_digest"],
execution_model_version=item["execution_model_version"],
execution_model_digest=item["execution_model_digest"],
cost_model_version=item["cost_model_version"],
cost_model_digest=item["cost_model_digest"],
random_seed=item["random_seed"],
code_revision=item["code_revision"],
environment_lock_digest=item["environment_lock_digest"],
configuration_digest=item["configuration_digest"],
evaluation_at=item["evaluation_at"],
computed_at=item["computed_at"],
parent=parent,
replay_reason=item["replay_reason"],
replay_attempt=item["replay_attempt"],
)
expected = rebuilt.to_dict()
for name in field_names:
if item[name] != expected[name]:
_backtest_fail(
BacktestContractErrorCode.IDENTITY_MISMATCH,
f"$.{name}",
"serialized run reference does not match admitted authorities",
)
return rebuilt
def _required_text(value: str, name: str) -> str: def _required_text(value: str, name: str) -> str:
normalized = value.strip() normalized = value.strip()
if not normalized: if not normalized:
+17
View File
@@ -0,0 +1,17 @@
{
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"replay_spec_digest": "sha256:20f07fcb526bc38b4ba3d71d6c3b00a8c63ad96cab0fecd869326503ab42d98b",
"manifest_id": "rhbacktestevidencev1:sha256:47e86c53e696672a8931e1cb496b6a48344a4ffbf84bbec4818ea2671fee8d42",
"evidence_digest": "sha256:907775054f4316bcb2f559a3068493210cebaa97bb90e462ba39ae0762eea98e",
"table_content_digests": {
"run": "sha256:7d947ec93f714641669cbb14bd69dd8cf30387aabb7f3a086918fc2e878ab05c",
"signals": "sha256:72dc15064cc45d7c51d2dd4b8c3a6d8c7d4155232d5d70d3c9e7697fab70ce50",
"trades": "sha256:2b8b9321e7993941ac486cf50cab5b4c6425b2571a4701f0eb02273ec9a96c53",
"positions": "sha256:b456a48fab51742b05084ca6dcaf01c03dfe5b39ad71215ada070e9b1f59f7ea",
"nav": "sha256:25649efce860b76410f87dbd36c81085887620086dc0b48b07099918f65d9c78",
"performance": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
"attribution": "sha256:ad0f12f668d0a2d9ae5b3636d29989ab4bafcfb95f5de77abeb52a6d9e95d366",
"attribution_daily": "sha256:d4459ad650f88871d7b1e40392033037b5818ef92fdf1ee021868929fc1455a5",
"risk": "sha256:4816dd4812b5ff2e97e74bf34ca2221bfde675b387a96e277683557f0ad7d975"
}
}
+13 -6
View File
@@ -16,19 +16,26 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower() prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
for term in ("investment advice", "live order", "credentials", "source facts"): for term in ("investment advice", "live order", "credentials", "source facts"):
assert term in prohibited assert term in prohibited
assert spec["authority"]["revision"] == 2 assert spec["authority"]["revision"] == 3
assert { assert {
(item["contract_id"], item["version"]) (item["contract_id"], item["version"])
for item in spec["contracts"]["provides"] for item in spec["contracts"]["provides"]
} == { } == {
("researchhub.factor-definition", "1.0.0"), ("researchhub.factor-definition", "1.0.0"),
("researchhub.factor-set-ref", "1.0.0"), ("researchhub.factor-set-ref", "1.0.0"),
("researchhub.backtest-run-ref", "1.0.0"),
("researchhub.backtest-evidence-manifest", "1.0.0"),
} }
assert all( expected_paths = {
item["authority"] == "quant_engine" "researchhub.factor-definition": "src/quant_engine/factor_contracts.py",
and item["path"] == "src/quant_engine/factor_contracts.py" "researchhub.factor-set-ref": "src/quant_engine/factor_contracts.py",
for item in spec["contracts"]["provides"] "researchhub.backtest-run-ref": "src/quant_engine/governed_pipeline.py",
) "researchhub.backtest-evidence-manifest": "src/quant_engine/artifact.py",
}
assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"])
assert {
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
} == expected_paths
assert { assert {
(item["contract_id"], item["version"]) (item["contract_id"], item["version"])
for item in spec["contracts"]["consumes"] for item in spec["contracts"]["consumes"]
+533
View File
@@ -0,0 +1,533 @@
"""Backtest run-reference and closed-evidence contract conformance."""
from __future__ import annotations
import copy
import hashlib
import json
from dataclasses import replace
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
import pandas as pd
import pytest
from quant_engine.artifact import (
BacktestEvidenceManifest,
EvidenceQualification,
ResearchRunArtifact,
build_backtest_evidence_manifest,
build_legacy_backtest_evidence_manifest,
build_research_run_artifact,
)
from quant_engine.execution import ExecutionConfig
from quant_engine.factor_contracts import (
ActorIdentity,
AvailabilityMode,
Causation,
DataFoundationEnvelope,
DatasetSnapshotEnvelope,
FactorInput,
FactorSetRef,
InputBinding,
OutputArtifactRef,
OutputCoverage,
OutputQuality,
OutputQualityCheck,
ProducerIdentity,
ViewAvailability,
canonical_json_bytes,
factor_definition_from_alpha158,
factor_input_schema_digest,
)
from quant_engine.governed_pipeline import (
BacktestContractError,
BacktestContractErrorCode,
BacktestRun,
BacktestRunRef,
)
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
ROOT = Path(__file__).resolve().parents[1]
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
BACKTEST_FIXTURE = ROOT / "tests" / "fixtures" / "backtest-evidence-v1.golden.json"
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
def _sha256(value: bytes) -> str:
return f"sha256:{hashlib.sha256(value).hexdigest()}"
def _accepted_authorities() -> tuple[
DatasetSnapshotEnvelope,
DataFoundationEnvelope,
FactorSetRef,
]:
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
definition = factor_definition_from_alpha158(
"alpha_005",
version="1.0.0",
parameters={},
inputs=(factor_input,),
implementation_digest="sha256:" + "1" * 64,
input_schema_digest=factor_input_schema_digest((factor_input,)),
valid_from="2026-01-01T00:00:00.000000Z",
valid_until="2027-01-01T00:00:00Z",
warmup_sessions=10,
lag_sessions=1,
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
)
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
output_content_bytes = canonical_json_bytes(fixture["output_content"])
artifact_ref = OutputArtifactRef.create(
schema_digest=_sha256(output_schema_bytes),
content_digest=_sha256(output_content_bytes),
)
factor_set = FactorSetRef.create(
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
selected_view_ref_ids=(VIEW_REF_ID,),
input_bindings=(
InputBinding(definition.definition_id, "market", VIEW_REF_ID, VIEW_SCHEMA_DIGEST),
),
view_availability=(
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
),
output_quality=OutputQuality(
"passed",
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
),
output_coverage=OutputCoverage(
"complete", 1, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64
),
output_schema_bytes=output_schema_bytes,
output_content_bytes=output_content_bytes,
output_artifact_ref=artifact_ref,
availability_mode=AvailabilityMode.AS_AVAILABLE,
evaluation_at="2026-01-03T11:00:00Z",
computed_at="2026-01-03T10:15:00Z",
artifact_available_at="2026-01-03T10:20:00Z",
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
actor=ActorIdentity("service", "factor_worker_v1"),
correlation_id="research_run_001",
causation=Causation("foundation", foundation.foundation_id),
evidence_scope="synthetic_fixture",
decision_eligible=False,
)
return snapshot, foundation, factor_set
def _config_digest(parameters: dict[str, object] | None = None) -> str:
encoded = json.dumps(
PARAMETERS if parameters is None else parameters,
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
).encode("utf-8")
return _sha256(encoded)
def _run_ref(**overrides: Any) -> BacktestRunRef:
snapshot, foundation, factor_set = _accepted_authorities()
arguments: dict[str, Any] = {
"dataset_snapshot": snapshot,
"foundation": foundation,
"factor_set": factor_set,
"universe_digest": "sha256:" + "5" * 64,
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
"strategy_id": "alpha-top1",
"strategy_version": "1.0.0",
"strategy_digest": "sha256:" + "6" * 64,
"execution_model_version": "1.0.0",
"execution_model_digest": "sha256:" + "7" * 64,
"cost_model_version": "1.0.0",
"cost_model_digest": "sha256:" + "8" * 64,
"random_seed": 7,
"code_revision": "d" * 40,
"environment_lock_digest": "sha256:" + "9" * 64,
"configuration_digest": _config_digest(),
"evaluation_at": "2026-01-08T01:00:00Z",
"computed_at": "2026-01-08T02:00:00Z",
}
arguments.update(overrides)
return BacktestRunRef.create(**arguments)
def _backtest_result() -> FactorBacktestResult:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
opens = pd.DataFrame(
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
index=dates,
)
closes = pd.DataFrame(
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
index=dates,
)
return run_factor_backtest_research(
scores,
opens,
closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
),
)
def _artifact(run_ref: BacktestRunRef, *, run_id: str | None = None) -> ResearchRunArtifact:
result = _backtest_result()
benchmark = pd.Series(
[0.0, 0.01, -0.01, 0.02],
index=result.returns.index,
name="benchmark_return",
)
return build_research_run_artifact(
result,
run_id=run_ref.run_id if run_id is None else run_id,
strategy_id=run_ref.strategy_id,
strategy_name="Alpha Top 1",
strategy_version=run_ref.strategy_version,
engine_version="1.2.0",
code_revision=run_ref.code_revision,
data_snapshot_id=run_ref.dataset_snapshot_id,
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-01-08T10:00:00+08:00",
finished_at="2026-01-08T10:01:00+08:00",
parameters=PARAMETERS,
benchmark_id="000300.SH",
benchmark_returns=benchmark,
)
def _assert_error(
error: pytest.ExceptionInfo[BacktestContractError],
code: BacktestContractErrorCode,
path: str,
) -> None:
assert error.value.code is code
assert error.value.path == path
def test_backtest_run_ref_is_deterministic_and_binds_only_opaque_authorities() -> None:
first = _run_ref()
second = _run_ref()
assert first == second
assert first.run_id.startswith("rhbacktestrunv1:sha256:")
assert first.replay_spec_digest.startswith("sha256:")
assert first.dataset_snapshot_id.startswith("rhdsv1:sha256:")
assert first.foundation_id.startswith("rhdfv1:sha256:")
assert first.factor_set_id.startswith("rhfactorsetv1:sha256:")
assert first.trading_calendar_revision_ids == (CALENDAR_REVISION_ID,)
assert first.corporate_action_revision_ids == (ACTION_REVISION_ID,)
assert first.replay_parent_run_id is None
assert first.replay_attempt == 0
assert first.replay_ancestor_run_ids == ()
snapshot, foundation, factor_set = _accepted_authorities()
assert BacktestRunRef.from_dict(
first.to_dict(),
dataset_snapshot=snapshot,
foundation=foundation,
factor_set=factor_set,
) == first
forbidden = ("latest", "locator", "uri", "credential", "provider", "broker")
assert not any(token in first.to_json().lower() for token in forbidden)
@pytest.mark.parametrize(
("field", "value"),
[
("universe_digest", "sha256:" + "a" * 64),
("strategy_digest", "sha256:" + "b" * 64),
("execution_model_digest", "sha256:" + "c" * 64),
("cost_model_digest", "sha256:" + "e" * 64),
("random_seed", 8),
("code_revision", "e" * 40),
("environment_lock_digest", "sha256:" + "f" * 64),
("configuration_digest", "sha256:" + "0" * 64),
("evaluation_at", "2026-01-08T01:00:01Z"),
("computed_at", "2026-01-08T02:00:01Z"),
],
)
def test_every_governed_run_input_mutation_changes_run_identity(
field: str,
value: object,
) -> None:
assert _run_ref(**{field: value}).run_id != _run_ref().run_id
def test_backtest_run_ref_rejects_unclosed_upstream_and_unsafe_scalars() -> None:
with pytest.raises(BacktestContractError) as wrong_calendar:
_run_ref(trading_calendar_revision_ids=())
_assert_error(
wrong_calendar,
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.trading_calendar_revision_ids",
)
with pytest.raises(BacktestContractError) as wrong_action:
_run_ref(corporate_action_revision_ids=())
_assert_error(
wrong_action,
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.corporate_action_revision_ids",
)
with pytest.raises(BacktestContractError) as bool_seed:
_run_ref(random_seed=True)
_assert_error(bool_seed, BacktestContractErrorCode.TYPE_ERROR, "$.random_seed")
with pytest.raises(BacktestContractError) as bad_revision:
_run_ref(code_revision="abc")
_assert_error(bad_revision, BacktestContractErrorCode.INVALID_FORMAT, "$.code_revision")
with pytest.raises(BacktestContractError) as bad_digest:
_run_ref(universe_digest="5" * 64)
_assert_error(bad_digest, BacktestContractErrorCode.INVALID_FORMAT, "$.universe_digest")
with pytest.raises(BacktestContractError) as lookahead:
_run_ref(computed_at="2026-01-08T00:59:59Z")
_assert_error(lookahead, BacktestContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
with pytest.raises(BacktestContractError) as factor_type:
_run_ref(factor_set="rhfactorsetv1:sha256:" + "0" * 64)
_assert_error(factor_type, BacktestContractErrorCode.TYPE_ERROR, "$.factor_set")
def test_replay_lineage_is_acyclic_and_cannot_claim_changed_inputs() -> None:
parent = _run_ref()
replay = _run_ref(
computed_at="2026-01-08T03:00:00Z",
parent=parent,
replay_reason="deterministic_reproduction",
replay_attempt=1,
)
assert replay.run_id != parent.run_id
assert replay.replay_spec_digest == parent.replay_spec_digest
assert replay.replay_parent_run_id == parent.run_id
assert replay.replay_ancestor_run_ids == (parent.run_id,)
with pytest.raises(BacktestContractError) as changed_input:
_run_ref(
universe_digest="sha256:" + "a" * 64,
computed_at="2026-01-08T03:00:00Z",
parent=parent,
replay_reason="changed_universe",
replay_attempt=1,
)
_assert_error(
changed_input,
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_spec_digest",
)
with pytest.raises(BacktestContractError) as skipped_attempt:
_run_ref(
computed_at="2026-01-08T03:00:00Z",
parent=parent,
replay_reason="skipped_attempt",
replay_attempt=2,
)
_assert_error(
skipped_attempt,
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_attempt",
)
def test_offline_research_manifest_closes_exact_existing_evidence_mapping() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
first = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
)
second = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
)
assert first == second
assert first.manifest_id.startswith("rhbacktestevidencev1:sha256:")
assert first.run_id == run_ref.run_id
assert first.profile == "offline_research_v1"
assert first.qualification is EvidenceQualification.CONTRACT_QUALIFIED
mapping = {
item.category: tuple(table.logical_name for table in item.tables)
for item in first.evidence
}
assert mapping == {
"run": ("run",),
"signal": ("signals",),
"fill": ("trades",),
"position_nav": ("positions", "nav"),
"performance": ("performance",),
"attribution": ("attribution", "attribution_daily"),
"risk_snapshot": ("risk",),
"replay": (),
}
assert "order" not in mapping
assert "rejection" not in mapping
risk = next(item for item in first.evidence if item.category == "risk_snapshot")
assert risk.tables[0].row_count == 0
assert risk.tables[0].schema_digest.startswith("sha256:")
changed_performance = artifact.performance
changed_performance.loc[0, "n_days"] += 1
changed_artifact = replace(artifact, _performance=changed_performance)
changed = build_backtest_evidence_manifest(
run_ref,
changed_artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
assert changed.manifest_id != first.manifest_id
assert run_ref.run_id == first.run_id == changed.run_id
def test_manifest_rejects_missing_mismatched_or_duplicate_evidence() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
with pytest.raises(BacktestContractError) as wrong_run:
build_backtest_evidence_manifest(
run_ref,
_artifact(run_ref, run_id="different-run"),
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
wrong_run,
BacktestContractErrorCode.IDENTITY_MISMATCH,
"$.artifact.tables.run.run_id",
)
missing_signals = replace(artifact, _signals=None) # type: ignore[arg-type]
with pytest.raises(BacktestContractError) as missing_table:
build_backtest_evidence_manifest(
run_ref,
missing_signals,
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
missing_table,
BacktestContractErrorCode.TYPE_ERROR,
"$.artifact.tables.signals",
)
with pytest.raises(BacktestContractError) as digest_mismatch:
build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
expected_table_digests={"performance": "sha256:" + "0" * 64},
)
_assert_error(
digest_mismatch,
BacktestContractErrorCode.EVIDENCE_MISMATCH,
"$.artifact.tables.performance.content_digest",
)
manifest = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
duplicate = manifest.to_dict()
duplicate["evidence"].append(copy.deepcopy(duplicate["evidence"][0]))
with pytest.raises(BacktestContractError) as duplicate_category:
BacktestEvidenceManifest.from_dict(
duplicate,
backtest_run_ref=run_ref,
artifact=artifact,
)
_assert_error(
duplicate_category,
BacktestContractErrorCode.INVALID_VALUE,
"$.evidence[8].category",
)
def test_legacy_bridge_is_explicit_and_cannot_be_contract_qualified() -> None:
run_ref = _run_ref()
legacy_run = BacktestRun(
run_id="legacy-run-001",
dataset_snapshot_id=run_ref.dataset_snapshot_id,
factor_version_id="alpha_005@1.0.0",
strategy_version_id="alpha-top1@1.0.0",
code_revision=run_ref.code_revision,
config_hash=_config_digest().removeprefix("sha256:"),
created_at=datetime(2026, 1, 8, 2, 0, tzinfo=UTC),
)
artifact = _artifact(run_ref, run_id=legacy_run.run_id)
manifest = build_legacy_backtest_evidence_manifest(
legacy_run,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
assert manifest.qualification is EvidenceQualification.LEGACY_EXPLORATORY
assert manifest.run_id == legacy_run.run_id
assert manifest.backtest_run_ref is None
assert manifest.to_dict()["run_reference"]["kind"] == "legacy_backtest_run"
with pytest.raises(BacktestContractError) as implicit_promotion:
build_backtest_evidence_manifest( # type: ignore[arg-type]
legacy_run,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
)
_assert_error(
implicit_promotion,
BacktestContractErrorCode.TYPE_ERROR,
"$.backtest_run_ref",
)
def test_golden_contract_and_architecture_boundary() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
manifest = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
golden = json.loads(BACKTEST_FIXTURE.read_text(encoding="utf-8"))
table_digests = {
table.logical_name: table.content_digest
for item in manifest.evidence
for table in item.tables
}
assert golden == {
"run_id": run_ref.run_id,
"replay_spec_digest": run_ref.replay_spec_digest,
"manifest_id": manifest.manifest_id,
"evidence_digest": manifest.evidence_digest,
"table_content_digests": table_digests,
}
governed_source = (ROOT / "src" / "quant_engine" / "governed_pipeline.py").read_text(
encoding="utf-8"
)
artifact_source = (ROOT / "src" / "quant_engine" / "artifact.py").read_text(
encoding="utf-8"
)
assert "from quant_engine.artifact" not in governed_source
assert "BacktestRunRef" in governed_source
assert "BacktestEvidenceManifest" not in governed_source
assert "BacktestEvidenceManifest" in artifact_source
assert not (ROOT / "src" / "quant_engine" / "backtest_contracts.py").exists()