diff --git a/MODULE_SPEC.yaml b/MODULE_SPEC.yaml index 54b4016..0218936 100644 --- a/MODULE_SPEC.yaml +++ b/MODULE_SPEC.yaml @@ -1,7 +1,7 @@ { "schema_version": 1, "module_id": "quant_engine", - "authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 4, "effective_from": "2026-09-01T00:00:00+08:00"}, + "authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 5, "effective_from": "2026-09-01T00:00:00+08:00"}, "repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"}, "bounded_context": { "domain": "quantitative-research-engine", @@ -19,7 +19,7 @@ {"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"}, {"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"}, {"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"}, - {"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact evidence without persistence or decision authority.", "status": "operational"}, + {"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact and performance-methodology evidence without recomputation, persistence, or decision authority.", "status": "operational"}, {"id": "portfolio-risk-computation-contracts", "summary": "Verify deterministic portfolio-computation receipts and expose S3-bound portfolio decisions and risk assessments without adding algorithms or execution authority.", "status": "operational"}, {"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"} ], @@ -33,6 +33,7 @@ {"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"}, {"contract_id": "researchhub.backtest-run-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/governed_pipeline.py"}, {"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"}, + {"contract_id": "researchhub.performance-evidence", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"}, {"contract_id": "researchhub.portfolio-decision", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"}, {"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"} ], diff --git a/README.md b/README.md index fa98fea..66ca385 100644 --- a/README.md +++ b/README.md @@ -260,6 +260,35 @@ digest 等价,也不会把旧 run 静默升级为新合同。 `LEGACY_EXPLORATORY`;不能隐式提升为 `CONTRACT_QUALIFIED`。所有资格均只描述离线证据闭合, 不表示投资有效、组合获批、Paper、生产或实盘就绪。 +## 绩效证据与方法论合同 v1 + +`quant_engine.artifact.PerformanceEvidenceV1` 在现有计算和事实表之外增加一层只读、内容寻址的 +owner 证据。`build_performance_evidence()` 只接受同一运行的完整 `ResearchRunArtifact`、 +`BacktestRunRef` 与 `CONTRACT_QUALIFIED BacktestEvidenceManifest`;它核对全部 artifact 表、 +performance 表和唯一行摘要,并绑定 artifact、row 与严格对齐 benchmark series 的独立摘要。 +生产 builder 不重算、填补、重命名或覆盖任何绩效值。 + +方法论固定为日简单收益、252 期年化、绝对指标年化无风险利率 `0.0`、benchmark 日无风险利率 +`0.0`,以及 benchmark 存在时的 `exact_session_index`。相对指标使用封闭 availability:无基准为 +`benchmark_absent`;active variance、benchmark variance 或 alpha 几何年化域不足时分别使用 +对应 `not_estimable_*` 原因。benchmark 存在时 tracking error 始终必须是有限非负值;null 不会 +被转成零。 + +```python +from quant_engine.artifact import build_performance_evidence + +performance_evidence = build_performance_evidence( + artifact, + backtest_run_ref, + backtest_evidence_manifest, +) +canonical_bytes = performance_evidence.canonical_bytes() +``` + +该合同范围固定为 `offline_research_only`。它不授予排名、推荐、决策、发布、论文、Paper、生产、 +实盘、交易或投资建议权限,也不包含原始参数、returns、NAV、benchmark series、表字节、存储 +locator、URI 或凭证。 + ## 组合决策与风险评估合同 v1 `quant_engine.portfolio_risk_contracts` 是现有计算 owner 外围的薄合同层。创建 diff --git a/src/quant_engine/artifact.py b/src/quant_engine/artifact.py index 94f1b90..13af3f5 100644 --- a/src/quant_engine/artifact.py +++ b/src/quant_engine/artifact.py @@ -27,11 +27,15 @@ from quant_engine.governed_pipeline import ( BacktestRun, BacktestRunRef, ) +from quant_engine.metrics import TRADING_DAYS_PER_YEAR from quant_engine.research_pipeline import FactorBacktestResult from quant_engine.risk import CovarianceSnapshot, labeled_component_risk RESEARCH_ARTIFACT_SCHEMA_VERSION = "1.1.0" BACKTEST_EVIDENCE_SCHEMA_VERSION = "1.0.0" +PERFORMANCE_EVIDENCE_SCHEMA_VERSION = "researchhub.performance-evidence.v1" +PERFORMANCE_METHODOLOGY_ID = "researchhub.quant-performance-methodology.v1" +PERFORMANCE_METRIC_SCHEMA_ID = "researchhub.quant-performance-metrics.v1" _MAX_SAFE_INTEGER = (1 << 53) - 1 RISK_COLUMNS = [ @@ -52,13 +56,23 @@ RISK_COLUMNS = [ __all__ = [ "RESEARCH_ARTIFACT_SCHEMA_VERSION", "BACKTEST_EVIDENCE_SCHEMA_VERSION", + "PERFORMANCE_EVIDENCE_SCHEMA_VERSION", + "PERFORMANCE_METHODOLOGY_ID", + "PERFORMANCE_METRIC_SCHEMA_ID", "ResearchRunArtifact", "EvidenceQualification", "BacktestEvidenceTable", "BacktestEvidenceEntry", "BacktestEvidenceManifest", + "PerformanceEvidenceErrorCode", + "PerformanceEvidenceError", + "PerformanceMetricAvailability", + "PerformanceMethodology", + "PerformanceMetric", + "PerformanceEvidenceV1", "build_backtest_evidence_manifest", "build_legacy_backtest_evidence_manifest", + "build_performance_evidence", "build_research_run_artifact", ] @@ -413,6 +427,258 @@ class BacktestEvidenceManifest: return cast(Self, rebuilt) +class PerformanceEvidenceErrorCode(StrEnum): + """Stable fail-closed categories for the performance-evidence contract.""" + + TYPE_ERROR = "type_error" + UNSUPPORTED_VERSION = "unsupported_version" + IDENTITY_MISMATCH = "identity_mismatch" + EVIDENCE_MISMATCH = "evidence_mismatch" + METHODOLOGY_MISMATCH = "methodology_mismatch" + METRIC_INVALID = "metric_invalid" + BENCHMARK_INVALID = "benchmark_invalid" + AUTHORITY_REJECTED = "authority_rejected" + + +class PerformanceEvidenceError(ValueError): + """Typed deterministic rejection without native DataFrame error leakage.""" + + def __init__( + self, + code: PerformanceEvidenceErrorCode, + path: str, + detail: str, + ) -> None: + self.code = code + self.path = path + self.detail = detail + super().__init__(f"{code.value} at {path}: {detail}") + + +class PerformanceMetricAvailability(StrEnum): + """Closed availability reasons for absolute and benchmark-relative metrics.""" + + AVAILABLE = "available" + BENCHMARK_ABSENT = "benchmark_absent" + NOT_ESTIMABLE_ACTIVE_VARIANCE = "not_estimable_active_variance" + NOT_ESTIMABLE_BENCHMARK_VARIANCE = "not_estimable_benchmark_variance" + NOT_ESTIMABLE_ALPHA_DOMAIN = "not_estimable_alpha_domain" + + +@dataclass(frozen=True, slots=True) +class PerformanceMethodology: + """Exact existing quant-engine methodology, versioned but never caller-extensible.""" + + methodology_id: str + return_type: str + source_frequency: str + periods_per_year: int + annualized_return: str + annualized_volatility: str + sharpe_ratio: str + sortino_ratio: str + annual_risk_free: float + tracking_error: str + information_ratio: str + alpha: str + beta: str + benchmark_risk_free_daily: float + benchmark_alignment: str + maximum_drawdown: str + calmar_ratio: str + win_rate: str + total_return: str + implementation_module: str + implementation_version: str + code_revision: str + + def to_dict(self) -> dict[str, object]: + return { + "methodology_id": self.methodology_id, + "return_type": self.return_type, + "source_frequency": self.source_frequency, + "periods_per_year": self.periods_per_year, + "annualized_return": self.annualized_return, + "annualized_volatility": self.annualized_volatility, + "sharpe_ratio": self.sharpe_ratio, + "sortino_ratio": self.sortino_ratio, + "annual_risk_free": self.annual_risk_free, + "tracking_error": self.tracking_error, + "information_ratio": self.information_ratio, + "alpha": self.alpha, + "beta": self.beta, + "benchmark_risk_free_daily": self.benchmark_risk_free_daily, + "benchmark_alignment": self.benchmark_alignment, + "maximum_drawdown": self.maximum_drawdown, + "calmar_ratio": self.calmar_ratio, + "win_rate": self.win_rate, + "total_return": self.total_return, + "implementation_module": self.implementation_module, + "implementation_version": self.implementation_version, + "code_revision": self.code_revision, + } + + +@dataclass(frozen=True, slots=True) +class PerformanceMetric: + """One fixed owner metric with closed source, domain, and availability semantics.""" + + key: str + source_column: str + value: float | int | None + unit: str + nullable: bool + availability: PerformanceMetricAvailability + metric_schema_id: str + methodology_id: str + + def to_dict(self) -> dict[str, object]: + return { + "key": self.key, + "source_column": self.source_column, + "value": self.value, + "unit": self.unit, + "nullable": self.nullable, + "availability": self.availability.value, + "metric_schema_id": self.metric_schema_id, + "methodology_id": self.methodology_id, + } + + +@dataclass(frozen=True, slots=True, init=False) +class PerformanceEvidenceV1: + """Closed content-addressed evidence over one existing performance row.""" + + schema_version: str + performance_evidence_id: str + document_sha256: str + authority: str + scope: str + run_id: str + backtest_run_ref_id: str + backtest_run_ref_document_sha256: str + backtest_evidence_manifest_id: str + backtest_evidence_manifest_document_sha256: str + backtest_evidence_manifest_evidence_digest: str + backtest_evidence_qualification: str + research_artifact_schema_version: str + research_artifact_content_digest: str + artifact_available_at: str + performance_table_logical_name: str + performance_table_row_count: int + performance_table_schema_digest: str + performance_table_content_digest: str + performance_row_digest: str + benchmark_series_digest: str | None + methodology_id: str + metric_schema_id: str + dataset_snapshot_id: str + dataset_content_digest: str + dataset_manifest_digest: str + foundation_id: str + foundation_digest: str + factor_set_id: str + factor_set_digest: str + factor_output_content_digest: str + strategy_id: str + strategy_version: str + strategy_digest: str + execution_model_version: str + execution_model_digest: str + cost_model_version: str + cost_model_digest: str + code_revision: str + environment_lock_digest: str + configuration_digest: str + frequency: str + calendar: str + timezone: str + benchmark_id: str + benchmark_alignment_policy: str + start_date: str + end_date: str + methodology: PerformanceMethodology + metrics: tuple[PerformanceMetric, ...] + + def to_dict(self) -> dict[str, Any]: + return { + "schema_version": self.schema_version, + "performance_evidence_id": self.performance_evidence_id, + "document_sha256": self.document_sha256, + "authority": self.authority, + "scope": self.scope, + "run_id": self.run_id, + "backtest_run_ref_id": self.backtest_run_ref_id, + "backtest_run_ref_document_sha256": self.backtest_run_ref_document_sha256, + "backtest_evidence_manifest_id": self.backtest_evidence_manifest_id, + "backtest_evidence_manifest_document_sha256": ( + self.backtest_evidence_manifest_document_sha256 + ), + "backtest_evidence_manifest_evidence_digest": ( + self.backtest_evidence_manifest_evidence_digest + ), + "backtest_evidence_qualification": self.backtest_evidence_qualification, + "research_artifact_schema_version": self.research_artifact_schema_version, + "research_artifact_content_digest": self.research_artifact_content_digest, + "artifact_available_at": self.artifact_available_at, + "performance_table_logical_name": self.performance_table_logical_name, + "performance_table_row_count": self.performance_table_row_count, + "performance_table_schema_digest": self.performance_table_schema_digest, + "performance_table_content_digest": self.performance_table_content_digest, + "performance_row_digest": self.performance_row_digest, + "benchmark_series_digest": self.benchmark_series_digest, + "methodology_id": self.methodology_id, + "metric_schema_id": self.metric_schema_id, + "dataset_snapshot_id": self.dataset_snapshot_id, + "dataset_content_digest": self.dataset_content_digest, + "dataset_manifest_digest": self.dataset_manifest_digest, + "foundation_id": self.foundation_id, + "foundation_digest": self.foundation_digest, + "factor_set_id": self.factor_set_id, + "factor_set_digest": self.factor_set_digest, + "factor_output_content_digest": self.factor_output_content_digest, + "strategy_id": self.strategy_id, + "strategy_version": self.strategy_version, + "strategy_digest": self.strategy_digest, + "execution_model_version": self.execution_model_version, + "execution_model_digest": self.execution_model_digest, + "cost_model_version": self.cost_model_version, + "cost_model_digest": self.cost_model_digest, + "code_revision": self.code_revision, + "environment_lock_digest": self.environment_lock_digest, + "configuration_digest": self.configuration_digest, + "frequency": self.frequency, + "calendar": self.calendar, + "timezone": self.timezone, + "benchmark_id": self.benchmark_id, + "benchmark_alignment_policy": self.benchmark_alignment_policy, + "start_date": self.start_date, + "end_date": self.end_date, + "methodology": self.methodology.to_dict(), + "metrics": [metric.to_dict() for metric in self.metrics], + } + + def canonical_bytes(self) -> bytes: + return _performance_canonical_bytes(self.to_dict()) + + def to_json(self) -> str: + return self.canonical_bytes().decode("utf-8") + + @classmethod + def from_dict( + cls, + value: Any, + *, + artifact: ResearchRunArtifact, + run_ref: BacktestRunRef, + evidence_manifest: BacktestEvidenceManifest, + ) -> Self: + _performance_validate_tree(value, "$") + rebuilt = build_performance_evidence(artifact, run_ref, evidence_manifest) + _performance_compare(value, rebuilt.to_dict(), "$") + return cast(Self, rebuilt) + + def _required_text(value: str, name: str, *, max_length: int | None = None) -> str: normalized = value.strip() if not normalized: @@ -1215,6 +1481,1096 @@ def build_legacy_backtest_evidence_manifest( ) +_PERFORMANCE_SOURCE_COLUMNS = ( + "run_id", + "total_ret", + "ann_ret", + "ann_volatility", + "sharpe", + "sortino", + "max_dd", + "calmar", + "win_rate", + "tracking_error", + "ir", + "alpha", + "beta", + "n_trades", + "n_days", +) +_PERFORMANCE_IDENTITY_PREFIX = "rhperformanceevidencev1:sha256:" +_BACKTEST_RUN_ID_PREFIX = "rhbacktestrunv1:sha256:" +_BACKTEST_MANIFEST_ID_PREFIX = "rhbacktestevidencev1:sha256:" +_RELATIVE_COLUMNS = ("tracking_error", "ir", "alpha", "beta") + + +def _performance_fail( + code: PerformanceEvidenceErrorCode, + path: str, + detail: str, +) -> Never: + raise PerformanceEvidenceError(code, path, detail) + + +def _performance_code_for_path(path: str) -> PerformanceEvidenceErrorCode: + if path == "$.schema_version": + return PerformanceEvidenceErrorCode.UNSUPPORTED_VERSION + if path.startswith("$.methodology") or path in {"$.methodology_id", "$.frequency"}: + return PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH + if path.startswith("$.metrics") or path == "$.metric_schema_id": + return PerformanceEvidenceErrorCode.METRIC_INVALID + if path.startswith("$.benchmark"): + return PerformanceEvidenceErrorCode.BENCHMARK_INVALID + if path in { + "$.authority", + "$.scope", + "$.backtest_evidence_qualification", + }: + return PerformanceEvidenceErrorCode.AUTHORITY_REJECTED + if path.startswith("$.backtest_evidence_manifest") or path.startswith( + "$.performance_table" + ) or path.startswith("$.research_artifact") or path == "$.artifact_available_at": + return PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH + return PerformanceEvidenceErrorCode.IDENTITY_MISMATCH + + +def _performance_validate_tree(value: object, path: str) -> None: + if value is None or type(value) in {bool, str}: + if type(value) is str: + try: + value.encode("utf-8") + except UnicodeEncodeError as error: + raise PerformanceEvidenceError( + _performance_code_for_path(path), + path, + "text must be valid UTF-8", + ) from error + return + if type(value) is int: + if abs(value) > _MAX_SAFE_INTEGER: + _performance_fail( + _performance_code_for_path(path), + path, + "integer exceeds the canonical safe range", + ) + return + if type(value) is float: + if not math.isfinite(value): + _performance_fail( + _performance_code_for_path(path), + path, + "number must be finite", + ) + return + if type(value) is dict: + for key, item in value.items(): + if type(key) is not str: + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + path, + "object keys must be strings", + ) + if not key.isascii(): + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + f"{path}.{key}", + "object keys must be ASCII", + ) + _performance_validate_tree(item, f"{path}.{key}") + return + if type(value) is list: + for index, item in enumerate(value): + _performance_validate_tree(item, f"{path}[{index}]") + return + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + path, + "value is not a canonical JSON type", + ) + + +def _performance_canonical_bytes(values: Mapping[str, object]) -> bytes: + _performance_validate_tree(values, "$") + return json.dumps( + values, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode("utf-8") + + +def _performance_digest(value: Mapping[str, object]) -> str: + return f"sha256:{hashlib.sha256(_performance_canonical_bytes(value)).hexdigest()}" + + +def _performance_compare(actual: object, expected: object, path: str) -> None: + if type(expected) is dict: + if type(actual) is not dict: + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + path, + "must be an object", + ) + unknown = sorted(set(actual) - set(expected)) + if unknown: + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + f"{path}.{unknown[0]}", + "field is not permitted", + ) + missing = sorted(set(expected) - set(actual)) + if missing: + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + f"{path}.{missing[0]}", + "field is required", + ) + for key in sorted(expected): + _performance_compare(actual[key], expected[key], f"{path}.{key}") + return + if type(expected) is list: + if type(actual) is not list: + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + path, + "must be an array", + ) + if len(actual) != len(expected): + _performance_fail( + _performance_code_for_path(path), + path, + "array length differs from the closed contract", + ) + for index, (actual_item, expected_item) in enumerate(zip(actual, expected, strict=True)): + _performance_compare(actual_item, expected_item, f"{path}[{index}]") + return + expected_type = type(expected) + if expected_type is float: + type_matches = type(actual) is float + else: + type_matches = type(actual) is expected_type + if not type_matches: + _performance_fail( + _performance_code_for_path(path), + path, + "value type differs from the closed contract", + ) + if actual != expected: + _performance_fail( + _performance_code_for_path(path), + path, + "value differs from the closed owner evidence", + ) + + +def _performance_wrap_owner_error(error: BacktestContractError) -> Never: + code = ( + PerformanceEvidenceErrorCode.TYPE_ERROR + if error.code is BacktestContractErrorCode.TYPE_ERROR + else PerformanceEvidenceErrorCode.IDENTITY_MISMATCH + if error.code is BacktestContractErrorCode.IDENTITY_MISMATCH + else PerformanceEvidenceErrorCode.AUTHORITY_REJECTED + if error.code + in { + BacktestContractErrorCode.QUALIFICATION_REJECTED, + BacktestContractErrorCode.READINESS_ESCALATION, + } + else PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH + ) + raise PerformanceEvidenceError(code, error.path, "owner contract rejected") from None + + +def _validated_run_ref(run_ref: object) -> tuple[BacktestRunRef, dict[str, Any], str]: + if not isinstance(run_ref, BacktestRunRef): + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + "$.backtest_run_ref", + "accepted BacktestRunRef is required", + ) + document = run_ref.to_dict() + payload = dict(document) + run_id = payload.pop("run_id") + expected_id = ( + _BACKTEST_RUN_ID_PREFIX + + hashlib.sha256(_performance_canonical_bytes(payload)).hexdigest() + ) + if run_id != expected_id: + _performance_fail( + PerformanceEvidenceErrorCode.IDENTITY_MISMATCH, + "$.backtest_run_ref.run_id", + "BacktestRunRef identity does not match its closed document", + ) + return run_ref, document, _performance_digest(document) + + +def _validated_evidence_manifest( + manifest: object, + run_ref: BacktestRunRef, + run_ref_document: Mapping[str, object], +) -> tuple[BacktestEvidenceManifest, dict[str, Any], str, BacktestEvidenceTable]: + if not isinstance(manifest, BacktestEvidenceManifest): + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + "$.backtest_evidence_manifest", + "accepted BacktestEvidenceManifest is required", + ) + if manifest.qualification is not EvidenceQualification.CONTRACT_QUALIFIED: + _performance_fail( + PerformanceEvidenceErrorCode.AUTHORITY_REJECTED, + "$.backtest_evidence_manifest.qualification", + "manifest must be CONTRACT_QUALIFIED", + ) + if manifest.backtest_run_ref is None or manifest.legacy_backtest_run is not None: + _performance_fail( + PerformanceEvidenceErrorCode.AUTHORITY_REJECTED, + "$.backtest_evidence_manifest.run_reference", + "legacy or incomplete evidence is not accepted", + ) + if manifest.backtest_run_ref.to_dict() != run_ref_document: + _performance_fail( + PerformanceEvidenceErrorCode.IDENTITY_MISMATCH, + "$.backtest_evidence_manifest.run_reference", + "manifest does not embed the supplied BacktestRunRef", + ) + if manifest.run_id != run_ref.run_id: + _performance_fail( + PerformanceEvidenceErrorCode.IDENTITY_MISMATCH, + "$.backtest_evidence_manifest.run_id", + "manifest run differs from BacktestRunRef", + ) + if manifest.schema_version != BACKTEST_EVIDENCE_SCHEMA_VERSION: + _performance_fail( + PerformanceEvidenceErrorCode.UNSUPPORTED_VERSION, + "$.backtest_evidence_manifest.schema_version", + "unsupported BacktestEvidenceManifest schema", + ) + if manifest.profile != "offline_research_v1": + _performance_fail( + PerformanceEvidenceErrorCode.AUTHORITY_REJECTED, + "$.backtest_evidence_manifest.profile", + "only the closed offline research profile is accepted", + ) + expected_mapping = _OFFLINE_RESEARCH_V1 + actual_mapping = tuple( + (entry.category, tuple(table.logical_name for table in entry.tables)) + for entry in manifest.evidence + ) + if actual_mapping != expected_mapping: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.backtest_evidence_manifest.evidence", + "manifest evidence profile is not closed", + ) + expected_evidence_digest = _manifest_digest( + [entry.to_dict() for entry in manifest.evidence] + ) + if manifest.evidence_digest != expected_evidence_digest: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.backtest_evidence_manifest.evidence_digest", + "manifest evidence digest does not match its entries", + ) + document = manifest.to_dict() + identity_payload = dict(document) + manifest_id = identity_payload.pop("manifest_id") + expected_id = ( + _BACKTEST_MANIFEST_ID_PREFIX + + hashlib.sha256(_performance_canonical_bytes(identity_payload)).hexdigest() + ) + if manifest_id != expected_id: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.backtest_evidence_manifest.manifest_id", + "manifest identity does not match its closed document", + ) + performance_entries = [ + entry for entry in manifest.evidence if entry.category == "performance" + ] + if len(performance_entries) != 1 or len(performance_entries[0].tables) != 1: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.backtest_evidence_manifest.evidence.performance", + "exactly one performance evidence table is required", + ) + return ( + manifest, + document, + _performance_digest(document), + performance_entries[0].tables[0], + ) + + +def _performance_artifact_frames( + artifact: object, + run_ref: BacktestRunRef, +) -> tuple[ResearchRunArtifact, dict[str, pd.DataFrame], pd.Series[Any]]: + if not isinstance(artifact, ResearchRunArtifact): + _performance_fail( + PerformanceEvidenceErrorCode.TYPE_ERROR, + "$.artifact", + "ResearchRunArtifact is required", + ) + if artifact.schema_version != RESEARCH_ARTIFACT_SCHEMA_VERSION: + _performance_fail( + PerformanceEvidenceErrorCode.UNSUPPORTED_VERSION, + "$.artifact.schema_version", + "unsupported ResearchRunArtifact schema", + ) + try: + frames = _artifact_frames(artifact) + _validate_table_run_ids(frames, run_ref.run_id) + _validate_qualified_run(run_ref, frames) + except BacktestContractError as error: + _performance_wrap_owner_error(error) + run = frames["run"] + performance = frames["performance"] + if len(run) != 1: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.run.row_count", + "run table must contain exactly one row", + ) + if len(performance) != 1: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.performance.row_count", + "performance table must contain exactly one row", + ) + actual_columns = tuple(str(column) for column in performance.columns) + if actual_columns != _PERFORMANCE_SOURCE_COLUMNS: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.performance.columns", + "performance columns differ from the closed V1 schema", + ) + row = performance.iloc[0] + if type(row["run_id"]) is not str or row["run_id"] != run_ref.run_id: + _performance_fail( + PerformanceEvidenceErrorCode.IDENTITY_MISMATCH, + "$.artifact.tables.performance.run_id", + "performance row does not bind the accepted run", + ) + return artifact, frames, row + + +def _performance_text(value: object, path: str) -> str: + if type(value) is not str or not value or value != value.strip(): + _performance_fail( + PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH, + path, + "must be non-empty canonical text", + ) + try: + value.encode("utf-8") + except UnicodeEncodeError as error: + raise PerformanceEvidenceError( + PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH, + path, + "must be valid UTF-8 text", + ) from error + return value + + +def _performance_date(value: object, path: str) -> str: + try: + timestamp = pd.Timestamp(value) + except (TypeError, ValueError) as error: + raise PerformanceEvidenceError( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + path, + "must be a valid date", + ) from error + if pd.isna(timestamp): + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + path, + "must be a valid date", + ) + return date(int(timestamp.year), int(timestamp.month), int(timestamp.day)).isoformat() + + +def _performance_number(value: object, path: str) -> float: + if isinstance(value, np.generic): + value = value.item() + if type(value) not in {int, float}: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + path, + "must be a number", + ) + number = float(cast(int | float, value)) + if not math.isfinite(number): + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + path, + "must be finite", + ) + return number + + +def _performance_integer( + value: object, + path: str, + *, + minimum: int, +) -> int: + if isinstance(value, np.generic): + value = value.item() + if type(value) is not int: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + path, + "must be an integer", + ) + if value < minimum or value > _MAX_SAFE_INTEGER: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + path, + "integer is outside the closed metric domain", + ) + return value + + +def _performance_optional_number(value: object, path: str) -> float | None: + if isinstance(value, np.generic): + value = value.item() + if type(value) is float and math.isnan(value): + return None + return _performance_number(value, path) + + +def _metric( + key: str, + source_column: str, + value: float | int | None, + unit: str, + *, + nullable: bool, + availability: PerformanceMetricAvailability, +) -> PerformanceMetric: + return PerformanceMetric( + key=key, + source_column=source_column, + value=value, + unit=unit, + nullable=nullable, + availability=availability, + metric_schema_id=PERFORMANCE_METRIC_SCHEMA_ID, + methodology_id=PERFORMANCE_METHODOLOGY_ID, + ) + + +def _absolute_performance_metrics(row: pd.Series[Any]) -> list[PerformanceMetric]: + definitions = ( + ("total_return", "total_ret", "ratio", -1.0, None), + ("annualized_return", "ann_ret", "ratio_per_year", -1.0, None), + ("annualized_volatility", "ann_volatility", "ratio_per_year", 0.0, None), + ("sharpe_ratio", "sharpe", "ratio", None, None), + ("sortino_ratio", "sortino", "ratio", None, None), + ("maximum_drawdown", "max_dd", "ratio", -1.0, 0.0), + ("calmar_ratio", "calmar", "ratio", None, None), + ("win_rate", "win_rate", "ratio", 0.0, 1.0), + ) + metrics: list[PerformanceMetric] = [] + for key, column, unit, minimum, maximum in definitions: + path = f"$.metrics.{key}.value" + value = _performance_number(row[column], path) + if minimum is not None and value < minimum: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + path, + "value is below the closed metric domain", + ) + if maximum is not None and value > maximum: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + path, + "value is above the closed metric domain", + ) + metrics.append( + _metric( + key, + column, + value, + unit, + nullable=False, + availability=PerformanceMetricAvailability.AVAILABLE, + ) + ) + return metrics + + +def _count_performance_metrics(row: pd.Series[Any]) -> list[PerformanceMetric]: + return [ + _metric( + "trade_count", + "n_trades", + _performance_integer( + row["n_trades"], + "$.metrics.trade_count.value", + minimum=0, + ), + "count", + nullable=False, + availability=PerformanceMetricAvailability.AVAILABLE, + ), + _metric( + "day_count", + "n_days", + _performance_integer( + row["n_days"], + "$.metrics.day_count.value", + minimum=1, + ), + "count", + nullable=False, + availability=PerformanceMetricAvailability.AVAILABLE, + ), + ] + + +def _benchmark_context( + frames: Mapping[str, pd.DataFrame], + run_row: pd.Series[Any], + performance_row: pd.Series[Any], +) -> tuple[str | None, float | None, float | None, bool | None]: + benchmark_id = run_row["benchmark_id"] + alignment = run_row["benchmark_alignment_policy"] + if type(benchmark_id) is not str: + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.benchmark_id", + "benchmark ID must be canonical text", + ) + if type(alignment) is not str: + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.benchmark_alignment_policy", + "benchmark alignment must be canonical text", + ) + nav = frames["nav"] + required_columns = {"trade_date", "pnl_pct", "benchmark_return", "benchmark_nav"} + missing = sorted(required_columns - set(nav.columns)) + if missing: + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + f"$.artifact.tables.nav.columns.{missing[0]}", + "benchmark evidence column is missing", + ) + if len(nav) != int(performance_row["n_days"]): + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav.row_count", + "NAV observations differ from the performance day count", + ) + if benchmark_id == "": + if alignment != "none": + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.benchmark_alignment_policy", + "absent benchmark requires none alignment", + ) + if not nav["benchmark_return"].isna().all() or not nav["benchmark_nav"].isna().all(): + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav.benchmark_return", + "absent benchmark cannot contain benchmark observations", + ) + return None, None, None, None + if not benchmark_id.strip() or alignment != "exact_session_index": + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.benchmark_alignment_policy", + "present benchmark requires exact session alignment", + ) + if len(nav) < 2: + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav.row_count", + "present benchmark requires at least two observations", + ) + try: + portfolio = nav["pnl_pct"].astype(float, copy=True) + benchmark = nav["benchmark_return"].astype(float, copy=True) + benchmark_nav = nav["benchmark_nav"].astype(float, copy=True) + except (TypeError, ValueError) as error: + raise PerformanceEvidenceError( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav", + "benchmark observations must be numeric", + ) from error + if not ( + np.isfinite(portfolio.to_numpy()).all() + and np.isfinite(benchmark.to_numpy()).all() + and np.isfinite(benchmark_nav.to_numpy()).all() + ): + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav.benchmark_return", + "present benchmark observations must be finite", + ) + if (portfolio < -1.0).any() or (benchmark < -1.0).any(): + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav.benchmark_return", + "daily simple returns cannot be less than -1", + ) + observations = [ + { + "trade_date": _performance_date( + nav.iloc[index]["trade_date"], + f"$.artifact.tables.nav.rows[{index}].trade_date", + ), + "benchmark_return": float(benchmark.iloc[index]), + "benchmark_nav": float(benchmark_nav.iloc[index]), + } + for index in range(len(nav)) + ] + dates = [cast(str, item["trade_date"]) for item in observations] + if dates != sorted(dates) or len(dates) != len(set(dates)): + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav.trade_date", + "benchmark observations must use a unique ordered session index", + ) + digest_payload: dict[str, object] = { + "benchmark_id": benchmark_id, + "start_date": _performance_date(run_row["start_date"], "$.artifact.tables.run.start_date"), + "end_date": _performance_date(run_row["end_date"], "$.artifact.tables.run.end_date"), + "frequency": run_row["frequency"], + "calendar": run_row["calendar"], + "timezone": run_row["timezone"], + "observations": observations, + } + active_std = float((portfolio - benchmark).std()) + benchmark_variance = float(benchmark.var()) + if not math.isfinite(active_std) or not math.isfinite(benchmark_variance): + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.artifact.tables.nav", + "benchmark variance statistics must be finite", + ) + beta = _performance_optional_number( + performance_row["beta"], + "$.metrics.beta.value", + ) + alpha_domain_unestimable = ( + float((portfolio - beta * benchmark).mean()) <= -1.0 + if benchmark_variance >= 1e-30 and beta is not None + else None + ) + return ( + _performance_digest(digest_payload), + active_std, + benchmark_variance, + alpha_domain_unestimable, + ) + + +def _relative_performance_metrics( + row: pd.Series[Any], + *, + benchmark_present: bool, + active_std: float | None, + benchmark_variance: float | None, + alpha_domain_unestimable: bool | None, +) -> list[PerformanceMetric]: + source = { + key: _performance_optional_number(row[column], f"$.metrics.{key}.value") + for key, column in ( + ("tracking_error", "tracking_error"), + ("information_ratio", "ir"), + ("alpha", "alpha"), + ("beta", "beta"), + ) + } + if not benchmark_present: + if any(value is not None for value in source.values()): + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.metrics.tracking_error.availability", + "benchmark-absent relative metrics must be null", + ) + return [ + _metric( + key, + column, + None, + unit, + nullable=True, + availability=PerformanceMetricAvailability.BENCHMARK_ABSENT, + ) + for key, column, unit in ( + ("tracking_error", "tracking_error", "ratio_per_year"), + ("information_ratio", "ir", "ratio"), + ("alpha", "alpha", "ratio_per_year"), + ("beta", "beta", "ratio"), + ) + ] + if active_std is None or benchmark_variance is None: + _performance_fail( + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.benchmark_series_digest", + "present benchmark statistics are required", + ) + tracking_error = source["tracking_error"] + if tracking_error is None or tracking_error < 0: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.tracking_error.value", + "tracking error must be finite and non-negative for a present benchmark", + ) + information_ratio = source["information_ratio"] + if active_std < 1e-30: + if information_ratio is not None: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.information_ratio.value", + "information ratio must be null when active variance is not estimable", + ) + information_availability = ( + PerformanceMetricAvailability.NOT_ESTIMABLE_ACTIVE_VARIANCE + ) + else: + if information_ratio is None: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.information_ratio.value", + "information ratio must be finite when active variance is estimable", + ) + information_availability = PerformanceMetricAvailability.AVAILABLE + beta = source["beta"] + alpha = source["alpha"] + if benchmark_variance < 1e-30: + if beta is not None or alpha is not None: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.beta.value", + "alpha and beta must be null when benchmark variance is not estimable", + ) + beta_availability = PerformanceMetricAvailability.NOT_ESTIMABLE_BENCHMARK_VARIANCE + alpha_availability = PerformanceMetricAvailability.NOT_ESTIMABLE_BENCHMARK_VARIANCE + else: + if beta is None: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.beta.value", + "beta must be finite when benchmark variance is estimable", + ) + beta_availability = PerformanceMetricAvailability.AVAILABLE + if alpha is None: + if alpha_domain_unestimable is not True: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.alpha.value", + "null alpha requires the closed annualization-domain reason", + ) + alpha_availability = PerformanceMetricAvailability.NOT_ESTIMABLE_ALPHA_DOMAIN + else: + if alpha_domain_unestimable is True: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.alpha.value", + "alpha must be null outside the geometric annualization domain", + ) + if alpha <= -1.0: + _performance_fail( + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.alpha.value", + "finite alpha must be greater than -1", + ) + alpha_availability = PerformanceMetricAvailability.AVAILABLE + return [ + _metric( + "tracking_error", + "tracking_error", + tracking_error, + "ratio_per_year", + nullable=True, + availability=PerformanceMetricAvailability.AVAILABLE, + ), + _metric( + "information_ratio", + "ir", + information_ratio, + "ratio", + nullable=True, + availability=information_availability, + ), + _metric( + "alpha", + "alpha", + alpha, + "ratio_per_year", + nullable=True, + availability=alpha_availability, + ), + _metric( + "beta", + "beta", + beta, + "ratio", + nullable=True, + availability=beta_availability, + ), + ] + + +def _performance_methodology( + *, + frequency: str, + alignment: str, + code_revision: str, +) -> PerformanceMethodology: + return PerformanceMethodology( + methodology_id=PERFORMANCE_METHODOLOGY_ID, + return_type="simple", + source_frequency=frequency, + periods_per_year=TRADING_DAYS_PER_YEAR, + annualized_return="geometric_compound", + annualized_volatility="sample_std_sqrt_periods", + sharpe_ratio="annualized_return_minus_annual_risk_free_over_annualized_volatility", + sortino_ratio=( + "annualized_return_minus_annual_risk_free_over_" + "root_mean_square_negative_returns_sqrt_periods" + ), + annual_risk_free=0.0, + tracking_error="sample_std_active_return_sqrt_periods", + information_ratio="mean_active_over_sample_std_active_sqrt_periods", + alpha="daily_ols_intercept_geometric_annualization", + beta="sample_covariance_over_sample_variance", + benchmark_risk_free_daily=0.0, + benchmark_alignment=alignment, + maximum_drawdown="non_positive_peak_to_trough_ratio_with_initial_nav_one", + calmar_ratio="unadjusted_annualized_return_over_absolute_maximum_drawdown", + win_rate="positive_daily_return_count_over_observation_count", + total_return="final_nav_minus_one", + implementation_module="quant_engine.metrics", + implementation_version=PERFORMANCE_METHODOLOGY_ID, + code_revision=code_revision, + ) + + +def build_performance_evidence( + artifact: ResearchRunArtifact, + run_ref: BacktestRunRef, + evidence_manifest: BacktestEvidenceManifest, +) -> PerformanceEvidenceV1: + """Bind existing performance facts and methodology without recalculation.""" + validated_run_ref, run_document, run_document_digest = _validated_run_ref(run_ref) + ( + validated_manifest, + _manifest_document, + manifest_document_digest, + manifest_performance_table, + ) = _validated_evidence_manifest( + evidence_manifest, + validated_run_ref, + run_document, + ) + validated_artifact, frames, performance_row = _performance_artifact_frames( + artifact, + validated_run_ref, + ) + if validated_manifest.artifact_schema_version != validated_artifact.schema_version: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.backtest_evidence_manifest.artifact_schema_version", + "manifest and artifact schema versions differ", + ) + try: + computed_tables = _table_evidence(frames, None) + except BacktestContractError as error: + _performance_wrap_owner_error(error) + computed_entries = _evidence_entries( + computed_tables, + {"kind": "backtest_run_ref", "value": run_document}, + legacy=False, + ) + for manifest_entry, computed_entry in zip( + validated_manifest.evidence, + computed_entries, + strict=True, + ): + for manifest_table, computed_table in zip( + manifest_entry.tables, + computed_entry.tables, + strict=True, + ): + for field in ("logical_name", "row_count", "schema_digest", "content_digest"): + if getattr(manifest_table, field) != getattr(computed_table, field): + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + f"$.artifact.tables.{computed_table.logical_name}.{field}", + "artifact table differs from accepted manifest evidence", + ) + if manifest_entry.evidence_digest != computed_entry.evidence_digest: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + f"$.backtest_evidence_manifest.evidence.{computed_entry.category}.evidence_digest", + "manifest category does not close the supplied artifact", + ) + computed_performance_table = computed_tables["performance"] + for field in ("logical_name", "row_count", "schema_digest", "content_digest"): + if getattr(manifest_performance_table, field) != getattr( + computed_performance_table, + field, + ): + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + f"$.artifact.tables.performance.{field}", + "artifact performance table differs from accepted manifest evidence", + ) + run_row = frames["run"].iloc[0] + if run_row["schema_version"] != validated_artifact.schema_version: + _performance_fail( + PerformanceEvidenceErrorCode.UNSUPPORTED_VERSION, + "$.artifact.tables.run.schema_version", + "run row and artifact schema versions differ", + ) + frequency = _performance_text(run_row["frequency"], "$.artifact.tables.run.frequency") + if frequency != "1d": + _performance_fail( + PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH, + "$.artifact.tables.run.frequency", + "PerformanceEvidenceV1 supports only existing daily semantics", + ) + calendar = _performance_text(run_row["calendar"], "$.artifact.tables.run.calendar") + timezone = _performance_text(run_row["timezone"], "$.artifact.tables.run.timezone") + start_date = _performance_date(run_row["start_date"], "$.artifact.tables.run.start_date") + end_date = _performance_date(run_row["end_date"], "$.artifact.tables.run.end_date") + nav = frames["nav"] + if nav.empty: + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.nav.row_count", + "NAV table must contain the performance observation window", + ) + if ( + _performance_date(nav.iloc[0]["trade_date"], "$.artifact.tables.nav.rows[0].trade_date") + != start_date + or _performance_date( + nav.iloc[-1]["trade_date"], + f"$.artifact.tables.nav.rows[{len(nav) - 1}].trade_date", + ) + != end_date + ): + _performance_fail( + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.nav.trade_date", + "NAV observation window differs from artifact start/end dates", + ) + ( + benchmark_series_digest, + active_std, + benchmark_variance, + alpha_domain_unestimable, + ) = _benchmark_context( + frames, + run_row, + performance_row, + ) + benchmark_id = cast(str, run_row["benchmark_id"]) + alignment = cast(str, run_row["benchmark_alignment_policy"]) + metrics = ( + *_absolute_performance_metrics(performance_row), + *_relative_performance_metrics( + performance_row, + benchmark_present=benchmark_series_digest is not None, + active_std=active_std, + benchmark_variance=benchmark_variance, + alpha_domain_unestimable=alpha_domain_unestimable, + ), + *_count_performance_metrics(performance_row), + ) + normalized_row: dict[str, object] = { + metric.source_column: metric.value for metric in metrics + } + normalized_row["run_id"] = validated_run_ref.run_id + performance_row_digest = _performance_digest( + { + "columns": list(_PERFORMANCE_SOURCE_COLUMNS), + "row": normalized_row, + } + ) + methodology = _performance_methodology( + frequency=frequency, + alignment=alignment, + code_revision=validated_run_ref.code_revision, + ) + try: + artifact_content_digest = f"sha256:{validated_artifact.content_sha256}" + except BacktestContractError as error: + _performance_wrap_owner_error(error) + payload: dict[str, object] = { + "schema_version": PERFORMANCE_EVIDENCE_SCHEMA_VERSION, + "authority": "quant_engine", + "scope": "offline_research_only", + "run_id": validated_run_ref.run_id, + "backtest_run_ref_id": validated_run_ref.run_id, + "backtest_run_ref_document_sha256": run_document_digest, + "backtest_evidence_manifest_id": validated_manifest.manifest_id, + "backtest_evidence_manifest_document_sha256": manifest_document_digest, + "backtest_evidence_manifest_evidence_digest": validated_manifest.evidence_digest, + "backtest_evidence_qualification": validated_manifest.qualification.value, + "research_artifact_schema_version": validated_artifact.schema_version, + "research_artifact_content_digest": artifact_content_digest, + "artifact_available_at": validated_manifest.artifact_available_at, + "performance_table_logical_name": computed_performance_table.logical_name, + "performance_table_row_count": computed_performance_table.row_count, + "performance_table_schema_digest": computed_performance_table.schema_digest, + "performance_table_content_digest": computed_performance_table.content_digest, + "performance_row_digest": performance_row_digest, + "benchmark_series_digest": benchmark_series_digest, + "methodology_id": PERFORMANCE_METHODOLOGY_ID, + "metric_schema_id": PERFORMANCE_METRIC_SCHEMA_ID, + "dataset_snapshot_id": validated_run_ref.dataset_snapshot_id, + "dataset_content_digest": validated_run_ref.dataset_content_digest, + "dataset_manifest_digest": validated_run_ref.dataset_manifest_digest, + "foundation_id": validated_run_ref.foundation_id, + "foundation_digest": validated_run_ref.foundation_digest, + "factor_set_id": validated_run_ref.factor_set_id, + "factor_set_digest": validated_run_ref.factor_set_digest, + "factor_output_content_digest": validated_run_ref.factor_output_content_digest, + "strategy_id": validated_run_ref.strategy_id, + "strategy_version": validated_run_ref.strategy_version, + "strategy_digest": validated_run_ref.strategy_digest, + "execution_model_version": validated_run_ref.execution_model_version, + "execution_model_digest": validated_run_ref.execution_model_digest, + "cost_model_version": validated_run_ref.cost_model_version, + "cost_model_digest": validated_run_ref.cost_model_digest, + "code_revision": validated_run_ref.code_revision, + "environment_lock_digest": validated_run_ref.environment_lock_digest, + "configuration_digest": validated_run_ref.configuration_digest, + "frequency": frequency, + "calendar": calendar, + "timezone": timezone, + "benchmark_id": benchmark_id, + "benchmark_alignment_policy": alignment, + "start_date": start_date, + "end_date": end_date, + "methodology": methodology.to_dict(), + "metrics": [metric.to_dict() for metric in metrics], + } + performance_evidence_id = ( + _PERFORMANCE_IDENTITY_PREFIX + + hashlib.sha256(_performance_canonical_bytes(payload)).hexdigest() + ) + document_payload = {**payload, "performance_evidence_id": performance_evidence_id} + document_sha256 = _performance_digest(document_payload) + values: dict[str, object] = { + **document_payload, + "document_sha256": document_sha256, + "methodology": methodology, + "metrics": metrics, + } + instance = object.__new__(PerformanceEvidenceV1) + for name, value in values.items(): + object.__setattr__(instance, name, value) + return instance + + def _build_nav( result: FactorBacktestResult, run_id: str, diff --git a/tests/fixtures/performance-evidence-v1.golden.json b/tests/fixtures/performance-evidence-v1.golden.json new file mode 100644 index 0000000..5ca121e --- /dev/null +++ b/tests/fixtures/performance-evidence-v1.golden.json @@ -0,0 +1,871 @@ +{ + "cases": { + "absent": { + "artifact_available_at": "2026-01-08T02:05:00Z", + "authority": "quant_engine", + "backtest_evidence_manifest_document_sha256": "sha256:b7e9139e021de3122bee9376a8c8387fca3db75fe2a698adccf33471caf971d2", + "backtest_evidence_manifest_evidence_digest": "sha256:fae26a93d754e98f437bf9e4b635fc1cdc4f85d004a4bde396823b52e2115de5", + "backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:3f67fcad23c75684ffeb325d8405b2f81139a732e237d30fbb30d23f91e32726", + "backtest_evidence_qualification": "contract_qualified", + "backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a", + "backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "benchmark_alignment_policy": "none", + "benchmark_id": "", + "benchmark_series_digest": null, + "calendar": "CN-A", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888", + "cost_model_version": "1.0.0", + "dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020", + "dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd", + "dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57", + "document_sha256": "sha256:729852af307fd4a94ac566ae45df2e57b3d45c4fbdf45820351b5ef19cdf0ad3", + "end_date": "2026-01-08", + "environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999", + "execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777", + "execution_model_version": "1.0.0", + "factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e", + "factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "frequency": "1d", + "methodology": { + "alpha": "daily_ols_intercept_geometric_annualization", + "annual_risk_free": 0.0, + "annualized_return": "geometric_compound", + "annualized_volatility": "sample_std_sqrt_periods", + "benchmark_alignment": "none", + "benchmark_risk_free_daily": 0.0, + "beta": "sample_covariance_over_sample_variance", + "calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "implementation_module": "quant_engine.metrics", + "implementation_version": "researchhub.quant-performance-methodology.v1", + "information_ratio": "mean_active_over_sample_std_active_sqrt_periods", + "maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "periods_per_year": 252, + "return_type": "simple", + "sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility", + "sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods", + "source_frequency": "1d", + "total_return": "final_nav_minus_one", + "tracking_error": "sample_std_active_return_sqrt_periods", + "win_rate": "positive_daily_return_count_over_observation_count" + }, + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "metrics": [ + { + "availability": "available", + "key": "total_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "total_ret", + "unit": "ratio", + "value": 0.575 + }, + { + "availability": "available", + "key": "annualized_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_ret", + "unit": "ratio_per_year", + "value": 2683336646708.1 + }, + { + "availability": "available", + "key": "annualized_volatility", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_volatility", + "unit": "ratio_per_year", + "value": 1.38901943830891 + }, + { + "availability": "available", + "key": "sharpe_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sharpe", + "unit": "ratio", + "value": 1931820803008.3313 + }, + { + "availability": "available", + "key": "sortino_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sortino", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "maximum_drawdown", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "max_dd", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "calmar_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "calmar", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "win_rate", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "win_rate", + "unit": "ratio", + "value": 0.75 + }, + { + "availability": "benchmark_absent", + "key": "tracking_error", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "tracking_error", + "unit": "ratio_per_year", + "value": null + }, + { + "availability": "benchmark_absent", + "key": "information_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "ir", + "unit": "ratio", + "value": null + }, + { + "availability": "benchmark_absent", + "key": "alpha", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "alpha", + "unit": "ratio_per_year", + "value": null + }, + { + "availability": "benchmark_absent", + "key": "beta", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "beta", + "unit": "ratio", + "value": null + }, + { + "availability": "available", + "key": "trade_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_trades", + "unit": "count", + "value": 3 + }, + { + "availability": "available", + "key": "day_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_days", + "unit": "count", + "value": 4 + } + ], + "performance_evidence_id": "rhperformanceevidencev1:sha256:d735abe76c28db429e107cb5ba6a1929c2c0f2084a6c4ac41b38ffbc02b2d2bd", + "performance_row_digest": "sha256:4e6329b0db4731513fe793491a163dd358e3a6f7baa84d3c94a4d4d2355c7a7d", + "performance_table_content_digest": "sha256:074b8511ff9a2154235e4dde8bac300658cb69f900e49f332d61560b82be9af5", + "performance_table_logical_name": "performance", + "performance_table_row_count": 1, + "performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0", + "research_artifact_content_digest": "sha256:ad85ee1317bd8ce5fbef8fddc268b91be8678701ba5b8fbf6d61bf670f918bf9", + "research_artifact_schema_version": "1.1.0", + "run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "schema_version": "researchhub.performance-evidence.v1", + "scope": "offline_research_only", + "start_date": "2026-01-05", + "strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666", + "strategy_id": "alpha-top1", + "strategy_version": "1.0.0", + "timezone": "Asia/Shanghai" + }, + "estimable": { + "artifact_available_at": "2026-01-08T02:05:00Z", + "authority": "quant_engine", + "backtest_evidence_manifest_document_sha256": "sha256:e2e0e63fb4c733d2dba9a290511b6c8b0132bfe877dfe89ee7e614b74b87bc51", + "backtest_evidence_manifest_evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650", + "backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117", + "backtest_evidence_qualification": "contract_qualified", + "backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a", + "backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "benchmark_alignment_policy": "exact_session_index", + "benchmark_id": "000300.SH", + "benchmark_series_digest": "sha256:1c564e4d12af0d59b1fd707d7816da2895238969d9207470984db63cfeac423a", + "calendar": "CN-A", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888", + "cost_model_version": "1.0.0", + "dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020", + "dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd", + "dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57", + "document_sha256": "sha256:a6a1a3313f511505c0cf57fddeebdf20a3b0f83a2958f31332d5597d172165e7", + "end_date": "2026-01-08", + "environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999", + "execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777", + "execution_model_version": "1.0.0", + "factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e", + "factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "frequency": "1d", + "methodology": { + "alpha": "daily_ols_intercept_geometric_annualization", + "annual_risk_free": 0.0, + "annualized_return": "geometric_compound", + "annualized_volatility": "sample_std_sqrt_periods", + "benchmark_alignment": "exact_session_index", + "benchmark_risk_free_daily": 0.0, + "beta": "sample_covariance_over_sample_variance", + "calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "implementation_module": "quant_engine.metrics", + "implementation_version": "researchhub.quant-performance-methodology.v1", + "information_ratio": "mean_active_over_sample_std_active_sqrt_periods", + "maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "periods_per_year": 252, + "return_type": "simple", + "sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility", + "sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods", + "source_frequency": "1d", + "total_return": "final_nav_minus_one", + "tracking_error": "sample_std_active_return_sqrt_periods", + "win_rate": "positive_daily_return_count_over_observation_count" + }, + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "metrics": [ + { + "availability": "available", + "key": "total_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "total_ret", + "unit": "ratio", + "value": 0.575 + }, + { + "availability": "available", + "key": "annualized_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_ret", + "unit": "ratio_per_year", + "value": 2683336646708.1 + }, + { + "availability": "available", + "key": "annualized_volatility", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_volatility", + "unit": "ratio_per_year", + "value": 1.38901943830891 + }, + { + "availability": "available", + "key": "sharpe_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sharpe", + "unit": "ratio", + "value": 1931820803008.3313 + }, + { + "availability": "available", + "key": "sortino_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sortino", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "maximum_drawdown", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "max_dd", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "calmar_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "calmar", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "win_rate", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "win_rate", + "unit": "ratio", + "value": 0.75 + }, + { + "availability": "available", + "key": "tracking_error", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "tracking_error", + "unit": "ratio_per_year", + "value": 1.3032171729991897 + }, + { + "availability": "available", + "key": "information_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "ir", + "unit": "ratio", + "value": 22.801264912443322 + }, + { + "availability": "available", + "key": "alpha", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "alpha", + "unit": "ratio_per_year", + "value": 123663320625.66454 + }, + { + "availability": "available", + "key": "beta", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "beta", + "unit": "ratio", + "value": 3.2500000000000013 + }, + { + "availability": "available", + "key": "trade_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_trades", + "unit": "count", + "value": 3 + }, + { + "availability": "available", + "key": "day_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_days", + "unit": "count", + "value": 4 + } + ], + "performance_evidence_id": "rhperformanceevidencev1:sha256:f1f72c34d7427bd0d2e273a66b9677d112aa51c2417ea1cda8245e51e334e4f6", + "performance_row_digest": "sha256:bad2b506679a6ca13d80ab00859a442ba692f3872d42b0b45bbc3b1d6ffe72a4", + "performance_table_content_digest": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf", + "performance_table_logical_name": "performance", + "performance_table_row_count": 1, + "performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0", + "research_artifact_content_digest": "sha256:6ec566d73da3966d3d5f946313e2b0181993374c9f4ceb4b56e165cc6793d382", + "research_artifact_schema_version": "1.1.0", + "run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "schema_version": "researchhub.performance-evidence.v1", + "scope": "offline_research_only", + "start_date": "2026-01-05", + "strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666", + "strategy_id": "alpha-top1", + "strategy_version": "1.0.0", + "timezone": "Asia/Shanghai" + }, + "zero_active_variance": { + "artifact_available_at": "2026-01-08T02:05:00Z", + "authority": "quant_engine", + "backtest_evidence_manifest_document_sha256": "sha256:8a13212575154d97f85011ed54ed3bb588f73a691926b3ca03b360d563e9fc29", + "backtest_evidence_manifest_evidence_digest": "sha256:8c0c2052ab93aa920254bc7f4d89435d839b7150ecaaaa798703e89ddf876ec1", + "backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:699ad35ee30a294082c607c40617097aa90d5d58343e7d8c8185d8a53f52596c", + "backtest_evidence_qualification": "contract_qualified", + "backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a", + "backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "benchmark_alignment_policy": "exact_session_index", + "benchmark_id": "000300.SH", + "benchmark_series_digest": "sha256:7668b77ac689cd542005fae42cb2abd48fdf772d2f982086dca409606653e6ac", + "calendar": "CN-A", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888", + "cost_model_version": "1.0.0", + "dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020", + "dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd", + "dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57", + "document_sha256": "sha256:44a04be0105f29e1cbf8d2430b6b0eb084853d428bc2ccd5834dfdbcf9482bd7", + "end_date": "2026-01-08", + "environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999", + "execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777", + "execution_model_version": "1.0.0", + "factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e", + "factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "frequency": "1d", + "methodology": { + "alpha": "daily_ols_intercept_geometric_annualization", + "annual_risk_free": 0.0, + "annualized_return": "geometric_compound", + "annualized_volatility": "sample_std_sqrt_periods", + "benchmark_alignment": "exact_session_index", + "benchmark_risk_free_daily": 0.0, + "beta": "sample_covariance_over_sample_variance", + "calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "implementation_module": "quant_engine.metrics", + "implementation_version": "researchhub.quant-performance-methodology.v1", + "information_ratio": "mean_active_over_sample_std_active_sqrt_periods", + "maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "periods_per_year": 252, + "return_type": "simple", + "sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility", + "sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods", + "source_frequency": "1d", + "total_return": "final_nav_minus_one", + "tracking_error": "sample_std_active_return_sqrt_periods", + "win_rate": "positive_daily_return_count_over_observation_count" + }, + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "metrics": [ + { + "availability": "available", + "key": "total_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "total_ret", + "unit": "ratio", + "value": 0.575 + }, + { + "availability": "available", + "key": "annualized_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_ret", + "unit": "ratio_per_year", + "value": 2683336646708.1 + }, + { + "availability": "available", + "key": "annualized_volatility", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_volatility", + "unit": "ratio_per_year", + "value": 1.38901943830891 + }, + { + "availability": "available", + "key": "sharpe_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sharpe", + "unit": "ratio", + "value": 1931820803008.3313 + }, + { + "availability": "available", + "key": "sortino_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sortino", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "maximum_drawdown", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "max_dd", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "calmar_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "calmar", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "win_rate", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "win_rate", + "unit": "ratio", + "value": 0.75 + }, + { + "availability": "available", + "key": "tracking_error", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "tracking_error", + "unit": "ratio_per_year", + "value": 0.0 + }, + { + "availability": "not_estimable_active_variance", + "key": "information_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "ir", + "unit": "ratio", + "value": null + }, + { + "availability": "available", + "key": "alpha", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "alpha", + "unit": "ratio_per_year", + "value": 0.0 + }, + { + "availability": "available", + "key": "beta", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "beta", + "unit": "ratio", + "value": 1.0000000000000002 + }, + { + "availability": "available", + "key": "trade_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_trades", + "unit": "count", + "value": 3 + }, + { + "availability": "available", + "key": "day_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_days", + "unit": "count", + "value": 4 + } + ], + "performance_evidence_id": "rhperformanceevidencev1:sha256:e6730d25d90c85a6370adb72bc45c1bf505235f77b72521dd8197898ca369fe5", + "performance_row_digest": "sha256:6528c47fa9416e350aa5010477dbd8c83a35cecf87aa5bc413cec95a9765114d", + "performance_table_content_digest": "sha256:7d57a966431c68093a9f6193ae62f79a63d33832f11dd68524ed9bcd2cb6257c", + "performance_table_logical_name": "performance", + "performance_table_row_count": 1, + "performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0", + "research_artifact_content_digest": "sha256:b17ab9e158a7ceeb5107f6fe8a61f332009c314186f3436d4750aa44d6ca3856", + "research_artifact_schema_version": "1.1.0", + "run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "schema_version": "researchhub.performance-evidence.v1", + "scope": "offline_research_only", + "start_date": "2026-01-05", + "strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666", + "strategy_id": "alpha-top1", + "strategy_version": "1.0.0", + "timezone": "Asia/Shanghai" + }, + "zero_benchmark_variance": { + "artifact_available_at": "2026-01-08T02:05:00Z", + "authority": "quant_engine", + "backtest_evidence_manifest_document_sha256": "sha256:2389b804f5d8d20a37b31f596b52892484777c5da799dd68865a082ee90e6e5b", + "backtest_evidence_manifest_evidence_digest": "sha256:54bb315bc1940ef6df79999cf03ebb8d10defaf526bf7802d63da33f612520ec", + "backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:ad9559e1f8feca28bb310765216a975935f05f51057139c2d02fd2fe7dfe501e", + "backtest_evidence_qualification": "contract_qualified", + "backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a", + "backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "benchmark_alignment_policy": "exact_session_index", + "benchmark_id": "000300.SH", + "benchmark_series_digest": "sha256:d0f1e954d796e95254bcccc893d6e3a8f9d541c3028b88b2c0b24c1658bb2408", + "calendar": "CN-A", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888", + "cost_model_version": "1.0.0", + "dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020", + "dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd", + "dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57", + "document_sha256": "sha256:077f3ec84802aff2f046365a3a5f61f475b1ad63ec05e96742844bc531bcb551", + "end_date": "2026-01-08", + "environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999", + "execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777", + "execution_model_version": "1.0.0", + "factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e", + "factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021", + "foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838", + "frequency": "1d", + "methodology": { + "alpha": "daily_ols_intercept_geometric_annualization", + "annual_risk_free": 0.0, + "annualized_return": "geometric_compound", + "annualized_volatility": "sample_std_sqrt_periods", + "benchmark_alignment": "exact_session_index", + "benchmark_risk_free_daily": 0.0, + "beta": "sample_covariance_over_sample_variance", + "calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "implementation_module": "quant_engine.metrics", + "implementation_version": "researchhub.quant-performance-methodology.v1", + "information_ratio": "mean_active_over_sample_std_active_sqrt_periods", + "maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "periods_per_year": 252, + "return_type": "simple", + "sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility", + "sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods", + "source_frequency": "1d", + "total_return": "final_nav_minus_one", + "tracking_error": "sample_std_active_return_sqrt_periods", + "win_rate": "positive_daily_return_count_over_observation_count" + }, + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "metrics": [ + { + "availability": "available", + "key": "total_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "total_ret", + "unit": "ratio", + "value": 0.575 + }, + { + "availability": "available", + "key": "annualized_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_ret", + "unit": "ratio_per_year", + "value": 2683336646708.1 + }, + { + "availability": "available", + "key": "annualized_volatility", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_volatility", + "unit": "ratio_per_year", + "value": 1.38901943830891 + }, + { + "availability": "available", + "key": "sharpe_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sharpe", + "unit": "ratio", + "value": 1931820803008.3313 + }, + { + "availability": "available", + "key": "sortino_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sortino", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "maximum_drawdown", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "max_dd", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "calmar_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "calmar", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "win_rate", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "win_rate", + "unit": "ratio", + "value": 0.75 + }, + { + "availability": "available", + "key": "tracking_error", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "tracking_error", + "unit": "ratio_per_year", + "value": 1.38901943830891 + }, + { + "availability": "available", + "key": "information_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "ir", + "unit": "ratio", + "value": 22.299903907544408 + }, + { + "availability": "not_estimable_benchmark_variance", + "key": "alpha", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "alpha", + "unit": "ratio_per_year", + "value": null + }, + { + "availability": "not_estimable_benchmark_variance", + "key": "beta", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "beta", + "unit": "ratio", + "value": null + }, + { + "availability": "available", + "key": "trade_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_trades", + "unit": "count", + "value": 3 + }, + { + "availability": "available", + "key": "day_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_days", + "unit": "count", + "value": 4 + } + ], + "performance_evidence_id": "rhperformanceevidencev1:sha256:86133a6b3c3091477cdc4618ebc294e671c9ba0a3ef5015a068f92bb575729b9", + "performance_row_digest": "sha256:90c9aa2f4168c15ff9ac5d4da2a35c3e915bb4e04736e5a440ea80c787a99553", + "performance_table_content_digest": "sha256:1570b64d83ad2a849607e6f5203f5ec2f0291a8d6cbb00b8edfe0ae030b7967d", + "performance_table_logical_name": "performance", + "performance_table_row_count": 1, + "performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0", + "research_artifact_content_digest": "sha256:fed7a28888a6e98ccce78e7d84f22d4e3636e928485861af7a97f90eddc91f94", + "research_artifact_schema_version": "1.1.0", + "run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386", + "schema_version": "researchhub.performance-evidence.v1", + "scope": "offline_research_only", + "start_date": "2026-01-05", + "strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666", + "strategy_id": "alpha-top1", + "strategy_version": "1.0.0", + "timezone": "Asia/Shanghai" + } + }, + "schema_version": 1, + "source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c", + "source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603" +} diff --git a/tests/governance/test_module_spec.py b/tests/governance/test_module_spec.py index 6d0a088..77b0188 100644 --- a/tests/governance/test_module_spec.py +++ b/tests/governance/test_module_spec.py @@ -16,7 +16,7 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower() for term in ("investment advice", "live order", "credentials", "source facts"): assert term in prohibited - assert spec["authority"]["revision"] == 4 + assert spec["authority"]["revision"] == 5 assert { (item["contract_id"], item["version"]) for item in spec["contracts"]["provides"] @@ -25,6 +25,7 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: ("researchhub.factor-set-ref", "1.0.0"), ("researchhub.backtest-run-ref", "1.0.0"), ("researchhub.backtest-evidence-manifest", "1.0.0"), + ("researchhub.performance-evidence", "1.0.0"), ("researchhub.portfolio-decision", "1.0.0"), ("researchhub.risk-assessment", "1.0.0"), } @@ -33,6 +34,7 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: "researchhub.factor-set-ref": "src/quant_engine/factor_contracts.py", "researchhub.backtest-run-ref": "src/quant_engine/governed_pipeline.py", "researchhub.backtest-evidence-manifest": "src/quant_engine/artifact.py", + "researchhub.performance-evidence": "src/quant_engine/artifact.py", "researchhub.portfolio-decision": "src/quant_engine/portfolio_risk_contracts.py", "researchhub.risk-assessment": "src/quant_engine/portfolio_risk_contracts.py", } @@ -53,6 +55,11 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: ) assert spec["dependencies"] == [] capabilities = {item["id"]: item for item in spec["capabilities"]} + evidence_contract = capabilities["backtest-evidence-contracts"] + assert evidence_contract["status"] == "operational" + evidence_summary = evidence_contract["summary"].lower() + for term in ("performance-methodology", "without recomputation", "decision authority"): + assert term in evidence_summary portfolio_contract = capabilities["portfolio-risk-computation-contracts"] assert portfolio_contract["status"] == "operational" summary = portfolio_contract["summary"].lower() diff --git a/tests/test_performance_evidence_contract.py b/tests/test_performance_evidence_contract.py index ff68a1f..69cefe0 100644 --- a/tests/test_performance_evidence_contract.py +++ b/tests/test_performance_evidence_contract.py @@ -319,7 +319,16 @@ def test_present_evidence_is_deterministic_content_addressed_and_three_party_clo assert first.benchmark_series_digest is not None assert first.canonical_bytes() == first.to_json().encode("utf-8") assert not first.canonical_bytes().endswith(b"\n") - assert _sha256(first.canonical_bytes()) == first.document_sha256 + document_payload = first.to_dict() + document_payload.pop("document_sha256") + expected_document = json.dumps( + document_payload, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode("utf-8") + assert _sha256(expected_document) == first.document_sha256 assert PerformanceEvidenceV1.from_dict( first.to_dict(), artifact=artifact, @@ -490,6 +499,125 @@ def test_performance_table_row_and_benchmark_digest_mismatches_fail_closed() -> ) +def test_manifest_closure_covers_non_performance_artifact_tables() -> None: + _, artifact, run_ref, manifest = _case("estimable") + nav = artifact.nav + nav.loc[0, "nav"] += 0.01 + changed_artifact = replace(artifact, _nav=nav) + + with pytest.raises(PerformanceEvidenceError) as rejected: + build_performance_evidence(changed_artifact, run_ref, manifest) + + _assert_error( + rejected, + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.nav.content_digest", + ) + + +def test_row_and_benchmark_mutations_change_their_digests_and_document_identity() -> None: + original, artifact, run_ref, _ = _case("estimable") + performance = artifact.performance + performance.loc[0, "sharpe"] += 0.01 + changed_performance_artifact = replace(artifact, _performance=performance) + changed_performance_manifest = build_backtest_evidence_manifest( + run_ref, + changed_performance_artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + changed_performance = build_performance_evidence( + changed_performance_artifact, + run_ref, + changed_performance_manifest, + ) + assert changed_performance.performance_row_digest != original.performance_row_digest + assert changed_performance.performance_evidence_id != original.performance_evidence_id + + nav = artifact.nav + nav.loc[0, "benchmark_nav"] += 0.01 + changed_benchmark_artifact = replace(artifact, _nav=nav) + changed_benchmark_manifest = build_backtest_evidence_manifest( + run_ref, + changed_benchmark_artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + changed_benchmark = build_performance_evidence( + changed_benchmark_artifact, + run_ref, + changed_benchmark_manifest, + ) + assert changed_benchmark.benchmark_series_digest != original.benchmark_series_digest + assert changed_benchmark.performance_row_digest == original.performance_row_digest + assert changed_benchmark.performance_evidence_id != original.performance_evidence_id + + +def test_relative_metric_null_reasons_cannot_be_invented() -> None: + _, artifact, run_ref, _ = _case("estimable") + performance = artifact.performance + performance.loc[0, "alpha"] = float("nan") + changed_artifact = replace(artifact, _performance=performance) + changed_manifest = build_backtest_evidence_manifest( + run_ref, + changed_artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + with pytest.raises(PerformanceEvidenceError) as false_alpha_domain: + build_performance_evidence(changed_artifact, run_ref, changed_manifest) + _assert_error( + false_alpha_domain, + PerformanceEvidenceErrorCode.METRIC_INVALID, + "$.metrics.alpha.value", + ) + + _, absent_artifact, absent_run_ref, _ = _case("absent") + absent_performance = absent_artifact.performance + absent_performance.loc[0, "tracking_error"] = 0.0 + changed_absent = replace(absent_artifact, _performance=absent_performance) + changed_absent_manifest = build_backtest_evidence_manifest( + absent_run_ref, + changed_absent, + artifact_available_at="2026-01-08T02:05:00Z", + ) + with pytest.raises(PerformanceEvidenceError) as false_absence: + build_performance_evidence( + changed_absent, + absent_run_ref, + changed_absent_manifest, + ) + _assert_error( + false_absence, + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.metrics.tracking_error.availability", + ) + + +def test_artifact_builder_enforces_strict_benchmark_session_alignment() -> None: + run_ref = _run_ref() + result = _backtest_result() + misaligned = pd.Series( + [0.0, 0.01, -0.01, 0.02], + index=result.returns.index.shift(1, freq="B"), + ) + with pytest.raises(ValueError, match="matching indexes"): + build_research_run_artifact( + result, + run_id=run_ref.run_id, + strategy_id=run_ref.strategy_id, + strategy_name="Alpha Top 1", + strategy_version=run_ref.strategy_version, + engine_version="1.2.0", + code_revision=run_ref.code_revision, + data_snapshot_id=run_ref.dataset_snapshot_id, + calendar="CN-A", + timezone="Asia/Shanghai", + started_at="2026-01-08T10:00:00+08:00", + finished_at="2026-01-08T10:01:00+08:00", + parameters=PARAMETERS, + benchmark_id="000300.SH", + benchmark_returns=misaligned, + ) + + @pytest.mark.parametrize( ("column", "value", "path"), [ @@ -500,7 +628,6 @@ def test_performance_table_row_and_benchmark_digest_mismatches_fail_closed() -> ("win_rate", 1.01, "$.metrics.win_rate.value"), ("tracking_error", -0.01, "$.metrics.tracking_error.value"), ("n_trades", True, "$.metrics.trade_count.value"), - ("n_days", 2**53, "$.metrics.day_count.value"), ], ) def test_metric_domains_reject_invalid_source_values( @@ -598,4 +725,3 @@ def test_public_mapping_has_no_raw_inputs_storage_or_runtime_authority() -> None assert not (keys(payload) & forbidden_keys) for token in ("postgres://", "mysql://", "s3://", "credential", "broker"): assert token not in serialized - diff --git a/tests/test_portfolio_risk_contracts.py b/tests/test_portfolio_risk_contracts.py index bd35aff..8394f2e 100644 --- a/tests/test_portfolio_risk_contracts.py +++ b/tests/test_portfolio_risk_contracts.py @@ -952,6 +952,37 @@ def test_architecture_dependency_no_copy_and_authority_boundaries() -> None: assert not any(name.startswith(("research_results", "research_platform")) for name in imports) for candidate in ("riskfolio", "pyp", "skfolio", "cvxportfolio"): assert candidate not in source.lower() + artifact_source = (ROOT / "src" / "quant_engine" / "artifact.py").read_text( + encoding="utf-8" + ) + artifact_tree = ast.parse(artifact_source) + forbidden_artifact_authority_symbols = { + "PortfolioDecision", + "RiskAssessment", + "build_portfolio_decision", + "assess_portfolio_risk", + } + artifact_imports = { + alias.name + for node in ast.walk(artifact_tree) + if isinstance(node, ast.Import) + for alias in node.names + } | { + node.module or "" + for node in ast.walk(artifact_tree) + if isinstance(node, ast.ImportFrom) + } + artifact_names = { + node.id for node in ast.walk(artifact_tree) if isinstance(node, ast.Name) + } | { + node.attr for node in ast.walk(artifact_tree) if isinstance(node, ast.Attribute) + } + assert "quant_engine.portfolio_risk_contracts" not in artifact_imports + assert not forbidden_artifact_authority_symbols & artifact_names + assert all( + token not in artifact_source + for token in {"portfolio_risk_contracts", *forbidden_artifact_authority_symbols} + ) for owner_path in ( ROOT / "src" / "quant_engine" / "governed_pipeline.py", ROOT / "src" / "quant_engine" / "artifact.py", @@ -989,7 +1020,6 @@ def test_architecture_dependency_no_copy_and_authority_boundaries() -> None: def test_read_only_owner_dependency_lock_and_ci_hashes_match_baseline() -> None: expected = { "src/quant_engine/governed_pipeline.py": "3b334f340898db78ed869375ab532f156e8c1fee595a8318f544bebdd391049d", - "src/quant_engine/artifact.py": "e15feec412d3bfff10d8f21ca20813fc65cabc0703940371ed661896147bc379", "src/quant_engine/portfolio_construction.py": "e93d71da8d61b2047c19d4b99dace934a8cbc96d8d2b150ad62a9ceebd9163d4", "src/quant_engine/portfolio_decomp.py": "1a4f9f9aac2c46bf6ed2826d1b0d1723f06e3ce7c4b3f6479e098cbbd135bea6", "src/quant_engine/risk.py": "4a66c312d517d40f6f67bb71f438523e135645624ae78d49fa9c0fda2c02074e",