"""Closed performance-evidence contract conformance tests.""" from __future__ import annotations import copy import hashlib import json from dataclasses import replace from pathlib import Path from typing import Any import numpy as np import pandas as pd import pytest from quant_engine.artifact import ( PERFORMANCE_EVIDENCE_SCHEMA_VERSION, PERFORMANCE_METRIC_SCHEMA_ID, PERFORMANCE_METHODOLOGY_ID, BacktestEvidenceManifest, EvidenceQualification, PerformanceEvidenceError, PerformanceEvidenceErrorCode, PerformanceEvidenceV1, PerformanceMetricAvailability, ResearchRunArtifact, build_backtest_evidence_manifest, build_performance_evidence, build_research_run_artifact, ) from quant_engine.execution import ExecutionConfig from quant_engine.factor_contracts import ( ActorIdentity, AvailabilityMode, Causation, DataFoundationEnvelope, DatasetSnapshotEnvelope, FactorInput, FactorSetRef, InputBinding, OutputArtifactRef, OutputCoverage, OutputQuality, OutputQualityCheck, ProducerIdentity, ViewAvailability, canonical_json_bytes, factor_definition_from_alpha158, factor_input_schema_digest, ) from quant_engine.governed_pipeline import BacktestRunRef from quant_engine.metrics import TRADING_DAYS_PER_YEAR, benchmark_summary, summary from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research ROOT = Path(__file__).resolve().parents[1] FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json" PERFORMANCE_FIXTURE = ( ROOT / "tests" / "fixtures" / "performance-evidence-v1.golden.json" ) VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c" VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef" CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078" ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275" PARAMETERS = {"lag_sessions": 1, "top_k": 1} def _sha256(value: bytes) -> str: return f"sha256:{hashlib.sha256(value).hexdigest()}" def _accepted_authorities() -> tuple[ DatasetSnapshotEnvelope, DataFoundationEnvelope, FactorSetRef, ]: fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8")) snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"]) foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"]) factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume")) definition = factor_definition_from_alpha158( "alpha_005", version="1.0.0", parameters={}, inputs=(factor_input,), implementation_digest="sha256:" + "1" * 64, input_schema_digest=factor_input_schema_digest((factor_input,)), valid_from="2026-01-01T00:00:00.000000Z", valid_until="2027-01-01T00:00:00Z", warmup_sessions=10, lag_sessions=1, producer=ProducerIdentity("quant_engine", "1.0.0"), code_revision="c" * 40, ) output_schema_bytes = canonical_json_bytes(fixture["output_schema"]) output_content_bytes = canonical_json_bytes(fixture["output_content"]) artifact_ref = OutputArtifactRef.create( schema_digest=_sha256(output_schema_bytes), content_digest=_sha256(output_content_bytes), ) factor_set = FactorSetRef.create( definitions=(definition,), dataset_snapshot=snapshot, foundation=foundation, selected_view_ref_ids=(VIEW_REF_ID,), input_bindings=( InputBinding( definition.definition_id, "market", VIEW_REF_ID, VIEW_SCHEMA_DIGEST, ), ), view_availability=( ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64), ), output_quality=OutputQuality( "passed", (OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),), ), output_coverage=OutputCoverage( "complete", 1, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64, ), output_schema_bytes=output_schema_bytes, output_content_bytes=output_content_bytes, output_artifact_ref=artifact_ref, availability_mode=AvailabilityMode.AS_AVAILABLE, evaluation_at="2026-01-03T11:00:00Z", computed_at="2026-01-03T10:15:00Z", artifact_available_at="2026-01-03T10:20:00Z", producer=ProducerIdentity("quant_engine", "1.0.0"), code_revision="c" * 40, actor=ActorIdentity("service", "factor_worker_v1"), correlation_id="research_run_001", causation=Causation("foundation", foundation.foundation_id), evidence_scope="synthetic_fixture", decision_eligible=False, ) return snapshot, foundation, factor_set def _configuration_digest() -> str: return _sha256( json.dumps( PARAMETERS, ensure_ascii=False, sort_keys=True, separators=(",", ":"), allow_nan=False, ).encode("utf-8") ) def _run_ref(**overrides: Any) -> BacktestRunRef: snapshot, foundation, factor_set = _accepted_authorities() arguments: dict[str, Any] = { "dataset_snapshot": snapshot, "foundation": foundation, "factor_set": factor_set, "universe_digest": "sha256:" + "5" * 64, "trading_calendar_revision_ids": (CALENDAR_REVISION_ID,), "corporate_action_revision_ids": (ACTION_REVISION_ID,), "strategy_id": "alpha-top1", "strategy_version": "1.0.0", "strategy_digest": "sha256:" + "6" * 64, "execution_model_version": "1.0.0", "execution_model_digest": "sha256:" + "7" * 64, "cost_model_version": "1.0.0", "cost_model_digest": "sha256:" + "8" * 64, "random_seed": 7, "code_revision": "d" * 40, "environment_lock_digest": "sha256:" + "9" * 64, "configuration_digest": _configuration_digest(), "evaluation_at": "2026-01-08T01:00:00Z", "computed_at": "2026-01-08T02:00:00Z", } arguments.update(overrides) return BacktestRunRef.create(**arguments) def _backtest_result() -> FactorBacktestResult: dates = pd.date_range("2026-01-05", periods=4, freq="B") scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2]) opens = pd.DataFrame( {"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]}, index=dates, ) closes = pd.DataFrame( {"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]}, index=dates, ) return run_factor_backtest_research( scores, opens, closes, top_k=1, execution_price_field="open", valuation_price_field="close", initial_cash=1_000.0, config=ExecutionConfig( commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0, ), ) def _artifact( run_ref: BacktestRunRef, benchmark_kind: str, ) -> tuple[ResearchRunArtifact, FactorBacktestResult]: result = _backtest_result() benchmark_id: str | None benchmark_returns: pd.Series | None if benchmark_kind == "absent": benchmark_id = None benchmark_returns = None elif benchmark_kind == "estimable": benchmark_id = "000300.SH" benchmark_returns = pd.Series( [0.0, 0.01, -0.01, 0.02], index=result.returns.index, name="benchmark_return", ) elif benchmark_kind == "zero_active_variance": benchmark_id = "000300.SH" benchmark_returns = result.returns.rename("benchmark_return") elif benchmark_kind == "zero_benchmark_variance": benchmark_id = "000300.SH" benchmark_returns = pd.Series( np.zeros(len(result.returns)), index=result.returns.index, name="benchmark_return", ) else: raise AssertionError(f"unknown benchmark_kind: {benchmark_kind}") artifact = build_research_run_artifact( result, run_id=run_ref.run_id, strategy_id=run_ref.strategy_id, strategy_name="Alpha Top 1", strategy_version=run_ref.strategy_version, engine_version="1.2.0", code_revision=run_ref.code_revision, data_snapshot_id=run_ref.dataset_snapshot_id, calendar="CN-A", timezone="Asia/Shanghai", started_at="2026-01-08T10:00:00+08:00", finished_at="2026-01-08T10:01:00+08:00", parameters=PARAMETERS, benchmark_id=benchmark_id, benchmark_returns=benchmark_returns, ) return artifact, result def _case( benchmark_kind: str, ) -> tuple[PerformanceEvidenceV1, ResearchRunArtifact, BacktestRunRef, BacktestEvidenceManifest]: run_ref = _run_ref() artifact, _ = _artifact(run_ref, benchmark_kind) manifest = build_backtest_evidence_manifest( run_ref, artifact, artifact_available_at="2026-01-08T02:05:00Z", qualification=EvidenceQualification.CONTRACT_QUALIFIED, ) return ( build_performance_evidence(artifact, run_ref, manifest), artifact, run_ref, manifest, ) def _metric_map(evidence: PerformanceEvidenceV1) -> dict[str, Any]: return {metric.key: metric for metric in evidence.metrics} def _mutate_frozen(value: Any, field: str, replacement: object) -> Any: changed = copy.copy(value) object.__setattr__(changed, field, replacement) return changed def _assert_error( error: pytest.ExceptionInfo[PerformanceEvidenceError], code: PerformanceEvidenceErrorCode, path: str, ) -> None: assert error.value.code is code assert error.value.path == path def test_present_evidence_is_deterministic_content_addressed_and_three_party_closed() -> None: first, artifact, run_ref, manifest = _case("estimable") second = build_performance_evidence(artifact, run_ref, manifest) assert first == second assert first.schema_version == PERFORMANCE_EVIDENCE_SCHEMA_VERSION assert first.performance_evidence_id.startswith("rhperformanceevidencev1:sha256:") assert first.document_sha256.startswith("sha256:") assert first.authority == "quant_engine" assert first.scope == "offline_research_only" assert first.run_id == first.backtest_run_ref_id == run_ref.run_id == manifest.run_id assert first.backtest_evidence_manifest_id == manifest.manifest_id assert first.backtest_evidence_manifest_evidence_digest == manifest.evidence_digest assert first.backtest_evidence_qualification == "contract_qualified" assert first.research_artifact_content_digest == f"sha256:{artifact.content_sha256}" assert first.performance_table_logical_name == "performance" assert first.performance_table_row_count == 1 assert first.performance_row_digest.startswith("sha256:") assert first.benchmark_series_digest is not None assert first.canonical_bytes() == first.to_json().encode("utf-8") assert not first.canonical_bytes().endswith(b"\n") document_payload = first.to_dict() document_payload.pop("document_sha256") expected_document = json.dumps( document_payload, ensure_ascii=False, sort_keys=True, separators=(",", ":"), allow_nan=False, ).encode("utf-8") assert _sha256(expected_document) == first.document_sha256 assert PerformanceEvidenceV1.from_dict( first.to_dict(), artifact=artifact, run_ref=run_ref, evidence_manifest=manifest, ) == first def test_methodology_and_metrics_bind_the_actual_artifact_builder_path() -> None: evidence, artifact, _, _ = _case("estimable") result = _backtest_result() expected_absolute = summary(result.returns, rf=0.0) benchmark = artifact.nav.set_index("trade_date")["benchmark_return"] benchmark.index = result.returns.index expected_relative = benchmark_summary( result.returns, benchmark, risk_free_daily=0.0, annualization=TRADING_DAYS_PER_YEAR, ) metrics = _metric_map(evidence) assert evidence.methodology.methodology_id == PERFORMANCE_METHODOLOGY_ID assert evidence.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID assert evidence.methodology.return_type == "simple" assert evidence.methodology.source_frequency == "1d" assert evidence.methodology.periods_per_year == TRADING_DAYS_PER_YEAR == 252 assert evidence.methodology.annual_risk_free == 0.0 assert evidence.methodology.benchmark_risk_free_daily == 0.0 assert evidence.methodology.benchmark_alignment == "exact_session_index" assert metrics["annualized_return"].value == pytest.approx( expected_absolute["ann_return"] ) assert metrics["sharpe_ratio"].value == pytest.approx(expected_absolute["sharpe"]) assert metrics["tracking_error"].value == pytest.approx( expected_relative["tracking_error"] ) assert metrics["alpha"].value == pytest.approx(expected_relative["alpha"]) assert all(metric.methodology_id == PERFORMANCE_METHODOLOGY_ID for metric in metrics.values()) assert all(metric.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID for metric in metrics.values()) def test_relative_metric_availability_is_closed_for_present_absent_and_unestimable() -> None: present, *_ = _case("estimable") absent, *_ = _case("absent") zero_active, *_ = _case("zero_active_variance") zero_benchmark, *_ = _case("zero_benchmark_variance") present_metrics = _metric_map(present) assert all( present_metrics[key].availability is PerformanceMetricAvailability.AVAILABLE for key in ("tracking_error", "information_ratio", "alpha", "beta") ) absent_metrics = _metric_map(absent) assert absent.benchmark_series_digest is None assert absent.benchmark_id == "" assert absent.benchmark_alignment_policy == "none" assert all( absent_metrics[key].value is None and absent_metrics[key].availability is PerformanceMetricAvailability.BENCHMARK_ABSENT for key in ("tracking_error", "information_ratio", "alpha", "beta") ) zero_active_metrics = _metric_map(zero_active) assert zero_active_metrics["tracking_error"].value == pytest.approx(0.0) assert ( zero_active_metrics["information_ratio"].availability is PerformanceMetricAvailability.NOT_ESTIMABLE_ACTIVE_VARIANCE ) assert zero_active_metrics["information_ratio"].value is None zero_benchmark_metrics = _metric_map(zero_benchmark) assert np.isfinite(zero_benchmark_metrics["tracking_error"].value) for key in ("alpha", "beta"): assert zero_benchmark_metrics[key].value is None assert ( zero_benchmark_metrics[key].availability is PerformanceMetricAvailability.NOT_ESTIMABLE_BENCHMARK_VARIANCE ) def test_golden_covers_present_absent_and_both_unestimable_states() -> None: expected = { "schema_version": 1, "source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c", "source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603", "cases": { name: _case(name)[0].to_dict() for name in ( "estimable", "zero_active_variance", "zero_benchmark_variance", "absent", ) }, } assert json.loads(PERFORMANCE_FIXTURE.read_text(encoding="utf-8")) == expected @pytest.mark.parametrize( ("owner", "field", "replacement", "code", "path"), [ ( "run_ref", "run_id", "rhbacktestrunv1:sha256:" + "0" * 64, PerformanceEvidenceErrorCode.IDENTITY_MISMATCH, "$.backtest_run_ref.run_id", ), ( "manifest", "manifest_id", "rhbacktestevidencev1:sha256:" + "0" * 64, PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, "$.backtest_evidence_manifest.manifest_id", ), ( "manifest", "qualification", EvidenceQualification.EXPLORATORY, PerformanceEvidenceErrorCode.AUTHORITY_REJECTED, "$.backtest_evidence_manifest.qualification", ), ], ) def test_owner_identity_and_authority_mismatches_fail_closed( owner: str, field: str, replacement: object, code: PerformanceEvidenceErrorCode, path: str, ) -> None: _, artifact, run_ref, manifest = _case("estimable") changed_run_ref = _mutate_frozen(run_ref, field, replacement) if owner == "run_ref" else run_ref changed_manifest = ( _mutate_frozen(manifest, field, replacement) if owner == "manifest" else manifest ) with pytest.raises(PerformanceEvidenceError) as rejected: build_performance_evidence(artifact, changed_run_ref, changed_manifest) _assert_error(rejected, code, path) def test_performance_table_row_and_benchmark_digest_mismatches_fail_closed() -> None: evidence, artifact, run_ref, manifest = _case("estimable") performance = artifact.performance performance.loc[0, "n_days"] += 1 changed_artifact = replace(artifact, _performance=performance) with pytest.raises(PerformanceEvidenceError) as table_mismatch: build_performance_evidence(changed_artifact, run_ref, manifest) _assert_error( table_mismatch, PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, "$.artifact.tables.performance.content_digest", ) payload = evidence.to_dict() payload["benchmark_series_digest"] = "sha256:" + "0" * 64 with pytest.raises(PerformanceEvidenceError) as benchmark_mismatch: PerformanceEvidenceV1.from_dict( payload, artifact=artifact, run_ref=run_ref, evidence_manifest=manifest, ) _assert_error( benchmark_mismatch, PerformanceEvidenceErrorCode.BENCHMARK_INVALID, "$.benchmark_series_digest", ) def test_manifest_closure_covers_non_performance_artifact_tables() -> None: _, artifact, run_ref, manifest = _case("estimable") nav = artifact.nav nav.loc[0, "nav"] += 0.01 changed_artifact = replace(artifact, _nav=nav) with pytest.raises(PerformanceEvidenceError) as rejected: build_performance_evidence(changed_artifact, run_ref, manifest) _assert_error( rejected, PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, "$.artifact.tables.nav.content_digest", ) def test_row_and_benchmark_mutations_change_their_digests_and_document_identity() -> None: original, artifact, run_ref, _ = _case("estimable") performance = artifact.performance performance.loc[0, "sharpe"] += 0.01 changed_performance_artifact = replace(artifact, _performance=performance) changed_performance_manifest = build_backtest_evidence_manifest( run_ref, changed_performance_artifact, artifact_available_at="2026-01-08T02:05:00Z", ) changed_performance = build_performance_evidence( changed_performance_artifact, run_ref, changed_performance_manifest, ) assert changed_performance.performance_row_digest != original.performance_row_digest assert changed_performance.performance_evidence_id != original.performance_evidence_id nav = artifact.nav nav.loc[0, "benchmark_nav"] += 0.01 changed_benchmark_artifact = replace(artifact, _nav=nav) changed_benchmark_manifest = build_backtest_evidence_manifest( run_ref, changed_benchmark_artifact, artifact_available_at="2026-01-08T02:05:00Z", ) changed_benchmark = build_performance_evidence( changed_benchmark_artifact, run_ref, changed_benchmark_manifest, ) assert changed_benchmark.benchmark_series_digest != original.benchmark_series_digest assert changed_benchmark.performance_row_digest == original.performance_row_digest assert changed_benchmark.performance_evidence_id != original.performance_evidence_id def test_relative_metric_null_reasons_cannot_be_invented() -> None: _, artifact, run_ref, _ = _case("estimable") performance = artifact.performance performance.loc[0, "alpha"] = float("nan") changed_artifact = replace(artifact, _performance=performance) changed_manifest = build_backtest_evidence_manifest( run_ref, changed_artifact, artifact_available_at="2026-01-08T02:05:00Z", ) with pytest.raises(PerformanceEvidenceError) as false_alpha_domain: build_performance_evidence(changed_artifact, run_ref, changed_manifest) _assert_error( false_alpha_domain, PerformanceEvidenceErrorCode.METRIC_INVALID, "$.metrics.alpha.value", ) _, absent_artifact, absent_run_ref, _ = _case("absent") absent_performance = absent_artifact.performance absent_performance.loc[0, "tracking_error"] = 0.0 changed_absent = replace(absent_artifact, _performance=absent_performance) changed_absent_manifest = build_backtest_evidence_manifest( absent_run_ref, changed_absent, artifact_available_at="2026-01-08T02:05:00Z", ) with pytest.raises(PerformanceEvidenceError) as false_absence: build_performance_evidence( changed_absent, absent_run_ref, changed_absent_manifest, ) _assert_error( false_absence, PerformanceEvidenceErrorCode.BENCHMARK_INVALID, "$.metrics.tracking_error.availability", ) def test_artifact_builder_enforces_strict_benchmark_session_alignment() -> None: run_ref = _run_ref() result = _backtest_result() misaligned = pd.Series( [0.0, 0.01, -0.01, 0.02], index=result.returns.index.shift(1, freq="B"), ) with pytest.raises(ValueError, match="matching indexes"): build_research_run_artifact( result, run_id=run_ref.run_id, strategy_id=run_ref.strategy_id, strategy_name="Alpha Top 1", strategy_version=run_ref.strategy_version, engine_version="1.2.0", code_revision=run_ref.code_revision, data_snapshot_id=run_ref.dataset_snapshot_id, calendar="CN-A", timezone="Asia/Shanghai", started_at="2026-01-08T10:00:00+08:00", finished_at="2026-01-08T10:01:00+08:00", parameters=PARAMETERS, benchmark_id="000300.SH", benchmark_returns=misaligned, ) @pytest.mark.parametrize( ("column", "value", "path"), [ ("total_ret", -1.01, "$.metrics.total_return.value"), ("ann_ret", -1.01, "$.metrics.annualized_return.value"), ("ann_volatility", -0.01, "$.metrics.annualized_volatility.value"), ("max_dd", 0.01, "$.metrics.maximum_drawdown.value"), ("win_rate", 1.01, "$.metrics.win_rate.value"), ("tracking_error", -0.01, "$.metrics.tracking_error.value"), ("n_trades", True, "$.metrics.trade_count.value"), ], ) def test_metric_domains_reject_invalid_source_values( column: str, value: object, path: str, ) -> None: _, artifact, run_ref, _ = _case("estimable") performance = artifact.performance.astype(object) performance.at[0, column] = value changed_artifact = replace(artifact, _performance=performance) changed_manifest = build_backtest_evidence_manifest( run_ref, changed_artifact, artifact_available_at="2026-01-08T02:05:00Z", ) with pytest.raises(PerformanceEvidenceError) as rejected: build_performance_evidence(changed_artifact, run_ref, changed_manifest) _assert_error(rejected, PerformanceEvidenceErrorCode.METRIC_INVALID, path) def test_closed_parser_rejects_unknown_non_ascii_non_finite_bool_and_unsafe_integer() -> None: evidence, artifact, run_ref, manifest = _case("estimable") mutations: list[tuple[dict[str, Any], PerformanceEvidenceErrorCode, str]] = [] unknown = evidence.to_dict() unknown["unexpected"] = "value" mutations.append((unknown, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.unexpected")) non_ascii = evidence.to_dict() non_ascii["métric"] = "value" mutations.append((non_ascii, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.métric")) non_finite = evidence.to_dict() non_finite["metrics"][0]["value"] = float("inf") mutations.append( (non_finite, PerformanceEvidenceErrorCode.METRIC_INVALID, "$.metrics[0].value") ) bool_number = evidence.to_dict() bool_number["methodology"]["periods_per_year"] = True mutations.append( ( bool_number, PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH, "$.methodology.periods_per_year", ) ) unsafe = evidence.to_dict() unsafe["performance_table_row_count"] = 2**53 mutations.append( ( unsafe, PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, "$.performance_table_row_count", ) ) for payload, code, path in mutations: with pytest.raises(PerformanceEvidenceError) as rejected: PerformanceEvidenceV1.from_dict( payload, artifact=artifact, run_ref=run_ref, evidence_manifest=manifest, ) _assert_error(rejected, code, path) def test_public_mapping_has_no_raw_inputs_storage_or_runtime_authority() -> None: evidence, *_ = _case("estimable") payload = evidence.to_dict() serialized = evidence.to_json().lower() forbidden_keys = { "parameters", "params_json", "returns", "nav", "benchmark_series", "table_bytes", "locator", "uri", "credential", "decision_eligible", "publication_eligible", "paper_trading", "live_trading", "investment_advice", } def keys(value: object) -> set[str]: if isinstance(value, dict): return set(value) | {key for item in value.values() for key in keys(item)} if isinstance(value, list): return {key for item in value for key in keys(item)} return set() assert not (keys(payload) & forbidden_keys) for token in ("postgres://", "mysql://", "s3://", "credential", "broker"): assert token not in serialized