From e7a6e20826f85bb785aa5b78a1e8f98908d609f7 Mon Sep 17 00:00:00 2001 From: ao gong <41768719+ageorge156@users.noreply.github.com> Date: Tue, 1 Sep 2026 22:08:01 +0800 Subject: [PATCH] test(performance): define S3.1 evidence contract RED --- tests/test_performance_evidence_contract.py | 601 ++++++++++++++++++++ 1 file changed, 601 insertions(+) create mode 100644 tests/test_performance_evidence_contract.py diff --git a/tests/test_performance_evidence_contract.py b/tests/test_performance_evidence_contract.py new file mode 100644 index 0000000..ff68a1f --- /dev/null +++ b/tests/test_performance_evidence_contract.py @@ -0,0 +1,601 @@ +"""Closed performance-evidence contract conformance tests.""" + +from __future__ import annotations + +import copy +import hashlib +import json +from dataclasses import replace +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd +import pytest + +from quant_engine.artifact import ( + PERFORMANCE_EVIDENCE_SCHEMA_VERSION, + PERFORMANCE_METRIC_SCHEMA_ID, + PERFORMANCE_METHODOLOGY_ID, + BacktestEvidenceManifest, + EvidenceQualification, + PerformanceEvidenceError, + PerformanceEvidenceErrorCode, + PerformanceEvidenceV1, + PerformanceMetricAvailability, + ResearchRunArtifact, + build_backtest_evidence_manifest, + build_performance_evidence, + build_research_run_artifact, +) +from quant_engine.execution import ExecutionConfig +from quant_engine.factor_contracts import ( + ActorIdentity, + AvailabilityMode, + Causation, + DataFoundationEnvelope, + DatasetSnapshotEnvelope, + FactorInput, + FactorSetRef, + InputBinding, + OutputArtifactRef, + OutputCoverage, + OutputQuality, + OutputQualityCheck, + ProducerIdentity, + ViewAvailability, + canonical_json_bytes, + factor_definition_from_alpha158, + factor_input_schema_digest, +) +from quant_engine.governed_pipeline import BacktestRunRef +from quant_engine.metrics import TRADING_DAYS_PER_YEAR, benchmark_summary, summary +from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research + + +ROOT = Path(__file__).resolve().parents[1] +FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json" +PERFORMANCE_FIXTURE = ( + ROOT / "tests" / "fixtures" / "performance-evidence-v1.golden.json" +) +VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c" +VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef" +CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078" +ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275" +PARAMETERS = {"lag_sessions": 1, "top_k": 1} + + +def _sha256(value: bytes) -> str: + return f"sha256:{hashlib.sha256(value).hexdigest()}" + + +def _accepted_authorities() -> tuple[ + DatasetSnapshotEnvelope, + DataFoundationEnvelope, + FactorSetRef, +]: + fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8")) + snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"]) + foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"]) + factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume")) + definition = factor_definition_from_alpha158( + "alpha_005", + version="1.0.0", + parameters={}, + inputs=(factor_input,), + implementation_digest="sha256:" + "1" * 64, + input_schema_digest=factor_input_schema_digest((factor_input,)), + valid_from="2026-01-01T00:00:00.000000Z", + valid_until="2027-01-01T00:00:00Z", + warmup_sessions=10, + lag_sessions=1, + producer=ProducerIdentity("quant_engine", "1.0.0"), + code_revision="c" * 40, + ) + output_schema_bytes = canonical_json_bytes(fixture["output_schema"]) + output_content_bytes = canonical_json_bytes(fixture["output_content"]) + artifact_ref = OutputArtifactRef.create( + schema_digest=_sha256(output_schema_bytes), + content_digest=_sha256(output_content_bytes), + ) + factor_set = FactorSetRef.create( + definitions=(definition,), + dataset_snapshot=snapshot, + foundation=foundation, + selected_view_ref_ids=(VIEW_REF_ID,), + input_bindings=( + InputBinding( + definition.definition_id, + "market", + VIEW_REF_ID, + VIEW_SCHEMA_DIGEST, + ), + ), + view_availability=( + ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64), + ), + output_quality=OutputQuality( + "passed", + (OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),), + ), + output_coverage=OutputCoverage( + "complete", + 1, + 1, + "row", + "alpha_005.cn_a", + "sha256:" + "4" * 64, + ), + output_schema_bytes=output_schema_bytes, + output_content_bytes=output_content_bytes, + output_artifact_ref=artifact_ref, + availability_mode=AvailabilityMode.AS_AVAILABLE, + evaluation_at="2026-01-03T11:00:00Z", + computed_at="2026-01-03T10:15:00Z", + artifact_available_at="2026-01-03T10:20:00Z", + producer=ProducerIdentity("quant_engine", "1.0.0"), + code_revision="c" * 40, + actor=ActorIdentity("service", "factor_worker_v1"), + correlation_id="research_run_001", + causation=Causation("foundation", foundation.foundation_id), + evidence_scope="synthetic_fixture", + decision_eligible=False, + ) + return snapshot, foundation, factor_set + + +def _configuration_digest() -> str: + return _sha256( + json.dumps( + PARAMETERS, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode("utf-8") + ) + + +def _run_ref(**overrides: Any) -> BacktestRunRef: + snapshot, foundation, factor_set = _accepted_authorities() + arguments: dict[str, Any] = { + "dataset_snapshot": snapshot, + "foundation": foundation, + "factor_set": factor_set, + "universe_digest": "sha256:" + "5" * 64, + "trading_calendar_revision_ids": (CALENDAR_REVISION_ID,), + "corporate_action_revision_ids": (ACTION_REVISION_ID,), + "strategy_id": "alpha-top1", + "strategy_version": "1.0.0", + "strategy_digest": "sha256:" + "6" * 64, + "execution_model_version": "1.0.0", + "execution_model_digest": "sha256:" + "7" * 64, + "cost_model_version": "1.0.0", + "cost_model_digest": "sha256:" + "8" * 64, + "random_seed": 7, + "code_revision": "d" * 40, + "environment_lock_digest": "sha256:" + "9" * 64, + "configuration_digest": _configuration_digest(), + "evaluation_at": "2026-01-08T01:00:00Z", + "computed_at": "2026-01-08T02:00:00Z", + } + arguments.update(overrides) + return BacktestRunRef.create(**arguments) + + +def _backtest_result() -> FactorBacktestResult: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2]) + opens = pd.DataFrame( + {"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]}, + index=dates, + ) + return run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ), + ) + + +def _artifact( + run_ref: BacktestRunRef, + benchmark_kind: str, +) -> tuple[ResearchRunArtifact, FactorBacktestResult]: + result = _backtest_result() + benchmark_id: str | None + benchmark_returns: pd.Series | None + if benchmark_kind == "absent": + benchmark_id = None + benchmark_returns = None + elif benchmark_kind == "estimable": + benchmark_id = "000300.SH" + benchmark_returns = pd.Series( + [0.0, 0.01, -0.01, 0.02], + index=result.returns.index, + name="benchmark_return", + ) + elif benchmark_kind == "zero_active_variance": + benchmark_id = "000300.SH" + benchmark_returns = result.returns.rename("benchmark_return") + elif benchmark_kind == "zero_benchmark_variance": + benchmark_id = "000300.SH" + benchmark_returns = pd.Series( + np.zeros(len(result.returns)), + index=result.returns.index, + name="benchmark_return", + ) + else: + raise AssertionError(f"unknown benchmark_kind: {benchmark_kind}") + artifact = build_research_run_artifact( + result, + run_id=run_ref.run_id, + strategy_id=run_ref.strategy_id, + strategy_name="Alpha Top 1", + strategy_version=run_ref.strategy_version, + engine_version="1.2.0", + code_revision=run_ref.code_revision, + data_snapshot_id=run_ref.dataset_snapshot_id, + calendar="CN-A", + timezone="Asia/Shanghai", + started_at="2026-01-08T10:00:00+08:00", + finished_at="2026-01-08T10:01:00+08:00", + parameters=PARAMETERS, + benchmark_id=benchmark_id, + benchmark_returns=benchmark_returns, + ) + return artifact, result + + +def _case( + benchmark_kind: str, +) -> tuple[PerformanceEvidenceV1, ResearchRunArtifact, BacktestRunRef, BacktestEvidenceManifest]: + run_ref = _run_ref() + artifact, _ = _artifact(run_ref, benchmark_kind) + manifest = build_backtest_evidence_manifest( + run_ref, + artifact, + artifact_available_at="2026-01-08T02:05:00Z", + qualification=EvidenceQualification.CONTRACT_QUALIFIED, + ) + return ( + build_performance_evidence(artifact, run_ref, manifest), + artifact, + run_ref, + manifest, + ) + + +def _metric_map(evidence: PerformanceEvidenceV1) -> dict[str, Any]: + return {metric.key: metric for metric in evidence.metrics} + + +def _mutate_frozen(value: Any, field: str, replacement: object) -> Any: + changed = copy.copy(value) + object.__setattr__(changed, field, replacement) + return changed + + +def _assert_error( + error: pytest.ExceptionInfo[PerformanceEvidenceError], + code: PerformanceEvidenceErrorCode, + path: str, +) -> None: + assert error.value.code is code + assert error.value.path == path + + +def test_present_evidence_is_deterministic_content_addressed_and_three_party_closed() -> None: + first, artifact, run_ref, manifest = _case("estimable") + second = build_performance_evidence(artifact, run_ref, manifest) + + assert first == second + assert first.schema_version == PERFORMANCE_EVIDENCE_SCHEMA_VERSION + assert first.performance_evidence_id.startswith("rhperformanceevidencev1:sha256:") + assert first.document_sha256.startswith("sha256:") + assert first.authority == "quant_engine" + assert first.scope == "offline_research_only" + assert first.run_id == first.backtest_run_ref_id == run_ref.run_id == manifest.run_id + assert first.backtest_evidence_manifest_id == manifest.manifest_id + assert first.backtest_evidence_manifest_evidence_digest == manifest.evidence_digest + assert first.backtest_evidence_qualification == "contract_qualified" + assert first.research_artifact_content_digest == f"sha256:{artifact.content_sha256}" + assert first.performance_table_logical_name == "performance" + assert first.performance_table_row_count == 1 + assert first.performance_row_digest.startswith("sha256:") + assert first.benchmark_series_digest is not None + assert first.canonical_bytes() == first.to_json().encode("utf-8") + assert not first.canonical_bytes().endswith(b"\n") + assert _sha256(first.canonical_bytes()) == first.document_sha256 + assert PerformanceEvidenceV1.from_dict( + first.to_dict(), + artifact=artifact, + run_ref=run_ref, + evidence_manifest=manifest, + ) == first + + +def test_methodology_and_metrics_bind_the_actual_artifact_builder_path() -> None: + evidence, artifact, _, _ = _case("estimable") + result = _backtest_result() + expected_absolute = summary(result.returns, rf=0.0) + benchmark = artifact.nav.set_index("trade_date")["benchmark_return"] + benchmark.index = result.returns.index + expected_relative = benchmark_summary( + result.returns, + benchmark, + risk_free_daily=0.0, + annualization=TRADING_DAYS_PER_YEAR, + ) + metrics = _metric_map(evidence) + + assert evidence.methodology.methodology_id == PERFORMANCE_METHODOLOGY_ID + assert evidence.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID + assert evidence.methodology.return_type == "simple" + assert evidence.methodology.source_frequency == "1d" + assert evidence.methodology.periods_per_year == TRADING_DAYS_PER_YEAR == 252 + assert evidence.methodology.annual_risk_free == 0.0 + assert evidence.methodology.benchmark_risk_free_daily == 0.0 + assert evidence.methodology.benchmark_alignment == "exact_session_index" + assert metrics["annualized_return"].value == pytest.approx( + expected_absolute["ann_return"] + ) + assert metrics["sharpe_ratio"].value == pytest.approx(expected_absolute["sharpe"]) + assert metrics["tracking_error"].value == pytest.approx( + expected_relative["tracking_error"] + ) + assert metrics["alpha"].value == pytest.approx(expected_relative["alpha"]) + assert all(metric.methodology_id == PERFORMANCE_METHODOLOGY_ID for metric in metrics.values()) + assert all(metric.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID for metric in metrics.values()) + + +def test_relative_metric_availability_is_closed_for_present_absent_and_unestimable() -> None: + present, *_ = _case("estimable") + absent, *_ = _case("absent") + zero_active, *_ = _case("zero_active_variance") + zero_benchmark, *_ = _case("zero_benchmark_variance") + + present_metrics = _metric_map(present) + assert all( + present_metrics[key].availability is PerformanceMetricAvailability.AVAILABLE + for key in ("tracking_error", "information_ratio", "alpha", "beta") + ) + absent_metrics = _metric_map(absent) + assert absent.benchmark_series_digest is None + assert absent.benchmark_id == "" + assert absent.benchmark_alignment_policy == "none" + assert all( + absent_metrics[key].value is None + and absent_metrics[key].availability + is PerformanceMetricAvailability.BENCHMARK_ABSENT + for key in ("tracking_error", "information_ratio", "alpha", "beta") + ) + zero_active_metrics = _metric_map(zero_active) + assert zero_active_metrics["tracking_error"].value == pytest.approx(0.0) + assert ( + zero_active_metrics["information_ratio"].availability + is PerformanceMetricAvailability.NOT_ESTIMABLE_ACTIVE_VARIANCE + ) + assert zero_active_metrics["information_ratio"].value is None + zero_benchmark_metrics = _metric_map(zero_benchmark) + assert np.isfinite(zero_benchmark_metrics["tracking_error"].value) + for key in ("alpha", "beta"): + assert zero_benchmark_metrics[key].value is None + assert ( + zero_benchmark_metrics[key].availability + is PerformanceMetricAvailability.NOT_ESTIMABLE_BENCHMARK_VARIANCE + ) + + +def test_golden_covers_present_absent_and_both_unestimable_states() -> None: + expected = { + "schema_version": 1, + "source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c", + "source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603", + "cases": { + name: _case(name)[0].to_dict() + for name in ( + "estimable", + "zero_active_variance", + "zero_benchmark_variance", + "absent", + ) + }, + } + assert json.loads(PERFORMANCE_FIXTURE.read_text(encoding="utf-8")) == expected + + +@pytest.mark.parametrize( + ("owner", "field", "replacement", "code", "path"), + [ + ( + "run_ref", + "run_id", + "rhbacktestrunv1:sha256:" + "0" * 64, + PerformanceEvidenceErrorCode.IDENTITY_MISMATCH, + "$.backtest_run_ref.run_id", + ), + ( + "manifest", + "manifest_id", + "rhbacktestevidencev1:sha256:" + "0" * 64, + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.backtest_evidence_manifest.manifest_id", + ), + ( + "manifest", + "qualification", + EvidenceQualification.EXPLORATORY, + PerformanceEvidenceErrorCode.AUTHORITY_REJECTED, + "$.backtest_evidence_manifest.qualification", + ), + ], +) +def test_owner_identity_and_authority_mismatches_fail_closed( + owner: str, + field: str, + replacement: object, + code: PerformanceEvidenceErrorCode, + path: str, +) -> None: + _, artifact, run_ref, manifest = _case("estimable") + changed_run_ref = _mutate_frozen(run_ref, field, replacement) if owner == "run_ref" else run_ref + changed_manifest = ( + _mutate_frozen(manifest, field, replacement) if owner == "manifest" else manifest + ) + with pytest.raises(PerformanceEvidenceError) as rejected: + build_performance_evidence(artifact, changed_run_ref, changed_manifest) + _assert_error(rejected, code, path) + + +def test_performance_table_row_and_benchmark_digest_mismatches_fail_closed() -> None: + evidence, artifact, run_ref, manifest = _case("estimable") + performance = artifact.performance + performance.loc[0, "n_days"] += 1 + changed_artifact = replace(artifact, _performance=performance) + with pytest.raises(PerformanceEvidenceError) as table_mismatch: + build_performance_evidence(changed_artifact, run_ref, manifest) + _assert_error( + table_mismatch, + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.artifact.tables.performance.content_digest", + ) + + payload = evidence.to_dict() + payload["benchmark_series_digest"] = "sha256:" + "0" * 64 + with pytest.raises(PerformanceEvidenceError) as benchmark_mismatch: + PerformanceEvidenceV1.from_dict( + payload, + artifact=artifact, + run_ref=run_ref, + evidence_manifest=manifest, + ) + _assert_error( + benchmark_mismatch, + PerformanceEvidenceErrorCode.BENCHMARK_INVALID, + "$.benchmark_series_digest", + ) + + +@pytest.mark.parametrize( + ("column", "value", "path"), + [ + ("total_ret", -1.01, "$.metrics.total_return.value"), + ("ann_ret", -1.01, "$.metrics.annualized_return.value"), + ("ann_volatility", -0.01, "$.metrics.annualized_volatility.value"), + ("max_dd", 0.01, "$.metrics.maximum_drawdown.value"), + ("win_rate", 1.01, "$.metrics.win_rate.value"), + ("tracking_error", -0.01, "$.metrics.tracking_error.value"), + ("n_trades", True, "$.metrics.trade_count.value"), + ("n_days", 2**53, "$.metrics.day_count.value"), + ], +) +def test_metric_domains_reject_invalid_source_values( + column: str, + value: object, + path: str, +) -> None: + _, artifact, run_ref, _ = _case("estimable") + performance = artifact.performance.astype(object) + performance.at[0, column] = value + changed_artifact = replace(artifact, _performance=performance) + changed_manifest = build_backtest_evidence_manifest( + run_ref, + changed_artifact, + artifact_available_at="2026-01-08T02:05:00Z", + ) + with pytest.raises(PerformanceEvidenceError) as rejected: + build_performance_evidence(changed_artifact, run_ref, changed_manifest) + _assert_error(rejected, PerformanceEvidenceErrorCode.METRIC_INVALID, path) + + +def test_closed_parser_rejects_unknown_non_ascii_non_finite_bool_and_unsafe_integer() -> None: + evidence, artifact, run_ref, manifest = _case("estimable") + + mutations: list[tuple[dict[str, Any], PerformanceEvidenceErrorCode, str]] = [] + unknown = evidence.to_dict() + unknown["unexpected"] = "value" + mutations.append((unknown, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.unexpected")) + non_ascii = evidence.to_dict() + non_ascii["métric"] = "value" + mutations.append((non_ascii, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.métric")) + non_finite = evidence.to_dict() + non_finite["metrics"][0]["value"] = float("inf") + mutations.append( + (non_finite, PerformanceEvidenceErrorCode.METRIC_INVALID, "$.metrics[0].value") + ) + bool_number = evidence.to_dict() + bool_number["methodology"]["periods_per_year"] = True + mutations.append( + ( + bool_number, + PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH, + "$.methodology.periods_per_year", + ) + ) + unsafe = evidence.to_dict() + unsafe["performance_table_row_count"] = 2**53 + mutations.append( + ( + unsafe, + PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH, + "$.performance_table_row_count", + ) + ) + + for payload, code, path in mutations: + with pytest.raises(PerformanceEvidenceError) as rejected: + PerformanceEvidenceV1.from_dict( + payload, + artifact=artifact, + run_ref=run_ref, + evidence_manifest=manifest, + ) + _assert_error(rejected, code, path) + + +def test_public_mapping_has_no_raw_inputs_storage_or_runtime_authority() -> None: + evidence, *_ = _case("estimable") + payload = evidence.to_dict() + serialized = evidence.to_json().lower() + forbidden_keys = { + "parameters", + "params_json", + "returns", + "nav", + "benchmark_series", + "table_bytes", + "locator", + "uri", + "credential", + "decision_eligible", + "publication_eligible", + "paper_trading", + "live_trading", + "investment_advice", + } + + def keys(value: object) -> set[str]: + if isinstance(value, dict): + return set(value) | {key for item in value.values() for key in keys(item)} + if isinstance(value, list): + return {key for item in value for key in keys(item)} + return set() + + assert not (keys(payload) & forbidden_keys) + for token in ("postgres://", "mysql://", "s3://", "credential", "broker"): + assert token not in serialized +