feat: add explicit retrospective v2 computation contracts (#20)
CI / lite (push) Successful in 19s
CI / lite (push) Successful in 19s
This commit was merged in pull request #20.
This commit is contained in:
+1215
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,175 @@
|
||||
{
|
||||
"contract_name": "researchhub.data-foundation",
|
||||
"schema_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"published_at": "2026-09-08T01:05:00Z",
|
||||
"instrument_routes": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"symbol": "SIM0",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb",
|
||||
"route_revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9"
|
||||
},
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"symbol": "SIM1",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799",
|
||||
"route_revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
}
|
||||
],
|
||||
"trading_calendar_revisions": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"session_date": "2018-01-02",
|
||||
"status": "open",
|
||||
"sessions": [
|
||||
{
|
||||
"opens_at": "2018-01-02T01:30:00Z",
|
||||
"closes_at": "2018-01-02T07:00:00Z"
|
||||
}
|
||||
],
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a",
|
||||
"calendar_revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
}
|
||||
],
|
||||
"corporate_action_revisions": [],
|
||||
"standardized_views": [
|
||||
{
|
||||
"view_id": "rhview:66666666666666666666666666666666",
|
||||
"view_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"transformation_digest": "sha256:b2376dbe4424b38e5b27874c998f082aecc086179f611896ebaa0fdeb5362c68",
|
||||
"instrument_route_revision_ids": [
|
||||
"rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
],
|
||||
"trading_calendar_revision_ids": [
|
||||
"rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
],
|
||||
"corporate_action_revision_ids": [],
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"available_at": "2026-09-08T01:04:00Z",
|
||||
"view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309"
|
||||
}
|
||||
],
|
||||
"observation_lineage": [
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb"
|
||||
},
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799"
|
||||
},
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a"
|
||||
}
|
||||
],
|
||||
"corporate_action_coverage": [
|
||||
{
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:233cc2968985e953e59e408b6619a837a792ae17f8afba32261f861de8fe9c7c"
|
||||
]
|
||||
},
|
||||
{
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:21f2594fc42e46168a65d7cac7e9b52aeabf24ecc5fb409530c07ad56ed3986c"
|
||||
]
|
||||
}
|
||||
],
|
||||
"readiness": {
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"contract_validation": {
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:62f2abbbb79f4ff9164e10cc4b21df37f0964224a4c546f978b97794241ff5b3"
|
||||
]
|
||||
},
|
||||
"real_data_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"production_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"live_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
}
|
||||
},
|
||||
"foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260"
|
||||
}
|
||||
@@ -0,0 +1,120 @@
|
||||
{
|
||||
"contract_name": "researchhub.dataset-snapshot",
|
||||
"schema_version": "2.0.0",
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"descriptor": {
|
||||
"dataset": {
|
||||
"dataset_id": "rhdataset:market:44444444444444444444444444444444",
|
||||
"dataset_kind": "market",
|
||||
"record_schema_version": "2.0.0",
|
||||
"dimensions": [
|
||||
"instrument_id",
|
||||
"effective_time"
|
||||
]
|
||||
},
|
||||
"published_at": "2026-09-08T01:03:00Z",
|
||||
"time_semantics": {
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"historical_availability": "not_established"
|
||||
},
|
||||
"content": {
|
||||
"digest_algorithm": "sha256",
|
||||
"canonicalization": "RFC8785",
|
||||
"record_order": "canonical-record-byte-order",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"logical_manifest": {
|
||||
"record_count": 2,
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2
|
||||
}
|
||||
]
|
||||
},
|
||||
"manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7",
|
||||
"record_count": 2
|
||||
},
|
||||
"observation_manifest": {
|
||||
"batches": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": "sha256:6f35f19a2ede0610d461b458a0f2cbc96f40cee6e90fa6592fd66daf617d3ec0"
|
||||
}
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"publisher": {
|
||||
"id": "researchhub.data",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"transformation": {
|
||||
"id": "rhtransform:55555555555555555555555555555555",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"upstream_snapshot_ids": [],
|
||||
"upstream_content_digests": []
|
||||
},
|
||||
"quality": {
|
||||
"status": "passed",
|
||||
"checks": [
|
||||
{
|
||||
"check_id": "completeness",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:519b01c60ad85c2b03d6c4f29027fe9b3672a8f6ae3ca8a2d831b281a61c5095"
|
||||
},
|
||||
{
|
||||
"check_id": "duplicate_identity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:7c585c7471ab45886fce037061b5ab0d5f981aede5d5190021963676b0ec0fcc"
|
||||
},
|
||||
{
|
||||
"check_id": "observation_coverage",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:dde5693c3ad79a01a162f9fc5a40c69d3c284965e283c91c5cd456904e0ef737"
|
||||
},
|
||||
{
|
||||
"check_id": "historical_claim_policy",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:f4cf8da48f92e03d75bb28b2569061e649f382ba360d879fef8e41e35169c228"
|
||||
},
|
||||
{
|
||||
"check_id": "range_validity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:6af08355c5e29d846e1c48783dccde791e8a3f48f8416149413217b7c2cbb52a"
|
||||
},
|
||||
{
|
||||
"check_id": "schema_conformance",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:fefdc5b81eba128af92f15a60feb7bb6f40b2e7277bf2beb48c1bb9399ee564b"
|
||||
}
|
||||
]
|
||||
},
|
||||
"qualification": {
|
||||
"status": "qualified",
|
||||
"usage": "retrospective_research",
|
||||
"policy_id": "researchhub.dataset-snapshot.retrospective",
|
||||
"policy_version": "2.0.0",
|
||||
"evaluated_at": "2026-09-08T01:02:00Z",
|
||||
"evidence_digest": "sha256:70c3ddcadf095877f17a1365c8d75a2e5a7078a27722fb8b698e61d350c6bb2d"
|
||||
}
|
||||
},
|
||||
"snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905"
|
||||
}
|
||||
@@ -16,7 +16,7 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
|
||||
for term in ("investment advice", "live order", "credentials", "source facts"):
|
||||
assert term in prohibited
|
||||
assert spec["authority"]["revision"] == 5
|
||||
assert spec["authority"]["revision"] == 6
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["provides"]
|
||||
@@ -28,6 +28,13 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
("researchhub.performance-evidence", "1.0.0"),
|
||||
("researchhub.portfolio-decision", "1.0.0"),
|
||||
("researchhub.risk-assessment", "1.0.0"),
|
||||
("researchhub.factor-set-ref", "2.0.0"),
|
||||
("researchhub.backtest-run-ref", "2.0.0"),
|
||||
("researchhub.backtest-evidence-manifest", "2.0.0"),
|
||||
("researchhub.performance-evidence", "2.0.0"),
|
||||
("researchhub.portfolio-target", "2.0.0"),
|
||||
("researchhub.portfolio-decision", "2.0.0"),
|
||||
("researchhub.risk-assessment", "2.0.0"),
|
||||
}
|
||||
expected_paths = {
|
||||
"researchhub.factor-definition": "src/quant_engine/factor_contracts.py",
|
||||
@@ -41,13 +48,28 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"])
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "1.0.0"
|
||||
} == expected_paths
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "2.0.0"
|
||||
} == {
|
||||
"researchhub.factor-set-ref": "src/quant_engine/retrospective_factor_contracts.py",
|
||||
"researchhub.backtest-run-ref": "src/quant_engine/retrospective_backtest_contracts.py",
|
||||
"researchhub.backtest-evidence-manifest": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.performance-evidence": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.portfolio-target": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.portfolio-decision": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.risk-assessment": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
}
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["consumes"]
|
||||
} == {
|
||||
("researchhub.dataset-snapshot", "1.0.0"),
|
||||
("researchhub.data-foundation", "1.0.0"),
|
||||
("researchhub.dataset-snapshot", "2.0.0"),
|
||||
("researchhub.data-foundation", "2.0.0"),
|
||||
}
|
||||
assert all(
|
||||
item["authority"] == "researchhub.data"
|
||||
@@ -65,6 +87,10 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
summary = portfolio_contract["summary"].lower()
|
||||
for term in ("receipt", "portfolio decisions", "risk assessments", "without"):
|
||||
assert term in summary
|
||||
retrospective = capabilities["retrospective-computation-contracts"]
|
||||
assert retrospective["status"] == "operational"
|
||||
for term in ("two clocks", "retrospective", "no historical-availability", "no execution"):
|
||||
assert term in retrospective["summary"].lower()
|
||||
for term in ("approval", "maker-checker", "publication", "paper", "live"):
|
||||
assert term in prohibited
|
||||
assert all(
|
||||
|
||||
@@ -0,0 +1,311 @@
|
||||
"""Fresh synthetic artifact assembly; no reuse or relabelling of real outputs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
EvidenceQualification,
|
||||
PerformanceEvidenceError,
|
||||
ResearchRunArtifact,
|
||||
build_research_run_artifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_performance_evidence,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.research_pipeline import run_factor_backtest_research
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
build_retrospective_performance_evidence,
|
||||
RetrospectiveBacktestEvidenceManifest,
|
||||
RetrospectivePerformanceEvidence,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import identify, replace_at
|
||||
|
||||
CONTRACT_ERRORS = (FactorContractError, BacktestContractError, PerformanceEvidenceError)
|
||||
|
||||
|
||||
def synthetic_artifact(run: RetrospectiveBacktestRunRef) -> ResearchRunArtifact:
|
||||
# Artifact-envelope tests, not an end-to-end proof of factor/source authenticity.
|
||||
# The existing financial methods receive new, in-memory synthetic matrices.
|
||||
dates = pd.date_range("2018-01-02", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"SIM0": [2.0, 0.0], "SIM1": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"SIM0": [10.0, 10.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 20.0, 21.0]}, index=dates
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"SIM0": [10.0, 12.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 18.0, 21.0]}, index=dates
|
||||
)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0
|
||||
),
|
||||
)
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02], index=result.returns.index, name="benchmark_return"
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id=run.run_id,
|
||||
strategy_id=run.strategy_id,
|
||||
strategy_name="Synthetic Top 1",
|
||||
strategy_version=run.strategy_version,
|
||||
engine_version="0.1.0",
|
||||
code_revision=run.code_revision,
|
||||
data_snapshot_id=run.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at=run.evaluation_at,
|
||||
finished_at=run.computed_at,
|
||||
parameters={"lag_sessions": 1, "top_k": 1},
|
||||
benchmark_id="synthetic.benchmark",
|
||||
benchmark_returns=benchmark,
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_closes_all_nine_existing_tables_without_changing_their_schema() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
wire = manifest.to_dict()
|
||||
assert wire["schema_version"] == "2.0.0"
|
||||
assert wire["artifact_schema_version"] == "1.1.0"
|
||||
assert wire["run_id"] == run.run_id
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["execution_validation"] == "not_validated"
|
||||
assert wire["decision_eligible"] is False
|
||||
assert manifest.manifest_id.startswith("rhbacktestevidencev2:sha256:")
|
||||
assert len({table.logical_name for item in manifest.evidence for table in item.tables}) == 9
|
||||
|
||||
|
||||
def test_performance_v2_keeps_existing_metric_methods_and_binds_all_upstreams() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
wire = evidence.to_dict()
|
||||
assert wire["schema_version"] == "researchhub.performance-evidence.v2"
|
||||
assert wire["methodology_id"] == "researchhub.quant-performance-methodology.v1"
|
||||
assert wire["metric_schema_id"] == "researchhub.quant-performance-metrics.v1"
|
||||
assert wire["research_artifact_schema_version"] == "1.1.0"
|
||||
assert wire["backtest_run_ref_id"] == run.run_id
|
||||
assert wire["backtest_evidence_manifest_id"] == manifest.manifest_id
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["start_date"] == "2018-01-02"
|
||||
assert wire["end_date"] == "2018-01-05"
|
||||
assert wire["artifact_available_at"] == "2026-09-08T01:11:00Z"
|
||||
assert evidence.run_id == run.run_id
|
||||
assert evidence.performance_evidence_id.startswith("rhperformancev2:sha256:")
|
||||
for metric in evidence.metrics:
|
||||
if metric.value is not None:
|
||||
assert metric.value == artifact.performance.iloc[0][metric.source_column]
|
||||
assert (
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
evidence.to_json(), artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
== evidence
|
||||
)
|
||||
assert (
|
||||
RetrospectiveBacktestEvidenceManifest.from_json(
|
||||
manifest.to_json(), artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
== manifest
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("run_id", "rhbacktestrunv2:sha256:" + "0" * 64),
|
||||
("profile", "offline_research_v1"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("evidence_scope", "real_data"),
|
||||
("artifact_available_at", "2026-09-08T01:09:00Z"),
|
||||
("artifact_schema_version", "2.0.0"),
|
||||
("qualification", "legacy_exploratory"),
|
||||
("evidence_digest", "sha256:" + "0" * 64),
|
||||
("evidence.0.tables.0.row_count", True),
|
||||
("evidence.0.tables.0.content_digest", "sha256:" + "0" * 64),
|
||||
("run_reference.value.factor_set_id", "rhfactorsetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_reidentified_claims_without_actual_table_closure(
|
||||
path: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
row = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "manifest_id", "rhbacktestevidencev2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
row, artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("table", "column", "value"),
|
||||
[
|
||||
("run", "data_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("run", "config_hash", "0" * 64),
|
||||
("run", "code_revision", "0" * 40),
|
||||
("run", "started_at", "2018-01-02T07:00:00Z"),
|
||||
("run", "finished_at", "2026-09-08T01:12:00Z"),
|
||||
("signals", "asset_id", "/private/data.csv"),
|
||||
("nav", "run_id", "old.run"),
|
||||
("performance", "run_id", "old.run"),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_artifact_identity_time_or_private_data_mismatch(
|
||||
table: str, column: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
frame = getattr(artifact, table)
|
||||
frame.loc[frame.index[0], column] = value
|
||||
forged = replace(artifact, **{"_" + table: frame})
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run, forged, artifact_available_at="2026-09-08T01:13:00Z"
|
||||
)
|
||||
|
||||
|
||||
def test_each_table_is_reconciled_and_legacy_apis_cannot_accept_v2() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
for item in manifest.evidence:
|
||||
for table in item.tables:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
expected_table_digests={table.logical_name: "sha256:" + "0" * 64},
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_performance_evidence(artifact, run, manifest)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
qualification=EvidenceQualification.LEGACY_EXPLORATORY,
|
||||
)
|
||||
|
||||
|
||||
def seal_performance(row: dict[str, Any]) -> None:
|
||||
def sha(document: Any) -> str:
|
||||
return (
|
||||
"sha256:"
|
||||
+ hashlib.sha256(
|
||||
json.dumps(
|
||||
document,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode()
|
||||
).hexdigest()
|
||||
)
|
||||
|
||||
row.pop("document_sha256", None)
|
||||
row.pop("performance_evidence_id", None)
|
||||
row["performance_evidence_id"] = "rhperformancev2:" + sha(row)
|
||||
row["document_sha256"] = sha(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "researchhub.performance-evidence.v1"),
|
||||
("scope", "live"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("evidence_scope", "real_data"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("backtest_run_ref_document_sha256", "sha256:" + "0" * 64),
|
||||
("backtest_evidence_manifest_id", "rhbacktestevidencev2:sha256:" + "0" * 64),
|
||||
("methodology.periods_per_year", 365),
|
||||
("metric_schema_id", "new.metric"),
|
||||
("metrics.0.value", 0.0),
|
||||
("metrics.0.nullable", True),
|
||||
("start_date", "2017-01-01"),
|
||||
("artifact_available_at", "2018-01-02T07:00:00Z"),
|
||||
],
|
||||
)
|
||||
def test_performance_never_accepts_reidentified_changed_facts(path: str, value: Any) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
row = build_retrospective_performance_evidence(artifact, run, manifest).to_dict()
|
||||
replace_at(row, path, value)
|
||||
seal_performance(row)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_dict(
|
||||
row, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
|
||||
|
||||
def test_performance_has_immutable_finite_canonical_payload_and_current_tables() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
assert evidence.document_sha256.startswith("sha256:")
|
||||
exported = evidence.to_dict()
|
||||
exported["metrics"][0]["value"] = 9.0
|
||||
assert evidence.to_dict()["metrics"][0]["value"] != 9.0
|
||||
for data in (
|
||||
evidence.to_json() + "\n",
|
||||
'{"schema_version":"x",' + evidence.to_json()[1:],
|
||||
"null",
|
||||
"{bad",
|
||||
):
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
data, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
frame = artifact.performance
|
||||
frame.loc[0, "total_ret"] = 0.0
|
||||
forged = replace(artifact, _performance=frame)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_performance_evidence(forged, run, manifest)
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Offline synthetic v2 backtest evidence and replay boundaries."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import FactorContractError, PayloadValidation
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef
|
||||
from test_retrospective_data_contracts import digest, identify, replace_at
|
||||
from test_retrospective_factor_contracts import factor_arguments, decoding_arguments
|
||||
|
||||
|
||||
def run_arguments() -> dict[str, Any]:
|
||||
arguments = factor_arguments()
|
||||
factor = RetrospectiveFactorSetRef.create(**arguments)
|
||||
view = next(iter(arguments["foundation"].views.values()))
|
||||
return {
|
||||
"dataset_snapshot": arguments["dataset_snapshot"],
|
||||
"foundation": arguments["foundation"],
|
||||
"factor_set": factor,
|
||||
"universe_digest": digest({"synthetic_universe": 2}),
|
||||
"trading_calendar_revision_ids": view.trading_calendar_revision_ids,
|
||||
"corporate_action_revision_ids": view.corporate_action_revision_ids,
|
||||
"strategy_id": "synthetic.top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": digest({"synthetic_strategy": "top1"}),
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": digest({"synthetic_execution": 1}),
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": digest({"synthetic_cost": 1}),
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": digest({"synthetic_lock": 1}),
|
||||
"configuration_digest": digest({"lag_sessions": 1, "top_k": 1}),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
}
|
||||
|
||||
|
||||
def run_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("dataset_snapshot", "foundation", "factor_set")}
|
||||
|
||||
|
||||
def test_run_identity_closes_observation_inputs_and_preserves_configurations() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
document = run.to_dict()
|
||||
assert document["schema_version"] == "2.0.0"
|
||||
assert run.run_id.startswith("rhbacktestrunv2:sha256:")
|
||||
assert run.dataset_snapshot_id == arguments["dataset_snapshot"].snapshot_id
|
||||
assert run.foundation_id == arguments["foundation"].foundation_id
|
||||
assert run.factor_set_id == arguments["factor_set"].factor_set_id
|
||||
assert document["usage"] == "retrospective_research"
|
||||
assert document["historical_availability"] == "not_established"
|
||||
assert document["decision_eligible"] is False
|
||||
assert document["execution_validation"] == "not_validated"
|
||||
assert document["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert document["replay_attempt"] == 0
|
||||
assert RetrospectiveBacktestRunRef.from_json(run.to_json(), **run_context(arguments)) == run
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("historical_availability", "established"),
|
||||
("usage", "as_available"),
|
||||
("execution_validation", "validated"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("dataset_content_digest", "sha256:" + "0" * 64),
|
||||
("foundation_digest", "sha256:" + "0" * 64),
|
||||
("factor_set_digest", "sha256:" + "0" * 64),
|
||||
("factor_output_content_digest", "sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("evidence_scope", "real_data"),
|
||||
("trading_calendar_revision_ids", []),
|
||||
("corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("evaluation_at", "2026-09-08T01:07:00Z"),
|
||||
("computed_at", "2026-09-08T01:08:00Z"),
|
||||
("computed_at", "2026-09-08T01:10:00.0000001Z"),
|
||||
("random_seed", True),
|
||||
("strategy_version", "latest"),
|
||||
("configuration_digest", "../private/a"),
|
||||
("code_revision", "unknown"),
|
||||
("replay_attempt", 1),
|
||||
("replay_reason", "retry"),
|
||||
("replay_spec_digest", "sha256:" + "0" * 64),
|
||||
("replay_ancestor_run_ids", ["rhbacktestrunv2:sha256:" + "0" * 64]),
|
||||
],
|
||||
)
|
||||
def test_reidentified_run_must_match_exact_context(path: str, value: Any) -> None:
|
||||
arguments = run_arguments()
|
||||
row = RetrospectiveBacktestRunRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "run_id", "rhbacktestrunv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.from_dict(row, **run_context(arguments))
|
||||
|
||||
|
||||
def test_replay_keeps_input_spec_but_requires_new_actual_attempt_times() -> None:
|
||||
arguments = run_arguments()
|
||||
root = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
replay_args = {
|
||||
**arguments,
|
||||
"parent": root,
|
||||
"replay_reason": "synthetic.retry",
|
||||
"replay_attempt": 1,
|
||||
"evaluation_at": "2026-09-08T01:12:00Z",
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
replay = RetrospectiveBacktestRunRef.create(**replay_args)
|
||||
assert replay.replay_spec_digest == root.replay_spec_digest
|
||||
assert replay.run_id != root.run_id
|
||||
assert replay.replay_ancestor_run_ids == (root.run_id,)
|
||||
assert replay.evaluation_at != root.evaluation_at
|
||||
assert (
|
||||
RetrospectiveBacktestRunRef.from_json(
|
||||
replay.to_json(), **run_context(arguments), parent=root
|
||||
)
|
||||
== replay
|
||||
)
|
||||
for changes in (
|
||||
{"random_seed": 9},
|
||||
{"configuration_digest": digest({"different_configuration": 1})},
|
||||
{"evaluation_at": root.evaluation_at},
|
||||
{"replay_attempt": 2},
|
||||
{"replay_reason": None},
|
||||
{"parent": None},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**replay_args, **changes})
|
||||
|
||||
|
||||
def test_reference_only_factors_can_be_read_but_not_used_to_create_new_runs() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
factor = arguments["factor_set"]
|
||||
reference = RetrospectiveFactorSetRef.from_dict(
|
||||
factor.to_dict(),
|
||||
definitions=factor._definitions,
|
||||
dataset_snapshot=arguments["dataset_snapshot"],
|
||||
foundation=arguments["foundation"],
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**arguments, "factor_set": reference})
|
||||
restored = RetrospectiveBacktestRunRef.from_dict(
|
||||
run.to_dict(), **{**run_context(arguments), "factor_set": reference}
|
||||
)
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
with pytest.raises(FactorContractError):
|
||||
restored.require_inputs_revalidated()
|
||||
run.require_inputs_revalidated()
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Frozen synthetic interoperability vector, not end-to-end source provenance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from quant_engine.artifact import _evidence_frame_records
|
||||
from quant_engine.retrospective_artifact_contracts import build_retrospective_performance_evidence
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_factor_contracts import factor_arguments
|
||||
from test_retrospective_portfolio_risk_contracts import portfolio_arguments, risk_arguments
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
VECTOR = ROOT / "tests" / "fixtures" / "retrospective-computation-v2.golden.json"
|
||||
|
||||
|
||||
def build_vector() -> dict[str, Any]:
|
||||
portfolio = portfolio_arguments()
|
||||
risk = risk_arguments(portfolio)
|
||||
run = portfolio["backtest_run_ref"]
|
||||
manifest = portfolio["manifest"]
|
||||
artifact = manifest._artifact
|
||||
factor = factor_arguments()
|
||||
return {
|
||||
"fixture_kind": "synthetic_retrospective_contract_vector",
|
||||
"artifact_data_provenance": "envelope_test_only_not_end_to_end",
|
||||
"source_authenticity": "not_established",
|
||||
"factor_definitions": [item.to_dict() for item in factor["definitions"]],
|
||||
"dataset_chunks": factor["dataset_chunks"],
|
||||
"resolved_view_schema": {"synthetic_schema": "neutral_close_v2"},
|
||||
"factor_output_schema": json.loads(factor["output_schema_bytes"]),
|
||||
"factor_output_records": json.loads(factor["output_content_bytes"]),
|
||||
"factor_set": run._factor_set.to_dict(),
|
||||
"backtest_run_ref": run.to_dict(),
|
||||
"artifact_tables": {
|
||||
name: _evidence_frame_records(frame, name)
|
||||
for name, frame in artifact.table_frames().items()
|
||||
},
|
||||
"backtest_evidence_manifest": manifest.to_dict(),
|
||||
"performance_evidence": build_retrospective_performance_evidence(
|
||||
artifact, run, manifest
|
||||
).to_dict(),
|
||||
"portfolio_target": portfolio["target"].to_dict(),
|
||||
"portfolio_decision": risk["portfolio_decision"].to_dict(),
|
||||
"risk_assessment": assess_retrospective_portfolio_risk(**risk).to_dict(),
|
||||
"covariance_matrix": risk["covariance"].covariance.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
def test_synthetic_vector_matches_all_current_owner_serializers() -> None:
|
||||
expected = VECTOR.read_text(encoding="utf-8")
|
||||
actual = (
|
||||
json.dumps(build_vector(), ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False)
|
||||
+ "\n"
|
||||
)
|
||||
assert actual == expected
|
||||
|
||||
|
||||
def test_v2_modules_do_not_import_data_owners_publishers_or_execution_authority() -> None:
|
||||
modules = sorted((ROOT / "src" / "quant_engine").glob("retrospective_*_contracts.py"))
|
||||
assert len(modules) == 5
|
||||
for path in modules:
|
||||
tree = ast.parse(path.read_text(encoding="utf-8"))
|
||||
imports = {
|
||||
alias.name
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Import)
|
||||
for alias in node.names
|
||||
} | {node.module or "" for node in ast.walk(tree) if isinstance(node, ast.ImportFrom)}
|
||||
assert not any(
|
||||
name.startswith(("research_results", "research_platform", "edb_data_core"))
|
||||
for name in imports
|
||||
)
|
||||
called = {
|
||||
node.func.id
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
|
||||
}
|
||||
assert not called & {
|
||||
"create_paper_order_intent",
|
||||
"run_governed_factor_slice",
|
||||
"evaluate_portfolio_risk",
|
||||
}
|
||||
@@ -0,0 +1,636 @@
|
||||
"""QE-owned decoding tests against RP's synthetic public v2 golden vectors."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from copy import deepcopy
|
||||
from dataclasses import FrozenInstanceError
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorContractError,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
|
||||
def golden(kind: str) -> dict[str, Any]:
|
||||
return json.loads((FIXTURES / f"retrospective-{kind}-v2.golden.json").read_text())
|
||||
|
||||
|
||||
def digest(value: Any) -> str:
|
||||
return "sha256:" + hashlib.sha256(canonical_json_bytes(value)).hexdigest()
|
||||
|
||||
|
||||
def identify(value: dict[str, Any], field: str, prefix: str) -> None:
|
||||
value[field] = prefix + digest({key: item for key, item in value.items() if key != field})
|
||||
|
||||
|
||||
def records() -> list[dict[str, Any]]:
|
||||
return [
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "1" * 32,
|
||||
"metric": "close",
|
||||
"value": "101.25",
|
||||
},
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "2" * 32,
|
||||
"metric": "close",
|
||||
"value": "87.50",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def bind_records(source: dict[str, Any], chunks: list[list[dict[str, Any]]]) -> None:
|
||||
def records_digest(rows: list[dict[str, Any]]) -> str:
|
||||
data = b"[" + b",".join(sorted(canonical_json_bytes(row) for row in rows)) + b"]"
|
||||
return "sha256:" + hashlib.sha256(data).hexdigest()
|
||||
|
||||
manifest = {
|
||||
"record_count": sum(len(rows) for rows in chunks),
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": index,
|
||||
"content_digest": records_digest(rows),
|
||||
"record_count": len(rows),
|
||||
}
|
||||
for index, rows in enumerate(chunks)
|
||||
],
|
||||
}
|
||||
source["descriptor"]["content"].update(
|
||||
{
|
||||
"record_count": manifest["record_count"],
|
||||
"logical_manifest": manifest,
|
||||
"manifest_digest": digest(manifest),
|
||||
"content_digest": records_digest([row for chunk in chunks for row in chunk]),
|
||||
}
|
||||
)
|
||||
source["descriptor"]["observation_manifest"]["batches"] = [
|
||||
{
|
||||
**chunk,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_receipt": index}),
|
||||
}
|
||||
for index, chunk in enumerate(manifest["chunks"])
|
||||
]
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
|
||||
|
||||
def replace_at(source: dict[str, Any], path: str, value: Any) -> None:
|
||||
target: Any = source
|
||||
keys = path.split(".")
|
||||
for key in keys[:-1]:
|
||||
target = target[int(key)] if isinstance(target, list) else target[key]
|
||||
target[int(keys[-1]) if isinstance(target, list) else keys[-1]] = value
|
||||
|
||||
|
||||
COLLECTIONS = (
|
||||
(
|
||||
"instrument_routes",
|
||||
"route_revision_id",
|
||||
"rhroutev2:",
|
||||
"instrument_route",
|
||||
"instrument_route_revision_ids",
|
||||
),
|
||||
(
|
||||
"trading_calendar_revisions",
|
||||
"calendar_revision_id",
|
||||
"rhcalv2:",
|
||||
"trading_calendar",
|
||||
"trading_calendar_revision_ids",
|
||||
),
|
||||
(
|
||||
"corporate_action_revisions",
|
||||
"action_revision_id",
|
||||
"rhcav2:",
|
||||
"corporate_action",
|
||||
"corporate_action_revision_ids",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def seal_foundation(source: dict[str, Any], *, rebuild_lineage: bool = True) -> None:
|
||||
lineage = []
|
||||
for name, key, prefix, kind, view_key in COLLECTIONS:
|
||||
replacements = {}
|
||||
for row in sorted(source[name], key=lambda row: row["observation_sequence"]):
|
||||
old = row[key]
|
||||
if "supersedes_observation_id" in row:
|
||||
row["supersedes_observation_id"] = replacements.get(
|
||||
row["supersedes_observation_id"], row["supersedes_observation_id"]
|
||||
)
|
||||
identify(row, key, prefix)
|
||||
replacements[old] = row[key]
|
||||
lineage.append(
|
||||
{
|
||||
"revision_kind": kind,
|
||||
"revision_id": row[key],
|
||||
**{
|
||||
field: row[field]
|
||||
for field in (
|
||||
"observation_sequence",
|
||||
"observed_by",
|
||||
"earliest_external_knowledge",
|
||||
"history_completeness",
|
||||
"evidence_digest",
|
||||
"supersedes_observation_id",
|
||||
)
|
||||
if field in row
|
||||
},
|
||||
}
|
||||
)
|
||||
for view in source["standardized_views"]:
|
||||
view[view_key] = [replacements.get(item, item) for item in view[view_key]]
|
||||
if rebuild_lineage:
|
||||
source["observation_lineage"] = lineage
|
||||
for view in source["standardized_views"]:
|
||||
identify(view, "view_ref_id", "rhviewrefv2:")
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
|
||||
|
||||
def parse_foundation(source: dict[str, Any]) -> RetrospectiveFoundationEnvelope:
|
||||
return RetrospectiveFoundationEnvelope.from_dict(
|
||||
source, snapshot=RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
)
|
||||
|
||||
|
||||
def test_public_snapshot_golden_is_an_explicit_observation_contract() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert snapshot.to_dict() == source
|
||||
assert snapshot.snapshot_id == source["snapshot_id"]
|
||||
assert snapshot.observation_cutoff == "2026-09-08T01:01:00Z"
|
||||
assert snapshot.evidence_scope == "synthetic_fixture"
|
||||
assert snapshot.earliest_external_knowledge == {"status": "unknown"}
|
||||
assert not hasattr(snapshot, "pit_cutoff")
|
||||
assert not hasattr(snapshot, "knowledge_time")
|
||||
snapshot.require_qualified()
|
||||
assert RetrospectiveSnapshotEnvelope.from_json(snapshot.to_json()) == snapshot
|
||||
|
||||
|
||||
def test_public_foundation_golden_binds_exact_typed_snapshot() -> None:
|
||||
source = golden("data-foundation")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert foundation.to_dict() == source
|
||||
assert foundation.dataset_snapshot_id == snapshot.snapshot_id
|
||||
assert foundation.observation_cutoff == snapshot.observation_cutoff
|
||||
assert foundation.evidence_scope == snapshot.evidence_scope
|
||||
assert foundation.real_data_validation_status == "not_validated"
|
||||
assert not hasattr(foundation, "pit_cutoff")
|
||||
assert (
|
||||
RetrospectiveFoundationEnvelope.from_json(foundation.to_json(), snapshot=snapshot)
|
||||
== foundation
|
||||
)
|
||||
|
||||
|
||||
def test_empty_logical_dimension_is_rejected_after_rebinding_all_bytes() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["instrument_id"] = ""
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError, match="dimension"):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_boolean_lineage_sequence_does_not_equal_integer_one() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["observation_lineage"][0]["observation_sequence"] = True
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("descriptor.time_semantics.pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.knowledge_time", {"start_inclusive": "2018-01-02T07:00:00Z"}),
|
||||
("descriptor.time_semantics.historical_availability", "established"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T01:01:00.0000001Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T09:01:00+08:00"),
|
||||
(
|
||||
"descriptor.time_semantics.earliest_external_knowledge",
|
||||
{"status": "unknown", "evidence_digest": "sha256:" + "0" * 64},
|
||||
),
|
||||
("descriptor.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("descriptor.published_at", "2026-02-30T00:00:00Z"),
|
||||
("descriptor.published_at", "2026-09-08T01:00:00Z"),
|
||||
("descriptor.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.qualification.usage", "as_available"),
|
||||
("descriptor.qualification.policy_version", "1.0.0"),
|
||||
("descriptor.quality.checks.0.severity", "advisory"),
|
||||
("descriptor.quality.checks.0.status", "failed"),
|
||||
("descriptor.quality.checks.0.check_id", "schema_conformance"),
|
||||
("descriptor.quality.status", "failed"),
|
||||
("descriptor.content.record_count", True),
|
||||
("descriptor.content.record_count", 9007199254740992),
|
||||
("descriptor.content.record_count", 2.0),
|
||||
("descriptor.content.content_digest", "bad"),
|
||||
("descriptor.content.manifest_digest", "sha256:" + "0" * 64),
|
||||
("descriptor.observation_manifest.batches", []),
|
||||
("descriptor.observation_manifest.batches.0.record_count", 1),
|
||||
("descriptor.observation_manifest.batches.0.chunk_index", True),
|
||||
("descriptor.observation_manifest.batches.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("descriptor.observation_manifest.batches.0.observation_kind", "first_published_at"),
|
||||
("descriptor.lineage.transformation.id", "rhtransform:private"),
|
||||
("descriptor.dataset.dimensions", ["instrument_id", "knowledge_time"]),
|
||||
("descriptor.dataset.dataset_id", "rhdataset:macroeconomic:" + "4" * 32),
|
||||
],
|
||||
)
|
||||
def test_snapshot_rejects_reidentified_invalid_declarations(path: str, value: Any) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
replace_at(source, path, value)
|
||||
# Noncanonical numbers are rejected before identity formation.
|
||||
if type(value) is not float and value != 9007199254740992:
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_snapshot_materialized_records_bind_the_public_golden_and_chunks() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
snapshot.verify_materialized_records([records()])
|
||||
snapshot.verify_materialized_records([list(reversed(records()))])
|
||||
chunks = [[records()[0]], [records()[1]]]
|
||||
bind_records(source, chunks)
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records(chunks)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records(chunks)
|
||||
rows = records()
|
||||
rows[0]["value"] = "0"
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"mutation", ["duplicate", "legacy", "range", "location", "missing_effective"]
|
||||
)
|
||||
def test_materialized_bad_records_rejected_even_with_matching_digests(mutation: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
if mutation == "duplicate":
|
||||
rows.append(deepcopy(rows[0]))
|
||||
elif mutation == "legacy":
|
||||
rows[0]["knowledge_time"] = "2026-09-08T01:00:00Z"
|
||||
elif mutation == "range":
|
||||
rows[0]["effective_time"] = "2018-01-01T07:00:00Z"
|
||||
elif mutation == "location":
|
||||
rows[0]["value"] = "/private/records.csv"
|
||||
else:
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_unknown_or_evidenced_knowledge_never_changes_usage() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"] = {
|
||||
"status": "evidenced",
|
||||
"range": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z",
|
||||
},
|
||||
"evidence_digest": digest({"synthetic_earliest": True}),
|
||||
}
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
parsed = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert (
|
||||
parsed.to_dict()["descriptor"]["time_semantics"]["historical_availability"]
|
||||
== "not_established"
|
||||
)
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"]["range"][
|
||||
"end_inclusive"
|
||||
] = "2026-09-08T01:01:00Z"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_rejected_snapshot_remains_readable_but_cannot_support_foundation() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["qualification"]["status"] = "rejected"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.require_qualified()
|
||||
foundation = golden("data-foundation")
|
||||
foundation["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in foundation["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(foundation, snapshot=snapshot)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2026-09-08T01:00:00Z"),
|
||||
("published_at", "2026-09-08T01:03:00Z"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "established"),
|
||||
("instrument_routes.0.observation_sequence", 2),
|
||||
("instrument_routes.0.supersedes_observation_id", "rhroutev2:sha256:" + "0" * 64),
|
||||
("instrument_routes.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("instrument_routes.0.history_completeness", "complete"),
|
||||
(
|
||||
"instrument_routes.0.earliest_external_knowledge",
|
||||
{"status": "unknown", "earliest_at": "2018-01-01T00:00:00Z"},
|
||||
),
|
||||
("instrument_routes.0.instrument_type", "index"),
|
||||
("instrument_routes.0.symbol", "WIND.TEST"),
|
||||
("instrument_routes.0.symbol", "A" * 33),
|
||||
("instrument_routes.0.calendar_id", "rhcalendar:" + "0" * 32),
|
||||
("trading_calendar_revisions.0.status", "closed"),
|
||||
("trading_calendar_revisions.0.sessions", []),
|
||||
("trading_calendar_revisions.0.sessions.0.closes_at", "2018-01-02T00:00:00Z"),
|
||||
("trading_calendar_revisions.0.session_date", "2018-02-30"),
|
||||
("standardized_views.0.instrument_route_revision_ids", []),
|
||||
("standardized_views.0.trading_calendar_revision_ids", []),
|
||||
("standardized_views.0.corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("standardized_views.0.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:02:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:06:00Z"),
|
||||
("standardized_views.0.usage", "as_available"),
|
||||
("corporate_action_coverage", []),
|
||||
("corporate_action_coverage.0.instrument_id", "rhinstrument:" + "0" * 32),
|
||||
("corporate_action_coverage.0.effective_time.end_inclusive", "2018-01-01T07:00:00Z"),
|
||||
("corporate_action_coverage.0.effective_time.start_inclusive", "2018-01-02T07:00:01Z"),
|
||||
("corporate_action_coverage.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("corporate_action_coverage.0.evidence_digests", []),
|
||||
("readiness.evidence_scope", "real_data"),
|
||||
("readiness.contract_validation.evidence_digests", []),
|
||||
(
|
||||
"readiness.real_data_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.production_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.live_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_foundation_rejects_semantic_forgery_after_reidentification(path: str, value: Any) -> None:
|
||||
source = golden("data-foundation")
|
||||
replace_at(source, path, value)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_v1_and_v2_never_coerce_each_other() -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
DatasetSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
with pytest.raises(FactorContractError):
|
||||
DataFoundationEnvelope.from_dict(golden("data-foundation"))
|
||||
old = json.loads((FIXTURES / "factor-contracts-v1.golden.json").read_text())
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(old["dataset_snapshot"])
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(old["data_foundation"])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"),
|
||||
snapshot=DatasetSnapshotEnvelope.from_dict(old["dataset_snapshot"]),
|
||||
)
|
||||
|
||||
|
||||
def test_deep_immutability_and_strict_canonical_json() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
source["descriptor"]["quality"]["status"] = "failed"
|
||||
snapshot.to_dict()["descriptor"]["qualification"]["status"] = "rejected"
|
||||
snapshot.require_qualified()
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
snapshot._payload = {}
|
||||
with pytest.raises(TypeError):
|
||||
snapshot.earliest_external_knowledge["status"] = "evidenced"
|
||||
foundation = parse_foundation(golden("data-foundation"))
|
||||
with pytest.raises(TypeError):
|
||||
foundation.views["new"] = next(iter(foundation.views.values()))
|
||||
for decoder, document in (
|
||||
(RetrospectiveSnapshotEnvelope.from_json, golden("dataset-snapshot")),
|
||||
(
|
||||
lambda value: RetrospectiveFoundationEnvelope.from_json(value, snapshot=snapshot),
|
||||
golden("data-foundation"),
|
||||
),
|
||||
):
|
||||
wire = canonical_json_bytes(document)
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(wire + b"\n")
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(b'{"schema_version":"2.0.0",' + wire[1:])
|
||||
|
||||
|
||||
def test_macro_period_is_not_inferred_as_an_effective_instant() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["dataset"].update(
|
||||
dataset_id="rhdataset:macroeconomic:" + "4" * 32,
|
||||
dataset_kind="macroeconomic",
|
||||
dimensions=["series_id", "observation_period"],
|
||||
)
|
||||
rows = [
|
||||
{
|
||||
"series_id": "cpi",
|
||||
"observation_period": "2018-01",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"value": "2.1",
|
||||
}
|
||||
]
|
||||
bind_records(source, [rows])
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def with_successor() -> dict[str, Any]:
|
||||
source = golden("data-foundation")
|
||||
previous = source["instrument_routes"][0]
|
||||
successor = deepcopy(previous)
|
||||
successor.update(
|
||||
observation_sequence=2,
|
||||
observed_by="2026-09-08T01:00:30Z",
|
||||
symbol="SIM0B",
|
||||
supersedes_observation_id=previous["route_revision_id"],
|
||||
)
|
||||
identify(successor, "route_revision_id", "rhroutev2:")
|
||||
source["instrument_routes"].append(successor)
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].append(
|
||||
successor["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
return source
|
||||
|
||||
|
||||
def test_retained_successor_requires_parent_time_and_view_ancestry() -> None:
|
||||
source = with_successor()
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.foundation_id == source["foundation_id"]
|
||||
assert parsed.published_at.isoformat() == "2026-09-08T01:05:00+00:00"
|
||||
assert parsed.contract_evidence_digests
|
||||
assert len(next(iter(parsed.views.values())).instrument_route_revision_ids) == 3
|
||||
for mutation in (
|
||||
"missing_parent",
|
||||
"equal_time",
|
||||
"omitted_ancestor",
|
||||
"duplicate_sequence",
|
||||
"wrong_lineage",
|
||||
):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "missing_parent":
|
||||
forged["instrument_routes"][-1]["supersedes_observation_id"] = (
|
||||
"rhroutev2:sha256:" + "0" * 64
|
||||
)
|
||||
elif mutation == "equal_time":
|
||||
forged["instrument_routes"][-1]["observed_by"] = forged["instrument_routes"][0][
|
||||
"observed_by"
|
||||
]
|
||||
elif mutation == "omitted_ancestor":
|
||||
forged["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
forged["instrument_routes"][0]["route_revision_id"]
|
||||
)
|
||||
elif mutation == "duplicate_sequence":
|
||||
forged["instrument_routes"][-1]["observation_sequence"] = 1
|
||||
else:
|
||||
forged["observation_lineage"][0]["evidence_digest"] = "sha256:" + "0" * 64
|
||||
seal_foundation(forged, rebuild_lineage=mutation != "wrong_lineage")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(forged)
|
||||
|
||||
|
||||
def test_index_and_closed_calendar_are_explicit_non_execution_data() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["instrument_routes"][0].update(asset_class="index", instrument_type="index")
|
||||
source["trading_calendar_revisions"][0].update(status="closed", sessions=[])
|
||||
seal_foundation(source)
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.to_dict()["instrument_routes"][0]["instrument_type"] == "index"
|
||||
assert parsed.to_dict()["readiness"]["live_validation"]["status"] == "not_validated"
|
||||
|
||||
|
||||
def test_real_scope_is_still_a_declaration_with_separate_evidence_and_coverage() -> None:
|
||||
snapshot_source = golden("dataset-snapshot")
|
||||
snapshot_source["evidence_scope"] = "real_data"
|
||||
identify(snapshot_source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_source)
|
||||
source = golden("data-foundation")
|
||||
source["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in source["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
source["readiness"]["evidence_scope"] = "real_data"
|
||||
source["readiness"]["real_data_validation"] = {
|
||||
"status": "validated",
|
||||
"evidence_digests": [digest({"synthetic_real_claim_test": 1})],
|
||||
}
|
||||
seal_foundation(source)
|
||||
parsed = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert parsed.real_data_validation_status == "validated" # NOT actual real-data evidence.
|
||||
for mutation in ("coverage", "reuse"):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "coverage":
|
||||
forged["corporate_action_coverage"][0].update(
|
||||
status="not_validated", evidence_digests=[]
|
||||
)
|
||||
else:
|
||||
forged["readiness"]["real_data_validation"] = deepcopy(
|
||||
forged["readiness"]["contract_validation"]
|
||||
)
|
||||
seal_foundation(forged)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(forged, snapshot=snapshot)
|
||||
synthetic = golden("data-foundation")
|
||||
synthetic["corporate_action_coverage"][0].update(status="not_validated", evidence_digests=[])
|
||||
seal_foundation(synthetic)
|
||||
assert parse_foundation(synthetic).real_data_validation_status == "not_validated"
|
||||
|
||||
|
||||
def test_action_must_belong_to_view_selected_instrument() -> None:
|
||||
source = golden("data-foundation")
|
||||
route = source["instrument_routes"][0]
|
||||
action = {
|
||||
"action_id": "rhaction:" + "7" * 32,
|
||||
"instrument_id": route["instrument_id"],
|
||||
"observation_sequence": 1,
|
||||
"observed_by": route["observed_by"],
|
||||
"earliest_external_knowledge": {
|
||||
"status": "evidenced",
|
||||
"earliest_at": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_action_earliest": 1}),
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": digest({"synthetic_action": 1}),
|
||||
"action_type": "cash_dividend",
|
||||
"status": "confirmed",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"terms_digest": digest({"synthetic_terms": 1}),
|
||||
}
|
||||
identify(action, "action_revision_id", "rhcav2:")
|
||||
source["corporate_action_revisions"] = [action]
|
||||
source["standardized_views"][0]["corporate_action_revision_ids"] = [
|
||||
action["action_revision_id"]
|
||||
]
|
||||
seal_foundation(source)
|
||||
assert len(parse_foundation(source).to_dict()["corporate_action_revisions"]) == 1
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
route["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_views_cannot_borrow_calendar_selection_from_each_other() -> None:
|
||||
source = golden("data-foundation")
|
||||
calendar = deepcopy(source["trading_calendar_revisions"][0])
|
||||
calendar["calendar_id"] = "rhcalendar:" + "9" * 32
|
||||
identify(calendar, "calendar_revision_id", "rhcalv2:")
|
||||
source["trading_calendar_revisions"].append(calendar)
|
||||
view = deepcopy(source["standardized_views"][0])
|
||||
view["view_id"] = "rhview:" + "8" * 32
|
||||
view["trading_calendar_revision_ids"] = [calendar["calendar_revision_id"]]
|
||||
source["standardized_views"].append(view)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"location", ["/private/a", "./a", "../a", "C:\\data\\a", "\\\\host\\a", "s3://private/a"]
|
||||
)
|
||||
def test_materialized_values_cannot_carry_physical_locations(location: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["value"] = location
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
@@ -0,0 +1,367 @@
|
||||
"""Synthetic v2 computation boundaries; never source authentication."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
FactorContractError,
|
||||
FactorDefinition,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
canonical_json_bytes,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
from quant_engine.retrospective_factor_contracts import (
|
||||
ResolvedRetrospectiveView,
|
||||
RetrospectiveCausation,
|
||||
RetrospectiveFactorSetRef,
|
||||
RetrospectiveInputBinding,
|
||||
RetrospectiveViewAvailability,
|
||||
)
|
||||
from test_retrospective_data_contracts import (
|
||||
digest,
|
||||
golden,
|
||||
identify,
|
||||
records,
|
||||
replace_at,
|
||||
seal_foundation,
|
||||
)
|
||||
|
||||
|
||||
def factor_arguments() -> dict[str, Any]:
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"), snapshot=snapshot
|
||||
)
|
||||
view = next(iter(foundation.views.values()))
|
||||
factor_inputs = (FactorInput("market", view.schema_digest, ("value",)),)
|
||||
definition = FactorDefinition.create(
|
||||
factor_id="neutral_close",
|
||||
version="1.0.0",
|
||||
formula="value",
|
||||
parameters={},
|
||||
implementation_digest=digest({"synthetic_formula": "identity"}),
|
||||
input_schema_digest=factor_input_schema_digest(factor_inputs),
|
||||
inputs=factor_inputs,
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=0,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "0.1.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
schema = {"fields": ["instrument_id", "value"]}
|
||||
output = [{"instrument_id": row["instrument_id"], "value": row["value"]} for row in records()]
|
||||
return {
|
||||
"definitions": (definition,),
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"selected_view_ref_ids": (view.view_ref_id,),
|
||||
"input_bindings": (
|
||||
RetrospectiveInputBinding(
|
||||
definition.definition_id, "market", view.view_ref_id, view.schema_digest
|
||||
),
|
||||
),
|
||||
"view_availability": (
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view_receipt": 1})
|
||||
),
|
||||
),
|
||||
"dataset_chunks": [records()],
|
||||
"resolved_views": (
|
||||
ResolvedRetrospectiveView(
|
||||
view.view_ref_id,
|
||||
canonical_json_bytes({"synthetic_schema": "neutral_close_v2"}),
|
||||
canonical_json_bytes(sorted(records(), key=canonical_json_bytes)),
|
||||
),
|
||||
),
|
||||
"output_quality": OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", digest({"synthetic_quality": 1})),),
|
||||
),
|
||||
"output_coverage": OutputCoverage(
|
||||
"complete", 2, 2, "records", "synthetic_market", digest({"synthetic_coverage": 1})
|
||||
),
|
||||
"output_schema_bytes": canonical_json_bytes(schema),
|
||||
"output_content_bytes": canonical_json_bytes(output),
|
||||
"output_artifact_ref": OutputArtifactRef.create(
|
||||
schema_digest=digest(schema), content_digest=digest(output)
|
||||
),
|
||||
"evaluation_at": "2026-09-08T01:06:00Z",
|
||||
"computed_at": "2026-09-08T01:07:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:08:00Z",
|
||||
"producer": ProducerIdentity("quant_engine", "0.1.0"),
|
||||
"code_revision": "d" * 40,
|
||||
"actor": ActorIdentity("service", "synthetic.research"),
|
||||
"correlation_id": "synthetic.retrospective",
|
||||
"causation": RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def decoding_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("definitions", "dataset_snapshot", "foundation")}
|
||||
|
||||
|
||||
def test_factor_result_closes_v2_inputs_and_preserves_v1_definition() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
wire = result.to_dict()
|
||||
assert result.schema_version == "2.0.0"
|
||||
assert result.factor_set_id.startswith("rhfactorsetv2:sha256:")
|
||||
assert result.definition_ids == (arguments["definitions"][0].definition_id,)
|
||||
assert result.definition_ids[0].startswith("rhfactorv1:")
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["availability_mode"] == "retrospective_replay"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert wire["decision_eligible"] is False
|
||||
assert "pit_cutoff" not in wire
|
||||
assert result.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
assert result.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
restored = RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json(), **decoding_arguments(arguments)
|
||||
)
|
||||
assert restored.to_dict() == wire
|
||||
assert restored.payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("contract_name", "researchhub.dataset-snapshot"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("foundation_id", "rhdfv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("selected_view_ref_ids", []),
|
||||
("selected_view_ref_ids", ["rhviewrefv2:sha256:" + "0" * 64]),
|
||||
("definition_ids", ["rhfactorv1:sha256:" + "0" * 64]),
|
||||
("input_bindings", []),
|
||||
("input_bindings.0.input_name", "volume"),
|
||||
("input_bindings.0.schema_digest", "sha256:" + "0" * 64),
|
||||
("input_bindings.0.view_ref_id", "rhviewrefv2:sha256:" + "0" * 64),
|
||||
("view_availability", []),
|
||||
("view_availability.0.available_at", "2026-09-08T01:03:30Z"),
|
||||
("view_availability.0.available_at", "2026-09-08T01:04:00.0000001Z"),
|
||||
("upstream_evidence.quality.checks.0.status", "failed"),
|
||||
("upstream_evidence.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("upstream_evidence.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("evidence_scope", "real_data"),
|
||||
("output_quality.status", "failed"),
|
||||
("output_quality.checks.0.status", "failed"),
|
||||
("output_coverage.status", "incomplete"),
|
||||
("output_coverage.observed_count", 1),
|
||||
("output_schema_digest", "sha256:" + "0" * 64),
|
||||
("output_artifact_ref.artifact_id", "rhfactoroutputv1:sha256:" + "0" * 64),
|
||||
("availability_mode", "as_available"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "declared_as_available"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("evaluation_at", "2018-01-02T07:00:00Z"),
|
||||
("evaluation_at", "2026-09-08T01:04:00Z"),
|
||||
("computed_at", "2026-09-08T01:05:00Z"),
|
||||
("artifact_available_at", "2026-09-08T01:06:00Z"),
|
||||
("producer.id", "research_platform"),
|
||||
("code_revision", "unknown"),
|
||||
("actor.id", "https://private/a"),
|
||||
("causation.id", "rhdfv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_factor_rejects_reidentified_semantic_forgery(path: str, value: Any) -> None:
|
||||
arguments = factor_arguments()
|
||||
row = RetrospectiveFactorSetRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "factor_set_id", "rhfactorsetv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(row, **decoding_arguments(arguments))
|
||||
|
||||
|
||||
def test_payload_validation_is_never_inherited_from_serialization() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
kwargs = decoding_arguments(arguments)
|
||||
reference = RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs)
|
||||
with pytest.raises(FactorContractError):
|
||||
reference.require_payloads_revalidated()
|
||||
checked = RetrospectiveFactorSetRef.from_dict(
|
||||
result.to_dict(),
|
||||
**kwargs,
|
||||
**{
|
||||
key: arguments[key]
|
||||
for key in (
|
||||
"output_schema_bytes",
|
||||
"output_content_bytes",
|
||||
"dataset_chunks",
|
||||
"resolved_views",
|
||||
)
|
||||
},
|
||||
)
|
||||
checked.require_payloads_revalidated()
|
||||
assert checked == result
|
||||
for extra in (
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"]},
|
||||
{"dataset_chunks": arguments["dataset_chunks"]},
|
||||
{"resolved_views": arguments["resolved_views"]},
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"], "output_content_bytes": b"[]"},
|
||||
{"dataset_chunks": arguments["dataset_chunks"], "resolved_views": []},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs, **extra)
|
||||
|
||||
|
||||
def test_create_verifies_actual_snapshot_and_each_resolved_view() -> None:
|
||||
for mutation in (
|
||||
"content",
|
||||
"schema",
|
||||
"snapshot",
|
||||
"duplicate_view",
|
||||
"noncanonical",
|
||||
"unknown_view",
|
||||
):
|
||||
arguments = factor_arguments()
|
||||
view = arguments["resolved_views"][0]
|
||||
if mutation == "content":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, view.schema_bytes, b"[]"),
|
||||
)
|
||||
elif mutation == "schema":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, b"{}", view.content_bytes),
|
||||
)
|
||||
elif mutation == "snapshot":
|
||||
arguments["dataset_chunks"][0][0]["value"] = "0"
|
||||
elif mutation == "duplicate_view":
|
||||
arguments["resolved_views"] = (view, view)
|
||||
elif mutation == "unknown_view":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(
|
||||
"rhviewrefv2:sha256:" + "0" * 64, view.schema_bytes, view.content_bytes
|
||||
),
|
||||
)
|
||||
else:
|
||||
arguments["output_content_bytes"] += b"\n"
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_parent_requires_exact_correlation_scope_and_actual_availability() -> None:
|
||||
arguments = factor_arguments()
|
||||
parent = RetrospectiveFactorSetRef.create(**arguments)
|
||||
child_args = {
|
||||
**arguments,
|
||||
"parent": parent,
|
||||
"causation": RetrospectiveCausation("factor_set", parent.factor_set_id),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:11:00Z",
|
||||
}
|
||||
child = RetrospectiveFactorSetRef.create(**child_args)
|
||||
assert child.factor_set_id != parent.factor_set_id
|
||||
assert (
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
child.to_json(), **decoding_arguments(arguments), parent=parent
|
||||
)
|
||||
== child
|
||||
)
|
||||
for changes in (
|
||||
{"parent": None},
|
||||
{"correlation_id": "different.correlation"},
|
||||
{"causation": RetrospectiveCausation("factor_set", "rhfactorsetv2:sha256:" + "0" * 64)},
|
||||
{"evaluation_at": "2026-09-08T01:07:59Z"},
|
||||
{"causation": arguments["causation"]},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**{**child_args, **changes})
|
||||
|
||||
|
||||
def test_definition_validity_is_checked_at_actual_evaluation() -> None:
|
||||
arguments = factor_arguments()
|
||||
arguments.update(
|
||||
evaluation_at="2027-01-01T00:00:00Z",
|
||||
computed_at="2027-01-01T00:01:00Z",
|
||||
artifact_available_at="2027-01-01T00:02:00Z",
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_factor_contract_is_immutable_and_v1_does_not_accept_it() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
exported = result.to_dict()
|
||||
exported["upstream_evidence"]["quality"]["status"] = "failed"
|
||||
assert result.upstream_evidence["quality"]["status"] == "passed"
|
||||
with pytest.raises(TypeError):
|
||||
result.upstream_evidence["quality"]["status"] = "failed"
|
||||
with pytest.raises(FactorContractError):
|
||||
FactorSetRef.from_dict(result.to_dict(), **decoding_arguments(arguments))
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json() + "\n", **decoding_arguments(arguments)
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
"rhviewrefv1:sha256:" + "0" * 64,
|
||||
"sha256:" + "0" * 64,
|
||||
)
|
||||
|
||||
|
||||
def test_missing_synthetic_or_real_readiness_cannot_be_relabelled() -> None:
|
||||
arguments = factor_arguments()
|
||||
snapshot_row = arguments["dataset_snapshot"].to_dict()
|
||||
snapshot_row["evidence_scope"] = "real_data"
|
||||
identify(snapshot_row, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_row)
|
||||
foundation_row = arguments["foundation"].to_dict()
|
||||
foundation_row["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
foundation_row["readiness"]["evidence_scope"] = "real_data"
|
||||
for view in foundation_row["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation_row)
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation_row, snapshot=snapshot)
|
||||
view = next(iter(foundation.views.values()))
|
||||
arguments.update(
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
evidence_scope="real_data",
|
||||
selected_view_ref_ids=(view.view_ref_id,),
|
||||
input_bindings=(
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
view.view_ref_id,
|
||||
view.schema_digest,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view": 1})
|
||||
),
|
||||
),
|
||||
causation=RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="real-data"):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
@@ -0,0 +1,655 @@
|
||||
"""New synthetic S4 evidence; historical valuation is not actual availability."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import FrozenInstanceError, replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
PortfolioRiskContractError,
|
||||
RiskAssessmentStatus,
|
||||
RiskFindingCode,
|
||||
)
|
||||
from quant_engine.artifact import EvidenceQualification, PerformanceEvidenceError
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.risk import ComponentRiskResult, CovarianceSnapshot, labeled_component_risk
|
||||
import quant_engine.retrospective_portfolio_risk_contracts as contracts
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
RetrospectivePortfolioDecision,
|
||||
RetrospectivePortfolioTarget,
|
||||
RetrospectiveRiskAssessment,
|
||||
build_retrospective_portfolio_decision,
|
||||
compute_retrospective_portfolio_receipt_digests,
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_artifact_contracts import synthetic_artifact
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import digest, replace_at
|
||||
|
||||
ASSETS = ("rhinstrument:" + "1" * 32, "rhinstrument:" + "2" * 32)
|
||||
CONTRACT_ERRORS = (
|
||||
FactorContractError,
|
||||
PortfolioRiskContractError,
|
||||
BacktestContractError,
|
||||
PerformanceEvidenceError,
|
||||
)
|
||||
|
||||
|
||||
def portfolio_arguments() -> dict[str, Any]:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
target = RetrospectivePortfolioTarget.create(
|
||||
backtest_run_id=run.run_id,
|
||||
dataset_snapshot_id=run.dataset_snapshot_id,
|
||||
weights={ASSETS[0]: 0.6, ASSETS[1]: 0.4},
|
||||
effective_at="2018-01-05T07:00:00Z",
|
||||
created_at="2026-09-08T01:12:00Z",
|
||||
)
|
||||
return {
|
||||
"backtest_run_ref": run,
|
||||
"manifest": manifest,
|
||||
"target": target,
|
||||
"objective_name": "synthetic_allocation",
|
||||
"objective_version": "1.0.0",
|
||||
"objective_digest": digest({"synthetic_objective": 1}),
|
||||
"model_name": "bounded_weights",
|
||||
"model_version": "1.0.0",
|
||||
"model_digest": digest({"synthetic_model": 1}),
|
||||
"expected_return_digest": digest({"synthetic_returns": 1}),
|
||||
"covariance_digest": "sha256:" + "a" * 64,
|
||||
"scenario_digest": digest({"synthetic_scenario": 1}),
|
||||
"constraints": ConstraintSetV1(
|
||||
gross_exposure_max=1.0,
|
||||
net_exposure_min=1.0,
|
||||
net_exposure_max=1.0,
|
||||
single_asset_min=0.2,
|
||||
single_asset_max=0.7,
|
||||
position_count_max=2,
|
||||
turnover_max=0.2,
|
||||
),
|
||||
"freshness_policy": FreshnessPolicy(
|
||||
max_manifest_age_seconds=3600, max_covariance_age_days=0
|
||||
),
|
||||
"prior_weights": {ASSETS[0]: 0.5, ASSETS[1]: 0.5},
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
|
||||
|
||||
def portfolio_receipt(arguments: dict[str, Any], **changes: Any) -> ComputationReceipt:
|
||||
values = compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key not in {"computed_at", "receipt"}}
|
||||
)
|
||||
return ComputationReceipt(
|
||||
**{
|
||||
"algorithm": "bounded_weights",
|
||||
"algorithm_version": "1.0.0",
|
||||
"implementation_digest": digest({"synthetic_implementation": 1}),
|
||||
"parameter_digest": digest({"synthetic_parameters": 1}),
|
||||
"input_digest": values["input_digest"],
|
||||
"constraint_digest": values["constraint_digest"],
|
||||
"output_digest": values["output_digest"],
|
||||
"status": "completed",
|
||||
"solver_required": False,
|
||||
"solver_name": None,
|
||||
"solver_version": None,
|
||||
"solver_config_digest": None,
|
||||
"iterations": None,
|
||||
"objective_value": None,
|
||||
"max_constraint_residual": values["max_constraint_residual"],
|
||||
"tolerance": 1e-12,
|
||||
"computed_at": arguments["computed_at"],
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_target_separates_historical_effective_time_from_actual_creation() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
target = arguments["target"]
|
||||
assert target.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert target.created_at == "2026-09-08T01:12:00Z"
|
||||
assert target.target_id.startswith("rhportfoliotargetv2:sha256:")
|
||||
assert target.to_dict()["usage"] == "retrospective_research"
|
||||
|
||||
|
||||
def test_portfolio_decision_preserves_constraints_and_actual_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
assert decision.decision_id.startswith("rhportfoliodecisionv2:sha256:")
|
||||
assert decision.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert decision.created_at == "2026-09-08T01:12:00Z"
|
||||
assert decision.computed_at == "2026-09-08T01:13:00Z"
|
||||
assert decision.gross_exposure == 1.0
|
||||
assert decision.position_count == 2
|
||||
assert decision.to_dict()["decision_eligible"] is False
|
||||
|
||||
|
||||
def covariance(arguments: dict[str, Any], **changes: Any) -> CovarianceSnapshot:
|
||||
return CovarianceSnapshot(
|
||||
**{
|
||||
"snapshot_id": "covariance:synthetic-retrospective",
|
||||
"as_of_date": "2018-01-05",
|
||||
"covariance": pd.DataFrame([[0.04, 0.01], [0.01, 0.09]], index=ASSETS, columns=ASSETS),
|
||||
"return_frequency": "1d",
|
||||
"periods_per_year": 252,
|
||||
"method": "provided",
|
||||
"window_start_date": "2018-01-02",
|
||||
"window_end_date": "2018-01-05",
|
||||
"observations": 4,
|
||||
"lookback_sessions": 4,
|
||||
"missing_policy": "complete_case",
|
||||
"data_snapshot_id": arguments["backtest_run_ref"].dataset_snapshot_id,
|
||||
"input_sha256": "a" * 64,
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def risk_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
return {
|
||||
"portfolio_decision": decision,
|
||||
"backtest_run_ref": arguments["backtest_run_ref"],
|
||||
"manifest": arguments["manifest"],
|
||||
"covariance": covariance(arguments),
|
||||
"risk_model_name": "euler_volatility",
|
||||
"risk_model_version": "1.0.0",
|
||||
"risk_model_digest": digest({"synthetic_risk_model": 1}),
|
||||
"risk_budget": {ASSETS[0]: 0.8, ASSETS[1]: 0.8},
|
||||
"portfolio_volatility_limit": 10.0,
|
||||
"groups": {ASSETS[0]: "equity", ASSETS[1]: "fixed_income"},
|
||||
"computed_at": "2026-09-08T01:14:00Z",
|
||||
}
|
||||
|
||||
|
||||
def test_risk_uses_historical_business_age_and_actual_computation_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.assessment_id.startswith("rhriskassessmentv2:sha256:")
|
||||
assert result.qualified is True
|
||||
assert result.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert result.computed_at == "2026-09-08T01:14:00Z"
|
||||
assert result.to_dict()["decision_eligible"] is False
|
||||
assert result.to_dict()["execution_validation"] == "not_validated"
|
||||
assert sum(result.percentage_risk.values()) == pytest.approx(1.0)
|
||||
assert sum(result.component_risk.values()) == pytest.approx(result.portfolio_volatility)
|
||||
|
||||
|
||||
def test_new_risk_computation_cannot_reuse_stale_actual_manifest_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments["computed_at"] = "2026-09-08T02:11:01Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
def target_with(arguments: dict[str, Any], **changes: Any) -> RetrospectivePortfolioTarget:
|
||||
row = arguments["target"].to_dict()
|
||||
return RetrospectivePortfolioTarget.create(
|
||||
**{
|
||||
key: value
|
||||
for key, value in {**row, **changes}.items()
|
||||
if key
|
||||
in {"backtest_run_id", "dataset_snapshot_id", "weights", "effective_at", "created_at"}
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def decision_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("backtest_run_ref", "manifest", "target")}
|
||||
|
||||
|
||||
def assessment_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
key: arguments[key]
|
||||
for key in ("portfolio_decision", "backtest_run_ref", "manifest", "covariance")
|
||||
}
|
||||
|
||||
|
||||
def reidentify(row: dict[str, Any], field: str, prefix: str) -> None:
|
||||
row.pop(field, None)
|
||||
encoded = json.dumps(
|
||||
row, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False
|
||||
)
|
||||
row[field] = prefix + "sha256:" + hashlib.sha256(encoded.encode()).hexdigest()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"parser",
|
||||
[RetrospectivePortfolioTarget, RetrospectivePortfolioDecision, RetrospectiveRiskAssessment],
|
||||
)
|
||||
def test_json_syntax_failures_use_typed_contract_errors(parser: Any) -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
parser.from_json(b"{")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"method": "alternate_estimator"},
|
||||
{"window_start_date": "2018-01-03"},
|
||||
{"window_end_date": "2018-01-04"},
|
||||
{"observations": 3},
|
||||
{"lookback_sessions": 5},
|
||||
{"missing_policy": "alternate_missing_policy"},
|
||||
],
|
||||
)
|
||||
def test_covariance_estimation_context_is_bound_into_the_result_identity(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
original = assess_retrospective_portfolio_risk(**arguments)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
changed = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert changed.assessment_id != original.assessment_id
|
||||
|
||||
|
||||
def test_canonical_roundtrips_and_immutable_results() -> None:
|
||||
base = portfolio_arguments()
|
||||
target = base["target"]
|
||||
assert RetrospectivePortfolioTarget.from_json(target.to_json().encode()) == target
|
||||
decision = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base))
|
||||
assert (
|
||||
RetrospectivePortfolioDecision.from_json(decision.to_json(), **decision_context(base))
|
||||
== decision
|
||||
)
|
||||
arguments = risk_arguments(base)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert (
|
||||
RetrospectiveRiskAssessment.from_json(
|
||||
result.to_json().encode(), **assessment_context(arguments)
|
||||
)
|
||||
== result
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
target.weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
target.created_at = "2018-01-05T07:00:00Z"
|
||||
with pytest.raises(TypeError):
|
||||
decision.target_weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(TypeError):
|
||||
result.component_risk[ASSETS[0]] = 0.1
|
||||
detached = result.to_dict()
|
||||
detached["component_risk"][ASSETS[0]] = 0.1
|
||||
assert detached != result.to_dict()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"weights": {}},
|
||||
{"weights": {"SIM0": 1.0}},
|
||||
{"weights": {ASSETS[0]: float("nan")}},
|
||||
{"weights": {ASSETS[0]: True}},
|
||||
{"backtest_run_id": "rhbacktestrunv1:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhds:sha256:" + "0" * 64},
|
||||
{"effective_at": "2026-09-09T01:00:00Z"},
|
||||
{"created_at": "2026-09-08T01:12:00.1234567Z"},
|
||||
{"effective_at": "2018-01-05T15:00:00+08:00"},
|
||||
],
|
||||
)
|
||||
def test_target_rejects_legacy_ambiguous_and_nonfinite_inputs(change: dict[str, Any]) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
target_with(portfolio_arguments(), **change)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("usage", "live"),
|
||||
("historical_availability", "established"),
|
||||
("schema_version", "1.0.0"),
|
||||
("extra", True),
|
||||
("target_id", "rhportfoliotargetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_target_rejects_wire_mutations(path: str, value: Any) -> None:
|
||||
row = portfolio_arguments()["target"].to_dict()
|
||||
row[path] = value
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_dict(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field", ["input_digest", "constraint_digest", "output_digest"])
|
||||
def test_receipt_digests_are_recomputed(field: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments, **{field: "sha256:" + "0" * 64})
|
||||
with pytest.raises(FactorContractError, match="independently recomputed"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", ["failed", "fallback"])
|
||||
def test_failed_or_fallback_solver_cannot_form_a_decision(status: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status=status,
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="failed/fallback"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"backtest_run_id": "rhbacktestrunv2:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"weights": {"rhinstrument:" + "f" * 32: 1.0}},
|
||||
{"created_at": "2026-09-08T01:10:00Z"},
|
||||
{"created_at": "2026-09-08T01:14:00Z"},
|
||||
],
|
||||
)
|
||||
def test_decision_closes_target_identity_assets_and_actual_time(change: dict[str, Any]) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
arguments["target"] = target_with(arguments, **change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_actual_manifest_freshness_boundary_and_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00Z"
|
||||
assert (
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
).computed_at
|
||||
== arguments["computed_at"]
|
||||
)
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00.000001Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=portfolio_receipt(arguments))
|
||||
arguments["computed_at"] = "2026-09-08T01:13:00Z"
|
||||
with pytest.raises(FactorContractError, match="receipt actual time"):
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments, computed_at="2026-09-08T01:13:01Z")
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_tables_and_qualification_are_revalidated_at_s4_boundary() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
manifest = arguments["manifest"]
|
||||
artifact = manifest._artifact
|
||||
arguments["manifest"] = build_retrospective_backtest_evidence_manifest(
|
||||
arguments["backtest_run_ref"],
|
||||
artifact,
|
||||
artifact_available_at=manifest.artifact_available_at,
|
||||
qualification=EvidenceQualification.EXPLORATORY,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="contract-qualified"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments["manifest"] = manifest
|
||||
# Public access is an isolated copy. Simulate corruption of the retained bytes,
|
||||
# beyond that normal interface, to exercise the consumer's independent recheck.
|
||||
artifact._performance.loc[0, "n_days"] += 1
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_constraint_residuals_and_prior_assets_cannot_be_bypassed() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["constraints"] = ConstraintSetV1(gross_exposure_max=0.9)
|
||||
# A solver may report convergence within its tolerance; actual contract constraints still bind.
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status="converged",
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
tolerance=0.2,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="violates supported constraints"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments = portfolio_arguments()
|
||||
arguments["prior_weights"] = {"rhinstrument:" + "f" * 32: 0.5}
|
||||
with pytest.raises(FactorContractError, match="prior assets"):
|
||||
compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key != "computed_at"}
|
||||
)
|
||||
arguments["prior_weights"] = None
|
||||
with pytest.raises(PortfolioRiskContractError, match="prior"):
|
||||
portfolio_receipt(arguments)
|
||||
|
||||
|
||||
def test_optional_prior_budget_limit_and_groups_have_explicit_empty_semantics() -> None:
|
||||
base = portfolio_arguments()
|
||||
base["constraints"] = ConstraintSetV1(gross_exposure_max=1.0)
|
||||
base["prior_weights"] = None
|
||||
arguments = risk_arguments(base)
|
||||
arguments.update(risk_budget=None, portfolio_volatility_limit=None, groups=None)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.qualified is True
|
||||
assert result.risk_budget == {}
|
||||
assert result.group_exposure == {}
|
||||
assert result.groups is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("gross_exposure", True),
|
||||
("position_count", 2.0),
|
||||
("target_weights." + ASSETS[0], 0.5),
|
||||
("schema_version", "1.0.0"),
|
||||
("observation_cutoff", "2018-01-05T07:00:00Z"),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_decision_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
base = portfolio_arguments()
|
||||
row = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base)).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "decision_id", "rhportfoliodecisionv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioDecision.from_dict(row, **decision_context(base))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"as_of_date": "2018-01-06"},
|
||||
{"as_of_date": "2018-01-04", "window_end_date": "2018-01-04"},
|
||||
{"window_start_date": None, "window_end_date": None},
|
||||
{"data_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"input_sha256": "b" * 64},
|
||||
],
|
||||
)
|
||||
def test_covariance_business_time_bounds_and_source_binding(change: dict[str, Any]) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
with pytest.raises(FactorContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,index,columns",
|
||||
[
|
||||
([[float("nan"), 0.0], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.1], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], ASSETS[0]), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], "unknown"), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], ASSETS, (ASSETS[0], "unknown")),
|
||||
],
|
||||
)
|
||||
def test_covariance_structure_is_checked_before_computation(
|
||||
matrix: Any, index: Any, columns: Any
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=index, columns=columns)
|
||||
)
|
||||
with pytest.raises(PortfolioRiskContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"risk_budget": {ASSETS[0]: -0.1}},
|
||||
{"risk_budget": {"unknown": 0.1}},
|
||||
{"portfolio_volatility_limit": -0.1},
|
||||
{"groups": {ASSETS[0]: "equity"}},
|
||||
{"groups": []},
|
||||
{"risk_model_version": "latest"},
|
||||
{"risk_model_name": "/private/model"},
|
||||
{"computed_at": "2026-09-08T01:12:59Z"},
|
||||
{"portfolio_decision": object()},
|
||||
{"covariance": object()},
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_invalid_models_budgets_clocks_and_untyped_inputs(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,finding",
|
||||
[
|
||||
([[1.0, 2.0], [2.0, 1.0]], RiskFindingCode.COVARIANCE_NOT_PSD),
|
||||
([[0.0, 0.0], [0.0, 0.0]], RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE),
|
||||
],
|
||||
)
|
||||
def test_numerical_unavailability_is_not_qualification(
|
||||
matrix: Any, finding: RiskFindingCode
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=ASSETS, columns=ASSETS)
|
||||
)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.qualified is False
|
||||
assert result.findings == (finding,)
|
||||
assert result.portfolio_volatility is None
|
||||
|
||||
|
||||
def test_risk_uses_the_existing_numeric_implementation_exactly_once(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
calls = []
|
||||
|
||||
def recorded(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
calls.append((weights, matrix))
|
||||
return labeled_component_risk(weights, matrix)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", recorded)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert len(calls) == 1
|
||||
expected = labeled_component_risk(*calls[0])
|
||||
assert result.component_risk == expected.component.to_dict()
|
||||
assert result.portfolio_volatility == expected.portfolio_volatility
|
||||
|
||||
|
||||
def test_unknown_numeric_failures_are_sanitized(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def failed(*args: Any) -> ComponentRiskResult:
|
||||
raise ValueError("synthetic internal detail")
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", failed)
|
||||
with pytest.raises(PortfolioRiskContractError, match="risk computation failed") as error:
|
||||
assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert "internal detail" not in str(error.value)
|
||||
|
||||
|
||||
def test_nonclosed_decomposition_is_unavailable(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def nonclosed(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
output = labeled_component_risk(weights, matrix)
|
||||
return replace(output, component=output.component * 0.5)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", nonclosed)
|
||||
result = assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.findings == (RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED,)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change", [{"portfolio_volatility_limit": 0.0}, {"risk_budget": {ASSETS[0]: 0.0}}]
|
||||
)
|
||||
def test_budget_breach_keeps_ready_but_unqualified_evidence(change: dict[str, Any]) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.READY
|
||||
assert result.qualified is False
|
||||
assert result.findings == (RiskFindingCode.RISK_BUDGET_BREACH,)
|
||||
assert result.decision_eligible is False
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("qualified", 1),
|
||||
("portfolio_volatility", 1.0),
|
||||
("component_risk." + ASSETS[0], 1.0),
|
||||
("schema_version", "1.0.0"),
|
||||
("covariance_matrix_digest", "sha256:" + "0" * 64),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
row = assess_retrospective_portfolio_risk(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "assessment_id", "rhriskassessmentv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveRiskAssessment.from_dict(row, **assessment_context(arguments))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw",
|
||||
[
|
||||
b'{"x":1,"x":2}',
|
||||
b'{ "x":1}',
|
||||
b"[]",
|
||||
b'{"x":NaN}',
|
||||
b'{"x":Infinity}',
|
||||
b'{"x":9007199254740992}',
|
||||
1,
|
||||
],
|
||||
)
|
||||
def test_json_profiles_reject_ambiguous_nonfinite_and_noncanonical_input(raw: Any) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_json(raw)
|
||||
Reference in New Issue
Block a user