diff --git a/MODULE_SPEC.yaml b/MODULE_SPEC.yaml index 0218936..1c8fba3 100644 --- a/MODULE_SPEC.yaml +++ b/MODULE_SPEC.yaml @@ -1,7 +1,7 @@ { "schema_version": 1, "module_id": "quant_engine", - "authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 5, "effective_from": "2026-09-01T00:00:00+08:00"}, + "authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 6, "effective_from": "2026-09-08T19:33:40+08:00"}, "repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"}, "bounded_context": { "domain": "quantitative-research-engine", @@ -21,6 +21,7 @@ {"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"}, {"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact and performance-methodology evidence without recomputation, persistence, or decision authority.", "status": "operational"}, {"id": "portfolio-risk-computation-contracts", "summary": "Verify deterministic portfolio-computation receipts and expose S3-bound portfolio decisions and risk assessments without adding algorithms or execution authority.", "status": "operational"}, + {"id": "retrospective-computation-contracts", "summary": "Decode observation-aware v2 data and expose explicit retrospective factor, backtest, portfolio and risk contracts with two clocks, no historical-availability claim and no execution authority.", "status": "operational"}, {"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"} ], "data": {"owns": [ @@ -35,11 +36,20 @@ {"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"}, {"contract_id": "researchhub.performance-evidence", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"}, {"contract_id": "researchhub.portfolio-decision", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"}, - {"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"} + {"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"}, + {"contract_id": "researchhub.factor-set-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_factor_contracts.py"}, + {"contract_id": "researchhub.backtest-run-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_backtest_contracts.py"}, + {"contract_id": "researchhub.backtest-evidence-manifest", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"}, + {"contract_id": "researchhub.performance-evidence", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"}, + {"contract_id": "researchhub.portfolio-target", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"}, + {"contract_id": "researchhub.portfolio-decision", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"}, + {"contract_id": "researchhub.risk-assessment", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"} ], "consumes": [ {"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"}, - {"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"} + {"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"}, + {"contract_id": "researchhub.dataset-snapshot", "version": "2.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_retrospective_envelope_and_materialized_chunks"}, + {"contract_id": "researchhub.data-foundation", "version": "2.0.0", "authority": "researchhub.data", "admission": "observation_bound_selected_views_and_materialized_bytes"} ] }, "dependencies": [], diff --git a/README.md b/README.md index 66ca385..ba5bca8 100644 --- a/README.md +++ b/README.md @@ -29,6 +29,7 @@ - `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef` - `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest` - `portfolio_risk_contracts` — S3 证据闭合的 `PortfolioDecision` / `RiskAssessment` v1;独立复核 freshness、约束与 computation receipt,并复用既有标签安全风险分解 +- `retrospective_*_contracts` — 未发布的显式 v2 回顾性合同:区分历史业务日期与实际可得/计算时间,保留 v1 和现有金融公式,不授予历史可得性、发布或执行权限;见 [v2 接口说明](docs/RETROSPECTIVE_COMPUTATION_V2.md) - `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计 - `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效 - `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test) diff --git a/docs/RETROSPECTIVE_COMPUTATION_V2.md b/docs/RETROSPECTIVE_COMPUTATION_V2.md new file mode 100644 index 0000000..0ae14fd --- /dev/null +++ b/docs/RETROSPECTIVE_COMPUTATION_V2.md @@ -0,0 +1,145 @@ +# Retrospective computation contracts v2 (unreleased) + +This pure, storage-neutral compatibility path consumes the separate data-contract +major 2.0.0. It does not migrate, reinterpret or relax the accepted v1 contracts. +No financial formula, execution simulation, dependency lock, production database, +publisher or live/paper-order interface changes here. Package version is unchanged; +the new contract major is not a package release or deployment. + +## Explicit public boundaries + +| Module | Public types/builders | Changed wire identity | +| --- | --- | --- | +| `retrospective_data_contracts` | `RetrospectiveSnapshotEnvelope`, `RetrospectiveFoundationEnvelope` | `rhdsv2`, `rhdfv2`; consume RP-owned 2.0.0 data semantics | +| `retrospective_factor_contracts` | `RetrospectiveFactorSetRef`, typed input/view/causation bindings, `ResolvedRetrospectiveView` | `rhfactorsetv2` | +| `retrospective_backtest_contracts` | `RetrospectiveBacktestRunRef` | `rhbacktestrunv2` | +| `retrospective_artifact_contracts` | `RetrospectiveBacktestEvidenceManifest`, `RetrospectivePerformanceEvidence` and their builders | `rhbacktestevidencev2`, `rhperformancev2` | +| `retrospective_portfolio_risk_contracts` | `RetrospectivePortfolioTarget`, `RetrospectivePortfolioDecision`, `RetrospectiveRiskAssessment`; receipt-digest, decision and assessment builders | `rhportfoliotargetv2`, `rhportfoliodecisionv2`, `rhriskassessmentv2` | + +These are separate types and domain-separated content identities. There is no +automatic v1-to-v2 cast. Unknown schema versions and fields are rejected. The +performance wire keeps its named schema `researchhub.performance-evidence.v2`; +the other new computation contracts use `schema_version: 2.0.0`. + +FactorDefinition, factor-output byte references, output quality/coverage, +ConstraintSetV1, FreshnessPolicy, ComputationReceipt, CovarianceSnapshot, financial +algorithms, performance metric/methodology IDs and the nine ResearchRunArtifact +tables keep their existing semantics. The table schema remains **1.1.0**. Reusing +these neutral primitives does not make a new-major upstream reference v1-compatible. + +## Two clocks, not backdated evidence + +Every new result fixes `usage=retrospective_research` and +`historical_availability=not_established`. A business date describes the historical +period being researched. Observation, publication, evaluation, artifact availability, +target creation and computation describe actual events, and must not be backdated. +Public v2 instants require UTC `Z` with at most six fractional digits. + +`observation_cutoff` and chunk `observed_by` are upper-bound observations. They are +not the earliest public knowledge time or a PIT cutoff. Unknown earliest knowledge +stays unknown; a supplied knowledge-evidence digest is not authenticated by parsing. +Foundation observation sequences describe retained revisions, not complete original +history. Selected view routes, calendars, corporate-action coverage and lineage +must close exactly within the supplied Foundation. + +Required actual order for factor/backtest evidence is: + +1. Foundation publication <= factor evaluation <= factor computation <= factor availability. +2. Factor availability <= backtest evaluation <= artifact start <= artifact finish + <= backtest computation <= artifact availability. +3. Artifact availability <= target creation <= portfolio computation <= risk computation. + +RetrospectivePortfolioTarget has a historical `effective_at` and a distinct actual +`created_at`. PortfolioDecision carries both plus actual `computed_at`. Covariance +window end <= covariance as-of date <= the historical effective date; covariance +maximum age is measured against that historical date. Manifest maximum age is +measured against **actual** portfolio and risk computation separately. Passing one +age check cannot substitute for the other. Generic v1 receipt timestamps retain +their original normalization; binding compares parsed actual instants. + +## Materialized bytes and reference-only reads + +Snapshot decoding checks structure, all six blocking-quality declarations, +qualification/time ordering, observation receipts and identities. +`verify_materialized_records` additionally checks supplied chunks, per-chunk and +aggregate content, counts, dimensions, effective ranges and macro effective instants. +Provider/physical paths are forbidden in public metadata and materialized records. + +Factor creation requires actual snapshot chunks, selected view schema/content bytes, +and factor-output schema/content bytes. Definition inputs, view availability, +Foundation ancestry and computed digests must close. Reference-only deserialization +is allowed for display/inspection, but input/output validation flags are derived from +supplied bytes, are not serialized claims, and must be re-established for new +computation. Backtest creation requires a factor whose payloads were revalidated. +Reference decoding cannot turn an unverified factor into an admitted compute input. + +Backtest manifest decoding rebuilds evidence from the supplied typed run and all +nine actual artifact tables. It checks table/run/config/strategy bindings and time +ordering. Portfolio composition revalidates those retained tables again, rather +than trusting a serialized manifest or mutable Python context. A table digest proves +content binding, not that those tables were produced by the claimed computation. + +All content-addressed IDs exclude their own ID field and bind the remainder of the +closed payload. Data/factor/backtest/manifest JSON retains the strict data profile +(no JSON floating-point numbers; financial record decimals are strings). Performance +and S4 preserve the existing finite numeric JSON profile: finite floats, safe ints, +exact booleans, sorted keys, compact separators, UTF-8. Duplicate keys, NaN, +Infinity, noncanonical JSON and extra fields are rejected. Wire revalidation uses +type-sensitive comparisons, including `true` versus `1`. Serializers return +detached copies; internal public maps are immutable. + +## Replay, receipts and risk + +Backtest v2 replay specification binds immutable input identities, selected calendar +and actions, strategy/execution/cost versions and digests, configuration, code, +environment lock and random seed. It excludes **both actual evaluation and actual +computation time**. These actual times remain in each run's identity. A replay must +retain the same replay specification, append its unique full ancestry, increment +attempt by one, and have parent computation < new actual evaluation <= computation. +This explicit new-major rule allows a later genuine replay without pretending its +evaluation happened at the parent's clock time. + +Portfolio computation-input v2 binds the full run and manifest document digests, +new target (including both clocks), objective/model versions and digests, declared +expected returns/covariance/scenario inputs, freshness policy and prior weights. +The receipt separately binds that input, constraints and recomputed outputs/residuals. +Targets and prior holdings must use selected logical instrument IDs, not ad-hoc +symbol matches. Failed/fallback receipts and any actual constraint residual are +rejected, even if a solver declares convergence within a permissive tolerance. + +Risk uses the existing labelled Euler decomposition exactly once. Its result binds +the supplied matrix content plus covariance method, bounded estimation window, +observation count, lookback, missing policy, annualization, source dataset/input, +model/budgets/groups and actual computation time. It checks exact labels, finite +symmetry, covariance-source binding and both freshness clocks. Non-PSD, +non-positive portfolio variance or non-closed contributions produce an unavailable, +unqualified result. A budget breach is a ready but unqualified calculation result. +`qualified=true` means only that these calculation checks passed. Every result +remains `decision_eligible=false`, `execution_validation=not_validated`; no portfolio +approval, maker-checker, publication, paper or live permission is granted here. + +## Trust, ownership and test evidence + +Pure builders accept declarations. Hashes, typed objects, model names, successful +constraint checks and synthetic fixtures do **not** authenticate data or compute +producers. Trusted owner-version bindings and receipt/qualification/view/clock +admission ports remain mandatory. RP owns governance and presentation; Research +Results owns publication. QE supplies validated calculation facts only. + +The two data fixtures are public RP candidate vectors from +`7da27e5bd33dc6d06f2c7c60f47029111156293a` (PR #100). EDB producer candidate +`88433dfc9d865ef782465498cdf9454c73920abd` (PR #13) is not a runtime dependency or +accepted owner binding. Acceptance/review gates remain separate from local tests. + +`tests/fixtures/retrospective-computation-v2.golden.json` freezes newly constructed +synthetic factor/run/manifest/performance/portfolio/risk payloads and their inputs. +Its artifact matrices are separate synthetic envelope-test inputs: the one-day +public data fixture is **not** claimed to have produced the four-day artifact. +The vector is not an end-to-end data/computation provenance proof or real-data run. +Its deterministic IDs are contract-regression evidence, not admitted source facts. + +Focused tests cover v2 goldens, mutation and strict JSON, bytes versus references, +two-clock freshness, replay ancestry, exact table bindings, constraints, receipts, +covariance provenance and numerical findings. Existing v1 tests must also pass. +Rollback is disabling the explicit v2 entry path while retaining v1 and original +immutable results; never retag old results or silently downgrade failed v2 admission. diff --git a/handoff-retrospective-contract-v2.md b/handoff-retrospective-contract-v2.md new file mode 100644 index 0000000..89d18bc --- /dev/null +++ b/handoff-retrospective-contract-v2.md @@ -0,0 +1,88 @@ +# Quant Engine retrospective v2 compatibility + +Scope: implement the user-authorized retrospective v2 compatibility without changing +v1 semantics, financial algorithms, original results, production databases, deployment +or trading. No claim of complete Quant OS delivery or real-data qualification. + +Branch: `codex/research-quant-os-retrospective-contract-v2-20260908`. +Declared base: accepted `main@68dd68392a26251391fbdae40c22eee370adb56e`. +One isolated delivery worktree; the old primary checkout is preserved. This is not +reactivation of an old registered stage or creation of a new stage ledger. + +## Dependency baseline + +Public RP data-contract candidate: PR #100, initial schemas/goldens at +`7da27e5bd33dc6d06f2c7c60f47029111156293a` (review/acceptance pending). +EDB mapping candidate: PR #13, initial implementation `88433df`, local full +validation passed. Neither candidate is silently treated as accepted owner evidence. +The shared public major is 2.0.0; preserve the accepted v1 paths independently. + +Order: public data contracts -> EDB mapping/Foundation -> Quant Engine typed +factor/backtest/portfolio/risk -> RP governance -> Research Results -> RP read. +Accepted owner-version bindings and runtime admission must still close every boundary. + +## Internal reuse decision + +Need: carry observation-aware inputs and retrospective-only claims through computation. +Existing: strict canonical JSON, immutable envelopes, factor definitions, input/output +closure, numerical algorithms, governed backtest and portfolio/risk contracts. +External candidates: not needed; this is project-owned semantics, not a missing library. +Approach: reuse those primitives and algorithms; introduce explicit new-major wrappers +only where upstream identity, time or usage semantics change. +Risk: reusing the v1 decoder or coercing observed-by into knowledge/PIT would make a +false historical claim. Unknown versions and unsupported usages must fail closed. + +## Implemented, not yet accepted or released + +Five separate v2 modules now implement immutable DatasetSnapshot/Foundation decoding +and materialized-content verification, FactorSet with explicit v2 nested bindings, +BacktestRunRef and replay ancestry, nine-table BacktestEvidenceManifest, +PerformanceEvidence, PortfolioTarget/Decision and RiskAssessment. The metadata +registers the new major alongside every existing v1 entry. See +`docs/RETROSPECTIVE_COMPUTATION_V2.md` for normative clocks, JSON profiles, input +closure, replay and owner-port boundaries. + +Factor definitions, generic output/receipt/constraint/covariance primitives and +financial implementations are reused without semantic edits. Table schema remains +1.1.0; v1 business source, v1 goldens, `pyproject.toml`, `uv.lock` and `ci-profile.yml` +are unchanged. Package version remains unreleased. Only module metadata, its exact +inventory test and README gain v2 alongside the new files. + +The frozen synthetic computation vector includes fresh factor/backtest/manifest/ +performance/target/portfolio/risk documents and synthetic artifact tables. It is +explicitly **envelope-only**, not an end-to-end claim that the one-day data fixture +produced the four-day synthetic financial artifact. No old real run was rerun, +retagged or backdated. + +## Local verification (2026-09-08) + +- Full repository unit suite: **1039 passed**, 1166 warnings, 31.50 seconds. +- S4 focused new + unchanged v1 contracts: **120 passed**; new S4 332 statements, + 20 branches, 100% measured coverage. Coverage is not source authentication or + proof of complete business semantics. +- All five new source modules passed mypy; all six new test modules, five new + sources and the updated metadata test passed Ruff. +- The combined synthetic vector and metadata smoke checks: **3 passed**. +- Actual negative tests reproduced and fixed missing covariance-estimation context + in result identity, untyped malformed-JSON errors, and risk-time stale-manifest + reuse. Other modules' earlier RED/GREEN evidence remains part of the same turn. + +The full suite was run directly against the frozen local environment. This is not +the same claim as remote CI or central ship acceptance; the unchanged declared CI +profile is `lite` with the module-metadata smoke command. Central validation and +Draft PR creation follow the implementation commit. No Ready, merge, accepted +upstream binding or independent-review pass is claimed here. + +Actual computation/admission times are distinct from simulated business dates. New +formal outputs cannot inherit the old run's producer identity or be backdated to it. +Real receipt/qualification/view/clock ports remain mandatory; typed objects and hashes +are not source authentication. The optional independent reviewer delegation is still +awaiting the already-requested user choice. + +Next: preserve the candidate for review, then carry explicit v2 facts through +RP governance -> Research Results publication -> RP read compatibility. Bind final +accepted upstream versions only when actual acceptance evidence exists. The entire +Quant OS goal is not complete at this intermediate owner unit. + +Rollback: disable the explicit v2 path and retain v1 plus immutable artifacts; never +retag v2 into v1 or silently use synthetic evidence for real admission. diff --git a/src/quant_engine/retrospective_artifact_contracts.py b/src/quant_engine/retrospective_artifact_contracts.py new file mode 100644 index 0000000..a62b974 --- /dev/null +++ b/src/quant_engine/retrospective_artifact_contracts.py @@ -0,0 +1,489 @@ +"""Retrospective-only evidence wrappers over the unchanged research fact tables.""" + +from __future__ import annotations + +import json +from collections.abc import Mapping +from dataclasses import dataclass, field +from types import MappingProxyType +from typing import Any, Self, cast + +import pandas as pd + +from quant_engine.artifact import ( + BacktestEvidenceEntry, + EvidenceQualification, + ResearchRunArtifact, + PERFORMANCE_METRIC_SCHEMA_ID, + PERFORMANCE_METHODOLOGY_ID, + PerformanceMethodology, + PerformanceMetric, + _PERFORMANCE_SOURCE_COLUMNS, + _absolute_performance_metrics, + _benchmark_context, + _count_performance_metrics, + _performance_canonical_bytes, + _performance_compare, + _performance_date, + _performance_digest, + _performance_methodology, + _performance_text, + _performance_validate_tree, + _relative_performance_metrics, + _artifact_frames, + _evidence_entries, + _evidence_frame_records, + _manifest_instant, + _run_row, + _table_evidence, + _validate_table_run_ids, +) +from quant_engine.factor_contracts import ( + ContractErrorCode, + FactorContractError, + _assert_canonical_profile, + _content_address, + _digest_bytes, + _duplicate_key_pairs, + _freeze_json, + _parse_json_object, + _parse_utc, + _thaw_json, + canonical_json, + canonical_json_bytes, +) +from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef +from quant_engine.retrospective_data_contracts import _check, _public, _shape + + +def _validated_run(run: Any) -> RetrospectiveBacktestRunRef: + _check( + type(run) is RetrospectiveBacktestRunRef, + "$.run_ref", + "explicit v2 run reference required", + ContractErrorCode.TYPE_ERROR, + ) + factor = run._factor_set + return RetrospectiveBacktestRunRef.from_dict( + run.to_dict(), + dataset_snapshot=factor._dataset_snapshot, + foundation=factor._foundation, + factor_set=factor, + parent=run._parent, + ) + + +def _validated_frames( + artifact: ResearchRunArtifact, run: RetrospectiveBacktestRunRef +) -> dict[str, pd.DataFrame]: + frames = _artifact_frames(artifact) + _validate_table_run_ids(frames, run.run_id) + row = _run_row(frames) + expected = { + "run_id": run.run_id, + "data_snapshot_id": run.dataset_snapshot_id, + "strategy_id": run.strategy_id, + "strategy_version": run.strategy_version, + "code_revision": run.code_revision, + "config_hash": run.configuration_digest.removeprefix("sha256:"), + "schema_version": artifact.schema_version, + } + _check( + set(expected) | {"started_at", "finished_at"} <= set(row.index), + "$.artifact.tables.run", + "run schema fields missing", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + for key, value in expected.items(): + _check( + type(row[key]) is str and row[key] == value, + f"$.artifact.tables.run.{key}", + "artifact does not bind exact v2 run", + ContractErrorCode.IDENTITY_MISMATCH, + ) + # The unchanged artifact 1.1 timestamp profile admits offsets; public v2 + # envelope times remain strict UTC. No knowledge-time inference is performed. + _, started = _manifest_instant(row["started_at"], "$.artifact.tables.run.started_at") + _, finished = _manifest_instant(row["finished_at"], "$.artifact.tables.run.finished_at") + _check( + _parse_utc(run.evaluation_at, "$.run_ref.evaluation_at") + <= started + <= finished + <= _parse_utc(run.computed_at, "$.run_ref.computed_at"), + "$.artifact.tables.run", + "actual evaluation <= start <= finish <= computed required", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + for name, frame in frames.items(): + _public(_evidence_frame_records(frame, name), f"$.artifact.tables.{name}") + return frames + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectiveBacktestEvidenceManifest: + """Exact artifact closure, not authenticity, historical or execution authority.""" + + contract_name: str + schema_version: str + manifest_id: str + run_id: str + profile: str + artifact_schema_version: str + artifact_available_at: str + qualification: EvidenceQualification + evidence_digest: str + evidence: tuple[BacktestEvidenceEntry, ...] + backtest_run_ref: RetrospectiveBacktestRunRef + evidence_scope: str + usage: str + historical_availability: str + observation_cutoff: str + decision_eligible: bool + execution_validation: str + _payload: Mapping[str, Any] = field(repr=False, compare=False) + _artifact: ResearchRunArtifact = field(repr=False, compare=False) + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return canonical_json(self.to_dict()) + + @classmethod + def from_dict( + cls, + value: Any, + *, + artifact: ResearchRunArtifact, + backtest_run_ref: RetrospectiveBacktestRunRef, + ) -> Self: + _assert_canonical_profile(value) + row = _shape( + value, + "$", + "contract_name schema_version manifest_id run_id profile artifact_schema_version artifact_available_at qualification " + "run_reference evidence_digest evidence evidence_scope usage historical_availability observation_cutoff decision_eligible execution_validation", + ) + _check( + type(row["qualification"]) is str + and row["qualification"] in {"exploratory", "contract_qualified"}, + "$.qualification", + "explicit non-legacy contract qualification required", + ContractErrorCode.QUALIFICATION_REJECTED, + ) + rebuilt = build_retrospective_backtest_evidence_manifest( + backtest_run_ref, + artifact, + artifact_available_at=row["artifact_available_at"], + qualification=EvidenceQualification(row["qualification"]), + ) + _check( + canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()), + "$", + "manifest differs from actual run/table closure", + ContractErrorCode.IDENTITY_MISMATCH, + ) + return cast(Self, rebuilt) + + @classmethod + def from_json(cls, value: str | bytes, **kwargs: Any) -> Self: + return cls.from_dict(_parse_json_object(value, "$"), **kwargs) + + +def build_retrospective_backtest_evidence_manifest( + backtest_run_ref: RetrospectiveBacktestRunRef, + artifact: ResearchRunArtifact, + *, + artifact_available_at: str, + qualification: EvidenceQualification = EvidenceQualification.CONTRACT_QUALIFIED, + expected_table_digests: Mapping[str, str] | None = None, +) -> RetrospectiveBacktestEvidenceManifest: + """Close new in-memory artifact bytes; never promote an old exploratory run.""" + run = _validated_run(backtest_run_ref) + _check( + type(qualification) is EvidenceQualification + and qualification is not EvidenceQualification.LEGACY_EXPLORATORY, + "$.qualification", + "legacy evidence cannot enter the v2 path", + ContractErrorCode.QUALIFICATION_REJECTED, + ) + available = _parse_utc(artifact_available_at, "$.artifact_available_at") + _check( + _parse_utc(run.computed_at, "$.run_ref.computed_at") <= available, + "$.artifact_available_at", + "artifact precedes actual computation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + frames = _validated_frames(artifact, run) + summaries = _table_evidence(frames, expected_table_digests) + reference: dict[str, object] = {"kind": "backtest_run_ref", "value": run.to_dict()} + # Table categories and canonical content hashing have not changed semantics. + evidence = _evidence_entries(summaries, reference, legacy=False) + evidence_digest = _digest_bytes(canonical_json_bytes([item.to_dict() for item in evidence])) + payload = { + "contract_name": "researchhub.backtest-evidence-manifest", + "schema_version": "2.0.0", + "run_id": run.run_id, + "profile": "offline_research_retrospective_v2", + "artifact_schema_version": artifact.schema_version, + "artifact_available_at": artifact_available_at, + "qualification": qualification.value, + "run_reference": reference, + "evidence_digest": evidence_digest, + "evidence": [item.to_dict() for item in evidence], + "evidence_scope": run.evidence_scope, + "usage": run.usage, + "historical_availability": run.historical_availability, + "observation_cutoff": run.observation_cutoff, + "decision_eligible": False, + "execution_validation": "not_validated", + } + payload["manifest_id"] = _content_address( + payload, "manifest_id", "rhbacktestevidencev2:sha256:" + ) + instance = object.__new__(RetrospectiveBacktestEvidenceManifest) + values = { + **payload, + "qualification": qualification, + "evidence": evidence, + "backtest_run_ref": run, + "_payload": _freeze_json(payload), + "_artifact": artifact, + } + del values["run_reference"] + for name, value in values.items(): + object.__setattr__(instance, name, value) + return instance + + +def _freeze_numeric_evidence(value: Any) -> Any: + """Freeze the existing finite-number metric profile, not the data JSON profile.""" + if type(value) is dict: + return MappingProxyType( + {key: _freeze_numeric_evidence(item) for key, item in value.items()} + ) + if type(value) is list: + return tuple(_freeze_numeric_evidence(item) for item in value) + return value + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectivePerformanceEvidence: + """New upstream/time identity; unchanged finite-number metric/methodology v1.""" + + methodology: PerformanceMethodology + metrics: tuple[PerformanceMetric, ...] + _payload: Mapping[str, Any] = field(repr=False) + + @property + def performance_evidence_id(self) -> str: + return cast(str, self._payload["performance_evidence_id"]) + + @property + def document_sha256(self) -> str: + return cast(str, self._payload["document_sha256"]) + + @property + def run_id(self) -> str: + return cast(str, self._payload["run_id"]) + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def canonical_bytes(self) -> bytes: + return _performance_canonical_bytes(self.to_dict()) + + def to_json(self) -> str: + return self.canonical_bytes().decode("utf-8") + + @classmethod + def from_dict( + cls, + value: Any, + *, + artifact: ResearchRunArtifact, + run_ref: RetrospectiveBacktestRunRef, + evidence_manifest: RetrospectiveBacktestEvidenceManifest, + ) -> Self: + _performance_validate_tree(value, "$") + rebuilt = build_retrospective_performance_evidence(artifact, run_ref, evidence_manifest) + _performance_compare(value, rebuilt.to_dict(), "$") + return cast(Self, rebuilt) + + @classmethod + def from_json(cls, value: str | bytes, **kwargs: Any) -> Self: + _check( + type(value) in {str, bytes}, + "$", + "canonical JSON text/bytes required", + ContractErrorCode.TYPE_ERROR, + ) + raw = value.encode("utf-8") if isinstance(value, str) else value + try: + document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs) + except (UnicodeDecodeError, json.JSONDecodeError) as error: + raise FactorContractError( + ContractErrorCode.INVALID_FORMAT, "$", "invalid performance evidence JSON" + ) from error + _check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR) + _check( + _performance_canonical_bytes(document) == raw, + "$", + "canonical finite-number JSON required", + ContractErrorCode.INVALID_FORMAT, + ) + return cls.from_dict(document, **kwargs) + + +def build_retrospective_performance_evidence( + artifact: ResearchRunArtifact, + run_ref: RetrospectiveBacktestRunRef, + evidence_manifest: RetrospectiveBacktestEvidenceManifest, +) -> RetrospectivePerformanceEvidence: + """Bind current tables and existing methodology; no performance recalculation.""" + run = _validated_run(run_ref) + _check( + type(evidence_manifest) is RetrospectiveBacktestEvidenceManifest, + "$.evidence_manifest", + "explicit v2 manifest required", + ContractErrorCode.TYPE_ERROR, + ) + manifest = RetrospectiveBacktestEvidenceManifest.from_dict( + evidence_manifest.to_dict(), artifact=artifact, backtest_run_ref=run + ) + frames = _validated_frames(artifact, run) + performance = frames["performance"] + _check( + len(performance) == 1 + and tuple(str(column) for column in performance.columns) == _PERFORMANCE_SOURCE_COLUMNS, + "$.artifact.tables.performance", + "one row in the unchanged closed performance schema required", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + performance_row = performance.iloc[0] + run_row = _run_row(frames) + frequency = _performance_text(run_row["frequency"], "$.artifact.tables.run.frequency") + _check( + frequency == "1d", + "$.artifact.tables.run.frequency", + "only existing daily methodology is supported", + ) + calendar = _performance_text(run_row["calendar"], "$.artifact.tables.run.calendar") + timezone = _performance_text(run_row["timezone"], "$.artifact.tables.run.timezone") + start_date = _performance_date(run_row["start_date"], "$.artifact.tables.run.start_date") + end_date = _performance_date(run_row["end_date"], "$.artifact.tables.run.end_date") + nav = frames["nav"] + _check( + not nav.empty, + "$.artifact.tables.nav", + "NAV observation window required", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + _check( + _performance_date(nav.iloc[0]["trade_date"], "$.artifact.tables.nav.start") == start_date + and _performance_date(nav.iloc[-1]["trade_date"], "$.artifact.tables.nav.end") == end_date, + "$.artifact.tables.nav", + "observation window differs from artifact dates", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + benchmark_digest, active_std, benchmark_variance, alpha_domain_unestimable = _benchmark_context( + frames, run_row, performance_row + ) + metrics = ( + *_absolute_performance_metrics(performance_row), + *_relative_performance_metrics( + performance_row, + benchmark_present=benchmark_digest is not None, + active_std=active_std, + benchmark_variance=benchmark_variance, + alpha_domain_unestimable=alpha_domain_unestimable, + ), + *_count_performance_metrics(performance_row), + ) + normalized_row: dict[str, object] = {metric.source_column: metric.value for metric in metrics} + normalized_row["run_id"] = run.run_id + row_digest = _performance_digest( + {"columns": list(_PERFORMANCE_SOURCE_COLUMNS), "row": normalized_row} + ) + alignment = cast(str, run_row["benchmark_alignment_policy"]) + methodology = _performance_methodology( + frequency=frequency, alignment=alignment, code_revision=run.code_revision + ) + performance_table = next( + table + for entry in manifest.evidence + for table in entry.tables + if table.logical_name == "performance" + ) + run_document = run.to_dict() + payload: dict[str, Any] = { + "schema_version": "researchhub.performance-evidence.v2", + "authority": "quant_engine", + "scope": "offline_retrospective_research_only", + "run_id": run.run_id, + "usage": run.usage, + "historical_availability": run.historical_availability, + "evidence_scope": run.evidence_scope, + "observation_cutoff": run.observation_cutoff, + "decision_eligible": False, + "execution_validation": "not_validated", + "backtest_run_ref_id": run.run_id, + "backtest_run_ref_document_sha256": _digest_bytes(canonical_json_bytes(run_document)), + "backtest_evidence_manifest_id": manifest.manifest_id, + "backtest_evidence_manifest_document_sha256": _digest_bytes( + canonical_json_bytes(manifest.to_dict()) + ), + "backtest_evidence_manifest_evidence_digest": manifest.evidence_digest, + "backtest_evidence_qualification": manifest.qualification.value, + "research_artifact_schema_version": artifact.schema_version, + "research_artifact_content_digest": "sha256:" + artifact.content_sha256, + "artifact_available_at": manifest.artifact_available_at, + "computed_at": run.computed_at, + "performance_table_logical_name": performance_table.logical_name, + "performance_table_row_count": performance_table.row_count, + "performance_table_schema_digest": performance_table.schema_digest, + "performance_table_content_digest": performance_table.content_digest, + "performance_row_digest": row_digest, + "benchmark_series_digest": benchmark_digest, + "methodology_id": PERFORMANCE_METHODOLOGY_ID, + "metric_schema_id": PERFORMANCE_METRIC_SCHEMA_ID, + **{ + key: run_document[key] + for key in ( + "dataset_snapshot_id", + "dataset_content_digest", + "dataset_manifest_digest", + "foundation_id", + "foundation_digest", + "factor_set_id", + "factor_set_digest", + "factor_output_content_digest", + "strategy_id", + "strategy_version", + "strategy_digest", + "execution_model_version", + "execution_model_digest", + "cost_model_version", + "cost_model_digest", + "code_revision", + "environment_lock_digest", + "configuration_digest", + ) + }, + "frequency": frequency, + "calendar": calendar, + "timezone": timezone, + "benchmark_id": run_row["benchmark_id"], + "benchmark_alignment_policy": alignment, + "start_date": start_date, + "end_date": end_date, + "methodology": methodology.to_dict(), + "metrics": [metric.to_dict() for metric in metrics], + } + payload["performance_evidence_id"] = "rhperformancev2:" + _performance_digest(payload) + payload["document_sha256"] = _performance_digest(payload) + instance = object.__new__(RetrospectivePerformanceEvidence) + object.__setattr__(instance, "_payload", _freeze_numeric_evidence(payload)) + object.__setattr__(instance, "methodology", methodology) + object.__setattr__(instance, "metrics", metrics) + return instance diff --git a/src/quant_engine/retrospective_backtest_contracts.py b/src/quant_engine/retrospective_backtest_contracts.py new file mode 100644 index 0000000..a36fd67 --- /dev/null +++ b/src/quant_engine/retrospective_backtest_contracts.py @@ -0,0 +1,422 @@ +"""Explicit retrospective v2 run identities and offline artifact evidence.""" + +from __future__ import annotations + +from collections.abc import Mapping, Sequence +from dataclasses import dataclass, field +from typing import Any, Self, cast + +from quant_engine.factor_contracts import ( + ContractErrorCode, + PayloadValidation, + _assert_canonical_profile, + _content_address, + _digest, + _digest_bytes, + _freeze_json, + _git_revision, + _logical_id, + _parse_json_object, + _parse_utc, + _safe_integer, + _semver, + _thaw_json, + canonical_json, + canonical_json_bytes, +) +from quant_engine.retrospective_data_contracts import ( + RetrospectiveFoundationEnvelope, + RetrospectiveSnapshotEnvelope, + _IDS, + _check, + _public, + _shape, + _strings, +) +from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef, _context + +_CONFIG_FIELDS = ( + "universe_digest", + "strategy_id", + "strategy_version", + "strategy_digest", + "execution_model_version", + "execution_model_digest", + "cost_model_version", + "cost_model_digest", + "random_seed", + "code_revision", + "environment_lock_digest", + "configuration_digest", +) +_RUN_FIELDS = ( + "contract_name schema_version run_id dataset_snapshot_id dataset_content_digest dataset_manifest_digest " + "foundation_id foundation_digest factor_set_id factor_set_digest factor_output_content_digest " + "observation_cutoff evidence_scope usage historical_availability decision_eligible execution_validation " + "universe_digest trading_calendar_revision_ids trading_calendar_digest corporate_action_revision_ids corporate_action_digest " + "strategy_id strategy_version strategy_digest execution_model_version execution_model_digest cost_model_version cost_model_digest " + "random_seed code_revision environment_lock_digest configuration_digest evaluation_at computed_at replay_spec_digest " + "replay_parent_run_id replay_reason replay_attempt replay_ancestor_run_ids" +) + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectiveBacktestRunRef: + """New-major deterministic-input identity with separate actual attempt times.""" + + contract_name: str + schema_version: str + run_id: str + dataset_snapshot_id: str + dataset_content_digest: str + dataset_manifest_digest: str + foundation_id: str + foundation_digest: str + factor_set_id: str + factor_set_digest: str + factor_output_content_digest: str + observation_cutoff: str + evidence_scope: str + usage: str + historical_availability: str + decision_eligible: bool + execution_validation: str + universe_digest: str + trading_calendar_revision_ids: tuple[str, ...] + trading_calendar_digest: str + corporate_action_revision_ids: tuple[str, ...] + corporate_action_digest: str + strategy_id: str + strategy_version: str + strategy_digest: str + execution_model_version: str + execution_model_digest: str + cost_model_version: str + cost_model_digest: str + random_seed: int + code_revision: str + environment_lock_digest: str + configuration_digest: str + evaluation_at: str + computed_at: str + replay_spec_digest: str + replay_parent_run_id: str | None + replay_reason: str | None + replay_attempt: int + replay_ancestor_run_ids: tuple[str, ...] + input_payload_validation: PayloadValidation = field(compare=False) + _payload: Mapping[str, Any] = field(repr=False, compare=False) + _factor_set: RetrospectiveFactorSetRef = field(repr=False, compare=False) + _parent: RetrospectiveBacktestRunRef | None = field(repr=False, compare=False) + + @classmethod + def create( + cls, + *, + dataset_snapshot: RetrospectiveSnapshotEnvelope, + foundation: RetrospectiveFoundationEnvelope, + factor_set: RetrospectiveFactorSetRef, + universe_digest: str, + trading_calendar_revision_ids: Sequence[str], + corporate_action_revision_ids: Sequence[str], + strategy_id: str, + strategy_version: str, + strategy_digest: str, + execution_model_version: str, + execution_model_digest: str, + cost_model_version: str, + cost_model_digest: str, + random_seed: int, + code_revision: str, + environment_lock_digest: str, + configuration_digest: str, + evaluation_at: str, + computed_at: str, + parent: RetrospectiveBacktestRunRef | None = None, + replay_reason: str | None = None, + replay_attempt: int = 0, + ) -> Self: + _check( + type(factor_set) is RetrospectiveFactorSetRef, + "$.factor_set", + "explicit v2 factor result required", + ContractErrorCode.TYPE_ERROR, + ) + factor_set.require_payloads_revalidated() + return cls._build( + dataset_snapshot=dataset_snapshot, + foundation=foundation, + factor_set=factor_set, + trading_calendar_revision_ids=trading_calendar_revision_ids, + corporate_action_revision_ids=corporate_action_revision_ids, + configuration={ + "universe_digest": universe_digest, + "strategy_id": strategy_id, + "strategy_version": strategy_version, + "strategy_digest": strategy_digest, + "execution_model_version": execution_model_version, + "execution_model_digest": execution_model_digest, + "cost_model_version": cost_model_version, + "cost_model_digest": cost_model_digest, + "random_seed": random_seed, + "code_revision": code_revision, + "environment_lock_digest": environment_lock_digest, + "configuration_digest": configuration_digest, + }, + evaluation_at=evaluation_at, + computed_at=computed_at, + parent=parent, + replay_reason=replay_reason, + replay_attempt=replay_attempt, + ) + + @classmethod + def _build( + cls, + *, + dataset_snapshot: Any, + foundation: Any, + factor_set: Any, + trading_calendar_revision_ids: Any, + corporate_action_revision_ids: Any, + configuration: dict[str, Any], + evaluation_at: Any, + computed_at: Any, + parent: RetrospectiveBacktestRunRef | None, + replay_reason: Any, + replay_attempt: Any, + ) -> Self: + _check( + type(factor_set) is RetrospectiveFactorSetRef, + "$.factor_set", + "explicit v2 factor result required", + ContractErrorCode.TYPE_ERROR, + ) + definitions, snapshot, foundation = _context( + factor_set._definitions, dataset_snapshot, foundation + ) + # Reconstruct the serialized factor boundary against the exact supplied inputs. + checked_factor = RetrospectiveFactorSetRef.from_dict( + factor_set.to_dict(), + definitions=definitions, + dataset_snapshot=snapshot, + foundation=foundation, + parent=factor_set._parent, + ) + closures: dict[str, tuple[str, ...]] = {} + for field_name, supplied, kind in ( + ("trading_calendar_revision_ids", trading_calendar_revision_ids, "calendar_revision"), + ("corporate_action_revision_ids", corporate_action_revision_ids, "action_revision"), + ): + _check( + type(supplied) in {tuple, list}, + f"$.{field_name}", + "list/tuple required", + ContractErrorCode.TYPE_ERROR, + ) + supplied_ids = tuple( + sorted( + _strings( + list(supplied), + f"$.{field_name}", + _IDS[kind], + 1 if kind == "calendar_revision" else 0, + ) + ) + ) + expected_ids = tuple( + sorted( + { + identity + for view_id in checked_factor.selected_view_ref_ids + for identity in getattr(foundation.views[view_id], field_name) + } + ) + ) + _check( + supplied_ids == expected_ids, + f"$.{field_name}", + "exact selected observation ancestry required", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + closures[field_name] = supplied_ids + _shape(configuration, "$.configuration", " ".join(_CONFIG_FIELDS)) + for name, value in configuration.items(): + if name.endswith("_digest"): + _digest(value, f"$.{name}") + elif name.endswith("_version"): + _semver(value, f"$.{name}") + elif name == "random_seed": + _safe_integer(value, f"$.{name}", minimum=0) + elif name == "code_revision": + _git_revision(value, f"$.{name}") + else: + _logical_id(value, f"$.{name}") + _public(configuration, "$.configuration") + evaluation = _parse_utc(evaluation_at, "$.evaluation_at") + computed = _parse_utc(computed_at, "$.computed_at") + _check( + _parse_utc(checked_factor.artifact_available_at, "$.factor_set.artifact_available_at") + <= evaluation + <= computed, + "$.computed_at", + "factor availability <= actual evaluation <= computation required", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + content = snapshot.to_dict()["descriptor"]["content"] + spec: dict[str, Any] = { + "dataset_snapshot_id": snapshot.snapshot_id, + "dataset_content_digest": content["content_digest"], + "dataset_manifest_digest": content["manifest_digest"], + "foundation_id": foundation.foundation_id, + "foundation_digest": foundation.foundation_id.removeprefix("rhdfv2:"), + "factor_set_id": checked_factor.factor_set_id, + "factor_set_digest": checked_factor.factor_set_id.removeprefix("rhfactorsetv2:"), + "factor_output_content_digest": checked_factor.output_content_digest, + "observation_cutoff": foundation.observation_cutoff, + "evidence_scope": checked_factor.evidence_scope, + "usage": "retrospective_research", + "historical_availability": "not_established", + "decision_eligible": False, + "execution_validation": "not_validated", + **configuration, + } + for field_name, identities in closures.items(): + spec[field_name] = list(identities) + digest_field = ( + "trading_calendar_digest" + if field_name == "trading_calendar_revision_ids" + else "corporate_action_digest" + ) + spec[digest_field] = _digest_bytes(canonical_json_bytes(list(identities))) + # v2 replay specification excludes BOTH actual attempt times. They remain in + # run_id, so a replay never backdates evaluation to manufacture equality. + replay_spec_digest = _digest_bytes(canonical_json_bytes(spec)) + replay_count = _safe_integer(replay_attempt, "$.replay_attempt", minimum=0) + if parent is None: + _check( + replay_reason is None and replay_count == 0, + "$.replay_attempt", + "root must use zero attempt and no reason", + ContractErrorCode.LINEAGE_VIOLATION, + ) + parent_id = None + ancestors: tuple[str, ...] = () + else: + _check( + type(parent) is RetrospectiveBacktestRunRef, + "$.parent", + "exact v2 run parent required", + ContractErrorCode.TYPE_ERROR, + ) + _logical_id(replay_reason, "$.replay_reason") + _check( + replay_count == parent.replay_attempt + 1 + and replay_spec_digest == parent.replay_spec_digest, + "$.replay_spec_digest", + "replay requires unchanged inputs and the next attempt", + ContractErrorCode.LINEAGE_VIOLATION, + ) + _check( + _parse_utc(parent.computed_at, "$.parent.computed_at") < evaluation <= computed, + "$.evaluation_at", + "new actual attempt must follow parent computation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + parent_id = parent.run_id + ancestors = (*parent.replay_ancestor_run_ids, parent_id) + _check( + len(ancestors) == len(set(ancestors)), + "$.replay_ancestor_run_ids", + "replay cycle", + ContractErrorCode.LINEAGE_VIOLATION, + ) + payload = { + "contract_name": "researchhub.backtest-run-ref", + "schema_version": "2.0.0", + **spec, + "evaluation_at": evaluation_at, + "computed_at": computed_at, + "replay_spec_digest": replay_spec_digest, + "replay_parent_run_id": parent_id, + "replay_reason": replay_reason, + "replay_attempt": replay_count, + "replay_ancestor_run_ids": list(ancestors), + } + payload["run_id"] = _content_address(payload, "run_id", "rhbacktestrunv2:sha256:") + _check( + payload["run_id"] not in ancestors, + "$.run_id", + "self-parent cycle", + ContractErrorCode.LINEAGE_VIOLATION, + ) + verified = ( + factor_set.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED + and factor_set.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED + ) + instance = object.__new__(cls) + for name, value in { + **payload, + **closures, + "replay_ancestor_run_ids": ancestors, + "input_payload_validation": PayloadValidation.PAYLOAD_REVALIDATED + if verified + else PayloadValidation.REFERENCE_ONLY, + "_payload": _freeze_json(payload), + "_factor_set": factor_set, + "_parent": parent, + }.items(): + object.__setattr__(instance, name, value) + return instance + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return canonical_json(self.to_dict()) + + def require_inputs_revalidated(self) -> None: + _check( + self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED, + "$.input_payload_validation", + "reference-only factors cannot admit a new computation", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + + @classmethod + def from_dict( + cls, + value: Any, + *, + dataset_snapshot: RetrospectiveSnapshotEnvelope, + foundation: RetrospectiveFoundationEnvelope, + factor_set: RetrospectiveFactorSetRef, + parent: RetrospectiveBacktestRunRef | None = None, + ) -> Self: + _assert_canonical_profile(value) + _public(value) + row = _shape(value, "$", _RUN_FIELDS) + rebuilt = cls._build( + dataset_snapshot=dataset_snapshot, + foundation=foundation, + factor_set=factor_set, + trading_calendar_revision_ids=row["trading_calendar_revision_ids"], + corporate_action_revision_ids=row["corporate_action_revision_ids"], + configuration={key: row[key] for key in _CONFIG_FIELDS}, + evaluation_at=row["evaluation_at"], + computed_at=row["computed_at"], + parent=parent, + replay_reason=row["replay_reason"], + replay_attempt=row["replay_attempt"], + ) + _check( + canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()), + "$", + "serialized run differs from exact v2 input/configuration/lineage closure", + ContractErrorCode.IDENTITY_MISMATCH, + ) + return rebuilt + + @classmethod + def from_json(cls, value: str | bytes, **kwargs: Any) -> Self: + return cls.from_dict(_parse_json_object(value, "$"), **kwargs) diff --git a/src/quant_engine/retrospective_data_contracts.py b/src/quant_engine/retrospective_data_contracts.py new file mode 100644 index 0000000..3e20a43 --- /dev/null +++ b/src/quant_engine/retrospective_data_contracts.py @@ -0,0 +1,1063 @@ +"""Immutable, closed-world v2 decoding and materialized-content verification. + +These pure functions validate declared facts, not source authenticity. Real admission +still needs trusted receipt, qualification, view-resolution and clock ports. The v1 +wire formats and their knowledge/PIT semantics are deliberately not reused here. +""" + +from __future__ import annotations + +import re +from collections import defaultdict +from collections.abc import Mapping, Sequence +from dataclasses import dataclass, field +from datetime import datetime +from itertools import pairwise +from types import MappingProxyType +from typing import Any, Self, cast + +from quant_engine.factor_contracts import ( + ContractErrorCode, + _array, + _assert_canonical_profile, + _content_address, + _digest, + _digest_bytes, + _fail, + _freeze_json, + _object, + _parse_date, + _parse_json_object, + _parse_utc, + _safe_integer, + _string, + _thaw_json, + canonical_json, + canonical_json_bytes, +) + + +CHECKS = frozenset( + { + "completeness", + "duplicate_identity", + "observation_coverage", + "historical_claim_policy", + "range_validity", + "schema_conformance", + } +) +_FORBIDDEN = frozenset( + { + "latest", + "provider", + "table", + "sql", + "locator", + "dsn", + "credential", + "tushare", + "wind", + "bloomberg", + "akshare", + } +) +_LOCATION = re.compile(r"^(?:/|\./|\.\./|\\|[a-z]:[\\/]|[a-z][a-z0-9+.-]*://)", re.I) +_VERSION = re.compile(r"^[0-9]+\.[0-9]+\.[0-9]+$") +_IDS = { + name: re.compile(rf"^{prefix}:[0-9a-f]{{{size}}}$") + for name, prefix, size in ( + ("snapshot", "rhdsv2:sha256", 64), + ("foundation", "rhdfv2:sha256", 64), + ("route", "rhroutev2:sha256", 64), + ("calendar_revision", "rhcalv2:sha256", 64), + ("action_revision", "rhcav2:sha256", 64), + ("view_ref", "rhviewrefv2:sha256", 64), + ("instrument", "rhinstrument", 32), + ("calendar", "rhcalendar", 32), + ("action", "rhaction", 32), + ("view", "rhview", 32), + ) +} + + +def _check( + condition: bool, + path: str, + detail: str, + code: ContractErrorCode = ContractErrorCode.INVALID_VALUE, +) -> None: + if not condition: + _fail(code, path, detail) + + +def _shape(value: Any, path: str, required: str, optional: str = "") -> dict[str, Any]: + return _object(value, path, required.split(), optional.split()) + + +def _choice(value: Any, path: str, choices: Sequence[str] | set[str] | frozenset[str]) -> str: + text = _string(value, path) + _check(text in choices, path, "unsupported value") + return text + + +def _public(value: Any, path: str = "$") -> None: + """Keep physical details out of both known fields and opaque public strings.""" + if type(value) is dict: + for key, child in value.items(): + _public(key, path) + _public(child, f"{path}.{key}") + elif type(value) is list: + for index, child in enumerate(value): + _public(child, f"{path}[{index}]") + elif type(value) is str: + _check( + _LOCATION.match(value) is None + and not set(re.findall(r"[a-z0-9]+", value.lower())) & _FORBIDDEN, + path, + "physical/private detail in public contract", + ) + + +def _range(value: Any, path: str) -> tuple[datetime, datetime]: + row = _shape(value, path, "start_inclusive end_inclusive") + start = _parse_utc(row["start_inclusive"], f"{path}.start_inclusive") + end = _parse_utc(row["end_inclusive"], f"{path}.end_inclusive") + _check(start <= end, path, "inverted range", ContractErrorCode.TIME_ORDER_VIOLATION) + return start, end + + +def _knowledge(value: Any, path: str, observed: datetime, *, ranged: bool = False) -> None: + row = _shape( + value, path, "status", "range evidence_digest" if ranged else "earliest_at evidence_digest" + ) + status = _choice(row["status"], f"{path}.status", {"unknown", "evidenced"}) + if status == "unknown": + _shape(row, path, "status") + return + _shape( + row, + path, + "status range evidence_digest" if ranged else "status earliest_at evidence_digest", + ) + _digest(row["evidence_digest"], f"{path}.evidence_digest") + end = ( + _range(row["range"], f"{path}.range")[1] + if ranged + else _parse_utc(row["earliest_at"], f"{path}.earliest_at") + ) + _check( + end <= observed, + path, + "knowledge exceeds observation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + + +def _identity(document: dict[str, Any], field_name: str, kind: str, prefix: str, path: str) -> None: + identity = _string(document[field_name], f"{path}.{field_name}", _IDS[kind]) + _check( + identity == _content_address(document, field_name, prefix + ":sha256:"), + f"{path}.{field_name}", + "content address mismatch", + ContractErrorCode.IDENTITY_MISMATCH, + ) + + +def _restrictions(row: dict[str, Any], path: str) -> None: + _choice(row["usage"], f"{path}.usage", {"retrospective_research"}) + _choice(row["historical_availability"], f"{path}.historical_availability", {"not_established"}) + + +def _strings(value: Any, path: str, pattern: re.Pattern[str], minimum: int = 0) -> tuple[str, ...]: + return tuple( + _string(item, f"{path}[{index}]", pattern) + for index, item in enumerate(_array(value, path, minimum=minimum, unique=True)) + ) + + +def _validate_snapshot(document: dict[str, Any]) -> None: + _assert_canonical_profile(document) + _public(document) + _shape(document, "$", "contract_name schema_version evidence_scope snapshot_id descriptor") + _choice(document["contract_name"], "$.contract_name", {"researchhub.dataset-snapshot"}) + _choice(document["schema_version"], "$.schema_version", {"2.0.0"}) + _choice(document["evidence_scope"], "$.evidence_scope", {"synthetic_fixture", "real_data"}) + descriptor = _shape( + document["descriptor"], + "$.descriptor", + "dataset published_at time_semantics content observation_manifest lineage quality qualification", + ) + dataset = _shape( + descriptor["dataset"], + "$.descriptor.dataset", + "dataset_id dataset_kind record_schema_version dimensions", + ) + kind = _choice( + dataset["dataset_kind"], "$.descriptor.dataset.dataset_kind", {"market", "macroeconomic"} + ) + _string( + dataset["dataset_id"], + "$.descriptor.dataset.dataset_id", + re.compile(rf"^rhdataset:{kind}:[0-9a-f]{{32}}$"), + ) + _string( + dataset["record_schema_version"], "$.descriptor.dataset.record_schema_version", _VERSION + ) + _array(dataset["dimensions"], "$.descriptor.dataset.dimensions", minimum=1, unique=True) + expected = ( + ["instrument_id", "effective_time"] + if kind == "market" + else ["series_id", "observation_period"] + ) + _check( + dataset["dimensions"] == expected, + "$.descriptor.dataset.dimensions", + "kind dimensions mismatch", + ) + temporal = _shape( + descriptor["time_semantics"], + "$.descriptor.time_semantics", + "effective_time observation_cutoff earliest_external_knowledge historical_availability", + ) + _range(temporal["effective_time"], "$.descriptor.time_semantics.effective_time") + _choice( + temporal["historical_availability"], + "$.descriptor.time_semantics.historical_availability", + {"not_established"}, + ) + cutoff = _parse_utc( + temporal["observation_cutoff"], "$.descriptor.time_semantics.observation_cutoff" + ) + published = _parse_utc(descriptor["published_at"], "$.descriptor.published_at") + content = _shape( + descriptor["content"], + "$.descriptor.content", + "digest_algorithm canonicalization record_order content_digest logical_manifest manifest_digest record_count", + ) + for key, literal in ( + ("digest_algorithm", "sha256"), + ("canonicalization", "RFC8785"), + ("record_order", "canonical-record-byte-order"), + ): + _choice(content[key], f"$.descriptor.content.{key}", {literal}) + _digest(content["content_digest"], "$.descriptor.content.content_digest") + count = _safe_integer(content["record_count"], "$.descriptor.content.record_count", minimum=1) + manifest = _shape( + content["logical_manifest"], "$.descriptor.content.logical_manifest", "record_count chunks" + ) + manifest_count = _safe_integer( + manifest["record_count"], "$.descriptor.content.logical_manifest.record_count", minimum=1 + ) + chunks = _array( + manifest["chunks"], "$.descriptor.content.logical_manifest.chunks", minimum=1, unique=True + ) + observation = _shape( + descriptor["observation_manifest"], "$.descriptor.observation_manifest", "batches" + ) + batches = _array( + observation["batches"], "$.descriptor.observation_manifest.batches", minimum=1, unique=True + ) + _check( + len(batches) == len(chunks), + "$.descriptor.observation_manifest", + "exact chunk coverage required", + ) + total = 0 + observed_times: list[datetime] = [] + for index, (raw_chunk, raw_batch) in enumerate(zip(chunks, batches, strict=True)): + path = f"$.descriptor.content.logical_manifest.chunks[{index}]" + chunk = _shape(raw_chunk, path, "chunk_index content_digest record_count") + _check( + _safe_integer(chunk["chunk_index"], f"{path}.chunk_index", minimum=0) == index, + path, + "non-contiguous chunk index", + ) + _digest(chunk["content_digest"], f"{path}.content_digest") + total += _safe_integer(chunk["record_count"], f"{path}.record_count", minimum=1) + path = f"$.descriptor.observation_manifest.batches[{index}]" + batch = _shape( + raw_batch, + path, + "chunk_index content_digest record_count observation_kind observed_by evidence_digest", + ) + _safe_integer(batch["chunk_index"], f"{path}.chunk_index", minimum=0) + _safe_integer(batch["record_count"], f"{path}.record_count", minimum=1) + _check( + all(batch[key] == chunk[key] for key in chunk), path, "batch does not bind exact chunk" + ) + _choice(batch["observation_kind"], f"{path}.observation_kind", {"observed_by"}) + _digest(batch["evidence_digest"], f"{path}.evidence_digest") + observed = _parse_utc(batch["observed_by"], f"{path}.observed_by") + _check( + observed <= cutoff, + path, + "observation exceeds cutoff", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + observed_times.append(observed) + _check(count == manifest_count == total, "$.descriptor.content", "record counts mismatch") + _digest(content["manifest_digest"], "$.descriptor.content.manifest_digest") + _check( + content["manifest_digest"] == _digest_bytes(canonical_json_bytes(manifest)), + "$.descriptor.content.manifest_digest", + "manifest mismatch", + ContractErrorCode.IDENTITY_MISMATCH, + ) + _knowledge( + temporal["earliest_external_knowledge"], + "$.descriptor.time_semantics.earliest_external_knowledge", + max(observed_times), + ranged=True, + ) + lineage = _shape( + descriptor["lineage"], + "$.descriptor.lineage", + "publisher transformation upstream_snapshot_ids upstream_content_digests", + ) + for name, pattern in ( + ("publisher", re.compile(r"^researchhub\.data$")), + ("transformation", re.compile(r"^rhtransform:[0-9a-f]{32}$")), + ): + item = _shape(lineage[name], f"$.descriptor.lineage.{name}", "id version") + _string(item["id"], f"$.descriptor.lineage.{name}.id", pattern) + _string(item["version"], f"$.descriptor.lineage.{name}.version", _VERSION) + _strings( + lineage["upstream_snapshot_ids"], + "$.descriptor.lineage.upstream_snapshot_ids", + re.compile(r"^rhdsv[12]:sha256:[0-9a-f]{64}$"), + ) + _strings( + lineage["upstream_content_digests"], + "$.descriptor.lineage.upstream_content_digests", + re.compile(r"^sha256:[0-9a-f]{64}$"), + ) + quality = _shape(descriptor["quality"], "$.descriptor.quality", "status checks") + status = _choice(quality["status"], "$.descriptor.quality.status", {"passed", "failed"}) + checks = _array(quality["checks"], "$.descriptor.quality.checks", minimum=6, unique=True) + ids: list[str] = [] + passed = True + for index, raw in enumerate(checks): + path = f"$.descriptor.quality.checks[{index}]" + check = _shape(raw, path, "check_id status severity evidence_digest") + ids.append(_choice(check["check_id"], f"{path}.check_id", CHECKS)) + item_status = _choice(check["status"], f"{path}.status", {"passed", "failed"}) + passed = passed and item_status == "passed" + _choice(check["severity"], f"{path}.severity", {"blocking"}) + _digest(check["evidence_digest"], f"{path}.evidence_digest") + _check( + len(ids) == len(CHECKS) and set(ids) == CHECKS, + "$.descriptor.quality.checks", + "each required check must occur once", + ) + _check((status == "passed") == passed, "$.descriptor.quality", "summary differs from checks") + qualification = _shape( + descriptor["qualification"], + "$.descriptor.qualification", + "status usage policy_id policy_version evaluated_at evidence_digest", + ) + qualification_status = _choice( + qualification["status"], "$.descriptor.qualification.status", {"qualified", "rejected"} + ) + for key, literal in ( + ("usage", "retrospective_research"), + ("policy_id", "researchhub.dataset-snapshot.retrospective"), + ("policy_version", "2.0.0"), + ): + _choice(qualification[key], f"$.descriptor.qualification.{key}", {literal}) + _digest(qualification["evidence_digest"], "$.descriptor.qualification.evidence_digest") + evaluated = _parse_utc(qualification["evaluated_at"], "$.descriptor.qualification.evaluated_at") + _check( + cutoff <= evaluated <= published, + "$.descriptor", + "cutoff <= qualification <= publication required", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + _check( + qualification_status != "qualified" or passed, + "$.descriptor.qualification", + "failed quality", + ContractErrorCode.QUALIFICATION_REJECTED, + ) + _identity(document, "snapshot_id", "snapshot", "rhdsv2", "$") + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectiveSnapshotEnvelope: + """Validated declarations; materialized records and owner evidence are separate.""" + + _payload: Mapping[str, Any] = field(repr=False) + + @classmethod + def from_dict(cls, value: Any) -> Self: + _validate_snapshot(value) + instance = object.__new__(cls) + object.__setattr__(instance, "_payload", _freeze_json(value)) + return instance + + @classmethod + def from_json(cls, value: str | bytes) -> Self: + return cls.from_dict(_parse_json_object(value, "$")) + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return canonical_json(self.to_dict()) + + @property + def snapshot_id(self) -> str: + return cast(str, self._payload["snapshot_id"]) + + @property + def evidence_scope(self) -> str: + return cast(str, self._payload["evidence_scope"]) + + @property + def observation_cutoff(self) -> str: + return cast(str, self._payload["descriptor"]["time_semantics"]["observation_cutoff"]) + + @property + def published_at(self) -> datetime: + return _parse_utc(self._payload["descriptor"]["published_at"], "$.descriptor.published_at") + + @property + def earliest_external_knowledge(self) -> Mapping[str, Any]: + return cast( + Mapping[str, Any], + self._payload["descriptor"]["time_semantics"]["earliest_external_knowledge"], + ) + + @property + def effective_range(self) -> tuple[datetime, datetime]: + return _range( + self.to_dict()["descriptor"]["time_semantics"]["effective_time"], + "$.descriptor.time_semantics.effective_time", + ) + + def require_qualified(self) -> None: + descriptor = self._payload["descriptor"] + _check( + descriptor["qualification"]["status"] == "qualified" + and descriptor["quality"]["status"] == "passed", + "$.descriptor.qualification", + "snapshot is not qualified", + ContractErrorCode.QUALIFICATION_REJECTED, + ) + + def verify_materialized_records(self, chunks: Any) -> None: + """Bind bytes, dimensions and effective range; never authenticate receipts.""" + _assert_canonical_profile(chunks) + _public(chunks) + values = _array(chunks, "$.chunks", minimum=1) + descriptor = self._payload["descriptor"] + content = descriptor["content"] + declarations = content["logical_manifest"]["chunks"] + _check( + len(values) == len(declarations), + "$.chunks", + "chunk count mismatch", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + records: list[dict[str, Any]] = [] + for index, (value, declaration) in enumerate(zip(values, declarations, strict=True)): + rows = _array(value, f"$.chunks[{index}]", minimum=1) + _check( + all(type(row) is dict for row in rows), + f"$.chunks[{index}]", + "records must be objects", + ContractErrorCode.TYPE_ERROR, + ) + _check( + len(rows) == declaration["record_count"] + and _records_digest(rows) == declaration["content_digest"], + f"$.chunks[{index}]", + "materialized chunk mismatch", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + records.extend(rows) + _check( + _records_digest(records) == content["content_digest"], + "$.chunks", + "whole content mismatch", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + dimensions = descriptor["dataset"]["dimensions"] + keys: set[bytes] = set() + instants: list[datetime] = [] + for index, row in enumerate(records): + path = f"$.records[{index}]" + _check( + not ({"knowledge_time", "pit_cutoff", "available_at"} & row.keys()), + path, + "legacy time coercion prohibited", + ) + for dimension in dimensions: + dimension_value = _string(row.get(dimension), f"{path}.{dimension}") + _check( + bool(dimension_value), + f"{path}.{dimension}", + "logical dimension cannot be empty", + ) + key = canonical_json_bytes([row[name] for name in dimensions]) + _check(key not in keys, path, "duplicate logical record identity") + keys.add(key) + instants.append(_parse_utc(row.get("effective_time"), f"{path}.effective_time")) + _check( + (min(instants), max(instants)) == self.effective_range, + "$.records", + "effective range mismatch", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + + +def _records_digest(rows: list[dict[str, Any]]) -> str: + return _digest_bytes(b"[" + b",".join(sorted(canonical_json_bytes(row) for row in rows)) + b"]") + + +_COLLECTIONS = ( + ( + "instrument_routes", + "route_revision_id", + "route", + "rhroutev2", + "instrument_route", + ("instrument_id",), + ), + ( + "trading_calendar_revisions", + "calendar_revision_id", + "calendar_revision", + "rhcalv2", + "trading_calendar", + ("calendar_id", "session_date"), + ), + ( + "corporate_action_revisions", + "action_revision_id", + "action_revision", + "rhcav2", + "corporate_action", + ("action_id",), + ), +) +_OBSERVATION_FIELDS = ( + "observation_sequence", + "observed_by", + "earliest_external_knowledge", + "history_completeness", + "evidence_digest", + "supersedes_observation_id", +) +_ASSET_TYPES = { + "equity": "stock", + "fund": "etf", + "fixed_income": "bond", + "future": "future", + "option": "option", + "index": "index", +} +_COMMON_OBSERVATION = "observation_sequence observed_by earliest_external_knowledge history_completeness evidence_digest" + + +def _validate_observation(value: Any, kind: str, path: str, cutoff: datetime) -> dict[str, Any]: + details = { + "route": "route_revision_id instrument_id symbol mic currency asset_class instrument_type calendar_id effective_from", + "calendar_revision": "calendar_revision_id calendar_id session_date status sessions", + "action_revision": "action_revision_id action_id instrument_id action_type status effective_time terms_digest", + } + row = _shape( + value, path, _COMMON_OBSERVATION + " " + details[kind], "supersedes_observation_id" + ) + _safe_integer(row["observation_sequence"], f"{path}.observation_sequence", minimum=1) + observed = _parse_utc(row["observed_by"], f"{path}.observed_by") + _check( + observed <= cutoff, + path, + "observation exceeds cutoff", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + _knowledge(row["earliest_external_knowledge"], f"{path}.earliest_external_knowledge", observed) + _choice(row["history_completeness"], f"{path}.history_completeness", {"not_established"}) + _digest(row["evidence_digest"], f"{path}.evidence_digest") + if "supersedes_observation_id" in row: + _string(row["supersedes_observation_id"], f"{path}.supersedes_observation_id", _IDS[kind]) + if kind != "calendar_revision": + _string(row["instrument_id"], f"{path}.instrument_id", _IDS["instrument"]) + if kind == "route": + symbol = _string(row["symbol"], f"{path}.symbol", re.compile(r"^[A-Z0-9][A-Z0-9.-]*$")) + _check(len(symbol) <= 32, f"{path}.symbol", "symbol too long") + _string(row["mic"], f"{path}.mic", re.compile(r"^[A-Z0-9]{4}$")) + _string(row["currency"], f"{path}.currency", re.compile(r"^[A-Z]{3}$")) + asset = _choice(row["asset_class"], f"{path}.asset_class", set(_ASSET_TYPES)) + _choice(row["instrument_type"], f"{path}.instrument_type", {_ASSET_TYPES[asset]}) + _string(row["calendar_id"], f"{path}.calendar_id", _IDS["calendar"]) + _parse_utc(row["effective_from"], f"{path}.effective_from") + elif kind == "calendar_revision": + _string(row["calendar_id"], f"{path}.calendar_id", _IDS["calendar"]) + _parse_date(row["session_date"], f"{path}.session_date") + status = _choice(row["status"], f"{path}.status", {"open", "closed"}) + sessions = _array(row["sessions"], f"{path}.sessions", unique=True) + _check( + bool(sessions) == (status == "open"), f"{path}.sessions", "open/closed session mismatch" + ) + spans = [] + for index, session in enumerate(sessions): + segment = _shape(session, f"{path}.sessions[{index}]", "opens_at closes_at") + start = _parse_utc(segment["opens_at"], f"{path}.sessions[{index}].opens_at") + end = _parse_utc(segment["closes_at"], f"{path}.sessions[{index}].closes_at") + _check( + start < end, + f"{path}.sessions[{index}]", + "inverted session", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + spans.append((start, end)) + _check( + spans == sorted(spans) and all(left[1] <= right[0] for left, right in pairwise(spans)), + f"{path}.sessions", + "unordered/overlapping sessions", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + else: + _string(row["action_id"], f"{path}.action_id", _IDS["action"]) + _choice( + row["action_type"], + f"{path}.action_type", + { + "cash_dividend", + "stock_dividend", + "split", + "rights_issue", + "symbol_change", + "delisting", + }, + ) + _choice(row["status"], f"{path}.status", {"announced", "confirmed", "cancelled"}) + _parse_utc(row["effective_time"], f"{path}.effective_time") + _digest(row["terms_digest"], f"{path}.terms_digest") + return row + + +def _chain(rows: list[dict[str, Any]], identity: str, grouping: tuple[str, ...], path: str) -> None: + groups: dict[tuple[str, ...], list[dict[str, Any]]] = defaultdict(list) + for row in rows: + groups[tuple(row[key] for key in grouping)].append(row) + for group in groups.values(): + ordered = sorted(group, key=lambda row: row["observation_sequence"]) + _check( + [row["observation_sequence"] for row in ordered] == list(range(1, len(ordered) + 1)), + path, + "observation sequence gap/duplicate", + ContractErrorCode.LINEAGE_VIOLATION, + ) + _check( + "supersedes_observation_id" not in ordered[0], + path, + "first retained observation has missing parent", + ContractErrorCode.LINEAGE_VIOLATION, + ) + for previous, current in pairwise(ordered): + _check( + current.get("supersedes_observation_id") == previous[identity], + path, + "observation parent mismatch", + ContractErrorCode.LINEAGE_VIOLATION, + ) + _check( + _parse_utc(previous["observed_by"], path) + < _parse_utc(current["observed_by"], path), + path, + "observations must increase strictly", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + + +def _selected( + value: Any, rows: dict[str, dict[str, Any]], kind: str, path: str, minimum: int +) -> tuple[str, ...]: + selected = _strings(value, path, _IDS[kind], minimum) + _check( + set(selected) <= rows.keys(), + path, + "unknown selected observation", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + for identity in selected: + parent = rows[identity].get("supersedes_observation_id") + _check( + parent is None or parent in selected, + path, + "selected ancestry missing", + ContractErrorCode.LINEAGE_VIOLATION, + ) + return selected + + +def _readiness_level(value: Any, path: str, allowed: set[str]) -> tuple[str, tuple[str, ...]]: + level = _shape(value, path, "status evidence_digests") + status = _choice(level["status"], f"{path}.status", allowed) + evidence = tuple( + _digest(item, f"{path}.evidence_digests[{index}]") + for index, item in enumerate( + _array(level["evidence_digests"], f"{path}.evidence_digests", unique=True) + ) + ) + _check( + bool(evidence) == (status == "validated"), + path, + "evidence/status mismatch", + ContractErrorCode.READINESS_ESCALATION, + ) + return status, evidence + + +def _validate_foundation(document: Any, snapshot: RetrospectiveSnapshotEnvelope) -> None: + _check( + type(snapshot) is RetrospectiveSnapshotEnvelope, + "$.snapshot", + "explicit v2 snapshot required", + ContractErrorCode.TYPE_ERROR, + ) + # Revalidate the actual wire payload; never trust independently supplied summary fields. + snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot.to_dict()) + snapshot.require_qualified() + _assert_canonical_profile(document) + _public(document) + root = _shape( + document, + "$", + "contract_name schema_version foundation_id dataset_snapshot_id observation_cutoff instrument_routes " + "trading_calendar_revisions corporate_action_revisions standardized_views observation_lineage readiness " + "usage historical_availability published_at corporate_action_coverage", + ) + _choice(root["contract_name"], "$.contract_name", {"researchhub.data-foundation"}) + _choice(root["schema_version"], "$.schema_version", {"2.0.0"}) + _restrictions(root, "$") + _string(root["dataset_snapshot_id"], "$.dataset_snapshot_id", _IDS["snapshot"]) + _check( + root["dataset_snapshot_id"] == snapshot.snapshot_id, + "$.dataset_snapshot_id", + "snapshot mismatch", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + cutoff = _parse_utc(root["observation_cutoff"], "$.observation_cutoff") + published = _parse_utc(root["published_at"], "$.published_at") + _check( + _parse_utc(snapshot.observation_cutoff, "$.snapshot.observation_cutoff") + <= cutoff + <= published + and snapshot.published_at <= published, + "$", + "foundation precedes its inputs", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + collections: dict[str, dict[str, dict[str, Any]]] = {} + expected_lineage: dict[str, dict[str, Any]] = {} + for name, identity, kind, prefix, lineage_kind, grouping in _COLLECTIONS: + raw = _array( + root[name], f"$.{name}", minimum=0 if kind == "action_revision" else 1, unique=True + ) + rows = [ + _validate_observation(row, kind, f"$.{name}[{index}]", cutoff) + for index, row in enumerate(raw) + ] + index_by_id: dict[str, dict[str, Any]] = {} + for index, row in enumerate(rows): + _identity(row, identity, kind, prefix, f"$.{name}[{index}]") + _check(row[identity] not in index_by_id, f"$.{name}", "duplicate observation identity") + index_by_id[row[identity]] = row + expected_lineage[row[identity]] = { + "revision_kind": lineage_kind, + "revision_id": row[identity], + **{key: row[key] for key in _OBSERVATION_FIELDS if key in row}, + } + _chain(rows, identity, grouping, f"$.{name}") + collections[name] = index_by_id + lineage = _array(root["observation_lineage"], "$.observation_lineage", minimum=1, unique=True) + lineage_ids = [] + for index, row in enumerate(lineage): + path = f"$.observation_lineage[{index}]" + _shape( + row, + path, + "revision_kind revision_id " + _COMMON_OBSERVATION, + "supersedes_observation_id", + ) + _safe_integer(row["observation_sequence"], f"{path}.observation_sequence", minimum=1) + revision_id = _string(row["revision_id"], f"{path}.revision_id") + _check( + row == expected_lineage.get(revision_id), + path, + "lineage facts mismatch", + ContractErrorCode.LINEAGE_VIOLATION, + ) + lineage_ids.append(revision_id) + _check( + len(lineage_ids) == len(expected_lineage) and set(lineage_ids) == expected_lineage.keys(), + "$.observation_lineage", + "lineage must cover observations exactly", + ContractErrorCode.LINEAGE_VIOLATION, + ) + routes = collections["instrument_routes"] + calendars = collections["trading_calendar_revisions"] + actions = collections["corporate_action_revisions"] + instruments = {row["instrument_id"] for row in routes.values()} + calendar_ids = {row["calendar_id"] for row in calendars.values()} + _check( + all(row["calendar_id"] in calendar_ids for row in routes.values()), + "$.instrument_routes", + "route has no calendar", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + _check( + all(row["instrument_id"] in instruments for row in actions.values()), + "$.corporate_action_revisions", + "action has no instrument", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + coverage: dict[str, dict[str, Any]] = {} + for index, row in enumerate( + _array( + root["corporate_action_coverage"], "$.corporate_action_coverage", minimum=1, unique=True + ) + ): + path = f"$.corporate_action_coverage[{index}]" + _shape(row, path, "instrument_id effective_time observed_by status evidence_digests") + instrument = _string(row["instrument_id"], f"{path}.instrument_id", _IDS["instrument"]) + _check( + instrument in instruments and instrument not in coverage, + path, + "unknown/duplicate coverage instrument", + ) + _range(row["effective_time"], f"{path}.effective_time") + _check( + _parse_utc(row["observed_by"], f"{path}.observed_by") <= cutoff, + path, + "coverage exceeds cutoff", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + _readiness_level( + {"status": row["status"], "evidence_digests": row["evidence_digests"]}, + path, + {"validated", "not_validated"}, + ) + coverage[instrument] = row + selected_instruments: set[str] = set() + views = _array(root["standardized_views"], "$.standardized_views", minimum=1, unique=True) + view_ids: set[str] = set() + for index, row in enumerate(views): + path = f"$.standardized_views[{index}]" + _shape( + row, + path, + "view_ref_id view_id view_version dataset_snapshot_id observation_cutoff schema_digest content_digest " + "transformation_digest instrument_route_revision_ids trading_calendar_revision_ids corporate_action_revision_ids " + "usage historical_availability available_at", + ) + _identity(row, "view_ref_id", "view_ref", "rhviewrefv2", path) + _check(row["view_ref_id"] not in view_ids, path, "duplicate view identity") + view_ids.add(row["view_ref_id"]) + _string(row["view_id"], f"{path}.view_id", _IDS["view"]) + _string(row["view_version"], f"{path}.view_version", _VERSION) + _check( + row["dataset_snapshot_id"] == snapshot.snapshot_id + and row["observation_cutoff"] == root["observation_cutoff"], + path, + "view input/cutoff mismatch", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + _restrictions(row, path) + available = _parse_utc(row["available_at"], f"{path}.available_at") + _check( + max(cutoff, snapshot.published_at) <= available <= published, + path, + "view availability out of order", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + for key in ("schema_digest", "content_digest", "transformation_digest"): + _digest(row[key], f"{path}.{key}") + selected_routes = _selected( + row["instrument_route_revision_ids"], + routes, + "route", + f"{path}.instrument_route_revision_ids", + 1, + ) + selected_calendars = _selected( + row["trading_calendar_revision_ids"], + calendars, + "calendar_revision", + f"{path}.trading_calendar_revision_ids", + 1, + ) + selected_actions = _selected( + row["corporate_action_revision_ids"], + actions, + "action_revision", + f"{path}.corporate_action_revision_ids", + 0, + ) + _check( + {routes[key]["calendar_id"] for key in selected_routes} + <= {calendars[key]["calendar_id"] for key in selected_calendars}, + path, + "view must select its own route calendars", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + selected = {routes[key]["instrument_id"] for key in selected_routes} + _check( + all(actions[key]["instrument_id"] in selected for key in selected_actions), + path, + "action lacks selected instrument", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + _check( + selected <= coverage.keys(), + path, + "explicit action coverage missing", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + for instrument in selected: + item = coverage[instrument] + if item["status"] == "validated": + start, end = _range(item["effective_time"], f"{path}.coverage") + _check( + start <= snapshot.effective_range[0] <= snapshot.effective_range[1] <= end, + path, + "action coverage does not span input range", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + selected_instruments.update(selected) + readiness = _shape( + root["readiness"], + "$.readiness", + "evidence_scope contract_validation real_data_validation production_validation live_validation", + ) + _check( + readiness["evidence_scope"] == snapshot.evidence_scope, + "$.readiness.evidence_scope", + "scope mismatch", + ContractErrorCode.READINESS_ESCALATION, + ) + seen: set[str] = set() + for name in ( + "contract_validation", + "real_data_validation", + "production_validation", + "live_validation", + ): + allowed = ( + {"validated"} + if name == "contract_validation" + else ( + {"validated", "not_validated"} + if name == "real_data_validation" and snapshot.evidence_scope == "real_data" + else {"not_validated"} + ) + ) + status, evidence = _readiness_level(readiness[name], f"$.readiness.{name}", allowed) + _check( + not seen.intersection(evidence), + f"$.readiness.{name}", + "evidence reused across readiness levels", + ContractErrorCode.READINESS_ESCALATION, + ) + seen.update(evidence) + if name == "real_data_validation" and status == "validated": + _check( + all(coverage[key]["status"] == "validated" for key in selected_instruments), + f"$.readiness.{name}", + "real validation needs complete validated coverage", + ContractErrorCode.READINESS_ESCALATION, + ) + _identity(root, "foundation_id", "foundation", "rhdfv2", "$") + + +@dataclass(frozen=True, slots=True) +class RetrospectiveViewFacts: + view_ref_id: str + schema_digest: str + content_digest: str + transformation_digest: str + available_at: str + instrument_route_revision_ids: tuple[str, ...] + trading_calendar_revision_ids: tuple[str, ...] + corporate_action_revision_ids: tuple[str, ...] + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectiveFoundationEnvelope: + """Snapshot-bound declarations; not authentication or operational readiness.""" + + _payload: Mapping[str, Any] = field(repr=False) + + @classmethod + def from_dict(cls, value: Any, *, snapshot: RetrospectiveSnapshotEnvelope) -> Self: + _validate_foundation(value, snapshot) + instance = object.__new__(cls) + object.__setattr__(instance, "_payload", _freeze_json(value)) + return instance + + @classmethod + def from_json(cls, value: str | bytes, *, snapshot: RetrospectiveSnapshotEnvelope) -> Self: + return cls.from_dict(_parse_json_object(value, "$"), snapshot=snapshot) + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return canonical_json(self.to_dict()) + + @property + def foundation_id(self) -> str: + return cast(str, self._payload["foundation_id"]) + + @property + def dataset_snapshot_id(self) -> str: + return cast(str, self._payload["dataset_snapshot_id"]) + + @property + def observation_cutoff(self) -> str: + return cast(str, self._payload["observation_cutoff"]) + + @property + def published_at(self) -> datetime: + return _parse_utc(self._payload["published_at"], "$.published_at") + + @property + def evidence_scope(self) -> str: + return cast(str, self._payload["readiness"]["evidence_scope"]) + + @property + def real_data_validation_status(self) -> str: + return cast(str, self._payload["readiness"]["real_data_validation"]["status"]) + + @property + def contract_evidence_digests(self) -> tuple[str, ...]: + return cast( + tuple[str, ...], self._payload["readiness"]["contract_validation"]["evidence_digests"] + ) + + @property + def views(self) -> Mapping[str, RetrospectiveViewFacts]: + return MappingProxyType( + { + row["view_ref_id"]: RetrospectiveViewFacts( + **{ + key: row[key] + for key in ( + "view_ref_id", + "schema_digest", + "content_digest", + "transformation_digest", + "available_at", + "instrument_route_revision_ids", + "trading_calendar_revision_ids", + "corporate_action_revision_ids", + ) + } + ) + for row in self._payload["standardized_views"] + } + ) diff --git a/src/quant_engine/retrospective_factor_contracts.py b/src/quant_engine/retrospective_factor_contracts.py new file mode 100644 index 0000000..9e009bc --- /dev/null +++ b/src/quant_engine/retrospective_factor_contracts.py @@ -0,0 +1,721 @@ +"""Observation-aware factor results; no historical, governance or execution grant.""" + +from __future__ import annotations + +import json +import re +from collections.abc import Mapping, Sequence +from dataclasses import dataclass, field +from typing import Any, Self, cast + +from quant_engine.factor_contracts import ( + ActorIdentity, + ContractErrorCode, + FactorDefinition, + OutputArtifactRef, + OutputCoverage, + OutputQuality, + PayloadValidation, + ProducerIdentity, + _DEFINITION_ID, + _FIELD_NAME, + _array, + _assert_canonical_profile, + _canonical_evidence_bytes, + _content_address, + _digest, + _digest_bytes, + _freeze_json, + _git_revision, + _logical_id, + _parse_json_object, + _parse_utc, + _string, + _thaw_json, + canonical_json, + canonical_json_bytes, + validate_factor_catalog, +) +from quant_engine.retrospective_data_contracts import ( + RetrospectiveFoundationEnvelope, + RetrospectiveSnapshotEnvelope, + _IDS, + _check, + _choice, + _public, + _restrictions, + _shape, + _strings, +) + +_FACTOR_SET_ID = re.compile(r"^rhfactorsetv2:sha256:[0-9a-f]{64}$") + + +def _instant_text(value: Any, path: str) -> str: + _parse_utc(value, path) + return cast(str, value) + + +@dataclass(frozen=True, slots=True) +class RetrospectiveInputBinding: + definition_id: str + input_name: str + view_ref_id: str + schema_digest: str + + def __post_init__(self) -> None: + _string(self.definition_id, "$.input_bindings[].definition_id", _DEFINITION_ID) + _string(self.input_name, "$.input_bindings[].input_name", _FIELD_NAME) + _string(self.view_ref_id, "$.input_bindings[].view_ref_id", _IDS["view_ref"]) + _digest(self.schema_digest, "$.input_bindings[].schema_digest") + + def to_dict(self) -> dict[str, Any]: + return { + "definition_id": self.definition_id, + "input_name": self.input_name, + "view_ref_id": self.view_ref_id, + "schema_digest": self.schema_digest, + } + + @classmethod + def from_dict(cls, value: Any, path: str = "$.input_bindings[]") -> Self: + row = _shape(value, path, "definition_id input_name view_ref_id schema_digest") + return cls(**row) + + +@dataclass(frozen=True, slots=True) +class RetrospectiveViewAvailability: + view_ref_id: str + available_at: str + evidence_digest: str + + def __post_init__(self) -> None: + _string(self.view_ref_id, "$.view_availability[].view_ref_id", _IDS["view_ref"]) + _instant_text(self.available_at, "$.view_availability[].available_at") + _digest(self.evidence_digest, "$.view_availability[].evidence_digest") + + def to_dict(self) -> dict[str, Any]: + return { + "view_ref_id": self.view_ref_id, + "available_at": self.available_at, + "evidence_digest": self.evidence_digest, + } + + @classmethod + def from_dict(cls, value: Any, path: str = "$.view_availability[]") -> Self: + return cls(**_shape(value, path, "view_ref_id available_at evidence_digest")) + + +@dataclass(frozen=True, slots=True) +class RetrospectiveCausation: + kind: str + id: str + + def __post_init__(self) -> None: + kind = _choice(self.kind, "$.causation.kind", {"foundation", "factor_set"}) + _string( + self.id, + "$.causation.id", + _IDS["foundation"] if kind == "foundation" else _FACTOR_SET_ID, + ) + + def to_dict(self) -> dict[str, Any]: + return {"kind": self.kind, "id": self.id} + + @classmethod + def from_dict(cls, value: Any) -> Self: + return cls(**_shape(value, "$.causation", "kind id")) + + +@dataclass(frozen=True, slots=True) +class ResolvedRetrospectiveView: + """In-memory logical bytes; no locator, source authentication or transformation claim.""" + + view_ref_id: str + schema_bytes: bytes + content_bytes: bytes + + def __post_init__(self) -> None: + _string(self.view_ref_id, "$.resolved_views[].view_ref_id", _IDS["view_ref"]) + for key, value in ( + ("schema_bytes", self.schema_bytes), + ("content_bytes", self.content_bytes), + ): + _canonical_evidence_bytes(value, f"$.resolved_views[].{key}") + _public(json.loads(value), f"$.resolved_views[].{key}") + + +def _typed(values: Any, expected: type[Any], path: str) -> tuple[Any, ...]: + _check( + type(values) in {tuple, list}, + path, + "typed list/tuple required", + ContractErrorCode.TYPE_ERROR, + ) + _check( + all(type(value) is expected for value in values), + path, + f"{expected.__name__} required", + ContractErrorCode.TYPE_ERROR, + ) + return tuple(values) + + +def _context( + definitions: Sequence[FactorDefinition], + snapshot: Any, + foundation: Any, +) -> tuple[ + tuple[FactorDefinition, ...], RetrospectiveSnapshotEnvelope, RetrospectiveFoundationEnvelope +]: + _check( + type(snapshot) is RetrospectiveSnapshotEnvelope, + "$.dataset_snapshot", + "explicit v2 snapshot required", + ContractErrorCode.TYPE_ERROR, + ) + _check( + type(foundation) is RetrospectiveFoundationEnvelope, + "$.foundation", + "explicit v2 foundation required", + ContractErrorCode.TYPE_ERROR, + ) + snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot.to_dict()) + snapshot.require_qualified() + foundation = RetrospectiveFoundationEnvelope.from_dict(foundation.to_dict(), snapshot=snapshot) + supplied = _typed(definitions, FactorDefinition, "$.definitions") + # Definitions stay v1, but are parsed again so mutable/caller summaries are not authority. + normalized = validate_factor_catalog( + tuple(FactorDefinition.from_dict(item.to_dict()) for item in supplied) + ) + return normalized, snapshot, foundation + + +def _upstream( + snapshot: RetrospectiveSnapshotEnvelope, foundation: RetrospectiveFoundationEnvelope +) -> dict[str, Any]: + descriptor = snapshot.to_dict()["descriptor"] + return { + "dataset_snapshot_id": snapshot.snapshot_id, + "foundation_id": foundation.foundation_id, + "evidence_scope": snapshot.evidence_scope, + "content_digest": descriptor["content"]["content_digest"], + "manifest_digest": descriptor["content"]["manifest_digest"], + "observation_manifest_digest": _digest_bytes( + canonical_json_bytes(descriptor["observation_manifest"]) + ), + "time_semantics": descriptor["time_semantics"], + "quality": descriptor["quality"], + "qualification": descriptor["qualification"], + "foundation_readiness": foundation.to_dict()["readiness"], + } + + +def _input_payloads( + snapshot: RetrospectiveSnapshotEnvelope, + foundation: RetrospectiveFoundationEnvelope, + selected: tuple[str, ...], + dataset_chunks: Any, + resolved_views: Sequence[ResolvedRetrospectiveView] | None, +) -> PayloadValidation: + _check( + (dataset_chunks is None) == (resolved_views is None), + "$.input_payloads", + "snapshot chunks and resolved views must be supplied together", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + if dataset_chunks is None: + return PayloadValidation.REFERENCE_ONLY + snapshot.verify_materialized_records(dataset_chunks) + views = _typed(resolved_views, ResolvedRetrospectiveView, "$.resolved_views") + view_ids = [view.view_ref_id for view in views] + _check( + len(view_ids) == len(selected) and set(view_ids) == set(selected), + "$.resolved_views", + "resolved view closure mismatch", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + for item in views: + declared = foundation.views[item.view_ref_id] + # Recheck canonical bytes even for caller-constructed typed payloads. + schema = _canonical_evidence_bytes(item.schema_bytes, "$.resolved_views[].schema_bytes") + content = _canonical_evidence_bytes(item.content_bytes, "$.resolved_views[].content_bytes") + _public(json.loads(schema)) + _public(json.loads(content)) + _check( + _digest_bytes(schema) == declared.schema_digest + and _digest_bytes(content) == declared.content_digest, + "$.resolved_views", + "view bytes do not match Foundation", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + return PayloadValidation.PAYLOAD_REVALIDATED + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectiveFactorSetRef: + contract_name: str + schema_version: str + factor_set_id: str + definition_ids: tuple[str, ...] + dataset_snapshot_id: str + foundation_id: str + observation_cutoff: str + selected_view_ref_ids: tuple[str, ...] + input_bindings: tuple[RetrospectiveInputBinding, ...] + view_availability: tuple[RetrospectiveViewAvailability, ...] + upstream_evidence: Mapping[str, Any] + output_quality: OutputQuality + output_coverage: OutputCoverage + output_schema_digest: str + output_content_digest: str + output_artifact_ref: OutputArtifactRef + availability_mode: str + usage: str + historical_availability: str + evaluation_at: str + computed_at: str + artifact_available_at: str + producer: ProducerIdentity + code_revision: str + actor: ActorIdentity + correlation_id: str + causation: RetrospectiveCausation + evidence_scope: str + decision_eligible: bool + payload_validation: PayloadValidation = field(compare=False) + input_payload_validation: PayloadValidation = field(compare=False) + _payload: Mapping[str, Any] = field(repr=False, compare=False) + _definitions: tuple[FactorDefinition, ...] = field(repr=False, compare=False) + _dataset_snapshot: RetrospectiveSnapshotEnvelope = field(repr=False, compare=False) + _foundation: RetrospectiveFoundationEnvelope = field(repr=False, compare=False) + _parent: RetrospectiveFactorSetRef | None = field(repr=False, compare=False) + + @classmethod + def create( + cls, + *, + definitions: Sequence[FactorDefinition], + dataset_snapshot: RetrospectiveSnapshotEnvelope, + foundation: RetrospectiveFoundationEnvelope, + selected_view_ref_ids: Sequence[str], + input_bindings: Sequence[RetrospectiveInputBinding], + view_availability: Sequence[RetrospectiveViewAvailability], + dataset_chunks: Any, + resolved_views: Sequence[ResolvedRetrospectiveView], + output_quality: OutputQuality, + output_coverage: OutputCoverage, + output_schema_bytes: bytes, + output_content_bytes: bytes, + output_artifact_ref: OutputArtifactRef, + evaluation_at: str, + computed_at: str, + artifact_available_at: str, + producer: ProducerIdentity, + code_revision: str, + actor: ActorIdentity, + correlation_id: str, + causation: RetrospectiveCausation, + evidence_scope: str, + decision_eligible: bool, + parent: RetrospectiveFactorSetRef | None = None, + ) -> Self: + definitions, dataset_snapshot, foundation = _context( + definitions, dataset_snapshot, foundation + ) + _check( + type(selected_view_ref_ids) in {list, tuple}, + "$.selected_view_ref_ids", + "list/tuple required", + ContractErrorCode.TYPE_ERROR, + ) + selected = sorted( + _strings(list(selected_view_ref_ids), "$.selected_view_ref_ids", _IDS["view_ref"], 1) + ) + bindings = sorted( + _typed(input_bindings, RetrospectiveInputBinding, "$.input_bindings"), + key=lambda item: (item.definition_id, item.input_name), + ) + availability = sorted( + _typed(view_availability, RetrospectiveViewAvailability, "$.view_availability"), + key=lambda item: item.view_ref_id, + ) + for value, expected, path in ( + (output_quality, OutputQuality, "$.output_quality"), + (output_coverage, OutputCoverage, "$.output_coverage"), + (output_artifact_ref, OutputArtifactRef, "$.output_artifact_ref"), + (producer, ProducerIdentity, "$.producer"), + (actor, ActorIdentity, "$.actor"), + (causation, RetrospectiveCausation, "$.causation"), + ): + _check( + type(value) is expected, + path, + f"{expected.__name__} required", + ContractErrorCode.TYPE_ERROR, + ) + schema_digest = _digest_bytes( + _canonical_evidence_bytes(output_schema_bytes, "$.output_schema_bytes") + ) + content_digest = _digest_bytes( + _canonical_evidence_bytes(output_content_bytes, "$.output_content_bytes") + ) + document = { + "contract_name": "researchhub.factor-set-ref", + "schema_version": "2.0.0", + "definition_ids": [definition.definition_id for definition in definitions], + "dataset_snapshot_id": dataset_snapshot.snapshot_id, + "foundation_id": foundation.foundation_id, + "observation_cutoff": foundation.observation_cutoff, + "selected_view_ref_ids": selected, + "input_bindings": [item.to_dict() for item in bindings], + "view_availability": [item.to_dict() for item in availability], + "upstream_evidence": _upstream(dataset_snapshot, foundation), + "output_quality": output_quality.to_dict(), + "output_coverage": output_coverage.to_dict(), + "output_schema_digest": schema_digest, + "output_content_digest": content_digest, + "output_artifact_ref": output_artifact_ref.to_dict(), + "availability_mode": "retrospective_replay", + "usage": "retrospective_research", + "historical_availability": "not_established", + "evaluation_at": evaluation_at, + "computed_at": computed_at, + "artifact_available_at": artifact_available_at, + "producer": producer.to_dict(), + "code_revision": code_revision, + "actor": actor.to_dict(), + "correlation_id": correlation_id, + "causation": causation.to_dict(), + "evidence_scope": evidence_scope, + "decision_eligible": decision_eligible, + } + document["factor_set_id"] = _content_address( + document, "factor_set_id", "rhfactorsetv2:sha256:" + ) + result = cls.from_dict( + document, + definitions=definitions, + dataset_snapshot=dataset_snapshot, + foundation=foundation, + parent=parent, + output_schema_bytes=output_schema_bytes, + output_content_bytes=output_content_bytes, + dataset_chunks=dataset_chunks, + resolved_views=resolved_views, + ) + result.require_payloads_revalidated() + return result + + @classmethod + def from_dict( + cls, + value: Any, + *, + definitions: Sequence[FactorDefinition], + dataset_snapshot: RetrospectiveSnapshotEnvelope, + foundation: RetrospectiveFoundationEnvelope, + parent: RetrospectiveFactorSetRef | None = None, + output_schema_bytes: bytes | None = None, + output_content_bytes: bytes | None = None, + dataset_chunks: Any = None, + resolved_views: Sequence[ResolvedRetrospectiveView] | None = None, + ) -> Self: + definitions, dataset_snapshot, foundation = _context( + definitions, dataset_snapshot, foundation + ) + _assert_canonical_profile(value) + _public(value) + row = _shape( + value, + "$", + "contract_name schema_version factor_set_id definition_ids dataset_snapshot_id foundation_id observation_cutoff " + "selected_view_ref_ids input_bindings view_availability upstream_evidence output_quality output_coverage " + "output_schema_digest output_content_digest output_artifact_ref availability_mode usage historical_availability " + "evaluation_at computed_at artifact_available_at producer code_revision actor correlation_id causation evidence_scope decision_eligible", + ) + _choice(row["contract_name"], "$.contract_name", {"researchhub.factor-set-ref"}) + _choice(row["schema_version"], "$.schema_version", {"2.0.0"}) + _choice(row["availability_mode"], "$.availability_mode", {"retrospective_replay"}) + _restrictions(row, "$") + _check( + type(row["decision_eligible"]) is bool and not row["decision_eligible"], + "$.decision_eligible", + "computation is never decision eligible", + ContractErrorCode.READINESS_ESCALATION, + ) + _check( + row["dataset_snapshot_id"] == dataset_snapshot.snapshot_id + and row["foundation_id"] == foundation.foundation_id + and row["observation_cutoff"] == foundation.observation_cutoff, + "$.foundation_id", + "exact snapshot/foundation/cutoff required", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + definition_ids = _strings(row["definition_ids"], "$.definition_ids", _DEFINITION_ID, 1) + _check( + definition_ids == tuple(item.definition_id for item in definitions), + "$.definition_ids", + "normalized exact definitions required", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + selected = _strings( + row["selected_view_ref_ids"], "$.selected_view_ref_ids", _IDS["view_ref"], 1 + ) + _check( + tuple(sorted(selected)) == selected and set(selected) <= foundation.views.keys(), + "$.selected_view_ref_ids", + "unknown/unnormalized selected views", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + bindings = tuple( + RetrospectiveInputBinding.from_dict(item) + for item in _array(row["input_bindings"], "$.input_bindings", minimum=1, unique=True) + ) + keys = [(item.definition_id, item.input_name) for item in bindings] + expected = { + (item.definition_id, input_spec.input_name): input_spec + for item in definitions + for input_spec in item.inputs + } + _check( + len(keys) == len(expected) and set(keys) == expected.keys() and keys == sorted(keys), + "$.input_bindings", + "exact normalized factor input closure required", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + for binding in bindings: + _check( + binding.view_ref_id in selected, + "$.input_bindings", + "unselected view", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + _check( + binding.schema_digest + == expected[(binding.definition_id, binding.input_name)].schema_digest + == foundation.views[binding.view_ref_id].schema_digest, + "$.input_bindings", + "schema mismatch", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + _check( + {item.view_ref_id for item in bindings} == set(selected), + "$.selected_view_ref_ids", + "unused selected view", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + availability = tuple( + RetrospectiveViewAvailability.from_dict(item) + for item in _array( + row["view_availability"], "$.view_availability", minimum=1, unique=True + ) + ) + _check( + tuple(item.view_ref_id for item in availability) == selected, + "$.view_availability", + "exact normalized selected view availability required", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + for item in availability: + _check( + item.available_at == foundation.views[item.view_ref_id].available_at, + "$.view_availability", + "availability must equal its Foundation fact", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + upstream = _upstream(dataset_snapshot, foundation) + _check( + canonical_json_bytes(row["upstream_evidence"]) == canonical_json_bytes(upstream), + "$.upstream_evidence", + "upstream evidence differs from complete input envelopes", + ContractErrorCode.IDENTITY_MISMATCH, + ) + _check( + row["evidence_scope"] == dataset_snapshot.evidence_scope == foundation.evidence_scope, + "$.evidence_scope", + "scope must equal both inputs", + ContractErrorCode.READINESS_ESCALATION, + ) + if row["evidence_scope"] == "real_data": + _check( + foundation.real_data_validation_status == "validated", + "$.evidence_scope", + "real-data Foundation validation required", + ContractErrorCode.READINESS_ESCALATION, + ) + quality = OutputQuality.from_dict(row["output_quality"]) + coverage = OutputCoverage.from_dict(row["output_coverage"]) + _check( + quality.status == "passed" and all(item.status == "passed" for item in quality.checks), + "$.output_quality", + "all output checks must pass", + ) + _check( + coverage.status == "complete" and coverage.observed_count == coverage.expected_count, + "$.output_coverage", + "complete output coverage required", + ) + artifact = OutputArtifactRef.from_dict(row["output_artifact_ref"]) + schema_digest = _digest(row["output_schema_digest"], "$.output_schema_digest") + content_digest = _digest(row["output_content_digest"], "$.output_content_digest") + _check( + artifact.schema_digest == schema_digest and artifact.content_digest == content_digest, + "$.output_artifact_ref", + "output artifact mismatch", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + _check( + (output_schema_bytes is None) == (output_content_bytes is None), + "$.output_artifact_ref", + "both output payloads required together", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + validation = PayloadValidation.REFERENCE_ONLY + if output_schema_bytes is not None and output_content_bytes is not None: + for data, expected_digest, path in ( + (output_schema_bytes, schema_digest, "$.output_schema_bytes"), + (output_content_bytes, content_digest, "$.output_content_bytes"), + ): + canonical = _canonical_evidence_bytes(data, path) + _public(json.loads(canonical), path) + _check( + _digest_bytes(canonical) == expected_digest, + path, + "output bytes mismatch", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + validation = PayloadValidation.PAYLOAD_REVALIDATED + input_validation = _input_payloads( + dataset_snapshot, foundation, selected, dataset_chunks, resolved_views + ) + evaluation = _parse_utc(row["evaluation_at"], "$.evaluation_at") + computed = _parse_utc(row["computed_at"], "$.computed_at") + available = _parse_utc(row["artifact_available_at"], "$.artifact_available_at") + _check( + foundation.published_at <= evaluation <= computed <= available, + "$.computed_at", + "input publication <= actual evaluation <= computation <= artifact required", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + for definition in definitions: + _check( + _parse_utc(definition.valid_from, "$.definitions[].valid_from") + <= evaluation + < _parse_utc(definition.valid_until, "$.definitions[].valid_until"), + "$.definitions", + "factor definition is not valid at actual evaluation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + producer = ProducerIdentity.from_dict(row["producer"]) + _check( + producer.id == "quant_engine", + "$.producer.id", + "computation owner must be quant_engine", + ContractErrorCode.LINEAGE_VIOLATION, + ) + _git_revision(row["code_revision"], "$.code_revision") + actor = ActorIdentity.from_dict(row["actor"]) + correlation = _logical_id(row["correlation_id"], "$.correlation_id") + cause = RetrospectiveCausation.from_dict(row["causation"]) + for name, parsed in ( + ("output_quality", quality), + ("output_coverage", coverage), + ("output_artifact_ref", artifact), + ("producer", producer), + ("actor", actor), + ("causation", cause), + ): + _check( + canonical_json_bytes(row[name]) == canonical_json_bytes(parsed.to_dict()), + f"$.{name}", + "nested contract is not normalized", + ContractErrorCode.INVALID_FORMAT, + ) + if cause.kind == "foundation": + _check( + cause.id == foundation.foundation_id and parent is None, + "$.causation", + "exact Foundation cause required", + ContractErrorCode.LINEAGE_VIOLATION, + ) + else: + _check( + type(parent) is RetrospectiveFactorSetRef, + "$.causation", + "exact v2 parent object required", + ContractErrorCode.LINEAGE_VIOLATION, + ) + assert parent is not None + _check( + cause.id == parent.factor_set_id + and correlation == parent.correlation_id + and row["evidence_scope"] == parent.evidence_scope, + "$.causation", + "parent identity/correlation/scope mismatch", + ContractErrorCode.LINEAGE_VIOLATION, + ) + _check( + _parse_utc(parent.artifact_available_at, "$.parent.artifact_available_at") + <= evaluation, + "$.causation", + "parent artifact postdates child evaluation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + factor_set_id = _string(row["factor_set_id"], "$.factor_set_id", _FACTOR_SET_ID) + _check( + factor_set_id == _content_address(row, "factor_set_id", "rhfactorsetv2:sha256:"), + "$.factor_set_id", + "factor result identity mismatch", + ContractErrorCode.IDENTITY_MISMATCH, + ) + _check( + cause.id != factor_set_id, + "$.causation", + "self parent is forbidden", + ContractErrorCode.LINEAGE_VIOLATION, + ) + instance = object.__new__(cls) + values = { + **row, + "definition_ids": definition_ids, + "selected_view_ref_ids": selected, + "input_bindings": bindings, + "view_availability": availability, + "upstream_evidence": _freeze_json(upstream), + "output_quality": quality, + "output_coverage": coverage, + "output_artifact_ref": artifact, + "producer": producer, + "actor": actor, + "causation": cause, + "payload_validation": validation, + "input_payload_validation": input_validation, + "_payload": _freeze_json(row), + "_definitions": definitions, + "_dataset_snapshot": dataset_snapshot, + "_foundation": foundation, + "_parent": parent, + } + for name, item in values.items(): + object.__setattr__(instance, name, item) + return instance + + def require_payloads_revalidated(self) -> None: + _check( + self.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED + and self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED, + "$.payload_validation", + "reference-only data is not computation admission", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return canonical_json(self.to_dict()) + + @classmethod + def from_json(cls, value: str | bytes, **kwargs: Any) -> Self: + return cls.from_dict(_parse_json_object(value, "$"), **kwargs) diff --git a/src/quant_engine/retrospective_portfolio_risk_contracts.py b/src/quant_engine/retrospective_portfolio_risk_contracts.py new file mode 100644 index 0000000..35eff01 --- /dev/null +++ b/src/quant_engine/retrospective_portfolio_risk_contracts.py @@ -0,0 +1,995 @@ +"""Retrospective-only portfolio/risk evidence with separate business/actual clocks.""" + +from __future__ import annotations + +import json +import math +import re +from collections.abc import Mapping +from dataclasses import dataclass, field +from types import MappingProxyType +from typing import Any, Self, TypedDict, cast + +import pandas as pd + +from quant_engine.artifact import ( + EvidenceQualification, + _performance_compare, + _performance_validate_tree, +) +from quant_engine.factor_contracts import ( + ContractErrorCode, + FactorContractError, + _duplicate_key_pairs, + _parse_utc, + _string, + _thaw_json, +) +from quant_engine.portfolio_risk_contracts import ( + ComputationReceipt, + ConstraintSetV1, + FreshnessPolicy, + ReceiptStatus, + PortfolioRiskContractError, + PortfolioRiskContractErrorCode, + RiskAssessmentStatus, + RiskFindingCode, + _CLOSURE_ATOL, + _CLOSURE_RTOL, + _finite_number, + _series_mapping, + _validate_covariance_structure, + _canonical_json, + _constraint_metrics, + _constraint_residuals, + _digest, + _document_sha256, + _immutable_float_mapping, + _mapping_dict, + _payload_digest, + _semver, + _text, +) +from quant_engine.retrospective_artifact_contracts import ( + RetrospectiveBacktestEvidenceManifest, + _freeze_numeric_evidence, + _validated_run, +) +from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef +from quant_engine.retrospective_data_contracts import _IDS, _check, _public, _shape +from quant_engine.risk import CovarianceSnapshot, labeled_component_risk + +_RUN_ID = re.compile(r"^rhbacktestrunv2:sha256:[0-9a-f]{64}$") + + +def _json_object(value: str | bytes) -> dict[str, Any]: + _check( + type(value) in {str, bytes}, + "$", + "canonical JSON text/bytes required", + ContractErrorCode.TYPE_ERROR, + ) + try: + raw = value.encode("utf-8") if isinstance(value, str) else value + document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs) + except (json.JSONDecodeError, UnicodeError) as error: + raise FactorContractError( + ContractErrorCode.INVALID_FORMAT, "$", "valid UTF-8 JSON required" + ) from error + _check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR) + _performance_validate_tree(document, "$") + _check( + _canonical_json(document).encode() == raw, + "$", + "canonical numeric JSON required", + ContractErrorCode.INVALID_FORMAT, + ) + return cast(dict[str, Any], document) + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectivePortfolioTarget: + contract_name: str + schema_version: str + target_id: str + backtest_run_id: str + dataset_snapshot_id: str + weights: Mapping[str, float] + effective_at: str + created_at: str + usage: str + historical_availability: str + _payload: Mapping[str, Any] = field(repr=False, compare=False) + + @classmethod + def create( + cls, + *, + backtest_run_id: str, + dataset_snapshot_id: str, + weights: Mapping[str, float], + effective_at: str, + created_at: str, + ) -> Self: + _string(backtest_run_id, "$.backtest_run_id", _RUN_ID) + _string(dataset_snapshot_id, "$.dataset_snapshot_id", _IDS["snapshot"]) + normalized = _immutable_float_mapping(weights, "$.weights") + _check(bool(normalized), "$.weights", "non-empty target asset set required") + for instrument in normalized: + _string(instrument, "$.weights.keys", _IDS["instrument"]) + effective = _parse_utc(effective_at, "$.effective_at") + created = _parse_utc(created_at, "$.created_at") + _check( + effective <= created, + "$.effective_at", + "historical effective time exceeds actual creation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + payload = { + "contract_name": "researchhub.portfolio-target", + "schema_version": "2.0.0", + "backtest_run_id": backtest_run_id, + "dataset_snapshot_id": dataset_snapshot_id, + "weights": _mapping_dict(normalized), + "effective_at": effective_at, + "created_at": created_at, + "usage": "retrospective_research", + "historical_availability": "not_established", + } + _public(payload) + payload["target_id"] = "rhportfoliotargetv2:" + _payload_digest(payload) + instance = object.__new__(cls) + for key, value in { + **payload, + "weights": normalized, + "_payload": _freeze_numeric_evidence(payload), + }.items(): + object.__setattr__(instance, key, value) + return instance + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return _canonical_json(self.to_dict()) + + @classmethod + def from_dict(cls, value: Any) -> Self: + _performance_validate_tree(value, "$") + row = _shape( + value, + "$", + "contract_name schema_version target_id backtest_run_id dataset_snapshot_id weights effective_at created_at usage historical_availability", + ) + rebuilt = cls.create( + **{ + key: row[key] + for key in ( + "backtest_run_id", + "dataset_snapshot_id", + "weights", + "effective_at", + "created_at", + ) + } + ) + _performance_compare(row, rebuilt.to_dict(), "$") + return rebuilt + + @classmethod + def from_json(cls, value: str | bytes) -> Self: + return cls.from_dict(_json_object(value)) + + +@dataclass(frozen=True, slots=True) +class _PortfolioInputs: + run: RetrospectiveBacktestRunRef + manifest: RetrospectiveBacktestEvidenceManifest + target: RetrospectivePortfolioTarget + constraints: ConstraintSetV1 + freshness: FreshnessPolicy + weights: Mapping[str, float] + prior: Mapping[str, float] | None + metrics: dict[str, float | int | None] + residuals: dict[str, float] + input_payload: dict[str, object] + + +def _material( + *, + backtest_run_ref: RetrospectiveBacktestRunRef, + manifest: RetrospectiveBacktestEvidenceManifest, + target: RetrospectivePortfolioTarget, + objective_name: str, + objective_version: str, + objective_digest: str, + model_name: str, + model_version: str, + model_digest: str, + expected_return_digest: str, + covariance_digest: str, + scenario_digest: str, + constraints: ConstraintSetV1, + freshness_policy: FreshnessPolicy, + prior_weights: Mapping[str, float] | None = None, +) -> _PortfolioInputs: + run = _validated_run(backtest_run_ref) + for item, expected, path in ( + (manifest, RetrospectiveBacktestEvidenceManifest, "$.manifest"), + (target, RetrospectivePortfolioTarget, "$.target"), + (constraints, ConstraintSetV1, "$.constraints"), + (freshness_policy, FreshnessPolicy, "$.freshness_policy"), + ): + _check( + type(item) is expected, + path, + f"explicit {expected.__name__} required", + ContractErrorCode.TYPE_ERROR, + ) + checked_manifest = RetrospectiveBacktestEvidenceManifest.from_dict( + manifest.to_dict(), artifact=manifest._artifact, backtest_run_ref=run + ) + checked_target = RetrospectivePortfolioTarget.from_dict(target.to_dict()) + _check( + checked_manifest.qualification is EvidenceQualification.CONTRACT_QUALIFIED, + "$.manifest.qualification", + "contract-qualified retrospective S3 required", + ContractErrorCode.QUALIFICATION_REJECTED, + ) + _check( + checked_target.backtest_run_id == run.run_id + and checked_target.dataset_snapshot_id == run.dataset_snapshot_id, + "$.target", + "target and S3 identities differ", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + foundation = run._factor_set._foundation + selected_routes = { + identity + for view_id in run._factor_set.selected_view_ref_ids + for identity in foundation.views[view_id].instrument_route_revision_ids + } + selected_instruments = { + row["instrument_id"] + for row in foundation.to_dict()["instrument_routes"] + if row["route_revision_id"] in selected_routes + } + _check( + set(checked_target.weights) <= selected_instruments, + "$.target.weights", + "target assets must be selected logical instruments", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + constraints = ConstraintSetV1.from_dict(constraints.to_dict()) + freshness_policy = FreshnessPolicy.from_dict(freshness_policy.to_dict()) + prior = ( + None + if prior_weights is None + else _immutable_float_mapping(prior_weights, "$.prior_weights") + ) + if prior is not None: + _check( + set(prior) <= selected_instruments, + "$.prior_weights", + "prior assets outside selected instruments", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + weights = checked_target.weights + metrics = _constraint_metrics(weights, prior) + residuals = _constraint_residuals(constraints, weights, metrics) + payload: dict[str, object] = { + "contract_name": "researchhub.portfolio-computation-input", + "schema_version": "2.0.0", + "usage": run.usage, + "historical_availability": run.historical_availability, + "evidence_scope": run.evidence_scope, + "run_ref_document_sha256": _document_sha256(run.to_json()), + "manifest_document_sha256": _document_sha256(checked_manifest.to_json()), + "portfolio_target": checked_target.to_dict(), + "objective": { + "name": _text(objective_name, "$.objective_name"), + "version": _semver(objective_version, "$.objective_version"), + "digest": _digest(objective_digest, "$.objective_digest"), + }, + "model": { + "name": _text(model_name, "$.model_name"), + "version": _semver(model_version, "$.model_version"), + "digest": _digest(model_digest, "$.model_digest"), + }, + "expected_return_digest": _digest(expected_return_digest, "$.expected_return_digest"), + "covariance_digest": _digest(covariance_digest, "$.covariance_digest"), + "scenario_digest": _digest(scenario_digest, "$.scenario_digest"), + "freshness_policy_digest": _payload_digest(freshness_policy.to_dict()), + "prior_weights": None if prior is None else _mapping_dict(prior), + } + _public(payload) + return _PortfolioInputs( + run, + checked_manifest, + checked_target, + constraints, + freshness_policy, + weights, + prior, + metrics, + residuals, + payload, + ) + + +def _receipt_digests(inputs: _PortfolioInputs) -> dict[str, str | float]: + return { + "input_digest": _payload_digest(inputs.input_payload), + "constraint_digest": _payload_digest(inputs.constraints.to_dict()), + "output_digest": _payload_digest( + { + "weights": _mapping_dict(inputs.weights), + "metrics": inputs.metrics, + "constraint_residuals": inputs.residuals, + } + ), + "max_constraint_residual": max(inputs.residuals.values(), default=0.0), + } + + +def compute_retrospective_portfolio_receipt_digests(**kwargs: Any) -> Mapping[str, str | float]: + """Recompute receipt claims; the returned digests are not producer authentication.""" + return MappingProxyType(_receipt_digests(_material(**kwargs))) + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectivePortfolioDecision: + contract_name: str + schema_version: str + decision_id: str + run_id: str + manifest_id: str + evidence_digest: str + dataset_snapshot_id: str + run_ref_document_sha256: str + manifest_document_sha256: str + source_universe_digest: str + portfolio_asset_set_digest: str + target_id: str + target_weights: Mapping[str, float] + prior_weights: Mapping[str, float] | None + objective_name: str + objective_version: str + objective_digest: str + model_name: str + model_version: str + model_digest: str + expected_return_digest: str + covariance_digest: str + scenario_digest: str + constraints: ConstraintSetV1 + freshness_policy: FreshnessPolicy + receipt: ComputationReceipt + gross_exposure: float + net_exposure: float + turnover_l1: float | None + position_count: int + constraint_residuals: Mapping[str, float] + output_digest: str + effective_at: str + created_at: str + computed_at: str + observation_cutoff: str + evidence_scope: str + usage: str + historical_availability: str + decision_eligible: bool + execution_validation: str + _payload: Mapping[str, Any] = field(repr=False, compare=False) + _target: RetrospectivePortfolioTarget = field(repr=False, compare=False) + _run: RetrospectiveBacktestRunRef = field(repr=False, compare=False) + _manifest: RetrospectiveBacktestEvidenceManifest = field(repr=False, compare=False) + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return _canonical_json(self.to_dict()) + + @classmethod + def from_dict( + cls, + value: Any, + *, + backtest_run_ref: RetrospectiveBacktestRunRef, + manifest: RetrospectiveBacktestEvidenceManifest, + target: RetrospectivePortfolioTarget, + ) -> Self: + _performance_validate_tree(value, "$") + # Rebuild from independent typed inputs, not from a self-approved target in the wire. + row = _shape( + value, + "$", + " ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")), + ) + arguments = { + key: row[key] + for key in ( + "objective_name", + "objective_version", + "objective_digest", + "model_name", + "model_version", + "model_digest", + "expected_return_digest", + "covariance_digest", + "scenario_digest", + "computed_at", + "prior_weights", + ) + } + rebuilt = build_retrospective_portfolio_decision( + **arguments, + backtest_run_ref=backtest_run_ref, + manifest=manifest, + target=target, + constraints=ConstraintSetV1.from_dict(row["constraints"]), + freshness_policy=FreshnessPolicy.from_dict(row["freshness_policy"]), + receipt=ComputationReceipt.from_dict(row["receipt"]), + ) + _performance_compare(row, rebuilt.to_dict(), "$") + return cast(Self, rebuilt) + + @classmethod + def from_json(cls, value: str | bytes, **kwargs: Any) -> Self: + return cls.from_dict(_json_object(value), **kwargs) + + +def build_retrospective_portfolio_decision( + *, + backtest_run_ref: RetrospectiveBacktestRunRef, + manifest: RetrospectiveBacktestEvidenceManifest, + target: RetrospectivePortfolioTarget, + objective_name: str, + objective_version: str, + objective_digest: str, + model_name: str, + model_version: str, + model_digest: str, + expected_return_digest: str, + covariance_digest: str, + scenario_digest: str, + constraints: ConstraintSetV1, + freshness_policy: FreshnessPolicy, + receipt: ComputationReceipt, + computed_at: str, + prior_weights: Mapping[str, float] | None = None, +) -> RetrospectivePortfolioDecision: + """Verify the existing constraints and receipt, with two explicitly different clocks.""" + _check( + type(receipt) is ComputationReceipt, + "$.receipt", + "typed computation receipt required", + ContractErrorCode.TYPE_ERROR, + ) + receipt = ComputationReceipt.from_dict(receipt.to_dict()) + inputs = _material( + backtest_run_ref=backtest_run_ref, + manifest=manifest, + target=target, + objective_name=objective_name, + objective_version=objective_version, + objective_digest=objective_digest, + model_name=model_name, + model_version=model_version, + model_digest=model_digest, + expected_return_digest=expected_return_digest, + covariance_digest=covariance_digest, + scenario_digest=scenario_digest, + constraints=constraints, + freshness_policy=freshness_policy, + prior_weights=prior_weights, + ) + run, manifest, target = inputs.run, inputs.manifest, inputs.target + computed = _parse_utc(computed_at, "$.computed_at") + created = _parse_utc(target.created_at, "$.target.created_at") + available = _parse_utc(manifest.artifact_available_at, "$.manifest.artifact_available_at") + _check( + available <= created <= computed, + "$.target.created_at", + "artifact availability <= actual target creation <= computation required", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + _check( + _parse_utc(receipt.computed_at, "$.receipt.computed_at") == computed, + "$.receipt.computed_at", + "receipt actual time differs from computation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + _check( + (computed - available).total_seconds() <= inputs.freshness.max_manifest_age_seconds, + "$.manifest.artifact_available_at", + "manifest is stale at actual computation", + ) + _check( + receipt.status not in {ReceiptStatus.FAILED, ReceiptStatus.FALLBACK}, + "$.receipt.status", + "failed/fallback computation cannot form a result", + ContractErrorCode.QUALIFICATION_REJECTED, + ) + digests = _receipt_digests(inputs) + for key, expected in digests.items(): + _check( + getattr(receipt, key) == expected, + f"$.receipt.{key}", + "receipt differs from independently recomputed evidence", + ContractErrorCode.ARTIFACT_MISMATCH, + ) + _check( + digests["max_constraint_residual"] == 0.0, + "$.constraints", + "target violates supported constraints", + ) + payload = { + "contract_name": "researchhub.portfolio-decision", + "schema_version": "2.0.0", + "run_id": run.run_id, + "manifest_id": manifest.manifest_id, + "evidence_digest": manifest.evidence_digest, + "dataset_snapshot_id": run.dataset_snapshot_id, + "run_ref_document_sha256": _document_sha256(run.to_json()), + "manifest_document_sha256": _document_sha256(manifest.to_json()), + "source_universe_digest": run.universe_digest, + "portfolio_asset_set_digest": _payload_digest(sorted(inputs.weights)), + "target_id": target.target_id, + "target_weights": _mapping_dict(inputs.weights), + "prior_weights": None if inputs.prior is None else _mapping_dict(inputs.prior), + "objective_name": objective_name, + "objective_version": objective_version, + "objective_digest": objective_digest, + "model_name": model_name, + "model_version": model_version, + "model_digest": model_digest, + "expected_return_digest": expected_return_digest, + "covariance_digest": covariance_digest, + "scenario_digest": scenario_digest, + "constraints": inputs.constraints.to_dict(), + "freshness_policy": inputs.freshness.to_dict(), + "receipt": receipt.to_dict(), + **inputs.metrics, + "constraint_residuals": inputs.residuals, + "output_digest": digests["output_digest"], + "effective_at": target.effective_at, + "created_at": target.created_at, + "computed_at": computed_at, + "observation_cutoff": run.observation_cutoff, + "evidence_scope": run.evidence_scope, + "usage": run.usage, + "historical_availability": run.historical_availability, + "decision_eligible": False, + "execution_validation": "not_validated", + } + _public(payload) + payload["decision_id"] = "rhportfoliodecisionv2:" + _payload_digest(payload) + instance = object.__new__(RetrospectivePortfolioDecision) + values = { + **payload, + "target_weights": inputs.weights, + "prior_weights": inputs.prior, + "constraints": inputs.constraints, + "freshness_policy": inputs.freshness, + "receipt": receipt, + "constraint_residuals": MappingProxyType(inputs.residuals), + "_payload": _freeze_numeric_evidence(payload), + "_target": target, + "_run": run, + "_manifest": manifest, + } + for key, value in values.items(): + object.__setattr__(instance, key, value) + return instance + + +@dataclass(frozen=True, slots=True, init=False) +class RetrospectiveRiskAssessment: + contract_name: str + schema_version: str + assessment_id: str + decision_id: str + run_id: str + manifest_id: str + dataset_snapshot_id: str + covariance_data_snapshot_id: str + covariance_snapshot_id: str + covariance_as_of_date: str + covariance_method: str + covariance_window_start_date: str + covariance_window_end_date: str + covariance_observations: int | None + covariance_lookback_sessions: int | None + covariance_missing_policy: str + covariance_input_digest: str + covariance_matrix_digest: str + return_frequency: str + periods_per_year: int + risk_model_name: str + risk_model_version: str + risk_model_digest: str + freshness_policy_digest: str + scenario_digest: str + portfolio_volatility_limit: float | None + risk_budget: Mapping[str, float] + groups: Mapping[str, str] | None + marginal_risk: Mapping[str, float] + component_risk: Mapping[str, float] + percentage_risk: Mapping[str, float] + portfolio_volatility: float | None + group_exposure: Mapping[str, float] + findings: tuple[RiskFindingCode, ...] + status: RiskAssessmentStatus + qualified: bool + effective_at: str + computed_at: str + observation_cutoff: str + evidence_scope: str + usage: str + historical_availability: str + decision_eligible: bool + execution_validation: str + _payload: Mapping[str, Any] = field(repr=False, compare=False) + + def to_dict(self) -> dict[str, Any]: + return cast(dict[str, Any], _thaw_json(self._payload)) + + def to_json(self) -> str: + return _canonical_json(self.to_dict()) + + @classmethod + def from_dict( + cls, + value: Any, + *, + portfolio_decision: RetrospectivePortfolioDecision, + backtest_run_ref: RetrospectiveBacktestRunRef, + manifest: RetrospectiveBacktestEvidenceManifest, + covariance: CovarianceSnapshot, + ) -> Self: + _performance_validate_tree(value, "$") + row = _shape( + value, + "$", + " ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")), + ) + rebuilt = assess_retrospective_portfolio_risk( + portfolio_decision=portfolio_decision, + backtest_run_ref=backtest_run_ref, + manifest=manifest, + covariance=covariance, + **{ + key: row[key] + for key in ( + "risk_model_name", + "risk_model_version", + "risk_model_digest", + "risk_budget", + "portfolio_volatility_limit", + "groups", + "computed_at", + ) + }, + ) + _performance_compare(row, rebuilt.to_dict(), "$") + return cast(Self, rebuilt) + + @classmethod + def from_json(cls, value: str | bytes, **kwargs: Any) -> Self: + return cls.from_dict(_json_object(value), **kwargs) + + +class _RiskContext(TypedDict): + decision: RetrospectivePortfolioDecision + covariance: CovarianceSnapshot + matrix_digest: str + risk_model_name: str + risk_model_version: str + risk_model_digest: str + portfolio_volatility_limit: float | None + risk_budget: Mapping[str, float] + groups: Mapping[str, str] | None + computed_at: str + + +def _risk_result( + *, + decision: RetrospectivePortfolioDecision, + covariance: CovarianceSnapshot, + matrix_digest: str, + risk_model_name: str, + risk_model_version: str, + risk_model_digest: str, + portfolio_volatility_limit: float | None, + risk_budget: Mapping[str, float], + groups: Mapping[str, str] | None, + marginal: Mapping[str, float], + component: Mapping[str, float], + percentage: Mapping[str, float], + volatility: float | None, + grouped: Mapping[str, float], + findings: tuple[RiskFindingCode, ...], + status: RiskAssessmentStatus, + qualified: bool, + computed_at: str, +) -> RetrospectiveRiskAssessment: + assert covariance.window_start_date is not None + assert covariance.window_end_date is not None + payload = { + "contract_name": "researchhub.risk-assessment", + "schema_version": "2.0.0", + "decision_id": decision.decision_id, + "run_id": decision.run_id, + "manifest_id": decision.manifest_id, + "dataset_snapshot_id": decision.dataset_snapshot_id, + "covariance_data_snapshot_id": covariance.data_snapshot_id, + "covariance_snapshot_id": covariance.snapshot_id, + "covariance_as_of_date": covariance.as_of_date.isoformat(), + "covariance_method": covariance.method, + "covariance_window_start_date": covariance.window_start_date.isoformat(), + "covariance_window_end_date": covariance.window_end_date.isoformat(), + "covariance_observations": covariance.observations, + "covariance_lookback_sessions": covariance.lookback_sessions, + "covariance_missing_policy": covariance.missing_policy, + "covariance_input_digest": "sha256:" + covariance.input_sha256, + "covariance_matrix_digest": matrix_digest, + "return_frequency": covariance.return_frequency, + "periods_per_year": covariance.periods_per_year, + "risk_model_name": risk_model_name, + "risk_model_version": risk_model_version, + "risk_model_digest": risk_model_digest, + "freshness_policy_digest": _payload_digest(decision.freshness_policy.to_dict()), + "scenario_digest": decision.scenario_digest, + "portfolio_volatility_limit": portfolio_volatility_limit, + "risk_budget": _mapping_dict(risk_budget), + "groups": None if groups is None else dict(groups), + "marginal_risk": _mapping_dict(marginal), + "component_risk": _mapping_dict(component), + "percentage_risk": _mapping_dict(percentage), + "portfolio_volatility": volatility, + "group_exposure": _mapping_dict(grouped), + "findings": [finding.value for finding in findings], + "status": status.value, + "qualified": qualified, + "effective_at": decision.effective_at, + "computed_at": computed_at, + "observation_cutoff": decision.observation_cutoff, + "evidence_scope": decision.evidence_scope, + "usage": decision.usage, + "historical_availability": decision.historical_availability, + "decision_eligible": False, + "execution_validation": "not_validated", + } + _public(payload) + payload["assessment_id"] = "rhriskassessmentv2:" + _payload_digest(payload) + instance = object.__new__(RetrospectiveRiskAssessment) + values = { + **payload, + "risk_budget": risk_budget, + "groups": groups, + "marginal_risk": marginal, + "component_risk": component, + "percentage_risk": percentage, + "group_exposure": grouped, + "findings": findings, + "status": status, + "_payload": _freeze_numeric_evidence(payload), + } + for key, value in values.items(): + object.__setattr__(instance, key, value) + return instance + + +def assess_retrospective_portfolio_risk( + *, + portfolio_decision: RetrospectivePortfolioDecision, + backtest_run_ref: RetrospectiveBacktestRunRef, + manifest: RetrospectiveBacktestEvidenceManifest, + covariance: CovarianceSnapshot, + risk_model_name: str, + risk_model_version: str, + risk_model_digest: str, + computed_at: str, + risk_budget: Mapping[str, float] | None = None, + portfolio_volatility_limit: float | None = None, + groups: Mapping[str, str] | None = None, +) -> RetrospectiveRiskAssessment: + """Use the existing Euler decomposition once; distinguish the two freshness clocks.""" + _check( + type(portfolio_decision) is RetrospectivePortfolioDecision, + "$.portfolio_decision", + "explicit v2 portfolio result required", + ContractErrorCode.TYPE_ERROR, + ) + _check( + type(covariance) is CovarianceSnapshot, + "$.covariance", + "typed covariance required", + ContractErrorCode.TYPE_ERROR, + ) + decision = RetrospectivePortfolioDecision.from_dict( + portfolio_decision.to_dict(), + backtest_run_ref=backtest_run_ref, + manifest=manifest, + target=portfolio_decision._target, + ) + actual_computed = _parse_utc(computed_at, "$.computed_at") + _check( + _parse_utc(decision.computed_at, "$.portfolio_decision.computed_at") <= actual_computed, + "$.computed_at", + "risk computation precedes portfolio computation", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + manifest_age = ( + actual_computed + - _parse_utc(decision._manifest.artifact_available_at, "$.manifest.artifact_available_at") + ).total_seconds() + _check( + 0 <= manifest_age <= decision.freshness_policy.max_manifest_age_seconds, + "$.manifest.artifact_available_at", + "manifest is stale at actual risk computation", + ) + _check( + covariance.data_snapshot_id == decision.dataset_snapshot_id, + "$.covariance.data_snapshot_id", + "covariance and decision data differ", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + business_date = _parse_utc(decision.effective_at, "$.portfolio_decision.effective_at").date() + _check( + covariance.window_start_date is not None and covariance.window_end_date is not None, + "$.covariance", + "bounded covariance window required", + ) + assert covariance.window_start_date is not None + assert covariance.window_end_date is not None + _check( + covariance.window_start_date + <= covariance.window_end_date + <= covariance.as_of_date + <= business_date, + "$.covariance.as_of_date", + "covariance business dates exceed the historical target date", + ContractErrorCode.TIME_ORDER_VIOLATION, + ) + _check( + (business_date - covariance.as_of_date).days + <= decision.freshness_policy.max_covariance_age_days, + "$.covariance.as_of_date", + "covariance is stale at historical target date", + ) + _check( + _digest("sha256:" + covariance.input_sha256, "$.covariance.input_sha256") + == decision.covariance_digest, + "$.covariance.input_sha256", + "covariance input differs from portfolio receipt", + ContractErrorCode.IDENTITY_MISMATCH, + ) + name = _text(risk_model_name, "$.risk_model_name") + version = _semver(risk_model_version, "$.risk_model_version") + model_digest = _digest(risk_model_digest, "$.risk_model_digest") + limit = ( + None + if portfolio_volatility_limit is None + else _finite_number( + portfolio_volatility_limit, "$.portfolio_volatility_limit", non_negative=True + ) + ) + budget: Mapping[str, float] = ( + MappingProxyType({}) + if risk_budget is None + else _immutable_float_mapping(risk_budget, "$.risk_budget") + ) + _check( + all(value >= 0 for value in budget.values()) + and set(budget) <= decision.target_weights.keys(), + "$.risk_budget", + "risk budgets must be non-negative and use target labels", + ) + normalized_groups = None + if groups is not None: + _check( + isinstance(groups, Mapping), + "$.groups", + "mapping required", + ContractErrorCode.TYPE_ERROR, + ) + group_values = { + _text(key, "$.groups.keys"): _text(value, "$.groups.values") + for key, value in groups.items() + } + _check( + set(group_values) == decision.target_weights.keys(), + "$.groups", + "groups must label every target exactly once", + ContractErrorCode.INPUT_CLOSURE_VIOLATION, + ) + normalized_groups = MappingProxyType(dict(sorted(group_values.items()))) + aligned = _validate_covariance_structure(decision.target_weights, covariance) + matrix_digest = _payload_digest( + { + "assets": sorted(decision.target_weights), + "matrix": aligned.to_numpy(dtype=float).tolist(), + } + ) + arguments: _RiskContext = { + "decision": decision, + "covariance": covariance, + "matrix_digest": matrix_digest, + "risk_model_name": name, + "risk_model_version": version, + "risk_model_digest": model_digest, + "portfolio_volatility_limit": limit, + "risk_budget": budget, + "groups": normalized_groups, + "computed_at": computed_at, + } + empty: Mapping[str, float] = MappingProxyType({}) + + def unavailable(finding: RiskFindingCode) -> RetrospectiveRiskAssessment: + return _risk_result( + **arguments, + marginal=empty, + component=empty, + percentage=empty, + volatility=None, + grouped=empty, + findings=(finding,), + status=RiskAssessmentStatus.UNAVAILABLE, + qualified=False, + ) + + weights = pd.Series(_mapping_dict(decision.target_weights), dtype=float, name="weight") + try: + decomposition = labeled_component_risk(weights, aligned * covariance.periods_per_year) + except ValueError as error: + finding = { + "covariance must be positive semidefinite": RiskFindingCode.COVARIANCE_NOT_PSD, + "weights and covariance must produce positive portfolio variance": RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE, + }.get(str(error)) + if finding is None: + raise PortfolioRiskContractError( + PortfolioRiskContractErrorCode.COMPUTATION_FAILURE, + "$.covariance", + "risk computation failed", + ) from error + return unavailable(finding) + marginal = _series_mapping(decomposition.marginal) + component = _series_mapping(decomposition.component) + percentage = _series_mapping(decomposition.percentage) + volatility = _finite_number( + decomposition.portfolio_volatility, "$.risk_output.portfolio_volatility", non_negative=True + ) + if not ( + set(marginal) == set(component) == set(percentage) == decision.target_weights.keys() + and math.isclose( + sum(component.values()), volatility, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL + ) + and math.isclose( + sum(percentage.values()), 1.0, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL + ) + ): + return unavailable(RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED) + grouped = ( + empty + if normalized_groups is None + else _series_mapping( + decomposition.grouped_component(pd.Series(dict(normalized_groups), dtype="object")) + ) + ) + breached = (limit is not None and volatility > limit + _CLOSURE_ATOL) or any( + percentage[label] > maximum + _CLOSURE_ATOL for label, maximum in budget.items() + ) + return _risk_result( + **arguments, + marginal=marginal, + component=component, + percentage=percentage, + volatility=volatility, + grouped=grouped, + findings=(RiskFindingCode.RISK_BUDGET_BREACH,) if breached else (), + status=RiskAssessmentStatus.READY, + qualified=not breached, + ) diff --git a/tests/fixtures/retrospective-computation-v2.golden.json b/tests/fixtures/retrospective-computation-v2.golden.json new file mode 100644 index 0000000..8a51abe --- /dev/null +++ b/tests/fixtures/retrospective-computation-v2.golden.json @@ -0,0 +1,1215 @@ +{ + "artifact_data_provenance": "envelope_test_only_not_end_to_end", + "artifact_tables": { + "attribution": [ + { + "asset_id": "SIM0", + "asset_total": 0.0, + "intraday": 0.0, + "overnight": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-02" + }, + { + "asset_id": "SIM1", + "asset_total": 0.0, + "intraday": 0.0, + "overnight": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-02" + }, + { + "asset_id": "SIM0", + "asset_total": 0.2, + "intraday": 0.2, + "overnight": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-03" + }, + { + "asset_id": "SIM1", + "asset_total": 0.0, + "intraday": 0.0, + "overnight": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-03" + }, + { + "asset_id": "SIM0", + "asset_total": 0.25, + "intraday": 0.0, + "overnight": 0.25, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-04" + }, + { + "asset_id": "SIM1", + "asset_total": -0.125, + "intraday": -0.125, + "overnight": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-04" + }, + { + "asset_id": "SIM0", + "asset_total": 0.0, + "intraday": 0.0, + "overnight": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-05" + }, + { + "asset_id": "SIM1", + "asset_total": 0.16666666666666666, + "intraday": 0.0, + "overnight": 0.16666666666666666, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-05" + } + ], + "attribution_daily": [ + { + "explained_return": 0.0, + "residual": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_return": 0.0, + "trade_date": "2018-01-02", + "transaction_cost": 0.0 + }, + { + "explained_return": 0.2, + "residual": -5.551115123125783e-17, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_return": 0.19999999999999996, + "trade_date": "2018-01-03", + "transaction_cost": -0.0 + }, + { + "explained_return": 0.125, + "residual": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_return": 0.125, + "trade_date": "2018-01-04", + "transaction_cost": -0.0 + }, + { + "explained_return": 0.16666666666666666, + "residual": 8.326672684688674e-17, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_return": 0.16666666666666674, + "trade_date": "2018-01-05", + "transaction_cost": 0.0 + } + ], + "nav": [ + { + "benchmark_nav": 1.0, + "benchmark_return": 0.0, + "cash": 1000.0, + "excess_ret": 0.0, + "nav": 1.0, + "pnl": 0.0, + "pnl_pct": 0.0, + "portfolio_value": 1000.0, + "position_value": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_cost": 0.0, + "trade_date": "2018-01-02", + "turnover": 0.0 + }, + { + "benchmark_nav": 1.01, + "benchmark_return": 0.01, + "cash": 0.0, + "excess_ret": 0.18999999999999995, + "nav": 1.2, + "pnl": 200.0, + "pnl_pct": 0.19999999999999996, + "portfolio_value": 1200.0, + "position_value": 1200.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_cost": 0.0, + "trade_date": "2018-01-03", + "turnover": 1.0 + }, + { + "benchmark_nav": 0.9999, + "benchmark_return": -0.01, + "cash": 0.0, + "excess_ret": 0.135, + "nav": 1.35, + "pnl": 150.0, + "pnl_pct": 0.125, + "portfolio_value": 1350.0, + "position_value": 1350.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_cost": 0.0, + "trade_date": "2018-01-04", + "turnover": 2.0 + }, + { + "benchmark_nav": 1.019898, + "benchmark_return": 0.02, + "cash": 0.0, + "excess_ret": 0.14666666666666675, + "nav": 1.575, + "pnl": 225.0, + "pnl_pct": 0.16666666666666674, + "portfolio_value": 1575.0, + "position_value": 1575.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "total_cost": 0.0, + "trade_date": "2018-01-05", + "turnover": 0.0 + } + ], + "performance": [ + { + "alpha": 123663320625.66454, + "ann_ret": 2683336646708.1, + "ann_volatility": 1.38901943830891, + "beta": 3.2500000000000013, + "calmar": 0.0, + "ir": 22.801264912443322, + "max_dd": 0.0, + "n_days": 4, + "n_trades": 3, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "sharpe": 1931820803008.3313, + "sortino": 0.0, + "total_ret": 0.575, + "tracking_error": 1.3032171729991897, + "win_rate": 0.75 + } + ], + "positions": [ + { + "asset_id": "CASH", + "asset_type": "cash", + "mark_price": 1.0, + "market_value": 1000.0, + "quantity": 1000.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-02", + "weight": 1.0 + }, + { + "asset_id": "SIM0", + "asset_type": "security", + "mark_price": 12.0, + "market_value": 1200.0, + "quantity": 100.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-03", + "weight": 1.0 + }, + { + "asset_id": "CASH", + "asset_type": "cash", + "mark_price": 1.0, + "market_value": 0.0, + "quantity": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-03", + "weight": 0.0 + }, + { + "asset_id": "SIM1", + "asset_type": "security", + "mark_price": 18.0, + "market_value": 1350.0, + "quantity": 75.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-04", + "weight": 1.0 + }, + { + "asset_id": "CASH", + "asset_type": "cash", + "mark_price": 1.0, + "market_value": 0.0, + "quantity": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-04", + "weight": 0.0 + }, + { + "asset_id": "SIM1", + "asset_type": "security", + "mark_price": 21.0, + "market_value": 1575.0, + "quantity": 75.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-05", + "weight": 1.0 + }, + { + "asset_id": "CASH", + "asset_type": "cash", + "mark_price": 1.0, + "market_value": 0.0, + "quantity": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "trade_date": "2018-01-05", + "weight": 0.0 + } + ], + "risk": [], + "run": [ + { + "benchmark_alignment_policy": "exact_session_index", + "benchmark_id": "synthetic.benchmark", + "calendar": "CN-A", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "config_hash": "d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "data_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "end_date": "2018-01-05", + "engine_version": "0.1.0", + "finished_at": "2026-09-08T01:10:00+00:00", + "frequency": "1d", + "initial_capital": 1000.0, + "params_json": "{\"lag_sessions\":1,\"top_k\":1}", + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "schema_version": "1.1.0", + "start_date": "2018-01-02", + "started_at": "2026-09-08T01:09:00+00:00", + "status": "success", + "strategy_id": "synthetic.top1", + "strategy_name": "Synthetic Top 1", + "strategy_version": "1.0.0", + "timezone": "Asia/Shanghai" + } + ], + "signals": [ + { + "asset_id": "SIM0", + "execution_date": "2018-01-03", + "factor_score": 2.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "signal_date": "2018-01-02", + "target_weight": 1.0 + }, + { + "asset_id": "SIM1", + "execution_date": "2018-01-03", + "factor_score": 1.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "signal_date": "2018-01-02", + "target_weight": 0.0 + }, + { + "asset_id": "SIM0", + "execution_date": "2018-01-04", + "factor_score": 0.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "signal_date": "2018-01-03", + "target_weight": 0.0 + }, + { + "asset_id": "SIM1", + "execution_date": "2018-01-04", + "factor_score": 3.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "signal_date": "2018-01-03", + "target_weight": 1.0 + } + ], + "trades": [ + { + "amount": 1000.0, + "fee": 0.0, + "price": 10.0, + "qty": 100.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "side": "buy", + "signal_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044:signal:2018-01-02", + "slippage": 0.0, + "total_cost": 0.0, + "trade_date": "2018-01-03", + "trade_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044:00000001", + "ts_code": "SIM0" + }, + { + "amount": 1500.0, + "fee": 0.0, + "price": 15.0, + "qty": 100.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "side": "sell", + "signal_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044:signal:2018-01-03", + "slippage": 0.0, + "total_cost": 0.0, + "trade_date": "2018-01-04", + "trade_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044:00000002", + "ts_code": "SIM0" + }, + { + "amount": 1500.0, + "fee": 0.0, + "price": 20.0, + "qty": 75.0, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "side": "buy", + "signal_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044:signal:2018-01-03", + "slippage": 0.0, + "total_cost": 0.0, + "trade_date": "2018-01-04", + "trade_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044:00000003", + "ts_code": "SIM1" + } + ] + }, + "backtest_evidence_manifest": { + "artifact_available_at": "2026-09-08T01:11:00Z", + "artifact_schema_version": "1.1.0", + "contract_name": "researchhub.backtest-evidence-manifest", + "decision_eligible": false, + "evidence": [ + { + "category": "run", + "evidence_digest": "sha256:79d678006cadf6e464de382549af86a50b0f656dd7ed68ca25cbce11a6148567", + "reconciliation": "closed", + "tables": [ + { + "content_digest": "sha256:9933c041fe06d58e5bc97e05e820e05ec29d6058fd2b4f9f445946242b2b9fba", + "logical_name": "run", + "row_count": 1, + "schema_digest": "sha256:3463e62fa4cd38d6155f80086ad30839272cb1fb05c1d466908c7bb2842d6e4a" + } + ] + }, + { + "category": "signal", + "evidence_digest": "sha256:7107c9908d8d39c920f05539c75167b7b6800e8be2f1773ca2e0986ee53dd316", + "reconciliation": "closed", + "tables": [ + { + "content_digest": "sha256:0b3f38967717e6ce9436421f950a595eef71504613ab0fe7d160801bcdce70f8", + "logical_name": "signals", + "row_count": 4, + "schema_digest": "sha256:cd113e3c8b9f1480c6f8839e33659926f94ba739f5ef7fb923ec0c5b2cce0456" + } + ] + }, + { + "category": "fill", + "evidence_digest": "sha256:b141526c0e53e061da82a2d6d096ddf74dd17d170a09c2b9b57b79b5b3f2e317", + "reconciliation": "closed", + "tables": [ + { + "content_digest": "sha256:1460a49d3487de51582377cdbba29c8a60b601ce65ce01699289c0473ad7006a", + "logical_name": "trades", + "row_count": 3, + "schema_digest": "sha256:e64396c22a1dc301aea664601cea9929e826b68b13f764c63866f83c757c47ab" + } + ] + }, + { + "category": "position_nav", + "evidence_digest": "sha256:d324bbc9a5403f5ff96cc8cdfc0dad6ce2c84afe667b8d1c64d048840eeb2b03", + "reconciliation": "closed", + "tables": [ + { + "content_digest": "sha256:744a39f19ba451c8c930d7603aa5a93d534d7ea48bdf0aa84daae0367099025c", + "logical_name": "positions", + "row_count": 7, + "schema_digest": "sha256:9d4735b06437f5c0c47df60776cebf1b6ec0214d696daeedfc37af4a9934a1ce" + }, + { + "content_digest": "sha256:9afca6628168e18c87ab9128795fe2cd91a01950f11028ed1d6fd13f5083a8f4", + "logical_name": "nav", + "row_count": 4, + "schema_digest": "sha256:3fc0061f769c2f48ae0c282b53607a8984a2d1b4ee48b494a684311d7970d8e5" + } + ] + }, + { + "category": "performance", + "evidence_digest": "sha256:ebc4cab78e8f5570a8941140c984126d94643b881bed9beb47b6748f77e5392d", + "reconciliation": "closed", + "tables": [ + { + "content_digest": "sha256:8ecc2d9dd084efeac04c8961a44d39843b87a8ddd2042ec64883b5887d9aacc5", + "logical_name": "performance", + "row_count": 1, + "schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0" + } + ] + }, + { + "category": "attribution", + "evidence_digest": "sha256:3adc63b9e87febfccce47d4349538d3088d94a0a2540f33b5046f239d0b981b0", + "reconciliation": "closed", + "tables": [ + { + "content_digest": "sha256:63c748edc047306026e4c563935699469df488473e3cbbae49806b0decc7a793", + "logical_name": "attribution", + "row_count": 8, + "schema_digest": "sha256:4cb0d6a53d62bd4342d06a5ab397ff73f93b351d37bb7566a3016fd0f6040f7b" + }, + { + "content_digest": "sha256:0f13d9bc350bf865543e8eaa4782fdb6b8e896b852732fad8be059333addc96c", + "logical_name": "attribution_daily", + "row_count": 4, + "schema_digest": "sha256:7785620e47a7472886ac9b7b385e04a5597aad45200a3e3c53059032b183b178" + } + ] + }, + { + "category": "risk_snapshot", + "evidence_digest": "sha256:1fae7ee94a7c1b1bb993e81712a6ab01f3eba3992befb0d76046770a68e55a64", + "reconciliation": "closed", + "tables": [ + { + "content_digest": "sha256:4816dd4812b5ff2e97e74bf34ca2221bfde675b387a96e277683557f0ad7d975", + "logical_name": "risk", + "row_count": 0, + "schema_digest": "sha256:b844eeae60edc2c997502eb65e15c33040c7a9f1c1c4b12e74a81c5e282e44b5" + } + ] + }, + { + "category": "replay", + "evidence_digest": "sha256:3e0389ff96f2d7e4093ad8f439d29b707c464aef509134298dc6d7ae859ba9de", + "reconciliation": "closed", + "tables": [] + } + ], + "evidence_digest": "sha256:1b47221b0f0116682cadffc182fe9f535e2a4fc1931d29253a3cc29da336a838", + "evidence_scope": "synthetic_fixture", + "execution_validation": "not_validated", + "historical_availability": "not_established", + "manifest_id": "rhbacktestevidencev2:sha256:2281f48c107b85fa00cd01221d7fbb300a54688cecd0e8a980650ecd6e629696", + "observation_cutoff": "2026-09-08T01:01:00Z", + "profile": "offline_research_retrospective_v2", + "qualification": "contract_qualified", + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "run_reference": { + "kind": "backtest_run_ref", + "value": { + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "computed_at": "2026-09-08T01:10:00Z", + "configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "contract_name": "researchhub.backtest-run-ref", + "corporate_action_digest": "sha256:4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945", + "corporate_action_revision_ids": [], + "cost_model_digest": "sha256:59f01f62f4455be06bf3d8148e2c3d99263f1cf0f7673af25adfaa3851687483", + "cost_model_version": "1.0.0", + "dataset_content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "dataset_manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "decision_eligible": false, + "environment_lock_digest": "sha256:b3131d3051dcf6796451ebc214763a7dd822e8ac1d07e46e7bcc3962a2fb69bb", + "evaluation_at": "2026-09-08T01:09:00Z", + "evidence_scope": "synthetic_fixture", + "execution_model_digest": "sha256:ec494dc34f36f055ffe03155836ea82d2eb1af8deee7d25681eec13f6ffc64c5", + "execution_model_version": "1.0.0", + "execution_validation": "not_validated", + "factor_output_content_digest": "sha256:0a5e12736ed241958fc7bd5f58a3577cb69558ba3c67ac87068e340fe4beb709", + "factor_set_digest": "sha256:4116afc8b17608e033df46319bb68b343f88a23e3a0d04ecf81453c336e0b4dc", + "factor_set_id": "rhfactorsetv2:sha256:4116afc8b17608e033df46319bb68b343f88a23e3a0d04ecf81453c336e0b4dc", + "foundation_digest": "sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "historical_availability": "not_established", + "observation_cutoff": "2026-09-08T01:01:00Z", + "random_seed": 7, + "replay_ancestor_run_ids": [], + "replay_attempt": 0, + "replay_parent_run_id": null, + "replay_reason": null, + "replay_spec_digest": "sha256:e4c94920a8d5616a223624c607908d42453bdabf87dd44fdd8f7fbe24b6e91f1", + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "schema_version": "2.0.0", + "strategy_digest": "sha256:b76f507a34775c831fecbbe38499f67b0424e5126af5c61832b31860916b79dd", + "strategy_id": "synthetic.top1", + "strategy_version": "1.0.0", + "trading_calendar_digest": "sha256:baae76476f6902b44ce83170191adeafed5faae9a3faef62b8f9d3d232bc63ab", + "trading_calendar_revision_ids": [ + "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25" + ], + "universe_digest": "sha256:37722cf3accfd32550732e05a57fd8f9532f9752a26e62fb09fed80313613164", + "usage": "retrospective_research" + } + }, + "schema_version": "2.0.0", + "usage": "retrospective_research" + }, + "backtest_run_ref": { + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "computed_at": "2026-09-08T01:10:00Z", + "configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "contract_name": "researchhub.backtest-run-ref", + "corporate_action_digest": "sha256:4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945", + "corporate_action_revision_ids": [], + "cost_model_digest": "sha256:59f01f62f4455be06bf3d8148e2c3d99263f1cf0f7673af25adfaa3851687483", + "cost_model_version": "1.0.0", + "dataset_content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "dataset_manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "decision_eligible": false, + "environment_lock_digest": "sha256:b3131d3051dcf6796451ebc214763a7dd822e8ac1d07e46e7bcc3962a2fb69bb", + "evaluation_at": "2026-09-08T01:09:00Z", + "evidence_scope": "synthetic_fixture", + "execution_model_digest": "sha256:ec494dc34f36f055ffe03155836ea82d2eb1af8deee7d25681eec13f6ffc64c5", + "execution_model_version": "1.0.0", + "execution_validation": "not_validated", + "factor_output_content_digest": "sha256:0a5e12736ed241958fc7bd5f58a3577cb69558ba3c67ac87068e340fe4beb709", + "factor_set_digest": "sha256:4116afc8b17608e033df46319bb68b343f88a23e3a0d04ecf81453c336e0b4dc", + "factor_set_id": "rhfactorsetv2:sha256:4116afc8b17608e033df46319bb68b343f88a23e3a0d04ecf81453c336e0b4dc", + "foundation_digest": "sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "historical_availability": "not_established", + "observation_cutoff": "2026-09-08T01:01:00Z", + "random_seed": 7, + "replay_ancestor_run_ids": [], + "replay_attempt": 0, + "replay_parent_run_id": null, + "replay_reason": null, + "replay_spec_digest": "sha256:e4c94920a8d5616a223624c607908d42453bdabf87dd44fdd8f7fbe24b6e91f1", + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "schema_version": "2.0.0", + "strategy_digest": "sha256:b76f507a34775c831fecbbe38499f67b0424e5126af5c61832b31860916b79dd", + "strategy_id": "synthetic.top1", + "strategy_version": "1.0.0", + "trading_calendar_digest": "sha256:baae76476f6902b44ce83170191adeafed5faae9a3faef62b8f9d3d232bc63ab", + "trading_calendar_revision_ids": [ + "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25" + ], + "universe_digest": "sha256:37722cf3accfd32550732e05a57fd8f9532f9752a26e62fb09fed80313613164", + "usage": "retrospective_research" + }, + "covariance_matrix": { + "rhinstrument:11111111111111111111111111111111": { + "rhinstrument:11111111111111111111111111111111": 0.04, + "rhinstrument:22222222222222222222222222222222": 0.01 + }, + "rhinstrument:22222222222222222222222222222222": { + "rhinstrument:11111111111111111111111111111111": 0.01, + "rhinstrument:22222222222222222222222222222222": 0.09 + } + }, + "dataset_chunks": [ + [ + { + "effective_time": "2018-01-02T07:00:00Z", + "instrument_id": "rhinstrument:11111111111111111111111111111111", + "metric": "close", + "value": "101.25" + }, + { + "effective_time": "2018-01-02T07:00:00Z", + "instrument_id": "rhinstrument:22222222222222222222222222222222", + "metric": "close", + "value": "87.50" + } + ] + ], + "factor_definitions": [ + { + "code_revision": "cccccccccccccccccccccccccccccccccccccccc", + "contract_name": "researchhub.factor-definition", + "definition_id": "rhfactorv1:sha256:a3a94c3f9776aab7c71901c84fee834e2dc57ca96a012f63b02aae3423b95957", + "factor_id": "neutral_close", + "formula": "value", + "implementation_digest": "sha256:0fa01f5bdde718e21e1ef22cd3d59959dc58306ee9b3ca8b869298eda05302c1", + "input_schema_digest": "sha256:a49478618a6b07300720a0936fa6ce1ede0229bcc21cc1ca2ff3d596b1a2b289", + "inputs": [ + { + "input_name": "market", + "required_columns": [ + "value" + ], + "schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d" + } + ], + "lag_sessions": 1, + "parameters": {}, + "producer": { + "id": "quant_engine", + "version": "0.1.0" + }, + "schema_version": "1.0.0", + "valid_from": "2026-01-01T00:00:00Z", + "valid_until": "2027-01-01T00:00:00Z", + "version": "1.0.0", + "warmup_sessions": 0 + } + ], + "factor_output_records": [ + { + "instrument_id": "rhinstrument:11111111111111111111111111111111", + "value": "101.25" + }, + { + "instrument_id": "rhinstrument:22222222222222222222222222222222", + "value": "87.50" + } + ], + "factor_output_schema": { + "fields": [ + "instrument_id", + "value" + ] + }, + "factor_set": { + "actor": { + "id": "synthetic.research", + "kind": "service" + }, + "artifact_available_at": "2026-09-08T01:08:00Z", + "availability_mode": "retrospective_replay", + "causation": { + "id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "kind": "foundation" + }, + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "computed_at": "2026-09-08T01:07:00Z", + "contract_name": "researchhub.factor-set-ref", + "correlation_id": "synthetic.retrospective", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "decision_eligible": false, + "definition_ids": [ + "rhfactorv1:sha256:a3a94c3f9776aab7c71901c84fee834e2dc57ca96a012f63b02aae3423b95957" + ], + "evaluation_at": "2026-09-08T01:06:00Z", + "evidence_scope": "synthetic_fixture", + "factor_set_id": "rhfactorsetv2:sha256:4116afc8b17608e033df46319bb68b343f88a23e3a0d04ecf81453c336e0b4dc", + "foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "historical_availability": "not_established", + "input_bindings": [ + { + "definition_id": "rhfactorv1:sha256:a3a94c3f9776aab7c71901c84fee834e2dc57ca96a012f63b02aae3423b95957", + "input_name": "market", + "schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d", + "view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309" + } + ], + "observation_cutoff": "2026-09-08T01:01:00Z", + "output_artifact_ref": { + "artifact_id": "rhfactoroutputv1:sha256:cc93ed5d16a89ae529f45373b2b86a600375892267f5e8d39fc1e3d7acdaf520", + "content_digest": "sha256:0a5e12736ed241958fc7bd5f58a3577cb69558ba3c67ac87068e340fe4beb709", + "schema_digest": "sha256:fbe288c6a7ad0ba0aa337719533767d3cd815fe16f408ebee79d0bf8994a860b" + }, + "output_content_digest": "sha256:0a5e12736ed241958fc7bd5f58a3577cb69558ba3c67ac87068e340fe4beb709", + "output_coverage": { + "domain": "synthetic_market", + "evidence_digest": "sha256:b6ab14382af395b7271d8344f394ecd6dc56077bc7a08d6a92a54b2b07f39120", + "expected_count": 2, + "observed_count": 2, + "status": "complete", + "unit": "records" + }, + "output_quality": { + "checks": [ + { + "check_id": "finite_values", + "evidence_digest": "sha256:800eb85775bb67c51646b5c0a72d071eb81026d8a71f93615947806ba22327b5", + "status": "passed" + } + ], + "status": "passed" + }, + "output_schema_digest": "sha256:fbe288c6a7ad0ba0aa337719533767d3cd815fe16f408ebee79d0bf8994a860b", + "producer": { + "id": "quant_engine", + "version": "0.1.0" + }, + "schema_version": "2.0.0", + "selected_view_ref_ids": [ + "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309" + ], + "upstream_evidence": { + "content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "evidence_scope": "synthetic_fixture", + "foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "foundation_readiness": { + "contract_validation": { + "evidence_digests": [ + "sha256:62f2abbbb79f4ff9164e10cc4b21df37f0964224a4c546f978b97794241ff5b3" + ], + "status": "validated" + }, + "evidence_scope": "synthetic_fixture", + "live_validation": { + "evidence_digests": [], + "status": "not_validated" + }, + "production_validation": { + "evidence_digests": [], + "status": "not_validated" + }, + "real_data_validation": { + "evidence_digests": [], + "status": "not_validated" + } + }, + "manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7", + "observation_manifest_digest": "sha256:8722b288ada6bd1bf436040c7f1783222a41027f3a89c3bef7f244abcc78c7ec", + "qualification": { + "evaluated_at": "2026-09-08T01:02:00Z", + "evidence_digest": "sha256:70c3ddcadf095877f17a1365c8d75a2e5a7078a27722fb8b698e61d350c6bb2d", + "policy_id": "researchhub.dataset-snapshot.retrospective", + "policy_version": "2.0.0", + "status": "qualified", + "usage": "retrospective_research" + }, + "quality": { + "checks": [ + { + "check_id": "completeness", + "evidence_digest": "sha256:519b01c60ad85c2b03d6c4f29027fe9b3672a8f6ae3ca8a2d831b281a61c5095", + "severity": "blocking", + "status": "passed" + }, + { + "check_id": "duplicate_identity", + "evidence_digest": "sha256:7c585c7471ab45886fce037061b5ab0d5f981aede5d5190021963676b0ec0fcc", + "severity": "blocking", + "status": "passed" + }, + { + "check_id": "observation_coverage", + "evidence_digest": "sha256:dde5693c3ad79a01a162f9fc5a40c69d3c284965e283c91c5cd456904e0ef737", + "severity": "blocking", + "status": "passed" + }, + { + "check_id": "historical_claim_policy", + "evidence_digest": "sha256:f4cf8da48f92e03d75bb28b2569061e649f382ba360d879fef8e41e35169c228", + "severity": "blocking", + "status": "passed" + }, + { + "check_id": "range_validity", + "evidence_digest": "sha256:6af08355c5e29d846e1c48783dccde791e8a3f48f8416149413217b7c2cbb52a", + "severity": "blocking", + "status": "passed" + }, + { + "check_id": "schema_conformance", + "evidence_digest": "sha256:fefdc5b81eba128af92f15a60feb7bb6f40b2e7277bf2beb48c1bb9399ee564b", + "severity": "blocking", + "status": "passed" + } + ], + "status": "passed" + }, + "time_semantics": { + "earliest_external_knowledge": { + "status": "unknown" + }, + "effective_time": { + "end_inclusive": "2018-01-02T07:00:00Z", + "start_inclusive": "2018-01-02T07:00:00Z" + }, + "historical_availability": "not_established", + "observation_cutoff": "2026-09-08T01:01:00Z" + } + }, + "usage": "retrospective_research", + "view_availability": [ + { + "available_at": "2026-09-08T01:04:00Z", + "evidence_digest": "sha256:1bc7878d67fe5c8d3f5cefdfdba0ee606ac196e5fbe8c53fdde1833a1665e9fa", + "view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309" + } + ] + }, + "fixture_kind": "synthetic_retrospective_contract_vector", + "performance_evidence": { + "artifact_available_at": "2026-09-08T01:11:00Z", + "authority": "quant_engine", + "backtest_evidence_manifest_document_sha256": "sha256:91a68769b3844b436690e4a0f18befd8a7376b0f6349fed2a0816c02f396ca72", + "backtest_evidence_manifest_evidence_digest": "sha256:1b47221b0f0116682cadffc182fe9f535e2a4fc1931d29253a3cc29da336a838", + "backtest_evidence_manifest_id": "rhbacktestevidencev2:sha256:2281f48c107b85fa00cd01221d7fbb300a54688cecd0e8a980650ecd6e629696", + "backtest_evidence_qualification": "contract_qualified", + "backtest_run_ref_document_sha256": "sha256:e1d0182752256ff68932636c8b132d8d370858be67935e9a250bcfe8aa9e8eea", + "backtest_run_ref_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "benchmark_alignment_policy": "exact_session_index", + "benchmark_id": "synthetic.benchmark", + "benchmark_series_digest": "sha256:817fd6bbfa00326cfdc2b4150d06831bf395dcc1e4bd8e06c2aa7fb7a492bf78", + "calendar": "CN-A", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "computed_at": "2026-09-08T01:10:00Z", + "configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d", + "cost_model_digest": "sha256:59f01f62f4455be06bf3d8148e2c3d99263f1cf0f7673af25adfaa3851687483", + "cost_model_version": "1.0.0", + "dataset_content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "dataset_manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "decision_eligible": false, + "document_sha256": "sha256:5a846defd41665666d91f3310b9ba95afc7d5ecb24878067ff8f385c486baf78", + "end_date": "2018-01-05", + "environment_lock_digest": "sha256:b3131d3051dcf6796451ebc214763a7dd822e8ac1d07e46e7bcc3962a2fb69bb", + "evidence_scope": "synthetic_fixture", + "execution_model_digest": "sha256:ec494dc34f36f055ffe03155836ea82d2eb1af8deee7d25681eec13f6ffc64c5", + "execution_model_version": "1.0.0", + "execution_validation": "not_validated", + "factor_output_content_digest": "sha256:0a5e12736ed241958fc7bd5f58a3577cb69558ba3c67ac87068e340fe4beb709", + "factor_set_digest": "sha256:4116afc8b17608e033df46319bb68b343f88a23e3a0d04ecf81453c336e0b4dc", + "factor_set_id": "rhfactorsetv2:sha256:4116afc8b17608e033df46319bb68b343f88a23e3a0d04ecf81453c336e0b4dc", + "foundation_digest": "sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260", + "frequency": "1d", + "historical_availability": "not_established", + "methodology": { + "alpha": "daily_ols_intercept_geometric_annualization", + "annual_risk_free": 0.0, + "annualized_return": "geometric_compound", + "annualized_volatility": "sample_std_sqrt_periods", + "benchmark_alignment": "exact_session_index", + "benchmark_risk_free_daily": 0.0, + "beta": "sample_covariance_over_sample_variance", + "calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown", + "code_revision": "dddddddddddddddddddddddddddddddddddddddd", + "implementation_module": "quant_engine.metrics", + "implementation_version": "researchhub.quant-performance-methodology.v1", + "information_ratio": "mean_active_over_sample_std_active_sqrt_periods", + "maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "periods_per_year": 252, + "return_type": "simple", + "sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility", + "sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods", + "source_frequency": "1d", + "total_return": "final_nav_minus_one", + "tracking_error": "sample_std_active_return_sqrt_periods", + "win_rate": "positive_daily_return_count_over_observation_count" + }, + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "metrics": [ + { + "availability": "available", + "key": "total_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "total_ret", + "unit": "ratio", + "value": 0.575 + }, + { + "availability": "available", + "key": "annualized_return", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_ret", + "unit": "ratio_per_year", + "value": 2683336646708.1 + }, + { + "availability": "available", + "key": "annualized_volatility", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "ann_volatility", + "unit": "ratio_per_year", + "value": 1.38901943830891 + }, + { + "availability": "available", + "key": "sharpe_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sharpe", + "unit": "ratio", + "value": 1931820803008.3313 + }, + { + "availability": "available", + "key": "sortino_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "sortino", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "maximum_drawdown", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "max_dd", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "calmar_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "calmar", + "unit": "ratio", + "value": 0.0 + }, + { + "availability": "available", + "key": "win_rate", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "win_rate", + "unit": "ratio", + "value": 0.75 + }, + { + "availability": "available", + "key": "tracking_error", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "tracking_error", + "unit": "ratio_per_year", + "value": 1.3032171729991897 + }, + { + "availability": "available", + "key": "information_ratio", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "ir", + "unit": "ratio", + "value": 22.801264912443322 + }, + { + "availability": "available", + "key": "alpha", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "alpha", + "unit": "ratio_per_year", + "value": 123663320625.66454 + }, + { + "availability": "available", + "key": "beta", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": true, + "source_column": "beta", + "unit": "ratio", + "value": 3.2500000000000013 + }, + { + "availability": "available", + "key": "trade_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_trades", + "unit": "count", + "value": 3 + }, + { + "availability": "available", + "key": "day_count", + "methodology_id": "researchhub.quant-performance-methodology.v1", + "metric_schema_id": "researchhub.quant-performance-metrics.v1", + "nullable": false, + "source_column": "n_days", + "unit": "count", + "value": 4 + } + ], + "observation_cutoff": "2026-09-08T01:01:00Z", + "performance_evidence_id": "rhperformancev2:sha256:fa27254a1229f16b548495037dfbaef1e3ec2506c9d9a1f9fb4956363ecc0e07", + "performance_row_digest": "sha256:910cb63d4054d2ce76762c28cabc8beaf36767a123a4082d6a8f83e2c322171f", + "performance_table_content_digest": "sha256:8ecc2d9dd084efeac04c8961a44d39843b87a8ddd2042ec64883b5887d9aacc5", + "performance_table_logical_name": "performance", + "performance_table_row_count": 1, + "performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0", + "research_artifact_content_digest": "sha256:c34804a255fadb3d825aea1a87735d4cdd9985620d1506d5818325ed6413ae54", + "research_artifact_schema_version": "1.1.0", + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "schema_version": "researchhub.performance-evidence.v2", + "scope": "offline_retrospective_research_only", + "start_date": "2018-01-02", + "strategy_digest": "sha256:b76f507a34775c831fecbbe38499f67b0424e5126af5c61832b31860916b79dd", + "strategy_id": "synthetic.top1", + "strategy_version": "1.0.0", + "timezone": "Asia/Shanghai", + "usage": "retrospective_research" + }, + "portfolio_decision": { + "computed_at": "2026-09-08T01:13:00Z", + "constraint_residuals": { + "gross_exposure_max": 0.0, + "net_exposure_max": 0.0, + "net_exposure_min": 0.0, + "position_count_max": 0.0, + "single_asset_max": 0.0, + "single_asset_min": 0.0, + "turnover_max": 0.0 + }, + "constraints": { + "gross_exposure_max": 1.0, + "net_exposure_max": 1.0, + "net_exposure_min": 1.0, + "position_count_max": 2, + "schema_version": "1.0.0", + "single_asset_max": 0.7, + "single_asset_min": 0.2, + "turnover_max": 0.2 + }, + "contract_name": "researchhub.portfolio-decision", + "covariance_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "created_at": "2026-09-08T01:12:00Z", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "decision_eligible": false, + "decision_id": "rhportfoliodecisionv2:sha256:d6987931d7feb7bdbee421d9be280f6b42a50d9c80a02369049b618f7e2159e4", + "effective_at": "2018-01-05T07:00:00Z", + "evidence_digest": "sha256:1b47221b0f0116682cadffc182fe9f535e2a4fc1931d29253a3cc29da336a838", + "evidence_scope": "synthetic_fixture", + "execution_validation": "not_validated", + "expected_return_digest": "sha256:bb364736b6128fe1caf2dd01d1470ee5e4ee063b45aa98dfdc32e6b7a89e3659", + "freshness_policy": { + "max_covariance_age_days": 0, + "max_manifest_age_seconds": 3600, + "schema_version": "1.0.0" + }, + "gross_exposure": 1.0, + "historical_availability": "not_established", + "manifest_document_sha256": "91a68769b3844b436690e4a0f18befd8a7376b0f6349fed2a0816c02f396ca72", + "manifest_id": "rhbacktestevidencev2:sha256:2281f48c107b85fa00cd01221d7fbb300a54688cecd0e8a980650ecd6e629696", + "model_digest": "sha256:f8fc77be8bcdc147f5725349353a71a8396440f4c89142007b62b328d8a87926", + "model_name": "bounded_weights", + "model_version": "1.0.0", + "net_exposure": 1.0, + "objective_digest": "sha256:208b735cfbcb81c4911b9c27b38bcf51ac06ba3eb416a62f1a690ca8c8fe4f85", + "objective_name": "synthetic_allocation", + "objective_version": "1.0.0", + "observation_cutoff": "2026-09-08T01:01:00Z", + "output_digest": "sha256:3213fda3ff396470fd2c680976e7118a6d816a5a07b05dadfd5994db5d01e8d0", + "portfolio_asset_set_digest": "sha256:330735ad79464ff1e3e101d51790433df8c8dda97b64b5e7cc45fd8335248753", + "position_count": 2, + "prior_weights": { + "rhinstrument:11111111111111111111111111111111": 0.5, + "rhinstrument:22222222222222222222222222222222": 0.5 + }, + "receipt": { + "algorithm": "bounded_weights", + "algorithm_version": "1.0.0", + "computed_at": "2026-09-08T01:13:00Z", + "constraint_digest": "sha256:34df0e5c00f503748ff936f9cd415a8947169181f8ca0917bce97dd606e08e94", + "implementation_digest": "sha256:d82e7d69c3929680c87ea9ff959f0f197c82e4f2b447cdf880132fe2e49f4bdf", + "input_digest": "sha256:dcb70bc59eedd7c031f86c096485aae556d2a04f968d1da64243c1b3698686b0", + "iterations": null, + "max_constraint_residual": 0.0, + "objective_value": null, + "output_digest": "sha256:3213fda3ff396470fd2c680976e7118a6d816a5a07b05dadfd5994db5d01e8d0", + "parameter_digest": "sha256:9ce595f121b5d9cfc99b65c938040f945f256a0eae60d59e8047c7de2f417a31", + "schema_version": "1.0.0", + "solver_config_digest": null, + "solver_name": null, + "solver_required": false, + "solver_version": null, + "status": "completed", + "tolerance": 1e-12 + }, + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "run_ref_document_sha256": "e1d0182752256ff68932636c8b132d8d370858be67935e9a250bcfe8aa9e8eea", + "scenario_digest": "sha256:a6fd5ec0216d7b36ed844b21fc827dd6b7aab37bf77adf675b1d270855fc955e", + "schema_version": "2.0.0", + "source_universe_digest": "sha256:37722cf3accfd32550732e05a57fd8f9532f9752a26e62fb09fed80313613164", + "target_id": "rhportfoliotargetv2:sha256:0f896072a074f5895b72a1a5240c41771abcc4056c34776af219f5d8b4614c9e", + "target_weights": { + "rhinstrument:11111111111111111111111111111111": 0.6, + "rhinstrument:22222222222222222222222222222222": 0.4 + }, + "turnover_l1": 0.19999999999999996, + "usage": "retrospective_research" + }, + "portfolio_target": { + "backtest_run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "contract_name": "researchhub.portfolio-target", + "created_at": "2026-09-08T01:12:00Z", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "effective_at": "2018-01-05T07:00:00Z", + "historical_availability": "not_established", + "schema_version": "2.0.0", + "target_id": "rhportfoliotargetv2:sha256:0f896072a074f5895b72a1a5240c41771abcc4056c34776af219f5d8b4614c9e", + "usage": "retrospective_research", + "weights": { + "rhinstrument:11111111111111111111111111111111": 0.6, + "rhinstrument:22222222222222222222222222222222": 0.4 + } + }, + "resolved_view_schema": { + "synthetic_schema": "neutral_close_v2" + }, + "risk_assessment": { + "assessment_id": "rhriskassessmentv2:sha256:fad4085f21bf90aaa2b63424e71228c8b9dd6387c44321e4890763ad51bab044", + "component_risk": { + "rhinstrument:11111111111111111111111111111111": 1.4549226783578566, + "rhinstrument:22222222222222222222222222222222": 1.4549226783578568 + }, + "computed_at": "2026-09-08T01:14:00Z", + "contract_name": "researchhub.risk-assessment", + "covariance_as_of_date": "2018-01-05", + "covariance_data_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "covariance_input_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "covariance_lookback_sessions": 4, + "covariance_matrix_digest": "sha256:31f52dfc29d9b4c1d99ebd0e4400916880ef7e98c359c07cc89a9f55d6364a28", + "covariance_method": "provided", + "covariance_missing_policy": "complete_case", + "covariance_observations": 4, + "covariance_snapshot_id": "covariance:synthetic-retrospective", + "covariance_window_end_date": "2018-01-05", + "covariance_window_start_date": "2018-01-02", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "decision_eligible": false, + "decision_id": "rhportfoliodecisionv2:sha256:d6987931d7feb7bdbee421d9be280f6b42a50d9c80a02369049b618f7e2159e4", + "effective_at": "2018-01-05T07:00:00Z", + "evidence_scope": "synthetic_fixture", + "execution_validation": "not_validated", + "findings": [], + "freshness_policy_digest": "sha256:833f58f4d1056f0450f7369fd4edd70534cc74e9e55cd60f05b2b3d7a2979763", + "group_exposure": { + "equity": 1.4549226783578566, + "fixed_income": 1.4549226783578568 + }, + "groups": { + "rhinstrument:11111111111111111111111111111111": "equity", + "rhinstrument:22222222222222222222222222222222": "fixed_income" + }, + "historical_availability": "not_established", + "manifest_id": "rhbacktestevidencev2:sha256:2281f48c107b85fa00cd01221d7fbb300a54688cecd0e8a980650ecd6e629696", + "marginal_risk": { + "rhinstrument:11111111111111111111111111111111": 2.424871130596428, + "rhinstrument:22222222222222222222222222222222": 3.637306695894642 + }, + "observation_cutoff": "2026-09-08T01:01:00Z", + "percentage_risk": { + "rhinstrument:11111111111111111111111111111111": 0.49999999999999983, + "rhinstrument:22222222222222222222222222222222": 0.49999999999999994 + }, + "periods_per_year": 252, + "portfolio_volatility": 2.909845356715714, + "portfolio_volatility_limit": 10.0, + "qualified": true, + "return_frequency": "1d", + "risk_budget": { + "rhinstrument:11111111111111111111111111111111": 0.8, + "rhinstrument:22222222222222222222222222222222": 0.8 + }, + "risk_model_digest": "sha256:b70e3fd4fe74354b6cfcea70a04a8261661ac56199bb58b435db921f2f2adfdb", + "risk_model_name": "euler_volatility", + "risk_model_version": "1.0.0", + "run_id": "rhbacktestrunv2:sha256:42ed87936dd3b4cd639d11ba73be021405b81bab536df18597cec4c35b14e044", + "scenario_digest": "sha256:a6fd5ec0216d7b36ed844b21fc827dd6b7aab37bf77adf675b1d270855fc955e", + "schema_version": "2.0.0", + "status": "ready", + "usage": "retrospective_research" + }, + "source_authenticity": "not_established" +} diff --git a/tests/fixtures/retrospective-data-foundation-v2.golden.json b/tests/fixtures/retrospective-data-foundation-v2.golden.json new file mode 100644 index 0000000..54bc480 --- /dev/null +++ b/tests/fixtures/retrospective-data-foundation-v2.golden.json @@ -0,0 +1,175 @@ +{ + "contract_name": "researchhub.data-foundation", + "schema_version": "2.0.0", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "observation_cutoff": "2026-09-08T01:01:00Z", + "usage": "retrospective_research", + "historical_availability": "not_established", + "published_at": "2026-09-08T01:05:00Z", + "instrument_routes": [ + { + "observation_sequence": 1, + "observed_by": "2026-09-08T01:00:00Z", + "earliest_external_knowledge": { + "status": "unknown" + }, + "history_completeness": "not_established", + "instrument_id": "rhinstrument:11111111111111111111111111111111", + "symbol": "SIM0", + "mic": "XSHG", + "currency": "CNY", + "asset_class": "equity", + "instrument_type": "stock", + "calendar_id": "rhcalendar:33333333333333333333333333333333", + "effective_from": "2018-01-01T00:00:00Z", + "evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb", + "route_revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9" + }, + { + "observation_sequence": 1, + "observed_by": "2026-09-08T01:00:00Z", + "earliest_external_knowledge": { + "status": "unknown" + }, + "history_completeness": "not_established", + "instrument_id": "rhinstrument:22222222222222222222222222222222", + "symbol": "SIM1", + "mic": "XSHG", + "currency": "CNY", + "asset_class": "equity", + "instrument_type": "stock", + "calendar_id": "rhcalendar:33333333333333333333333333333333", + "effective_from": "2018-01-01T00:00:00Z", + "evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799", + "route_revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92" + } + ], + "trading_calendar_revisions": [ + { + "observation_sequence": 1, + "observed_by": "2026-09-08T01:00:00Z", + "earliest_external_knowledge": { + "status": "unknown" + }, + "history_completeness": "not_established", + "calendar_id": "rhcalendar:33333333333333333333333333333333", + "session_date": "2018-01-02", + "status": "open", + "sessions": [ + { + "opens_at": "2018-01-02T01:30:00Z", + "closes_at": "2018-01-02T07:00:00Z" + } + ], + "evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a", + "calendar_revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25" + } + ], + "corporate_action_revisions": [], + "standardized_views": [ + { + "view_id": "rhview:66666666666666666666666666666666", + "view_version": "2.0.0", + "dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905", + "observation_cutoff": "2026-09-08T01:01:00Z", + "schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d", + "content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "transformation_digest": "sha256:b2376dbe4424b38e5b27874c998f082aecc086179f611896ebaa0fdeb5362c68", + "instrument_route_revision_ids": [ + "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9", + "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92" + ], + "trading_calendar_revision_ids": [ + "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25" + ], + "corporate_action_revision_ids": [], + "usage": "retrospective_research", + "historical_availability": "not_established", + "available_at": "2026-09-08T01:04:00Z", + "view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309" + } + ], + "observation_lineage": [ + { + "revision_kind": "instrument_route", + "revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9", + "observation_sequence": 1, + "observed_by": "2026-09-08T01:00:00Z", + "earliest_external_knowledge": { + "status": "unknown" + }, + "history_completeness": "not_established", + "evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb" + }, + { + "revision_kind": "instrument_route", + "revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92", + "observation_sequence": 1, + "observed_by": "2026-09-08T01:00:00Z", + "earliest_external_knowledge": { + "status": "unknown" + }, + "history_completeness": "not_established", + "evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799" + }, + { + "revision_kind": "trading_calendar", + "revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25", + "observation_sequence": 1, + "observed_by": "2026-09-08T01:00:00Z", + "earliest_external_knowledge": { + "status": "unknown" + }, + "history_completeness": "not_established", + "evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a" + } + ], + "corporate_action_coverage": [ + { + "instrument_id": "rhinstrument:11111111111111111111111111111111", + "effective_time": { + "start_inclusive": "2018-01-02T07:00:00Z", + "end_inclusive": "2018-01-02T07:00:00Z" + }, + "observed_by": "2026-09-08T01:00:00Z", + "status": "validated", + "evidence_digests": [ + "sha256:233cc2968985e953e59e408b6619a837a792ae17f8afba32261f861de8fe9c7c" + ] + }, + { + "instrument_id": "rhinstrument:22222222222222222222222222222222", + "effective_time": { + "start_inclusive": "2018-01-02T07:00:00Z", + "end_inclusive": "2018-01-02T07:00:00Z" + }, + "observed_by": "2026-09-08T01:00:00Z", + "status": "validated", + "evidence_digests": [ + "sha256:21f2594fc42e46168a65d7cac7e9b52aeabf24ecc5fb409530c07ad56ed3986c" + ] + } + ], + "readiness": { + "evidence_scope": "synthetic_fixture", + "contract_validation": { + "status": "validated", + "evidence_digests": [ + "sha256:62f2abbbb79f4ff9164e10cc4b21df37f0964224a4c546f978b97794241ff5b3" + ] + }, + "real_data_validation": { + "status": "not_validated", + "evidence_digests": [] + }, + "production_validation": { + "status": "not_validated", + "evidence_digests": [] + }, + "live_validation": { + "status": "not_validated", + "evidence_digests": [] + } + }, + "foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260" +} diff --git a/tests/fixtures/retrospective-dataset-snapshot-v2.golden.json b/tests/fixtures/retrospective-dataset-snapshot-v2.golden.json new file mode 100644 index 0000000..9c18387 --- /dev/null +++ b/tests/fixtures/retrospective-dataset-snapshot-v2.golden.json @@ -0,0 +1,120 @@ +{ + "contract_name": "researchhub.dataset-snapshot", + "schema_version": "2.0.0", + "evidence_scope": "synthetic_fixture", + "descriptor": { + "dataset": { + "dataset_id": "rhdataset:market:44444444444444444444444444444444", + "dataset_kind": "market", + "record_schema_version": "2.0.0", + "dimensions": [ + "instrument_id", + "effective_time" + ] + }, + "published_at": "2026-09-08T01:03:00Z", + "time_semantics": { + "effective_time": { + "start_inclusive": "2018-01-02T07:00:00Z", + "end_inclusive": "2018-01-02T07:00:00Z" + }, + "observation_cutoff": "2026-09-08T01:01:00Z", + "earliest_external_knowledge": { + "status": "unknown" + }, + "historical_availability": "not_established" + }, + "content": { + "digest_algorithm": "sha256", + "canonicalization": "RFC8785", + "record_order": "canonical-record-byte-order", + "content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "logical_manifest": { + "record_count": 2, + "chunks": [ + { + "chunk_index": 0, + "content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "record_count": 2 + } + ] + }, + "manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7", + "record_count": 2 + }, + "observation_manifest": { + "batches": [ + { + "chunk_index": 0, + "content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a", + "record_count": 2, + "observation_kind": "observed_by", + "observed_by": "2026-09-08T01:00:00Z", + "evidence_digest": "sha256:6f35f19a2ede0610d461b458a0f2cbc96f40cee6e90fa6592fd66daf617d3ec0" + } + ] + }, + "lineage": { + "publisher": { + "id": "researchhub.data", + "version": "2.0.0" + }, + "transformation": { + "id": "rhtransform:55555555555555555555555555555555", + "version": "2.0.0" + }, + "upstream_snapshot_ids": [], + "upstream_content_digests": [] + }, + "quality": { + "status": "passed", + "checks": [ + { + "check_id": "completeness", + "status": "passed", + "severity": "blocking", + "evidence_digest": "sha256:519b01c60ad85c2b03d6c4f29027fe9b3672a8f6ae3ca8a2d831b281a61c5095" + }, + { + "check_id": "duplicate_identity", + "status": "passed", + "severity": "blocking", + "evidence_digest": "sha256:7c585c7471ab45886fce037061b5ab0d5f981aede5d5190021963676b0ec0fcc" + }, + { + "check_id": "observation_coverage", + "status": "passed", + "severity": "blocking", + "evidence_digest": "sha256:dde5693c3ad79a01a162f9fc5a40c69d3c284965e283c91c5cd456904e0ef737" + }, + { + "check_id": "historical_claim_policy", + "status": "passed", + "severity": "blocking", + "evidence_digest": "sha256:f4cf8da48f92e03d75bb28b2569061e649f382ba360d879fef8e41e35169c228" + }, + { + "check_id": "range_validity", + "status": "passed", + "severity": "blocking", + "evidence_digest": "sha256:6af08355c5e29d846e1c48783dccde791e8a3f48f8416149413217b7c2cbb52a" + }, + { + "check_id": "schema_conformance", + "status": "passed", + "severity": "blocking", + "evidence_digest": "sha256:fefdc5b81eba128af92f15a60feb7bb6f40b2e7277bf2beb48c1bb9399ee564b" + } + ] + }, + "qualification": { + "status": "qualified", + "usage": "retrospective_research", + "policy_id": "researchhub.dataset-snapshot.retrospective", + "policy_version": "2.0.0", + "evaluated_at": "2026-09-08T01:02:00Z", + "evidence_digest": "sha256:70c3ddcadf095877f17a1365c8d75a2e5a7078a27722fb8b698e61d350c6bb2d" + } + }, + "snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905" +} diff --git a/tests/governance/test_module_spec.py b/tests/governance/test_module_spec.py index 77b0188..6bc3060 100644 --- a/tests/governance/test_module_spec.py +++ b/tests/governance/test_module_spec.py @@ -16,7 +16,7 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower() for term in ("investment advice", "live order", "credentials", "source facts"): assert term in prohibited - assert spec["authority"]["revision"] == 5 + assert spec["authority"]["revision"] == 6 assert { (item["contract_id"], item["version"]) for item in spec["contracts"]["provides"] @@ -28,6 +28,13 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: ("researchhub.performance-evidence", "1.0.0"), ("researchhub.portfolio-decision", "1.0.0"), ("researchhub.risk-assessment", "1.0.0"), + ("researchhub.factor-set-ref", "2.0.0"), + ("researchhub.backtest-run-ref", "2.0.0"), + ("researchhub.backtest-evidence-manifest", "2.0.0"), + ("researchhub.performance-evidence", "2.0.0"), + ("researchhub.portfolio-target", "2.0.0"), + ("researchhub.portfolio-decision", "2.0.0"), + ("researchhub.risk-assessment", "2.0.0"), } expected_paths = { "researchhub.factor-definition": "src/quant_engine/factor_contracts.py", @@ -41,13 +48,28 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"]) assert { item["contract_id"]: item["path"] for item in spec["contracts"]["provides"] + if item["version"] == "1.0.0" } == expected_paths + assert { + item["contract_id"]: item["path"] for item in spec["contracts"]["provides"] + if item["version"] == "2.0.0" + } == { + "researchhub.factor-set-ref": "src/quant_engine/retrospective_factor_contracts.py", + "researchhub.backtest-run-ref": "src/quant_engine/retrospective_backtest_contracts.py", + "researchhub.backtest-evidence-manifest": "src/quant_engine/retrospective_artifact_contracts.py", + "researchhub.performance-evidence": "src/quant_engine/retrospective_artifact_contracts.py", + "researchhub.portfolio-target": "src/quant_engine/retrospective_portfolio_risk_contracts.py", + "researchhub.portfolio-decision": "src/quant_engine/retrospective_portfolio_risk_contracts.py", + "researchhub.risk-assessment": "src/quant_engine/retrospective_portfolio_risk_contracts.py", + } assert { (item["contract_id"], item["version"]) for item in spec["contracts"]["consumes"] } == { ("researchhub.dataset-snapshot", "1.0.0"), ("researchhub.data-foundation", "1.0.0"), + ("researchhub.dataset-snapshot", "2.0.0"), + ("researchhub.data-foundation", "2.0.0"), } assert all( item["authority"] == "researchhub.data" @@ -65,6 +87,10 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None: summary = portfolio_contract["summary"].lower() for term in ("receipt", "portfolio decisions", "risk assessments", "without"): assert term in summary + retrospective = capabilities["retrospective-computation-contracts"] + assert retrospective["status"] == "operational" + for term in ("two clocks", "retrospective", "no historical-availability", "no execution"): + assert term in retrospective["summary"].lower() for term in ("approval", "maker-checker", "publication", "paper", "live"): assert term in prohibited assert all( diff --git a/tests/test_retrospective_artifact_contracts.py b/tests/test_retrospective_artifact_contracts.py new file mode 100644 index 0000000..cd5cb1f --- /dev/null +++ b/tests/test_retrospective_artifact_contracts.py @@ -0,0 +1,311 @@ +"""Fresh synthetic artifact assembly; no reuse or relabelling of real outputs.""" + +from __future__ import annotations + +import hashlib +import json +from dataclasses import replace +from typing import Any + +import pandas as pd +import pytest + +from quant_engine.artifact import ( + EvidenceQualification, + PerformanceEvidenceError, + ResearchRunArtifact, + build_research_run_artifact, + build_backtest_evidence_manifest, + build_performance_evidence, +) +from quant_engine.execution import ExecutionConfig +from quant_engine.factor_contracts import FactorContractError +from quant_engine.governed_pipeline import BacktestContractError +from quant_engine.research_pipeline import run_factor_backtest_research +from quant_engine.retrospective_artifact_contracts import ( + build_retrospective_backtest_evidence_manifest, + build_retrospective_performance_evidence, + RetrospectiveBacktestEvidenceManifest, + RetrospectivePerformanceEvidence, +) +from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef +from test_retrospective_backtest_contracts import run_arguments +from test_retrospective_data_contracts import identify, replace_at + +CONTRACT_ERRORS = (FactorContractError, BacktestContractError, PerformanceEvidenceError) + + +def synthetic_artifact(run: RetrospectiveBacktestRunRef) -> ResearchRunArtifact: + # Artifact-envelope tests, not an end-to-end proof of factor/source authenticity. + # The existing financial methods receive new, in-memory synthetic matrices. + dates = pd.date_range("2018-01-02", periods=4, freq="B") + scores = pd.DataFrame({"SIM0": [2.0, 0.0], "SIM1": [1.0, 3.0]}, index=dates[:2]) + opens = pd.DataFrame( + {"SIM0": [10.0, 10.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 20.0, 21.0]}, index=dates + ) + closes = pd.DataFrame( + {"SIM0": [10.0, 12.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 18.0, 21.0]}, index=dates + ) + result = run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1000.0, + config=ExecutionConfig( + commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0 + ), + ) + benchmark = pd.Series( + [0.0, 0.01, -0.01, 0.02], index=result.returns.index, name="benchmark_return" + ) + return build_research_run_artifact( + result, + run_id=run.run_id, + strategy_id=run.strategy_id, + strategy_name="Synthetic Top 1", + strategy_version=run.strategy_version, + engine_version="0.1.0", + code_revision=run.code_revision, + data_snapshot_id=run.dataset_snapshot_id, + calendar="CN-A", + timezone="Asia/Shanghai", + started_at=run.evaluation_at, + finished_at=run.computed_at, + parameters={"lag_sessions": 1, "top_k": 1}, + benchmark_id="synthetic.benchmark", + benchmark_returns=benchmark, + ) + + +def test_manifest_closes_all_nine_existing_tables_without_changing_their_schema() -> None: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + manifest = build_retrospective_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ) + wire = manifest.to_dict() + assert wire["schema_version"] == "2.0.0" + assert wire["artifact_schema_version"] == "1.1.0" + assert wire["run_id"] == run.run_id + assert wire["usage"] == "retrospective_research" + assert wire["historical_availability"] == "not_established" + assert wire["execution_validation"] == "not_validated" + assert wire["decision_eligible"] is False + assert manifest.manifest_id.startswith("rhbacktestevidencev2:sha256:") + assert len({table.logical_name for item in manifest.evidence for table in item.tables}) == 9 + + +def test_performance_v2_keeps_existing_metric_methods_and_binds_all_upstreams() -> None: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + manifest = build_retrospective_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ) + evidence = build_retrospective_performance_evidence(artifact, run, manifest) + wire = evidence.to_dict() + assert wire["schema_version"] == "researchhub.performance-evidence.v2" + assert wire["methodology_id"] == "researchhub.quant-performance-methodology.v1" + assert wire["metric_schema_id"] == "researchhub.quant-performance-metrics.v1" + assert wire["research_artifact_schema_version"] == "1.1.0" + assert wire["backtest_run_ref_id"] == run.run_id + assert wire["backtest_evidence_manifest_id"] == manifest.manifest_id + assert wire["historical_availability"] == "not_established" + assert wire["usage"] == "retrospective_research" + assert wire["start_date"] == "2018-01-02" + assert wire["end_date"] == "2018-01-05" + assert wire["artifact_available_at"] == "2026-09-08T01:11:00Z" + assert evidence.run_id == run.run_id + assert evidence.performance_evidence_id.startswith("rhperformancev2:sha256:") + for metric in evidence.metrics: + if metric.value is not None: + assert metric.value == artifact.performance.iloc[0][metric.source_column] + assert ( + RetrospectivePerformanceEvidence.from_json( + evidence.to_json(), artifact=artifact, run_ref=run, evidence_manifest=manifest + ) + == evidence + ) + assert ( + RetrospectiveBacktestEvidenceManifest.from_json( + manifest.to_json(), artifact=artifact, backtest_run_ref=run + ) + == manifest + ) + + +@pytest.mark.parametrize( + ("path", "value"), + [ + ("schema_version", "1.0.0"), + ("run_id", "rhbacktestrunv2:sha256:" + "0" * 64), + ("profile", "offline_research_v1"), + ("historical_availability", "established"), + ("decision_eligible", True), + ("execution_validation", "validated"), + ("evidence_scope", "real_data"), + ("artifact_available_at", "2026-09-08T01:09:00Z"), + ("artifact_schema_version", "2.0.0"), + ("qualification", "legacy_exploratory"), + ("evidence_digest", "sha256:" + "0" * 64), + ("evidence.0.tables.0.row_count", True), + ("evidence.0.tables.0.content_digest", "sha256:" + "0" * 64), + ("run_reference.value.factor_set_id", "rhfactorsetv2:sha256:" + "0" * 64), + ], +) +def test_manifest_rejects_reidentified_claims_without_actual_table_closure( + path: str, value: Any +) -> None: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + row = build_retrospective_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ).to_dict() + replace_at(row, path, value) + identify(row, "manifest_id", "rhbacktestevidencev2:") + with pytest.raises(CONTRACT_ERRORS): + RetrospectiveBacktestEvidenceManifest.from_dict( + row, artifact=artifact, backtest_run_ref=run + ) + + +@pytest.mark.parametrize( + ("table", "column", "value"), + [ + ("run", "data_snapshot_id", "rhdsv2:sha256:" + "0" * 64), + ("run", "config_hash", "0" * 64), + ("run", "code_revision", "0" * 40), + ("run", "started_at", "2018-01-02T07:00:00Z"), + ("run", "finished_at", "2026-09-08T01:12:00Z"), + ("signals", "asset_id", "/private/data.csv"), + ("nav", "run_id", "old.run"), + ("performance", "run_id", "old.run"), + ], +) +def test_manifest_rejects_artifact_identity_time_or_private_data_mismatch( + table: str, column: str, value: Any +) -> None: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + frame = getattr(artifact, table) + frame.loc[frame.index[0], column] = value + forged = replace(artifact, **{"_" + table: frame}) + with pytest.raises(CONTRACT_ERRORS): + build_retrospective_backtest_evidence_manifest( + run, forged, artifact_available_at="2026-09-08T01:13:00Z" + ) + + +def test_each_table_is_reconciled_and_legacy_apis_cannot_accept_v2() -> None: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + manifest = build_retrospective_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ) + for item in manifest.evidence: + for table in item.tables: + with pytest.raises(CONTRACT_ERRORS): + build_retrospective_backtest_evidence_manifest( + run, + artifact, + artifact_available_at="2026-09-08T01:11:00Z", + expected_table_digests={table.logical_name: "sha256:" + "0" * 64}, + ) + with pytest.raises(CONTRACT_ERRORS): + build_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ) + with pytest.raises(CONTRACT_ERRORS): + build_performance_evidence(artifact, run, manifest) + with pytest.raises(CONTRACT_ERRORS): + build_retrospective_backtest_evidence_manifest( + run, + artifact, + artifact_available_at="2026-09-08T01:11:00Z", + qualification=EvidenceQualification.LEGACY_EXPLORATORY, + ) + + +def seal_performance(row: dict[str, Any]) -> None: + def sha(document: Any) -> str: + return ( + "sha256:" + + hashlib.sha256( + json.dumps( + document, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ).encode() + ).hexdigest() + ) + + row.pop("document_sha256", None) + row.pop("performance_evidence_id", None) + row["performance_evidence_id"] = "rhperformancev2:" + sha(row) + row["document_sha256"] = sha(row) + + +@pytest.mark.parametrize( + ("path", "value"), + [ + ("schema_version", "researchhub.performance-evidence.v1"), + ("scope", "live"), + ("historical_availability", "established"), + ("decision_eligible", True), + ("evidence_scope", "real_data"), + ("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64), + ("backtest_run_ref_document_sha256", "sha256:" + "0" * 64), + ("backtest_evidence_manifest_id", "rhbacktestevidencev2:sha256:" + "0" * 64), + ("methodology.periods_per_year", 365), + ("metric_schema_id", "new.metric"), + ("metrics.0.value", 0.0), + ("metrics.0.nullable", True), + ("start_date", "2017-01-01"), + ("artifact_available_at", "2018-01-02T07:00:00Z"), + ], +) +def test_performance_never_accepts_reidentified_changed_facts(path: str, value: Any) -> None: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + manifest = build_retrospective_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ) + row = build_retrospective_performance_evidence(artifact, run, manifest).to_dict() + replace_at(row, path, value) + seal_performance(row) + with pytest.raises(CONTRACT_ERRORS): + RetrospectivePerformanceEvidence.from_dict( + row, artifact=artifact, run_ref=run, evidence_manifest=manifest + ) + + +def test_performance_has_immutable_finite_canonical_payload_and_current_tables() -> None: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + manifest = build_retrospective_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ) + evidence = build_retrospective_performance_evidence(artifact, run, manifest) + assert evidence.document_sha256.startswith("sha256:") + exported = evidence.to_dict() + exported["metrics"][0]["value"] = 9.0 + assert evidence.to_dict()["metrics"][0]["value"] != 9.0 + for data in ( + evidence.to_json() + "\n", + '{"schema_version":"x",' + evidence.to_json()[1:], + "null", + "{bad", + ): + with pytest.raises(CONTRACT_ERRORS): + RetrospectivePerformanceEvidence.from_json( + data, artifact=artifact, run_ref=run, evidence_manifest=manifest + ) + frame = artifact.performance + frame.loc[0, "total_ret"] = 0.0 + forged = replace(artifact, _performance=frame) + with pytest.raises(CONTRACT_ERRORS): + build_retrospective_performance_evidence(forged, run, manifest) diff --git a/tests/test_retrospective_backtest_contracts.py b/tests/test_retrospective_backtest_contracts.py new file mode 100644 index 0000000..0317325 --- /dev/null +++ b/tests/test_retrospective_backtest_contracts.py @@ -0,0 +1,157 @@ +"""Offline synthetic v2 backtest evidence and replay boundaries.""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from quant_engine.factor_contracts import FactorContractError, PayloadValidation +from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef +from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef +from test_retrospective_data_contracts import digest, identify, replace_at +from test_retrospective_factor_contracts import factor_arguments, decoding_arguments + + +def run_arguments() -> dict[str, Any]: + arguments = factor_arguments() + factor = RetrospectiveFactorSetRef.create(**arguments) + view = next(iter(arguments["foundation"].views.values())) + return { + "dataset_snapshot": arguments["dataset_snapshot"], + "foundation": arguments["foundation"], + "factor_set": factor, + "universe_digest": digest({"synthetic_universe": 2}), + "trading_calendar_revision_ids": view.trading_calendar_revision_ids, + "corporate_action_revision_ids": view.corporate_action_revision_ids, + "strategy_id": "synthetic.top1", + "strategy_version": "1.0.0", + "strategy_digest": digest({"synthetic_strategy": "top1"}), + "execution_model_version": "1.0.0", + "execution_model_digest": digest({"synthetic_execution": 1}), + "cost_model_version": "1.0.0", + "cost_model_digest": digest({"synthetic_cost": 1}), + "random_seed": 7, + "code_revision": "d" * 40, + "environment_lock_digest": digest({"synthetic_lock": 1}), + "configuration_digest": digest({"lag_sessions": 1, "top_k": 1}), + "evaluation_at": "2026-09-08T01:09:00Z", + "computed_at": "2026-09-08T01:10:00Z", + } + + +def run_context(arguments: dict[str, Any]) -> dict[str, Any]: + return {key: arguments[key] for key in ("dataset_snapshot", "foundation", "factor_set")} + + +def test_run_identity_closes_observation_inputs_and_preserves_configurations() -> None: + arguments = run_arguments() + run = RetrospectiveBacktestRunRef.create(**arguments) + document = run.to_dict() + assert document["schema_version"] == "2.0.0" + assert run.run_id.startswith("rhbacktestrunv2:sha256:") + assert run.dataset_snapshot_id == arguments["dataset_snapshot"].snapshot_id + assert run.foundation_id == arguments["foundation"].foundation_id + assert run.factor_set_id == arguments["factor_set"].factor_set_id + assert document["usage"] == "retrospective_research" + assert document["historical_availability"] == "not_established" + assert document["decision_eligible"] is False + assert document["execution_validation"] == "not_validated" + assert document["observation_cutoff"] == arguments["foundation"].observation_cutoff + assert document["replay_attempt"] == 0 + assert RetrospectiveBacktestRunRef.from_json(run.to_json(), **run_context(arguments)) == run + + +@pytest.mark.parametrize( + ("path", "value"), + [ + ("schema_version", "1.0.0"), + ("schema_version", "2.1.0"), + ("historical_availability", "established"), + ("usage", "as_available"), + ("execution_validation", "validated"), + ("decision_eligible", True), + ("decision_eligible", 0), + ("dataset_content_digest", "sha256:" + "0" * 64), + ("foundation_digest", "sha256:" + "0" * 64), + ("factor_set_digest", "sha256:" + "0" * 64), + ("factor_output_content_digest", "sha256:" + "0" * 64), + ("observation_cutoff", "2018-01-02T07:00:00Z"), + ("evidence_scope", "real_data"), + ("trading_calendar_revision_ids", []), + ("corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]), + ("evaluation_at", "2026-09-08T01:07:00Z"), + ("computed_at", "2026-09-08T01:08:00Z"), + ("computed_at", "2026-09-08T01:10:00.0000001Z"), + ("random_seed", True), + ("strategy_version", "latest"), + ("configuration_digest", "../private/a"), + ("code_revision", "unknown"), + ("replay_attempt", 1), + ("replay_reason", "retry"), + ("replay_spec_digest", "sha256:" + "0" * 64), + ("replay_ancestor_run_ids", ["rhbacktestrunv2:sha256:" + "0" * 64]), + ], +) +def test_reidentified_run_must_match_exact_context(path: str, value: Any) -> None: + arguments = run_arguments() + row = RetrospectiveBacktestRunRef.create(**arguments).to_dict() + replace_at(row, path, value) + identify(row, "run_id", "rhbacktestrunv2:") + with pytest.raises(FactorContractError): + RetrospectiveBacktestRunRef.from_dict(row, **run_context(arguments)) + + +def test_replay_keeps_input_spec_but_requires_new_actual_attempt_times() -> None: + arguments = run_arguments() + root = RetrospectiveBacktestRunRef.create(**arguments) + replay_args = { + **arguments, + "parent": root, + "replay_reason": "synthetic.retry", + "replay_attempt": 1, + "evaluation_at": "2026-09-08T01:12:00Z", + "computed_at": "2026-09-08T01:13:00Z", + } + replay = RetrospectiveBacktestRunRef.create(**replay_args) + assert replay.replay_spec_digest == root.replay_spec_digest + assert replay.run_id != root.run_id + assert replay.replay_ancestor_run_ids == (root.run_id,) + assert replay.evaluation_at != root.evaluation_at + assert ( + RetrospectiveBacktestRunRef.from_json( + replay.to_json(), **run_context(arguments), parent=root + ) + == replay + ) + for changes in ( + {"random_seed": 9}, + {"configuration_digest": digest({"different_configuration": 1})}, + {"evaluation_at": root.evaluation_at}, + {"replay_attempt": 2}, + {"replay_reason": None}, + {"parent": None}, + ): + with pytest.raises(FactorContractError): + RetrospectiveBacktestRunRef.create(**{**replay_args, **changes}) + + +def test_reference_only_factors_can_be_read_but_not_used_to_create_new_runs() -> None: + arguments = run_arguments() + run = RetrospectiveBacktestRunRef.create(**arguments) + factor = arguments["factor_set"] + reference = RetrospectiveFactorSetRef.from_dict( + factor.to_dict(), + definitions=factor._definitions, + dataset_snapshot=arguments["dataset_snapshot"], + foundation=arguments["foundation"], + ) + with pytest.raises(FactorContractError): + RetrospectiveBacktestRunRef.create(**{**arguments, "factor_set": reference}) + restored = RetrospectiveBacktestRunRef.from_dict( + run.to_dict(), **{**run_context(arguments), "factor_set": reference} + ) + assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY + with pytest.raises(FactorContractError): + restored.require_inputs_revalidated() + run.require_inputs_revalidated() diff --git a/tests/test_retrospective_computation_v2_golden.py b/tests/test_retrospective_computation_v2_golden.py new file mode 100644 index 0000000..4acaa76 --- /dev/null +++ b/tests/test_retrospective_computation_v2_golden.py @@ -0,0 +1,88 @@ +"""Frozen synthetic interoperability vector, not end-to-end source provenance.""" + +from __future__ import annotations + +import ast +import json +from pathlib import Path +from typing import Any + +from quant_engine.artifact import _evidence_frame_records +from quant_engine.retrospective_artifact_contracts import build_retrospective_performance_evidence +from quant_engine.retrospective_portfolio_risk_contracts import ( + assess_retrospective_portfolio_risk, +) +from test_retrospective_factor_contracts import factor_arguments +from test_retrospective_portfolio_risk_contracts import portfolio_arguments, risk_arguments + +ROOT = Path(__file__).resolve().parents[1] +VECTOR = ROOT / "tests" / "fixtures" / "retrospective-computation-v2.golden.json" + + +def build_vector() -> dict[str, Any]: + portfolio = portfolio_arguments() + risk = risk_arguments(portfolio) + run = portfolio["backtest_run_ref"] + manifest = portfolio["manifest"] + artifact = manifest._artifact + factor = factor_arguments() + return { + "fixture_kind": "synthetic_retrospective_contract_vector", + "artifact_data_provenance": "envelope_test_only_not_end_to_end", + "source_authenticity": "not_established", + "factor_definitions": [item.to_dict() for item in factor["definitions"]], + "dataset_chunks": factor["dataset_chunks"], + "resolved_view_schema": {"synthetic_schema": "neutral_close_v2"}, + "factor_output_schema": json.loads(factor["output_schema_bytes"]), + "factor_output_records": json.loads(factor["output_content_bytes"]), + "factor_set": run._factor_set.to_dict(), + "backtest_run_ref": run.to_dict(), + "artifact_tables": { + name: _evidence_frame_records(frame, name) + for name, frame in artifact.table_frames().items() + }, + "backtest_evidence_manifest": manifest.to_dict(), + "performance_evidence": build_retrospective_performance_evidence( + artifact, run, manifest + ).to_dict(), + "portfolio_target": portfolio["target"].to_dict(), + "portfolio_decision": risk["portfolio_decision"].to_dict(), + "risk_assessment": assess_retrospective_portfolio_risk(**risk).to_dict(), + "covariance_matrix": risk["covariance"].covariance.to_dict(), + } + + +def test_synthetic_vector_matches_all_current_owner_serializers() -> None: + expected = VECTOR.read_text(encoding="utf-8") + actual = ( + json.dumps(build_vector(), ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False) + + "\n" + ) + assert actual == expected + + +def test_v2_modules_do_not_import_data_owners_publishers_or_execution_authority() -> None: + modules = sorted((ROOT / "src" / "quant_engine").glob("retrospective_*_contracts.py")) + assert len(modules) == 5 + for path in modules: + tree = ast.parse(path.read_text(encoding="utf-8")) + imports = { + alias.name + for node in ast.walk(tree) + if isinstance(node, ast.Import) + for alias in node.names + } | {node.module or "" for node in ast.walk(tree) if isinstance(node, ast.ImportFrom)} + assert not any( + name.startswith(("research_results", "research_platform", "edb_data_core")) + for name in imports + ) + called = { + node.func.id + for node in ast.walk(tree) + if isinstance(node, ast.Call) and isinstance(node.func, ast.Name) + } + assert not called & { + "create_paper_order_intent", + "run_governed_factor_slice", + "evaluate_portfolio_risk", + } diff --git a/tests/test_retrospective_data_contracts.py b/tests/test_retrospective_data_contracts.py new file mode 100644 index 0000000..a978f29 --- /dev/null +++ b/tests/test_retrospective_data_contracts.py @@ -0,0 +1,636 @@ +"""QE-owned decoding tests against RP's synthetic public v2 golden vectors.""" + +from __future__ import annotations + +import hashlib +import json +from copy import deepcopy +from dataclasses import FrozenInstanceError +from pathlib import Path +from typing import Any + +import pytest + +from quant_engine.factor_contracts import ( + DataFoundationEnvelope, + DatasetSnapshotEnvelope, + FactorContractError, + canonical_json_bytes, +) +from quant_engine.retrospective_data_contracts import ( + RetrospectiveFoundationEnvelope, + RetrospectiveSnapshotEnvelope, +) + +FIXTURES = Path(__file__).parent / "fixtures" + + +def golden(kind: str) -> dict[str, Any]: + return json.loads((FIXTURES / f"retrospective-{kind}-v2.golden.json").read_text()) + + +def digest(value: Any) -> str: + return "sha256:" + hashlib.sha256(canonical_json_bytes(value)).hexdigest() + + +def identify(value: dict[str, Any], field: str, prefix: str) -> None: + value[field] = prefix + digest({key: item for key, item in value.items() if key != field}) + + +def records() -> list[dict[str, Any]]: + return [ + { + "effective_time": "2018-01-02T07:00:00Z", + "instrument_id": "rhinstrument:" + "1" * 32, + "metric": "close", + "value": "101.25", + }, + { + "effective_time": "2018-01-02T07:00:00Z", + "instrument_id": "rhinstrument:" + "2" * 32, + "metric": "close", + "value": "87.50", + }, + ] + + +def bind_records(source: dict[str, Any], chunks: list[list[dict[str, Any]]]) -> None: + def records_digest(rows: list[dict[str, Any]]) -> str: + data = b"[" + b",".join(sorted(canonical_json_bytes(row) for row in rows)) + b"]" + return "sha256:" + hashlib.sha256(data).hexdigest() + + manifest = { + "record_count": sum(len(rows) for rows in chunks), + "chunks": [ + { + "chunk_index": index, + "content_digest": records_digest(rows), + "record_count": len(rows), + } + for index, rows in enumerate(chunks) + ], + } + source["descriptor"]["content"].update( + { + "record_count": manifest["record_count"], + "logical_manifest": manifest, + "manifest_digest": digest(manifest), + "content_digest": records_digest([row for chunk in chunks for row in chunk]), + } + ) + source["descriptor"]["observation_manifest"]["batches"] = [ + { + **chunk, + "observation_kind": "observed_by", + "observed_by": "2026-09-08T01:00:00Z", + "evidence_digest": digest({"synthetic_receipt": index}), + } + for index, chunk in enumerate(manifest["chunks"]) + ] + identify(source, "snapshot_id", "rhdsv2:") + + +def replace_at(source: dict[str, Any], path: str, value: Any) -> None: + target: Any = source + keys = path.split(".") + for key in keys[:-1]: + target = target[int(key)] if isinstance(target, list) else target[key] + target[int(keys[-1]) if isinstance(target, list) else keys[-1]] = value + + +COLLECTIONS = ( + ( + "instrument_routes", + "route_revision_id", + "rhroutev2:", + "instrument_route", + "instrument_route_revision_ids", + ), + ( + "trading_calendar_revisions", + "calendar_revision_id", + "rhcalv2:", + "trading_calendar", + "trading_calendar_revision_ids", + ), + ( + "corporate_action_revisions", + "action_revision_id", + "rhcav2:", + "corporate_action", + "corporate_action_revision_ids", + ), +) + + +def seal_foundation(source: dict[str, Any], *, rebuild_lineage: bool = True) -> None: + lineage = [] + for name, key, prefix, kind, view_key in COLLECTIONS: + replacements = {} + for row in sorted(source[name], key=lambda row: row["observation_sequence"]): + old = row[key] + if "supersedes_observation_id" in row: + row["supersedes_observation_id"] = replacements.get( + row["supersedes_observation_id"], row["supersedes_observation_id"] + ) + identify(row, key, prefix) + replacements[old] = row[key] + lineage.append( + { + "revision_kind": kind, + "revision_id": row[key], + **{ + field: row[field] + for field in ( + "observation_sequence", + "observed_by", + "earliest_external_knowledge", + "history_completeness", + "evidence_digest", + "supersedes_observation_id", + ) + if field in row + }, + } + ) + for view in source["standardized_views"]: + view[view_key] = [replacements.get(item, item) for item in view[view_key]] + if rebuild_lineage: + source["observation_lineage"] = lineage + for view in source["standardized_views"]: + identify(view, "view_ref_id", "rhviewrefv2:") + identify(source, "foundation_id", "rhdfv2:") + + +def parse_foundation(source: dict[str, Any]) -> RetrospectiveFoundationEnvelope: + return RetrospectiveFoundationEnvelope.from_dict( + source, snapshot=RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot")) + ) + + +def test_public_snapshot_golden_is_an_explicit_observation_contract() -> None: + source = golden("dataset-snapshot") + snapshot = RetrospectiveSnapshotEnvelope.from_dict(source) + assert snapshot.to_dict() == source + assert snapshot.snapshot_id == source["snapshot_id"] + assert snapshot.observation_cutoff == "2026-09-08T01:01:00Z" + assert snapshot.evidence_scope == "synthetic_fixture" + assert snapshot.earliest_external_knowledge == {"status": "unknown"} + assert not hasattr(snapshot, "pit_cutoff") + assert not hasattr(snapshot, "knowledge_time") + snapshot.require_qualified() + assert RetrospectiveSnapshotEnvelope.from_json(snapshot.to_json()) == snapshot + + +def test_public_foundation_golden_binds_exact_typed_snapshot() -> None: + source = golden("data-foundation") + snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot")) + foundation = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot) + assert foundation.to_dict() == source + assert foundation.dataset_snapshot_id == snapshot.snapshot_id + assert foundation.observation_cutoff == snapshot.observation_cutoff + assert foundation.evidence_scope == snapshot.evidence_scope + assert foundation.real_data_validation_status == "not_validated" + assert not hasattr(foundation, "pit_cutoff") + assert ( + RetrospectiveFoundationEnvelope.from_json(foundation.to_json(), snapshot=snapshot) + == foundation + ) + + +def test_empty_logical_dimension_is_rejected_after_rebinding_all_bytes() -> None: + source = golden("dataset-snapshot") + rows = records() + rows[0]["instrument_id"] = "" + bind_records(source, [rows]) + snapshot = RetrospectiveSnapshotEnvelope.from_dict(source) + with pytest.raises(FactorContractError, match="dimension"): + snapshot.verify_materialized_records([rows]) + + +def test_boolean_lineage_sequence_does_not_equal_integer_one() -> None: + source = golden("data-foundation") + source["observation_lineage"][0]["observation_sequence"] = True + identify(source, "foundation_id", "rhdfv2:") + with pytest.raises(FactorContractError): + parse_foundation(source) + + +@pytest.mark.parametrize( + ("path", "value"), + [ + ("schema_version", "1.0.0"), + ("schema_version", "2.1.0"), + ("descriptor.time_semantics.pit_cutoff", "2018-01-02T07:00:00Z"), + ("descriptor.time_semantics.knowledge_time", {"start_inclusive": "2018-01-02T07:00:00Z"}), + ("descriptor.time_semantics.historical_availability", "established"), + ("descriptor.time_semantics.observation_cutoff", "2018-01-02T07:00:00Z"), + ("descriptor.time_semantics.observation_cutoff", "2026-09-08T01:01:00.0000001Z"), + ("descriptor.time_semantics.observation_cutoff", "2026-09-08T09:01:00+08:00"), + ( + "descriptor.time_semantics.earliest_external_knowledge", + {"status": "unknown", "evidence_digest": "sha256:" + "0" * 64}, + ), + ("descriptor.time_semantics.earliest_external_knowledge", {"status": "evidenced"}), + ("descriptor.published_at", "2026-02-30T00:00:00Z"), + ("descriptor.published_at", "2026-09-08T01:00:00Z"), + ("descriptor.qualification.evaluated_at", "2018-01-02T07:00:00Z"), + ("descriptor.qualification.usage", "as_available"), + ("descriptor.qualification.policy_version", "1.0.0"), + ("descriptor.quality.checks.0.severity", "advisory"), + ("descriptor.quality.checks.0.status", "failed"), + ("descriptor.quality.checks.0.check_id", "schema_conformance"), + ("descriptor.quality.status", "failed"), + ("descriptor.content.record_count", True), + ("descriptor.content.record_count", 9007199254740992), + ("descriptor.content.record_count", 2.0), + ("descriptor.content.content_digest", "bad"), + ("descriptor.content.manifest_digest", "sha256:" + "0" * 64), + ("descriptor.observation_manifest.batches", []), + ("descriptor.observation_manifest.batches.0.record_count", 1), + ("descriptor.observation_manifest.batches.0.chunk_index", True), + ("descriptor.observation_manifest.batches.0.observed_by", "2026-09-08T01:02:00Z"), + ("descriptor.observation_manifest.batches.0.observation_kind", "first_published_at"), + ("descriptor.lineage.transformation.id", "rhtransform:private"), + ("descriptor.dataset.dimensions", ["instrument_id", "knowledge_time"]), + ("descriptor.dataset.dataset_id", "rhdataset:macroeconomic:" + "4" * 32), + ], +) +def test_snapshot_rejects_reidentified_invalid_declarations(path: str, value: Any) -> None: + source = golden("dataset-snapshot") + replace_at(source, path, value) + # Noncanonical numbers are rejected before identity formation. + if type(value) is not float and value != 9007199254740992: + identify(source, "snapshot_id", "rhdsv2:") + with pytest.raises(FactorContractError): + RetrospectiveSnapshotEnvelope.from_dict(source) + + +def test_snapshot_materialized_records_bind_the_public_golden_and_chunks() -> None: + source = golden("dataset-snapshot") + snapshot = RetrospectiveSnapshotEnvelope.from_dict(source) + snapshot.verify_materialized_records([records()]) + snapshot.verify_materialized_records([list(reversed(records()))]) + chunks = [[records()[0]], [records()[1]]] + bind_records(source, chunks) + RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records(chunks) + with pytest.raises(FactorContractError): + snapshot.verify_materialized_records(chunks) + rows = records() + rows[0]["value"] = "0" + with pytest.raises(FactorContractError): + snapshot.verify_materialized_records([rows]) + + +@pytest.mark.parametrize( + "mutation", ["duplicate", "legacy", "range", "location", "missing_effective"] +) +def test_materialized_bad_records_rejected_even_with_matching_digests(mutation: str) -> None: + source = golden("dataset-snapshot") + rows = records() + if mutation == "duplicate": + rows.append(deepcopy(rows[0])) + elif mutation == "legacy": + rows[0]["knowledge_time"] = "2026-09-08T01:00:00Z" + elif mutation == "range": + rows[0]["effective_time"] = "2018-01-01T07:00:00Z" + elif mutation == "location": + rows[0]["value"] = "/private/records.csv" + else: + del rows[0]["effective_time"] + bind_records(source, [rows]) + with pytest.raises(FactorContractError): + RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows]) + + +def test_unknown_or_evidenced_knowledge_never_changes_usage() -> None: + source = golden("dataset-snapshot") + source["descriptor"]["time_semantics"]["earliest_external_knowledge"] = { + "status": "evidenced", + "range": { + "start_inclusive": "2018-01-02T07:00:00Z", + "end_inclusive": "2018-01-02T07:00:00Z", + }, + "evidence_digest": digest({"synthetic_earliest": True}), + } + identify(source, "snapshot_id", "rhdsv2:") + parsed = RetrospectiveSnapshotEnvelope.from_dict(source) + assert ( + parsed.to_dict()["descriptor"]["time_semantics"]["historical_availability"] + == "not_established" + ) + source["descriptor"]["time_semantics"]["earliest_external_knowledge"]["range"][ + "end_inclusive" + ] = "2026-09-08T01:01:00Z" + identify(source, "snapshot_id", "rhdsv2:") + with pytest.raises(FactorContractError): + RetrospectiveSnapshotEnvelope.from_dict(source) + + +def test_rejected_snapshot_remains_readable_but_cannot_support_foundation() -> None: + source = golden("dataset-snapshot") + source["descriptor"]["qualification"]["status"] = "rejected" + identify(source, "snapshot_id", "rhdsv2:") + snapshot = RetrospectiveSnapshotEnvelope.from_dict(source) + with pytest.raises(FactorContractError): + snapshot.require_qualified() + foundation = golden("data-foundation") + foundation["dataset_snapshot_id"] = snapshot.snapshot_id + for view in foundation["standardized_views"]: + view["dataset_snapshot_id"] = snapshot.snapshot_id + seal_foundation(foundation) + with pytest.raises(FactorContractError): + RetrospectiveFoundationEnvelope.from_dict(foundation, snapshot=snapshot) + + +@pytest.mark.parametrize( + ("path", "value"), + [ + ("schema_version", "1.0.0"), + ("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64), + ("observation_cutoff", "2026-09-08T01:00:00Z"), + ("published_at", "2026-09-08T01:03:00Z"), + ("usage", "paper_trading"), + ("historical_availability", "established"), + ("instrument_routes.0.observation_sequence", 2), + ("instrument_routes.0.supersedes_observation_id", "rhroutev2:sha256:" + "0" * 64), + ("instrument_routes.0.observed_by", "2026-09-08T01:02:00Z"), + ("instrument_routes.0.history_completeness", "complete"), + ( + "instrument_routes.0.earliest_external_knowledge", + {"status": "unknown", "earliest_at": "2018-01-01T00:00:00Z"}, + ), + ("instrument_routes.0.instrument_type", "index"), + ("instrument_routes.0.symbol", "WIND.TEST"), + ("instrument_routes.0.symbol", "A" * 33), + ("instrument_routes.0.calendar_id", "rhcalendar:" + "0" * 32), + ("trading_calendar_revisions.0.status", "closed"), + ("trading_calendar_revisions.0.sessions", []), + ("trading_calendar_revisions.0.sessions.0.closes_at", "2018-01-02T00:00:00Z"), + ("trading_calendar_revisions.0.session_date", "2018-02-30"), + ("standardized_views.0.instrument_route_revision_ids", []), + ("standardized_views.0.trading_calendar_revision_ids", []), + ("standardized_views.0.corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]), + ("standardized_views.0.observation_cutoff", "2018-01-02T07:00:00Z"), + ("standardized_views.0.available_at", "2026-09-08T01:02:00Z"), + ("standardized_views.0.available_at", "2026-09-08T01:06:00Z"), + ("standardized_views.0.usage", "as_available"), + ("corporate_action_coverage", []), + ("corporate_action_coverage.0.instrument_id", "rhinstrument:" + "0" * 32), + ("corporate_action_coverage.0.effective_time.end_inclusive", "2018-01-01T07:00:00Z"), + ("corporate_action_coverage.0.effective_time.start_inclusive", "2018-01-02T07:00:01Z"), + ("corporate_action_coverage.0.observed_by", "2026-09-08T01:02:00Z"), + ("corporate_action_coverage.0.evidence_digests", []), + ("readiness.evidence_scope", "real_data"), + ("readiness.contract_validation.evidence_digests", []), + ( + "readiness.real_data_validation", + {"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]}, + ), + ( + "readiness.production_validation", + {"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]}, + ), + ( + "readiness.live_validation", + {"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]}, + ), + ], +) +def test_foundation_rejects_semantic_forgery_after_reidentification(path: str, value: Any) -> None: + source = golden("data-foundation") + replace_at(source, path, value) + seal_foundation(source) + with pytest.raises(FactorContractError): + parse_foundation(source) + + +def test_v1_and_v2_never_coerce_each_other() -> None: + with pytest.raises(FactorContractError): + DatasetSnapshotEnvelope.from_dict(golden("dataset-snapshot")) + with pytest.raises(FactorContractError): + DataFoundationEnvelope.from_dict(golden("data-foundation")) + old = json.loads((FIXTURES / "factor-contracts-v1.golden.json").read_text()) + with pytest.raises(FactorContractError): + RetrospectiveSnapshotEnvelope.from_dict(old["dataset_snapshot"]) + with pytest.raises(FactorContractError): + parse_foundation(old["data_foundation"]) + with pytest.raises(FactorContractError): + RetrospectiveFoundationEnvelope.from_dict( + golden("data-foundation"), + snapshot=DatasetSnapshotEnvelope.from_dict(old["dataset_snapshot"]), + ) + + +def test_deep_immutability_and_strict_canonical_json() -> None: + source = golden("dataset-snapshot") + snapshot = RetrospectiveSnapshotEnvelope.from_dict(source) + source["descriptor"]["quality"]["status"] = "failed" + snapshot.to_dict()["descriptor"]["qualification"]["status"] = "rejected" + snapshot.require_qualified() + with pytest.raises(FrozenInstanceError): + snapshot._payload = {} + with pytest.raises(TypeError): + snapshot.earliest_external_knowledge["status"] = "evidenced" + foundation = parse_foundation(golden("data-foundation")) + with pytest.raises(TypeError): + foundation.views["new"] = next(iter(foundation.views.values())) + for decoder, document in ( + (RetrospectiveSnapshotEnvelope.from_json, golden("dataset-snapshot")), + ( + lambda value: RetrospectiveFoundationEnvelope.from_json(value, snapshot=snapshot), + golden("data-foundation"), + ), + ): + wire = canonical_json_bytes(document) + with pytest.raises(FactorContractError): + decoder(wire + b"\n") + with pytest.raises(FactorContractError): + decoder(b'{"schema_version":"2.0.0",' + wire[1:]) + + +def test_macro_period_is_not_inferred_as_an_effective_instant() -> None: + source = golden("dataset-snapshot") + source["descriptor"]["dataset"].update( + dataset_id="rhdataset:macroeconomic:" + "4" * 32, + dataset_kind="macroeconomic", + dimensions=["series_id", "observation_period"], + ) + rows = [ + { + "series_id": "cpi", + "observation_period": "2018-01", + "effective_time": "2018-01-02T07:00:00Z", + "value": "2.1", + } + ] + bind_records(source, [rows]) + RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows]) + del rows[0]["effective_time"] + bind_records(source, [rows]) + with pytest.raises(FactorContractError): + RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows]) + + +def with_successor() -> dict[str, Any]: + source = golden("data-foundation") + previous = source["instrument_routes"][0] + successor = deepcopy(previous) + successor.update( + observation_sequence=2, + observed_by="2026-09-08T01:00:30Z", + symbol="SIM0B", + supersedes_observation_id=previous["route_revision_id"], + ) + identify(successor, "route_revision_id", "rhroutev2:") + source["instrument_routes"].append(successor) + source["standardized_views"][0]["instrument_route_revision_ids"].append( + successor["route_revision_id"] + ) + seal_foundation(source) + return source + + +def test_retained_successor_requires_parent_time_and_view_ancestry() -> None: + source = with_successor() + parsed = parse_foundation(source) + assert parsed.foundation_id == source["foundation_id"] + assert parsed.published_at.isoformat() == "2026-09-08T01:05:00+00:00" + assert parsed.contract_evidence_digests + assert len(next(iter(parsed.views.values())).instrument_route_revision_ids) == 3 + for mutation in ( + "missing_parent", + "equal_time", + "omitted_ancestor", + "duplicate_sequence", + "wrong_lineage", + ): + forged = deepcopy(source) + if mutation == "missing_parent": + forged["instrument_routes"][-1]["supersedes_observation_id"] = ( + "rhroutev2:sha256:" + "0" * 64 + ) + elif mutation == "equal_time": + forged["instrument_routes"][-1]["observed_by"] = forged["instrument_routes"][0][ + "observed_by" + ] + elif mutation == "omitted_ancestor": + forged["standardized_views"][0]["instrument_route_revision_ids"].remove( + forged["instrument_routes"][0]["route_revision_id"] + ) + elif mutation == "duplicate_sequence": + forged["instrument_routes"][-1]["observation_sequence"] = 1 + else: + forged["observation_lineage"][0]["evidence_digest"] = "sha256:" + "0" * 64 + seal_foundation(forged, rebuild_lineage=mutation != "wrong_lineage") + with pytest.raises(FactorContractError): + parse_foundation(forged) + + +def test_index_and_closed_calendar_are_explicit_non_execution_data() -> None: + source = golden("data-foundation") + source["instrument_routes"][0].update(asset_class="index", instrument_type="index") + source["trading_calendar_revisions"][0].update(status="closed", sessions=[]) + seal_foundation(source) + parsed = parse_foundation(source) + assert parsed.to_dict()["instrument_routes"][0]["instrument_type"] == "index" + assert parsed.to_dict()["readiness"]["live_validation"]["status"] == "not_validated" + + +def test_real_scope_is_still_a_declaration_with_separate_evidence_and_coverage() -> None: + snapshot_source = golden("dataset-snapshot") + snapshot_source["evidence_scope"] = "real_data" + identify(snapshot_source, "snapshot_id", "rhdsv2:") + snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_source) + source = golden("data-foundation") + source["dataset_snapshot_id"] = snapshot.snapshot_id + for view in source["standardized_views"]: + view["dataset_snapshot_id"] = snapshot.snapshot_id + source["readiness"]["evidence_scope"] = "real_data" + source["readiness"]["real_data_validation"] = { + "status": "validated", + "evidence_digests": [digest({"synthetic_real_claim_test": 1})], + } + seal_foundation(source) + parsed = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot) + assert parsed.real_data_validation_status == "validated" # NOT actual real-data evidence. + for mutation in ("coverage", "reuse"): + forged = deepcopy(source) + if mutation == "coverage": + forged["corporate_action_coverage"][0].update( + status="not_validated", evidence_digests=[] + ) + else: + forged["readiness"]["real_data_validation"] = deepcopy( + forged["readiness"]["contract_validation"] + ) + seal_foundation(forged) + with pytest.raises(FactorContractError): + RetrospectiveFoundationEnvelope.from_dict(forged, snapshot=snapshot) + synthetic = golden("data-foundation") + synthetic["corporate_action_coverage"][0].update(status="not_validated", evidence_digests=[]) + seal_foundation(synthetic) + assert parse_foundation(synthetic).real_data_validation_status == "not_validated" + + +def test_action_must_belong_to_view_selected_instrument() -> None: + source = golden("data-foundation") + route = source["instrument_routes"][0] + action = { + "action_id": "rhaction:" + "7" * 32, + "instrument_id": route["instrument_id"], + "observation_sequence": 1, + "observed_by": route["observed_by"], + "earliest_external_knowledge": { + "status": "evidenced", + "earliest_at": "2018-01-01T00:00:00Z", + "evidence_digest": digest({"synthetic_action_earliest": 1}), + }, + "history_completeness": "not_established", + "evidence_digest": digest({"synthetic_action": 1}), + "action_type": "cash_dividend", + "status": "confirmed", + "effective_time": "2018-01-02T07:00:00Z", + "terms_digest": digest({"synthetic_terms": 1}), + } + identify(action, "action_revision_id", "rhcav2:") + source["corporate_action_revisions"] = [action] + source["standardized_views"][0]["corporate_action_revision_ids"] = [ + action["action_revision_id"] + ] + seal_foundation(source) + assert len(parse_foundation(source).to_dict()["corporate_action_revisions"]) == 1 + source["standardized_views"][0]["instrument_route_revision_ids"].remove( + route["route_revision_id"] + ) + seal_foundation(source) + with pytest.raises(FactorContractError): + parse_foundation(source) + + +def test_views_cannot_borrow_calendar_selection_from_each_other() -> None: + source = golden("data-foundation") + calendar = deepcopy(source["trading_calendar_revisions"][0]) + calendar["calendar_id"] = "rhcalendar:" + "9" * 32 + identify(calendar, "calendar_revision_id", "rhcalv2:") + source["trading_calendar_revisions"].append(calendar) + view = deepcopy(source["standardized_views"][0]) + view["view_id"] = "rhview:" + "8" * 32 + view["trading_calendar_revision_ids"] = [calendar["calendar_revision_id"]] + source["standardized_views"].append(view) + seal_foundation(source) + with pytest.raises(FactorContractError): + parse_foundation(source) + + +@pytest.mark.parametrize( + "location", ["/private/a", "./a", "../a", "C:\\data\\a", "\\\\host\\a", "s3://private/a"] +) +def test_materialized_values_cannot_carry_physical_locations(location: str) -> None: + source = golden("dataset-snapshot") + rows = records() + rows[0]["value"] = location + bind_records(source, [rows]) + snapshot = RetrospectiveSnapshotEnvelope.from_dict(source) + with pytest.raises(FactorContractError): + snapshot.verify_materialized_records([rows]) diff --git a/tests/test_retrospective_factor_contracts.py b/tests/test_retrospective_factor_contracts.py new file mode 100644 index 0000000..517591a --- /dev/null +++ b/tests/test_retrospective_factor_contracts.py @@ -0,0 +1,367 @@ +"""Synthetic v2 computation boundaries; never source authentication.""" + +from __future__ import annotations + +from copy import deepcopy +from typing import Any + +import pytest + +from quant_engine.factor_contracts import ( + ActorIdentity, + FactorContractError, + FactorDefinition, + FactorInput, + FactorSetRef, + OutputArtifactRef, + OutputCoverage, + OutputQuality, + OutputQualityCheck, + PayloadValidation, + ProducerIdentity, + canonical_json_bytes, + factor_input_schema_digest, +) +from quant_engine.retrospective_data_contracts import ( + RetrospectiveFoundationEnvelope, + RetrospectiveSnapshotEnvelope, +) +from quant_engine.retrospective_factor_contracts import ( + ResolvedRetrospectiveView, + RetrospectiveCausation, + RetrospectiveFactorSetRef, + RetrospectiveInputBinding, + RetrospectiveViewAvailability, +) +from test_retrospective_data_contracts import ( + digest, + golden, + identify, + records, + replace_at, + seal_foundation, +) + + +def factor_arguments() -> dict[str, Any]: + snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot")) + foundation = RetrospectiveFoundationEnvelope.from_dict( + golden("data-foundation"), snapshot=snapshot + ) + view = next(iter(foundation.views.values())) + factor_inputs = (FactorInput("market", view.schema_digest, ("value",)),) + definition = FactorDefinition.create( + factor_id="neutral_close", + version="1.0.0", + formula="value", + parameters={}, + implementation_digest=digest({"synthetic_formula": "identity"}), + input_schema_digest=factor_input_schema_digest(factor_inputs), + inputs=factor_inputs, + valid_from="2026-01-01T00:00:00Z", + valid_until="2027-01-01T00:00:00Z", + warmup_sessions=0, + lag_sessions=1, + producer=ProducerIdentity("quant_engine", "0.1.0"), + code_revision="c" * 40, + ) + schema = {"fields": ["instrument_id", "value"]} + output = [{"instrument_id": row["instrument_id"], "value": row["value"]} for row in records()] + return { + "definitions": (definition,), + "dataset_snapshot": snapshot, + "foundation": foundation, + "selected_view_ref_ids": (view.view_ref_id,), + "input_bindings": ( + RetrospectiveInputBinding( + definition.definition_id, "market", view.view_ref_id, view.schema_digest + ), + ), + "view_availability": ( + RetrospectiveViewAvailability( + view.view_ref_id, view.available_at, digest({"synthetic_view_receipt": 1}) + ), + ), + "dataset_chunks": [records()], + "resolved_views": ( + ResolvedRetrospectiveView( + view.view_ref_id, + canonical_json_bytes({"synthetic_schema": "neutral_close_v2"}), + canonical_json_bytes(sorted(records(), key=canonical_json_bytes)), + ), + ), + "output_quality": OutputQuality( + "passed", + (OutputQualityCheck("finite_values", "passed", digest({"synthetic_quality": 1})),), + ), + "output_coverage": OutputCoverage( + "complete", 2, 2, "records", "synthetic_market", digest({"synthetic_coverage": 1}) + ), + "output_schema_bytes": canonical_json_bytes(schema), + "output_content_bytes": canonical_json_bytes(output), + "output_artifact_ref": OutputArtifactRef.create( + schema_digest=digest(schema), content_digest=digest(output) + ), + "evaluation_at": "2026-09-08T01:06:00Z", + "computed_at": "2026-09-08T01:07:00Z", + "artifact_available_at": "2026-09-08T01:08:00Z", + "producer": ProducerIdentity("quant_engine", "0.1.0"), + "code_revision": "d" * 40, + "actor": ActorIdentity("service", "synthetic.research"), + "correlation_id": "synthetic.retrospective", + "causation": RetrospectiveCausation("foundation", foundation.foundation_id), + "evidence_scope": "synthetic_fixture", + "decision_eligible": False, + } + + +def decoding_arguments(arguments: dict[str, Any]) -> dict[str, Any]: + return {key: arguments[key] for key in ("definitions", "dataset_snapshot", "foundation")} + + +def test_factor_result_closes_v2_inputs_and_preserves_v1_definition() -> None: + arguments = factor_arguments() + result = RetrospectiveFactorSetRef.create(**arguments) + wire = result.to_dict() + assert result.schema_version == "2.0.0" + assert result.factor_set_id.startswith("rhfactorsetv2:sha256:") + assert result.definition_ids == (arguments["definitions"][0].definition_id,) + assert result.definition_ids[0].startswith("rhfactorv1:") + assert wire["usage"] == "retrospective_research" + assert wire["availability_mode"] == "retrospective_replay" + assert wire["historical_availability"] == "not_established" + assert wire["observation_cutoff"] == arguments["foundation"].observation_cutoff + assert wire["decision_eligible"] is False + assert "pit_cutoff" not in wire + assert result.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED + assert result.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED + restored = RetrospectiveFactorSetRef.from_json( + result.to_json(), **decoding_arguments(arguments) + ) + assert restored.to_dict() == wire + assert restored.payload_validation is PayloadValidation.REFERENCE_ONLY + assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY + + +@pytest.mark.parametrize( + ("path", "value"), + [ + ("schema_version", "1.0.0"), + ("schema_version", "2.1.0"), + ("contract_name", "researchhub.dataset-snapshot"), + ("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64), + ("foundation_id", "rhdfv2:sha256:" + "0" * 64), + ("observation_cutoff", "2018-01-02T07:00:00Z"), + ("pit_cutoff", "2018-01-02T07:00:00Z"), + ("selected_view_ref_ids", []), + ("selected_view_ref_ids", ["rhviewrefv2:sha256:" + "0" * 64]), + ("definition_ids", ["rhfactorv1:sha256:" + "0" * 64]), + ("input_bindings", []), + ("input_bindings.0.input_name", "volume"), + ("input_bindings.0.schema_digest", "sha256:" + "0" * 64), + ("input_bindings.0.view_ref_id", "rhviewrefv2:sha256:" + "0" * 64), + ("view_availability", []), + ("view_availability.0.available_at", "2026-09-08T01:03:30Z"), + ("view_availability.0.available_at", "2026-09-08T01:04:00.0000001Z"), + ("upstream_evidence.quality.checks.0.status", "failed"), + ("upstream_evidence.qualification.evaluated_at", "2018-01-02T07:00:00Z"), + ("upstream_evidence.time_semantics.earliest_external_knowledge", {"status": "evidenced"}), + ("evidence_scope", "real_data"), + ("output_quality.status", "failed"), + ("output_quality.checks.0.status", "failed"), + ("output_coverage.status", "incomplete"), + ("output_coverage.observed_count", 1), + ("output_schema_digest", "sha256:" + "0" * 64), + ("output_artifact_ref.artifact_id", "rhfactoroutputv1:sha256:" + "0" * 64), + ("availability_mode", "as_available"), + ("usage", "paper_trading"), + ("historical_availability", "declared_as_available"), + ("decision_eligible", True), + ("decision_eligible", 0), + ("evaluation_at", "2018-01-02T07:00:00Z"), + ("evaluation_at", "2026-09-08T01:04:00Z"), + ("computed_at", "2026-09-08T01:05:00Z"), + ("artifact_available_at", "2026-09-08T01:06:00Z"), + ("producer.id", "research_platform"), + ("code_revision", "unknown"), + ("actor.id", "https://private/a"), + ("causation.id", "rhdfv2:sha256:" + "0" * 64), + ], +) +def test_factor_rejects_reidentified_semantic_forgery(path: str, value: Any) -> None: + arguments = factor_arguments() + row = RetrospectiveFactorSetRef.create(**arguments).to_dict() + replace_at(row, path, value) + identify(row, "factor_set_id", "rhfactorsetv2:") + with pytest.raises(FactorContractError): + RetrospectiveFactorSetRef.from_dict(row, **decoding_arguments(arguments)) + + +def test_payload_validation_is_never_inherited_from_serialization() -> None: + arguments = factor_arguments() + result = RetrospectiveFactorSetRef.create(**arguments) + kwargs = decoding_arguments(arguments) + reference = RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs) + with pytest.raises(FactorContractError): + reference.require_payloads_revalidated() + checked = RetrospectiveFactorSetRef.from_dict( + result.to_dict(), + **kwargs, + **{ + key: arguments[key] + for key in ( + "output_schema_bytes", + "output_content_bytes", + "dataset_chunks", + "resolved_views", + ) + }, + ) + checked.require_payloads_revalidated() + assert checked == result + for extra in ( + {"output_schema_bytes": arguments["output_schema_bytes"]}, + {"dataset_chunks": arguments["dataset_chunks"]}, + {"resolved_views": arguments["resolved_views"]}, + {"output_schema_bytes": arguments["output_schema_bytes"], "output_content_bytes": b"[]"}, + {"dataset_chunks": arguments["dataset_chunks"], "resolved_views": []}, + ): + with pytest.raises(FactorContractError): + RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs, **extra) + + +def test_create_verifies_actual_snapshot_and_each_resolved_view() -> None: + for mutation in ( + "content", + "schema", + "snapshot", + "duplicate_view", + "noncanonical", + "unknown_view", + ): + arguments = factor_arguments() + view = arguments["resolved_views"][0] + if mutation == "content": + arguments["resolved_views"] = ( + ResolvedRetrospectiveView(view.view_ref_id, view.schema_bytes, b"[]"), + ) + elif mutation == "schema": + arguments["resolved_views"] = ( + ResolvedRetrospectiveView(view.view_ref_id, b"{}", view.content_bytes), + ) + elif mutation == "snapshot": + arguments["dataset_chunks"][0][0]["value"] = "0" + elif mutation == "duplicate_view": + arguments["resolved_views"] = (view, view) + elif mutation == "unknown_view": + arguments["resolved_views"] = ( + ResolvedRetrospectiveView( + "rhviewrefv2:sha256:" + "0" * 64, view.schema_bytes, view.content_bytes + ), + ) + else: + arguments["output_content_bytes"] += b"\n" + with pytest.raises(FactorContractError): + RetrospectiveFactorSetRef.create(**arguments) + + +def test_parent_requires_exact_correlation_scope_and_actual_availability() -> None: + arguments = factor_arguments() + parent = RetrospectiveFactorSetRef.create(**arguments) + child_args = { + **arguments, + "parent": parent, + "causation": RetrospectiveCausation("factor_set", parent.factor_set_id), + "evaluation_at": "2026-09-08T01:09:00Z", + "computed_at": "2026-09-08T01:10:00Z", + "artifact_available_at": "2026-09-08T01:11:00Z", + } + child = RetrospectiveFactorSetRef.create(**child_args) + assert child.factor_set_id != parent.factor_set_id + assert ( + RetrospectiveFactorSetRef.from_json( + child.to_json(), **decoding_arguments(arguments), parent=parent + ) + == child + ) + for changes in ( + {"parent": None}, + {"correlation_id": "different.correlation"}, + {"causation": RetrospectiveCausation("factor_set", "rhfactorsetv2:sha256:" + "0" * 64)}, + {"evaluation_at": "2026-09-08T01:07:59Z"}, + {"causation": arguments["causation"]}, + ): + with pytest.raises(FactorContractError): + RetrospectiveFactorSetRef.create(**{**child_args, **changes}) + + +def test_definition_validity_is_checked_at_actual_evaluation() -> None: + arguments = factor_arguments() + arguments.update( + evaluation_at="2027-01-01T00:00:00Z", + computed_at="2027-01-01T00:01:00Z", + artifact_available_at="2027-01-01T00:02:00Z", + ) + with pytest.raises(FactorContractError): + RetrospectiveFactorSetRef.create(**arguments) + + +def test_factor_contract_is_immutable_and_v1_does_not_accept_it() -> None: + arguments = factor_arguments() + result = RetrospectiveFactorSetRef.create(**arguments) + exported = result.to_dict() + exported["upstream_evidence"]["quality"]["status"] = "failed" + assert result.upstream_evidence["quality"]["status"] == "passed" + with pytest.raises(TypeError): + result.upstream_evidence["quality"]["status"] = "failed" + with pytest.raises(FactorContractError): + FactorSetRef.from_dict(result.to_dict(), **decoding_arguments(arguments)) + with pytest.raises(FactorContractError): + RetrospectiveFactorSetRef.from_json( + result.to_json() + "\n", **decoding_arguments(arguments) + ) + with pytest.raises(FactorContractError): + RetrospectiveInputBinding( + arguments["definitions"][0].definition_id, + "market", + "rhviewrefv1:sha256:" + "0" * 64, + "sha256:" + "0" * 64, + ) + + +def test_missing_synthetic_or_real_readiness_cannot_be_relabelled() -> None: + arguments = factor_arguments() + snapshot_row = arguments["dataset_snapshot"].to_dict() + snapshot_row["evidence_scope"] = "real_data" + identify(snapshot_row, "snapshot_id", "rhdsv2:") + snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_row) + foundation_row = arguments["foundation"].to_dict() + foundation_row["dataset_snapshot_id"] = snapshot.snapshot_id + foundation_row["readiness"]["evidence_scope"] = "real_data" + for view in foundation_row["standardized_views"]: + view["dataset_snapshot_id"] = snapshot.snapshot_id + seal_foundation(foundation_row) + foundation = RetrospectiveFoundationEnvelope.from_dict(foundation_row, snapshot=snapshot) + view = next(iter(foundation.views.values())) + arguments.update( + dataset_snapshot=snapshot, + foundation=foundation, + evidence_scope="real_data", + selected_view_ref_ids=(view.view_ref_id,), + input_bindings=( + RetrospectiveInputBinding( + arguments["definitions"][0].definition_id, + "market", + view.view_ref_id, + view.schema_digest, + ), + ), + view_availability=( + RetrospectiveViewAvailability( + view.view_ref_id, view.available_at, digest({"synthetic_view": 1}) + ), + ), + causation=RetrospectiveCausation("foundation", foundation.foundation_id), + ) + with pytest.raises(FactorContractError, match="real-data"): + RetrospectiveFactorSetRef.create(**arguments) diff --git a/tests/test_retrospective_portfolio_risk_contracts.py b/tests/test_retrospective_portfolio_risk_contracts.py new file mode 100644 index 0000000..9c225d2 --- /dev/null +++ b/tests/test_retrospective_portfolio_risk_contracts.py @@ -0,0 +1,655 @@ +"""New synthetic S4 evidence; historical valuation is not actual availability.""" + +from __future__ import annotations + +import hashlib +import json +from dataclasses import FrozenInstanceError, replace +from typing import Any + +import pandas as pd +import pytest + +from quant_engine.portfolio_risk_contracts import ( + ComputationReceipt, + ConstraintSetV1, + FreshnessPolicy, + PortfolioRiskContractError, + RiskAssessmentStatus, + RiskFindingCode, +) +from quant_engine.artifact import EvidenceQualification, PerformanceEvidenceError +from quant_engine.factor_contracts import FactorContractError +from quant_engine.governed_pipeline import BacktestContractError +from quant_engine.risk import ComponentRiskResult, CovarianceSnapshot, labeled_component_risk +import quant_engine.retrospective_portfolio_risk_contracts as contracts +from quant_engine.retrospective_artifact_contracts import ( + build_retrospective_backtest_evidence_manifest, +) +from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef +from quant_engine.retrospective_portfolio_risk_contracts import ( + RetrospectivePortfolioDecision, + RetrospectivePortfolioTarget, + RetrospectiveRiskAssessment, + build_retrospective_portfolio_decision, + compute_retrospective_portfolio_receipt_digests, + assess_retrospective_portfolio_risk, +) +from test_retrospective_artifact_contracts import synthetic_artifact +from test_retrospective_backtest_contracts import run_arguments +from test_retrospective_data_contracts import digest, replace_at + +ASSETS = ("rhinstrument:" + "1" * 32, "rhinstrument:" + "2" * 32) +CONTRACT_ERRORS = ( + FactorContractError, + PortfolioRiskContractError, + BacktestContractError, + PerformanceEvidenceError, +) + + +def portfolio_arguments() -> dict[str, Any]: + run = RetrospectiveBacktestRunRef.create(**run_arguments()) + artifact = synthetic_artifact(run) + manifest = build_retrospective_backtest_evidence_manifest( + run, artifact, artifact_available_at="2026-09-08T01:11:00Z" + ) + target = RetrospectivePortfolioTarget.create( + backtest_run_id=run.run_id, + dataset_snapshot_id=run.dataset_snapshot_id, + weights={ASSETS[0]: 0.6, ASSETS[1]: 0.4}, + effective_at="2018-01-05T07:00:00Z", + created_at="2026-09-08T01:12:00Z", + ) + return { + "backtest_run_ref": run, + "manifest": manifest, + "target": target, + "objective_name": "synthetic_allocation", + "objective_version": "1.0.0", + "objective_digest": digest({"synthetic_objective": 1}), + "model_name": "bounded_weights", + "model_version": "1.0.0", + "model_digest": digest({"synthetic_model": 1}), + "expected_return_digest": digest({"synthetic_returns": 1}), + "covariance_digest": "sha256:" + "a" * 64, + "scenario_digest": digest({"synthetic_scenario": 1}), + "constraints": ConstraintSetV1( + gross_exposure_max=1.0, + net_exposure_min=1.0, + net_exposure_max=1.0, + single_asset_min=0.2, + single_asset_max=0.7, + position_count_max=2, + turnover_max=0.2, + ), + "freshness_policy": FreshnessPolicy( + max_manifest_age_seconds=3600, max_covariance_age_days=0 + ), + "prior_weights": {ASSETS[0]: 0.5, ASSETS[1]: 0.5}, + "computed_at": "2026-09-08T01:13:00Z", + } + + +def portfolio_receipt(arguments: dict[str, Any], **changes: Any) -> ComputationReceipt: + values = compute_retrospective_portfolio_receipt_digests( + **{key: value for key, value in arguments.items() if key not in {"computed_at", "receipt"}} + ) + return ComputationReceipt( + **{ + "algorithm": "bounded_weights", + "algorithm_version": "1.0.0", + "implementation_digest": digest({"synthetic_implementation": 1}), + "parameter_digest": digest({"synthetic_parameters": 1}), + "input_digest": values["input_digest"], + "constraint_digest": values["constraint_digest"], + "output_digest": values["output_digest"], + "status": "completed", + "solver_required": False, + "solver_name": None, + "solver_version": None, + "solver_config_digest": None, + "iterations": None, + "objective_value": None, + "max_constraint_residual": values["max_constraint_residual"], + "tolerance": 1e-12, + "computed_at": arguments["computed_at"], + **changes, + } + ) + + +def test_target_separates_historical_effective_time_from_actual_creation() -> None: + arguments = portfolio_arguments() + target = arguments["target"] + assert target.effective_at == "2018-01-05T07:00:00Z" + assert target.created_at == "2026-09-08T01:12:00Z" + assert target.target_id.startswith("rhportfoliotargetv2:sha256:") + assert target.to_dict()["usage"] == "retrospective_research" + + +def test_portfolio_decision_preserves_constraints_and_actual_receipt_time() -> None: + arguments = portfolio_arguments() + decision = build_retrospective_portfolio_decision( + **arguments, receipt=portfolio_receipt(arguments) + ) + assert decision.decision_id.startswith("rhportfoliodecisionv2:sha256:") + assert decision.effective_at == "2018-01-05T07:00:00Z" + assert decision.created_at == "2026-09-08T01:12:00Z" + assert decision.computed_at == "2026-09-08T01:13:00Z" + assert decision.gross_exposure == 1.0 + assert decision.position_count == 2 + assert decision.to_dict()["decision_eligible"] is False + + +def covariance(arguments: dict[str, Any], **changes: Any) -> CovarianceSnapshot: + return CovarianceSnapshot( + **{ + "snapshot_id": "covariance:synthetic-retrospective", + "as_of_date": "2018-01-05", + "covariance": pd.DataFrame([[0.04, 0.01], [0.01, 0.09]], index=ASSETS, columns=ASSETS), + "return_frequency": "1d", + "periods_per_year": 252, + "method": "provided", + "window_start_date": "2018-01-02", + "window_end_date": "2018-01-05", + "observations": 4, + "lookback_sessions": 4, + "missing_policy": "complete_case", + "data_snapshot_id": arguments["backtest_run_ref"].dataset_snapshot_id, + "input_sha256": "a" * 64, + **changes, + } + ) + + +def risk_arguments(arguments: dict[str, Any]) -> dict[str, Any]: + decision = build_retrospective_portfolio_decision( + **arguments, receipt=portfolio_receipt(arguments) + ) + return { + "portfolio_decision": decision, + "backtest_run_ref": arguments["backtest_run_ref"], + "manifest": arguments["manifest"], + "covariance": covariance(arguments), + "risk_model_name": "euler_volatility", + "risk_model_version": "1.0.0", + "risk_model_digest": digest({"synthetic_risk_model": 1}), + "risk_budget": {ASSETS[0]: 0.8, ASSETS[1]: 0.8}, + "portfolio_volatility_limit": 10.0, + "groups": {ASSETS[0]: "equity", ASSETS[1]: "fixed_income"}, + "computed_at": "2026-09-08T01:14:00Z", + } + + +def test_risk_uses_historical_business_age_and_actual_computation_time() -> None: + arguments = risk_arguments(portfolio_arguments()) + result = assess_retrospective_portfolio_risk(**arguments) + assert result.assessment_id.startswith("rhriskassessmentv2:sha256:") + assert result.qualified is True + assert result.effective_at == "2018-01-05T07:00:00Z" + assert result.computed_at == "2026-09-08T01:14:00Z" + assert result.to_dict()["decision_eligible"] is False + assert result.to_dict()["execution_validation"] == "not_validated" + assert sum(result.percentage_risk.values()) == pytest.approx(1.0) + assert sum(result.component_risk.values()) == pytest.approx(result.portfolio_volatility) + + +def test_new_risk_computation_cannot_reuse_stale_actual_manifest_time() -> None: + arguments = risk_arguments(portfolio_arguments()) + arguments["computed_at"] = "2026-09-08T02:11:01Z" + with pytest.raises(FactorContractError, match="stale"): + assess_retrospective_portfolio_risk(**arguments) + + +def target_with(arguments: dict[str, Any], **changes: Any) -> RetrospectivePortfolioTarget: + row = arguments["target"].to_dict() + return RetrospectivePortfolioTarget.create( + **{ + key: value + for key, value in {**row, **changes}.items() + if key + in {"backtest_run_id", "dataset_snapshot_id", "weights", "effective_at", "created_at"} + } + ) + + +def decision_context(arguments: dict[str, Any]) -> dict[str, Any]: + return {key: arguments[key] for key in ("backtest_run_ref", "manifest", "target")} + + +def assessment_context(arguments: dict[str, Any]) -> dict[str, Any]: + return { + key: arguments[key] + for key in ("portfolio_decision", "backtest_run_ref", "manifest", "covariance") + } + + +def reidentify(row: dict[str, Any], field: str, prefix: str) -> None: + row.pop(field, None) + encoded = json.dumps( + row, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False + ) + row[field] = prefix + "sha256:" + hashlib.sha256(encoded.encode()).hexdigest() + + +@pytest.mark.parametrize( + "parser", + [RetrospectivePortfolioTarget, RetrospectivePortfolioDecision, RetrospectiveRiskAssessment], +) +def test_json_syntax_failures_use_typed_contract_errors(parser: Any) -> None: + with pytest.raises(FactorContractError): + parser.from_json(b"{") + + +@pytest.mark.parametrize( + "change", + [ + {"method": "alternate_estimator"}, + {"window_start_date": "2018-01-03"}, + {"window_end_date": "2018-01-04"}, + {"observations": 3}, + {"lookback_sessions": 5}, + {"missing_policy": "alternate_missing_policy"}, + ], +) +def test_covariance_estimation_context_is_bound_into_the_result_identity( + change: dict[str, Any], +) -> None: + base = portfolio_arguments() + arguments = risk_arguments(base) + original = assess_retrospective_portfolio_risk(**arguments) + arguments["covariance"] = covariance(base, **change) + changed = assess_retrospective_portfolio_risk(**arguments) + assert changed.assessment_id != original.assessment_id + + +def test_canonical_roundtrips_and_immutable_results() -> None: + base = portfolio_arguments() + target = base["target"] + assert RetrospectivePortfolioTarget.from_json(target.to_json().encode()) == target + decision = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base)) + assert ( + RetrospectivePortfolioDecision.from_json(decision.to_json(), **decision_context(base)) + == decision + ) + arguments = risk_arguments(base) + result = assess_retrospective_portfolio_risk(**arguments) + assert ( + RetrospectiveRiskAssessment.from_json( + result.to_json().encode(), **assessment_context(arguments) + ) + == result + ) + with pytest.raises(TypeError): + target.weights[ASSETS[0]] = 0.1 + with pytest.raises(FrozenInstanceError): + target.created_at = "2018-01-05T07:00:00Z" + with pytest.raises(TypeError): + decision.target_weights[ASSETS[0]] = 0.1 + with pytest.raises(TypeError): + result.component_risk[ASSETS[0]] = 0.1 + detached = result.to_dict() + detached["component_risk"][ASSETS[0]] = 0.1 + assert detached != result.to_dict() + + +@pytest.mark.parametrize( + "change", + [ + {"weights": {}}, + {"weights": {"SIM0": 1.0}}, + {"weights": {ASSETS[0]: float("nan")}}, + {"weights": {ASSETS[0]: True}}, + {"backtest_run_id": "rhbacktestrunv1:sha256:" + "0" * 64}, + {"dataset_snapshot_id": "rhds:sha256:" + "0" * 64}, + {"effective_at": "2026-09-09T01:00:00Z"}, + {"created_at": "2026-09-08T01:12:00.1234567Z"}, + {"effective_at": "2018-01-05T15:00:00+08:00"}, + ], +) +def test_target_rejects_legacy_ambiguous_and_nonfinite_inputs(change: dict[str, Any]) -> None: + with pytest.raises(CONTRACT_ERRORS): + target_with(portfolio_arguments(), **change) + + +@pytest.mark.parametrize( + "path,value", + [ + ("usage", "live"), + ("historical_availability", "established"), + ("schema_version", "1.0.0"), + ("extra", True), + ("target_id", "rhportfoliotargetv2:sha256:" + "0" * 64), + ], +) +def test_target_rejects_wire_mutations(path: str, value: Any) -> None: + row = portfolio_arguments()["target"].to_dict() + row[path] = value + with pytest.raises(CONTRACT_ERRORS): + RetrospectivePortfolioTarget.from_dict(row) + + +@pytest.mark.parametrize("field", ["input_digest", "constraint_digest", "output_digest"]) +def test_receipt_digests_are_recomputed(field: str) -> None: + arguments = portfolio_arguments() + receipt = portfolio_receipt(arguments, **{field: "sha256:" + "0" * 64}) + with pytest.raises(FactorContractError, match="independently recomputed"): + build_retrospective_portfolio_decision(**arguments, receipt=receipt) + + +@pytest.mark.parametrize("status", ["failed", "fallback"]) +def test_failed_or_fallback_solver_cannot_form_a_decision(status: str) -> None: + arguments = portfolio_arguments() + receipt = portfolio_receipt( + arguments, + status=status, + solver_required=True, + solver_name="synthetic_solver", + solver_version="1.0.0", + solver_config_digest=digest({"synthetic_solver": 1}), + iterations=1, + objective_value=0.0, + ) + with pytest.raises(FactorContractError, match="failed/fallback"): + build_retrospective_portfolio_decision(**arguments, receipt=receipt) + + +@pytest.mark.parametrize( + "change", + [ + {"backtest_run_id": "rhbacktestrunv2:sha256:" + "0" * 64}, + {"dataset_snapshot_id": "rhdsv2:sha256:" + "0" * 64}, + {"weights": {"rhinstrument:" + "f" * 32: 1.0}}, + {"created_at": "2026-09-08T01:10:00Z"}, + {"created_at": "2026-09-08T01:14:00Z"}, + ], +) +def test_decision_closes_target_identity_assets_and_actual_time(change: dict[str, Any]) -> None: + arguments = portfolio_arguments() + receipt = portfolio_receipt(arguments) + arguments["target"] = target_with(arguments, **change) + with pytest.raises(CONTRACT_ERRORS): + build_retrospective_portfolio_decision(**arguments, receipt=receipt) + + +def test_actual_manifest_freshness_boundary_and_receipt_time() -> None: + arguments = portfolio_arguments() + arguments["computed_at"] = "2026-09-08T02:11:00Z" + assert ( + build_retrospective_portfolio_decision( + **arguments, receipt=portfolio_receipt(arguments) + ).computed_at + == arguments["computed_at"] + ) + arguments["computed_at"] = "2026-09-08T02:11:00.000001Z" + with pytest.raises(FactorContractError, match="stale"): + build_retrospective_portfolio_decision(**arguments, receipt=portfolio_receipt(arguments)) + arguments["computed_at"] = "2026-09-08T01:13:00Z" + with pytest.raises(FactorContractError, match="receipt actual time"): + build_retrospective_portfolio_decision( + **arguments, receipt=portfolio_receipt(arguments, computed_at="2026-09-08T01:13:01Z") + ) + + +def test_manifest_tables_and_qualification_are_revalidated_at_s4_boundary() -> None: + arguments = portfolio_arguments() + receipt = portfolio_receipt(arguments) + manifest = arguments["manifest"] + artifact = manifest._artifact + arguments["manifest"] = build_retrospective_backtest_evidence_manifest( + arguments["backtest_run_ref"], + artifact, + artifact_available_at=manifest.artifact_available_at, + qualification=EvidenceQualification.EXPLORATORY, + ) + with pytest.raises(FactorContractError, match="contract-qualified"): + build_retrospective_portfolio_decision(**arguments, receipt=receipt) + arguments["manifest"] = manifest + # Public access is an isolated copy. Simulate corruption of the retained bytes, + # beyond that normal interface, to exercise the consumer's independent recheck. + artifact._performance.loc[0, "n_days"] += 1 + with pytest.raises(CONTRACT_ERRORS): + build_retrospective_portfolio_decision(**arguments, receipt=receipt) + + +def test_constraint_residuals_and_prior_assets_cannot_be_bypassed() -> None: + arguments = portfolio_arguments() + arguments["constraints"] = ConstraintSetV1(gross_exposure_max=0.9) + # A solver may report convergence within its tolerance; actual contract constraints still bind. + receipt = portfolio_receipt( + arguments, + status="converged", + solver_required=True, + solver_name="synthetic_solver", + solver_version="1.0.0", + solver_config_digest=digest({"synthetic_solver": 1}), + iterations=1, + objective_value=0.0, + tolerance=0.2, + ) + with pytest.raises(FactorContractError, match="violates supported constraints"): + build_retrospective_portfolio_decision(**arguments, receipt=receipt) + arguments = portfolio_arguments() + arguments["prior_weights"] = {"rhinstrument:" + "f" * 32: 0.5} + with pytest.raises(FactorContractError, match="prior assets"): + compute_retrospective_portfolio_receipt_digests( + **{key: value for key, value in arguments.items() if key != "computed_at"} + ) + arguments["prior_weights"] = None + with pytest.raises(PortfolioRiskContractError, match="prior"): + portfolio_receipt(arguments) + + +def test_optional_prior_budget_limit_and_groups_have_explicit_empty_semantics() -> None: + base = portfolio_arguments() + base["constraints"] = ConstraintSetV1(gross_exposure_max=1.0) + base["prior_weights"] = None + arguments = risk_arguments(base) + arguments.update(risk_budget=None, portfolio_volatility_limit=None, groups=None) + result = assess_retrospective_portfolio_risk(**arguments) + assert result.qualified is True + assert result.risk_budget == {} + assert result.group_exposure == {} + assert result.groups is None + + +@pytest.mark.parametrize( + "path,value", + [ + ("decision_eligible", True), + ("execution_validation", "validated"), + ("historical_availability", "established"), + ("gross_exposure", True), + ("position_count", 2.0), + ("target_weights." + ASSETS[0], 0.5), + ("schema_version", "1.0.0"), + ("observation_cutoff", "2018-01-05T07:00:00Z"), + ("extra", True), + ], +) +def test_decision_rejects_reidentified_forged_wire(path: str, value: Any) -> None: + base = portfolio_arguments() + row = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base)).to_dict() + replace_at(row, path, value) + reidentify(row, "decision_id", "rhportfoliodecisionv2:") + with pytest.raises(CONTRACT_ERRORS): + RetrospectivePortfolioDecision.from_dict(row, **decision_context(base)) + + +@pytest.mark.parametrize( + "change", + [ + {"as_of_date": "2018-01-06"}, + {"as_of_date": "2018-01-04", "window_end_date": "2018-01-04"}, + {"window_start_date": None, "window_end_date": None}, + {"data_snapshot_id": "rhdsv2:sha256:" + "0" * 64}, + {"input_sha256": "b" * 64}, + ], +) +def test_covariance_business_time_bounds_and_source_binding(change: dict[str, Any]) -> None: + base = portfolio_arguments() + arguments = risk_arguments(base) + arguments["covariance"] = covariance(base, **change) + with pytest.raises(FactorContractError): + assess_retrospective_portfolio_risk(**arguments) + + +@pytest.mark.parametrize( + "matrix,index,columns", + [ + ([[float("nan"), 0.0], [0.0, 0.1]], ASSETS, ASSETS), + ([[0.1, 0.1], [0.0, 0.1]], ASSETS, ASSETS), + ([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], ASSETS[0]), ASSETS), + ([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], "unknown"), ASSETS), + ([[0.1, 0.0], [0.0, 0.1]], ASSETS, (ASSETS[0], "unknown")), + ], +) +def test_covariance_structure_is_checked_before_computation( + matrix: Any, index: Any, columns: Any +) -> None: + base = portfolio_arguments() + arguments = risk_arguments(base) + arguments["covariance"] = covariance( + base, covariance=pd.DataFrame(matrix, index=index, columns=columns) + ) + with pytest.raises(PortfolioRiskContractError): + assess_retrospective_portfolio_risk(**arguments) + + +@pytest.mark.parametrize( + "change", + [ + {"risk_budget": {ASSETS[0]: -0.1}}, + {"risk_budget": {"unknown": 0.1}}, + {"portfolio_volatility_limit": -0.1}, + {"groups": {ASSETS[0]: "equity"}}, + {"groups": []}, + {"risk_model_version": "latest"}, + {"risk_model_name": "/private/model"}, + {"computed_at": "2026-09-08T01:12:59Z"}, + {"portfolio_decision": object()}, + {"covariance": object()}, + ], +) +def test_risk_rejects_invalid_models_budgets_clocks_and_untyped_inputs( + change: dict[str, Any], +) -> None: + arguments = risk_arguments(portfolio_arguments()) + arguments.update(change) + with pytest.raises(CONTRACT_ERRORS): + assess_retrospective_portfolio_risk(**arguments) + + +@pytest.mark.parametrize( + "matrix,finding", + [ + ([[1.0, 2.0], [2.0, 1.0]], RiskFindingCode.COVARIANCE_NOT_PSD), + ([[0.0, 0.0], [0.0, 0.0]], RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE), + ], +) +def test_numerical_unavailability_is_not_qualification( + matrix: Any, finding: RiskFindingCode +) -> None: + base = portfolio_arguments() + arguments = risk_arguments(base) + arguments["covariance"] = covariance( + base, covariance=pd.DataFrame(matrix, index=ASSETS, columns=ASSETS) + ) + result = assess_retrospective_portfolio_risk(**arguments) + assert result.status is RiskAssessmentStatus.UNAVAILABLE + assert result.qualified is False + assert result.findings == (finding,) + assert result.portfolio_volatility is None + + +def test_risk_uses_the_existing_numeric_implementation_exactly_once( + monkeypatch: pytest.MonkeyPatch, +) -> None: + arguments = risk_arguments(portfolio_arguments()) + calls = [] + + def recorded(weights: Any, matrix: Any) -> ComponentRiskResult: + calls.append((weights, matrix)) + return labeled_component_risk(weights, matrix) + + monkeypatch.setattr(contracts, "labeled_component_risk", recorded) + result = assess_retrospective_portfolio_risk(**arguments) + assert len(calls) == 1 + expected = labeled_component_risk(*calls[0]) + assert result.component_risk == expected.component.to_dict() + assert result.portfolio_volatility == expected.portfolio_volatility + + +def test_unknown_numeric_failures_are_sanitized(monkeypatch: pytest.MonkeyPatch) -> None: + def failed(*args: Any) -> ComponentRiskResult: + raise ValueError("synthetic internal detail") + + monkeypatch.setattr(contracts, "labeled_component_risk", failed) + with pytest.raises(PortfolioRiskContractError, match="risk computation failed") as error: + assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments())) + assert "internal detail" not in str(error.value) + + +def test_nonclosed_decomposition_is_unavailable(monkeypatch: pytest.MonkeyPatch) -> None: + def nonclosed(weights: Any, matrix: Any) -> ComponentRiskResult: + output = labeled_component_risk(weights, matrix) + return replace(output, component=output.component * 0.5) + + monkeypatch.setattr(contracts, "labeled_component_risk", nonclosed) + result = assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments())) + assert result.status is RiskAssessmentStatus.UNAVAILABLE + assert result.findings == (RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED,) + + +@pytest.mark.parametrize( + "change", [{"portfolio_volatility_limit": 0.0}, {"risk_budget": {ASSETS[0]: 0.0}}] +) +def test_budget_breach_keeps_ready_but_unqualified_evidence(change: dict[str, Any]) -> None: + arguments = risk_arguments(portfolio_arguments()) + arguments.update(change) + result = assess_retrospective_portfolio_risk(**arguments) + assert result.status is RiskAssessmentStatus.READY + assert result.qualified is False + assert result.findings == (RiskFindingCode.RISK_BUDGET_BREACH,) + assert result.decision_eligible is False + + +@pytest.mark.parametrize( + "path,value", + [ + ("decision_eligible", True), + ("execution_validation", "validated"), + ("historical_availability", "established"), + ("qualified", 1), + ("portfolio_volatility", 1.0), + ("component_risk." + ASSETS[0], 1.0), + ("schema_version", "1.0.0"), + ("covariance_matrix_digest", "sha256:" + "0" * 64), + ("extra", True), + ], +) +def test_risk_rejects_reidentified_forged_wire(path: str, value: Any) -> None: + arguments = risk_arguments(portfolio_arguments()) + row = assess_retrospective_portfolio_risk(**arguments).to_dict() + replace_at(row, path, value) + reidentify(row, "assessment_id", "rhriskassessmentv2:") + with pytest.raises(CONTRACT_ERRORS): + RetrospectiveRiskAssessment.from_dict(row, **assessment_context(arguments)) + + +@pytest.mark.parametrize( + "raw", + [ + b'{"x":1,"x":2}', + b'{ "x":1}', + b"[]", + b'{"x":NaN}', + b'{"x":Infinity}', + b'{"x":9007199254740992}', + 1, + ], +) +def test_json_profiles_reject_ambiguous_nonfinite_and_noncanonical_input(raw: Any) -> None: + with pytest.raises(CONTRACT_ERRORS): + RetrospectivePortfolioTarget.from_json(raw)