feat: add explicit retrospective v2 computation contracts (#20)
CI / lite (push) Successful in 19s
CI / lite (push) Successful in 19s
This commit was merged in pull request #20.
This commit is contained in:
+13
-3
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"module_id": "quant_engine",
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 5, "effective_from": "2026-09-01T00:00:00+08:00"},
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 6, "effective_from": "2026-09-08T19:33:40+08:00"},
|
||||
"repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"},
|
||||
"bounded_context": {
|
||||
"domain": "quantitative-research-engine",
|
||||
@@ -21,6 +21,7 @@
|
||||
{"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"},
|
||||
{"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact and performance-methodology evidence without recomputation, persistence, or decision authority.", "status": "operational"},
|
||||
{"id": "portfolio-risk-computation-contracts", "summary": "Verify deterministic portfolio-computation receipts and expose S3-bound portfolio decisions and risk assessments without adding algorithms or execution authority.", "status": "operational"},
|
||||
{"id": "retrospective-computation-contracts", "summary": "Decode observation-aware v2 data and expose explicit retrospective factor, backtest, portfolio and risk contracts with two clocks, no historical-availability claim and no execution authority.", "status": "operational"},
|
||||
{"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"}
|
||||
],
|
||||
"data": {"owns": [
|
||||
@@ -35,11 +36,20 @@
|
||||
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
|
||||
{"contract_id": "researchhub.performance-evidence", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
|
||||
{"contract_id": "researchhub.portfolio-decision", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"}
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.factor-set-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_factor_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-run-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_backtest_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
|
||||
{"contract_id": "researchhub.performance-evidence", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
|
||||
{"contract_id": "researchhub.portfolio-target", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.portfolio-decision", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"}
|
||||
],
|
||||
"consumes": [
|
||||
{"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"},
|
||||
{"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"}
|
||||
{"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"},
|
||||
{"contract_id": "researchhub.dataset-snapshot", "version": "2.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_retrospective_envelope_and_materialized_chunks"},
|
||||
{"contract_id": "researchhub.data-foundation", "version": "2.0.0", "authority": "researchhub.data", "admission": "observation_bound_selected_views_and_materialized_bytes"}
|
||||
]
|
||||
},
|
||||
"dependencies": [],
|
||||
|
||||
@@ -29,6 +29,7 @@
|
||||
- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef`
|
||||
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest`
|
||||
- `portfolio_risk_contracts` — S3 证据闭合的 `PortfolioDecision` / `RiskAssessment` v1;独立复核 freshness、约束与 computation receipt,并复用既有标签安全风险分解
|
||||
- `retrospective_*_contracts` — 未发布的显式 v2 回顾性合同:区分历史业务日期与实际可得/计算时间,保留 v1 和现有金融公式,不授予历史可得性、发布或执行权限;见 [v2 接口说明](docs/RETROSPECTIVE_COMPUTATION_V2.md)
|
||||
- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计
|
||||
- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效
|
||||
- `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test)
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
# Retrospective computation contracts v2 (unreleased)
|
||||
|
||||
This pure, storage-neutral compatibility path consumes the separate data-contract
|
||||
major 2.0.0. It does not migrate, reinterpret or relax the accepted v1 contracts.
|
||||
No financial formula, execution simulation, dependency lock, production database,
|
||||
publisher or live/paper-order interface changes here. Package version is unchanged;
|
||||
the new contract major is not a package release or deployment.
|
||||
|
||||
## Explicit public boundaries
|
||||
|
||||
| Module | Public types/builders | Changed wire identity |
|
||||
| --- | --- | --- |
|
||||
| `retrospective_data_contracts` | `RetrospectiveSnapshotEnvelope`, `RetrospectiveFoundationEnvelope` | `rhdsv2`, `rhdfv2`; consume RP-owned 2.0.0 data semantics |
|
||||
| `retrospective_factor_contracts` | `RetrospectiveFactorSetRef`, typed input/view/causation bindings, `ResolvedRetrospectiveView` | `rhfactorsetv2` |
|
||||
| `retrospective_backtest_contracts` | `RetrospectiveBacktestRunRef` | `rhbacktestrunv2` |
|
||||
| `retrospective_artifact_contracts` | `RetrospectiveBacktestEvidenceManifest`, `RetrospectivePerformanceEvidence` and their builders | `rhbacktestevidencev2`, `rhperformancev2` |
|
||||
| `retrospective_portfolio_risk_contracts` | `RetrospectivePortfolioTarget`, `RetrospectivePortfolioDecision`, `RetrospectiveRiskAssessment`; receipt-digest, decision and assessment builders | `rhportfoliotargetv2`, `rhportfoliodecisionv2`, `rhriskassessmentv2` |
|
||||
|
||||
These are separate types and domain-separated content identities. There is no
|
||||
automatic v1-to-v2 cast. Unknown schema versions and fields are rejected. The
|
||||
performance wire keeps its named schema `researchhub.performance-evidence.v2`;
|
||||
the other new computation contracts use `schema_version: 2.0.0`.
|
||||
|
||||
FactorDefinition, factor-output byte references, output quality/coverage,
|
||||
ConstraintSetV1, FreshnessPolicy, ComputationReceipt, CovarianceSnapshot, financial
|
||||
algorithms, performance metric/methodology IDs and the nine ResearchRunArtifact
|
||||
tables keep their existing semantics. The table schema remains **1.1.0**. Reusing
|
||||
these neutral primitives does not make a new-major upstream reference v1-compatible.
|
||||
|
||||
## Two clocks, not backdated evidence
|
||||
|
||||
Every new result fixes `usage=retrospective_research` and
|
||||
`historical_availability=not_established`. A business date describes the historical
|
||||
period being researched. Observation, publication, evaluation, artifact availability,
|
||||
target creation and computation describe actual events, and must not be backdated.
|
||||
Public v2 instants require UTC `Z` with at most six fractional digits.
|
||||
|
||||
`observation_cutoff` and chunk `observed_by` are upper-bound observations. They are
|
||||
not the earliest public knowledge time or a PIT cutoff. Unknown earliest knowledge
|
||||
stays unknown; a supplied knowledge-evidence digest is not authenticated by parsing.
|
||||
Foundation observation sequences describe retained revisions, not complete original
|
||||
history. Selected view routes, calendars, corporate-action coverage and lineage
|
||||
must close exactly within the supplied Foundation.
|
||||
|
||||
Required actual order for factor/backtest evidence is:
|
||||
|
||||
1. Foundation publication <= factor evaluation <= factor computation <= factor availability.
|
||||
2. Factor availability <= backtest evaluation <= artifact start <= artifact finish
|
||||
<= backtest computation <= artifact availability.
|
||||
3. Artifact availability <= target creation <= portfolio computation <= risk computation.
|
||||
|
||||
RetrospectivePortfolioTarget has a historical `effective_at` and a distinct actual
|
||||
`created_at`. PortfolioDecision carries both plus actual `computed_at`. Covariance
|
||||
window end <= covariance as-of date <= the historical effective date; covariance
|
||||
maximum age is measured against that historical date. Manifest maximum age is
|
||||
measured against **actual** portfolio and risk computation separately. Passing one
|
||||
age check cannot substitute for the other. Generic v1 receipt timestamps retain
|
||||
their original normalization; binding compares parsed actual instants.
|
||||
|
||||
## Materialized bytes and reference-only reads
|
||||
|
||||
Snapshot decoding checks structure, all six blocking-quality declarations,
|
||||
qualification/time ordering, observation receipts and identities.
|
||||
`verify_materialized_records` additionally checks supplied chunks, per-chunk and
|
||||
aggregate content, counts, dimensions, effective ranges and macro effective instants.
|
||||
Provider/physical paths are forbidden in public metadata and materialized records.
|
||||
|
||||
Factor creation requires actual snapshot chunks, selected view schema/content bytes,
|
||||
and factor-output schema/content bytes. Definition inputs, view availability,
|
||||
Foundation ancestry and computed digests must close. Reference-only deserialization
|
||||
is allowed for display/inspection, but input/output validation flags are derived from
|
||||
supplied bytes, are not serialized claims, and must be re-established for new
|
||||
computation. Backtest creation requires a factor whose payloads were revalidated.
|
||||
Reference decoding cannot turn an unverified factor into an admitted compute input.
|
||||
|
||||
Backtest manifest decoding rebuilds evidence from the supplied typed run and all
|
||||
nine actual artifact tables. It checks table/run/config/strategy bindings and time
|
||||
ordering. Portfolio composition revalidates those retained tables again, rather
|
||||
than trusting a serialized manifest or mutable Python context. A table digest proves
|
||||
content binding, not that those tables were produced by the claimed computation.
|
||||
|
||||
All content-addressed IDs exclude their own ID field and bind the remainder of the
|
||||
closed payload. Data/factor/backtest/manifest JSON retains the strict data profile
|
||||
(no JSON floating-point numbers; financial record decimals are strings). Performance
|
||||
and S4 preserve the existing finite numeric JSON profile: finite floats, safe ints,
|
||||
exact booleans, sorted keys, compact separators, UTF-8. Duplicate keys, NaN,
|
||||
Infinity, noncanonical JSON and extra fields are rejected. Wire revalidation uses
|
||||
type-sensitive comparisons, including `true` versus `1`. Serializers return
|
||||
detached copies; internal public maps are immutable.
|
||||
|
||||
## Replay, receipts and risk
|
||||
|
||||
Backtest v2 replay specification binds immutable input identities, selected calendar
|
||||
and actions, strategy/execution/cost versions and digests, configuration, code,
|
||||
environment lock and random seed. It excludes **both actual evaluation and actual
|
||||
computation time**. These actual times remain in each run's identity. A replay must
|
||||
retain the same replay specification, append its unique full ancestry, increment
|
||||
attempt by one, and have parent computation < new actual evaluation <= computation.
|
||||
This explicit new-major rule allows a later genuine replay without pretending its
|
||||
evaluation happened at the parent's clock time.
|
||||
|
||||
Portfolio computation-input v2 binds the full run and manifest document digests,
|
||||
new target (including both clocks), objective/model versions and digests, declared
|
||||
expected returns/covariance/scenario inputs, freshness policy and prior weights.
|
||||
The receipt separately binds that input, constraints and recomputed outputs/residuals.
|
||||
Targets and prior holdings must use selected logical instrument IDs, not ad-hoc
|
||||
symbol matches. Failed/fallback receipts and any actual constraint residual are
|
||||
rejected, even if a solver declares convergence within a permissive tolerance.
|
||||
|
||||
Risk uses the existing labelled Euler decomposition exactly once. Its result binds
|
||||
the supplied matrix content plus covariance method, bounded estimation window,
|
||||
observation count, lookback, missing policy, annualization, source dataset/input,
|
||||
model/budgets/groups and actual computation time. It checks exact labels, finite
|
||||
symmetry, covariance-source binding and both freshness clocks. Non-PSD,
|
||||
non-positive portfolio variance or non-closed contributions produce an unavailable,
|
||||
unqualified result. A budget breach is a ready but unqualified calculation result.
|
||||
`qualified=true` means only that these calculation checks passed. Every result
|
||||
remains `decision_eligible=false`, `execution_validation=not_validated`; no portfolio
|
||||
approval, maker-checker, publication, paper or live permission is granted here.
|
||||
|
||||
## Trust, ownership and test evidence
|
||||
|
||||
Pure builders accept declarations. Hashes, typed objects, model names, successful
|
||||
constraint checks and synthetic fixtures do **not** authenticate data or compute
|
||||
producers. Trusted owner-version bindings and receipt/qualification/view/clock
|
||||
admission ports remain mandatory. RP owns governance and presentation; Research
|
||||
Results owns publication. QE supplies validated calculation facts only.
|
||||
|
||||
The two data fixtures are public RP candidate vectors from
|
||||
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (PR #100). EDB producer candidate
|
||||
`88433dfc9d865ef782465498cdf9454c73920abd` (PR #13) is not a runtime dependency or
|
||||
accepted owner binding. Acceptance/review gates remain separate from local tests.
|
||||
|
||||
`tests/fixtures/retrospective-computation-v2.golden.json` freezes newly constructed
|
||||
synthetic factor/run/manifest/performance/portfolio/risk payloads and their inputs.
|
||||
Its artifact matrices are separate synthetic envelope-test inputs: the one-day
|
||||
public data fixture is **not** claimed to have produced the four-day artifact.
|
||||
The vector is not an end-to-end data/computation provenance proof or real-data run.
|
||||
Its deterministic IDs are contract-regression evidence, not admitted source facts.
|
||||
|
||||
Focused tests cover v2 goldens, mutation and strict JSON, bytes versus references,
|
||||
two-clock freshness, replay ancestry, exact table bindings, constraints, receipts,
|
||||
covariance provenance and numerical findings. Existing v1 tests must also pass.
|
||||
Rollback is disabling the explicit v2 entry path while retaining v1 and original
|
||||
immutable results; never retag old results or silently downgrade failed v2 admission.
|
||||
@@ -0,0 +1,88 @@
|
||||
# Quant Engine retrospective v2 compatibility
|
||||
|
||||
Scope: implement the user-authorized retrospective v2 compatibility without changing
|
||||
v1 semantics, financial algorithms, original results, production databases, deployment
|
||||
or trading. No claim of complete Quant OS delivery or real-data qualification.
|
||||
|
||||
Branch: `codex/research-quant-os-retrospective-contract-v2-20260908`.
|
||||
Declared base: accepted `main@68dd68392a26251391fbdae40c22eee370adb56e`.
|
||||
One isolated delivery worktree; the old primary checkout is preserved. This is not
|
||||
reactivation of an old registered stage or creation of a new stage ledger.
|
||||
|
||||
## Dependency baseline
|
||||
|
||||
Public RP data-contract candidate: PR #100, initial schemas/goldens at
|
||||
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (review/acceptance pending).
|
||||
EDB mapping candidate: PR #13, initial implementation `88433df`, local full
|
||||
validation passed. Neither candidate is silently treated as accepted owner evidence.
|
||||
The shared public major is 2.0.0; preserve the accepted v1 paths independently.
|
||||
|
||||
Order: public data contracts -> EDB mapping/Foundation -> Quant Engine typed
|
||||
factor/backtest/portfolio/risk -> RP governance -> Research Results -> RP read.
|
||||
Accepted owner-version bindings and runtime admission must still close every boundary.
|
||||
|
||||
## Internal reuse decision
|
||||
|
||||
Need: carry observation-aware inputs and retrospective-only claims through computation.
|
||||
Existing: strict canonical JSON, immutable envelopes, factor definitions, input/output
|
||||
closure, numerical algorithms, governed backtest and portfolio/risk contracts.
|
||||
External candidates: not needed; this is project-owned semantics, not a missing library.
|
||||
Approach: reuse those primitives and algorithms; introduce explicit new-major wrappers
|
||||
only where upstream identity, time or usage semantics change.
|
||||
Risk: reusing the v1 decoder or coercing observed-by into knowledge/PIT would make a
|
||||
false historical claim. Unknown versions and unsupported usages must fail closed.
|
||||
|
||||
## Implemented, not yet accepted or released
|
||||
|
||||
Five separate v2 modules now implement immutable DatasetSnapshot/Foundation decoding
|
||||
and materialized-content verification, FactorSet with explicit v2 nested bindings,
|
||||
BacktestRunRef and replay ancestry, nine-table BacktestEvidenceManifest,
|
||||
PerformanceEvidence, PortfolioTarget/Decision and RiskAssessment. The metadata
|
||||
registers the new major alongside every existing v1 entry. See
|
||||
`docs/RETROSPECTIVE_COMPUTATION_V2.md` for normative clocks, JSON profiles, input
|
||||
closure, replay and owner-port boundaries.
|
||||
|
||||
Factor definitions, generic output/receipt/constraint/covariance primitives and
|
||||
financial implementations are reused without semantic edits. Table schema remains
|
||||
1.1.0; v1 business source, v1 goldens, `pyproject.toml`, `uv.lock` and `ci-profile.yml`
|
||||
are unchanged. Package version remains unreleased. Only module metadata, its exact
|
||||
inventory test and README gain v2 alongside the new files.
|
||||
|
||||
The frozen synthetic computation vector includes fresh factor/backtest/manifest/
|
||||
performance/target/portfolio/risk documents and synthetic artifact tables. It is
|
||||
explicitly **envelope-only**, not an end-to-end claim that the one-day data fixture
|
||||
produced the four-day synthetic financial artifact. No old real run was rerun,
|
||||
retagged or backdated.
|
||||
|
||||
## Local verification (2026-09-08)
|
||||
|
||||
- Full repository unit suite: **1039 passed**, 1166 warnings, 31.50 seconds.
|
||||
- S4 focused new + unchanged v1 contracts: **120 passed**; new S4 332 statements,
|
||||
20 branches, 100% measured coverage. Coverage is not source authentication or
|
||||
proof of complete business semantics.
|
||||
- All five new source modules passed mypy; all six new test modules, five new
|
||||
sources and the updated metadata test passed Ruff.
|
||||
- The combined synthetic vector and metadata smoke checks: **3 passed**.
|
||||
- Actual negative tests reproduced and fixed missing covariance-estimation context
|
||||
in result identity, untyped malformed-JSON errors, and risk-time stale-manifest
|
||||
reuse. Other modules' earlier RED/GREEN evidence remains part of the same turn.
|
||||
|
||||
The full suite was run directly against the frozen local environment. This is not
|
||||
the same claim as remote CI or central ship acceptance; the unchanged declared CI
|
||||
profile is `lite` with the module-metadata smoke command. Central validation and
|
||||
Draft PR creation follow the implementation commit. No Ready, merge, accepted
|
||||
upstream binding or independent-review pass is claimed here.
|
||||
|
||||
Actual computation/admission times are distinct from simulated business dates. New
|
||||
formal outputs cannot inherit the old run's producer identity or be backdated to it.
|
||||
Real receipt/qualification/view/clock ports remain mandatory; typed objects and hashes
|
||||
are not source authentication. The optional independent reviewer delegation is still
|
||||
awaiting the already-requested user choice.
|
||||
|
||||
Next: preserve the candidate for review, then carry explicit v2 facts through
|
||||
RP governance -> Research Results publication -> RP read compatibility. Bind final
|
||||
accepted upstream versions only when actual acceptance evidence exists. The entire
|
||||
Quant OS goal is not complete at this intermediate owner unit.
|
||||
|
||||
Rollback: disable the explicit v2 path and retain v1 plus immutable artifacts; never
|
||||
retag v2 into v1 or silently use synthetic evidence for real admission.
|
||||
@@ -0,0 +1,489 @@
|
||||
"""Retrospective-only evidence wrappers over the unchanged research fact tables."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Self, cast
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from quant_engine.artifact import (
|
||||
BacktestEvidenceEntry,
|
||||
EvidenceQualification,
|
||||
ResearchRunArtifact,
|
||||
PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
PERFORMANCE_METHODOLOGY_ID,
|
||||
PerformanceMethodology,
|
||||
PerformanceMetric,
|
||||
_PERFORMANCE_SOURCE_COLUMNS,
|
||||
_absolute_performance_metrics,
|
||||
_benchmark_context,
|
||||
_count_performance_metrics,
|
||||
_performance_canonical_bytes,
|
||||
_performance_compare,
|
||||
_performance_date,
|
||||
_performance_digest,
|
||||
_performance_methodology,
|
||||
_performance_text,
|
||||
_performance_validate_tree,
|
||||
_relative_performance_metrics,
|
||||
_artifact_frames,
|
||||
_evidence_entries,
|
||||
_evidence_frame_records,
|
||||
_manifest_instant,
|
||||
_run_row,
|
||||
_table_evidence,
|
||||
_validate_table_run_ids,
|
||||
)
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
FactorContractError,
|
||||
_assert_canonical_profile,
|
||||
_content_address,
|
||||
_digest_bytes,
|
||||
_duplicate_key_pairs,
|
||||
_freeze_json,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_data_contracts import _check, _public, _shape
|
||||
|
||||
|
||||
def _validated_run(run: Any) -> RetrospectiveBacktestRunRef:
|
||||
_check(
|
||||
type(run) is RetrospectiveBacktestRunRef,
|
||||
"$.run_ref",
|
||||
"explicit v2 run reference required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
factor = run._factor_set
|
||||
return RetrospectiveBacktestRunRef.from_dict(
|
||||
run.to_dict(),
|
||||
dataset_snapshot=factor._dataset_snapshot,
|
||||
foundation=factor._foundation,
|
||||
factor_set=factor,
|
||||
parent=run._parent,
|
||||
)
|
||||
|
||||
|
||||
def _validated_frames(
|
||||
artifact: ResearchRunArtifact, run: RetrospectiveBacktestRunRef
|
||||
) -> dict[str, pd.DataFrame]:
|
||||
frames = _artifact_frames(artifact)
|
||||
_validate_table_run_ids(frames, run.run_id)
|
||||
row = _run_row(frames)
|
||||
expected = {
|
||||
"run_id": run.run_id,
|
||||
"data_snapshot_id": run.dataset_snapshot_id,
|
||||
"strategy_id": run.strategy_id,
|
||||
"strategy_version": run.strategy_version,
|
||||
"code_revision": run.code_revision,
|
||||
"config_hash": run.configuration_digest.removeprefix("sha256:"),
|
||||
"schema_version": artifact.schema_version,
|
||||
}
|
||||
_check(
|
||||
set(expected) | {"started_at", "finished_at"} <= set(row.index),
|
||||
"$.artifact.tables.run",
|
||||
"run schema fields missing",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for key, value in expected.items():
|
||||
_check(
|
||||
type(row[key]) is str and row[key] == value,
|
||||
f"$.artifact.tables.run.{key}",
|
||||
"artifact does not bind exact v2 run",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
# The unchanged artifact 1.1 timestamp profile admits offsets; public v2
|
||||
# envelope times remain strict UTC. No knowledge-time inference is performed.
|
||||
_, started = _manifest_instant(row["started_at"], "$.artifact.tables.run.started_at")
|
||||
_, finished = _manifest_instant(row["finished_at"], "$.artifact.tables.run.finished_at")
|
||||
_check(
|
||||
_parse_utc(run.evaluation_at, "$.run_ref.evaluation_at")
|
||||
<= started
|
||||
<= finished
|
||||
<= _parse_utc(run.computed_at, "$.run_ref.computed_at"),
|
||||
"$.artifact.tables.run",
|
||||
"actual evaluation <= start <= finish <= computed required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
for name, frame in frames.items():
|
||||
_public(_evidence_frame_records(frame, name), f"$.artifact.tables.{name}")
|
||||
return frames
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveBacktestEvidenceManifest:
|
||||
"""Exact artifact closure, not authenticity, historical or execution authority."""
|
||||
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
manifest_id: str
|
||||
run_id: str
|
||||
profile: str
|
||||
artifact_schema_version: str
|
||||
artifact_available_at: str
|
||||
qualification: EvidenceQualification
|
||||
evidence_digest: str
|
||||
evidence: tuple[BacktestEvidenceEntry, ...]
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
observation_cutoff: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_artifact: ResearchRunArtifact = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
artifact: ResearchRunArtifact,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
) -> Self:
|
||||
_assert_canonical_profile(value)
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version manifest_id run_id profile artifact_schema_version artifact_available_at qualification "
|
||||
"run_reference evidence_digest evidence evidence_scope usage historical_availability observation_cutoff decision_eligible execution_validation",
|
||||
)
|
||||
_check(
|
||||
type(row["qualification"]) is str
|
||||
and row["qualification"] in {"exploratory", "contract_qualified"},
|
||||
"$.qualification",
|
||||
"explicit non-legacy contract qualification required",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
rebuilt = build_retrospective_backtest_evidence_manifest(
|
||||
backtest_run_ref,
|
||||
artifact,
|
||||
artifact_available_at=row["artifact_available_at"],
|
||||
qualification=EvidenceQualification(row["qualification"]),
|
||||
)
|
||||
_check(
|
||||
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
|
||||
"$",
|
||||
"manifest differs from actual run/table closure",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_backtest_evidence_manifest(
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
artifact: ResearchRunArtifact,
|
||||
*,
|
||||
artifact_available_at: str,
|
||||
qualification: EvidenceQualification = EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
expected_table_digests: Mapping[str, str] | None = None,
|
||||
) -> RetrospectiveBacktestEvidenceManifest:
|
||||
"""Close new in-memory artifact bytes; never promote an old exploratory run."""
|
||||
run = _validated_run(backtest_run_ref)
|
||||
_check(
|
||||
type(qualification) is EvidenceQualification
|
||||
and qualification is not EvidenceQualification.LEGACY_EXPLORATORY,
|
||||
"$.qualification",
|
||||
"legacy evidence cannot enter the v2 path",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
available = _parse_utc(artifact_available_at, "$.artifact_available_at")
|
||||
_check(
|
||||
_parse_utc(run.computed_at, "$.run_ref.computed_at") <= available,
|
||||
"$.artifact_available_at",
|
||||
"artifact precedes actual computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
frames = _validated_frames(artifact, run)
|
||||
summaries = _table_evidence(frames, expected_table_digests)
|
||||
reference: dict[str, object] = {"kind": "backtest_run_ref", "value": run.to_dict()}
|
||||
# Table categories and canonical content hashing have not changed semantics.
|
||||
evidence = _evidence_entries(summaries, reference, legacy=False)
|
||||
evidence_digest = _digest_bytes(canonical_json_bytes([item.to_dict() for item in evidence]))
|
||||
payload = {
|
||||
"contract_name": "researchhub.backtest-evidence-manifest",
|
||||
"schema_version": "2.0.0",
|
||||
"run_id": run.run_id,
|
||||
"profile": "offline_research_retrospective_v2",
|
||||
"artifact_schema_version": artifact.schema_version,
|
||||
"artifact_available_at": artifact_available_at,
|
||||
"qualification": qualification.value,
|
||||
"run_reference": reference,
|
||||
"evidence_digest": evidence_digest,
|
||||
"evidence": [item.to_dict() for item in evidence],
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
payload["manifest_id"] = _content_address(
|
||||
payload, "manifest_id", "rhbacktestevidencev2:sha256:"
|
||||
)
|
||||
instance = object.__new__(RetrospectiveBacktestEvidenceManifest)
|
||||
values = {
|
||||
**payload,
|
||||
"qualification": qualification,
|
||||
"evidence": evidence,
|
||||
"backtest_run_ref": run,
|
||||
"_payload": _freeze_json(payload),
|
||||
"_artifact": artifact,
|
||||
}
|
||||
del values["run_reference"]
|
||||
for name, value in values.items():
|
||||
object.__setattr__(instance, name, value)
|
||||
return instance
|
||||
|
||||
|
||||
def _freeze_numeric_evidence(value: Any) -> Any:
|
||||
"""Freeze the existing finite-number metric profile, not the data JSON profile."""
|
||||
if type(value) is dict:
|
||||
return MappingProxyType(
|
||||
{key: _freeze_numeric_evidence(item) for key, item in value.items()}
|
||||
)
|
||||
if type(value) is list:
|
||||
return tuple(_freeze_numeric_evidence(item) for item in value)
|
||||
return value
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePerformanceEvidence:
|
||||
"""New upstream/time identity; unchanged finite-number metric/methodology v1."""
|
||||
|
||||
methodology: PerformanceMethodology
|
||||
metrics: tuple[PerformanceMetric, ...]
|
||||
_payload: Mapping[str, Any] = field(repr=False)
|
||||
|
||||
@property
|
||||
def performance_evidence_id(self) -> str:
|
||||
return cast(str, self._payload["performance_evidence_id"])
|
||||
|
||||
@property
|
||||
def document_sha256(self) -> str:
|
||||
return cast(str, self._payload["document_sha256"])
|
||||
|
||||
@property
|
||||
def run_id(self) -> str:
|
||||
return cast(str, self._payload["run_id"])
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def canonical_bytes(self) -> bytes:
|
||||
return _performance_canonical_bytes(self.to_dict())
|
||||
|
||||
def to_json(self) -> str:
|
||||
return self.canonical_bytes().decode("utf-8")
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
artifact: ResearchRunArtifact,
|
||||
run_ref: RetrospectiveBacktestRunRef,
|
||||
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
rebuilt = build_retrospective_performance_evidence(artifact, run_ref, evidence_manifest)
|
||||
_performance_compare(value, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
_check(
|
||||
type(value) in {str, bytes},
|
||||
"$",
|
||||
"canonical JSON text/bytes required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
raw = value.encode("utf-8") if isinstance(value, str) else value
|
||||
try:
|
||||
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
|
||||
except (UnicodeDecodeError, json.JSONDecodeError) as error:
|
||||
raise FactorContractError(
|
||||
ContractErrorCode.INVALID_FORMAT, "$", "invalid performance evidence JSON"
|
||||
) from error
|
||||
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
|
||||
_check(
|
||||
_performance_canonical_bytes(document) == raw,
|
||||
"$",
|
||||
"canonical finite-number JSON required",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
return cls.from_dict(document, **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_performance_evidence(
|
||||
artifact: ResearchRunArtifact,
|
||||
run_ref: RetrospectiveBacktestRunRef,
|
||||
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
) -> RetrospectivePerformanceEvidence:
|
||||
"""Bind current tables and existing methodology; no performance recalculation."""
|
||||
run = _validated_run(run_ref)
|
||||
_check(
|
||||
type(evidence_manifest) is RetrospectiveBacktestEvidenceManifest,
|
||||
"$.evidence_manifest",
|
||||
"explicit v2 manifest required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
evidence_manifest.to_dict(), artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
frames = _validated_frames(artifact, run)
|
||||
performance = frames["performance"]
|
||||
_check(
|
||||
len(performance) == 1
|
||||
and tuple(str(column) for column in performance.columns) == _PERFORMANCE_SOURCE_COLUMNS,
|
||||
"$.artifact.tables.performance",
|
||||
"one row in the unchanged closed performance schema required",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
performance_row = performance.iloc[0]
|
||||
run_row = _run_row(frames)
|
||||
frequency = _performance_text(run_row["frequency"], "$.artifact.tables.run.frequency")
|
||||
_check(
|
||||
frequency == "1d",
|
||||
"$.artifact.tables.run.frequency",
|
||||
"only existing daily methodology is supported",
|
||||
)
|
||||
calendar = _performance_text(run_row["calendar"], "$.artifact.tables.run.calendar")
|
||||
timezone = _performance_text(run_row["timezone"], "$.artifact.tables.run.timezone")
|
||||
start_date = _performance_date(run_row["start_date"], "$.artifact.tables.run.start_date")
|
||||
end_date = _performance_date(run_row["end_date"], "$.artifact.tables.run.end_date")
|
||||
nav = frames["nav"]
|
||||
_check(
|
||||
not nav.empty,
|
||||
"$.artifact.tables.nav",
|
||||
"NAV observation window required",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
_performance_date(nav.iloc[0]["trade_date"], "$.artifact.tables.nav.start") == start_date
|
||||
and _performance_date(nav.iloc[-1]["trade_date"], "$.artifact.tables.nav.end") == end_date,
|
||||
"$.artifact.tables.nav",
|
||||
"observation window differs from artifact dates",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
benchmark_digest, active_std, benchmark_variance, alpha_domain_unestimable = _benchmark_context(
|
||||
frames, run_row, performance_row
|
||||
)
|
||||
metrics = (
|
||||
*_absolute_performance_metrics(performance_row),
|
||||
*_relative_performance_metrics(
|
||||
performance_row,
|
||||
benchmark_present=benchmark_digest is not None,
|
||||
active_std=active_std,
|
||||
benchmark_variance=benchmark_variance,
|
||||
alpha_domain_unestimable=alpha_domain_unestimable,
|
||||
),
|
||||
*_count_performance_metrics(performance_row),
|
||||
)
|
||||
normalized_row: dict[str, object] = {metric.source_column: metric.value for metric in metrics}
|
||||
normalized_row["run_id"] = run.run_id
|
||||
row_digest = _performance_digest(
|
||||
{"columns": list(_PERFORMANCE_SOURCE_COLUMNS), "row": normalized_row}
|
||||
)
|
||||
alignment = cast(str, run_row["benchmark_alignment_policy"])
|
||||
methodology = _performance_methodology(
|
||||
frequency=frequency, alignment=alignment, code_revision=run.code_revision
|
||||
)
|
||||
performance_table = next(
|
||||
table
|
||||
for entry in manifest.evidence
|
||||
for table in entry.tables
|
||||
if table.logical_name == "performance"
|
||||
)
|
||||
run_document = run.to_dict()
|
||||
payload: dict[str, Any] = {
|
||||
"schema_version": "researchhub.performance-evidence.v2",
|
||||
"authority": "quant_engine",
|
||||
"scope": "offline_retrospective_research_only",
|
||||
"run_id": run.run_id,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
"backtest_run_ref_id": run.run_id,
|
||||
"backtest_run_ref_document_sha256": _digest_bytes(canonical_json_bytes(run_document)),
|
||||
"backtest_evidence_manifest_id": manifest.manifest_id,
|
||||
"backtest_evidence_manifest_document_sha256": _digest_bytes(
|
||||
canonical_json_bytes(manifest.to_dict())
|
||||
),
|
||||
"backtest_evidence_manifest_evidence_digest": manifest.evidence_digest,
|
||||
"backtest_evidence_qualification": manifest.qualification.value,
|
||||
"research_artifact_schema_version": artifact.schema_version,
|
||||
"research_artifact_content_digest": "sha256:" + artifact.content_sha256,
|
||||
"artifact_available_at": manifest.artifact_available_at,
|
||||
"computed_at": run.computed_at,
|
||||
"performance_table_logical_name": performance_table.logical_name,
|
||||
"performance_table_row_count": performance_table.row_count,
|
||||
"performance_table_schema_digest": performance_table.schema_digest,
|
||||
"performance_table_content_digest": performance_table.content_digest,
|
||||
"performance_row_digest": row_digest,
|
||||
"benchmark_series_digest": benchmark_digest,
|
||||
"methodology_id": PERFORMANCE_METHODOLOGY_ID,
|
||||
"metric_schema_id": PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
**{
|
||||
key: run_document[key]
|
||||
for key in (
|
||||
"dataset_snapshot_id",
|
||||
"dataset_content_digest",
|
||||
"dataset_manifest_digest",
|
||||
"foundation_id",
|
||||
"foundation_digest",
|
||||
"factor_set_id",
|
||||
"factor_set_digest",
|
||||
"factor_output_content_digest",
|
||||
"strategy_id",
|
||||
"strategy_version",
|
||||
"strategy_digest",
|
||||
"execution_model_version",
|
||||
"execution_model_digest",
|
||||
"cost_model_version",
|
||||
"cost_model_digest",
|
||||
"code_revision",
|
||||
"environment_lock_digest",
|
||||
"configuration_digest",
|
||||
)
|
||||
},
|
||||
"frequency": frequency,
|
||||
"calendar": calendar,
|
||||
"timezone": timezone,
|
||||
"benchmark_id": run_row["benchmark_id"],
|
||||
"benchmark_alignment_policy": alignment,
|
||||
"start_date": start_date,
|
||||
"end_date": end_date,
|
||||
"methodology": methodology.to_dict(),
|
||||
"metrics": [metric.to_dict() for metric in metrics],
|
||||
}
|
||||
payload["performance_evidence_id"] = "rhperformancev2:" + _performance_digest(payload)
|
||||
payload["document_sha256"] = _performance_digest(payload)
|
||||
instance = object.__new__(RetrospectivePerformanceEvidence)
|
||||
object.__setattr__(instance, "_payload", _freeze_numeric_evidence(payload))
|
||||
object.__setattr__(instance, "methodology", methodology)
|
||||
object.__setattr__(instance, "metrics", metrics)
|
||||
return instance
|
||||
@@ -0,0 +1,422 @@
|
||||
"""Explicit retrospective v2 run identities and offline artifact evidence."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Self, cast
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
PayloadValidation,
|
||||
_assert_canonical_profile,
|
||||
_content_address,
|
||||
_digest,
|
||||
_digest_bytes,
|
||||
_freeze_json,
|
||||
_git_revision,
|
||||
_logical_id,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_safe_integer,
|
||||
_semver,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
_IDS,
|
||||
_check,
|
||||
_public,
|
||||
_shape,
|
||||
_strings,
|
||||
)
|
||||
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef, _context
|
||||
|
||||
_CONFIG_FIELDS = (
|
||||
"universe_digest",
|
||||
"strategy_id",
|
||||
"strategy_version",
|
||||
"strategy_digest",
|
||||
"execution_model_version",
|
||||
"execution_model_digest",
|
||||
"cost_model_version",
|
||||
"cost_model_digest",
|
||||
"random_seed",
|
||||
"code_revision",
|
||||
"environment_lock_digest",
|
||||
"configuration_digest",
|
||||
)
|
||||
_RUN_FIELDS = (
|
||||
"contract_name schema_version run_id dataset_snapshot_id dataset_content_digest dataset_manifest_digest "
|
||||
"foundation_id foundation_digest factor_set_id factor_set_digest factor_output_content_digest "
|
||||
"observation_cutoff evidence_scope usage historical_availability decision_eligible execution_validation "
|
||||
"universe_digest trading_calendar_revision_ids trading_calendar_digest corporate_action_revision_ids corporate_action_digest "
|
||||
"strategy_id strategy_version strategy_digest execution_model_version execution_model_digest cost_model_version cost_model_digest "
|
||||
"random_seed code_revision environment_lock_digest configuration_digest evaluation_at computed_at replay_spec_digest "
|
||||
"replay_parent_run_id replay_reason replay_attempt replay_ancestor_run_ids"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveBacktestRunRef:
|
||||
"""New-major deterministic-input identity with separate actual attempt times."""
|
||||
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
run_id: str
|
||||
dataset_snapshot_id: str
|
||||
dataset_content_digest: str
|
||||
dataset_manifest_digest: str
|
||||
foundation_id: str
|
||||
foundation_digest: str
|
||||
factor_set_id: str
|
||||
factor_set_digest: str
|
||||
factor_output_content_digest: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
universe_digest: str
|
||||
trading_calendar_revision_ids: tuple[str, ...]
|
||||
trading_calendar_digest: str
|
||||
corporate_action_revision_ids: tuple[str, ...]
|
||||
corporate_action_digest: str
|
||||
strategy_id: str
|
||||
strategy_version: str
|
||||
strategy_digest: str
|
||||
execution_model_version: str
|
||||
execution_model_digest: str
|
||||
cost_model_version: str
|
||||
cost_model_digest: str
|
||||
random_seed: int
|
||||
code_revision: str
|
||||
environment_lock_digest: str
|
||||
configuration_digest: str
|
||||
evaluation_at: str
|
||||
computed_at: str
|
||||
replay_spec_digest: str
|
||||
replay_parent_run_id: str | None
|
||||
replay_reason: str | None
|
||||
replay_attempt: int
|
||||
replay_ancestor_run_ids: tuple[str, ...]
|
||||
input_payload_validation: PayloadValidation = field(compare=False)
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_factor_set: RetrospectiveFactorSetRef = field(repr=False, compare=False)
|
||||
_parent: RetrospectiveBacktestRunRef | None = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
factor_set: RetrospectiveFactorSetRef,
|
||||
universe_digest: str,
|
||||
trading_calendar_revision_ids: Sequence[str],
|
||||
corporate_action_revision_ids: Sequence[str],
|
||||
strategy_id: str,
|
||||
strategy_version: str,
|
||||
strategy_digest: str,
|
||||
execution_model_version: str,
|
||||
execution_model_digest: str,
|
||||
cost_model_version: str,
|
||||
cost_model_digest: str,
|
||||
random_seed: int,
|
||||
code_revision: str,
|
||||
environment_lock_digest: str,
|
||||
configuration_digest: str,
|
||||
evaluation_at: str,
|
||||
computed_at: str,
|
||||
parent: RetrospectiveBacktestRunRef | None = None,
|
||||
replay_reason: str | None = None,
|
||||
replay_attempt: int = 0,
|
||||
) -> Self:
|
||||
_check(
|
||||
type(factor_set) is RetrospectiveFactorSetRef,
|
||||
"$.factor_set",
|
||||
"explicit v2 factor result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
factor_set.require_payloads_revalidated()
|
||||
return cls._build(
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
trading_calendar_revision_ids=trading_calendar_revision_ids,
|
||||
corporate_action_revision_ids=corporate_action_revision_ids,
|
||||
configuration={
|
||||
"universe_digest": universe_digest,
|
||||
"strategy_id": strategy_id,
|
||||
"strategy_version": strategy_version,
|
||||
"strategy_digest": strategy_digest,
|
||||
"execution_model_version": execution_model_version,
|
||||
"execution_model_digest": execution_model_digest,
|
||||
"cost_model_version": cost_model_version,
|
||||
"cost_model_digest": cost_model_digest,
|
||||
"random_seed": random_seed,
|
||||
"code_revision": code_revision,
|
||||
"environment_lock_digest": environment_lock_digest,
|
||||
"configuration_digest": configuration_digest,
|
||||
},
|
||||
evaluation_at=evaluation_at,
|
||||
computed_at=computed_at,
|
||||
parent=parent,
|
||||
replay_reason=replay_reason,
|
||||
replay_attempt=replay_attempt,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def _build(
|
||||
cls,
|
||||
*,
|
||||
dataset_snapshot: Any,
|
||||
foundation: Any,
|
||||
factor_set: Any,
|
||||
trading_calendar_revision_ids: Any,
|
||||
corporate_action_revision_ids: Any,
|
||||
configuration: dict[str, Any],
|
||||
evaluation_at: Any,
|
||||
computed_at: Any,
|
||||
parent: RetrospectiveBacktestRunRef | None,
|
||||
replay_reason: Any,
|
||||
replay_attempt: Any,
|
||||
) -> Self:
|
||||
_check(
|
||||
type(factor_set) is RetrospectiveFactorSetRef,
|
||||
"$.factor_set",
|
||||
"explicit v2 factor result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
definitions, snapshot, foundation = _context(
|
||||
factor_set._definitions, dataset_snapshot, foundation
|
||||
)
|
||||
# Reconstruct the serialized factor boundary against the exact supplied inputs.
|
||||
checked_factor = RetrospectiveFactorSetRef.from_dict(
|
||||
factor_set.to_dict(),
|
||||
definitions=definitions,
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
parent=factor_set._parent,
|
||||
)
|
||||
closures: dict[str, tuple[str, ...]] = {}
|
||||
for field_name, supplied, kind in (
|
||||
("trading_calendar_revision_ids", trading_calendar_revision_ids, "calendar_revision"),
|
||||
("corporate_action_revision_ids", corporate_action_revision_ids, "action_revision"),
|
||||
):
|
||||
_check(
|
||||
type(supplied) in {tuple, list},
|
||||
f"$.{field_name}",
|
||||
"list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
supplied_ids = tuple(
|
||||
sorted(
|
||||
_strings(
|
||||
list(supplied),
|
||||
f"$.{field_name}",
|
||||
_IDS[kind],
|
||||
1 if kind == "calendar_revision" else 0,
|
||||
)
|
||||
)
|
||||
)
|
||||
expected_ids = tuple(
|
||||
sorted(
|
||||
{
|
||||
identity
|
||||
for view_id in checked_factor.selected_view_ref_ids
|
||||
for identity in getattr(foundation.views[view_id], field_name)
|
||||
}
|
||||
)
|
||||
)
|
||||
_check(
|
||||
supplied_ids == expected_ids,
|
||||
f"$.{field_name}",
|
||||
"exact selected observation ancestry required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
closures[field_name] = supplied_ids
|
||||
_shape(configuration, "$.configuration", " ".join(_CONFIG_FIELDS))
|
||||
for name, value in configuration.items():
|
||||
if name.endswith("_digest"):
|
||||
_digest(value, f"$.{name}")
|
||||
elif name.endswith("_version"):
|
||||
_semver(value, f"$.{name}")
|
||||
elif name == "random_seed":
|
||||
_safe_integer(value, f"$.{name}", minimum=0)
|
||||
elif name == "code_revision":
|
||||
_git_revision(value, f"$.{name}")
|
||||
else:
|
||||
_logical_id(value, f"$.{name}")
|
||||
_public(configuration, "$.configuration")
|
||||
evaluation = _parse_utc(evaluation_at, "$.evaluation_at")
|
||||
computed = _parse_utc(computed_at, "$.computed_at")
|
||||
_check(
|
||||
_parse_utc(checked_factor.artifact_available_at, "$.factor_set.artifact_available_at")
|
||||
<= evaluation
|
||||
<= computed,
|
||||
"$.computed_at",
|
||||
"factor availability <= actual evaluation <= computation required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
content = snapshot.to_dict()["descriptor"]["content"]
|
||||
spec: dict[str, Any] = {
|
||||
"dataset_snapshot_id": snapshot.snapshot_id,
|
||||
"dataset_content_digest": content["content_digest"],
|
||||
"dataset_manifest_digest": content["manifest_digest"],
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"foundation_digest": foundation.foundation_id.removeprefix("rhdfv2:"),
|
||||
"factor_set_id": checked_factor.factor_set_id,
|
||||
"factor_set_digest": checked_factor.factor_set_id.removeprefix("rhfactorsetv2:"),
|
||||
"factor_output_content_digest": checked_factor.output_content_digest,
|
||||
"observation_cutoff": foundation.observation_cutoff,
|
||||
"evidence_scope": checked_factor.evidence_scope,
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
**configuration,
|
||||
}
|
||||
for field_name, identities in closures.items():
|
||||
spec[field_name] = list(identities)
|
||||
digest_field = (
|
||||
"trading_calendar_digest"
|
||||
if field_name == "trading_calendar_revision_ids"
|
||||
else "corporate_action_digest"
|
||||
)
|
||||
spec[digest_field] = _digest_bytes(canonical_json_bytes(list(identities)))
|
||||
# v2 replay specification excludes BOTH actual attempt times. They remain in
|
||||
# run_id, so a replay never backdates evaluation to manufacture equality.
|
||||
replay_spec_digest = _digest_bytes(canonical_json_bytes(spec))
|
||||
replay_count = _safe_integer(replay_attempt, "$.replay_attempt", minimum=0)
|
||||
if parent is None:
|
||||
_check(
|
||||
replay_reason is None and replay_count == 0,
|
||||
"$.replay_attempt",
|
||||
"root must use zero attempt and no reason",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
parent_id = None
|
||||
ancestors: tuple[str, ...] = ()
|
||||
else:
|
||||
_check(
|
||||
type(parent) is RetrospectiveBacktestRunRef,
|
||||
"$.parent",
|
||||
"exact v2 run parent required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_logical_id(replay_reason, "$.replay_reason")
|
||||
_check(
|
||||
replay_count == parent.replay_attempt + 1
|
||||
and replay_spec_digest == parent.replay_spec_digest,
|
||||
"$.replay_spec_digest",
|
||||
"replay requires unchanged inputs and the next attempt",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(parent.computed_at, "$.parent.computed_at") < evaluation <= computed,
|
||||
"$.evaluation_at",
|
||||
"new actual attempt must follow parent computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
parent_id = parent.run_id
|
||||
ancestors = (*parent.replay_ancestor_run_ids, parent_id)
|
||||
_check(
|
||||
len(ancestors) == len(set(ancestors)),
|
||||
"$.replay_ancestor_run_ids",
|
||||
"replay cycle",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.backtest-run-ref",
|
||||
"schema_version": "2.0.0",
|
||||
**spec,
|
||||
"evaluation_at": evaluation_at,
|
||||
"computed_at": computed_at,
|
||||
"replay_spec_digest": replay_spec_digest,
|
||||
"replay_parent_run_id": parent_id,
|
||||
"replay_reason": replay_reason,
|
||||
"replay_attempt": replay_count,
|
||||
"replay_ancestor_run_ids": list(ancestors),
|
||||
}
|
||||
payload["run_id"] = _content_address(payload, "run_id", "rhbacktestrunv2:sha256:")
|
||||
_check(
|
||||
payload["run_id"] not in ancestors,
|
||||
"$.run_id",
|
||||
"self-parent cycle",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
verified = (
|
||||
factor_set.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
and factor_set.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
)
|
||||
instance = object.__new__(cls)
|
||||
for name, value in {
|
||||
**payload,
|
||||
**closures,
|
||||
"replay_ancestor_run_ids": ancestors,
|
||||
"input_payload_validation": PayloadValidation.PAYLOAD_REVALIDATED
|
||||
if verified
|
||||
else PayloadValidation.REFERENCE_ONLY,
|
||||
"_payload": _freeze_json(payload),
|
||||
"_factor_set": factor_set,
|
||||
"_parent": parent,
|
||||
}.items():
|
||||
object.__setattr__(instance, name, value)
|
||||
return instance
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
def require_inputs_revalidated(self) -> None:
|
||||
_check(
|
||||
self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
|
||||
"$.input_payload_validation",
|
||||
"reference-only factors cannot admit a new computation",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
factor_set: RetrospectiveFactorSetRef,
|
||||
parent: RetrospectiveBacktestRunRef | None = None,
|
||||
) -> Self:
|
||||
_assert_canonical_profile(value)
|
||||
_public(value)
|
||||
row = _shape(value, "$", _RUN_FIELDS)
|
||||
rebuilt = cls._build(
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
trading_calendar_revision_ids=row["trading_calendar_revision_ids"],
|
||||
corporate_action_revision_ids=row["corporate_action_revision_ids"],
|
||||
configuration={key: row[key] for key in _CONFIG_FIELDS},
|
||||
evaluation_at=row["evaluation_at"],
|
||||
computed_at=row["computed_at"],
|
||||
parent=parent,
|
||||
replay_reason=row["replay_reason"],
|
||||
replay_attempt=row["replay_attempt"],
|
||||
)
|
||||
_check(
|
||||
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
|
||||
"$",
|
||||
"serialized run differs from exact v2 input/configuration/lineage closure",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
return rebuilt
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,721 @@
|
||||
"""Observation-aware factor results; no historical, governance or execution grant."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Self, cast
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
ContractErrorCode,
|
||||
FactorDefinition,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
_DEFINITION_ID,
|
||||
_FIELD_NAME,
|
||||
_array,
|
||||
_assert_canonical_profile,
|
||||
_canonical_evidence_bytes,
|
||||
_content_address,
|
||||
_digest,
|
||||
_digest_bytes,
|
||||
_freeze_json,
|
||||
_git_revision,
|
||||
_logical_id,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_string,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
validate_factor_catalog,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
_IDS,
|
||||
_check,
|
||||
_choice,
|
||||
_public,
|
||||
_restrictions,
|
||||
_shape,
|
||||
_strings,
|
||||
)
|
||||
|
||||
_FACTOR_SET_ID = re.compile(r"^rhfactorsetv2:sha256:[0-9a-f]{64}$")
|
||||
|
||||
|
||||
def _instant_text(value: Any, path: str) -> str:
|
||||
_parse_utc(value, path)
|
||||
return cast(str, value)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveInputBinding:
|
||||
definition_id: str
|
||||
input_name: str
|
||||
view_ref_id: str
|
||||
schema_digest: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.definition_id, "$.input_bindings[].definition_id", _DEFINITION_ID)
|
||||
_string(self.input_name, "$.input_bindings[].input_name", _FIELD_NAME)
|
||||
_string(self.view_ref_id, "$.input_bindings[].view_ref_id", _IDS["view_ref"])
|
||||
_digest(self.schema_digest, "$.input_bindings[].schema_digest")
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"definition_id": self.definition_id,
|
||||
"input_name": self.input_name,
|
||||
"view_ref_id": self.view_ref_id,
|
||||
"schema_digest": self.schema_digest,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any, path: str = "$.input_bindings[]") -> Self:
|
||||
row = _shape(value, path, "definition_id input_name view_ref_id schema_digest")
|
||||
return cls(**row)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveViewAvailability:
|
||||
view_ref_id: str
|
||||
available_at: str
|
||||
evidence_digest: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.view_ref_id, "$.view_availability[].view_ref_id", _IDS["view_ref"])
|
||||
_instant_text(self.available_at, "$.view_availability[].available_at")
|
||||
_digest(self.evidence_digest, "$.view_availability[].evidence_digest")
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"view_ref_id": self.view_ref_id,
|
||||
"available_at": self.available_at,
|
||||
"evidence_digest": self.evidence_digest,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any, path: str = "$.view_availability[]") -> Self:
|
||||
return cls(**_shape(value, path, "view_ref_id available_at evidence_digest"))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveCausation:
|
||||
kind: str
|
||||
id: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
kind = _choice(self.kind, "$.causation.kind", {"foundation", "factor_set"})
|
||||
_string(
|
||||
self.id,
|
||||
"$.causation.id",
|
||||
_IDS["foundation"] if kind == "foundation" else _FACTOR_SET_ID,
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {"kind": self.kind, "id": self.id}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any) -> Self:
|
||||
return cls(**_shape(value, "$.causation", "kind id"))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ResolvedRetrospectiveView:
|
||||
"""In-memory logical bytes; no locator, source authentication or transformation claim."""
|
||||
|
||||
view_ref_id: str
|
||||
schema_bytes: bytes
|
||||
content_bytes: bytes
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.view_ref_id, "$.resolved_views[].view_ref_id", _IDS["view_ref"])
|
||||
for key, value in (
|
||||
("schema_bytes", self.schema_bytes),
|
||||
("content_bytes", self.content_bytes),
|
||||
):
|
||||
_canonical_evidence_bytes(value, f"$.resolved_views[].{key}")
|
||||
_public(json.loads(value), f"$.resolved_views[].{key}")
|
||||
|
||||
|
||||
def _typed(values: Any, expected: type[Any], path: str) -> tuple[Any, ...]:
|
||||
_check(
|
||||
type(values) in {tuple, list},
|
||||
path,
|
||||
"typed list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
all(type(value) is expected for value in values),
|
||||
path,
|
||||
f"{expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
return tuple(values)
|
||||
|
||||
|
||||
def _context(
|
||||
definitions: Sequence[FactorDefinition],
|
||||
snapshot: Any,
|
||||
foundation: Any,
|
||||
) -> tuple[
|
||||
tuple[FactorDefinition, ...], RetrospectiveSnapshotEnvelope, RetrospectiveFoundationEnvelope
|
||||
]:
|
||||
_check(
|
||||
type(snapshot) is RetrospectiveSnapshotEnvelope,
|
||||
"$.dataset_snapshot",
|
||||
"explicit v2 snapshot required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
type(foundation) is RetrospectiveFoundationEnvelope,
|
||||
"$.foundation",
|
||||
"explicit v2 foundation required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot.to_dict())
|
||||
snapshot.require_qualified()
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation.to_dict(), snapshot=snapshot)
|
||||
supplied = _typed(definitions, FactorDefinition, "$.definitions")
|
||||
# Definitions stay v1, but are parsed again so mutable/caller summaries are not authority.
|
||||
normalized = validate_factor_catalog(
|
||||
tuple(FactorDefinition.from_dict(item.to_dict()) for item in supplied)
|
||||
)
|
||||
return normalized, snapshot, foundation
|
||||
|
||||
|
||||
def _upstream(
|
||||
snapshot: RetrospectiveSnapshotEnvelope, foundation: RetrospectiveFoundationEnvelope
|
||||
) -> dict[str, Any]:
|
||||
descriptor = snapshot.to_dict()["descriptor"]
|
||||
return {
|
||||
"dataset_snapshot_id": snapshot.snapshot_id,
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"evidence_scope": snapshot.evidence_scope,
|
||||
"content_digest": descriptor["content"]["content_digest"],
|
||||
"manifest_digest": descriptor["content"]["manifest_digest"],
|
||||
"observation_manifest_digest": _digest_bytes(
|
||||
canonical_json_bytes(descriptor["observation_manifest"])
|
||||
),
|
||||
"time_semantics": descriptor["time_semantics"],
|
||||
"quality": descriptor["quality"],
|
||||
"qualification": descriptor["qualification"],
|
||||
"foundation_readiness": foundation.to_dict()["readiness"],
|
||||
}
|
||||
|
||||
|
||||
def _input_payloads(
|
||||
snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
selected: tuple[str, ...],
|
||||
dataset_chunks: Any,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView] | None,
|
||||
) -> PayloadValidation:
|
||||
_check(
|
||||
(dataset_chunks is None) == (resolved_views is None),
|
||||
"$.input_payloads",
|
||||
"snapshot chunks and resolved views must be supplied together",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
if dataset_chunks is None:
|
||||
return PayloadValidation.REFERENCE_ONLY
|
||||
snapshot.verify_materialized_records(dataset_chunks)
|
||||
views = _typed(resolved_views, ResolvedRetrospectiveView, "$.resolved_views")
|
||||
view_ids = [view.view_ref_id for view in views]
|
||||
_check(
|
||||
len(view_ids) == len(selected) and set(view_ids) == set(selected),
|
||||
"$.resolved_views",
|
||||
"resolved view closure mismatch",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for item in views:
|
||||
declared = foundation.views[item.view_ref_id]
|
||||
# Recheck canonical bytes even for caller-constructed typed payloads.
|
||||
schema = _canonical_evidence_bytes(item.schema_bytes, "$.resolved_views[].schema_bytes")
|
||||
content = _canonical_evidence_bytes(item.content_bytes, "$.resolved_views[].content_bytes")
|
||||
_public(json.loads(schema))
|
||||
_public(json.loads(content))
|
||||
_check(
|
||||
_digest_bytes(schema) == declared.schema_digest
|
||||
and _digest_bytes(content) == declared.content_digest,
|
||||
"$.resolved_views",
|
||||
"view bytes do not match Foundation",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
return PayloadValidation.PAYLOAD_REVALIDATED
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveFactorSetRef:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
factor_set_id: str
|
||||
definition_ids: tuple[str, ...]
|
||||
dataset_snapshot_id: str
|
||||
foundation_id: str
|
||||
observation_cutoff: str
|
||||
selected_view_ref_ids: tuple[str, ...]
|
||||
input_bindings: tuple[RetrospectiveInputBinding, ...]
|
||||
view_availability: tuple[RetrospectiveViewAvailability, ...]
|
||||
upstream_evidence: Mapping[str, Any]
|
||||
output_quality: OutputQuality
|
||||
output_coverage: OutputCoverage
|
||||
output_schema_digest: str
|
||||
output_content_digest: str
|
||||
output_artifact_ref: OutputArtifactRef
|
||||
availability_mode: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
evaluation_at: str
|
||||
computed_at: str
|
||||
artifact_available_at: str
|
||||
producer: ProducerIdentity
|
||||
code_revision: str
|
||||
actor: ActorIdentity
|
||||
correlation_id: str
|
||||
causation: RetrospectiveCausation
|
||||
evidence_scope: str
|
||||
decision_eligible: bool
|
||||
payload_validation: PayloadValidation = field(compare=False)
|
||||
input_payload_validation: PayloadValidation = field(compare=False)
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_definitions: tuple[FactorDefinition, ...] = field(repr=False, compare=False)
|
||||
_dataset_snapshot: RetrospectiveSnapshotEnvelope = field(repr=False, compare=False)
|
||||
_foundation: RetrospectiveFoundationEnvelope = field(repr=False, compare=False)
|
||||
_parent: RetrospectiveFactorSetRef | None = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
definitions: Sequence[FactorDefinition],
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
selected_view_ref_ids: Sequence[str],
|
||||
input_bindings: Sequence[RetrospectiveInputBinding],
|
||||
view_availability: Sequence[RetrospectiveViewAvailability],
|
||||
dataset_chunks: Any,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView],
|
||||
output_quality: OutputQuality,
|
||||
output_coverage: OutputCoverage,
|
||||
output_schema_bytes: bytes,
|
||||
output_content_bytes: bytes,
|
||||
output_artifact_ref: OutputArtifactRef,
|
||||
evaluation_at: str,
|
||||
computed_at: str,
|
||||
artifact_available_at: str,
|
||||
producer: ProducerIdentity,
|
||||
code_revision: str,
|
||||
actor: ActorIdentity,
|
||||
correlation_id: str,
|
||||
causation: RetrospectiveCausation,
|
||||
evidence_scope: str,
|
||||
decision_eligible: bool,
|
||||
parent: RetrospectiveFactorSetRef | None = None,
|
||||
) -> Self:
|
||||
definitions, dataset_snapshot, foundation = _context(
|
||||
definitions, dataset_snapshot, foundation
|
||||
)
|
||||
_check(
|
||||
type(selected_view_ref_ids) in {list, tuple},
|
||||
"$.selected_view_ref_ids",
|
||||
"list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
selected = sorted(
|
||||
_strings(list(selected_view_ref_ids), "$.selected_view_ref_ids", _IDS["view_ref"], 1)
|
||||
)
|
||||
bindings = sorted(
|
||||
_typed(input_bindings, RetrospectiveInputBinding, "$.input_bindings"),
|
||||
key=lambda item: (item.definition_id, item.input_name),
|
||||
)
|
||||
availability = sorted(
|
||||
_typed(view_availability, RetrospectiveViewAvailability, "$.view_availability"),
|
||||
key=lambda item: item.view_ref_id,
|
||||
)
|
||||
for value, expected, path in (
|
||||
(output_quality, OutputQuality, "$.output_quality"),
|
||||
(output_coverage, OutputCoverage, "$.output_coverage"),
|
||||
(output_artifact_ref, OutputArtifactRef, "$.output_artifact_ref"),
|
||||
(producer, ProducerIdentity, "$.producer"),
|
||||
(actor, ActorIdentity, "$.actor"),
|
||||
(causation, RetrospectiveCausation, "$.causation"),
|
||||
):
|
||||
_check(
|
||||
type(value) is expected,
|
||||
path,
|
||||
f"{expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
schema_digest = _digest_bytes(
|
||||
_canonical_evidence_bytes(output_schema_bytes, "$.output_schema_bytes")
|
||||
)
|
||||
content_digest = _digest_bytes(
|
||||
_canonical_evidence_bytes(output_content_bytes, "$.output_content_bytes")
|
||||
)
|
||||
document = {
|
||||
"contract_name": "researchhub.factor-set-ref",
|
||||
"schema_version": "2.0.0",
|
||||
"definition_ids": [definition.definition_id for definition in definitions],
|
||||
"dataset_snapshot_id": dataset_snapshot.snapshot_id,
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"observation_cutoff": foundation.observation_cutoff,
|
||||
"selected_view_ref_ids": selected,
|
||||
"input_bindings": [item.to_dict() for item in bindings],
|
||||
"view_availability": [item.to_dict() for item in availability],
|
||||
"upstream_evidence": _upstream(dataset_snapshot, foundation),
|
||||
"output_quality": output_quality.to_dict(),
|
||||
"output_coverage": output_coverage.to_dict(),
|
||||
"output_schema_digest": schema_digest,
|
||||
"output_content_digest": content_digest,
|
||||
"output_artifact_ref": output_artifact_ref.to_dict(),
|
||||
"availability_mode": "retrospective_replay",
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"evaluation_at": evaluation_at,
|
||||
"computed_at": computed_at,
|
||||
"artifact_available_at": artifact_available_at,
|
||||
"producer": producer.to_dict(),
|
||||
"code_revision": code_revision,
|
||||
"actor": actor.to_dict(),
|
||||
"correlation_id": correlation_id,
|
||||
"causation": causation.to_dict(),
|
||||
"evidence_scope": evidence_scope,
|
||||
"decision_eligible": decision_eligible,
|
||||
}
|
||||
document["factor_set_id"] = _content_address(
|
||||
document, "factor_set_id", "rhfactorsetv2:sha256:"
|
||||
)
|
||||
result = cls.from_dict(
|
||||
document,
|
||||
definitions=definitions,
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
parent=parent,
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
dataset_chunks=dataset_chunks,
|
||||
resolved_views=resolved_views,
|
||||
)
|
||||
result.require_payloads_revalidated()
|
||||
return result
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
definitions: Sequence[FactorDefinition],
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
parent: RetrospectiveFactorSetRef | None = None,
|
||||
output_schema_bytes: bytes | None = None,
|
||||
output_content_bytes: bytes | None = None,
|
||||
dataset_chunks: Any = None,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView] | None = None,
|
||||
) -> Self:
|
||||
definitions, dataset_snapshot, foundation = _context(
|
||||
definitions, dataset_snapshot, foundation
|
||||
)
|
||||
_assert_canonical_profile(value)
|
||||
_public(value)
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version factor_set_id definition_ids dataset_snapshot_id foundation_id observation_cutoff "
|
||||
"selected_view_ref_ids input_bindings view_availability upstream_evidence output_quality output_coverage "
|
||||
"output_schema_digest output_content_digest output_artifact_ref availability_mode usage historical_availability "
|
||||
"evaluation_at computed_at artifact_available_at producer code_revision actor correlation_id causation evidence_scope decision_eligible",
|
||||
)
|
||||
_choice(row["contract_name"], "$.contract_name", {"researchhub.factor-set-ref"})
|
||||
_choice(row["schema_version"], "$.schema_version", {"2.0.0"})
|
||||
_choice(row["availability_mode"], "$.availability_mode", {"retrospective_replay"})
|
||||
_restrictions(row, "$")
|
||||
_check(
|
||||
type(row["decision_eligible"]) is bool and not row["decision_eligible"],
|
||||
"$.decision_eligible",
|
||||
"computation is never decision eligible",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
_check(
|
||||
row["dataset_snapshot_id"] == dataset_snapshot.snapshot_id
|
||||
and row["foundation_id"] == foundation.foundation_id
|
||||
and row["observation_cutoff"] == foundation.observation_cutoff,
|
||||
"$.foundation_id",
|
||||
"exact snapshot/foundation/cutoff required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
definition_ids = _strings(row["definition_ids"], "$.definition_ids", _DEFINITION_ID, 1)
|
||||
_check(
|
||||
definition_ids == tuple(item.definition_id for item in definitions),
|
||||
"$.definition_ids",
|
||||
"normalized exact definitions required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
selected = _strings(
|
||||
row["selected_view_ref_ids"], "$.selected_view_ref_ids", _IDS["view_ref"], 1
|
||||
)
|
||||
_check(
|
||||
tuple(sorted(selected)) == selected and set(selected) <= foundation.views.keys(),
|
||||
"$.selected_view_ref_ids",
|
||||
"unknown/unnormalized selected views",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
bindings = tuple(
|
||||
RetrospectiveInputBinding.from_dict(item)
|
||||
for item in _array(row["input_bindings"], "$.input_bindings", minimum=1, unique=True)
|
||||
)
|
||||
keys = [(item.definition_id, item.input_name) for item in bindings]
|
||||
expected = {
|
||||
(item.definition_id, input_spec.input_name): input_spec
|
||||
for item in definitions
|
||||
for input_spec in item.inputs
|
||||
}
|
||||
_check(
|
||||
len(keys) == len(expected) and set(keys) == expected.keys() and keys == sorted(keys),
|
||||
"$.input_bindings",
|
||||
"exact normalized factor input closure required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for binding in bindings:
|
||||
_check(
|
||||
binding.view_ref_id in selected,
|
||||
"$.input_bindings",
|
||||
"unselected view",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
binding.schema_digest
|
||||
== expected[(binding.definition_id, binding.input_name)].schema_digest
|
||||
== foundation.views[binding.view_ref_id].schema_digest,
|
||||
"$.input_bindings",
|
||||
"schema mismatch",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
{item.view_ref_id for item in bindings} == set(selected),
|
||||
"$.selected_view_ref_ids",
|
||||
"unused selected view",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
availability = tuple(
|
||||
RetrospectiveViewAvailability.from_dict(item)
|
||||
for item in _array(
|
||||
row["view_availability"], "$.view_availability", minimum=1, unique=True
|
||||
)
|
||||
)
|
||||
_check(
|
||||
tuple(item.view_ref_id for item in availability) == selected,
|
||||
"$.view_availability",
|
||||
"exact normalized selected view availability required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for item in availability:
|
||||
_check(
|
||||
item.available_at == foundation.views[item.view_ref_id].available_at,
|
||||
"$.view_availability",
|
||||
"availability must equal its Foundation fact",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
upstream = _upstream(dataset_snapshot, foundation)
|
||||
_check(
|
||||
canonical_json_bytes(row["upstream_evidence"]) == canonical_json_bytes(upstream),
|
||||
"$.upstream_evidence",
|
||||
"upstream evidence differs from complete input envelopes",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
row["evidence_scope"] == dataset_snapshot.evidence_scope == foundation.evidence_scope,
|
||||
"$.evidence_scope",
|
||||
"scope must equal both inputs",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
if row["evidence_scope"] == "real_data":
|
||||
_check(
|
||||
foundation.real_data_validation_status == "validated",
|
||||
"$.evidence_scope",
|
||||
"real-data Foundation validation required",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
quality = OutputQuality.from_dict(row["output_quality"])
|
||||
coverage = OutputCoverage.from_dict(row["output_coverage"])
|
||||
_check(
|
||||
quality.status == "passed" and all(item.status == "passed" for item in quality.checks),
|
||||
"$.output_quality",
|
||||
"all output checks must pass",
|
||||
)
|
||||
_check(
|
||||
coverage.status == "complete" and coverage.observed_count == coverage.expected_count,
|
||||
"$.output_coverage",
|
||||
"complete output coverage required",
|
||||
)
|
||||
artifact = OutputArtifactRef.from_dict(row["output_artifact_ref"])
|
||||
schema_digest = _digest(row["output_schema_digest"], "$.output_schema_digest")
|
||||
content_digest = _digest(row["output_content_digest"], "$.output_content_digest")
|
||||
_check(
|
||||
artifact.schema_digest == schema_digest and artifact.content_digest == content_digest,
|
||||
"$.output_artifact_ref",
|
||||
"output artifact mismatch",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
(output_schema_bytes is None) == (output_content_bytes is None),
|
||||
"$.output_artifact_ref",
|
||||
"both output payloads required together",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
validation = PayloadValidation.REFERENCE_ONLY
|
||||
if output_schema_bytes is not None and output_content_bytes is not None:
|
||||
for data, expected_digest, path in (
|
||||
(output_schema_bytes, schema_digest, "$.output_schema_bytes"),
|
||||
(output_content_bytes, content_digest, "$.output_content_bytes"),
|
||||
):
|
||||
canonical = _canonical_evidence_bytes(data, path)
|
||||
_public(json.loads(canonical), path)
|
||||
_check(
|
||||
_digest_bytes(canonical) == expected_digest,
|
||||
path,
|
||||
"output bytes mismatch",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
validation = PayloadValidation.PAYLOAD_REVALIDATED
|
||||
input_validation = _input_payloads(
|
||||
dataset_snapshot, foundation, selected, dataset_chunks, resolved_views
|
||||
)
|
||||
evaluation = _parse_utc(row["evaluation_at"], "$.evaluation_at")
|
||||
computed = _parse_utc(row["computed_at"], "$.computed_at")
|
||||
available = _parse_utc(row["artifact_available_at"], "$.artifact_available_at")
|
||||
_check(
|
||||
foundation.published_at <= evaluation <= computed <= available,
|
||||
"$.computed_at",
|
||||
"input publication <= actual evaluation <= computation <= artifact required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
for definition in definitions:
|
||||
_check(
|
||||
_parse_utc(definition.valid_from, "$.definitions[].valid_from")
|
||||
<= evaluation
|
||||
< _parse_utc(definition.valid_until, "$.definitions[].valid_until"),
|
||||
"$.definitions",
|
||||
"factor definition is not valid at actual evaluation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
producer = ProducerIdentity.from_dict(row["producer"])
|
||||
_check(
|
||||
producer.id == "quant_engine",
|
||||
"$.producer.id",
|
||||
"computation owner must be quant_engine",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_git_revision(row["code_revision"], "$.code_revision")
|
||||
actor = ActorIdentity.from_dict(row["actor"])
|
||||
correlation = _logical_id(row["correlation_id"], "$.correlation_id")
|
||||
cause = RetrospectiveCausation.from_dict(row["causation"])
|
||||
for name, parsed in (
|
||||
("output_quality", quality),
|
||||
("output_coverage", coverage),
|
||||
("output_artifact_ref", artifact),
|
||||
("producer", producer),
|
||||
("actor", actor),
|
||||
("causation", cause),
|
||||
):
|
||||
_check(
|
||||
canonical_json_bytes(row[name]) == canonical_json_bytes(parsed.to_dict()),
|
||||
f"$.{name}",
|
||||
"nested contract is not normalized",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
if cause.kind == "foundation":
|
||||
_check(
|
||||
cause.id == foundation.foundation_id and parent is None,
|
||||
"$.causation",
|
||||
"exact Foundation cause required",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
else:
|
||||
_check(
|
||||
type(parent) is RetrospectiveFactorSetRef,
|
||||
"$.causation",
|
||||
"exact v2 parent object required",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
assert parent is not None
|
||||
_check(
|
||||
cause.id == parent.factor_set_id
|
||||
and correlation == parent.correlation_id
|
||||
and row["evidence_scope"] == parent.evidence_scope,
|
||||
"$.causation",
|
||||
"parent identity/correlation/scope mismatch",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(parent.artifact_available_at, "$.parent.artifact_available_at")
|
||||
<= evaluation,
|
||||
"$.causation",
|
||||
"parent artifact postdates child evaluation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
factor_set_id = _string(row["factor_set_id"], "$.factor_set_id", _FACTOR_SET_ID)
|
||||
_check(
|
||||
factor_set_id == _content_address(row, "factor_set_id", "rhfactorsetv2:sha256:"),
|
||||
"$.factor_set_id",
|
||||
"factor result identity mismatch",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
cause.id != factor_set_id,
|
||||
"$.causation",
|
||||
"self parent is forbidden",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
instance = object.__new__(cls)
|
||||
values = {
|
||||
**row,
|
||||
"definition_ids": definition_ids,
|
||||
"selected_view_ref_ids": selected,
|
||||
"input_bindings": bindings,
|
||||
"view_availability": availability,
|
||||
"upstream_evidence": _freeze_json(upstream),
|
||||
"output_quality": quality,
|
||||
"output_coverage": coverage,
|
||||
"output_artifact_ref": artifact,
|
||||
"producer": producer,
|
||||
"actor": actor,
|
||||
"causation": cause,
|
||||
"payload_validation": validation,
|
||||
"input_payload_validation": input_validation,
|
||||
"_payload": _freeze_json(row),
|
||||
"_definitions": definitions,
|
||||
"_dataset_snapshot": dataset_snapshot,
|
||||
"_foundation": foundation,
|
||||
"_parent": parent,
|
||||
}
|
||||
for name, item in values.items():
|
||||
object.__setattr__(instance, name, item)
|
||||
return instance
|
||||
|
||||
def require_payloads_revalidated(self) -> None:
|
||||
_check(
|
||||
self.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
and self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
|
||||
"$.payload_validation",
|
||||
"reference-only data is not computation admission",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
@@ -0,0 +1,995 @@
|
||||
"""Retrospective-only portfolio/risk evidence with separate business/actual clocks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Self, TypedDict, cast
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from quant_engine.artifact import (
|
||||
EvidenceQualification,
|
||||
_performance_compare,
|
||||
_performance_validate_tree,
|
||||
)
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
FactorContractError,
|
||||
_duplicate_key_pairs,
|
||||
_parse_utc,
|
||||
_string,
|
||||
_thaw_json,
|
||||
)
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
ReceiptStatus,
|
||||
PortfolioRiskContractError,
|
||||
PortfolioRiskContractErrorCode,
|
||||
RiskAssessmentStatus,
|
||||
RiskFindingCode,
|
||||
_CLOSURE_ATOL,
|
||||
_CLOSURE_RTOL,
|
||||
_finite_number,
|
||||
_series_mapping,
|
||||
_validate_covariance_structure,
|
||||
_canonical_json,
|
||||
_constraint_metrics,
|
||||
_constraint_residuals,
|
||||
_digest,
|
||||
_document_sha256,
|
||||
_immutable_float_mapping,
|
||||
_mapping_dict,
|
||||
_payload_digest,
|
||||
_semver,
|
||||
_text,
|
||||
)
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
RetrospectiveBacktestEvidenceManifest,
|
||||
_freeze_numeric_evidence,
|
||||
_validated_run,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_data_contracts import _IDS, _check, _public, _shape
|
||||
from quant_engine.risk import CovarianceSnapshot, labeled_component_risk
|
||||
|
||||
_RUN_ID = re.compile(r"^rhbacktestrunv2:sha256:[0-9a-f]{64}$")
|
||||
|
||||
|
||||
def _json_object(value: str | bytes) -> dict[str, Any]:
|
||||
_check(
|
||||
type(value) in {str, bytes},
|
||||
"$",
|
||||
"canonical JSON text/bytes required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
try:
|
||||
raw = value.encode("utf-8") if isinstance(value, str) else value
|
||||
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
|
||||
except (json.JSONDecodeError, UnicodeError) as error:
|
||||
raise FactorContractError(
|
||||
ContractErrorCode.INVALID_FORMAT, "$", "valid UTF-8 JSON required"
|
||||
) from error
|
||||
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
|
||||
_performance_validate_tree(document, "$")
|
||||
_check(
|
||||
_canonical_json(document).encode() == raw,
|
||||
"$",
|
||||
"canonical numeric JSON required",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
return cast(dict[str, Any], document)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePortfolioTarget:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
target_id: str
|
||||
backtest_run_id: str
|
||||
dataset_snapshot_id: str
|
||||
weights: Mapping[str, float]
|
||||
effective_at: str
|
||||
created_at: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
backtest_run_id: str,
|
||||
dataset_snapshot_id: str,
|
||||
weights: Mapping[str, float],
|
||||
effective_at: str,
|
||||
created_at: str,
|
||||
) -> Self:
|
||||
_string(backtest_run_id, "$.backtest_run_id", _RUN_ID)
|
||||
_string(dataset_snapshot_id, "$.dataset_snapshot_id", _IDS["snapshot"])
|
||||
normalized = _immutable_float_mapping(weights, "$.weights")
|
||||
_check(bool(normalized), "$.weights", "non-empty target asset set required")
|
||||
for instrument in normalized:
|
||||
_string(instrument, "$.weights.keys", _IDS["instrument"])
|
||||
effective = _parse_utc(effective_at, "$.effective_at")
|
||||
created = _parse_utc(created_at, "$.created_at")
|
||||
_check(
|
||||
effective <= created,
|
||||
"$.effective_at",
|
||||
"historical effective time exceeds actual creation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.portfolio-target",
|
||||
"schema_version": "2.0.0",
|
||||
"backtest_run_id": backtest_run_id,
|
||||
"dataset_snapshot_id": dataset_snapshot_id,
|
||||
"weights": _mapping_dict(normalized),
|
||||
"effective_at": effective_at,
|
||||
"created_at": created_at,
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
}
|
||||
_public(payload)
|
||||
payload["target_id"] = "rhportfoliotargetv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(cls)
|
||||
for key, value in {
|
||||
**payload,
|
||||
"weights": normalized,
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
}.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version target_id backtest_run_id dataset_snapshot_id weights effective_at created_at usage historical_availability",
|
||||
)
|
||||
rebuilt = cls.create(
|
||||
**{
|
||||
key: row[key]
|
||||
for key in (
|
||||
"backtest_run_id",
|
||||
"dataset_snapshot_id",
|
||||
"weights",
|
||||
"effective_at",
|
||||
"created_at",
|
||||
)
|
||||
}
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return rebuilt
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes) -> Self:
|
||||
return cls.from_dict(_json_object(value))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _PortfolioInputs:
|
||||
run: RetrospectiveBacktestRunRef
|
||||
manifest: RetrospectiveBacktestEvidenceManifest
|
||||
target: RetrospectivePortfolioTarget
|
||||
constraints: ConstraintSetV1
|
||||
freshness: FreshnessPolicy
|
||||
weights: Mapping[str, float]
|
||||
prior: Mapping[str, float] | None
|
||||
metrics: dict[str, float | int | None]
|
||||
residuals: dict[str, float]
|
||||
input_payload: dict[str, object]
|
||||
|
||||
|
||||
def _material(
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
objective_name: str,
|
||||
objective_version: str,
|
||||
objective_digest: str,
|
||||
model_name: str,
|
||||
model_version: str,
|
||||
model_digest: str,
|
||||
expected_return_digest: str,
|
||||
covariance_digest: str,
|
||||
scenario_digest: str,
|
||||
constraints: ConstraintSetV1,
|
||||
freshness_policy: FreshnessPolicy,
|
||||
prior_weights: Mapping[str, float] | None = None,
|
||||
) -> _PortfolioInputs:
|
||||
run = _validated_run(backtest_run_ref)
|
||||
for item, expected, path in (
|
||||
(manifest, RetrospectiveBacktestEvidenceManifest, "$.manifest"),
|
||||
(target, RetrospectivePortfolioTarget, "$.target"),
|
||||
(constraints, ConstraintSetV1, "$.constraints"),
|
||||
(freshness_policy, FreshnessPolicy, "$.freshness_policy"),
|
||||
):
|
||||
_check(
|
||||
type(item) is expected,
|
||||
path,
|
||||
f"explicit {expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
checked_manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
manifest.to_dict(), artifact=manifest._artifact, backtest_run_ref=run
|
||||
)
|
||||
checked_target = RetrospectivePortfolioTarget.from_dict(target.to_dict())
|
||||
_check(
|
||||
checked_manifest.qualification is EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
"$.manifest.qualification",
|
||||
"contract-qualified retrospective S3 required",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
_check(
|
||||
checked_target.backtest_run_id == run.run_id
|
||||
and checked_target.dataset_snapshot_id == run.dataset_snapshot_id,
|
||||
"$.target",
|
||||
"target and S3 identities differ",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
foundation = run._factor_set._foundation
|
||||
selected_routes = {
|
||||
identity
|
||||
for view_id in run._factor_set.selected_view_ref_ids
|
||||
for identity in foundation.views[view_id].instrument_route_revision_ids
|
||||
}
|
||||
selected_instruments = {
|
||||
row["instrument_id"]
|
||||
for row in foundation.to_dict()["instrument_routes"]
|
||||
if row["route_revision_id"] in selected_routes
|
||||
}
|
||||
_check(
|
||||
set(checked_target.weights) <= selected_instruments,
|
||||
"$.target.weights",
|
||||
"target assets must be selected logical instruments",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
constraints = ConstraintSetV1.from_dict(constraints.to_dict())
|
||||
freshness_policy = FreshnessPolicy.from_dict(freshness_policy.to_dict())
|
||||
prior = (
|
||||
None
|
||||
if prior_weights is None
|
||||
else _immutable_float_mapping(prior_weights, "$.prior_weights")
|
||||
)
|
||||
if prior is not None:
|
||||
_check(
|
||||
set(prior) <= selected_instruments,
|
||||
"$.prior_weights",
|
||||
"prior assets outside selected instruments",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
weights = checked_target.weights
|
||||
metrics = _constraint_metrics(weights, prior)
|
||||
residuals = _constraint_residuals(constraints, weights, metrics)
|
||||
payload: dict[str, object] = {
|
||||
"contract_name": "researchhub.portfolio-computation-input",
|
||||
"schema_version": "2.0.0",
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"run_ref_document_sha256": _document_sha256(run.to_json()),
|
||||
"manifest_document_sha256": _document_sha256(checked_manifest.to_json()),
|
||||
"portfolio_target": checked_target.to_dict(),
|
||||
"objective": {
|
||||
"name": _text(objective_name, "$.objective_name"),
|
||||
"version": _semver(objective_version, "$.objective_version"),
|
||||
"digest": _digest(objective_digest, "$.objective_digest"),
|
||||
},
|
||||
"model": {
|
||||
"name": _text(model_name, "$.model_name"),
|
||||
"version": _semver(model_version, "$.model_version"),
|
||||
"digest": _digest(model_digest, "$.model_digest"),
|
||||
},
|
||||
"expected_return_digest": _digest(expected_return_digest, "$.expected_return_digest"),
|
||||
"covariance_digest": _digest(covariance_digest, "$.covariance_digest"),
|
||||
"scenario_digest": _digest(scenario_digest, "$.scenario_digest"),
|
||||
"freshness_policy_digest": _payload_digest(freshness_policy.to_dict()),
|
||||
"prior_weights": None if prior is None else _mapping_dict(prior),
|
||||
}
|
||||
_public(payload)
|
||||
return _PortfolioInputs(
|
||||
run,
|
||||
checked_manifest,
|
||||
checked_target,
|
||||
constraints,
|
||||
freshness_policy,
|
||||
weights,
|
||||
prior,
|
||||
metrics,
|
||||
residuals,
|
||||
payload,
|
||||
)
|
||||
|
||||
|
||||
def _receipt_digests(inputs: _PortfolioInputs) -> dict[str, str | float]:
|
||||
return {
|
||||
"input_digest": _payload_digest(inputs.input_payload),
|
||||
"constraint_digest": _payload_digest(inputs.constraints.to_dict()),
|
||||
"output_digest": _payload_digest(
|
||||
{
|
||||
"weights": _mapping_dict(inputs.weights),
|
||||
"metrics": inputs.metrics,
|
||||
"constraint_residuals": inputs.residuals,
|
||||
}
|
||||
),
|
||||
"max_constraint_residual": max(inputs.residuals.values(), default=0.0),
|
||||
}
|
||||
|
||||
|
||||
def compute_retrospective_portfolio_receipt_digests(**kwargs: Any) -> Mapping[str, str | float]:
|
||||
"""Recompute receipt claims; the returned digests are not producer authentication."""
|
||||
return MappingProxyType(_receipt_digests(_material(**kwargs)))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePortfolioDecision:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
decision_id: str
|
||||
run_id: str
|
||||
manifest_id: str
|
||||
evidence_digest: str
|
||||
dataset_snapshot_id: str
|
||||
run_ref_document_sha256: str
|
||||
manifest_document_sha256: str
|
||||
source_universe_digest: str
|
||||
portfolio_asset_set_digest: str
|
||||
target_id: str
|
||||
target_weights: Mapping[str, float]
|
||||
prior_weights: Mapping[str, float] | None
|
||||
objective_name: str
|
||||
objective_version: str
|
||||
objective_digest: str
|
||||
model_name: str
|
||||
model_version: str
|
||||
model_digest: str
|
||||
expected_return_digest: str
|
||||
covariance_digest: str
|
||||
scenario_digest: str
|
||||
constraints: ConstraintSetV1
|
||||
freshness_policy: FreshnessPolicy
|
||||
receipt: ComputationReceipt
|
||||
gross_exposure: float
|
||||
net_exposure: float
|
||||
turnover_l1: float | None
|
||||
position_count: int
|
||||
constraint_residuals: Mapping[str, float]
|
||||
output_digest: str
|
||||
effective_at: str
|
||||
created_at: str
|
||||
computed_at: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_target: RetrospectivePortfolioTarget = field(repr=False, compare=False)
|
||||
_run: RetrospectiveBacktestRunRef = field(repr=False, compare=False)
|
||||
_manifest: RetrospectiveBacktestEvidenceManifest = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
# Rebuild from independent typed inputs, not from a self-approved target in the wire.
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
|
||||
)
|
||||
arguments = {
|
||||
key: row[key]
|
||||
for key in (
|
||||
"objective_name",
|
||||
"objective_version",
|
||||
"objective_digest",
|
||||
"model_name",
|
||||
"model_version",
|
||||
"model_digest",
|
||||
"expected_return_digest",
|
||||
"covariance_digest",
|
||||
"scenario_digest",
|
||||
"computed_at",
|
||||
"prior_weights",
|
||||
)
|
||||
}
|
||||
rebuilt = build_retrospective_portfolio_decision(
|
||||
**arguments,
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=target,
|
||||
constraints=ConstraintSetV1.from_dict(row["constraints"]),
|
||||
freshness_policy=FreshnessPolicy.from_dict(row["freshness_policy"]),
|
||||
receipt=ComputationReceipt.from_dict(row["receipt"]),
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_json_object(value), **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_portfolio_decision(
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
objective_name: str,
|
||||
objective_version: str,
|
||||
objective_digest: str,
|
||||
model_name: str,
|
||||
model_version: str,
|
||||
model_digest: str,
|
||||
expected_return_digest: str,
|
||||
covariance_digest: str,
|
||||
scenario_digest: str,
|
||||
constraints: ConstraintSetV1,
|
||||
freshness_policy: FreshnessPolicy,
|
||||
receipt: ComputationReceipt,
|
||||
computed_at: str,
|
||||
prior_weights: Mapping[str, float] | None = None,
|
||||
) -> RetrospectivePortfolioDecision:
|
||||
"""Verify the existing constraints and receipt, with two explicitly different clocks."""
|
||||
_check(
|
||||
type(receipt) is ComputationReceipt,
|
||||
"$.receipt",
|
||||
"typed computation receipt required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
receipt = ComputationReceipt.from_dict(receipt.to_dict())
|
||||
inputs = _material(
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=target,
|
||||
objective_name=objective_name,
|
||||
objective_version=objective_version,
|
||||
objective_digest=objective_digest,
|
||||
model_name=model_name,
|
||||
model_version=model_version,
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness_policy,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
run, manifest, target = inputs.run, inputs.manifest, inputs.target
|
||||
computed = _parse_utc(computed_at, "$.computed_at")
|
||||
created = _parse_utc(target.created_at, "$.target.created_at")
|
||||
available = _parse_utc(manifest.artifact_available_at, "$.manifest.artifact_available_at")
|
||||
_check(
|
||||
available <= created <= computed,
|
||||
"$.target.created_at",
|
||||
"artifact availability <= actual target creation <= computation required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(receipt.computed_at, "$.receipt.computed_at") == computed,
|
||||
"$.receipt.computed_at",
|
||||
"receipt actual time differs from computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
(computed - available).total_seconds() <= inputs.freshness.max_manifest_age_seconds,
|
||||
"$.manifest.artifact_available_at",
|
||||
"manifest is stale at actual computation",
|
||||
)
|
||||
_check(
|
||||
receipt.status not in {ReceiptStatus.FAILED, ReceiptStatus.FALLBACK},
|
||||
"$.receipt.status",
|
||||
"failed/fallback computation cannot form a result",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
digests = _receipt_digests(inputs)
|
||||
for key, expected in digests.items():
|
||||
_check(
|
||||
getattr(receipt, key) == expected,
|
||||
f"$.receipt.{key}",
|
||||
"receipt differs from independently recomputed evidence",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
digests["max_constraint_residual"] == 0.0,
|
||||
"$.constraints",
|
||||
"target violates supported constraints",
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.portfolio-decision",
|
||||
"schema_version": "2.0.0",
|
||||
"run_id": run.run_id,
|
||||
"manifest_id": manifest.manifest_id,
|
||||
"evidence_digest": manifest.evidence_digest,
|
||||
"dataset_snapshot_id": run.dataset_snapshot_id,
|
||||
"run_ref_document_sha256": _document_sha256(run.to_json()),
|
||||
"manifest_document_sha256": _document_sha256(manifest.to_json()),
|
||||
"source_universe_digest": run.universe_digest,
|
||||
"portfolio_asset_set_digest": _payload_digest(sorted(inputs.weights)),
|
||||
"target_id": target.target_id,
|
||||
"target_weights": _mapping_dict(inputs.weights),
|
||||
"prior_weights": None if inputs.prior is None else _mapping_dict(inputs.prior),
|
||||
"objective_name": objective_name,
|
||||
"objective_version": objective_version,
|
||||
"objective_digest": objective_digest,
|
||||
"model_name": model_name,
|
||||
"model_version": model_version,
|
||||
"model_digest": model_digest,
|
||||
"expected_return_digest": expected_return_digest,
|
||||
"covariance_digest": covariance_digest,
|
||||
"scenario_digest": scenario_digest,
|
||||
"constraints": inputs.constraints.to_dict(),
|
||||
"freshness_policy": inputs.freshness.to_dict(),
|
||||
"receipt": receipt.to_dict(),
|
||||
**inputs.metrics,
|
||||
"constraint_residuals": inputs.residuals,
|
||||
"output_digest": digests["output_digest"],
|
||||
"effective_at": target.effective_at,
|
||||
"created_at": target.created_at,
|
||||
"computed_at": computed_at,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
_public(payload)
|
||||
payload["decision_id"] = "rhportfoliodecisionv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(RetrospectivePortfolioDecision)
|
||||
values = {
|
||||
**payload,
|
||||
"target_weights": inputs.weights,
|
||||
"prior_weights": inputs.prior,
|
||||
"constraints": inputs.constraints,
|
||||
"freshness_policy": inputs.freshness,
|
||||
"receipt": receipt,
|
||||
"constraint_residuals": MappingProxyType(inputs.residuals),
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
"_target": target,
|
||||
"_run": run,
|
||||
"_manifest": manifest,
|
||||
}
|
||||
for key, value in values.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveRiskAssessment:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
assessment_id: str
|
||||
decision_id: str
|
||||
run_id: str
|
||||
manifest_id: str
|
||||
dataset_snapshot_id: str
|
||||
covariance_data_snapshot_id: str
|
||||
covariance_snapshot_id: str
|
||||
covariance_as_of_date: str
|
||||
covariance_method: str
|
||||
covariance_window_start_date: str
|
||||
covariance_window_end_date: str
|
||||
covariance_observations: int | None
|
||||
covariance_lookback_sessions: int | None
|
||||
covariance_missing_policy: str
|
||||
covariance_input_digest: str
|
||||
covariance_matrix_digest: str
|
||||
return_frequency: str
|
||||
periods_per_year: int
|
||||
risk_model_name: str
|
||||
risk_model_version: str
|
||||
risk_model_digest: str
|
||||
freshness_policy_digest: str
|
||||
scenario_digest: str
|
||||
portfolio_volatility_limit: float | None
|
||||
risk_budget: Mapping[str, float]
|
||||
groups: Mapping[str, str] | None
|
||||
marginal_risk: Mapping[str, float]
|
||||
component_risk: Mapping[str, float]
|
||||
percentage_risk: Mapping[str, float]
|
||||
portfolio_volatility: float | None
|
||||
group_exposure: Mapping[str, float]
|
||||
findings: tuple[RiskFindingCode, ...]
|
||||
status: RiskAssessmentStatus
|
||||
qualified: bool
|
||||
effective_at: str
|
||||
computed_at: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
portfolio_decision: RetrospectivePortfolioDecision,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
covariance: CovarianceSnapshot,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
|
||||
)
|
||||
rebuilt = assess_retrospective_portfolio_risk(
|
||||
portfolio_decision=portfolio_decision,
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
covariance=covariance,
|
||||
**{
|
||||
key: row[key]
|
||||
for key in (
|
||||
"risk_model_name",
|
||||
"risk_model_version",
|
||||
"risk_model_digest",
|
||||
"risk_budget",
|
||||
"portfolio_volatility_limit",
|
||||
"groups",
|
||||
"computed_at",
|
||||
)
|
||||
},
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_json_object(value), **kwargs)
|
||||
|
||||
|
||||
class _RiskContext(TypedDict):
|
||||
decision: RetrospectivePortfolioDecision
|
||||
covariance: CovarianceSnapshot
|
||||
matrix_digest: str
|
||||
risk_model_name: str
|
||||
risk_model_version: str
|
||||
risk_model_digest: str
|
||||
portfolio_volatility_limit: float | None
|
||||
risk_budget: Mapping[str, float]
|
||||
groups: Mapping[str, str] | None
|
||||
computed_at: str
|
||||
|
||||
|
||||
def _risk_result(
|
||||
*,
|
||||
decision: RetrospectivePortfolioDecision,
|
||||
covariance: CovarianceSnapshot,
|
||||
matrix_digest: str,
|
||||
risk_model_name: str,
|
||||
risk_model_version: str,
|
||||
risk_model_digest: str,
|
||||
portfolio_volatility_limit: float | None,
|
||||
risk_budget: Mapping[str, float],
|
||||
groups: Mapping[str, str] | None,
|
||||
marginal: Mapping[str, float],
|
||||
component: Mapping[str, float],
|
||||
percentage: Mapping[str, float],
|
||||
volatility: float | None,
|
||||
grouped: Mapping[str, float],
|
||||
findings: tuple[RiskFindingCode, ...],
|
||||
status: RiskAssessmentStatus,
|
||||
qualified: bool,
|
||||
computed_at: str,
|
||||
) -> RetrospectiveRiskAssessment:
|
||||
assert covariance.window_start_date is not None
|
||||
assert covariance.window_end_date is not None
|
||||
payload = {
|
||||
"contract_name": "researchhub.risk-assessment",
|
||||
"schema_version": "2.0.0",
|
||||
"decision_id": decision.decision_id,
|
||||
"run_id": decision.run_id,
|
||||
"manifest_id": decision.manifest_id,
|
||||
"dataset_snapshot_id": decision.dataset_snapshot_id,
|
||||
"covariance_data_snapshot_id": covariance.data_snapshot_id,
|
||||
"covariance_snapshot_id": covariance.snapshot_id,
|
||||
"covariance_as_of_date": covariance.as_of_date.isoformat(),
|
||||
"covariance_method": covariance.method,
|
||||
"covariance_window_start_date": covariance.window_start_date.isoformat(),
|
||||
"covariance_window_end_date": covariance.window_end_date.isoformat(),
|
||||
"covariance_observations": covariance.observations,
|
||||
"covariance_lookback_sessions": covariance.lookback_sessions,
|
||||
"covariance_missing_policy": covariance.missing_policy,
|
||||
"covariance_input_digest": "sha256:" + covariance.input_sha256,
|
||||
"covariance_matrix_digest": matrix_digest,
|
||||
"return_frequency": covariance.return_frequency,
|
||||
"periods_per_year": covariance.periods_per_year,
|
||||
"risk_model_name": risk_model_name,
|
||||
"risk_model_version": risk_model_version,
|
||||
"risk_model_digest": risk_model_digest,
|
||||
"freshness_policy_digest": _payload_digest(decision.freshness_policy.to_dict()),
|
||||
"scenario_digest": decision.scenario_digest,
|
||||
"portfolio_volatility_limit": portfolio_volatility_limit,
|
||||
"risk_budget": _mapping_dict(risk_budget),
|
||||
"groups": None if groups is None else dict(groups),
|
||||
"marginal_risk": _mapping_dict(marginal),
|
||||
"component_risk": _mapping_dict(component),
|
||||
"percentage_risk": _mapping_dict(percentage),
|
||||
"portfolio_volatility": volatility,
|
||||
"group_exposure": _mapping_dict(grouped),
|
||||
"findings": [finding.value for finding in findings],
|
||||
"status": status.value,
|
||||
"qualified": qualified,
|
||||
"effective_at": decision.effective_at,
|
||||
"computed_at": computed_at,
|
||||
"observation_cutoff": decision.observation_cutoff,
|
||||
"evidence_scope": decision.evidence_scope,
|
||||
"usage": decision.usage,
|
||||
"historical_availability": decision.historical_availability,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
_public(payload)
|
||||
payload["assessment_id"] = "rhriskassessmentv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(RetrospectiveRiskAssessment)
|
||||
values = {
|
||||
**payload,
|
||||
"risk_budget": risk_budget,
|
||||
"groups": groups,
|
||||
"marginal_risk": marginal,
|
||||
"component_risk": component,
|
||||
"percentage_risk": percentage,
|
||||
"group_exposure": grouped,
|
||||
"findings": findings,
|
||||
"status": status,
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
}
|
||||
for key, value in values.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
|
||||
def assess_retrospective_portfolio_risk(
|
||||
*,
|
||||
portfolio_decision: RetrospectivePortfolioDecision,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
covariance: CovarianceSnapshot,
|
||||
risk_model_name: str,
|
||||
risk_model_version: str,
|
||||
risk_model_digest: str,
|
||||
computed_at: str,
|
||||
risk_budget: Mapping[str, float] | None = None,
|
||||
portfolio_volatility_limit: float | None = None,
|
||||
groups: Mapping[str, str] | None = None,
|
||||
) -> RetrospectiveRiskAssessment:
|
||||
"""Use the existing Euler decomposition once; distinguish the two freshness clocks."""
|
||||
_check(
|
||||
type(portfolio_decision) is RetrospectivePortfolioDecision,
|
||||
"$.portfolio_decision",
|
||||
"explicit v2 portfolio result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
type(covariance) is CovarianceSnapshot,
|
||||
"$.covariance",
|
||||
"typed covariance required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
decision = RetrospectivePortfolioDecision.from_dict(
|
||||
portfolio_decision.to_dict(),
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=portfolio_decision._target,
|
||||
)
|
||||
actual_computed = _parse_utc(computed_at, "$.computed_at")
|
||||
_check(
|
||||
_parse_utc(decision.computed_at, "$.portfolio_decision.computed_at") <= actual_computed,
|
||||
"$.computed_at",
|
||||
"risk computation precedes portfolio computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
manifest_age = (
|
||||
actual_computed
|
||||
- _parse_utc(decision._manifest.artifact_available_at, "$.manifest.artifact_available_at")
|
||||
).total_seconds()
|
||||
_check(
|
||||
0 <= manifest_age <= decision.freshness_policy.max_manifest_age_seconds,
|
||||
"$.manifest.artifact_available_at",
|
||||
"manifest is stale at actual risk computation",
|
||||
)
|
||||
_check(
|
||||
covariance.data_snapshot_id == decision.dataset_snapshot_id,
|
||||
"$.covariance.data_snapshot_id",
|
||||
"covariance and decision data differ",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
business_date = _parse_utc(decision.effective_at, "$.portfolio_decision.effective_at").date()
|
||||
_check(
|
||||
covariance.window_start_date is not None and covariance.window_end_date is not None,
|
||||
"$.covariance",
|
||||
"bounded covariance window required",
|
||||
)
|
||||
assert covariance.window_start_date is not None
|
||||
assert covariance.window_end_date is not None
|
||||
_check(
|
||||
covariance.window_start_date
|
||||
<= covariance.window_end_date
|
||||
<= covariance.as_of_date
|
||||
<= business_date,
|
||||
"$.covariance.as_of_date",
|
||||
"covariance business dates exceed the historical target date",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
(business_date - covariance.as_of_date).days
|
||||
<= decision.freshness_policy.max_covariance_age_days,
|
||||
"$.covariance.as_of_date",
|
||||
"covariance is stale at historical target date",
|
||||
)
|
||||
_check(
|
||||
_digest("sha256:" + covariance.input_sha256, "$.covariance.input_sha256")
|
||||
== decision.covariance_digest,
|
||||
"$.covariance.input_sha256",
|
||||
"covariance input differs from portfolio receipt",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
name = _text(risk_model_name, "$.risk_model_name")
|
||||
version = _semver(risk_model_version, "$.risk_model_version")
|
||||
model_digest = _digest(risk_model_digest, "$.risk_model_digest")
|
||||
limit = (
|
||||
None
|
||||
if portfolio_volatility_limit is None
|
||||
else _finite_number(
|
||||
portfolio_volatility_limit, "$.portfolio_volatility_limit", non_negative=True
|
||||
)
|
||||
)
|
||||
budget: Mapping[str, float] = (
|
||||
MappingProxyType({})
|
||||
if risk_budget is None
|
||||
else _immutable_float_mapping(risk_budget, "$.risk_budget")
|
||||
)
|
||||
_check(
|
||||
all(value >= 0 for value in budget.values())
|
||||
and set(budget) <= decision.target_weights.keys(),
|
||||
"$.risk_budget",
|
||||
"risk budgets must be non-negative and use target labels",
|
||||
)
|
||||
normalized_groups = None
|
||||
if groups is not None:
|
||||
_check(
|
||||
isinstance(groups, Mapping),
|
||||
"$.groups",
|
||||
"mapping required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
group_values = {
|
||||
_text(key, "$.groups.keys"): _text(value, "$.groups.values")
|
||||
for key, value in groups.items()
|
||||
}
|
||||
_check(
|
||||
set(group_values) == decision.target_weights.keys(),
|
||||
"$.groups",
|
||||
"groups must label every target exactly once",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
normalized_groups = MappingProxyType(dict(sorted(group_values.items())))
|
||||
aligned = _validate_covariance_structure(decision.target_weights, covariance)
|
||||
matrix_digest = _payload_digest(
|
||||
{
|
||||
"assets": sorted(decision.target_weights),
|
||||
"matrix": aligned.to_numpy(dtype=float).tolist(),
|
||||
}
|
||||
)
|
||||
arguments: _RiskContext = {
|
||||
"decision": decision,
|
||||
"covariance": covariance,
|
||||
"matrix_digest": matrix_digest,
|
||||
"risk_model_name": name,
|
||||
"risk_model_version": version,
|
||||
"risk_model_digest": model_digest,
|
||||
"portfolio_volatility_limit": limit,
|
||||
"risk_budget": budget,
|
||||
"groups": normalized_groups,
|
||||
"computed_at": computed_at,
|
||||
}
|
||||
empty: Mapping[str, float] = MappingProxyType({})
|
||||
|
||||
def unavailable(finding: RiskFindingCode) -> RetrospectiveRiskAssessment:
|
||||
return _risk_result(
|
||||
**arguments,
|
||||
marginal=empty,
|
||||
component=empty,
|
||||
percentage=empty,
|
||||
volatility=None,
|
||||
grouped=empty,
|
||||
findings=(finding,),
|
||||
status=RiskAssessmentStatus.UNAVAILABLE,
|
||||
qualified=False,
|
||||
)
|
||||
|
||||
weights = pd.Series(_mapping_dict(decision.target_weights), dtype=float, name="weight")
|
||||
try:
|
||||
decomposition = labeled_component_risk(weights, aligned * covariance.periods_per_year)
|
||||
except ValueError as error:
|
||||
finding = {
|
||||
"covariance must be positive semidefinite": RiskFindingCode.COVARIANCE_NOT_PSD,
|
||||
"weights and covariance must produce positive portfolio variance": RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE,
|
||||
}.get(str(error))
|
||||
if finding is None:
|
||||
raise PortfolioRiskContractError(
|
||||
PortfolioRiskContractErrorCode.COMPUTATION_FAILURE,
|
||||
"$.covariance",
|
||||
"risk computation failed",
|
||||
) from error
|
||||
return unavailable(finding)
|
||||
marginal = _series_mapping(decomposition.marginal)
|
||||
component = _series_mapping(decomposition.component)
|
||||
percentage = _series_mapping(decomposition.percentage)
|
||||
volatility = _finite_number(
|
||||
decomposition.portfolio_volatility, "$.risk_output.portfolio_volatility", non_negative=True
|
||||
)
|
||||
if not (
|
||||
set(marginal) == set(component) == set(percentage) == decision.target_weights.keys()
|
||||
and math.isclose(
|
||||
sum(component.values()), volatility, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
|
||||
)
|
||||
and math.isclose(
|
||||
sum(percentage.values()), 1.0, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
|
||||
)
|
||||
):
|
||||
return unavailable(RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED)
|
||||
grouped = (
|
||||
empty
|
||||
if normalized_groups is None
|
||||
else _series_mapping(
|
||||
decomposition.grouped_component(pd.Series(dict(normalized_groups), dtype="object"))
|
||||
)
|
||||
)
|
||||
breached = (limit is not None and volatility > limit + _CLOSURE_ATOL) or any(
|
||||
percentage[label] > maximum + _CLOSURE_ATOL for label, maximum in budget.items()
|
||||
)
|
||||
return _risk_result(
|
||||
**arguments,
|
||||
marginal=marginal,
|
||||
component=component,
|
||||
percentage=percentage,
|
||||
volatility=volatility,
|
||||
grouped=grouped,
|
||||
findings=(RiskFindingCode.RISK_BUDGET_BREACH,) if breached else (),
|
||||
status=RiskAssessmentStatus.READY,
|
||||
qualified=not breached,
|
||||
)
|
||||
+1215
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,175 @@
|
||||
{
|
||||
"contract_name": "researchhub.data-foundation",
|
||||
"schema_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"published_at": "2026-09-08T01:05:00Z",
|
||||
"instrument_routes": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"symbol": "SIM0",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb",
|
||||
"route_revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9"
|
||||
},
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"symbol": "SIM1",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799",
|
||||
"route_revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
}
|
||||
],
|
||||
"trading_calendar_revisions": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"session_date": "2018-01-02",
|
||||
"status": "open",
|
||||
"sessions": [
|
||||
{
|
||||
"opens_at": "2018-01-02T01:30:00Z",
|
||||
"closes_at": "2018-01-02T07:00:00Z"
|
||||
}
|
||||
],
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a",
|
||||
"calendar_revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
}
|
||||
],
|
||||
"corporate_action_revisions": [],
|
||||
"standardized_views": [
|
||||
{
|
||||
"view_id": "rhview:66666666666666666666666666666666",
|
||||
"view_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"transformation_digest": "sha256:b2376dbe4424b38e5b27874c998f082aecc086179f611896ebaa0fdeb5362c68",
|
||||
"instrument_route_revision_ids": [
|
||||
"rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
],
|
||||
"trading_calendar_revision_ids": [
|
||||
"rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
],
|
||||
"corporate_action_revision_ids": [],
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"available_at": "2026-09-08T01:04:00Z",
|
||||
"view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309"
|
||||
}
|
||||
],
|
||||
"observation_lineage": [
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb"
|
||||
},
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799"
|
||||
},
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a"
|
||||
}
|
||||
],
|
||||
"corporate_action_coverage": [
|
||||
{
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:233cc2968985e953e59e408b6619a837a792ae17f8afba32261f861de8fe9c7c"
|
||||
]
|
||||
},
|
||||
{
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:21f2594fc42e46168a65d7cac7e9b52aeabf24ecc5fb409530c07ad56ed3986c"
|
||||
]
|
||||
}
|
||||
],
|
||||
"readiness": {
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"contract_validation": {
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:62f2abbbb79f4ff9164e10cc4b21df37f0964224a4c546f978b97794241ff5b3"
|
||||
]
|
||||
},
|
||||
"real_data_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"production_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"live_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
}
|
||||
},
|
||||
"foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260"
|
||||
}
|
||||
@@ -0,0 +1,120 @@
|
||||
{
|
||||
"contract_name": "researchhub.dataset-snapshot",
|
||||
"schema_version": "2.0.0",
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"descriptor": {
|
||||
"dataset": {
|
||||
"dataset_id": "rhdataset:market:44444444444444444444444444444444",
|
||||
"dataset_kind": "market",
|
||||
"record_schema_version": "2.0.0",
|
||||
"dimensions": [
|
||||
"instrument_id",
|
||||
"effective_time"
|
||||
]
|
||||
},
|
||||
"published_at": "2026-09-08T01:03:00Z",
|
||||
"time_semantics": {
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"historical_availability": "not_established"
|
||||
},
|
||||
"content": {
|
||||
"digest_algorithm": "sha256",
|
||||
"canonicalization": "RFC8785",
|
||||
"record_order": "canonical-record-byte-order",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"logical_manifest": {
|
||||
"record_count": 2,
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2
|
||||
}
|
||||
]
|
||||
},
|
||||
"manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7",
|
||||
"record_count": 2
|
||||
},
|
||||
"observation_manifest": {
|
||||
"batches": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": "sha256:6f35f19a2ede0610d461b458a0f2cbc96f40cee6e90fa6592fd66daf617d3ec0"
|
||||
}
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"publisher": {
|
||||
"id": "researchhub.data",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"transformation": {
|
||||
"id": "rhtransform:55555555555555555555555555555555",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"upstream_snapshot_ids": [],
|
||||
"upstream_content_digests": []
|
||||
},
|
||||
"quality": {
|
||||
"status": "passed",
|
||||
"checks": [
|
||||
{
|
||||
"check_id": "completeness",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:519b01c60ad85c2b03d6c4f29027fe9b3672a8f6ae3ca8a2d831b281a61c5095"
|
||||
},
|
||||
{
|
||||
"check_id": "duplicate_identity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:7c585c7471ab45886fce037061b5ab0d5f981aede5d5190021963676b0ec0fcc"
|
||||
},
|
||||
{
|
||||
"check_id": "observation_coverage",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:dde5693c3ad79a01a162f9fc5a40c69d3c284965e283c91c5cd456904e0ef737"
|
||||
},
|
||||
{
|
||||
"check_id": "historical_claim_policy",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:f4cf8da48f92e03d75bb28b2569061e649f382ba360d879fef8e41e35169c228"
|
||||
},
|
||||
{
|
||||
"check_id": "range_validity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:6af08355c5e29d846e1c48783dccde791e8a3f48f8416149413217b7c2cbb52a"
|
||||
},
|
||||
{
|
||||
"check_id": "schema_conformance",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:fefdc5b81eba128af92f15a60feb7bb6f40b2e7277bf2beb48c1bb9399ee564b"
|
||||
}
|
||||
]
|
||||
},
|
||||
"qualification": {
|
||||
"status": "qualified",
|
||||
"usage": "retrospective_research",
|
||||
"policy_id": "researchhub.dataset-snapshot.retrospective",
|
||||
"policy_version": "2.0.0",
|
||||
"evaluated_at": "2026-09-08T01:02:00Z",
|
||||
"evidence_digest": "sha256:70c3ddcadf095877f17a1365c8d75a2e5a7078a27722fb8b698e61d350c6bb2d"
|
||||
}
|
||||
},
|
||||
"snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905"
|
||||
}
|
||||
@@ -16,7 +16,7 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
|
||||
for term in ("investment advice", "live order", "credentials", "source facts"):
|
||||
assert term in prohibited
|
||||
assert spec["authority"]["revision"] == 5
|
||||
assert spec["authority"]["revision"] == 6
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["provides"]
|
||||
@@ -28,6 +28,13 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
("researchhub.performance-evidence", "1.0.0"),
|
||||
("researchhub.portfolio-decision", "1.0.0"),
|
||||
("researchhub.risk-assessment", "1.0.0"),
|
||||
("researchhub.factor-set-ref", "2.0.0"),
|
||||
("researchhub.backtest-run-ref", "2.0.0"),
|
||||
("researchhub.backtest-evidence-manifest", "2.0.0"),
|
||||
("researchhub.performance-evidence", "2.0.0"),
|
||||
("researchhub.portfolio-target", "2.0.0"),
|
||||
("researchhub.portfolio-decision", "2.0.0"),
|
||||
("researchhub.risk-assessment", "2.0.0"),
|
||||
}
|
||||
expected_paths = {
|
||||
"researchhub.factor-definition": "src/quant_engine/factor_contracts.py",
|
||||
@@ -41,13 +48,28 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"])
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "1.0.0"
|
||||
} == expected_paths
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "2.0.0"
|
||||
} == {
|
||||
"researchhub.factor-set-ref": "src/quant_engine/retrospective_factor_contracts.py",
|
||||
"researchhub.backtest-run-ref": "src/quant_engine/retrospective_backtest_contracts.py",
|
||||
"researchhub.backtest-evidence-manifest": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.performance-evidence": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.portfolio-target": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.portfolio-decision": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.risk-assessment": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
}
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["consumes"]
|
||||
} == {
|
||||
("researchhub.dataset-snapshot", "1.0.0"),
|
||||
("researchhub.data-foundation", "1.0.0"),
|
||||
("researchhub.dataset-snapshot", "2.0.0"),
|
||||
("researchhub.data-foundation", "2.0.0"),
|
||||
}
|
||||
assert all(
|
||||
item["authority"] == "researchhub.data"
|
||||
@@ -65,6 +87,10 @@ def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
summary = portfolio_contract["summary"].lower()
|
||||
for term in ("receipt", "portfolio decisions", "risk assessments", "without"):
|
||||
assert term in summary
|
||||
retrospective = capabilities["retrospective-computation-contracts"]
|
||||
assert retrospective["status"] == "operational"
|
||||
for term in ("two clocks", "retrospective", "no historical-availability", "no execution"):
|
||||
assert term in retrospective["summary"].lower()
|
||||
for term in ("approval", "maker-checker", "publication", "paper", "live"):
|
||||
assert term in prohibited
|
||||
assert all(
|
||||
|
||||
@@ -0,0 +1,311 @@
|
||||
"""Fresh synthetic artifact assembly; no reuse or relabelling of real outputs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
EvidenceQualification,
|
||||
PerformanceEvidenceError,
|
||||
ResearchRunArtifact,
|
||||
build_research_run_artifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_performance_evidence,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.research_pipeline import run_factor_backtest_research
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
build_retrospective_performance_evidence,
|
||||
RetrospectiveBacktestEvidenceManifest,
|
||||
RetrospectivePerformanceEvidence,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import identify, replace_at
|
||||
|
||||
CONTRACT_ERRORS = (FactorContractError, BacktestContractError, PerformanceEvidenceError)
|
||||
|
||||
|
||||
def synthetic_artifact(run: RetrospectiveBacktestRunRef) -> ResearchRunArtifact:
|
||||
# Artifact-envelope tests, not an end-to-end proof of factor/source authenticity.
|
||||
# The existing financial methods receive new, in-memory synthetic matrices.
|
||||
dates = pd.date_range("2018-01-02", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"SIM0": [2.0, 0.0], "SIM1": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"SIM0": [10.0, 10.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 20.0, 21.0]}, index=dates
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"SIM0": [10.0, 12.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 18.0, 21.0]}, index=dates
|
||||
)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0
|
||||
),
|
||||
)
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02], index=result.returns.index, name="benchmark_return"
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id=run.run_id,
|
||||
strategy_id=run.strategy_id,
|
||||
strategy_name="Synthetic Top 1",
|
||||
strategy_version=run.strategy_version,
|
||||
engine_version="0.1.0",
|
||||
code_revision=run.code_revision,
|
||||
data_snapshot_id=run.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at=run.evaluation_at,
|
||||
finished_at=run.computed_at,
|
||||
parameters={"lag_sessions": 1, "top_k": 1},
|
||||
benchmark_id="synthetic.benchmark",
|
||||
benchmark_returns=benchmark,
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_closes_all_nine_existing_tables_without_changing_their_schema() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
wire = manifest.to_dict()
|
||||
assert wire["schema_version"] == "2.0.0"
|
||||
assert wire["artifact_schema_version"] == "1.1.0"
|
||||
assert wire["run_id"] == run.run_id
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["execution_validation"] == "not_validated"
|
||||
assert wire["decision_eligible"] is False
|
||||
assert manifest.manifest_id.startswith("rhbacktestevidencev2:sha256:")
|
||||
assert len({table.logical_name for item in manifest.evidence for table in item.tables}) == 9
|
||||
|
||||
|
||||
def test_performance_v2_keeps_existing_metric_methods_and_binds_all_upstreams() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
wire = evidence.to_dict()
|
||||
assert wire["schema_version"] == "researchhub.performance-evidence.v2"
|
||||
assert wire["methodology_id"] == "researchhub.quant-performance-methodology.v1"
|
||||
assert wire["metric_schema_id"] == "researchhub.quant-performance-metrics.v1"
|
||||
assert wire["research_artifact_schema_version"] == "1.1.0"
|
||||
assert wire["backtest_run_ref_id"] == run.run_id
|
||||
assert wire["backtest_evidence_manifest_id"] == manifest.manifest_id
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["start_date"] == "2018-01-02"
|
||||
assert wire["end_date"] == "2018-01-05"
|
||||
assert wire["artifact_available_at"] == "2026-09-08T01:11:00Z"
|
||||
assert evidence.run_id == run.run_id
|
||||
assert evidence.performance_evidence_id.startswith("rhperformancev2:sha256:")
|
||||
for metric in evidence.metrics:
|
||||
if metric.value is not None:
|
||||
assert metric.value == artifact.performance.iloc[0][metric.source_column]
|
||||
assert (
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
evidence.to_json(), artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
== evidence
|
||||
)
|
||||
assert (
|
||||
RetrospectiveBacktestEvidenceManifest.from_json(
|
||||
manifest.to_json(), artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
== manifest
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("run_id", "rhbacktestrunv2:sha256:" + "0" * 64),
|
||||
("profile", "offline_research_v1"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("evidence_scope", "real_data"),
|
||||
("artifact_available_at", "2026-09-08T01:09:00Z"),
|
||||
("artifact_schema_version", "2.0.0"),
|
||||
("qualification", "legacy_exploratory"),
|
||||
("evidence_digest", "sha256:" + "0" * 64),
|
||||
("evidence.0.tables.0.row_count", True),
|
||||
("evidence.0.tables.0.content_digest", "sha256:" + "0" * 64),
|
||||
("run_reference.value.factor_set_id", "rhfactorsetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_reidentified_claims_without_actual_table_closure(
|
||||
path: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
row = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "manifest_id", "rhbacktestevidencev2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
row, artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("table", "column", "value"),
|
||||
[
|
||||
("run", "data_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("run", "config_hash", "0" * 64),
|
||||
("run", "code_revision", "0" * 40),
|
||||
("run", "started_at", "2018-01-02T07:00:00Z"),
|
||||
("run", "finished_at", "2026-09-08T01:12:00Z"),
|
||||
("signals", "asset_id", "/private/data.csv"),
|
||||
("nav", "run_id", "old.run"),
|
||||
("performance", "run_id", "old.run"),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_artifact_identity_time_or_private_data_mismatch(
|
||||
table: str, column: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
frame = getattr(artifact, table)
|
||||
frame.loc[frame.index[0], column] = value
|
||||
forged = replace(artifact, **{"_" + table: frame})
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run, forged, artifact_available_at="2026-09-08T01:13:00Z"
|
||||
)
|
||||
|
||||
|
||||
def test_each_table_is_reconciled_and_legacy_apis_cannot_accept_v2() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
for item in manifest.evidence:
|
||||
for table in item.tables:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
expected_table_digests={table.logical_name: "sha256:" + "0" * 64},
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_performance_evidence(artifact, run, manifest)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
qualification=EvidenceQualification.LEGACY_EXPLORATORY,
|
||||
)
|
||||
|
||||
|
||||
def seal_performance(row: dict[str, Any]) -> None:
|
||||
def sha(document: Any) -> str:
|
||||
return (
|
||||
"sha256:"
|
||||
+ hashlib.sha256(
|
||||
json.dumps(
|
||||
document,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode()
|
||||
).hexdigest()
|
||||
)
|
||||
|
||||
row.pop("document_sha256", None)
|
||||
row.pop("performance_evidence_id", None)
|
||||
row["performance_evidence_id"] = "rhperformancev2:" + sha(row)
|
||||
row["document_sha256"] = sha(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "researchhub.performance-evidence.v1"),
|
||||
("scope", "live"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("evidence_scope", "real_data"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("backtest_run_ref_document_sha256", "sha256:" + "0" * 64),
|
||||
("backtest_evidence_manifest_id", "rhbacktestevidencev2:sha256:" + "0" * 64),
|
||||
("methodology.periods_per_year", 365),
|
||||
("metric_schema_id", "new.metric"),
|
||||
("metrics.0.value", 0.0),
|
||||
("metrics.0.nullable", True),
|
||||
("start_date", "2017-01-01"),
|
||||
("artifact_available_at", "2018-01-02T07:00:00Z"),
|
||||
],
|
||||
)
|
||||
def test_performance_never_accepts_reidentified_changed_facts(path: str, value: Any) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
row = build_retrospective_performance_evidence(artifact, run, manifest).to_dict()
|
||||
replace_at(row, path, value)
|
||||
seal_performance(row)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_dict(
|
||||
row, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
|
||||
|
||||
def test_performance_has_immutable_finite_canonical_payload_and_current_tables() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
assert evidence.document_sha256.startswith("sha256:")
|
||||
exported = evidence.to_dict()
|
||||
exported["metrics"][0]["value"] = 9.0
|
||||
assert evidence.to_dict()["metrics"][0]["value"] != 9.0
|
||||
for data in (
|
||||
evidence.to_json() + "\n",
|
||||
'{"schema_version":"x",' + evidence.to_json()[1:],
|
||||
"null",
|
||||
"{bad",
|
||||
):
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
data, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
frame = artifact.performance
|
||||
frame.loc[0, "total_ret"] = 0.0
|
||||
forged = replace(artifact, _performance=frame)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_performance_evidence(forged, run, manifest)
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Offline synthetic v2 backtest evidence and replay boundaries."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import FactorContractError, PayloadValidation
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef
|
||||
from test_retrospective_data_contracts import digest, identify, replace_at
|
||||
from test_retrospective_factor_contracts import factor_arguments, decoding_arguments
|
||||
|
||||
|
||||
def run_arguments() -> dict[str, Any]:
|
||||
arguments = factor_arguments()
|
||||
factor = RetrospectiveFactorSetRef.create(**arguments)
|
||||
view = next(iter(arguments["foundation"].views.values()))
|
||||
return {
|
||||
"dataset_snapshot": arguments["dataset_snapshot"],
|
||||
"foundation": arguments["foundation"],
|
||||
"factor_set": factor,
|
||||
"universe_digest": digest({"synthetic_universe": 2}),
|
||||
"trading_calendar_revision_ids": view.trading_calendar_revision_ids,
|
||||
"corporate_action_revision_ids": view.corporate_action_revision_ids,
|
||||
"strategy_id": "synthetic.top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": digest({"synthetic_strategy": "top1"}),
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": digest({"synthetic_execution": 1}),
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": digest({"synthetic_cost": 1}),
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": digest({"synthetic_lock": 1}),
|
||||
"configuration_digest": digest({"lag_sessions": 1, "top_k": 1}),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
}
|
||||
|
||||
|
||||
def run_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("dataset_snapshot", "foundation", "factor_set")}
|
||||
|
||||
|
||||
def test_run_identity_closes_observation_inputs_and_preserves_configurations() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
document = run.to_dict()
|
||||
assert document["schema_version"] == "2.0.0"
|
||||
assert run.run_id.startswith("rhbacktestrunv2:sha256:")
|
||||
assert run.dataset_snapshot_id == arguments["dataset_snapshot"].snapshot_id
|
||||
assert run.foundation_id == arguments["foundation"].foundation_id
|
||||
assert run.factor_set_id == arguments["factor_set"].factor_set_id
|
||||
assert document["usage"] == "retrospective_research"
|
||||
assert document["historical_availability"] == "not_established"
|
||||
assert document["decision_eligible"] is False
|
||||
assert document["execution_validation"] == "not_validated"
|
||||
assert document["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert document["replay_attempt"] == 0
|
||||
assert RetrospectiveBacktestRunRef.from_json(run.to_json(), **run_context(arguments)) == run
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("historical_availability", "established"),
|
||||
("usage", "as_available"),
|
||||
("execution_validation", "validated"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("dataset_content_digest", "sha256:" + "0" * 64),
|
||||
("foundation_digest", "sha256:" + "0" * 64),
|
||||
("factor_set_digest", "sha256:" + "0" * 64),
|
||||
("factor_output_content_digest", "sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("evidence_scope", "real_data"),
|
||||
("trading_calendar_revision_ids", []),
|
||||
("corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("evaluation_at", "2026-09-08T01:07:00Z"),
|
||||
("computed_at", "2026-09-08T01:08:00Z"),
|
||||
("computed_at", "2026-09-08T01:10:00.0000001Z"),
|
||||
("random_seed", True),
|
||||
("strategy_version", "latest"),
|
||||
("configuration_digest", "../private/a"),
|
||||
("code_revision", "unknown"),
|
||||
("replay_attempt", 1),
|
||||
("replay_reason", "retry"),
|
||||
("replay_spec_digest", "sha256:" + "0" * 64),
|
||||
("replay_ancestor_run_ids", ["rhbacktestrunv2:sha256:" + "0" * 64]),
|
||||
],
|
||||
)
|
||||
def test_reidentified_run_must_match_exact_context(path: str, value: Any) -> None:
|
||||
arguments = run_arguments()
|
||||
row = RetrospectiveBacktestRunRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "run_id", "rhbacktestrunv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.from_dict(row, **run_context(arguments))
|
||||
|
||||
|
||||
def test_replay_keeps_input_spec_but_requires_new_actual_attempt_times() -> None:
|
||||
arguments = run_arguments()
|
||||
root = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
replay_args = {
|
||||
**arguments,
|
||||
"parent": root,
|
||||
"replay_reason": "synthetic.retry",
|
||||
"replay_attempt": 1,
|
||||
"evaluation_at": "2026-09-08T01:12:00Z",
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
replay = RetrospectiveBacktestRunRef.create(**replay_args)
|
||||
assert replay.replay_spec_digest == root.replay_spec_digest
|
||||
assert replay.run_id != root.run_id
|
||||
assert replay.replay_ancestor_run_ids == (root.run_id,)
|
||||
assert replay.evaluation_at != root.evaluation_at
|
||||
assert (
|
||||
RetrospectiveBacktestRunRef.from_json(
|
||||
replay.to_json(), **run_context(arguments), parent=root
|
||||
)
|
||||
== replay
|
||||
)
|
||||
for changes in (
|
||||
{"random_seed": 9},
|
||||
{"configuration_digest": digest({"different_configuration": 1})},
|
||||
{"evaluation_at": root.evaluation_at},
|
||||
{"replay_attempt": 2},
|
||||
{"replay_reason": None},
|
||||
{"parent": None},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**replay_args, **changes})
|
||||
|
||||
|
||||
def test_reference_only_factors_can_be_read_but_not_used_to_create_new_runs() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
factor = arguments["factor_set"]
|
||||
reference = RetrospectiveFactorSetRef.from_dict(
|
||||
factor.to_dict(),
|
||||
definitions=factor._definitions,
|
||||
dataset_snapshot=arguments["dataset_snapshot"],
|
||||
foundation=arguments["foundation"],
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**arguments, "factor_set": reference})
|
||||
restored = RetrospectiveBacktestRunRef.from_dict(
|
||||
run.to_dict(), **{**run_context(arguments), "factor_set": reference}
|
||||
)
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
with pytest.raises(FactorContractError):
|
||||
restored.require_inputs_revalidated()
|
||||
run.require_inputs_revalidated()
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Frozen synthetic interoperability vector, not end-to-end source provenance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from quant_engine.artifact import _evidence_frame_records
|
||||
from quant_engine.retrospective_artifact_contracts import build_retrospective_performance_evidence
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_factor_contracts import factor_arguments
|
||||
from test_retrospective_portfolio_risk_contracts import portfolio_arguments, risk_arguments
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
VECTOR = ROOT / "tests" / "fixtures" / "retrospective-computation-v2.golden.json"
|
||||
|
||||
|
||||
def build_vector() -> dict[str, Any]:
|
||||
portfolio = portfolio_arguments()
|
||||
risk = risk_arguments(portfolio)
|
||||
run = portfolio["backtest_run_ref"]
|
||||
manifest = portfolio["manifest"]
|
||||
artifact = manifest._artifact
|
||||
factor = factor_arguments()
|
||||
return {
|
||||
"fixture_kind": "synthetic_retrospective_contract_vector",
|
||||
"artifact_data_provenance": "envelope_test_only_not_end_to_end",
|
||||
"source_authenticity": "not_established",
|
||||
"factor_definitions": [item.to_dict() for item in factor["definitions"]],
|
||||
"dataset_chunks": factor["dataset_chunks"],
|
||||
"resolved_view_schema": {"synthetic_schema": "neutral_close_v2"},
|
||||
"factor_output_schema": json.loads(factor["output_schema_bytes"]),
|
||||
"factor_output_records": json.loads(factor["output_content_bytes"]),
|
||||
"factor_set": run._factor_set.to_dict(),
|
||||
"backtest_run_ref": run.to_dict(),
|
||||
"artifact_tables": {
|
||||
name: _evidence_frame_records(frame, name)
|
||||
for name, frame in artifact.table_frames().items()
|
||||
},
|
||||
"backtest_evidence_manifest": manifest.to_dict(),
|
||||
"performance_evidence": build_retrospective_performance_evidence(
|
||||
artifact, run, manifest
|
||||
).to_dict(),
|
||||
"portfolio_target": portfolio["target"].to_dict(),
|
||||
"portfolio_decision": risk["portfolio_decision"].to_dict(),
|
||||
"risk_assessment": assess_retrospective_portfolio_risk(**risk).to_dict(),
|
||||
"covariance_matrix": risk["covariance"].covariance.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
def test_synthetic_vector_matches_all_current_owner_serializers() -> None:
|
||||
expected = VECTOR.read_text(encoding="utf-8")
|
||||
actual = (
|
||||
json.dumps(build_vector(), ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False)
|
||||
+ "\n"
|
||||
)
|
||||
assert actual == expected
|
||||
|
||||
|
||||
def test_v2_modules_do_not_import_data_owners_publishers_or_execution_authority() -> None:
|
||||
modules = sorted((ROOT / "src" / "quant_engine").glob("retrospective_*_contracts.py"))
|
||||
assert len(modules) == 5
|
||||
for path in modules:
|
||||
tree = ast.parse(path.read_text(encoding="utf-8"))
|
||||
imports = {
|
||||
alias.name
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Import)
|
||||
for alias in node.names
|
||||
} | {node.module or "" for node in ast.walk(tree) if isinstance(node, ast.ImportFrom)}
|
||||
assert not any(
|
||||
name.startswith(("research_results", "research_platform", "edb_data_core"))
|
||||
for name in imports
|
||||
)
|
||||
called = {
|
||||
node.func.id
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
|
||||
}
|
||||
assert not called & {
|
||||
"create_paper_order_intent",
|
||||
"run_governed_factor_slice",
|
||||
"evaluate_portfolio_risk",
|
||||
}
|
||||
@@ -0,0 +1,636 @@
|
||||
"""QE-owned decoding tests against RP's synthetic public v2 golden vectors."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from copy import deepcopy
|
||||
from dataclasses import FrozenInstanceError
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorContractError,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
|
||||
def golden(kind: str) -> dict[str, Any]:
|
||||
return json.loads((FIXTURES / f"retrospective-{kind}-v2.golden.json").read_text())
|
||||
|
||||
|
||||
def digest(value: Any) -> str:
|
||||
return "sha256:" + hashlib.sha256(canonical_json_bytes(value)).hexdigest()
|
||||
|
||||
|
||||
def identify(value: dict[str, Any], field: str, prefix: str) -> None:
|
||||
value[field] = prefix + digest({key: item for key, item in value.items() if key != field})
|
||||
|
||||
|
||||
def records() -> list[dict[str, Any]]:
|
||||
return [
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "1" * 32,
|
||||
"metric": "close",
|
||||
"value": "101.25",
|
||||
},
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "2" * 32,
|
||||
"metric": "close",
|
||||
"value": "87.50",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def bind_records(source: dict[str, Any], chunks: list[list[dict[str, Any]]]) -> None:
|
||||
def records_digest(rows: list[dict[str, Any]]) -> str:
|
||||
data = b"[" + b",".join(sorted(canonical_json_bytes(row) for row in rows)) + b"]"
|
||||
return "sha256:" + hashlib.sha256(data).hexdigest()
|
||||
|
||||
manifest = {
|
||||
"record_count": sum(len(rows) for rows in chunks),
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": index,
|
||||
"content_digest": records_digest(rows),
|
||||
"record_count": len(rows),
|
||||
}
|
||||
for index, rows in enumerate(chunks)
|
||||
],
|
||||
}
|
||||
source["descriptor"]["content"].update(
|
||||
{
|
||||
"record_count": manifest["record_count"],
|
||||
"logical_manifest": manifest,
|
||||
"manifest_digest": digest(manifest),
|
||||
"content_digest": records_digest([row for chunk in chunks for row in chunk]),
|
||||
}
|
||||
)
|
||||
source["descriptor"]["observation_manifest"]["batches"] = [
|
||||
{
|
||||
**chunk,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_receipt": index}),
|
||||
}
|
||||
for index, chunk in enumerate(manifest["chunks"])
|
||||
]
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
|
||||
|
||||
def replace_at(source: dict[str, Any], path: str, value: Any) -> None:
|
||||
target: Any = source
|
||||
keys = path.split(".")
|
||||
for key in keys[:-1]:
|
||||
target = target[int(key)] if isinstance(target, list) else target[key]
|
||||
target[int(keys[-1]) if isinstance(target, list) else keys[-1]] = value
|
||||
|
||||
|
||||
COLLECTIONS = (
|
||||
(
|
||||
"instrument_routes",
|
||||
"route_revision_id",
|
||||
"rhroutev2:",
|
||||
"instrument_route",
|
||||
"instrument_route_revision_ids",
|
||||
),
|
||||
(
|
||||
"trading_calendar_revisions",
|
||||
"calendar_revision_id",
|
||||
"rhcalv2:",
|
||||
"trading_calendar",
|
||||
"trading_calendar_revision_ids",
|
||||
),
|
||||
(
|
||||
"corporate_action_revisions",
|
||||
"action_revision_id",
|
||||
"rhcav2:",
|
||||
"corporate_action",
|
||||
"corporate_action_revision_ids",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def seal_foundation(source: dict[str, Any], *, rebuild_lineage: bool = True) -> None:
|
||||
lineage = []
|
||||
for name, key, prefix, kind, view_key in COLLECTIONS:
|
||||
replacements = {}
|
||||
for row in sorted(source[name], key=lambda row: row["observation_sequence"]):
|
||||
old = row[key]
|
||||
if "supersedes_observation_id" in row:
|
||||
row["supersedes_observation_id"] = replacements.get(
|
||||
row["supersedes_observation_id"], row["supersedes_observation_id"]
|
||||
)
|
||||
identify(row, key, prefix)
|
||||
replacements[old] = row[key]
|
||||
lineage.append(
|
||||
{
|
||||
"revision_kind": kind,
|
||||
"revision_id": row[key],
|
||||
**{
|
||||
field: row[field]
|
||||
for field in (
|
||||
"observation_sequence",
|
||||
"observed_by",
|
||||
"earliest_external_knowledge",
|
||||
"history_completeness",
|
||||
"evidence_digest",
|
||||
"supersedes_observation_id",
|
||||
)
|
||||
if field in row
|
||||
},
|
||||
}
|
||||
)
|
||||
for view in source["standardized_views"]:
|
||||
view[view_key] = [replacements.get(item, item) for item in view[view_key]]
|
||||
if rebuild_lineage:
|
||||
source["observation_lineage"] = lineage
|
||||
for view in source["standardized_views"]:
|
||||
identify(view, "view_ref_id", "rhviewrefv2:")
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
|
||||
|
||||
def parse_foundation(source: dict[str, Any]) -> RetrospectiveFoundationEnvelope:
|
||||
return RetrospectiveFoundationEnvelope.from_dict(
|
||||
source, snapshot=RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
)
|
||||
|
||||
|
||||
def test_public_snapshot_golden_is_an_explicit_observation_contract() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert snapshot.to_dict() == source
|
||||
assert snapshot.snapshot_id == source["snapshot_id"]
|
||||
assert snapshot.observation_cutoff == "2026-09-08T01:01:00Z"
|
||||
assert snapshot.evidence_scope == "synthetic_fixture"
|
||||
assert snapshot.earliest_external_knowledge == {"status": "unknown"}
|
||||
assert not hasattr(snapshot, "pit_cutoff")
|
||||
assert not hasattr(snapshot, "knowledge_time")
|
||||
snapshot.require_qualified()
|
||||
assert RetrospectiveSnapshotEnvelope.from_json(snapshot.to_json()) == snapshot
|
||||
|
||||
|
||||
def test_public_foundation_golden_binds_exact_typed_snapshot() -> None:
|
||||
source = golden("data-foundation")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert foundation.to_dict() == source
|
||||
assert foundation.dataset_snapshot_id == snapshot.snapshot_id
|
||||
assert foundation.observation_cutoff == snapshot.observation_cutoff
|
||||
assert foundation.evidence_scope == snapshot.evidence_scope
|
||||
assert foundation.real_data_validation_status == "not_validated"
|
||||
assert not hasattr(foundation, "pit_cutoff")
|
||||
assert (
|
||||
RetrospectiveFoundationEnvelope.from_json(foundation.to_json(), snapshot=snapshot)
|
||||
== foundation
|
||||
)
|
||||
|
||||
|
||||
def test_empty_logical_dimension_is_rejected_after_rebinding_all_bytes() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["instrument_id"] = ""
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError, match="dimension"):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_boolean_lineage_sequence_does_not_equal_integer_one() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["observation_lineage"][0]["observation_sequence"] = True
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("descriptor.time_semantics.pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.knowledge_time", {"start_inclusive": "2018-01-02T07:00:00Z"}),
|
||||
("descriptor.time_semantics.historical_availability", "established"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T01:01:00.0000001Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T09:01:00+08:00"),
|
||||
(
|
||||
"descriptor.time_semantics.earliest_external_knowledge",
|
||||
{"status": "unknown", "evidence_digest": "sha256:" + "0" * 64},
|
||||
),
|
||||
("descriptor.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("descriptor.published_at", "2026-02-30T00:00:00Z"),
|
||||
("descriptor.published_at", "2026-09-08T01:00:00Z"),
|
||||
("descriptor.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.qualification.usage", "as_available"),
|
||||
("descriptor.qualification.policy_version", "1.0.0"),
|
||||
("descriptor.quality.checks.0.severity", "advisory"),
|
||||
("descriptor.quality.checks.0.status", "failed"),
|
||||
("descriptor.quality.checks.0.check_id", "schema_conformance"),
|
||||
("descriptor.quality.status", "failed"),
|
||||
("descriptor.content.record_count", True),
|
||||
("descriptor.content.record_count", 9007199254740992),
|
||||
("descriptor.content.record_count", 2.0),
|
||||
("descriptor.content.content_digest", "bad"),
|
||||
("descriptor.content.manifest_digest", "sha256:" + "0" * 64),
|
||||
("descriptor.observation_manifest.batches", []),
|
||||
("descriptor.observation_manifest.batches.0.record_count", 1),
|
||||
("descriptor.observation_manifest.batches.0.chunk_index", True),
|
||||
("descriptor.observation_manifest.batches.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("descriptor.observation_manifest.batches.0.observation_kind", "first_published_at"),
|
||||
("descriptor.lineage.transformation.id", "rhtransform:private"),
|
||||
("descriptor.dataset.dimensions", ["instrument_id", "knowledge_time"]),
|
||||
("descriptor.dataset.dataset_id", "rhdataset:macroeconomic:" + "4" * 32),
|
||||
],
|
||||
)
|
||||
def test_snapshot_rejects_reidentified_invalid_declarations(path: str, value: Any) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
replace_at(source, path, value)
|
||||
# Noncanonical numbers are rejected before identity formation.
|
||||
if type(value) is not float and value != 9007199254740992:
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_snapshot_materialized_records_bind_the_public_golden_and_chunks() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
snapshot.verify_materialized_records([records()])
|
||||
snapshot.verify_materialized_records([list(reversed(records()))])
|
||||
chunks = [[records()[0]], [records()[1]]]
|
||||
bind_records(source, chunks)
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records(chunks)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records(chunks)
|
||||
rows = records()
|
||||
rows[0]["value"] = "0"
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"mutation", ["duplicate", "legacy", "range", "location", "missing_effective"]
|
||||
)
|
||||
def test_materialized_bad_records_rejected_even_with_matching_digests(mutation: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
if mutation == "duplicate":
|
||||
rows.append(deepcopy(rows[0]))
|
||||
elif mutation == "legacy":
|
||||
rows[0]["knowledge_time"] = "2026-09-08T01:00:00Z"
|
||||
elif mutation == "range":
|
||||
rows[0]["effective_time"] = "2018-01-01T07:00:00Z"
|
||||
elif mutation == "location":
|
||||
rows[0]["value"] = "/private/records.csv"
|
||||
else:
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_unknown_or_evidenced_knowledge_never_changes_usage() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"] = {
|
||||
"status": "evidenced",
|
||||
"range": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z",
|
||||
},
|
||||
"evidence_digest": digest({"synthetic_earliest": True}),
|
||||
}
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
parsed = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert (
|
||||
parsed.to_dict()["descriptor"]["time_semantics"]["historical_availability"]
|
||||
== "not_established"
|
||||
)
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"]["range"][
|
||||
"end_inclusive"
|
||||
] = "2026-09-08T01:01:00Z"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_rejected_snapshot_remains_readable_but_cannot_support_foundation() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["qualification"]["status"] = "rejected"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.require_qualified()
|
||||
foundation = golden("data-foundation")
|
||||
foundation["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in foundation["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(foundation, snapshot=snapshot)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2026-09-08T01:00:00Z"),
|
||||
("published_at", "2026-09-08T01:03:00Z"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "established"),
|
||||
("instrument_routes.0.observation_sequence", 2),
|
||||
("instrument_routes.0.supersedes_observation_id", "rhroutev2:sha256:" + "0" * 64),
|
||||
("instrument_routes.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("instrument_routes.0.history_completeness", "complete"),
|
||||
(
|
||||
"instrument_routes.0.earliest_external_knowledge",
|
||||
{"status": "unknown", "earliest_at": "2018-01-01T00:00:00Z"},
|
||||
),
|
||||
("instrument_routes.0.instrument_type", "index"),
|
||||
("instrument_routes.0.symbol", "WIND.TEST"),
|
||||
("instrument_routes.0.symbol", "A" * 33),
|
||||
("instrument_routes.0.calendar_id", "rhcalendar:" + "0" * 32),
|
||||
("trading_calendar_revisions.0.status", "closed"),
|
||||
("trading_calendar_revisions.0.sessions", []),
|
||||
("trading_calendar_revisions.0.sessions.0.closes_at", "2018-01-02T00:00:00Z"),
|
||||
("trading_calendar_revisions.0.session_date", "2018-02-30"),
|
||||
("standardized_views.0.instrument_route_revision_ids", []),
|
||||
("standardized_views.0.trading_calendar_revision_ids", []),
|
||||
("standardized_views.0.corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("standardized_views.0.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:02:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:06:00Z"),
|
||||
("standardized_views.0.usage", "as_available"),
|
||||
("corporate_action_coverage", []),
|
||||
("corporate_action_coverage.0.instrument_id", "rhinstrument:" + "0" * 32),
|
||||
("corporate_action_coverage.0.effective_time.end_inclusive", "2018-01-01T07:00:00Z"),
|
||||
("corporate_action_coverage.0.effective_time.start_inclusive", "2018-01-02T07:00:01Z"),
|
||||
("corporate_action_coverage.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("corporate_action_coverage.0.evidence_digests", []),
|
||||
("readiness.evidence_scope", "real_data"),
|
||||
("readiness.contract_validation.evidence_digests", []),
|
||||
(
|
||||
"readiness.real_data_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.production_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.live_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_foundation_rejects_semantic_forgery_after_reidentification(path: str, value: Any) -> None:
|
||||
source = golden("data-foundation")
|
||||
replace_at(source, path, value)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_v1_and_v2_never_coerce_each_other() -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
DatasetSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
with pytest.raises(FactorContractError):
|
||||
DataFoundationEnvelope.from_dict(golden("data-foundation"))
|
||||
old = json.loads((FIXTURES / "factor-contracts-v1.golden.json").read_text())
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(old["dataset_snapshot"])
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(old["data_foundation"])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"),
|
||||
snapshot=DatasetSnapshotEnvelope.from_dict(old["dataset_snapshot"]),
|
||||
)
|
||||
|
||||
|
||||
def test_deep_immutability_and_strict_canonical_json() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
source["descriptor"]["quality"]["status"] = "failed"
|
||||
snapshot.to_dict()["descriptor"]["qualification"]["status"] = "rejected"
|
||||
snapshot.require_qualified()
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
snapshot._payload = {}
|
||||
with pytest.raises(TypeError):
|
||||
snapshot.earliest_external_knowledge["status"] = "evidenced"
|
||||
foundation = parse_foundation(golden("data-foundation"))
|
||||
with pytest.raises(TypeError):
|
||||
foundation.views["new"] = next(iter(foundation.views.values()))
|
||||
for decoder, document in (
|
||||
(RetrospectiveSnapshotEnvelope.from_json, golden("dataset-snapshot")),
|
||||
(
|
||||
lambda value: RetrospectiveFoundationEnvelope.from_json(value, snapshot=snapshot),
|
||||
golden("data-foundation"),
|
||||
),
|
||||
):
|
||||
wire = canonical_json_bytes(document)
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(wire + b"\n")
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(b'{"schema_version":"2.0.0",' + wire[1:])
|
||||
|
||||
|
||||
def test_macro_period_is_not_inferred_as_an_effective_instant() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["dataset"].update(
|
||||
dataset_id="rhdataset:macroeconomic:" + "4" * 32,
|
||||
dataset_kind="macroeconomic",
|
||||
dimensions=["series_id", "observation_period"],
|
||||
)
|
||||
rows = [
|
||||
{
|
||||
"series_id": "cpi",
|
||||
"observation_period": "2018-01",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"value": "2.1",
|
||||
}
|
||||
]
|
||||
bind_records(source, [rows])
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def with_successor() -> dict[str, Any]:
|
||||
source = golden("data-foundation")
|
||||
previous = source["instrument_routes"][0]
|
||||
successor = deepcopy(previous)
|
||||
successor.update(
|
||||
observation_sequence=2,
|
||||
observed_by="2026-09-08T01:00:30Z",
|
||||
symbol="SIM0B",
|
||||
supersedes_observation_id=previous["route_revision_id"],
|
||||
)
|
||||
identify(successor, "route_revision_id", "rhroutev2:")
|
||||
source["instrument_routes"].append(successor)
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].append(
|
||||
successor["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
return source
|
||||
|
||||
|
||||
def test_retained_successor_requires_parent_time_and_view_ancestry() -> None:
|
||||
source = with_successor()
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.foundation_id == source["foundation_id"]
|
||||
assert parsed.published_at.isoformat() == "2026-09-08T01:05:00+00:00"
|
||||
assert parsed.contract_evidence_digests
|
||||
assert len(next(iter(parsed.views.values())).instrument_route_revision_ids) == 3
|
||||
for mutation in (
|
||||
"missing_parent",
|
||||
"equal_time",
|
||||
"omitted_ancestor",
|
||||
"duplicate_sequence",
|
||||
"wrong_lineage",
|
||||
):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "missing_parent":
|
||||
forged["instrument_routes"][-1]["supersedes_observation_id"] = (
|
||||
"rhroutev2:sha256:" + "0" * 64
|
||||
)
|
||||
elif mutation == "equal_time":
|
||||
forged["instrument_routes"][-1]["observed_by"] = forged["instrument_routes"][0][
|
||||
"observed_by"
|
||||
]
|
||||
elif mutation == "omitted_ancestor":
|
||||
forged["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
forged["instrument_routes"][0]["route_revision_id"]
|
||||
)
|
||||
elif mutation == "duplicate_sequence":
|
||||
forged["instrument_routes"][-1]["observation_sequence"] = 1
|
||||
else:
|
||||
forged["observation_lineage"][0]["evidence_digest"] = "sha256:" + "0" * 64
|
||||
seal_foundation(forged, rebuild_lineage=mutation != "wrong_lineage")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(forged)
|
||||
|
||||
|
||||
def test_index_and_closed_calendar_are_explicit_non_execution_data() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["instrument_routes"][0].update(asset_class="index", instrument_type="index")
|
||||
source["trading_calendar_revisions"][0].update(status="closed", sessions=[])
|
||||
seal_foundation(source)
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.to_dict()["instrument_routes"][0]["instrument_type"] == "index"
|
||||
assert parsed.to_dict()["readiness"]["live_validation"]["status"] == "not_validated"
|
||||
|
||||
|
||||
def test_real_scope_is_still_a_declaration_with_separate_evidence_and_coverage() -> None:
|
||||
snapshot_source = golden("dataset-snapshot")
|
||||
snapshot_source["evidence_scope"] = "real_data"
|
||||
identify(snapshot_source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_source)
|
||||
source = golden("data-foundation")
|
||||
source["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in source["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
source["readiness"]["evidence_scope"] = "real_data"
|
||||
source["readiness"]["real_data_validation"] = {
|
||||
"status": "validated",
|
||||
"evidence_digests": [digest({"synthetic_real_claim_test": 1})],
|
||||
}
|
||||
seal_foundation(source)
|
||||
parsed = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert parsed.real_data_validation_status == "validated" # NOT actual real-data evidence.
|
||||
for mutation in ("coverage", "reuse"):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "coverage":
|
||||
forged["corporate_action_coverage"][0].update(
|
||||
status="not_validated", evidence_digests=[]
|
||||
)
|
||||
else:
|
||||
forged["readiness"]["real_data_validation"] = deepcopy(
|
||||
forged["readiness"]["contract_validation"]
|
||||
)
|
||||
seal_foundation(forged)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(forged, snapshot=snapshot)
|
||||
synthetic = golden("data-foundation")
|
||||
synthetic["corporate_action_coverage"][0].update(status="not_validated", evidence_digests=[])
|
||||
seal_foundation(synthetic)
|
||||
assert parse_foundation(synthetic).real_data_validation_status == "not_validated"
|
||||
|
||||
|
||||
def test_action_must_belong_to_view_selected_instrument() -> None:
|
||||
source = golden("data-foundation")
|
||||
route = source["instrument_routes"][0]
|
||||
action = {
|
||||
"action_id": "rhaction:" + "7" * 32,
|
||||
"instrument_id": route["instrument_id"],
|
||||
"observation_sequence": 1,
|
||||
"observed_by": route["observed_by"],
|
||||
"earliest_external_knowledge": {
|
||||
"status": "evidenced",
|
||||
"earliest_at": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_action_earliest": 1}),
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": digest({"synthetic_action": 1}),
|
||||
"action_type": "cash_dividend",
|
||||
"status": "confirmed",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"terms_digest": digest({"synthetic_terms": 1}),
|
||||
}
|
||||
identify(action, "action_revision_id", "rhcav2:")
|
||||
source["corporate_action_revisions"] = [action]
|
||||
source["standardized_views"][0]["corporate_action_revision_ids"] = [
|
||||
action["action_revision_id"]
|
||||
]
|
||||
seal_foundation(source)
|
||||
assert len(parse_foundation(source).to_dict()["corporate_action_revisions"]) == 1
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
route["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_views_cannot_borrow_calendar_selection_from_each_other() -> None:
|
||||
source = golden("data-foundation")
|
||||
calendar = deepcopy(source["trading_calendar_revisions"][0])
|
||||
calendar["calendar_id"] = "rhcalendar:" + "9" * 32
|
||||
identify(calendar, "calendar_revision_id", "rhcalv2:")
|
||||
source["trading_calendar_revisions"].append(calendar)
|
||||
view = deepcopy(source["standardized_views"][0])
|
||||
view["view_id"] = "rhview:" + "8" * 32
|
||||
view["trading_calendar_revision_ids"] = [calendar["calendar_revision_id"]]
|
||||
source["standardized_views"].append(view)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"location", ["/private/a", "./a", "../a", "C:\\data\\a", "\\\\host\\a", "s3://private/a"]
|
||||
)
|
||||
def test_materialized_values_cannot_carry_physical_locations(location: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["value"] = location
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
@@ -0,0 +1,367 @@
|
||||
"""Synthetic v2 computation boundaries; never source authentication."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
FactorContractError,
|
||||
FactorDefinition,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
canonical_json_bytes,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
from quant_engine.retrospective_factor_contracts import (
|
||||
ResolvedRetrospectiveView,
|
||||
RetrospectiveCausation,
|
||||
RetrospectiveFactorSetRef,
|
||||
RetrospectiveInputBinding,
|
||||
RetrospectiveViewAvailability,
|
||||
)
|
||||
from test_retrospective_data_contracts import (
|
||||
digest,
|
||||
golden,
|
||||
identify,
|
||||
records,
|
||||
replace_at,
|
||||
seal_foundation,
|
||||
)
|
||||
|
||||
|
||||
def factor_arguments() -> dict[str, Any]:
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"), snapshot=snapshot
|
||||
)
|
||||
view = next(iter(foundation.views.values()))
|
||||
factor_inputs = (FactorInput("market", view.schema_digest, ("value",)),)
|
||||
definition = FactorDefinition.create(
|
||||
factor_id="neutral_close",
|
||||
version="1.0.0",
|
||||
formula="value",
|
||||
parameters={},
|
||||
implementation_digest=digest({"synthetic_formula": "identity"}),
|
||||
input_schema_digest=factor_input_schema_digest(factor_inputs),
|
||||
inputs=factor_inputs,
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=0,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "0.1.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
schema = {"fields": ["instrument_id", "value"]}
|
||||
output = [{"instrument_id": row["instrument_id"], "value": row["value"]} for row in records()]
|
||||
return {
|
||||
"definitions": (definition,),
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"selected_view_ref_ids": (view.view_ref_id,),
|
||||
"input_bindings": (
|
||||
RetrospectiveInputBinding(
|
||||
definition.definition_id, "market", view.view_ref_id, view.schema_digest
|
||||
),
|
||||
),
|
||||
"view_availability": (
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view_receipt": 1})
|
||||
),
|
||||
),
|
||||
"dataset_chunks": [records()],
|
||||
"resolved_views": (
|
||||
ResolvedRetrospectiveView(
|
||||
view.view_ref_id,
|
||||
canonical_json_bytes({"synthetic_schema": "neutral_close_v2"}),
|
||||
canonical_json_bytes(sorted(records(), key=canonical_json_bytes)),
|
||||
),
|
||||
),
|
||||
"output_quality": OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", digest({"synthetic_quality": 1})),),
|
||||
),
|
||||
"output_coverage": OutputCoverage(
|
||||
"complete", 2, 2, "records", "synthetic_market", digest({"synthetic_coverage": 1})
|
||||
),
|
||||
"output_schema_bytes": canonical_json_bytes(schema),
|
||||
"output_content_bytes": canonical_json_bytes(output),
|
||||
"output_artifact_ref": OutputArtifactRef.create(
|
||||
schema_digest=digest(schema), content_digest=digest(output)
|
||||
),
|
||||
"evaluation_at": "2026-09-08T01:06:00Z",
|
||||
"computed_at": "2026-09-08T01:07:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:08:00Z",
|
||||
"producer": ProducerIdentity("quant_engine", "0.1.0"),
|
||||
"code_revision": "d" * 40,
|
||||
"actor": ActorIdentity("service", "synthetic.research"),
|
||||
"correlation_id": "synthetic.retrospective",
|
||||
"causation": RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def decoding_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("definitions", "dataset_snapshot", "foundation")}
|
||||
|
||||
|
||||
def test_factor_result_closes_v2_inputs_and_preserves_v1_definition() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
wire = result.to_dict()
|
||||
assert result.schema_version == "2.0.0"
|
||||
assert result.factor_set_id.startswith("rhfactorsetv2:sha256:")
|
||||
assert result.definition_ids == (arguments["definitions"][0].definition_id,)
|
||||
assert result.definition_ids[0].startswith("rhfactorv1:")
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["availability_mode"] == "retrospective_replay"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert wire["decision_eligible"] is False
|
||||
assert "pit_cutoff" not in wire
|
||||
assert result.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
assert result.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
restored = RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json(), **decoding_arguments(arguments)
|
||||
)
|
||||
assert restored.to_dict() == wire
|
||||
assert restored.payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("contract_name", "researchhub.dataset-snapshot"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("foundation_id", "rhdfv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("selected_view_ref_ids", []),
|
||||
("selected_view_ref_ids", ["rhviewrefv2:sha256:" + "0" * 64]),
|
||||
("definition_ids", ["rhfactorv1:sha256:" + "0" * 64]),
|
||||
("input_bindings", []),
|
||||
("input_bindings.0.input_name", "volume"),
|
||||
("input_bindings.0.schema_digest", "sha256:" + "0" * 64),
|
||||
("input_bindings.0.view_ref_id", "rhviewrefv2:sha256:" + "0" * 64),
|
||||
("view_availability", []),
|
||||
("view_availability.0.available_at", "2026-09-08T01:03:30Z"),
|
||||
("view_availability.0.available_at", "2026-09-08T01:04:00.0000001Z"),
|
||||
("upstream_evidence.quality.checks.0.status", "failed"),
|
||||
("upstream_evidence.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("upstream_evidence.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("evidence_scope", "real_data"),
|
||||
("output_quality.status", "failed"),
|
||||
("output_quality.checks.0.status", "failed"),
|
||||
("output_coverage.status", "incomplete"),
|
||||
("output_coverage.observed_count", 1),
|
||||
("output_schema_digest", "sha256:" + "0" * 64),
|
||||
("output_artifact_ref.artifact_id", "rhfactoroutputv1:sha256:" + "0" * 64),
|
||||
("availability_mode", "as_available"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "declared_as_available"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("evaluation_at", "2018-01-02T07:00:00Z"),
|
||||
("evaluation_at", "2026-09-08T01:04:00Z"),
|
||||
("computed_at", "2026-09-08T01:05:00Z"),
|
||||
("artifact_available_at", "2026-09-08T01:06:00Z"),
|
||||
("producer.id", "research_platform"),
|
||||
("code_revision", "unknown"),
|
||||
("actor.id", "https://private/a"),
|
||||
("causation.id", "rhdfv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_factor_rejects_reidentified_semantic_forgery(path: str, value: Any) -> None:
|
||||
arguments = factor_arguments()
|
||||
row = RetrospectiveFactorSetRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "factor_set_id", "rhfactorsetv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(row, **decoding_arguments(arguments))
|
||||
|
||||
|
||||
def test_payload_validation_is_never_inherited_from_serialization() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
kwargs = decoding_arguments(arguments)
|
||||
reference = RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs)
|
||||
with pytest.raises(FactorContractError):
|
||||
reference.require_payloads_revalidated()
|
||||
checked = RetrospectiveFactorSetRef.from_dict(
|
||||
result.to_dict(),
|
||||
**kwargs,
|
||||
**{
|
||||
key: arguments[key]
|
||||
for key in (
|
||||
"output_schema_bytes",
|
||||
"output_content_bytes",
|
||||
"dataset_chunks",
|
||||
"resolved_views",
|
||||
)
|
||||
},
|
||||
)
|
||||
checked.require_payloads_revalidated()
|
||||
assert checked == result
|
||||
for extra in (
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"]},
|
||||
{"dataset_chunks": arguments["dataset_chunks"]},
|
||||
{"resolved_views": arguments["resolved_views"]},
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"], "output_content_bytes": b"[]"},
|
||||
{"dataset_chunks": arguments["dataset_chunks"], "resolved_views": []},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs, **extra)
|
||||
|
||||
|
||||
def test_create_verifies_actual_snapshot_and_each_resolved_view() -> None:
|
||||
for mutation in (
|
||||
"content",
|
||||
"schema",
|
||||
"snapshot",
|
||||
"duplicate_view",
|
||||
"noncanonical",
|
||||
"unknown_view",
|
||||
):
|
||||
arguments = factor_arguments()
|
||||
view = arguments["resolved_views"][0]
|
||||
if mutation == "content":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, view.schema_bytes, b"[]"),
|
||||
)
|
||||
elif mutation == "schema":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, b"{}", view.content_bytes),
|
||||
)
|
||||
elif mutation == "snapshot":
|
||||
arguments["dataset_chunks"][0][0]["value"] = "0"
|
||||
elif mutation == "duplicate_view":
|
||||
arguments["resolved_views"] = (view, view)
|
||||
elif mutation == "unknown_view":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(
|
||||
"rhviewrefv2:sha256:" + "0" * 64, view.schema_bytes, view.content_bytes
|
||||
),
|
||||
)
|
||||
else:
|
||||
arguments["output_content_bytes"] += b"\n"
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_parent_requires_exact_correlation_scope_and_actual_availability() -> None:
|
||||
arguments = factor_arguments()
|
||||
parent = RetrospectiveFactorSetRef.create(**arguments)
|
||||
child_args = {
|
||||
**arguments,
|
||||
"parent": parent,
|
||||
"causation": RetrospectiveCausation("factor_set", parent.factor_set_id),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:11:00Z",
|
||||
}
|
||||
child = RetrospectiveFactorSetRef.create(**child_args)
|
||||
assert child.factor_set_id != parent.factor_set_id
|
||||
assert (
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
child.to_json(), **decoding_arguments(arguments), parent=parent
|
||||
)
|
||||
== child
|
||||
)
|
||||
for changes in (
|
||||
{"parent": None},
|
||||
{"correlation_id": "different.correlation"},
|
||||
{"causation": RetrospectiveCausation("factor_set", "rhfactorsetv2:sha256:" + "0" * 64)},
|
||||
{"evaluation_at": "2026-09-08T01:07:59Z"},
|
||||
{"causation": arguments["causation"]},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**{**child_args, **changes})
|
||||
|
||||
|
||||
def test_definition_validity_is_checked_at_actual_evaluation() -> None:
|
||||
arguments = factor_arguments()
|
||||
arguments.update(
|
||||
evaluation_at="2027-01-01T00:00:00Z",
|
||||
computed_at="2027-01-01T00:01:00Z",
|
||||
artifact_available_at="2027-01-01T00:02:00Z",
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_factor_contract_is_immutable_and_v1_does_not_accept_it() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
exported = result.to_dict()
|
||||
exported["upstream_evidence"]["quality"]["status"] = "failed"
|
||||
assert result.upstream_evidence["quality"]["status"] == "passed"
|
||||
with pytest.raises(TypeError):
|
||||
result.upstream_evidence["quality"]["status"] = "failed"
|
||||
with pytest.raises(FactorContractError):
|
||||
FactorSetRef.from_dict(result.to_dict(), **decoding_arguments(arguments))
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json() + "\n", **decoding_arguments(arguments)
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
"rhviewrefv1:sha256:" + "0" * 64,
|
||||
"sha256:" + "0" * 64,
|
||||
)
|
||||
|
||||
|
||||
def test_missing_synthetic_or_real_readiness_cannot_be_relabelled() -> None:
|
||||
arguments = factor_arguments()
|
||||
snapshot_row = arguments["dataset_snapshot"].to_dict()
|
||||
snapshot_row["evidence_scope"] = "real_data"
|
||||
identify(snapshot_row, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_row)
|
||||
foundation_row = arguments["foundation"].to_dict()
|
||||
foundation_row["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
foundation_row["readiness"]["evidence_scope"] = "real_data"
|
||||
for view in foundation_row["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation_row)
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation_row, snapshot=snapshot)
|
||||
view = next(iter(foundation.views.values()))
|
||||
arguments.update(
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
evidence_scope="real_data",
|
||||
selected_view_ref_ids=(view.view_ref_id,),
|
||||
input_bindings=(
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
view.view_ref_id,
|
||||
view.schema_digest,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view": 1})
|
||||
),
|
||||
),
|
||||
causation=RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="real-data"):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
@@ -0,0 +1,655 @@
|
||||
"""New synthetic S4 evidence; historical valuation is not actual availability."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import FrozenInstanceError, replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
PortfolioRiskContractError,
|
||||
RiskAssessmentStatus,
|
||||
RiskFindingCode,
|
||||
)
|
||||
from quant_engine.artifact import EvidenceQualification, PerformanceEvidenceError
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.risk import ComponentRiskResult, CovarianceSnapshot, labeled_component_risk
|
||||
import quant_engine.retrospective_portfolio_risk_contracts as contracts
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
RetrospectivePortfolioDecision,
|
||||
RetrospectivePortfolioTarget,
|
||||
RetrospectiveRiskAssessment,
|
||||
build_retrospective_portfolio_decision,
|
||||
compute_retrospective_portfolio_receipt_digests,
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_artifact_contracts import synthetic_artifact
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import digest, replace_at
|
||||
|
||||
ASSETS = ("rhinstrument:" + "1" * 32, "rhinstrument:" + "2" * 32)
|
||||
CONTRACT_ERRORS = (
|
||||
FactorContractError,
|
||||
PortfolioRiskContractError,
|
||||
BacktestContractError,
|
||||
PerformanceEvidenceError,
|
||||
)
|
||||
|
||||
|
||||
def portfolio_arguments() -> dict[str, Any]:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
target = RetrospectivePortfolioTarget.create(
|
||||
backtest_run_id=run.run_id,
|
||||
dataset_snapshot_id=run.dataset_snapshot_id,
|
||||
weights={ASSETS[0]: 0.6, ASSETS[1]: 0.4},
|
||||
effective_at="2018-01-05T07:00:00Z",
|
||||
created_at="2026-09-08T01:12:00Z",
|
||||
)
|
||||
return {
|
||||
"backtest_run_ref": run,
|
||||
"manifest": manifest,
|
||||
"target": target,
|
||||
"objective_name": "synthetic_allocation",
|
||||
"objective_version": "1.0.0",
|
||||
"objective_digest": digest({"synthetic_objective": 1}),
|
||||
"model_name": "bounded_weights",
|
||||
"model_version": "1.0.0",
|
||||
"model_digest": digest({"synthetic_model": 1}),
|
||||
"expected_return_digest": digest({"synthetic_returns": 1}),
|
||||
"covariance_digest": "sha256:" + "a" * 64,
|
||||
"scenario_digest": digest({"synthetic_scenario": 1}),
|
||||
"constraints": ConstraintSetV1(
|
||||
gross_exposure_max=1.0,
|
||||
net_exposure_min=1.0,
|
||||
net_exposure_max=1.0,
|
||||
single_asset_min=0.2,
|
||||
single_asset_max=0.7,
|
||||
position_count_max=2,
|
||||
turnover_max=0.2,
|
||||
),
|
||||
"freshness_policy": FreshnessPolicy(
|
||||
max_manifest_age_seconds=3600, max_covariance_age_days=0
|
||||
),
|
||||
"prior_weights": {ASSETS[0]: 0.5, ASSETS[1]: 0.5},
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
|
||||
|
||||
def portfolio_receipt(arguments: dict[str, Any], **changes: Any) -> ComputationReceipt:
|
||||
values = compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key not in {"computed_at", "receipt"}}
|
||||
)
|
||||
return ComputationReceipt(
|
||||
**{
|
||||
"algorithm": "bounded_weights",
|
||||
"algorithm_version": "1.0.0",
|
||||
"implementation_digest": digest({"synthetic_implementation": 1}),
|
||||
"parameter_digest": digest({"synthetic_parameters": 1}),
|
||||
"input_digest": values["input_digest"],
|
||||
"constraint_digest": values["constraint_digest"],
|
||||
"output_digest": values["output_digest"],
|
||||
"status": "completed",
|
||||
"solver_required": False,
|
||||
"solver_name": None,
|
||||
"solver_version": None,
|
||||
"solver_config_digest": None,
|
||||
"iterations": None,
|
||||
"objective_value": None,
|
||||
"max_constraint_residual": values["max_constraint_residual"],
|
||||
"tolerance": 1e-12,
|
||||
"computed_at": arguments["computed_at"],
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_target_separates_historical_effective_time_from_actual_creation() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
target = arguments["target"]
|
||||
assert target.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert target.created_at == "2026-09-08T01:12:00Z"
|
||||
assert target.target_id.startswith("rhportfoliotargetv2:sha256:")
|
||||
assert target.to_dict()["usage"] == "retrospective_research"
|
||||
|
||||
|
||||
def test_portfolio_decision_preserves_constraints_and_actual_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
assert decision.decision_id.startswith("rhportfoliodecisionv2:sha256:")
|
||||
assert decision.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert decision.created_at == "2026-09-08T01:12:00Z"
|
||||
assert decision.computed_at == "2026-09-08T01:13:00Z"
|
||||
assert decision.gross_exposure == 1.0
|
||||
assert decision.position_count == 2
|
||||
assert decision.to_dict()["decision_eligible"] is False
|
||||
|
||||
|
||||
def covariance(arguments: dict[str, Any], **changes: Any) -> CovarianceSnapshot:
|
||||
return CovarianceSnapshot(
|
||||
**{
|
||||
"snapshot_id": "covariance:synthetic-retrospective",
|
||||
"as_of_date": "2018-01-05",
|
||||
"covariance": pd.DataFrame([[0.04, 0.01], [0.01, 0.09]], index=ASSETS, columns=ASSETS),
|
||||
"return_frequency": "1d",
|
||||
"periods_per_year": 252,
|
||||
"method": "provided",
|
||||
"window_start_date": "2018-01-02",
|
||||
"window_end_date": "2018-01-05",
|
||||
"observations": 4,
|
||||
"lookback_sessions": 4,
|
||||
"missing_policy": "complete_case",
|
||||
"data_snapshot_id": arguments["backtest_run_ref"].dataset_snapshot_id,
|
||||
"input_sha256": "a" * 64,
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def risk_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
return {
|
||||
"portfolio_decision": decision,
|
||||
"backtest_run_ref": arguments["backtest_run_ref"],
|
||||
"manifest": arguments["manifest"],
|
||||
"covariance": covariance(arguments),
|
||||
"risk_model_name": "euler_volatility",
|
||||
"risk_model_version": "1.0.0",
|
||||
"risk_model_digest": digest({"synthetic_risk_model": 1}),
|
||||
"risk_budget": {ASSETS[0]: 0.8, ASSETS[1]: 0.8},
|
||||
"portfolio_volatility_limit": 10.0,
|
||||
"groups": {ASSETS[0]: "equity", ASSETS[1]: "fixed_income"},
|
||||
"computed_at": "2026-09-08T01:14:00Z",
|
||||
}
|
||||
|
||||
|
||||
def test_risk_uses_historical_business_age_and_actual_computation_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.assessment_id.startswith("rhriskassessmentv2:sha256:")
|
||||
assert result.qualified is True
|
||||
assert result.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert result.computed_at == "2026-09-08T01:14:00Z"
|
||||
assert result.to_dict()["decision_eligible"] is False
|
||||
assert result.to_dict()["execution_validation"] == "not_validated"
|
||||
assert sum(result.percentage_risk.values()) == pytest.approx(1.0)
|
||||
assert sum(result.component_risk.values()) == pytest.approx(result.portfolio_volatility)
|
||||
|
||||
|
||||
def test_new_risk_computation_cannot_reuse_stale_actual_manifest_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments["computed_at"] = "2026-09-08T02:11:01Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
def target_with(arguments: dict[str, Any], **changes: Any) -> RetrospectivePortfolioTarget:
|
||||
row = arguments["target"].to_dict()
|
||||
return RetrospectivePortfolioTarget.create(
|
||||
**{
|
||||
key: value
|
||||
for key, value in {**row, **changes}.items()
|
||||
if key
|
||||
in {"backtest_run_id", "dataset_snapshot_id", "weights", "effective_at", "created_at"}
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def decision_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("backtest_run_ref", "manifest", "target")}
|
||||
|
||||
|
||||
def assessment_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
key: arguments[key]
|
||||
for key in ("portfolio_decision", "backtest_run_ref", "manifest", "covariance")
|
||||
}
|
||||
|
||||
|
||||
def reidentify(row: dict[str, Any], field: str, prefix: str) -> None:
|
||||
row.pop(field, None)
|
||||
encoded = json.dumps(
|
||||
row, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False
|
||||
)
|
||||
row[field] = prefix + "sha256:" + hashlib.sha256(encoded.encode()).hexdigest()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"parser",
|
||||
[RetrospectivePortfolioTarget, RetrospectivePortfolioDecision, RetrospectiveRiskAssessment],
|
||||
)
|
||||
def test_json_syntax_failures_use_typed_contract_errors(parser: Any) -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
parser.from_json(b"{")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"method": "alternate_estimator"},
|
||||
{"window_start_date": "2018-01-03"},
|
||||
{"window_end_date": "2018-01-04"},
|
||||
{"observations": 3},
|
||||
{"lookback_sessions": 5},
|
||||
{"missing_policy": "alternate_missing_policy"},
|
||||
],
|
||||
)
|
||||
def test_covariance_estimation_context_is_bound_into_the_result_identity(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
original = assess_retrospective_portfolio_risk(**arguments)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
changed = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert changed.assessment_id != original.assessment_id
|
||||
|
||||
|
||||
def test_canonical_roundtrips_and_immutable_results() -> None:
|
||||
base = portfolio_arguments()
|
||||
target = base["target"]
|
||||
assert RetrospectivePortfolioTarget.from_json(target.to_json().encode()) == target
|
||||
decision = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base))
|
||||
assert (
|
||||
RetrospectivePortfolioDecision.from_json(decision.to_json(), **decision_context(base))
|
||||
== decision
|
||||
)
|
||||
arguments = risk_arguments(base)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert (
|
||||
RetrospectiveRiskAssessment.from_json(
|
||||
result.to_json().encode(), **assessment_context(arguments)
|
||||
)
|
||||
== result
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
target.weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
target.created_at = "2018-01-05T07:00:00Z"
|
||||
with pytest.raises(TypeError):
|
||||
decision.target_weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(TypeError):
|
||||
result.component_risk[ASSETS[0]] = 0.1
|
||||
detached = result.to_dict()
|
||||
detached["component_risk"][ASSETS[0]] = 0.1
|
||||
assert detached != result.to_dict()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"weights": {}},
|
||||
{"weights": {"SIM0": 1.0}},
|
||||
{"weights": {ASSETS[0]: float("nan")}},
|
||||
{"weights": {ASSETS[0]: True}},
|
||||
{"backtest_run_id": "rhbacktestrunv1:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhds:sha256:" + "0" * 64},
|
||||
{"effective_at": "2026-09-09T01:00:00Z"},
|
||||
{"created_at": "2026-09-08T01:12:00.1234567Z"},
|
||||
{"effective_at": "2018-01-05T15:00:00+08:00"},
|
||||
],
|
||||
)
|
||||
def test_target_rejects_legacy_ambiguous_and_nonfinite_inputs(change: dict[str, Any]) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
target_with(portfolio_arguments(), **change)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("usage", "live"),
|
||||
("historical_availability", "established"),
|
||||
("schema_version", "1.0.0"),
|
||||
("extra", True),
|
||||
("target_id", "rhportfoliotargetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_target_rejects_wire_mutations(path: str, value: Any) -> None:
|
||||
row = portfolio_arguments()["target"].to_dict()
|
||||
row[path] = value
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_dict(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field", ["input_digest", "constraint_digest", "output_digest"])
|
||||
def test_receipt_digests_are_recomputed(field: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments, **{field: "sha256:" + "0" * 64})
|
||||
with pytest.raises(FactorContractError, match="independently recomputed"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", ["failed", "fallback"])
|
||||
def test_failed_or_fallback_solver_cannot_form_a_decision(status: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status=status,
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="failed/fallback"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"backtest_run_id": "rhbacktestrunv2:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"weights": {"rhinstrument:" + "f" * 32: 1.0}},
|
||||
{"created_at": "2026-09-08T01:10:00Z"},
|
||||
{"created_at": "2026-09-08T01:14:00Z"},
|
||||
],
|
||||
)
|
||||
def test_decision_closes_target_identity_assets_and_actual_time(change: dict[str, Any]) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
arguments["target"] = target_with(arguments, **change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_actual_manifest_freshness_boundary_and_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00Z"
|
||||
assert (
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
).computed_at
|
||||
== arguments["computed_at"]
|
||||
)
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00.000001Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=portfolio_receipt(arguments))
|
||||
arguments["computed_at"] = "2026-09-08T01:13:00Z"
|
||||
with pytest.raises(FactorContractError, match="receipt actual time"):
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments, computed_at="2026-09-08T01:13:01Z")
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_tables_and_qualification_are_revalidated_at_s4_boundary() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
manifest = arguments["manifest"]
|
||||
artifact = manifest._artifact
|
||||
arguments["manifest"] = build_retrospective_backtest_evidence_manifest(
|
||||
arguments["backtest_run_ref"],
|
||||
artifact,
|
||||
artifact_available_at=manifest.artifact_available_at,
|
||||
qualification=EvidenceQualification.EXPLORATORY,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="contract-qualified"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments["manifest"] = manifest
|
||||
# Public access is an isolated copy. Simulate corruption of the retained bytes,
|
||||
# beyond that normal interface, to exercise the consumer's independent recheck.
|
||||
artifact._performance.loc[0, "n_days"] += 1
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_constraint_residuals_and_prior_assets_cannot_be_bypassed() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["constraints"] = ConstraintSetV1(gross_exposure_max=0.9)
|
||||
# A solver may report convergence within its tolerance; actual contract constraints still bind.
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status="converged",
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
tolerance=0.2,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="violates supported constraints"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments = portfolio_arguments()
|
||||
arguments["prior_weights"] = {"rhinstrument:" + "f" * 32: 0.5}
|
||||
with pytest.raises(FactorContractError, match="prior assets"):
|
||||
compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key != "computed_at"}
|
||||
)
|
||||
arguments["prior_weights"] = None
|
||||
with pytest.raises(PortfolioRiskContractError, match="prior"):
|
||||
portfolio_receipt(arguments)
|
||||
|
||||
|
||||
def test_optional_prior_budget_limit_and_groups_have_explicit_empty_semantics() -> None:
|
||||
base = portfolio_arguments()
|
||||
base["constraints"] = ConstraintSetV1(gross_exposure_max=1.0)
|
||||
base["prior_weights"] = None
|
||||
arguments = risk_arguments(base)
|
||||
arguments.update(risk_budget=None, portfolio_volatility_limit=None, groups=None)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.qualified is True
|
||||
assert result.risk_budget == {}
|
||||
assert result.group_exposure == {}
|
||||
assert result.groups is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("gross_exposure", True),
|
||||
("position_count", 2.0),
|
||||
("target_weights." + ASSETS[0], 0.5),
|
||||
("schema_version", "1.0.0"),
|
||||
("observation_cutoff", "2018-01-05T07:00:00Z"),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_decision_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
base = portfolio_arguments()
|
||||
row = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base)).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "decision_id", "rhportfoliodecisionv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioDecision.from_dict(row, **decision_context(base))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"as_of_date": "2018-01-06"},
|
||||
{"as_of_date": "2018-01-04", "window_end_date": "2018-01-04"},
|
||||
{"window_start_date": None, "window_end_date": None},
|
||||
{"data_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"input_sha256": "b" * 64},
|
||||
],
|
||||
)
|
||||
def test_covariance_business_time_bounds_and_source_binding(change: dict[str, Any]) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
with pytest.raises(FactorContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,index,columns",
|
||||
[
|
||||
([[float("nan"), 0.0], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.1], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], ASSETS[0]), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], "unknown"), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], ASSETS, (ASSETS[0], "unknown")),
|
||||
],
|
||||
)
|
||||
def test_covariance_structure_is_checked_before_computation(
|
||||
matrix: Any, index: Any, columns: Any
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=index, columns=columns)
|
||||
)
|
||||
with pytest.raises(PortfolioRiskContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"risk_budget": {ASSETS[0]: -0.1}},
|
||||
{"risk_budget": {"unknown": 0.1}},
|
||||
{"portfolio_volatility_limit": -0.1},
|
||||
{"groups": {ASSETS[0]: "equity"}},
|
||||
{"groups": []},
|
||||
{"risk_model_version": "latest"},
|
||||
{"risk_model_name": "/private/model"},
|
||||
{"computed_at": "2026-09-08T01:12:59Z"},
|
||||
{"portfolio_decision": object()},
|
||||
{"covariance": object()},
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_invalid_models_budgets_clocks_and_untyped_inputs(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,finding",
|
||||
[
|
||||
([[1.0, 2.0], [2.0, 1.0]], RiskFindingCode.COVARIANCE_NOT_PSD),
|
||||
([[0.0, 0.0], [0.0, 0.0]], RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE),
|
||||
],
|
||||
)
|
||||
def test_numerical_unavailability_is_not_qualification(
|
||||
matrix: Any, finding: RiskFindingCode
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=ASSETS, columns=ASSETS)
|
||||
)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.qualified is False
|
||||
assert result.findings == (finding,)
|
||||
assert result.portfolio_volatility is None
|
||||
|
||||
|
||||
def test_risk_uses_the_existing_numeric_implementation_exactly_once(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
calls = []
|
||||
|
||||
def recorded(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
calls.append((weights, matrix))
|
||||
return labeled_component_risk(weights, matrix)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", recorded)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert len(calls) == 1
|
||||
expected = labeled_component_risk(*calls[0])
|
||||
assert result.component_risk == expected.component.to_dict()
|
||||
assert result.portfolio_volatility == expected.portfolio_volatility
|
||||
|
||||
|
||||
def test_unknown_numeric_failures_are_sanitized(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def failed(*args: Any) -> ComponentRiskResult:
|
||||
raise ValueError("synthetic internal detail")
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", failed)
|
||||
with pytest.raises(PortfolioRiskContractError, match="risk computation failed") as error:
|
||||
assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert "internal detail" not in str(error.value)
|
||||
|
||||
|
||||
def test_nonclosed_decomposition_is_unavailable(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def nonclosed(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
output = labeled_component_risk(weights, matrix)
|
||||
return replace(output, component=output.component * 0.5)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", nonclosed)
|
||||
result = assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.findings == (RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED,)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change", [{"portfolio_volatility_limit": 0.0}, {"risk_budget": {ASSETS[0]: 0.0}}]
|
||||
)
|
||||
def test_budget_breach_keeps_ready_but_unqualified_evidence(change: dict[str, Any]) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.READY
|
||||
assert result.qualified is False
|
||||
assert result.findings == (RiskFindingCode.RISK_BUDGET_BREACH,)
|
||||
assert result.decision_eligible is False
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("qualified", 1),
|
||||
("portfolio_volatility", 1.0),
|
||||
("component_risk." + ASSETS[0], 1.0),
|
||||
("schema_version", "1.0.0"),
|
||||
("covariance_matrix_digest", "sha256:" + "0" * 64),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
row = assess_retrospective_portfolio_risk(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "assessment_id", "rhriskassessmentv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveRiskAssessment.from_dict(row, **assessment_context(arguments))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw",
|
||||
[
|
||||
b'{"x":1,"x":2}',
|
||||
b'{ "x":1}',
|
||||
b"[]",
|
||||
b'{"x":NaN}',
|
||||
b'{"x":Infinity}',
|
||||
b'{"x":9007199254740992}',
|
||||
1,
|
||||
],
|
||||
)
|
||||
def test_json_profiles_reject_ambiguous_nonfinite_and_noncanonical_input(raw: Any) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_json(raw)
|
||||
Reference in New Issue
Block a user