Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
861c1e97a8 | ||
|
|
68dd68392a | ||
|
|
a724e1e57a | ||
|
|
78d65b4db0 | ||
|
|
598c2b92a2 | ||
|
|
62ed09842d | ||
|
|
e782e223f7 | ||
|
|
015c1a3602 | ||
|
|
2bc8aea435 | ||
|
|
03e38d5123 | ||
|
|
90a43adda2 |
+29
-2
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"module_id": "quant_engine",
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 1, "effective_from": "2026-08-20T00:00:00+08:00"},
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 6, "effective_from": "2026-09-08T19:33:40+08:00"},
|
||||
"repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"},
|
||||
"bounded_context": {
|
||||
"domain": "quantitative-research-engine",
|
||||
@@ -11,6 +11,7 @@
|
||||
"Submitting live orders, routing trades, managing brokerage accounts, or claiming transaction execution",
|
||||
"Owning market-data source facts, research-result publication, or platform presentation state",
|
||||
"Loading provider credentials, brokerage credentials, or production secrets",
|
||||
"Granting portfolio approval, maker-checker decisions, publication eligibility, paper execution, or live execution authority",
|
||||
"Changing financial model semantics through module metadata"
|
||||
]
|
||||
},
|
||||
@@ -18,13 +19,39 @@
|
||||
{"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"},
|
||||
{"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"},
|
||||
{"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"},
|
||||
{"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact and performance-methodology evidence without recomputation, persistence, or decision authority.", "status": "operational"},
|
||||
{"id": "portfolio-risk-computation-contracts", "summary": "Verify deterministic portfolio-computation receipts and expose S3-bound portfolio decisions and risk assessments without adding algorithms or execution authority.", "status": "operational"},
|
||||
{"id": "retrospective-computation-contracts", "summary": "Decode observation-aware v2 data and expose explicit retrospective factor, backtest, portfolio and risk contracts with two clocks, no historical-availability claim and no execution authority.", "status": "operational"},
|
||||
{"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"}
|
||||
],
|
||||
"data": {"owns": [
|
||||
{"asset_id": "quantitative-model-implementations", "kind": "model", "classification": "internal"},
|
||||
{"asset_id": "simulation-and-metric-results", "kind": "artifact", "classification": "confidential"}
|
||||
]},
|
||||
"contracts": {"provides": [], "consumes": []},
|
||||
"contracts": {
|
||||
"provides": [
|
||||
{"contract_id": "researchhub.factor-definition", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
|
||||
{"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-run-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/governed_pipeline.py"},
|
||||
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
|
||||
{"contract_id": "researchhub.performance-evidence", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
|
||||
{"contract_id": "researchhub.portfolio-decision", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.factor-set-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_factor_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-run-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_backtest_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
|
||||
{"contract_id": "researchhub.performance-evidence", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
|
||||
{"contract_id": "researchhub.portfolio-target", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.portfolio-decision", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"}
|
||||
],
|
||||
"consumes": [
|
||||
{"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"},
|
||||
{"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"},
|
||||
{"contract_id": "researchhub.dataset-snapshot", "version": "2.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_retrospective_envelope_and_materialized_chunks"},
|
||||
{"contract_id": "researchhub.data-foundation", "version": "2.0.0", "authority": "researchhub.data", "admission": "observation_bound_selected_views_and_materialized_bytes"}
|
||||
]
|
||||
},
|
||||
"dependencies": [],
|
||||
"agent_context": {
|
||||
"default_entrypoints": [
|
||||
|
||||
@@ -19,13 +19,17 @@
|
||||
## 模块
|
||||
|
||||
- `alpha_factors` — 158 alpha 公式 + 24 基础算子(移植自 qlib alpha158)
|
||||
- `factor_contracts` — `FactorDefinition` / `FactorSetRef` v1 纯计算合同、严格 PIT/availability 输入准入与显式 legacy 投影
|
||||
- `execution` — A 股长仓执行仿真(成本/滑点/现金约束)+ 稀疏调仓/完整交易日 Ledger + 可投影成交与 NAV 审计;T+1、涨跌停、成交量与价差提供独立约束函数
|
||||
- `indicators` — 50+ 技术指标(MACD / KDJ / 布林 / ATR / ADX / 等)
|
||||
- `data_adapter` — 桥接 qtdb_pro 长表与新模块(rename / long-wide / 复权 / vwap 代理)
|
||||
- `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark)
|
||||
- `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表
|
||||
- `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排)
|
||||
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表与 manifest
|
||||
- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef`
|
||||
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest`
|
||||
- `portfolio_risk_contracts` — S3 证据闭合的 `PortfolioDecision` / `RiskAssessment` v1;独立复核 freshness、约束与 computation receipt,并复用既有标签安全风险分解
|
||||
- `retrospective_*_contracts` — 未发布的显式 v2 回顾性合同:区分历史业务日期与实际可得/计算时间,保留 v1 和现有金融公式,不授予历史可得性、发布或执行权限;见 [v2 接口说明](docs/RETROSPECTIVE_COMPUTATION_V2.md)
|
||||
- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计
|
||||
- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效
|
||||
- `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test)
|
||||
@@ -55,6 +59,9 @@ pytest # 单元测试
|
||||
pytest --cov=src # 覆盖率
|
||||
mypy --strict src/ # 类型检查
|
||||
ruff check src/ tests/ # lint
|
||||
|
||||
# 无网络、无数据库、无券商的架构烟测
|
||||
uv run python -m quant_engine.governed_pipeline
|
||||
```
|
||||
|
||||
## 使用
|
||||
@@ -191,6 +198,220 @@ print(backtest.stats())
|
||||
print(backtest.benchmark_report())
|
||||
```
|
||||
|
||||
## 因子/特征合同 v1
|
||||
|
||||
`quant_engine.factor_contracts` 提供 `researchhub.factor-definition` 与
|
||||
`researchhub.factor-set-ref` `1.0.0`。合同使用受限 canonical JSON:只接受 ASCII
|
||||
lower-snake-case object key、UTF-8 string、bool/null 和 safe integer;小数参数必须用显式
|
||||
canonical decimal string。定义、输入映射、上游证据、输出 schema/content 和 lineage 的任一
|
||||
语义变化都会产生新 identity。
|
||||
|
||||
创建 `FactorSetRef` 必须提供完整且可重算 identity 的 `DatasetSnapshotEnvelope` 与
|
||||
`DataFoundationEnvelope`,不能用 ID 字符串或布尔值代替资格证明。每个因子输入都要映射到一个
|
||||
实际选中的 `StandardizedViewRef`,schema 必须同时匹配定义和 view;未消费、缺失、重复或跨
|
||||
snapshot/Foundation/PIT 的 view 都会失败关闭。snapshot PIT 可以早于 Foundation/view PIT,
|
||||
但始终满足 knowledge ≤ snapshot PIT ≤ Foundation/view/FactorSet PIT ≤ evaluation。
|
||||
|
||||
```python
|
||||
from quant_engine.factor_contracts import (
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorDefinition,
|
||||
FactorSetRef,
|
||||
)
|
||||
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(dataset_snapshot_v1)
|
||||
foundation = DataFoundationEnvelope.from_dict(data_foundation_v1)
|
||||
|
||||
# definition 必须是完整的 FactorDefinition;FactorSetRef.create 还要求显式 input bindings、
|
||||
# view availability、output quality/coverage、canonical output bytes 和 immutable artifact ref。
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
**explicit_factor_set_evidence,
|
||||
)
|
||||
```
|
||||
|
||||
`availability_mode="as_available"` 声明 source/view 和计算产物在历史 evaluation 前实际可用;
|
||||
`"retrospective_replay"` 保留历史 evaluation,但要求真实 publication/view creation、compute 和
|
||||
artifact 时间位于之后,并固定 `historical_availability="not_established"`。两种模式都不会授予
|
||||
decision、real-data、production、paper 或 live readiness。
|
||||
|
||||
旧 `governed_pipeline.FactorVersion` 的四字段构造器、`factor_id@version`、run/target/risk/order
|
||||
identity 均保持不变。迁移只能通过 content-addressed `LegacyFactorBinding`,再显式调用
|
||||
`bind_legacy_factor()` 或 `project_legacy_factor()`;后者是有损投影,不表示旧 digest 与新定义
|
||||
digest 等价,也不会把旧 run 静默升级为新合同。
|
||||
|
||||
## 回测引用与证据合同 v1
|
||||
|
||||
`quant_engine.governed_pipeline.BacktestRunRef` 是合格回测运行身份的唯一权威。`run_id` 只由
|
||||
已验收的 Dataset Snapshot / Data Foundation / `FactorSetRef` 身份、universe、日历与公司行动
|
||||
祖先、策略、执行/成本模型、严格整数 seed、完整代码提交、环境锁、配置、时间和重放血缘决定;
|
||||
它不包含任何输出摘要。重放必须绑定直接父运行、连续 attempt 和不变的
|
||||
`replay_spec_digest`,输入漂移或血缘环会失败关闭。
|
||||
|
||||
`quant_engine.artifact.BacktestEvidenceManifest` 只摘要 `ResearchRunArtifact` 已有的九张事实表。
|
||||
固定 `offline_research_v1` 映射为 `run`、`signal`、`fill`、`position_nav`、`performance`、
|
||||
`attribution`、`risk_snapshot` 与 `replay`;每张表都保留列模式摘要、行数和内容摘要,空 risk
|
||||
表也必须有稳定 schema。`manifest_id` 由完整 RunRef 与输出证据决定,因此结果变化不会反向改变
|
||||
`run_id`。当前 artifact 不拥有订单或拒绝事实,所以此画像明确不声明 `order` / `rejection`。
|
||||
|
||||
旧 `BacktestRun` 只能通过 `build_legacy_backtest_evidence_manifest()` 显式映射为
|
||||
`LEGACY_EXPLORATORY`;不能隐式提升为 `CONTRACT_QUALIFIED`。所有资格均只描述离线证据闭合,
|
||||
不表示投资有效、组合获批、Paper、生产或实盘就绪。
|
||||
|
||||
## 绩效证据与方法论合同 v1
|
||||
|
||||
`quant_engine.artifact.PerformanceEvidenceV1` 在现有计算和事实表之外增加一层只读、内容寻址的
|
||||
owner 证据。`build_performance_evidence()` 只接受同一运行的完整 `ResearchRunArtifact`、
|
||||
`BacktestRunRef` 与 `CONTRACT_QUALIFIED BacktestEvidenceManifest`;它核对全部 artifact 表、
|
||||
performance 表和唯一行摘要,并绑定 artifact、row 与严格对齐 benchmark series 的独立摘要。
|
||||
生产 builder 不重算、填补、重命名或覆盖任何绩效值。
|
||||
|
||||
方法论固定为日简单收益、252 期年化、绝对指标年化无风险利率 `0.0`、benchmark 日无风险利率
|
||||
`0.0`,以及 benchmark 存在时的 `exact_session_index`。相对指标使用封闭 availability:无基准为
|
||||
`benchmark_absent`;active variance、benchmark variance 或 alpha 几何年化域不足时分别使用
|
||||
对应 `not_estimable_*` 原因。benchmark 存在时 tracking error 始终必须是有限非负值;null 不会
|
||||
被转成零。
|
||||
|
||||
```python
|
||||
from quant_engine.artifact import build_performance_evidence
|
||||
|
||||
performance_evidence = build_performance_evidence(
|
||||
artifact,
|
||||
backtest_run_ref,
|
||||
backtest_evidence_manifest,
|
||||
)
|
||||
canonical_bytes = performance_evidence.canonical_bytes()
|
||||
```
|
||||
|
||||
该合同范围固定为 `offline_research_only`。它不授予排名、推荐、决策、发布、论文、Paper、生产、
|
||||
实盘、交易或投资建议权限,也不包含原始参数、returns、NAV、benchmark series、表字节、存储
|
||||
locator、URI 或凭证。
|
||||
|
||||
## 组合决策与风险评估合同 v1
|
||||
|
||||
`quant_engine.portfolio_risk_contracts` 是现有计算 owner 外围的薄合同层。创建
|
||||
`PortfolioDecision` 必须同时提供完整 `BacktestRunRef`、嵌入同一 RunRef 的
|
||||
`CONTRACT_QUALIFIED` 非 legacy `BacktestEvidenceManifest`、现有 `PortfolioTarget`、
|
||||
`FreshnessPolicy`、`ConstraintSetV1` 与 `ComputationReceipt`。适配器会从权威输入独立重算
|
||||
receipt 的 input/constraint/output digest、敞口、持仓数和 L1 turnover 残差;receipt 自报
|
||||
成功、fallback 或放宽 tolerance 均不能替代复核。
|
||||
|
||||
```python
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
assess_portfolio_risk,
|
||||
build_portfolio_decision,
|
||||
compute_portfolio_receipt_digests,
|
||||
)
|
||||
|
||||
freshness = FreshnessPolicy(
|
||||
max_manifest_age_seconds=3600,
|
||||
max_covariance_age_days=5,
|
||||
)
|
||||
constraints = ConstraintSetV1(
|
||||
gross_exposure_max=1.0,
|
||||
single_asset_max=0.10,
|
||||
position_count_max=20,
|
||||
turnover_max=0.30,
|
||||
)
|
||||
|
||||
# 生产者先形成公开 canonical digest;decision 构建时仍会独立重算。
|
||||
expected = compute_portfolio_receipt_digests(
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
target=portfolio_target,
|
||||
objective_name="long_only_allocation",
|
||||
objective_version="1.0.0",
|
||||
objective_digest=objective_digest,
|
||||
model_name="factor_weighting",
|
||||
model_version="1.0.0",
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
|
||||
receipt = ComputationReceipt(
|
||||
algorithm="factor_weighting",
|
||||
algorithm_version="1.0.0",
|
||||
implementation_digest=implementation_digest,
|
||||
parameter_digest=parameter_digest,
|
||||
input_digest=expected["input_digest"],
|
||||
constraint_digest=expected["constraint_digest"],
|
||||
output_digest=expected["output_digest"],
|
||||
status="completed",
|
||||
solver_required=False,
|
||||
solver_name=None,
|
||||
solver_version=None,
|
||||
solver_config_digest=None,
|
||||
iterations=None,
|
||||
objective_value=None,
|
||||
max_constraint_residual=expected["max_constraint_residual"],
|
||||
tolerance=1e-12,
|
||||
computed_at=computed_at,
|
||||
)
|
||||
|
||||
decision = build_portfolio_decision(
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
target=portfolio_target,
|
||||
objective_name="long_only_allocation",
|
||||
objective_version="1.0.0",
|
||||
objective_digest=objective_digest,
|
||||
model_name="factor_weighting",
|
||||
model_version="1.0.0",
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness,
|
||||
receipt=receipt,
|
||||
computed_at=computed_at,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
|
||||
assessment = assess_portfolio_risk(
|
||||
portfolio_decision=decision,
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
covariance=covariance_snapshot,
|
||||
risk_model_name="euler_volatility",
|
||||
risk_model_version="1.0.0",
|
||||
risk_model_digest=risk_model_digest,
|
||||
)
|
||||
```
|
||||
|
||||
`source_universe_digest` 保留 S3 研究 universe 身份,`portfolio_asset_set_digest` 只描述实际
|
||||
目标资产标签;二者不会互相冒充成员证明。风险评估在任何数值计算前要求 covariance、target、
|
||||
RunRef 的 dataset identity 三方一致,并且只调用一次现有 `labeled_component_risk()`。合同中的
|
||||
`qualified` 仅表示 S4.1 计算证据闭合,不授予 maker-checker、发布、订单、Paper、生产或实盘权限。
|
||||
|
||||
## 治理垂直切片
|
||||
|
||||
`governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。
|
||||
调用方必须显式提供 `DatasetSnapshot`、`FactorVersion`、`StrategyVersion`、代码提交和
|
||||
`RiskPolicy`。模块只会生成 `environment="paper"` 的订单意图,不连接数据库、数据供应商或
|
||||
券商;风险决策为拒绝时,订单意图固定为空,直接调用创建函数也会失败关闭。
|
||||
|
||||
该切片对应 ResearchHub 架构的首个可执行验收链路:
|
||||
|
||||
```text
|
||||
DatasetSnapshot → FactorVersion → StrategyVersion → BacktestRun
|
||||
→ PortfolioTarget → RiskDecision → PaperOrderIntent
|
||||
```
|
||||
|
||||
平台总架构、五仓职责和十二层能力映射仍以 `research_platform/docs/architecture/` 为权威;
|
||||
本仓只拥有纯计算与离线模拟合同。
|
||||
|
||||
## 与 research_results 的关系
|
||||
|
||||
`research_results` 依赖 `quant_engine`(通过 re-export 保持向后兼容):
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
# Retrospective computation contracts v2 (unreleased)
|
||||
|
||||
This pure, storage-neutral compatibility path consumes the separate data-contract
|
||||
major 2.0.0. It does not migrate, reinterpret or relax the accepted v1 contracts.
|
||||
No financial formula, execution simulation, dependency lock, production database,
|
||||
publisher or live/paper-order interface changes here. Package version is unchanged;
|
||||
the new contract major is not a package release or deployment.
|
||||
|
||||
## Explicit public boundaries
|
||||
|
||||
| Module | Public types/builders | Changed wire identity |
|
||||
| --- | --- | --- |
|
||||
| `retrospective_data_contracts` | `RetrospectiveSnapshotEnvelope`, `RetrospectiveFoundationEnvelope` | `rhdsv2`, `rhdfv2`; consume RP-owned 2.0.0 data semantics |
|
||||
| `retrospective_factor_contracts` | `RetrospectiveFactorSetRef`, typed input/view/causation bindings, `ResolvedRetrospectiveView` | `rhfactorsetv2` |
|
||||
| `retrospective_backtest_contracts` | `RetrospectiveBacktestRunRef` | `rhbacktestrunv2` |
|
||||
| `retrospective_artifact_contracts` | `RetrospectiveBacktestEvidenceManifest`, `RetrospectivePerformanceEvidence` and their builders | `rhbacktestevidencev2`, `rhperformancev2` |
|
||||
| `retrospective_portfolio_risk_contracts` | `RetrospectivePortfolioTarget`, `RetrospectivePortfolioDecision`, `RetrospectiveRiskAssessment`; receipt-digest, decision and assessment builders | `rhportfoliotargetv2`, `rhportfoliodecisionv2`, `rhriskassessmentv2` |
|
||||
|
||||
These are separate types and domain-separated content identities. There is no
|
||||
automatic v1-to-v2 cast. Unknown schema versions and fields are rejected. The
|
||||
performance wire keeps its named schema `researchhub.performance-evidence.v2`;
|
||||
the other new computation contracts use `schema_version: 2.0.0`.
|
||||
|
||||
FactorDefinition, factor-output byte references, output quality/coverage,
|
||||
ConstraintSetV1, FreshnessPolicy, ComputationReceipt, CovarianceSnapshot, financial
|
||||
algorithms, performance metric/methodology IDs and the nine ResearchRunArtifact
|
||||
tables keep their existing semantics. The table schema remains **1.1.0**. Reusing
|
||||
these neutral primitives does not make a new-major upstream reference v1-compatible.
|
||||
|
||||
## Two clocks, not backdated evidence
|
||||
|
||||
Every new result fixes `usage=retrospective_research` and
|
||||
`historical_availability=not_established`. A business date describes the historical
|
||||
period being researched. Observation, publication, evaluation, artifact availability,
|
||||
target creation and computation describe actual events, and must not be backdated.
|
||||
Public v2 instants require UTC `Z` with at most six fractional digits.
|
||||
|
||||
`observation_cutoff` and chunk `observed_by` are upper-bound observations. They are
|
||||
not the earliest public knowledge time or a PIT cutoff. Unknown earliest knowledge
|
||||
stays unknown; a supplied knowledge-evidence digest is not authenticated by parsing.
|
||||
Foundation observation sequences describe retained revisions, not complete original
|
||||
history. Selected view routes, calendars, corporate-action coverage and lineage
|
||||
must close exactly within the supplied Foundation.
|
||||
|
||||
Required actual order for factor/backtest evidence is:
|
||||
|
||||
1. Foundation publication <= factor evaluation <= factor computation <= factor availability.
|
||||
2. Factor availability <= backtest evaluation <= artifact start <= artifact finish
|
||||
<= backtest computation <= artifact availability.
|
||||
3. Artifact availability <= target creation <= portfolio computation <= risk computation.
|
||||
|
||||
RetrospectivePortfolioTarget has a historical `effective_at` and a distinct actual
|
||||
`created_at`. PortfolioDecision carries both plus actual `computed_at`. Covariance
|
||||
window end <= covariance as-of date <= the historical effective date; covariance
|
||||
maximum age is measured against that historical date. Manifest maximum age is
|
||||
measured against **actual** portfolio and risk computation separately. Passing one
|
||||
age check cannot substitute for the other. Generic v1 receipt timestamps retain
|
||||
their original normalization; binding compares parsed actual instants.
|
||||
|
||||
## Materialized bytes and reference-only reads
|
||||
|
||||
Snapshot decoding checks structure, all six blocking-quality declarations,
|
||||
qualification/time ordering, observation receipts and identities.
|
||||
`verify_materialized_records` additionally checks supplied chunks, per-chunk and
|
||||
aggregate content, counts, dimensions, effective ranges and macro effective instants.
|
||||
Provider/physical paths are forbidden in public metadata and materialized records.
|
||||
|
||||
Factor creation requires actual snapshot chunks, selected view schema/content bytes,
|
||||
and factor-output schema/content bytes. Definition inputs, view availability,
|
||||
Foundation ancestry and computed digests must close. Reference-only deserialization
|
||||
is allowed for display/inspection, but input/output validation flags are derived from
|
||||
supplied bytes, are not serialized claims, and must be re-established for new
|
||||
computation. Backtest creation requires a factor whose payloads were revalidated.
|
||||
Reference decoding cannot turn an unverified factor into an admitted compute input.
|
||||
|
||||
Backtest manifest decoding rebuilds evidence from the supplied typed run and all
|
||||
nine actual artifact tables. It checks table/run/config/strategy bindings and time
|
||||
ordering. Portfolio composition revalidates those retained tables again, rather
|
||||
than trusting a serialized manifest or mutable Python context. A table digest proves
|
||||
content binding, not that those tables were produced by the claimed computation.
|
||||
|
||||
All content-addressed IDs exclude their own ID field and bind the remainder of the
|
||||
closed payload. Data/factor/backtest/manifest JSON retains the strict data profile
|
||||
(no JSON floating-point numbers; financial record decimals are strings). Performance
|
||||
and S4 preserve the existing finite numeric JSON profile: finite floats, safe ints,
|
||||
exact booleans, sorted keys, compact separators, UTF-8. Duplicate keys, NaN,
|
||||
Infinity, noncanonical JSON and extra fields are rejected. Wire revalidation uses
|
||||
type-sensitive comparisons, including `true` versus `1`. Serializers return
|
||||
detached copies; internal public maps are immutable.
|
||||
|
||||
## Replay, receipts and risk
|
||||
|
||||
Backtest v2 replay specification binds immutable input identities, selected calendar
|
||||
and actions, strategy/execution/cost versions and digests, configuration, code,
|
||||
environment lock and random seed. It excludes **both actual evaluation and actual
|
||||
computation time**. These actual times remain in each run's identity. A replay must
|
||||
retain the same replay specification, append its unique full ancestry, increment
|
||||
attempt by one, and have parent computation < new actual evaluation <= computation.
|
||||
This explicit new-major rule allows a later genuine replay without pretending its
|
||||
evaluation happened at the parent's clock time.
|
||||
|
||||
Portfolio computation-input v2 binds the full run and manifest document digests,
|
||||
new target (including both clocks), objective/model versions and digests, declared
|
||||
expected returns/covariance/scenario inputs, freshness policy and prior weights.
|
||||
The receipt separately binds that input, constraints and recomputed outputs/residuals.
|
||||
Targets and prior holdings must use selected logical instrument IDs, not ad-hoc
|
||||
symbol matches. Failed/fallback receipts and any actual constraint residual are
|
||||
rejected, even if a solver declares convergence within a permissive tolerance.
|
||||
|
||||
Risk uses the existing labelled Euler decomposition exactly once. Its result binds
|
||||
the supplied matrix content plus covariance method, bounded estimation window,
|
||||
observation count, lookback, missing policy, annualization, source dataset/input,
|
||||
model/budgets/groups and actual computation time. It checks exact labels, finite
|
||||
symmetry, covariance-source binding and both freshness clocks. Non-PSD,
|
||||
non-positive portfolio variance or non-closed contributions produce an unavailable,
|
||||
unqualified result. A budget breach is a ready but unqualified calculation result.
|
||||
`qualified=true` means only that these calculation checks passed. Every result
|
||||
remains `decision_eligible=false`, `execution_validation=not_validated`; no portfolio
|
||||
approval, maker-checker, publication, paper or live permission is granted here.
|
||||
|
||||
## Trust, ownership and test evidence
|
||||
|
||||
Pure builders accept declarations. Hashes, typed objects, model names, successful
|
||||
constraint checks and synthetic fixtures do **not** authenticate data or compute
|
||||
producers. Trusted owner-version bindings and receipt/qualification/view/clock
|
||||
admission ports remain mandatory. RP owns governance and presentation; Research
|
||||
Results owns publication. QE supplies validated calculation facts only.
|
||||
|
||||
The two data fixtures are public RP candidate vectors from
|
||||
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (PR #100). EDB producer candidate
|
||||
`88433dfc9d865ef782465498cdf9454c73920abd` (PR #13) is not a runtime dependency or
|
||||
accepted owner binding. Acceptance/review gates remain separate from local tests.
|
||||
|
||||
`tests/fixtures/retrospective-computation-v2.golden.json` freezes newly constructed
|
||||
synthetic factor/run/manifest/performance/portfolio/risk payloads and their inputs.
|
||||
Its artifact matrices are separate synthetic envelope-test inputs: the one-day
|
||||
public data fixture is **not** claimed to have produced the four-day artifact.
|
||||
The vector is not an end-to-end data/computation provenance proof or real-data run.
|
||||
Its deterministic IDs are contract-regression evidence, not admitted source facts.
|
||||
|
||||
Focused tests cover v2 goldens, mutation and strict JSON, bytes versus references,
|
||||
two-clock freshness, replay ancestry, exact table bindings, constraints, receipts,
|
||||
covariance provenance and numerical findings. Existing v1 tests must also pass.
|
||||
Rollback is disabling the explicit v2 entry path while retaining v1 and original
|
||||
immutable results; never retag old results or silently downgrade failed v2 admission.
|
||||
@@ -0,0 +1,88 @@
|
||||
# Quant Engine retrospective v2 compatibility
|
||||
|
||||
Scope: implement the user-authorized retrospective v2 compatibility without changing
|
||||
v1 semantics, financial algorithms, original results, production databases, deployment
|
||||
or trading. No claim of complete Quant OS delivery or real-data qualification.
|
||||
|
||||
Branch: `codex/research-quant-os-retrospective-contract-v2-20260908`.
|
||||
Declared base: accepted `main@68dd68392a26251391fbdae40c22eee370adb56e`.
|
||||
One isolated delivery worktree; the old primary checkout is preserved. This is not
|
||||
reactivation of an old registered stage or creation of a new stage ledger.
|
||||
|
||||
## Dependency baseline
|
||||
|
||||
Public RP data-contract candidate: PR #100, initial schemas/goldens at
|
||||
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (review/acceptance pending).
|
||||
EDB mapping candidate: PR #13, initial implementation `88433df`, local full
|
||||
validation passed. Neither candidate is silently treated as accepted owner evidence.
|
||||
The shared public major is 2.0.0; preserve the accepted v1 paths independently.
|
||||
|
||||
Order: public data contracts -> EDB mapping/Foundation -> Quant Engine typed
|
||||
factor/backtest/portfolio/risk -> RP governance -> Research Results -> RP read.
|
||||
Accepted owner-version bindings and runtime admission must still close every boundary.
|
||||
|
||||
## Internal reuse decision
|
||||
|
||||
Need: carry observation-aware inputs and retrospective-only claims through computation.
|
||||
Existing: strict canonical JSON, immutable envelopes, factor definitions, input/output
|
||||
closure, numerical algorithms, governed backtest and portfolio/risk contracts.
|
||||
External candidates: not needed; this is project-owned semantics, not a missing library.
|
||||
Approach: reuse those primitives and algorithms; introduce explicit new-major wrappers
|
||||
only where upstream identity, time or usage semantics change.
|
||||
Risk: reusing the v1 decoder or coercing observed-by into knowledge/PIT would make a
|
||||
false historical claim. Unknown versions and unsupported usages must fail closed.
|
||||
|
||||
## Implemented, not yet accepted or released
|
||||
|
||||
Five separate v2 modules now implement immutable DatasetSnapshot/Foundation decoding
|
||||
and materialized-content verification, FactorSet with explicit v2 nested bindings,
|
||||
BacktestRunRef and replay ancestry, nine-table BacktestEvidenceManifest,
|
||||
PerformanceEvidence, PortfolioTarget/Decision and RiskAssessment. The metadata
|
||||
registers the new major alongside every existing v1 entry. See
|
||||
`docs/RETROSPECTIVE_COMPUTATION_V2.md` for normative clocks, JSON profiles, input
|
||||
closure, replay and owner-port boundaries.
|
||||
|
||||
Factor definitions, generic output/receipt/constraint/covariance primitives and
|
||||
financial implementations are reused without semantic edits. Table schema remains
|
||||
1.1.0; v1 business source, v1 goldens, `pyproject.toml`, `uv.lock` and `ci-profile.yml`
|
||||
are unchanged. Package version remains unreleased. Only module metadata, its exact
|
||||
inventory test and README gain v2 alongside the new files.
|
||||
|
||||
The frozen synthetic computation vector includes fresh factor/backtest/manifest/
|
||||
performance/target/portfolio/risk documents and synthetic artifact tables. It is
|
||||
explicitly **envelope-only**, not an end-to-end claim that the one-day data fixture
|
||||
produced the four-day synthetic financial artifact. No old real run was rerun,
|
||||
retagged or backdated.
|
||||
|
||||
## Local verification (2026-09-08)
|
||||
|
||||
- Full repository unit suite: **1039 passed**, 1166 warnings, 31.50 seconds.
|
||||
- S4 focused new + unchanged v1 contracts: **120 passed**; new S4 332 statements,
|
||||
20 branches, 100% measured coverage. Coverage is not source authentication or
|
||||
proof of complete business semantics.
|
||||
- All five new source modules passed mypy; all six new test modules, five new
|
||||
sources and the updated metadata test passed Ruff.
|
||||
- The combined synthetic vector and metadata smoke checks: **3 passed**.
|
||||
- Actual negative tests reproduced and fixed missing covariance-estimation context
|
||||
in result identity, untyped malformed-JSON errors, and risk-time stale-manifest
|
||||
reuse. Other modules' earlier RED/GREEN evidence remains part of the same turn.
|
||||
|
||||
The full suite was run directly against the frozen local environment. This is not
|
||||
the same claim as remote CI or central ship acceptance; the unchanged declared CI
|
||||
profile is `lite` with the module-metadata smoke command. Central validation and
|
||||
Draft PR creation follow the implementation commit. No Ready, merge, accepted
|
||||
upstream binding or independent-review pass is claimed here.
|
||||
|
||||
Actual computation/admission times are distinct from simulated business dates. New
|
||||
formal outputs cannot inherit the old run's producer identity or be backdated to it.
|
||||
Real receipt/qualification/view/clock ports remain mandatory; typed objects and hashes
|
||||
are not source authentication. The optional independent reviewer delegation is still
|
||||
awaiting the already-requested user choice.
|
||||
|
||||
Next: preserve the candidate for review, then carry explicit v2 facts through
|
||||
RP governance -> Research Results publication -> RP read compatibility. Bind final
|
||||
accepted upstream versions only when actual acceptance evidence exists. The entire
|
||||
Quant OS goal is not complete at this intermediate owner unit.
|
||||
|
||||
Rollback: disable the explicit v2 path and retain v1 plus immutable artifacts; never
|
||||
retag v2 into v1 or silently use synthetic evidence for real admission.
|
||||
@@ -15,8 +15,9 @@ v1.2.0 Phase 0:5 个基础算子 + 5 个 alpha 公式(alpha001–alpha005)
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Callable
|
||||
from typing import Any
|
||||
from collections.abc import Callable, Mapping
|
||||
from types import MappingProxyType
|
||||
from typing import Any, cast
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
@@ -3092,6 +3093,541 @@ def parse_alpha_formula(formula_str: str) -> dict[str, Any]:
|
||||
return parsed
|
||||
|
||||
|
||||
# ── Phase 3 formula contract: frozen alpha001-alpha050 surface ──────────────
|
||||
|
||||
# Formula functions remain the implementation source of truth. This contract
|
||||
# freezes their callable surface separately from formula dependencies so that
|
||||
# historical compatibility-only arguments remain explicit without rewriting
|
||||
# formulas or changing direct-call APIs.
|
||||
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 = (
|
||||
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
|
||||
)
|
||||
|
||||
_PHASE3_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_001": alpha_001,
|
||||
"alpha_002": alpha_002,
|
||||
"alpha_003": alpha_003,
|
||||
"alpha_004": alpha_004,
|
||||
"alpha_005": alpha_005,
|
||||
"alpha_006": alpha_006,
|
||||
"alpha_007": alpha_007,
|
||||
"alpha_008": alpha_008,
|
||||
"alpha_009": alpha_009,
|
||||
"alpha_010": alpha_010,
|
||||
"alpha_011": alpha_011,
|
||||
"alpha_012": alpha_012,
|
||||
"alpha_013": alpha_013,
|
||||
"alpha_014": alpha_014,
|
||||
"alpha_015": alpha_015,
|
||||
"alpha_016": alpha_016,
|
||||
"alpha_017": alpha_017,
|
||||
"alpha_018": alpha_018,
|
||||
"alpha_019": alpha_019,
|
||||
"alpha_020": alpha_020,
|
||||
"alpha_021": alpha_021,
|
||||
"alpha_022": alpha_022,
|
||||
"alpha_023": alpha_023,
|
||||
"alpha_024": alpha_024,
|
||||
"alpha_025": alpha_025,
|
||||
"alpha_026": alpha_026,
|
||||
"alpha_027": alpha_027,
|
||||
"alpha_028": alpha_028,
|
||||
"alpha_029": alpha_029,
|
||||
"alpha_030": alpha_030,
|
||||
"alpha_031": alpha_031,
|
||||
"alpha_032": alpha_032,
|
||||
"alpha_033": alpha_033,
|
||||
"alpha_034": alpha_034,
|
||||
"alpha_035": alpha_035,
|
||||
"alpha_036": alpha_036,
|
||||
"alpha_037": alpha_037,
|
||||
"alpha_038": alpha_038,
|
||||
"alpha_039": alpha_039,
|
||||
"alpha_040": alpha_040,
|
||||
"alpha_041": alpha_041,
|
||||
"alpha_042": alpha_042,
|
||||
"alpha_043": alpha_043,
|
||||
"alpha_044": alpha_044,
|
||||
"alpha_045": alpha_045,
|
||||
"alpha_046": alpha_046,
|
||||
"alpha_047": alpha_047,
|
||||
"alpha_048": alpha_048,
|
||||
"alpha_049": alpha_049,
|
||||
"alpha_050": alpha_050,
|
||||
}
|
||||
|
||||
_PHASE3_FORMULA_INPUT_OVERRIDES: dict[str, list[str]] = {
|
||||
"alpha_011": ["close", "high", "low"],
|
||||
"alpha_035": ["volume"],
|
||||
"alpha_036": ["close"],
|
||||
"alpha_040": ["high", "low"],
|
||||
"alpha_042": ["close"],
|
||||
"alpha_043": ["volume"],
|
||||
}
|
||||
|
||||
_PHASE3_INPUT_CATEGORIES = {
|
||||
1: "single",
|
||||
2: "pair",
|
||||
3: "triple",
|
||||
4: "quadruple",
|
||||
}
|
||||
|
||||
|
||||
def _phase3_call_inputs(function: Callable[..., pd.Series]) -> list[str]:
|
||||
import inspect
|
||||
|
||||
parameters = list(inspect.signature(function).parameters.values())
|
||||
if any(
|
||||
parameter.kind is not inspect.Parameter.POSITIONAL_OR_KEYWORD
|
||||
or parameter.default is not inspect.Parameter.empty
|
||||
for parameter in parameters
|
||||
):
|
||||
raise RuntimeError(f"unsupported formula signature for {function.__name__}")
|
||||
return ["open" if parameter.name == "open_" else parameter.name for parameter in parameters]
|
||||
|
||||
|
||||
def _phase3_string_list(meta: dict[str, Any], field: str, alpha_id: str) -> list[str]:
|
||||
value = meta[field]
|
||||
if not isinstance(value, list) or not all(isinstance(item, str) for item in value):
|
||||
raise RuntimeError(f"{field} must be a list of strings for {alpha_id}")
|
||||
return list(value)
|
||||
|
||||
|
||||
def _build_phase3_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE3_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _PHASE3_FORMULA_INPUT_OVERRIDES.get(alpha_id, call_inputs)
|
||||
input_category = _PHASE3_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": list(formula_inputs),
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
def _freeze_phase3_formula_specs(
|
||||
specs: dict[str, dict[str, Any]],
|
||||
) -> Mapping[str, Mapping[str, Any]]:
|
||||
frozen_specs: dict[str, Mapping[str, Any]] = {}
|
||||
for alpha_id, spec in specs.items():
|
||||
frozen_specs[alpha_id] = MappingProxyType(
|
||||
{field: tuple(value) if isinstance(value, list) else value for field, value in spec.items()}
|
||||
)
|
||||
return MappingProxyType(frozen_specs)
|
||||
|
||||
|
||||
ALPHA158_PHASE3_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase3_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase3_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha001-alpha050 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE3_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase3_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 3 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE3_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE3_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE3_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 4 formula contract: frozen alpha051-alpha100 surface ──────────────
|
||||
|
||||
# Phase 4 extends the versioned formula contract without mutating the Phase 3
|
||||
# catalogue, digest, dispatch surface, or the existing formula functions.
|
||||
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 = (
|
||||
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
|
||||
)
|
||||
|
||||
_PHASE4_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_051": alpha_051,
|
||||
"alpha_052": alpha_052,
|
||||
"alpha_053": alpha_053,
|
||||
"alpha_054": alpha_054,
|
||||
"alpha_055": alpha_055,
|
||||
"alpha_056": alpha_056,
|
||||
"alpha_057": alpha_057,
|
||||
"alpha_058": alpha_058,
|
||||
"alpha_059": alpha_059,
|
||||
"alpha_060": alpha_060,
|
||||
"alpha_061": alpha_061,
|
||||
"alpha_062": alpha_062,
|
||||
"alpha_063": alpha_063,
|
||||
"alpha_064": alpha_064,
|
||||
"alpha_065": alpha_065,
|
||||
"alpha_066": alpha_066,
|
||||
"alpha_067": alpha_067,
|
||||
"alpha_068": alpha_068,
|
||||
"alpha_069": alpha_069,
|
||||
"alpha_070": alpha_070,
|
||||
"alpha_071": alpha_071,
|
||||
"alpha_072": alpha_072,
|
||||
"alpha_073": alpha_073,
|
||||
"alpha_074": alpha_074,
|
||||
"alpha_075": alpha_075,
|
||||
"alpha_076": alpha_076,
|
||||
"alpha_077": alpha_077,
|
||||
"alpha_078": alpha_078,
|
||||
"alpha_079": alpha_079,
|
||||
"alpha_080": alpha_080,
|
||||
"alpha_081": alpha_081,
|
||||
"alpha_082": alpha_082,
|
||||
"alpha_083": alpha_083,
|
||||
"alpha_084": alpha_084,
|
||||
"alpha_085": alpha_085,
|
||||
"alpha_086": alpha_086,
|
||||
"alpha_087": alpha_087,
|
||||
"alpha_088": alpha_088,
|
||||
"alpha_089": alpha_089,
|
||||
"alpha_090": alpha_090,
|
||||
"alpha_091": alpha_091,
|
||||
"alpha_092": alpha_092,
|
||||
"alpha_093": alpha_093,
|
||||
"alpha_094": alpha_094,
|
||||
"alpha_095": alpha_095,
|
||||
"alpha_096": alpha_096,
|
||||
"alpha_097": alpha_097,
|
||||
"alpha_098": alpha_098,
|
||||
"alpha_099": alpha_099,
|
||||
"alpha_100": alpha_100,
|
||||
}
|
||||
|
||||
_PHASE4_INPUT_CATEGORIES = {
|
||||
1: "single",
|
||||
2: "pair",
|
||||
3: "triple",
|
||||
4: "quadruple",
|
||||
5: "quintuple",
|
||||
}
|
||||
|
||||
|
||||
def _build_phase4_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE4_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE4_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase4_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase4_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha051-alpha100 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE4_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase4_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 4 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE4_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE4_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE4_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 5 formula contract: frozen alpha101-alpha150 surface ──────────────
|
||||
|
||||
# Phase 5 extends the versioned formula contract without mutating any earlier
|
||||
# catalogue, digest, dispatch surface, or existing formula implementation.
|
||||
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 = (
|
||||
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
|
||||
)
|
||||
|
||||
_PHASE5_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_101": alpha_101,
|
||||
"alpha_102": alpha_102,
|
||||
"alpha_103": alpha_103,
|
||||
"alpha_104": alpha_104,
|
||||
"alpha_105": alpha_105,
|
||||
"alpha_106": alpha_106,
|
||||
"alpha_107": alpha_107,
|
||||
"alpha_108": alpha_108,
|
||||
"alpha_109": alpha_109,
|
||||
"alpha_110": alpha_110,
|
||||
"alpha_111": alpha_111,
|
||||
"alpha_112": alpha_112,
|
||||
"alpha_113": alpha_113,
|
||||
"alpha_114": alpha_114,
|
||||
"alpha_115": alpha_115,
|
||||
"alpha_116": alpha_116,
|
||||
"alpha_117": alpha_117,
|
||||
"alpha_118": alpha_118,
|
||||
"alpha_119": alpha_119,
|
||||
"alpha_120": alpha_120,
|
||||
"alpha_121": alpha_121,
|
||||
"alpha_122": alpha_122,
|
||||
"alpha_123": alpha_123,
|
||||
"alpha_124": alpha_124,
|
||||
"alpha_125": alpha_125,
|
||||
"alpha_126": alpha_126,
|
||||
"alpha_127": alpha_127,
|
||||
"alpha_128": alpha_128,
|
||||
"alpha_129": alpha_129,
|
||||
"alpha_130": alpha_130,
|
||||
"alpha_131": alpha_131,
|
||||
"alpha_132": alpha_132,
|
||||
"alpha_133": alpha_133,
|
||||
"alpha_134": alpha_134,
|
||||
"alpha_135": alpha_135,
|
||||
"alpha_136": alpha_136,
|
||||
"alpha_137": alpha_137,
|
||||
"alpha_138": alpha_138,
|
||||
"alpha_139": alpha_139,
|
||||
"alpha_140": alpha_140,
|
||||
"alpha_141": alpha_141,
|
||||
"alpha_142": alpha_142,
|
||||
"alpha_143": alpha_143,
|
||||
"alpha_144": alpha_144,
|
||||
"alpha_145": alpha_145,
|
||||
"alpha_146": alpha_146,
|
||||
"alpha_147": alpha_147,
|
||||
"alpha_148": alpha_148,
|
||||
"alpha_149": alpha_149,
|
||||
"alpha_150": alpha_150,
|
||||
}
|
||||
|
||||
|
||||
def _build_phase5_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE5_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE5_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase5_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase5_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha101-alpha150 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE5_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase5_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 5 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE5_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE5_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if len(inputs[field]) != len(primary):
|
||||
raise ValueError(f"{field} length must match {primary_field}")
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE5_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 6 formula contract: frozen alpha151-alpha158 surface ──────────────
|
||||
|
||||
# Phase 6 completes the versioned formula contract without mutating any
|
||||
# earlier catalogue, digest, dispatch surface, or formula implementation.
|
||||
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 = (
|
||||
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
|
||||
)
|
||||
|
||||
_PHASE6_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_151": alpha_151,
|
||||
"alpha_152": alpha_152,
|
||||
"alpha_153": alpha_153,
|
||||
"alpha_154": alpha_154,
|
||||
"alpha_155": alpha_155,
|
||||
"alpha_156": alpha_156,
|
||||
"alpha_157": alpha_157,
|
||||
"alpha_158": alpha_158,
|
||||
}
|
||||
|
||||
|
||||
def _build_phase6_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE6_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE6_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase6_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase6_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha151-alpha158 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE6_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase6_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 6 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE6_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE6_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if len(inputs[field]) != len(primary):
|
||||
raise ValueError(f"{field} length must match {primary_field}")
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE6_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
__all__ = [
|
||||
"rank",
|
||||
"delta",
|
||||
@@ -3125,6 +3661,26 @@ __all__ = [
|
||||
"ALPHA158_PHASE2_OPERATOR_SPECS",
|
||||
"list_phase2_operators",
|
||||
"evaluate_phase2_operator",
|
||||
"ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE3_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE3_FORMULA_SPECS",
|
||||
"list_phase3_formulas",
|
||||
"evaluate_phase3_formula",
|
||||
"ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE4_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE4_FORMULA_SPECS",
|
||||
"list_phase4_formulas",
|
||||
"evaluate_phase4_formula",
|
||||
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE5_FORMULA_SPECS",
|
||||
"list_phase5_formulas",
|
||||
"evaluate_phase5_formula",
|
||||
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE6_FORMULA_SPECS",
|
||||
"list_phase6_formulas",
|
||||
"evaluate_phase6_formula",
|
||||
"alpha_001",
|
||||
"alpha_002",
|
||||
"alpha_003",
|
||||
|
||||
+2368
-19
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,489 @@
|
||||
"""Retrospective-only evidence wrappers over the unchanged research fact tables."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Self, cast
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from quant_engine.artifact import (
|
||||
BacktestEvidenceEntry,
|
||||
EvidenceQualification,
|
||||
ResearchRunArtifact,
|
||||
PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
PERFORMANCE_METHODOLOGY_ID,
|
||||
PerformanceMethodology,
|
||||
PerformanceMetric,
|
||||
_PERFORMANCE_SOURCE_COLUMNS,
|
||||
_absolute_performance_metrics,
|
||||
_benchmark_context,
|
||||
_count_performance_metrics,
|
||||
_performance_canonical_bytes,
|
||||
_performance_compare,
|
||||
_performance_date,
|
||||
_performance_digest,
|
||||
_performance_methodology,
|
||||
_performance_text,
|
||||
_performance_validate_tree,
|
||||
_relative_performance_metrics,
|
||||
_artifact_frames,
|
||||
_evidence_entries,
|
||||
_evidence_frame_records,
|
||||
_manifest_instant,
|
||||
_run_row,
|
||||
_table_evidence,
|
||||
_validate_table_run_ids,
|
||||
)
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
FactorContractError,
|
||||
_assert_canonical_profile,
|
||||
_content_address,
|
||||
_digest_bytes,
|
||||
_duplicate_key_pairs,
|
||||
_freeze_json,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_data_contracts import _check, _public, _shape
|
||||
|
||||
|
||||
def _validated_run(run: Any) -> RetrospectiveBacktestRunRef:
|
||||
_check(
|
||||
type(run) is RetrospectiveBacktestRunRef,
|
||||
"$.run_ref",
|
||||
"explicit v2 run reference required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
factor = run._factor_set
|
||||
return RetrospectiveBacktestRunRef.from_dict(
|
||||
run.to_dict(),
|
||||
dataset_snapshot=factor._dataset_snapshot,
|
||||
foundation=factor._foundation,
|
||||
factor_set=factor,
|
||||
parent=run._parent,
|
||||
)
|
||||
|
||||
|
||||
def _validated_frames(
|
||||
artifact: ResearchRunArtifact, run: RetrospectiveBacktestRunRef
|
||||
) -> dict[str, pd.DataFrame]:
|
||||
frames = _artifact_frames(artifact)
|
||||
_validate_table_run_ids(frames, run.run_id)
|
||||
row = _run_row(frames)
|
||||
expected = {
|
||||
"run_id": run.run_id,
|
||||
"data_snapshot_id": run.dataset_snapshot_id,
|
||||
"strategy_id": run.strategy_id,
|
||||
"strategy_version": run.strategy_version,
|
||||
"code_revision": run.code_revision,
|
||||
"config_hash": run.configuration_digest.removeprefix("sha256:"),
|
||||
"schema_version": artifact.schema_version,
|
||||
}
|
||||
_check(
|
||||
set(expected) | {"started_at", "finished_at"} <= set(row.index),
|
||||
"$.artifact.tables.run",
|
||||
"run schema fields missing",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for key, value in expected.items():
|
||||
_check(
|
||||
type(row[key]) is str and row[key] == value,
|
||||
f"$.artifact.tables.run.{key}",
|
||||
"artifact does not bind exact v2 run",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
# The unchanged artifact 1.1 timestamp profile admits offsets; public v2
|
||||
# envelope times remain strict UTC. No knowledge-time inference is performed.
|
||||
_, started = _manifest_instant(row["started_at"], "$.artifact.tables.run.started_at")
|
||||
_, finished = _manifest_instant(row["finished_at"], "$.artifact.tables.run.finished_at")
|
||||
_check(
|
||||
_parse_utc(run.evaluation_at, "$.run_ref.evaluation_at")
|
||||
<= started
|
||||
<= finished
|
||||
<= _parse_utc(run.computed_at, "$.run_ref.computed_at"),
|
||||
"$.artifact.tables.run",
|
||||
"actual evaluation <= start <= finish <= computed required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
for name, frame in frames.items():
|
||||
_public(_evidence_frame_records(frame, name), f"$.artifact.tables.{name}")
|
||||
return frames
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveBacktestEvidenceManifest:
|
||||
"""Exact artifact closure, not authenticity, historical or execution authority."""
|
||||
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
manifest_id: str
|
||||
run_id: str
|
||||
profile: str
|
||||
artifact_schema_version: str
|
||||
artifact_available_at: str
|
||||
qualification: EvidenceQualification
|
||||
evidence_digest: str
|
||||
evidence: tuple[BacktestEvidenceEntry, ...]
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
observation_cutoff: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_artifact: ResearchRunArtifact = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
artifact: ResearchRunArtifact,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
) -> Self:
|
||||
_assert_canonical_profile(value)
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version manifest_id run_id profile artifact_schema_version artifact_available_at qualification "
|
||||
"run_reference evidence_digest evidence evidence_scope usage historical_availability observation_cutoff decision_eligible execution_validation",
|
||||
)
|
||||
_check(
|
||||
type(row["qualification"]) is str
|
||||
and row["qualification"] in {"exploratory", "contract_qualified"},
|
||||
"$.qualification",
|
||||
"explicit non-legacy contract qualification required",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
rebuilt = build_retrospective_backtest_evidence_manifest(
|
||||
backtest_run_ref,
|
||||
artifact,
|
||||
artifact_available_at=row["artifact_available_at"],
|
||||
qualification=EvidenceQualification(row["qualification"]),
|
||||
)
|
||||
_check(
|
||||
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
|
||||
"$",
|
||||
"manifest differs from actual run/table closure",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_backtest_evidence_manifest(
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
artifact: ResearchRunArtifact,
|
||||
*,
|
||||
artifact_available_at: str,
|
||||
qualification: EvidenceQualification = EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
expected_table_digests: Mapping[str, str] | None = None,
|
||||
) -> RetrospectiveBacktestEvidenceManifest:
|
||||
"""Close new in-memory artifact bytes; never promote an old exploratory run."""
|
||||
run = _validated_run(backtest_run_ref)
|
||||
_check(
|
||||
type(qualification) is EvidenceQualification
|
||||
and qualification is not EvidenceQualification.LEGACY_EXPLORATORY,
|
||||
"$.qualification",
|
||||
"legacy evidence cannot enter the v2 path",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
available = _parse_utc(artifact_available_at, "$.artifact_available_at")
|
||||
_check(
|
||||
_parse_utc(run.computed_at, "$.run_ref.computed_at") <= available,
|
||||
"$.artifact_available_at",
|
||||
"artifact precedes actual computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
frames = _validated_frames(artifact, run)
|
||||
summaries = _table_evidence(frames, expected_table_digests)
|
||||
reference: dict[str, object] = {"kind": "backtest_run_ref", "value": run.to_dict()}
|
||||
# Table categories and canonical content hashing have not changed semantics.
|
||||
evidence = _evidence_entries(summaries, reference, legacy=False)
|
||||
evidence_digest = _digest_bytes(canonical_json_bytes([item.to_dict() for item in evidence]))
|
||||
payload = {
|
||||
"contract_name": "researchhub.backtest-evidence-manifest",
|
||||
"schema_version": "2.0.0",
|
||||
"run_id": run.run_id,
|
||||
"profile": "offline_research_retrospective_v2",
|
||||
"artifact_schema_version": artifact.schema_version,
|
||||
"artifact_available_at": artifact_available_at,
|
||||
"qualification": qualification.value,
|
||||
"run_reference": reference,
|
||||
"evidence_digest": evidence_digest,
|
||||
"evidence": [item.to_dict() for item in evidence],
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
payload["manifest_id"] = _content_address(
|
||||
payload, "manifest_id", "rhbacktestevidencev2:sha256:"
|
||||
)
|
||||
instance = object.__new__(RetrospectiveBacktestEvidenceManifest)
|
||||
values = {
|
||||
**payload,
|
||||
"qualification": qualification,
|
||||
"evidence": evidence,
|
||||
"backtest_run_ref": run,
|
||||
"_payload": _freeze_json(payload),
|
||||
"_artifact": artifact,
|
||||
}
|
||||
del values["run_reference"]
|
||||
for name, value in values.items():
|
||||
object.__setattr__(instance, name, value)
|
||||
return instance
|
||||
|
||||
|
||||
def _freeze_numeric_evidence(value: Any) -> Any:
|
||||
"""Freeze the existing finite-number metric profile, not the data JSON profile."""
|
||||
if type(value) is dict:
|
||||
return MappingProxyType(
|
||||
{key: _freeze_numeric_evidence(item) for key, item in value.items()}
|
||||
)
|
||||
if type(value) is list:
|
||||
return tuple(_freeze_numeric_evidence(item) for item in value)
|
||||
return value
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePerformanceEvidence:
|
||||
"""New upstream/time identity; unchanged finite-number metric/methodology v1."""
|
||||
|
||||
methodology: PerformanceMethodology
|
||||
metrics: tuple[PerformanceMetric, ...]
|
||||
_payload: Mapping[str, Any] = field(repr=False)
|
||||
|
||||
@property
|
||||
def performance_evidence_id(self) -> str:
|
||||
return cast(str, self._payload["performance_evidence_id"])
|
||||
|
||||
@property
|
||||
def document_sha256(self) -> str:
|
||||
return cast(str, self._payload["document_sha256"])
|
||||
|
||||
@property
|
||||
def run_id(self) -> str:
|
||||
return cast(str, self._payload["run_id"])
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def canonical_bytes(self) -> bytes:
|
||||
return _performance_canonical_bytes(self.to_dict())
|
||||
|
||||
def to_json(self) -> str:
|
||||
return self.canonical_bytes().decode("utf-8")
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
artifact: ResearchRunArtifact,
|
||||
run_ref: RetrospectiveBacktestRunRef,
|
||||
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
rebuilt = build_retrospective_performance_evidence(artifact, run_ref, evidence_manifest)
|
||||
_performance_compare(value, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
_check(
|
||||
type(value) in {str, bytes},
|
||||
"$",
|
||||
"canonical JSON text/bytes required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
raw = value.encode("utf-8") if isinstance(value, str) else value
|
||||
try:
|
||||
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
|
||||
except (UnicodeDecodeError, json.JSONDecodeError) as error:
|
||||
raise FactorContractError(
|
||||
ContractErrorCode.INVALID_FORMAT, "$", "invalid performance evidence JSON"
|
||||
) from error
|
||||
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
|
||||
_check(
|
||||
_performance_canonical_bytes(document) == raw,
|
||||
"$",
|
||||
"canonical finite-number JSON required",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
return cls.from_dict(document, **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_performance_evidence(
|
||||
artifact: ResearchRunArtifact,
|
||||
run_ref: RetrospectiveBacktestRunRef,
|
||||
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
) -> RetrospectivePerformanceEvidence:
|
||||
"""Bind current tables and existing methodology; no performance recalculation."""
|
||||
run = _validated_run(run_ref)
|
||||
_check(
|
||||
type(evidence_manifest) is RetrospectiveBacktestEvidenceManifest,
|
||||
"$.evidence_manifest",
|
||||
"explicit v2 manifest required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
evidence_manifest.to_dict(), artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
frames = _validated_frames(artifact, run)
|
||||
performance = frames["performance"]
|
||||
_check(
|
||||
len(performance) == 1
|
||||
and tuple(str(column) for column in performance.columns) == _PERFORMANCE_SOURCE_COLUMNS,
|
||||
"$.artifact.tables.performance",
|
||||
"one row in the unchanged closed performance schema required",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
performance_row = performance.iloc[0]
|
||||
run_row = _run_row(frames)
|
||||
frequency = _performance_text(run_row["frequency"], "$.artifact.tables.run.frequency")
|
||||
_check(
|
||||
frequency == "1d",
|
||||
"$.artifact.tables.run.frequency",
|
||||
"only existing daily methodology is supported",
|
||||
)
|
||||
calendar = _performance_text(run_row["calendar"], "$.artifact.tables.run.calendar")
|
||||
timezone = _performance_text(run_row["timezone"], "$.artifact.tables.run.timezone")
|
||||
start_date = _performance_date(run_row["start_date"], "$.artifact.tables.run.start_date")
|
||||
end_date = _performance_date(run_row["end_date"], "$.artifact.tables.run.end_date")
|
||||
nav = frames["nav"]
|
||||
_check(
|
||||
not nav.empty,
|
||||
"$.artifact.tables.nav",
|
||||
"NAV observation window required",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
_performance_date(nav.iloc[0]["trade_date"], "$.artifact.tables.nav.start") == start_date
|
||||
and _performance_date(nav.iloc[-1]["trade_date"], "$.artifact.tables.nav.end") == end_date,
|
||||
"$.artifact.tables.nav",
|
||||
"observation window differs from artifact dates",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
benchmark_digest, active_std, benchmark_variance, alpha_domain_unestimable = _benchmark_context(
|
||||
frames, run_row, performance_row
|
||||
)
|
||||
metrics = (
|
||||
*_absolute_performance_metrics(performance_row),
|
||||
*_relative_performance_metrics(
|
||||
performance_row,
|
||||
benchmark_present=benchmark_digest is not None,
|
||||
active_std=active_std,
|
||||
benchmark_variance=benchmark_variance,
|
||||
alpha_domain_unestimable=alpha_domain_unestimable,
|
||||
),
|
||||
*_count_performance_metrics(performance_row),
|
||||
)
|
||||
normalized_row: dict[str, object] = {metric.source_column: metric.value for metric in metrics}
|
||||
normalized_row["run_id"] = run.run_id
|
||||
row_digest = _performance_digest(
|
||||
{"columns": list(_PERFORMANCE_SOURCE_COLUMNS), "row": normalized_row}
|
||||
)
|
||||
alignment = cast(str, run_row["benchmark_alignment_policy"])
|
||||
methodology = _performance_methodology(
|
||||
frequency=frequency, alignment=alignment, code_revision=run.code_revision
|
||||
)
|
||||
performance_table = next(
|
||||
table
|
||||
for entry in manifest.evidence
|
||||
for table in entry.tables
|
||||
if table.logical_name == "performance"
|
||||
)
|
||||
run_document = run.to_dict()
|
||||
payload: dict[str, Any] = {
|
||||
"schema_version": "researchhub.performance-evidence.v2",
|
||||
"authority": "quant_engine",
|
||||
"scope": "offline_retrospective_research_only",
|
||||
"run_id": run.run_id,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
"backtest_run_ref_id": run.run_id,
|
||||
"backtest_run_ref_document_sha256": _digest_bytes(canonical_json_bytes(run_document)),
|
||||
"backtest_evidence_manifest_id": manifest.manifest_id,
|
||||
"backtest_evidence_manifest_document_sha256": _digest_bytes(
|
||||
canonical_json_bytes(manifest.to_dict())
|
||||
),
|
||||
"backtest_evidence_manifest_evidence_digest": manifest.evidence_digest,
|
||||
"backtest_evidence_qualification": manifest.qualification.value,
|
||||
"research_artifact_schema_version": artifact.schema_version,
|
||||
"research_artifact_content_digest": "sha256:" + artifact.content_sha256,
|
||||
"artifact_available_at": manifest.artifact_available_at,
|
||||
"computed_at": run.computed_at,
|
||||
"performance_table_logical_name": performance_table.logical_name,
|
||||
"performance_table_row_count": performance_table.row_count,
|
||||
"performance_table_schema_digest": performance_table.schema_digest,
|
||||
"performance_table_content_digest": performance_table.content_digest,
|
||||
"performance_row_digest": row_digest,
|
||||
"benchmark_series_digest": benchmark_digest,
|
||||
"methodology_id": PERFORMANCE_METHODOLOGY_ID,
|
||||
"metric_schema_id": PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
**{
|
||||
key: run_document[key]
|
||||
for key in (
|
||||
"dataset_snapshot_id",
|
||||
"dataset_content_digest",
|
||||
"dataset_manifest_digest",
|
||||
"foundation_id",
|
||||
"foundation_digest",
|
||||
"factor_set_id",
|
||||
"factor_set_digest",
|
||||
"factor_output_content_digest",
|
||||
"strategy_id",
|
||||
"strategy_version",
|
||||
"strategy_digest",
|
||||
"execution_model_version",
|
||||
"execution_model_digest",
|
||||
"cost_model_version",
|
||||
"cost_model_digest",
|
||||
"code_revision",
|
||||
"environment_lock_digest",
|
||||
"configuration_digest",
|
||||
)
|
||||
},
|
||||
"frequency": frequency,
|
||||
"calendar": calendar,
|
||||
"timezone": timezone,
|
||||
"benchmark_id": run_row["benchmark_id"],
|
||||
"benchmark_alignment_policy": alignment,
|
||||
"start_date": start_date,
|
||||
"end_date": end_date,
|
||||
"methodology": methodology.to_dict(),
|
||||
"metrics": [metric.to_dict() for metric in metrics],
|
||||
}
|
||||
payload["performance_evidence_id"] = "rhperformancev2:" + _performance_digest(payload)
|
||||
payload["document_sha256"] = _performance_digest(payload)
|
||||
instance = object.__new__(RetrospectivePerformanceEvidence)
|
||||
object.__setattr__(instance, "_payload", _freeze_numeric_evidence(payload))
|
||||
object.__setattr__(instance, "methodology", methodology)
|
||||
object.__setattr__(instance, "metrics", metrics)
|
||||
return instance
|
||||
@@ -0,0 +1,422 @@
|
||||
"""Explicit retrospective v2 run identities and offline artifact evidence."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Self, cast
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
PayloadValidation,
|
||||
_assert_canonical_profile,
|
||||
_content_address,
|
||||
_digest,
|
||||
_digest_bytes,
|
||||
_freeze_json,
|
||||
_git_revision,
|
||||
_logical_id,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_safe_integer,
|
||||
_semver,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
_IDS,
|
||||
_check,
|
||||
_public,
|
||||
_shape,
|
||||
_strings,
|
||||
)
|
||||
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef, _context
|
||||
|
||||
_CONFIG_FIELDS = (
|
||||
"universe_digest",
|
||||
"strategy_id",
|
||||
"strategy_version",
|
||||
"strategy_digest",
|
||||
"execution_model_version",
|
||||
"execution_model_digest",
|
||||
"cost_model_version",
|
||||
"cost_model_digest",
|
||||
"random_seed",
|
||||
"code_revision",
|
||||
"environment_lock_digest",
|
||||
"configuration_digest",
|
||||
)
|
||||
_RUN_FIELDS = (
|
||||
"contract_name schema_version run_id dataset_snapshot_id dataset_content_digest dataset_manifest_digest "
|
||||
"foundation_id foundation_digest factor_set_id factor_set_digest factor_output_content_digest "
|
||||
"observation_cutoff evidence_scope usage historical_availability decision_eligible execution_validation "
|
||||
"universe_digest trading_calendar_revision_ids trading_calendar_digest corporate_action_revision_ids corporate_action_digest "
|
||||
"strategy_id strategy_version strategy_digest execution_model_version execution_model_digest cost_model_version cost_model_digest "
|
||||
"random_seed code_revision environment_lock_digest configuration_digest evaluation_at computed_at replay_spec_digest "
|
||||
"replay_parent_run_id replay_reason replay_attempt replay_ancestor_run_ids"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveBacktestRunRef:
|
||||
"""New-major deterministic-input identity with separate actual attempt times."""
|
||||
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
run_id: str
|
||||
dataset_snapshot_id: str
|
||||
dataset_content_digest: str
|
||||
dataset_manifest_digest: str
|
||||
foundation_id: str
|
||||
foundation_digest: str
|
||||
factor_set_id: str
|
||||
factor_set_digest: str
|
||||
factor_output_content_digest: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
universe_digest: str
|
||||
trading_calendar_revision_ids: tuple[str, ...]
|
||||
trading_calendar_digest: str
|
||||
corporate_action_revision_ids: tuple[str, ...]
|
||||
corporate_action_digest: str
|
||||
strategy_id: str
|
||||
strategy_version: str
|
||||
strategy_digest: str
|
||||
execution_model_version: str
|
||||
execution_model_digest: str
|
||||
cost_model_version: str
|
||||
cost_model_digest: str
|
||||
random_seed: int
|
||||
code_revision: str
|
||||
environment_lock_digest: str
|
||||
configuration_digest: str
|
||||
evaluation_at: str
|
||||
computed_at: str
|
||||
replay_spec_digest: str
|
||||
replay_parent_run_id: str | None
|
||||
replay_reason: str | None
|
||||
replay_attempt: int
|
||||
replay_ancestor_run_ids: tuple[str, ...]
|
||||
input_payload_validation: PayloadValidation = field(compare=False)
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_factor_set: RetrospectiveFactorSetRef = field(repr=False, compare=False)
|
||||
_parent: RetrospectiveBacktestRunRef | None = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
factor_set: RetrospectiveFactorSetRef,
|
||||
universe_digest: str,
|
||||
trading_calendar_revision_ids: Sequence[str],
|
||||
corporate_action_revision_ids: Sequence[str],
|
||||
strategy_id: str,
|
||||
strategy_version: str,
|
||||
strategy_digest: str,
|
||||
execution_model_version: str,
|
||||
execution_model_digest: str,
|
||||
cost_model_version: str,
|
||||
cost_model_digest: str,
|
||||
random_seed: int,
|
||||
code_revision: str,
|
||||
environment_lock_digest: str,
|
||||
configuration_digest: str,
|
||||
evaluation_at: str,
|
||||
computed_at: str,
|
||||
parent: RetrospectiveBacktestRunRef | None = None,
|
||||
replay_reason: str | None = None,
|
||||
replay_attempt: int = 0,
|
||||
) -> Self:
|
||||
_check(
|
||||
type(factor_set) is RetrospectiveFactorSetRef,
|
||||
"$.factor_set",
|
||||
"explicit v2 factor result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
factor_set.require_payloads_revalidated()
|
||||
return cls._build(
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
trading_calendar_revision_ids=trading_calendar_revision_ids,
|
||||
corporate_action_revision_ids=corporate_action_revision_ids,
|
||||
configuration={
|
||||
"universe_digest": universe_digest,
|
||||
"strategy_id": strategy_id,
|
||||
"strategy_version": strategy_version,
|
||||
"strategy_digest": strategy_digest,
|
||||
"execution_model_version": execution_model_version,
|
||||
"execution_model_digest": execution_model_digest,
|
||||
"cost_model_version": cost_model_version,
|
||||
"cost_model_digest": cost_model_digest,
|
||||
"random_seed": random_seed,
|
||||
"code_revision": code_revision,
|
||||
"environment_lock_digest": environment_lock_digest,
|
||||
"configuration_digest": configuration_digest,
|
||||
},
|
||||
evaluation_at=evaluation_at,
|
||||
computed_at=computed_at,
|
||||
parent=parent,
|
||||
replay_reason=replay_reason,
|
||||
replay_attempt=replay_attempt,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def _build(
|
||||
cls,
|
||||
*,
|
||||
dataset_snapshot: Any,
|
||||
foundation: Any,
|
||||
factor_set: Any,
|
||||
trading_calendar_revision_ids: Any,
|
||||
corporate_action_revision_ids: Any,
|
||||
configuration: dict[str, Any],
|
||||
evaluation_at: Any,
|
||||
computed_at: Any,
|
||||
parent: RetrospectiveBacktestRunRef | None,
|
||||
replay_reason: Any,
|
||||
replay_attempt: Any,
|
||||
) -> Self:
|
||||
_check(
|
||||
type(factor_set) is RetrospectiveFactorSetRef,
|
||||
"$.factor_set",
|
||||
"explicit v2 factor result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
definitions, snapshot, foundation = _context(
|
||||
factor_set._definitions, dataset_snapshot, foundation
|
||||
)
|
||||
# Reconstruct the serialized factor boundary against the exact supplied inputs.
|
||||
checked_factor = RetrospectiveFactorSetRef.from_dict(
|
||||
factor_set.to_dict(),
|
||||
definitions=definitions,
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
parent=factor_set._parent,
|
||||
)
|
||||
closures: dict[str, tuple[str, ...]] = {}
|
||||
for field_name, supplied, kind in (
|
||||
("trading_calendar_revision_ids", trading_calendar_revision_ids, "calendar_revision"),
|
||||
("corporate_action_revision_ids", corporate_action_revision_ids, "action_revision"),
|
||||
):
|
||||
_check(
|
||||
type(supplied) in {tuple, list},
|
||||
f"$.{field_name}",
|
||||
"list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
supplied_ids = tuple(
|
||||
sorted(
|
||||
_strings(
|
||||
list(supplied),
|
||||
f"$.{field_name}",
|
||||
_IDS[kind],
|
||||
1 if kind == "calendar_revision" else 0,
|
||||
)
|
||||
)
|
||||
)
|
||||
expected_ids = tuple(
|
||||
sorted(
|
||||
{
|
||||
identity
|
||||
for view_id in checked_factor.selected_view_ref_ids
|
||||
for identity in getattr(foundation.views[view_id], field_name)
|
||||
}
|
||||
)
|
||||
)
|
||||
_check(
|
||||
supplied_ids == expected_ids,
|
||||
f"$.{field_name}",
|
||||
"exact selected observation ancestry required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
closures[field_name] = supplied_ids
|
||||
_shape(configuration, "$.configuration", " ".join(_CONFIG_FIELDS))
|
||||
for name, value in configuration.items():
|
||||
if name.endswith("_digest"):
|
||||
_digest(value, f"$.{name}")
|
||||
elif name.endswith("_version"):
|
||||
_semver(value, f"$.{name}")
|
||||
elif name == "random_seed":
|
||||
_safe_integer(value, f"$.{name}", minimum=0)
|
||||
elif name == "code_revision":
|
||||
_git_revision(value, f"$.{name}")
|
||||
else:
|
||||
_logical_id(value, f"$.{name}")
|
||||
_public(configuration, "$.configuration")
|
||||
evaluation = _parse_utc(evaluation_at, "$.evaluation_at")
|
||||
computed = _parse_utc(computed_at, "$.computed_at")
|
||||
_check(
|
||||
_parse_utc(checked_factor.artifact_available_at, "$.factor_set.artifact_available_at")
|
||||
<= evaluation
|
||||
<= computed,
|
||||
"$.computed_at",
|
||||
"factor availability <= actual evaluation <= computation required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
content = snapshot.to_dict()["descriptor"]["content"]
|
||||
spec: dict[str, Any] = {
|
||||
"dataset_snapshot_id": snapshot.snapshot_id,
|
||||
"dataset_content_digest": content["content_digest"],
|
||||
"dataset_manifest_digest": content["manifest_digest"],
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"foundation_digest": foundation.foundation_id.removeprefix("rhdfv2:"),
|
||||
"factor_set_id": checked_factor.factor_set_id,
|
||||
"factor_set_digest": checked_factor.factor_set_id.removeprefix("rhfactorsetv2:"),
|
||||
"factor_output_content_digest": checked_factor.output_content_digest,
|
||||
"observation_cutoff": foundation.observation_cutoff,
|
||||
"evidence_scope": checked_factor.evidence_scope,
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
**configuration,
|
||||
}
|
||||
for field_name, identities in closures.items():
|
||||
spec[field_name] = list(identities)
|
||||
digest_field = (
|
||||
"trading_calendar_digest"
|
||||
if field_name == "trading_calendar_revision_ids"
|
||||
else "corporate_action_digest"
|
||||
)
|
||||
spec[digest_field] = _digest_bytes(canonical_json_bytes(list(identities)))
|
||||
# v2 replay specification excludes BOTH actual attempt times. They remain in
|
||||
# run_id, so a replay never backdates evaluation to manufacture equality.
|
||||
replay_spec_digest = _digest_bytes(canonical_json_bytes(spec))
|
||||
replay_count = _safe_integer(replay_attempt, "$.replay_attempt", minimum=0)
|
||||
if parent is None:
|
||||
_check(
|
||||
replay_reason is None and replay_count == 0,
|
||||
"$.replay_attempt",
|
||||
"root must use zero attempt and no reason",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
parent_id = None
|
||||
ancestors: tuple[str, ...] = ()
|
||||
else:
|
||||
_check(
|
||||
type(parent) is RetrospectiveBacktestRunRef,
|
||||
"$.parent",
|
||||
"exact v2 run parent required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_logical_id(replay_reason, "$.replay_reason")
|
||||
_check(
|
||||
replay_count == parent.replay_attempt + 1
|
||||
and replay_spec_digest == parent.replay_spec_digest,
|
||||
"$.replay_spec_digest",
|
||||
"replay requires unchanged inputs and the next attempt",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(parent.computed_at, "$.parent.computed_at") < evaluation <= computed,
|
||||
"$.evaluation_at",
|
||||
"new actual attempt must follow parent computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
parent_id = parent.run_id
|
||||
ancestors = (*parent.replay_ancestor_run_ids, parent_id)
|
||||
_check(
|
||||
len(ancestors) == len(set(ancestors)),
|
||||
"$.replay_ancestor_run_ids",
|
||||
"replay cycle",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.backtest-run-ref",
|
||||
"schema_version": "2.0.0",
|
||||
**spec,
|
||||
"evaluation_at": evaluation_at,
|
||||
"computed_at": computed_at,
|
||||
"replay_spec_digest": replay_spec_digest,
|
||||
"replay_parent_run_id": parent_id,
|
||||
"replay_reason": replay_reason,
|
||||
"replay_attempt": replay_count,
|
||||
"replay_ancestor_run_ids": list(ancestors),
|
||||
}
|
||||
payload["run_id"] = _content_address(payload, "run_id", "rhbacktestrunv2:sha256:")
|
||||
_check(
|
||||
payload["run_id"] not in ancestors,
|
||||
"$.run_id",
|
||||
"self-parent cycle",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
verified = (
|
||||
factor_set.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
and factor_set.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
)
|
||||
instance = object.__new__(cls)
|
||||
for name, value in {
|
||||
**payload,
|
||||
**closures,
|
||||
"replay_ancestor_run_ids": ancestors,
|
||||
"input_payload_validation": PayloadValidation.PAYLOAD_REVALIDATED
|
||||
if verified
|
||||
else PayloadValidation.REFERENCE_ONLY,
|
||||
"_payload": _freeze_json(payload),
|
||||
"_factor_set": factor_set,
|
||||
"_parent": parent,
|
||||
}.items():
|
||||
object.__setattr__(instance, name, value)
|
||||
return instance
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
def require_inputs_revalidated(self) -> None:
|
||||
_check(
|
||||
self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
|
||||
"$.input_payload_validation",
|
||||
"reference-only factors cannot admit a new computation",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
factor_set: RetrospectiveFactorSetRef,
|
||||
parent: RetrospectiveBacktestRunRef | None = None,
|
||||
) -> Self:
|
||||
_assert_canonical_profile(value)
|
||||
_public(value)
|
||||
row = _shape(value, "$", _RUN_FIELDS)
|
||||
rebuilt = cls._build(
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
trading_calendar_revision_ids=row["trading_calendar_revision_ids"],
|
||||
corporate_action_revision_ids=row["corporate_action_revision_ids"],
|
||||
configuration={key: row[key] for key in _CONFIG_FIELDS},
|
||||
evaluation_at=row["evaluation_at"],
|
||||
computed_at=row["computed_at"],
|
||||
parent=parent,
|
||||
replay_reason=row["replay_reason"],
|
||||
replay_attempt=row["replay_attempt"],
|
||||
)
|
||||
_check(
|
||||
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
|
||||
"$",
|
||||
"serialized run differs from exact v2 input/configuration/lineage closure",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
return rebuilt
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,721 @@
|
||||
"""Observation-aware factor results; no historical, governance or execution grant."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Self, cast
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
ContractErrorCode,
|
||||
FactorDefinition,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
_DEFINITION_ID,
|
||||
_FIELD_NAME,
|
||||
_array,
|
||||
_assert_canonical_profile,
|
||||
_canonical_evidence_bytes,
|
||||
_content_address,
|
||||
_digest,
|
||||
_digest_bytes,
|
||||
_freeze_json,
|
||||
_git_revision,
|
||||
_logical_id,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_string,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
validate_factor_catalog,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
_IDS,
|
||||
_check,
|
||||
_choice,
|
||||
_public,
|
||||
_restrictions,
|
||||
_shape,
|
||||
_strings,
|
||||
)
|
||||
|
||||
_FACTOR_SET_ID = re.compile(r"^rhfactorsetv2:sha256:[0-9a-f]{64}$")
|
||||
|
||||
|
||||
def _instant_text(value: Any, path: str) -> str:
|
||||
_parse_utc(value, path)
|
||||
return cast(str, value)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveInputBinding:
|
||||
definition_id: str
|
||||
input_name: str
|
||||
view_ref_id: str
|
||||
schema_digest: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.definition_id, "$.input_bindings[].definition_id", _DEFINITION_ID)
|
||||
_string(self.input_name, "$.input_bindings[].input_name", _FIELD_NAME)
|
||||
_string(self.view_ref_id, "$.input_bindings[].view_ref_id", _IDS["view_ref"])
|
||||
_digest(self.schema_digest, "$.input_bindings[].schema_digest")
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"definition_id": self.definition_id,
|
||||
"input_name": self.input_name,
|
||||
"view_ref_id": self.view_ref_id,
|
||||
"schema_digest": self.schema_digest,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any, path: str = "$.input_bindings[]") -> Self:
|
||||
row = _shape(value, path, "definition_id input_name view_ref_id schema_digest")
|
||||
return cls(**row)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveViewAvailability:
|
||||
view_ref_id: str
|
||||
available_at: str
|
||||
evidence_digest: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.view_ref_id, "$.view_availability[].view_ref_id", _IDS["view_ref"])
|
||||
_instant_text(self.available_at, "$.view_availability[].available_at")
|
||||
_digest(self.evidence_digest, "$.view_availability[].evidence_digest")
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"view_ref_id": self.view_ref_id,
|
||||
"available_at": self.available_at,
|
||||
"evidence_digest": self.evidence_digest,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any, path: str = "$.view_availability[]") -> Self:
|
||||
return cls(**_shape(value, path, "view_ref_id available_at evidence_digest"))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveCausation:
|
||||
kind: str
|
||||
id: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
kind = _choice(self.kind, "$.causation.kind", {"foundation", "factor_set"})
|
||||
_string(
|
||||
self.id,
|
||||
"$.causation.id",
|
||||
_IDS["foundation"] if kind == "foundation" else _FACTOR_SET_ID,
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {"kind": self.kind, "id": self.id}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any) -> Self:
|
||||
return cls(**_shape(value, "$.causation", "kind id"))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ResolvedRetrospectiveView:
|
||||
"""In-memory logical bytes; no locator, source authentication or transformation claim."""
|
||||
|
||||
view_ref_id: str
|
||||
schema_bytes: bytes
|
||||
content_bytes: bytes
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.view_ref_id, "$.resolved_views[].view_ref_id", _IDS["view_ref"])
|
||||
for key, value in (
|
||||
("schema_bytes", self.schema_bytes),
|
||||
("content_bytes", self.content_bytes),
|
||||
):
|
||||
_canonical_evidence_bytes(value, f"$.resolved_views[].{key}")
|
||||
_public(json.loads(value), f"$.resolved_views[].{key}")
|
||||
|
||||
|
||||
def _typed(values: Any, expected: type[Any], path: str) -> tuple[Any, ...]:
|
||||
_check(
|
||||
type(values) in {tuple, list},
|
||||
path,
|
||||
"typed list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
all(type(value) is expected for value in values),
|
||||
path,
|
||||
f"{expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
return tuple(values)
|
||||
|
||||
|
||||
def _context(
|
||||
definitions: Sequence[FactorDefinition],
|
||||
snapshot: Any,
|
||||
foundation: Any,
|
||||
) -> tuple[
|
||||
tuple[FactorDefinition, ...], RetrospectiveSnapshotEnvelope, RetrospectiveFoundationEnvelope
|
||||
]:
|
||||
_check(
|
||||
type(snapshot) is RetrospectiveSnapshotEnvelope,
|
||||
"$.dataset_snapshot",
|
||||
"explicit v2 snapshot required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
type(foundation) is RetrospectiveFoundationEnvelope,
|
||||
"$.foundation",
|
||||
"explicit v2 foundation required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot.to_dict())
|
||||
snapshot.require_qualified()
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation.to_dict(), snapshot=snapshot)
|
||||
supplied = _typed(definitions, FactorDefinition, "$.definitions")
|
||||
# Definitions stay v1, but are parsed again so mutable/caller summaries are not authority.
|
||||
normalized = validate_factor_catalog(
|
||||
tuple(FactorDefinition.from_dict(item.to_dict()) for item in supplied)
|
||||
)
|
||||
return normalized, snapshot, foundation
|
||||
|
||||
|
||||
def _upstream(
|
||||
snapshot: RetrospectiveSnapshotEnvelope, foundation: RetrospectiveFoundationEnvelope
|
||||
) -> dict[str, Any]:
|
||||
descriptor = snapshot.to_dict()["descriptor"]
|
||||
return {
|
||||
"dataset_snapshot_id": snapshot.snapshot_id,
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"evidence_scope": snapshot.evidence_scope,
|
||||
"content_digest": descriptor["content"]["content_digest"],
|
||||
"manifest_digest": descriptor["content"]["manifest_digest"],
|
||||
"observation_manifest_digest": _digest_bytes(
|
||||
canonical_json_bytes(descriptor["observation_manifest"])
|
||||
),
|
||||
"time_semantics": descriptor["time_semantics"],
|
||||
"quality": descriptor["quality"],
|
||||
"qualification": descriptor["qualification"],
|
||||
"foundation_readiness": foundation.to_dict()["readiness"],
|
||||
}
|
||||
|
||||
|
||||
def _input_payloads(
|
||||
snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
selected: tuple[str, ...],
|
||||
dataset_chunks: Any,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView] | None,
|
||||
) -> PayloadValidation:
|
||||
_check(
|
||||
(dataset_chunks is None) == (resolved_views is None),
|
||||
"$.input_payloads",
|
||||
"snapshot chunks and resolved views must be supplied together",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
if dataset_chunks is None:
|
||||
return PayloadValidation.REFERENCE_ONLY
|
||||
snapshot.verify_materialized_records(dataset_chunks)
|
||||
views = _typed(resolved_views, ResolvedRetrospectiveView, "$.resolved_views")
|
||||
view_ids = [view.view_ref_id for view in views]
|
||||
_check(
|
||||
len(view_ids) == len(selected) and set(view_ids) == set(selected),
|
||||
"$.resolved_views",
|
||||
"resolved view closure mismatch",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for item in views:
|
||||
declared = foundation.views[item.view_ref_id]
|
||||
# Recheck canonical bytes even for caller-constructed typed payloads.
|
||||
schema = _canonical_evidence_bytes(item.schema_bytes, "$.resolved_views[].schema_bytes")
|
||||
content = _canonical_evidence_bytes(item.content_bytes, "$.resolved_views[].content_bytes")
|
||||
_public(json.loads(schema))
|
||||
_public(json.loads(content))
|
||||
_check(
|
||||
_digest_bytes(schema) == declared.schema_digest
|
||||
and _digest_bytes(content) == declared.content_digest,
|
||||
"$.resolved_views",
|
||||
"view bytes do not match Foundation",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
return PayloadValidation.PAYLOAD_REVALIDATED
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveFactorSetRef:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
factor_set_id: str
|
||||
definition_ids: tuple[str, ...]
|
||||
dataset_snapshot_id: str
|
||||
foundation_id: str
|
||||
observation_cutoff: str
|
||||
selected_view_ref_ids: tuple[str, ...]
|
||||
input_bindings: tuple[RetrospectiveInputBinding, ...]
|
||||
view_availability: tuple[RetrospectiveViewAvailability, ...]
|
||||
upstream_evidence: Mapping[str, Any]
|
||||
output_quality: OutputQuality
|
||||
output_coverage: OutputCoverage
|
||||
output_schema_digest: str
|
||||
output_content_digest: str
|
||||
output_artifact_ref: OutputArtifactRef
|
||||
availability_mode: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
evaluation_at: str
|
||||
computed_at: str
|
||||
artifact_available_at: str
|
||||
producer: ProducerIdentity
|
||||
code_revision: str
|
||||
actor: ActorIdentity
|
||||
correlation_id: str
|
||||
causation: RetrospectiveCausation
|
||||
evidence_scope: str
|
||||
decision_eligible: bool
|
||||
payload_validation: PayloadValidation = field(compare=False)
|
||||
input_payload_validation: PayloadValidation = field(compare=False)
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_definitions: tuple[FactorDefinition, ...] = field(repr=False, compare=False)
|
||||
_dataset_snapshot: RetrospectiveSnapshotEnvelope = field(repr=False, compare=False)
|
||||
_foundation: RetrospectiveFoundationEnvelope = field(repr=False, compare=False)
|
||||
_parent: RetrospectiveFactorSetRef | None = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
definitions: Sequence[FactorDefinition],
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
selected_view_ref_ids: Sequence[str],
|
||||
input_bindings: Sequence[RetrospectiveInputBinding],
|
||||
view_availability: Sequence[RetrospectiveViewAvailability],
|
||||
dataset_chunks: Any,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView],
|
||||
output_quality: OutputQuality,
|
||||
output_coverage: OutputCoverage,
|
||||
output_schema_bytes: bytes,
|
||||
output_content_bytes: bytes,
|
||||
output_artifact_ref: OutputArtifactRef,
|
||||
evaluation_at: str,
|
||||
computed_at: str,
|
||||
artifact_available_at: str,
|
||||
producer: ProducerIdentity,
|
||||
code_revision: str,
|
||||
actor: ActorIdentity,
|
||||
correlation_id: str,
|
||||
causation: RetrospectiveCausation,
|
||||
evidence_scope: str,
|
||||
decision_eligible: bool,
|
||||
parent: RetrospectiveFactorSetRef | None = None,
|
||||
) -> Self:
|
||||
definitions, dataset_snapshot, foundation = _context(
|
||||
definitions, dataset_snapshot, foundation
|
||||
)
|
||||
_check(
|
||||
type(selected_view_ref_ids) in {list, tuple},
|
||||
"$.selected_view_ref_ids",
|
||||
"list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
selected = sorted(
|
||||
_strings(list(selected_view_ref_ids), "$.selected_view_ref_ids", _IDS["view_ref"], 1)
|
||||
)
|
||||
bindings = sorted(
|
||||
_typed(input_bindings, RetrospectiveInputBinding, "$.input_bindings"),
|
||||
key=lambda item: (item.definition_id, item.input_name),
|
||||
)
|
||||
availability = sorted(
|
||||
_typed(view_availability, RetrospectiveViewAvailability, "$.view_availability"),
|
||||
key=lambda item: item.view_ref_id,
|
||||
)
|
||||
for value, expected, path in (
|
||||
(output_quality, OutputQuality, "$.output_quality"),
|
||||
(output_coverage, OutputCoverage, "$.output_coverage"),
|
||||
(output_artifact_ref, OutputArtifactRef, "$.output_artifact_ref"),
|
||||
(producer, ProducerIdentity, "$.producer"),
|
||||
(actor, ActorIdentity, "$.actor"),
|
||||
(causation, RetrospectiveCausation, "$.causation"),
|
||||
):
|
||||
_check(
|
||||
type(value) is expected,
|
||||
path,
|
||||
f"{expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
schema_digest = _digest_bytes(
|
||||
_canonical_evidence_bytes(output_schema_bytes, "$.output_schema_bytes")
|
||||
)
|
||||
content_digest = _digest_bytes(
|
||||
_canonical_evidence_bytes(output_content_bytes, "$.output_content_bytes")
|
||||
)
|
||||
document = {
|
||||
"contract_name": "researchhub.factor-set-ref",
|
||||
"schema_version": "2.0.0",
|
||||
"definition_ids": [definition.definition_id for definition in definitions],
|
||||
"dataset_snapshot_id": dataset_snapshot.snapshot_id,
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"observation_cutoff": foundation.observation_cutoff,
|
||||
"selected_view_ref_ids": selected,
|
||||
"input_bindings": [item.to_dict() for item in bindings],
|
||||
"view_availability": [item.to_dict() for item in availability],
|
||||
"upstream_evidence": _upstream(dataset_snapshot, foundation),
|
||||
"output_quality": output_quality.to_dict(),
|
||||
"output_coverage": output_coverage.to_dict(),
|
||||
"output_schema_digest": schema_digest,
|
||||
"output_content_digest": content_digest,
|
||||
"output_artifact_ref": output_artifact_ref.to_dict(),
|
||||
"availability_mode": "retrospective_replay",
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"evaluation_at": evaluation_at,
|
||||
"computed_at": computed_at,
|
||||
"artifact_available_at": artifact_available_at,
|
||||
"producer": producer.to_dict(),
|
||||
"code_revision": code_revision,
|
||||
"actor": actor.to_dict(),
|
||||
"correlation_id": correlation_id,
|
||||
"causation": causation.to_dict(),
|
||||
"evidence_scope": evidence_scope,
|
||||
"decision_eligible": decision_eligible,
|
||||
}
|
||||
document["factor_set_id"] = _content_address(
|
||||
document, "factor_set_id", "rhfactorsetv2:sha256:"
|
||||
)
|
||||
result = cls.from_dict(
|
||||
document,
|
||||
definitions=definitions,
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
parent=parent,
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
dataset_chunks=dataset_chunks,
|
||||
resolved_views=resolved_views,
|
||||
)
|
||||
result.require_payloads_revalidated()
|
||||
return result
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
definitions: Sequence[FactorDefinition],
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
parent: RetrospectiveFactorSetRef | None = None,
|
||||
output_schema_bytes: bytes | None = None,
|
||||
output_content_bytes: bytes | None = None,
|
||||
dataset_chunks: Any = None,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView] | None = None,
|
||||
) -> Self:
|
||||
definitions, dataset_snapshot, foundation = _context(
|
||||
definitions, dataset_snapshot, foundation
|
||||
)
|
||||
_assert_canonical_profile(value)
|
||||
_public(value)
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version factor_set_id definition_ids dataset_snapshot_id foundation_id observation_cutoff "
|
||||
"selected_view_ref_ids input_bindings view_availability upstream_evidence output_quality output_coverage "
|
||||
"output_schema_digest output_content_digest output_artifact_ref availability_mode usage historical_availability "
|
||||
"evaluation_at computed_at artifact_available_at producer code_revision actor correlation_id causation evidence_scope decision_eligible",
|
||||
)
|
||||
_choice(row["contract_name"], "$.contract_name", {"researchhub.factor-set-ref"})
|
||||
_choice(row["schema_version"], "$.schema_version", {"2.0.0"})
|
||||
_choice(row["availability_mode"], "$.availability_mode", {"retrospective_replay"})
|
||||
_restrictions(row, "$")
|
||||
_check(
|
||||
type(row["decision_eligible"]) is bool and not row["decision_eligible"],
|
||||
"$.decision_eligible",
|
||||
"computation is never decision eligible",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
_check(
|
||||
row["dataset_snapshot_id"] == dataset_snapshot.snapshot_id
|
||||
and row["foundation_id"] == foundation.foundation_id
|
||||
and row["observation_cutoff"] == foundation.observation_cutoff,
|
||||
"$.foundation_id",
|
||||
"exact snapshot/foundation/cutoff required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
definition_ids = _strings(row["definition_ids"], "$.definition_ids", _DEFINITION_ID, 1)
|
||||
_check(
|
||||
definition_ids == tuple(item.definition_id for item in definitions),
|
||||
"$.definition_ids",
|
||||
"normalized exact definitions required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
selected = _strings(
|
||||
row["selected_view_ref_ids"], "$.selected_view_ref_ids", _IDS["view_ref"], 1
|
||||
)
|
||||
_check(
|
||||
tuple(sorted(selected)) == selected and set(selected) <= foundation.views.keys(),
|
||||
"$.selected_view_ref_ids",
|
||||
"unknown/unnormalized selected views",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
bindings = tuple(
|
||||
RetrospectiveInputBinding.from_dict(item)
|
||||
for item in _array(row["input_bindings"], "$.input_bindings", minimum=1, unique=True)
|
||||
)
|
||||
keys = [(item.definition_id, item.input_name) for item in bindings]
|
||||
expected = {
|
||||
(item.definition_id, input_spec.input_name): input_spec
|
||||
for item in definitions
|
||||
for input_spec in item.inputs
|
||||
}
|
||||
_check(
|
||||
len(keys) == len(expected) and set(keys) == expected.keys() and keys == sorted(keys),
|
||||
"$.input_bindings",
|
||||
"exact normalized factor input closure required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for binding in bindings:
|
||||
_check(
|
||||
binding.view_ref_id in selected,
|
||||
"$.input_bindings",
|
||||
"unselected view",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
binding.schema_digest
|
||||
== expected[(binding.definition_id, binding.input_name)].schema_digest
|
||||
== foundation.views[binding.view_ref_id].schema_digest,
|
||||
"$.input_bindings",
|
||||
"schema mismatch",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
{item.view_ref_id for item in bindings} == set(selected),
|
||||
"$.selected_view_ref_ids",
|
||||
"unused selected view",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
availability = tuple(
|
||||
RetrospectiveViewAvailability.from_dict(item)
|
||||
for item in _array(
|
||||
row["view_availability"], "$.view_availability", minimum=1, unique=True
|
||||
)
|
||||
)
|
||||
_check(
|
||||
tuple(item.view_ref_id for item in availability) == selected,
|
||||
"$.view_availability",
|
||||
"exact normalized selected view availability required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for item in availability:
|
||||
_check(
|
||||
item.available_at == foundation.views[item.view_ref_id].available_at,
|
||||
"$.view_availability",
|
||||
"availability must equal its Foundation fact",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
upstream = _upstream(dataset_snapshot, foundation)
|
||||
_check(
|
||||
canonical_json_bytes(row["upstream_evidence"]) == canonical_json_bytes(upstream),
|
||||
"$.upstream_evidence",
|
||||
"upstream evidence differs from complete input envelopes",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
row["evidence_scope"] == dataset_snapshot.evidence_scope == foundation.evidence_scope,
|
||||
"$.evidence_scope",
|
||||
"scope must equal both inputs",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
if row["evidence_scope"] == "real_data":
|
||||
_check(
|
||||
foundation.real_data_validation_status == "validated",
|
||||
"$.evidence_scope",
|
||||
"real-data Foundation validation required",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
quality = OutputQuality.from_dict(row["output_quality"])
|
||||
coverage = OutputCoverage.from_dict(row["output_coverage"])
|
||||
_check(
|
||||
quality.status == "passed" and all(item.status == "passed" for item in quality.checks),
|
||||
"$.output_quality",
|
||||
"all output checks must pass",
|
||||
)
|
||||
_check(
|
||||
coverage.status == "complete" and coverage.observed_count == coverage.expected_count,
|
||||
"$.output_coverage",
|
||||
"complete output coverage required",
|
||||
)
|
||||
artifact = OutputArtifactRef.from_dict(row["output_artifact_ref"])
|
||||
schema_digest = _digest(row["output_schema_digest"], "$.output_schema_digest")
|
||||
content_digest = _digest(row["output_content_digest"], "$.output_content_digest")
|
||||
_check(
|
||||
artifact.schema_digest == schema_digest and artifact.content_digest == content_digest,
|
||||
"$.output_artifact_ref",
|
||||
"output artifact mismatch",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
(output_schema_bytes is None) == (output_content_bytes is None),
|
||||
"$.output_artifact_ref",
|
||||
"both output payloads required together",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
validation = PayloadValidation.REFERENCE_ONLY
|
||||
if output_schema_bytes is not None and output_content_bytes is not None:
|
||||
for data, expected_digest, path in (
|
||||
(output_schema_bytes, schema_digest, "$.output_schema_bytes"),
|
||||
(output_content_bytes, content_digest, "$.output_content_bytes"),
|
||||
):
|
||||
canonical = _canonical_evidence_bytes(data, path)
|
||||
_public(json.loads(canonical), path)
|
||||
_check(
|
||||
_digest_bytes(canonical) == expected_digest,
|
||||
path,
|
||||
"output bytes mismatch",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
validation = PayloadValidation.PAYLOAD_REVALIDATED
|
||||
input_validation = _input_payloads(
|
||||
dataset_snapshot, foundation, selected, dataset_chunks, resolved_views
|
||||
)
|
||||
evaluation = _parse_utc(row["evaluation_at"], "$.evaluation_at")
|
||||
computed = _parse_utc(row["computed_at"], "$.computed_at")
|
||||
available = _parse_utc(row["artifact_available_at"], "$.artifact_available_at")
|
||||
_check(
|
||||
foundation.published_at <= evaluation <= computed <= available,
|
||||
"$.computed_at",
|
||||
"input publication <= actual evaluation <= computation <= artifact required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
for definition in definitions:
|
||||
_check(
|
||||
_parse_utc(definition.valid_from, "$.definitions[].valid_from")
|
||||
<= evaluation
|
||||
< _parse_utc(definition.valid_until, "$.definitions[].valid_until"),
|
||||
"$.definitions",
|
||||
"factor definition is not valid at actual evaluation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
producer = ProducerIdentity.from_dict(row["producer"])
|
||||
_check(
|
||||
producer.id == "quant_engine",
|
||||
"$.producer.id",
|
||||
"computation owner must be quant_engine",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_git_revision(row["code_revision"], "$.code_revision")
|
||||
actor = ActorIdentity.from_dict(row["actor"])
|
||||
correlation = _logical_id(row["correlation_id"], "$.correlation_id")
|
||||
cause = RetrospectiveCausation.from_dict(row["causation"])
|
||||
for name, parsed in (
|
||||
("output_quality", quality),
|
||||
("output_coverage", coverage),
|
||||
("output_artifact_ref", artifact),
|
||||
("producer", producer),
|
||||
("actor", actor),
|
||||
("causation", cause),
|
||||
):
|
||||
_check(
|
||||
canonical_json_bytes(row[name]) == canonical_json_bytes(parsed.to_dict()),
|
||||
f"$.{name}",
|
||||
"nested contract is not normalized",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
if cause.kind == "foundation":
|
||||
_check(
|
||||
cause.id == foundation.foundation_id and parent is None,
|
||||
"$.causation",
|
||||
"exact Foundation cause required",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
else:
|
||||
_check(
|
||||
type(parent) is RetrospectiveFactorSetRef,
|
||||
"$.causation",
|
||||
"exact v2 parent object required",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
assert parent is not None
|
||||
_check(
|
||||
cause.id == parent.factor_set_id
|
||||
and correlation == parent.correlation_id
|
||||
and row["evidence_scope"] == parent.evidence_scope,
|
||||
"$.causation",
|
||||
"parent identity/correlation/scope mismatch",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(parent.artifact_available_at, "$.parent.artifact_available_at")
|
||||
<= evaluation,
|
||||
"$.causation",
|
||||
"parent artifact postdates child evaluation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
factor_set_id = _string(row["factor_set_id"], "$.factor_set_id", _FACTOR_SET_ID)
|
||||
_check(
|
||||
factor_set_id == _content_address(row, "factor_set_id", "rhfactorsetv2:sha256:"),
|
||||
"$.factor_set_id",
|
||||
"factor result identity mismatch",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
cause.id != factor_set_id,
|
||||
"$.causation",
|
||||
"self parent is forbidden",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
instance = object.__new__(cls)
|
||||
values = {
|
||||
**row,
|
||||
"definition_ids": definition_ids,
|
||||
"selected_view_ref_ids": selected,
|
||||
"input_bindings": bindings,
|
||||
"view_availability": availability,
|
||||
"upstream_evidence": _freeze_json(upstream),
|
||||
"output_quality": quality,
|
||||
"output_coverage": coverage,
|
||||
"output_artifact_ref": artifact,
|
||||
"producer": producer,
|
||||
"actor": actor,
|
||||
"causation": cause,
|
||||
"payload_validation": validation,
|
||||
"input_payload_validation": input_validation,
|
||||
"_payload": _freeze_json(row),
|
||||
"_definitions": definitions,
|
||||
"_dataset_snapshot": dataset_snapshot,
|
||||
"_foundation": foundation,
|
||||
"_parent": parent,
|
||||
}
|
||||
for name, item in values.items():
|
||||
object.__setattr__(instance, name, item)
|
||||
return instance
|
||||
|
||||
def require_payloads_revalidated(self) -> None:
|
||||
_check(
|
||||
self.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
and self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
|
||||
"$.payload_validation",
|
||||
"reference-only data is not computation admission",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
@@ -0,0 +1,995 @@
|
||||
"""Retrospective-only portfolio/risk evidence with separate business/actual clocks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Self, TypedDict, cast
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from quant_engine.artifact import (
|
||||
EvidenceQualification,
|
||||
_performance_compare,
|
||||
_performance_validate_tree,
|
||||
)
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
FactorContractError,
|
||||
_duplicate_key_pairs,
|
||||
_parse_utc,
|
||||
_string,
|
||||
_thaw_json,
|
||||
)
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
ReceiptStatus,
|
||||
PortfolioRiskContractError,
|
||||
PortfolioRiskContractErrorCode,
|
||||
RiskAssessmentStatus,
|
||||
RiskFindingCode,
|
||||
_CLOSURE_ATOL,
|
||||
_CLOSURE_RTOL,
|
||||
_finite_number,
|
||||
_series_mapping,
|
||||
_validate_covariance_structure,
|
||||
_canonical_json,
|
||||
_constraint_metrics,
|
||||
_constraint_residuals,
|
||||
_digest,
|
||||
_document_sha256,
|
||||
_immutable_float_mapping,
|
||||
_mapping_dict,
|
||||
_payload_digest,
|
||||
_semver,
|
||||
_text,
|
||||
)
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
RetrospectiveBacktestEvidenceManifest,
|
||||
_freeze_numeric_evidence,
|
||||
_validated_run,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_data_contracts import _IDS, _check, _public, _shape
|
||||
from quant_engine.risk import CovarianceSnapshot, labeled_component_risk
|
||||
|
||||
_RUN_ID = re.compile(r"^rhbacktestrunv2:sha256:[0-9a-f]{64}$")
|
||||
|
||||
|
||||
def _json_object(value: str | bytes) -> dict[str, Any]:
|
||||
_check(
|
||||
type(value) in {str, bytes},
|
||||
"$",
|
||||
"canonical JSON text/bytes required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
try:
|
||||
raw = value.encode("utf-8") if isinstance(value, str) else value
|
||||
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
|
||||
except (json.JSONDecodeError, UnicodeError) as error:
|
||||
raise FactorContractError(
|
||||
ContractErrorCode.INVALID_FORMAT, "$", "valid UTF-8 JSON required"
|
||||
) from error
|
||||
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
|
||||
_performance_validate_tree(document, "$")
|
||||
_check(
|
||||
_canonical_json(document).encode() == raw,
|
||||
"$",
|
||||
"canonical numeric JSON required",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
return cast(dict[str, Any], document)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePortfolioTarget:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
target_id: str
|
||||
backtest_run_id: str
|
||||
dataset_snapshot_id: str
|
||||
weights: Mapping[str, float]
|
||||
effective_at: str
|
||||
created_at: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
backtest_run_id: str,
|
||||
dataset_snapshot_id: str,
|
||||
weights: Mapping[str, float],
|
||||
effective_at: str,
|
||||
created_at: str,
|
||||
) -> Self:
|
||||
_string(backtest_run_id, "$.backtest_run_id", _RUN_ID)
|
||||
_string(dataset_snapshot_id, "$.dataset_snapshot_id", _IDS["snapshot"])
|
||||
normalized = _immutable_float_mapping(weights, "$.weights")
|
||||
_check(bool(normalized), "$.weights", "non-empty target asset set required")
|
||||
for instrument in normalized:
|
||||
_string(instrument, "$.weights.keys", _IDS["instrument"])
|
||||
effective = _parse_utc(effective_at, "$.effective_at")
|
||||
created = _parse_utc(created_at, "$.created_at")
|
||||
_check(
|
||||
effective <= created,
|
||||
"$.effective_at",
|
||||
"historical effective time exceeds actual creation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.portfolio-target",
|
||||
"schema_version": "2.0.0",
|
||||
"backtest_run_id": backtest_run_id,
|
||||
"dataset_snapshot_id": dataset_snapshot_id,
|
||||
"weights": _mapping_dict(normalized),
|
||||
"effective_at": effective_at,
|
||||
"created_at": created_at,
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
}
|
||||
_public(payload)
|
||||
payload["target_id"] = "rhportfoliotargetv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(cls)
|
||||
for key, value in {
|
||||
**payload,
|
||||
"weights": normalized,
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
}.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version target_id backtest_run_id dataset_snapshot_id weights effective_at created_at usage historical_availability",
|
||||
)
|
||||
rebuilt = cls.create(
|
||||
**{
|
||||
key: row[key]
|
||||
for key in (
|
||||
"backtest_run_id",
|
||||
"dataset_snapshot_id",
|
||||
"weights",
|
||||
"effective_at",
|
||||
"created_at",
|
||||
)
|
||||
}
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return rebuilt
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes) -> Self:
|
||||
return cls.from_dict(_json_object(value))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _PortfolioInputs:
|
||||
run: RetrospectiveBacktestRunRef
|
||||
manifest: RetrospectiveBacktestEvidenceManifest
|
||||
target: RetrospectivePortfolioTarget
|
||||
constraints: ConstraintSetV1
|
||||
freshness: FreshnessPolicy
|
||||
weights: Mapping[str, float]
|
||||
prior: Mapping[str, float] | None
|
||||
metrics: dict[str, float | int | None]
|
||||
residuals: dict[str, float]
|
||||
input_payload: dict[str, object]
|
||||
|
||||
|
||||
def _material(
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
objective_name: str,
|
||||
objective_version: str,
|
||||
objective_digest: str,
|
||||
model_name: str,
|
||||
model_version: str,
|
||||
model_digest: str,
|
||||
expected_return_digest: str,
|
||||
covariance_digest: str,
|
||||
scenario_digest: str,
|
||||
constraints: ConstraintSetV1,
|
||||
freshness_policy: FreshnessPolicy,
|
||||
prior_weights: Mapping[str, float] | None = None,
|
||||
) -> _PortfolioInputs:
|
||||
run = _validated_run(backtest_run_ref)
|
||||
for item, expected, path in (
|
||||
(manifest, RetrospectiveBacktestEvidenceManifest, "$.manifest"),
|
||||
(target, RetrospectivePortfolioTarget, "$.target"),
|
||||
(constraints, ConstraintSetV1, "$.constraints"),
|
||||
(freshness_policy, FreshnessPolicy, "$.freshness_policy"),
|
||||
):
|
||||
_check(
|
||||
type(item) is expected,
|
||||
path,
|
||||
f"explicit {expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
checked_manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
manifest.to_dict(), artifact=manifest._artifact, backtest_run_ref=run
|
||||
)
|
||||
checked_target = RetrospectivePortfolioTarget.from_dict(target.to_dict())
|
||||
_check(
|
||||
checked_manifest.qualification is EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
"$.manifest.qualification",
|
||||
"contract-qualified retrospective S3 required",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
_check(
|
||||
checked_target.backtest_run_id == run.run_id
|
||||
and checked_target.dataset_snapshot_id == run.dataset_snapshot_id,
|
||||
"$.target",
|
||||
"target and S3 identities differ",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
foundation = run._factor_set._foundation
|
||||
selected_routes = {
|
||||
identity
|
||||
for view_id in run._factor_set.selected_view_ref_ids
|
||||
for identity in foundation.views[view_id].instrument_route_revision_ids
|
||||
}
|
||||
selected_instruments = {
|
||||
row["instrument_id"]
|
||||
for row in foundation.to_dict()["instrument_routes"]
|
||||
if row["route_revision_id"] in selected_routes
|
||||
}
|
||||
_check(
|
||||
set(checked_target.weights) <= selected_instruments,
|
||||
"$.target.weights",
|
||||
"target assets must be selected logical instruments",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
constraints = ConstraintSetV1.from_dict(constraints.to_dict())
|
||||
freshness_policy = FreshnessPolicy.from_dict(freshness_policy.to_dict())
|
||||
prior = (
|
||||
None
|
||||
if prior_weights is None
|
||||
else _immutable_float_mapping(prior_weights, "$.prior_weights")
|
||||
)
|
||||
if prior is not None:
|
||||
_check(
|
||||
set(prior) <= selected_instruments,
|
||||
"$.prior_weights",
|
||||
"prior assets outside selected instruments",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
weights = checked_target.weights
|
||||
metrics = _constraint_metrics(weights, prior)
|
||||
residuals = _constraint_residuals(constraints, weights, metrics)
|
||||
payload: dict[str, object] = {
|
||||
"contract_name": "researchhub.portfolio-computation-input",
|
||||
"schema_version": "2.0.0",
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"run_ref_document_sha256": _document_sha256(run.to_json()),
|
||||
"manifest_document_sha256": _document_sha256(checked_manifest.to_json()),
|
||||
"portfolio_target": checked_target.to_dict(),
|
||||
"objective": {
|
||||
"name": _text(objective_name, "$.objective_name"),
|
||||
"version": _semver(objective_version, "$.objective_version"),
|
||||
"digest": _digest(objective_digest, "$.objective_digest"),
|
||||
},
|
||||
"model": {
|
||||
"name": _text(model_name, "$.model_name"),
|
||||
"version": _semver(model_version, "$.model_version"),
|
||||
"digest": _digest(model_digest, "$.model_digest"),
|
||||
},
|
||||
"expected_return_digest": _digest(expected_return_digest, "$.expected_return_digest"),
|
||||
"covariance_digest": _digest(covariance_digest, "$.covariance_digest"),
|
||||
"scenario_digest": _digest(scenario_digest, "$.scenario_digest"),
|
||||
"freshness_policy_digest": _payload_digest(freshness_policy.to_dict()),
|
||||
"prior_weights": None if prior is None else _mapping_dict(prior),
|
||||
}
|
||||
_public(payload)
|
||||
return _PortfolioInputs(
|
||||
run,
|
||||
checked_manifest,
|
||||
checked_target,
|
||||
constraints,
|
||||
freshness_policy,
|
||||
weights,
|
||||
prior,
|
||||
metrics,
|
||||
residuals,
|
||||
payload,
|
||||
)
|
||||
|
||||
|
||||
def _receipt_digests(inputs: _PortfolioInputs) -> dict[str, str | float]:
|
||||
return {
|
||||
"input_digest": _payload_digest(inputs.input_payload),
|
||||
"constraint_digest": _payload_digest(inputs.constraints.to_dict()),
|
||||
"output_digest": _payload_digest(
|
||||
{
|
||||
"weights": _mapping_dict(inputs.weights),
|
||||
"metrics": inputs.metrics,
|
||||
"constraint_residuals": inputs.residuals,
|
||||
}
|
||||
),
|
||||
"max_constraint_residual": max(inputs.residuals.values(), default=0.0),
|
||||
}
|
||||
|
||||
|
||||
def compute_retrospective_portfolio_receipt_digests(**kwargs: Any) -> Mapping[str, str | float]:
|
||||
"""Recompute receipt claims; the returned digests are not producer authentication."""
|
||||
return MappingProxyType(_receipt_digests(_material(**kwargs)))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePortfolioDecision:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
decision_id: str
|
||||
run_id: str
|
||||
manifest_id: str
|
||||
evidence_digest: str
|
||||
dataset_snapshot_id: str
|
||||
run_ref_document_sha256: str
|
||||
manifest_document_sha256: str
|
||||
source_universe_digest: str
|
||||
portfolio_asset_set_digest: str
|
||||
target_id: str
|
||||
target_weights: Mapping[str, float]
|
||||
prior_weights: Mapping[str, float] | None
|
||||
objective_name: str
|
||||
objective_version: str
|
||||
objective_digest: str
|
||||
model_name: str
|
||||
model_version: str
|
||||
model_digest: str
|
||||
expected_return_digest: str
|
||||
covariance_digest: str
|
||||
scenario_digest: str
|
||||
constraints: ConstraintSetV1
|
||||
freshness_policy: FreshnessPolicy
|
||||
receipt: ComputationReceipt
|
||||
gross_exposure: float
|
||||
net_exposure: float
|
||||
turnover_l1: float | None
|
||||
position_count: int
|
||||
constraint_residuals: Mapping[str, float]
|
||||
output_digest: str
|
||||
effective_at: str
|
||||
created_at: str
|
||||
computed_at: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_target: RetrospectivePortfolioTarget = field(repr=False, compare=False)
|
||||
_run: RetrospectiveBacktestRunRef = field(repr=False, compare=False)
|
||||
_manifest: RetrospectiveBacktestEvidenceManifest = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
# Rebuild from independent typed inputs, not from a self-approved target in the wire.
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
|
||||
)
|
||||
arguments = {
|
||||
key: row[key]
|
||||
for key in (
|
||||
"objective_name",
|
||||
"objective_version",
|
||||
"objective_digest",
|
||||
"model_name",
|
||||
"model_version",
|
||||
"model_digest",
|
||||
"expected_return_digest",
|
||||
"covariance_digest",
|
||||
"scenario_digest",
|
||||
"computed_at",
|
||||
"prior_weights",
|
||||
)
|
||||
}
|
||||
rebuilt = build_retrospective_portfolio_decision(
|
||||
**arguments,
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=target,
|
||||
constraints=ConstraintSetV1.from_dict(row["constraints"]),
|
||||
freshness_policy=FreshnessPolicy.from_dict(row["freshness_policy"]),
|
||||
receipt=ComputationReceipt.from_dict(row["receipt"]),
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_json_object(value), **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_portfolio_decision(
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
objective_name: str,
|
||||
objective_version: str,
|
||||
objective_digest: str,
|
||||
model_name: str,
|
||||
model_version: str,
|
||||
model_digest: str,
|
||||
expected_return_digest: str,
|
||||
covariance_digest: str,
|
||||
scenario_digest: str,
|
||||
constraints: ConstraintSetV1,
|
||||
freshness_policy: FreshnessPolicy,
|
||||
receipt: ComputationReceipt,
|
||||
computed_at: str,
|
||||
prior_weights: Mapping[str, float] | None = None,
|
||||
) -> RetrospectivePortfolioDecision:
|
||||
"""Verify the existing constraints and receipt, with two explicitly different clocks."""
|
||||
_check(
|
||||
type(receipt) is ComputationReceipt,
|
||||
"$.receipt",
|
||||
"typed computation receipt required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
receipt = ComputationReceipt.from_dict(receipt.to_dict())
|
||||
inputs = _material(
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=target,
|
||||
objective_name=objective_name,
|
||||
objective_version=objective_version,
|
||||
objective_digest=objective_digest,
|
||||
model_name=model_name,
|
||||
model_version=model_version,
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness_policy,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
run, manifest, target = inputs.run, inputs.manifest, inputs.target
|
||||
computed = _parse_utc(computed_at, "$.computed_at")
|
||||
created = _parse_utc(target.created_at, "$.target.created_at")
|
||||
available = _parse_utc(manifest.artifact_available_at, "$.manifest.artifact_available_at")
|
||||
_check(
|
||||
available <= created <= computed,
|
||||
"$.target.created_at",
|
||||
"artifact availability <= actual target creation <= computation required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(receipt.computed_at, "$.receipt.computed_at") == computed,
|
||||
"$.receipt.computed_at",
|
||||
"receipt actual time differs from computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
(computed - available).total_seconds() <= inputs.freshness.max_manifest_age_seconds,
|
||||
"$.manifest.artifact_available_at",
|
||||
"manifest is stale at actual computation",
|
||||
)
|
||||
_check(
|
||||
receipt.status not in {ReceiptStatus.FAILED, ReceiptStatus.FALLBACK},
|
||||
"$.receipt.status",
|
||||
"failed/fallback computation cannot form a result",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
digests = _receipt_digests(inputs)
|
||||
for key, expected in digests.items():
|
||||
_check(
|
||||
getattr(receipt, key) == expected,
|
||||
f"$.receipt.{key}",
|
||||
"receipt differs from independently recomputed evidence",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
digests["max_constraint_residual"] == 0.0,
|
||||
"$.constraints",
|
||||
"target violates supported constraints",
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.portfolio-decision",
|
||||
"schema_version": "2.0.0",
|
||||
"run_id": run.run_id,
|
||||
"manifest_id": manifest.manifest_id,
|
||||
"evidence_digest": manifest.evidence_digest,
|
||||
"dataset_snapshot_id": run.dataset_snapshot_id,
|
||||
"run_ref_document_sha256": _document_sha256(run.to_json()),
|
||||
"manifest_document_sha256": _document_sha256(manifest.to_json()),
|
||||
"source_universe_digest": run.universe_digest,
|
||||
"portfolio_asset_set_digest": _payload_digest(sorted(inputs.weights)),
|
||||
"target_id": target.target_id,
|
||||
"target_weights": _mapping_dict(inputs.weights),
|
||||
"prior_weights": None if inputs.prior is None else _mapping_dict(inputs.prior),
|
||||
"objective_name": objective_name,
|
||||
"objective_version": objective_version,
|
||||
"objective_digest": objective_digest,
|
||||
"model_name": model_name,
|
||||
"model_version": model_version,
|
||||
"model_digest": model_digest,
|
||||
"expected_return_digest": expected_return_digest,
|
||||
"covariance_digest": covariance_digest,
|
||||
"scenario_digest": scenario_digest,
|
||||
"constraints": inputs.constraints.to_dict(),
|
||||
"freshness_policy": inputs.freshness.to_dict(),
|
||||
"receipt": receipt.to_dict(),
|
||||
**inputs.metrics,
|
||||
"constraint_residuals": inputs.residuals,
|
||||
"output_digest": digests["output_digest"],
|
||||
"effective_at": target.effective_at,
|
||||
"created_at": target.created_at,
|
||||
"computed_at": computed_at,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
_public(payload)
|
||||
payload["decision_id"] = "rhportfoliodecisionv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(RetrospectivePortfolioDecision)
|
||||
values = {
|
||||
**payload,
|
||||
"target_weights": inputs.weights,
|
||||
"prior_weights": inputs.prior,
|
||||
"constraints": inputs.constraints,
|
||||
"freshness_policy": inputs.freshness,
|
||||
"receipt": receipt,
|
||||
"constraint_residuals": MappingProxyType(inputs.residuals),
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
"_target": target,
|
||||
"_run": run,
|
||||
"_manifest": manifest,
|
||||
}
|
||||
for key, value in values.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveRiskAssessment:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
assessment_id: str
|
||||
decision_id: str
|
||||
run_id: str
|
||||
manifest_id: str
|
||||
dataset_snapshot_id: str
|
||||
covariance_data_snapshot_id: str
|
||||
covariance_snapshot_id: str
|
||||
covariance_as_of_date: str
|
||||
covariance_method: str
|
||||
covariance_window_start_date: str
|
||||
covariance_window_end_date: str
|
||||
covariance_observations: int | None
|
||||
covariance_lookback_sessions: int | None
|
||||
covariance_missing_policy: str
|
||||
covariance_input_digest: str
|
||||
covariance_matrix_digest: str
|
||||
return_frequency: str
|
||||
periods_per_year: int
|
||||
risk_model_name: str
|
||||
risk_model_version: str
|
||||
risk_model_digest: str
|
||||
freshness_policy_digest: str
|
||||
scenario_digest: str
|
||||
portfolio_volatility_limit: float | None
|
||||
risk_budget: Mapping[str, float]
|
||||
groups: Mapping[str, str] | None
|
||||
marginal_risk: Mapping[str, float]
|
||||
component_risk: Mapping[str, float]
|
||||
percentage_risk: Mapping[str, float]
|
||||
portfolio_volatility: float | None
|
||||
group_exposure: Mapping[str, float]
|
||||
findings: tuple[RiskFindingCode, ...]
|
||||
status: RiskAssessmentStatus
|
||||
qualified: bool
|
||||
effective_at: str
|
||||
computed_at: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
portfolio_decision: RetrospectivePortfolioDecision,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
covariance: CovarianceSnapshot,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
|
||||
)
|
||||
rebuilt = assess_retrospective_portfolio_risk(
|
||||
portfolio_decision=portfolio_decision,
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
covariance=covariance,
|
||||
**{
|
||||
key: row[key]
|
||||
for key in (
|
||||
"risk_model_name",
|
||||
"risk_model_version",
|
||||
"risk_model_digest",
|
||||
"risk_budget",
|
||||
"portfolio_volatility_limit",
|
||||
"groups",
|
||||
"computed_at",
|
||||
)
|
||||
},
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_json_object(value), **kwargs)
|
||||
|
||||
|
||||
class _RiskContext(TypedDict):
|
||||
decision: RetrospectivePortfolioDecision
|
||||
covariance: CovarianceSnapshot
|
||||
matrix_digest: str
|
||||
risk_model_name: str
|
||||
risk_model_version: str
|
||||
risk_model_digest: str
|
||||
portfolio_volatility_limit: float | None
|
||||
risk_budget: Mapping[str, float]
|
||||
groups: Mapping[str, str] | None
|
||||
computed_at: str
|
||||
|
||||
|
||||
def _risk_result(
|
||||
*,
|
||||
decision: RetrospectivePortfolioDecision,
|
||||
covariance: CovarianceSnapshot,
|
||||
matrix_digest: str,
|
||||
risk_model_name: str,
|
||||
risk_model_version: str,
|
||||
risk_model_digest: str,
|
||||
portfolio_volatility_limit: float | None,
|
||||
risk_budget: Mapping[str, float],
|
||||
groups: Mapping[str, str] | None,
|
||||
marginal: Mapping[str, float],
|
||||
component: Mapping[str, float],
|
||||
percentage: Mapping[str, float],
|
||||
volatility: float | None,
|
||||
grouped: Mapping[str, float],
|
||||
findings: tuple[RiskFindingCode, ...],
|
||||
status: RiskAssessmentStatus,
|
||||
qualified: bool,
|
||||
computed_at: str,
|
||||
) -> RetrospectiveRiskAssessment:
|
||||
assert covariance.window_start_date is not None
|
||||
assert covariance.window_end_date is not None
|
||||
payload = {
|
||||
"contract_name": "researchhub.risk-assessment",
|
||||
"schema_version": "2.0.0",
|
||||
"decision_id": decision.decision_id,
|
||||
"run_id": decision.run_id,
|
||||
"manifest_id": decision.manifest_id,
|
||||
"dataset_snapshot_id": decision.dataset_snapshot_id,
|
||||
"covariance_data_snapshot_id": covariance.data_snapshot_id,
|
||||
"covariance_snapshot_id": covariance.snapshot_id,
|
||||
"covariance_as_of_date": covariance.as_of_date.isoformat(),
|
||||
"covariance_method": covariance.method,
|
||||
"covariance_window_start_date": covariance.window_start_date.isoformat(),
|
||||
"covariance_window_end_date": covariance.window_end_date.isoformat(),
|
||||
"covariance_observations": covariance.observations,
|
||||
"covariance_lookback_sessions": covariance.lookback_sessions,
|
||||
"covariance_missing_policy": covariance.missing_policy,
|
||||
"covariance_input_digest": "sha256:" + covariance.input_sha256,
|
||||
"covariance_matrix_digest": matrix_digest,
|
||||
"return_frequency": covariance.return_frequency,
|
||||
"periods_per_year": covariance.periods_per_year,
|
||||
"risk_model_name": risk_model_name,
|
||||
"risk_model_version": risk_model_version,
|
||||
"risk_model_digest": risk_model_digest,
|
||||
"freshness_policy_digest": _payload_digest(decision.freshness_policy.to_dict()),
|
||||
"scenario_digest": decision.scenario_digest,
|
||||
"portfolio_volatility_limit": portfolio_volatility_limit,
|
||||
"risk_budget": _mapping_dict(risk_budget),
|
||||
"groups": None if groups is None else dict(groups),
|
||||
"marginal_risk": _mapping_dict(marginal),
|
||||
"component_risk": _mapping_dict(component),
|
||||
"percentage_risk": _mapping_dict(percentage),
|
||||
"portfolio_volatility": volatility,
|
||||
"group_exposure": _mapping_dict(grouped),
|
||||
"findings": [finding.value for finding in findings],
|
||||
"status": status.value,
|
||||
"qualified": qualified,
|
||||
"effective_at": decision.effective_at,
|
||||
"computed_at": computed_at,
|
||||
"observation_cutoff": decision.observation_cutoff,
|
||||
"evidence_scope": decision.evidence_scope,
|
||||
"usage": decision.usage,
|
||||
"historical_availability": decision.historical_availability,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
_public(payload)
|
||||
payload["assessment_id"] = "rhriskassessmentv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(RetrospectiveRiskAssessment)
|
||||
values = {
|
||||
**payload,
|
||||
"risk_budget": risk_budget,
|
||||
"groups": groups,
|
||||
"marginal_risk": marginal,
|
||||
"component_risk": component,
|
||||
"percentage_risk": percentage,
|
||||
"group_exposure": grouped,
|
||||
"findings": findings,
|
||||
"status": status,
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
}
|
||||
for key, value in values.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
|
||||
def assess_retrospective_portfolio_risk(
|
||||
*,
|
||||
portfolio_decision: RetrospectivePortfolioDecision,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
covariance: CovarianceSnapshot,
|
||||
risk_model_name: str,
|
||||
risk_model_version: str,
|
||||
risk_model_digest: str,
|
||||
computed_at: str,
|
||||
risk_budget: Mapping[str, float] | None = None,
|
||||
portfolio_volatility_limit: float | None = None,
|
||||
groups: Mapping[str, str] | None = None,
|
||||
) -> RetrospectiveRiskAssessment:
|
||||
"""Use the existing Euler decomposition once; distinguish the two freshness clocks."""
|
||||
_check(
|
||||
type(portfolio_decision) is RetrospectivePortfolioDecision,
|
||||
"$.portfolio_decision",
|
||||
"explicit v2 portfolio result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
type(covariance) is CovarianceSnapshot,
|
||||
"$.covariance",
|
||||
"typed covariance required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
decision = RetrospectivePortfolioDecision.from_dict(
|
||||
portfolio_decision.to_dict(),
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=portfolio_decision._target,
|
||||
)
|
||||
actual_computed = _parse_utc(computed_at, "$.computed_at")
|
||||
_check(
|
||||
_parse_utc(decision.computed_at, "$.portfolio_decision.computed_at") <= actual_computed,
|
||||
"$.computed_at",
|
||||
"risk computation precedes portfolio computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
manifest_age = (
|
||||
actual_computed
|
||||
- _parse_utc(decision._manifest.artifact_available_at, "$.manifest.artifact_available_at")
|
||||
).total_seconds()
|
||||
_check(
|
||||
0 <= manifest_age <= decision.freshness_policy.max_manifest_age_seconds,
|
||||
"$.manifest.artifact_available_at",
|
||||
"manifest is stale at actual risk computation",
|
||||
)
|
||||
_check(
|
||||
covariance.data_snapshot_id == decision.dataset_snapshot_id,
|
||||
"$.covariance.data_snapshot_id",
|
||||
"covariance and decision data differ",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
business_date = _parse_utc(decision.effective_at, "$.portfolio_decision.effective_at").date()
|
||||
_check(
|
||||
covariance.window_start_date is not None and covariance.window_end_date is not None,
|
||||
"$.covariance",
|
||||
"bounded covariance window required",
|
||||
)
|
||||
assert covariance.window_start_date is not None
|
||||
assert covariance.window_end_date is not None
|
||||
_check(
|
||||
covariance.window_start_date
|
||||
<= covariance.window_end_date
|
||||
<= covariance.as_of_date
|
||||
<= business_date,
|
||||
"$.covariance.as_of_date",
|
||||
"covariance business dates exceed the historical target date",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
(business_date - covariance.as_of_date).days
|
||||
<= decision.freshness_policy.max_covariance_age_days,
|
||||
"$.covariance.as_of_date",
|
||||
"covariance is stale at historical target date",
|
||||
)
|
||||
_check(
|
||||
_digest("sha256:" + covariance.input_sha256, "$.covariance.input_sha256")
|
||||
== decision.covariance_digest,
|
||||
"$.covariance.input_sha256",
|
||||
"covariance input differs from portfolio receipt",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
name = _text(risk_model_name, "$.risk_model_name")
|
||||
version = _semver(risk_model_version, "$.risk_model_version")
|
||||
model_digest = _digest(risk_model_digest, "$.risk_model_digest")
|
||||
limit = (
|
||||
None
|
||||
if portfolio_volatility_limit is None
|
||||
else _finite_number(
|
||||
portfolio_volatility_limit, "$.portfolio_volatility_limit", non_negative=True
|
||||
)
|
||||
)
|
||||
budget: Mapping[str, float] = (
|
||||
MappingProxyType({})
|
||||
if risk_budget is None
|
||||
else _immutable_float_mapping(risk_budget, "$.risk_budget")
|
||||
)
|
||||
_check(
|
||||
all(value >= 0 for value in budget.values())
|
||||
and set(budget) <= decision.target_weights.keys(),
|
||||
"$.risk_budget",
|
||||
"risk budgets must be non-negative and use target labels",
|
||||
)
|
||||
normalized_groups = None
|
||||
if groups is not None:
|
||||
_check(
|
||||
isinstance(groups, Mapping),
|
||||
"$.groups",
|
||||
"mapping required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
group_values = {
|
||||
_text(key, "$.groups.keys"): _text(value, "$.groups.values")
|
||||
for key, value in groups.items()
|
||||
}
|
||||
_check(
|
||||
set(group_values) == decision.target_weights.keys(),
|
||||
"$.groups",
|
||||
"groups must label every target exactly once",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
normalized_groups = MappingProxyType(dict(sorted(group_values.items())))
|
||||
aligned = _validate_covariance_structure(decision.target_weights, covariance)
|
||||
matrix_digest = _payload_digest(
|
||||
{
|
||||
"assets": sorted(decision.target_weights),
|
||||
"matrix": aligned.to_numpy(dtype=float).tolist(),
|
||||
}
|
||||
)
|
||||
arguments: _RiskContext = {
|
||||
"decision": decision,
|
||||
"covariance": covariance,
|
||||
"matrix_digest": matrix_digest,
|
||||
"risk_model_name": name,
|
||||
"risk_model_version": version,
|
||||
"risk_model_digest": model_digest,
|
||||
"portfolio_volatility_limit": limit,
|
||||
"risk_budget": budget,
|
||||
"groups": normalized_groups,
|
||||
"computed_at": computed_at,
|
||||
}
|
||||
empty: Mapping[str, float] = MappingProxyType({})
|
||||
|
||||
def unavailable(finding: RiskFindingCode) -> RetrospectiveRiskAssessment:
|
||||
return _risk_result(
|
||||
**arguments,
|
||||
marginal=empty,
|
||||
component=empty,
|
||||
percentage=empty,
|
||||
volatility=None,
|
||||
grouped=empty,
|
||||
findings=(finding,),
|
||||
status=RiskAssessmentStatus.UNAVAILABLE,
|
||||
qualified=False,
|
||||
)
|
||||
|
||||
weights = pd.Series(_mapping_dict(decision.target_weights), dtype=float, name="weight")
|
||||
try:
|
||||
decomposition = labeled_component_risk(weights, aligned * covariance.periods_per_year)
|
||||
except ValueError as error:
|
||||
finding = {
|
||||
"covariance must be positive semidefinite": RiskFindingCode.COVARIANCE_NOT_PSD,
|
||||
"weights and covariance must produce positive portfolio variance": RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE,
|
||||
}.get(str(error))
|
||||
if finding is None:
|
||||
raise PortfolioRiskContractError(
|
||||
PortfolioRiskContractErrorCode.COMPUTATION_FAILURE,
|
||||
"$.covariance",
|
||||
"risk computation failed",
|
||||
) from error
|
||||
return unavailable(finding)
|
||||
marginal = _series_mapping(decomposition.marginal)
|
||||
component = _series_mapping(decomposition.component)
|
||||
percentage = _series_mapping(decomposition.percentage)
|
||||
volatility = _finite_number(
|
||||
decomposition.portfolio_volatility, "$.risk_output.portfolio_volatility", non_negative=True
|
||||
)
|
||||
if not (
|
||||
set(marginal) == set(component) == set(percentage) == decision.target_weights.keys()
|
||||
and math.isclose(
|
||||
sum(component.values()), volatility, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
|
||||
)
|
||||
and math.isclose(
|
||||
sum(percentage.values()), 1.0, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
|
||||
)
|
||||
):
|
||||
return unavailable(RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED)
|
||||
grouped = (
|
||||
empty
|
||||
if normalized_groups is None
|
||||
else _series_mapping(
|
||||
decomposition.grouped_component(pd.Series(dict(normalized_groups), dtype="object"))
|
||||
)
|
||||
)
|
||||
breached = (limit is not None and volatility > limit + _CLOSURE_ATOL) or any(
|
||||
percentage[label] > maximum + _CLOSURE_ATOL for label, maximum in budget.items()
|
||||
)
|
||||
return _risk_result(
|
||||
**arguments,
|
||||
marginal=marginal,
|
||||
component=component,
|
||||
percentage=percentage,
|
||||
volatility=volatility,
|
||||
grouped=grouped,
|
||||
findings=(RiskFindingCode.RISK_BUDGET_BREACH,) if breached else (),
|
||||
status=RiskAssessmentStatus.READY,
|
||||
qualified=not breached,
|
||||
)
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
{
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"replay_spec_digest": "sha256:20f07fcb526bc38b4ba3d71d6c3b00a8c63ad96cab0fecd869326503ab42d98b",
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117",
|
||||
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"table_content_digests": {
|
||||
"run": "sha256:7d947ec93f714641669cbb14bd69dd8cf30387aabb7f3a086918fc2e878ab05c",
|
||||
"signals": "sha256:72dc15064cc45d7c51d2dd4b8c3a6d8c7d4155232d5d70d3c9e7697fab70ce50",
|
||||
"trades": "sha256:2b8b9321e7993941ac486cf50cab5b4c6425b2571a4701f0eb02273ec9a96c53",
|
||||
"positions": "sha256:b456a48fab51742b05084ca6dcaf01c03dfe5b39ad71215ada070e9b1f59f7ea",
|
||||
"nav": "sha256:25649efce860b76410f87dbd36c81085887620086dc0b48b07099918f65d9c78",
|
||||
"performance": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
|
||||
"attribution": "sha256:ad0f12f668d0a2d9ae5b3636d29989ab4bafcfb95f5de77abeb52a6d9e95d366",
|
||||
"attribution_daily": "sha256:d4459ad650f88871d7b1e40392033037b5818ef92fdf1ee021868929fc1455a5",
|
||||
"risk": "sha256:4816dd4812b5ff2e97e74bf34ca2221bfde675b387a96e277683557f0ad7d975"
|
||||
}
|
||||
}
|
||||
+206
@@ -0,0 +1,206 @@
|
||||
{
|
||||
"dataset_snapshot": {
|
||||
"contract_name": "researchhub.dataset-snapshot",
|
||||
"schema_version": "1.0.0",
|
||||
"snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"descriptor": {
|
||||
"dataset": {
|
||||
"dataset_id": "rhdataset:market:0123456789abcdef0123456789abcdef",
|
||||
"dataset_kind": "market",
|
||||
"record_schema_version": "1.0.0",
|
||||
"dimensions": ["instrument_id", "effective_time"]
|
||||
},
|
||||
"published_at": "2026-01-02T07:05:00Z",
|
||||
"time_semantics": {
|
||||
"effective_time": {
|
||||
"start_inclusive": "2026-01-02T07:00:00Z",
|
||||
"end_inclusive": "2026-01-02T07:00:00Z"
|
||||
},
|
||||
"knowledge_time": {
|
||||
"start_inclusive": "2026-01-02T07:01:00Z",
|
||||
"end_inclusive": "2026-01-02T07:01:00Z"
|
||||
},
|
||||
"pit_cutoff": "2026-01-02T07:01:00Z"
|
||||
},
|
||||
"content": {
|
||||
"digest_algorithm": "sha256",
|
||||
"canonicalization": "RFC8785",
|
||||
"record_order": "canonical-record-byte-order",
|
||||
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"logical_manifest": {
|
||||
"record_count": 2,
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"record_count": 2
|
||||
}
|
||||
]
|
||||
},
|
||||
"manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"record_count": 2
|
||||
},
|
||||
"lineage": {
|
||||
"publisher": {"id": "researchhub.data", "version": "1.0.0"},
|
||||
"transformation": {
|
||||
"id": "rhtransform:00112233445566778899aabbccddeeff",
|
||||
"version": "1.0.0"
|
||||
},
|
||||
"upstream_snapshot_ids": [],
|
||||
"upstream_content_digests": []
|
||||
},
|
||||
"quality": {
|
||||
"status": "passed",
|
||||
"checks": [
|
||||
{
|
||||
"check_id": "completeness",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:876fc2fcc6414ddc3f824a47f475d34c82a53d2bda5dc72a234a3f3164e8e2ec"
|
||||
},
|
||||
{
|
||||
"check_id": "pit_time_integrity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:90a6cc46b9f2ab317a1c6d14dc173784e19b5621cd7fe956e8f41338cbdc5944"
|
||||
}
|
||||
]
|
||||
},
|
||||
"qualification": {
|
||||
"status": "qualified",
|
||||
"policy_id": "researchhub.dataset-snapshot.pit",
|
||||
"policy_version": "1.0.0",
|
||||
"evaluated_at": "2026-01-02T07:04:00Z",
|
||||
"evidence_digest": "sha256:e192462f9022f2b477f73cdbe9e6c9f891ebcfdc2b4ed4f8ddd7b1ff107ee6a6"
|
||||
}
|
||||
}
|
||||
},
|
||||
"data_foundation": {
|
||||
"contract_name": "researchhub.data-foundation",
|
||||
"schema_version": "1.0.0",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"pit_cutoff": "2026-01-03T00:00:00Z",
|
||||
"instrument_routes": [
|
||||
{
|
||||
"route_revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"revision_number": 1,
|
||||
"symbol": "600000",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:11112222333344445555666677778888",
|
||||
"effective_from": "2020-01-01T00:00:00Z",
|
||||
"knowledge_time": "2026-01-01T07:00:00Z",
|
||||
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
}
|
||||
],
|
||||
"trading_calendar_revisions": [
|
||||
{
|
||||
"calendar_revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
|
||||
"calendar_id": "rhcalendar:11112222333344445555666677778888",
|
||||
"session_date": "2026-01-02",
|
||||
"revision_number": 1,
|
||||
"status": "open",
|
||||
"sessions": [
|
||||
{"opens_at": "2026-01-02T01:30:00Z", "closes_at": "2026-01-02T07:00:00Z"}
|
||||
],
|
||||
"knowledge_time": "2026-01-01T08:00:00Z",
|
||||
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
}
|
||||
],
|
||||
"corporate_action_revisions": [
|
||||
{
|
||||
"action_revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
|
||||
"action_id": "rhaction:99998888777766665555444433332222",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"revision_number": 1,
|
||||
"action_type": "cash_dividend",
|
||||
"status": "confirmed",
|
||||
"effective_time": "2026-01-02T00:00:00Z",
|
||||
"knowledge_time": "2026-01-01T09:00:00Z",
|
||||
"terms_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
|
||||
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
|
||||
}
|
||||
],
|
||||
"standardized_views": [
|
||||
{
|
||||
"view_ref_id": "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c",
|
||||
"view_id": "rhview:abcdef0123456789abcdef0123456789",
|
||||
"view_version": "1.0.0",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"pit_cutoff": "2026-01-03T00:00:00Z",
|
||||
"schema_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
|
||||
"content_digest": "sha256:123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef0",
|
||||
"transformation_digest": "sha256:23456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef01",
|
||||
"instrument_route_revision_ids": [
|
||||
"rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86"
|
||||
],
|
||||
"trading_calendar_revision_ids": [
|
||||
"rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
],
|
||||
"corporate_action_revision_ids": [
|
||||
"rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
]
|
||||
}
|
||||
],
|
||||
"revision_lineage": [
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T07:00:00Z",
|
||||
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
},
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T08:00:00Z",
|
||||
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
},
|
||||
{
|
||||
"revision_kind": "corporate_action",
|
||||
"revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T09:00:00Z",
|
||||
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
|
||||
}
|
||||
],
|
||||
"readiness": {
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"contract_validation": {
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee"
|
||||
]
|
||||
},
|
||||
"real_data_validation": {"status": "not_validated", "evidence_digests": []},
|
||||
"production_validation": {"status": "not_validated", "evidence_digests": []},
|
||||
"live_validation": {"status": "not_validated", "evidence_digests": []}
|
||||
}
|
||||
},
|
||||
"output_schema": {
|
||||
"columns": ["evaluation_at", "factor_id", "instrument_id", "value"],
|
||||
"schema_version": "1.0.0"
|
||||
},
|
||||
"output_content": {
|
||||
"rows": [
|
||||
{
|
||||
"evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"factor_id": "alpha_005",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"value": "0.125"
|
||||
}
|
||||
]
|
||||
},
|
||||
"expected": {
|
||||
"definition_id": "rhfactorv1:sha256:978fb8000d318373844a5e044ca14bf377e01ebe8d85964b826ecd2af9085ce9",
|
||||
"input_schema_digest": "sha256:4501aeab99b4bcc25a1b8813ebe197fb498053fd710d73746bf20fc8eeb4bfa7",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"output_artifact_id": "rhfactoroutputv1:sha256:a4803b5ff66d12d3a0e7e8e5b8cca953cbc137e2bf41514b7c4e5f05da5ee68b",
|
||||
"legacy_binding_id": "rhlegacyfactorv1:sha256:541bc5a9469f9c8e4c2d696a9972fc5f2e6e2bef218b86f728823994b915dede"
|
||||
}
|
||||
}
|
||||
+871
@@ -0,0 +1,871 @@
|
||||
{
|
||||
"cases": {
|
||||
"absent": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:b7e9139e021de3122bee9376a8c8387fca3db75fe2a698adccf33471caf971d2",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:fae26a93d754e98f437bf9e4b635fc1cdc4f85d004a4bde396823b52e2115de5",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:3f67fcad23c75684ffeb325d8405b2f81139a732e237d30fbb30d23f91e32726",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "none",
|
||||
"benchmark_id": "",
|
||||
"benchmark_series_digest": null,
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:729852af307fd4a94ac566ae45df2e57b3d45c4fbdf45820351b5ef19cdf0ad3",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "none",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:d735abe76c28db429e107cb5ba6a1929c2c0f2084a6c4ac41b38ffbc02b2d2bd",
|
||||
"performance_row_digest": "sha256:4e6329b0db4731513fe793491a163dd358e3a6f7baa84d3c94a4d4d2355c7a7d",
|
||||
"performance_table_content_digest": "sha256:074b8511ff9a2154235e4dde8bac300658cb69f900e49f332d61560b82be9af5",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:ad85ee1317bd8ce5fbef8fddc268b91be8678701ba5b8fbf6d61bf670f918bf9",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
},
|
||||
"estimable": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:e2e0e63fb4c733d2dba9a290511b6c8b0132bfe877dfe89ee7e614b74b87bc51",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "exact_session_index",
|
||||
"benchmark_id": "000300.SH",
|
||||
"benchmark_series_digest": "sha256:1c564e4d12af0d59b1fd707d7816da2895238969d9207470984db63cfeac423a",
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:a6a1a3313f511505c0cf57fddeebdf20a3b0f83a2958f31332d5597d172165e7",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "exact_session_index",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.3032171729991897
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": 22.801264912443322
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 123663320625.66454
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": 3.2500000000000013
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:f1f72c34d7427bd0d2e273a66b9677d112aa51c2417ea1cda8245e51e334e4f6",
|
||||
"performance_row_digest": "sha256:bad2b506679a6ca13d80ab00859a442ba692f3872d42b0b45bbc3b1d6ffe72a4",
|
||||
"performance_table_content_digest": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:6ec566d73da3966d3d5f946313e2b0181993374c9f4ceb4b56e165cc6793d382",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
},
|
||||
"zero_active_variance": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:8a13212575154d97f85011ed54ed3bb588f73a691926b3ca03b360d563e9fc29",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:8c0c2052ab93aa920254bc7f4d89435d839b7150ecaaaa798703e89ddf876ec1",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:699ad35ee30a294082c607c40617097aa90d5d58343e7d8c8185d8a53f52596c",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "exact_session_index",
|
||||
"benchmark_id": "000300.SH",
|
||||
"benchmark_series_digest": "sha256:7668b77ac689cd542005fae42cb2abd48fdf772d2f982086dca409606653e6ac",
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:44a04be0105f29e1cbf8d2430b6b0eb084853d428bc2ccd5834dfdbcf9482bd7",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "exact_session_index",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "not_estimable_active_variance",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": 1.0000000000000002
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:e6730d25d90c85a6370adb72bc45c1bf505235f77b72521dd8197898ca369fe5",
|
||||
"performance_row_digest": "sha256:6528c47fa9416e350aa5010477dbd8c83a35cecf87aa5bc413cec95a9765114d",
|
||||
"performance_table_content_digest": "sha256:7d57a966431c68093a9f6193ae62f79a63d33832f11dd68524ed9bcd2cb6257c",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:b17ab9e158a7ceeb5107f6fe8a61f332009c314186f3436d4750aa44d6ca3856",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
},
|
||||
"zero_benchmark_variance": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:2389b804f5d8d20a37b31f596b52892484777c5da799dd68865a082ee90e6e5b",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:54bb315bc1940ef6df79999cf03ebb8d10defaf526bf7802d63da33f612520ec",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:ad9559e1f8feca28bb310765216a975935f05f51057139c2d02fd2fe7dfe501e",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "exact_session_index",
|
||||
"benchmark_id": "000300.SH",
|
||||
"benchmark_series_digest": "sha256:d0f1e954d796e95254bcccc893d6e3a8f9d541c3028b88b2c0b24c1658bb2408",
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:077f3ec84802aff2f046365a3a5f61f475b1ad63ec05e96742844bc531bcb551",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "exact_session_index",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": 22.299903907544408
|
||||
},
|
||||
{
|
||||
"availability": "not_estimable_benchmark_variance",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "not_estimable_benchmark_variance",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:86133a6b3c3091477cdc4618ebc294e671c9ba0a3ef5015a068f92bb575729b9",
|
||||
"performance_row_digest": "sha256:90c9aa2f4168c15ff9ac5d4da2a35c3e915bb4e04736e5a440ea80c787a99553",
|
||||
"performance_table_content_digest": "sha256:1570b64d83ad2a849607e6f5203f5ec2f0291a8d6cbb00b8edfe0ae030b7967d",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:fed7a28888a6e98ccce78e7d84f22d4e3636e928485861af7a97f90eddc91f94",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
}
|
||||
},
|
||||
"schema_version": 1,
|
||||
"source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c",
|
||||
"source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603"
|
||||
}
|
||||
@@ -0,0 +1,129 @@
|
||||
{
|
||||
"portfolio_decision": {
|
||||
"computed_at": "2026-01-08T03:01:00Z",
|
||||
"constraint_residuals": {
|
||||
"gross_exposure_max": 0.0,
|
||||
"net_exposure_max": 0.0,
|
||||
"net_exposure_min": 0.0,
|
||||
"position_count_max": 0.0,
|
||||
"single_asset_max": 0.0,
|
||||
"single_asset_min": 0.0,
|
||||
"turnover_max": 0.0
|
||||
},
|
||||
"constraints": {
|
||||
"gross_exposure_max": 1.0,
|
||||
"net_exposure_max": 1.0,
|
||||
"net_exposure_min": 1.0,
|
||||
"position_count_max": 2,
|
||||
"schema_version": "1.0.0",
|
||||
"single_asset_max": 0.7,
|
||||
"single_asset_min": 0.2,
|
||||
"turnover_max": 0.2
|
||||
},
|
||||
"contract_name": "researchhub.portfolio-decision",
|
||||
"covariance_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
|
||||
"effective_at": "2026-01-08T03:00:00Z",
|
||||
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"expected_return_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd",
|
||||
"freshness_policy": {
|
||||
"max_covariance_age_days": 0,
|
||||
"max_manifest_age_seconds": 3600,
|
||||
"schema_version": "1.0.0"
|
||||
},
|
||||
"gross_exposure": 1.0,
|
||||
"manifest_document_sha256": "fbf54218f770528978f9ccd35577e1ab00877143ea397a576e071e75f0afbab0",
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
|
||||
"model_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
|
||||
"model_name": "deterministic_weights",
|
||||
"model_version": "1.0.0",
|
||||
"net_exposure": 1.0,
|
||||
"objective_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
|
||||
"objective_name": "long_only_allocation",
|
||||
"objective_version": "1.0.0",
|
||||
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
|
||||
"portfolio_asset_set_digest": "sha256:b64e3448a83a5b86466465080361c1a7e1157a27ddccd4b68069cb18caffb74a",
|
||||
"position_count": 2,
|
||||
"prior_weights": {
|
||||
"A": 0.5,
|
||||
"B": 0.5
|
||||
},
|
||||
"receipt": {
|
||||
"algorithm": "bounded_allocation",
|
||||
"algorithm_version": "1.0.0",
|
||||
"computed_at": "2026-01-08T03:01:00Z",
|
||||
"constraint_digest": "sha256:34df0e5c00f503748ff936f9cd415a8947169181f8ca0917bce97dd606e08e94",
|
||||
"implementation_digest": "sha256:ffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffff",
|
||||
"input_digest": "sha256:ebd8ff957115e1adfd84eafa5cea49470356194d747b64ee5897858f9dd067b7",
|
||||
"iterations": null,
|
||||
"max_constraint_residual": 0.0,
|
||||
"objective_value": null,
|
||||
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
|
||||
"parameter_digest": "sha256:0000000000000000000000000000000000000000000000000000000000000000",
|
||||
"schema_version": "1.0.0",
|
||||
"solver_config_digest": null,
|
||||
"solver_name": null,
|
||||
"solver_required": false,
|
||||
"solver_version": null,
|
||||
"status": "completed",
|
||||
"tolerance": 1e-12
|
||||
},
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"run_ref_document_sha256": "6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
|
||||
"schema_version": "1.0.0",
|
||||
"source_universe_digest": "sha256:5555555555555555555555555555555555555555555555555555555555555555",
|
||||
"target_id": "portfolio-target:synthetic-v1",
|
||||
"target_weights": {
|
||||
"A": 0.6,
|
||||
"B": 0.4
|
||||
},
|
||||
"turnover_l1": 0.19999999999999996
|
||||
},
|
||||
"risk_assessment": {
|
||||
"assessment_id": "rhriskassessmentv1:sha256:dbc38825cffcf6d95bd0216d22dbba4e0d4a1359ca5924a3ee529b99a7d78b6d",
|
||||
"component_risk": {
|
||||
"A": 1.4549226783578566,
|
||||
"B": 1.4549226783578568
|
||||
},
|
||||
"contract_name": "researchhub.risk-assessment",
|
||||
"covariance_as_of_date": "2026-01-08",
|
||||
"covariance_data_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"covariance_input_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
|
||||
"covariance_snapshot_id": "covariance:synthetic-v1",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
|
||||
"findings": [],
|
||||
"freshness_policy_digest": "sha256:833f58f4d1056f0450f7369fd4edd70534cc74e9e55cd60f05b2b3d7a2979763",
|
||||
"group_exposure": {
|
||||
"equity": 1.4549226783578566,
|
||||
"fixed_income": 1.4549226783578568
|
||||
},
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
|
||||
"marginal_risk": {
|
||||
"A": 2.424871130596428,
|
||||
"B": 3.637306695894642
|
||||
},
|
||||
"percentage_risk": {
|
||||
"A": 0.49999999999999983,
|
||||
"B": 0.49999999999999994
|
||||
},
|
||||
"periods_per_year": 252,
|
||||
"portfolio_volatility": 2.909845356715714,
|
||||
"portfolio_volatility_limit": 10.0,
|
||||
"qualified": true,
|
||||
"return_frequency": "1d",
|
||||
"risk_budget": {
|
||||
"A": 0.8,
|
||||
"B": 0.8
|
||||
},
|
||||
"risk_model_digest": "sha256:2222222222222222222222222222222222222222222222222222222222222222",
|
||||
"risk_model_name": "euler_volatility",
|
||||
"risk_model_version": "1.0.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
|
||||
"schema_version": "1.0.0",
|
||||
"status": "ready"
|
||||
}
|
||||
}
|
||||
+1215
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,175 @@
|
||||
{
|
||||
"contract_name": "researchhub.data-foundation",
|
||||
"schema_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"published_at": "2026-09-08T01:05:00Z",
|
||||
"instrument_routes": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"symbol": "SIM0",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb",
|
||||
"route_revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9"
|
||||
},
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"symbol": "SIM1",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799",
|
||||
"route_revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
}
|
||||
],
|
||||
"trading_calendar_revisions": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"session_date": "2018-01-02",
|
||||
"status": "open",
|
||||
"sessions": [
|
||||
{
|
||||
"opens_at": "2018-01-02T01:30:00Z",
|
||||
"closes_at": "2018-01-02T07:00:00Z"
|
||||
}
|
||||
],
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a",
|
||||
"calendar_revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
}
|
||||
],
|
||||
"corporate_action_revisions": [],
|
||||
"standardized_views": [
|
||||
{
|
||||
"view_id": "rhview:66666666666666666666666666666666",
|
||||
"view_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"transformation_digest": "sha256:b2376dbe4424b38e5b27874c998f082aecc086179f611896ebaa0fdeb5362c68",
|
||||
"instrument_route_revision_ids": [
|
||||
"rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
],
|
||||
"trading_calendar_revision_ids": [
|
||||
"rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
],
|
||||
"corporate_action_revision_ids": [],
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"available_at": "2026-09-08T01:04:00Z",
|
||||
"view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309"
|
||||
}
|
||||
],
|
||||
"observation_lineage": [
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb"
|
||||
},
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799"
|
||||
},
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a"
|
||||
}
|
||||
],
|
||||
"corporate_action_coverage": [
|
||||
{
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:233cc2968985e953e59e408b6619a837a792ae17f8afba32261f861de8fe9c7c"
|
||||
]
|
||||
},
|
||||
{
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:21f2594fc42e46168a65d7cac7e9b52aeabf24ecc5fb409530c07ad56ed3986c"
|
||||
]
|
||||
}
|
||||
],
|
||||
"readiness": {
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"contract_validation": {
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:62f2abbbb79f4ff9164e10cc4b21df37f0964224a4c546f978b97794241ff5b3"
|
||||
]
|
||||
},
|
||||
"real_data_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"production_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"live_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
}
|
||||
},
|
||||
"foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260"
|
||||
}
|
||||
@@ -0,0 +1,120 @@
|
||||
{
|
||||
"contract_name": "researchhub.dataset-snapshot",
|
||||
"schema_version": "2.0.0",
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"descriptor": {
|
||||
"dataset": {
|
||||
"dataset_id": "rhdataset:market:44444444444444444444444444444444",
|
||||
"dataset_kind": "market",
|
||||
"record_schema_version": "2.0.0",
|
||||
"dimensions": [
|
||||
"instrument_id",
|
||||
"effective_time"
|
||||
]
|
||||
},
|
||||
"published_at": "2026-09-08T01:03:00Z",
|
||||
"time_semantics": {
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"historical_availability": "not_established"
|
||||
},
|
||||
"content": {
|
||||
"digest_algorithm": "sha256",
|
||||
"canonicalization": "RFC8785",
|
||||
"record_order": "canonical-record-byte-order",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"logical_manifest": {
|
||||
"record_count": 2,
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2
|
||||
}
|
||||
]
|
||||
},
|
||||
"manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7",
|
||||
"record_count": 2
|
||||
},
|
||||
"observation_manifest": {
|
||||
"batches": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": "sha256:6f35f19a2ede0610d461b458a0f2cbc96f40cee6e90fa6592fd66daf617d3ec0"
|
||||
}
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"publisher": {
|
||||
"id": "researchhub.data",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"transformation": {
|
||||
"id": "rhtransform:55555555555555555555555555555555",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"upstream_snapshot_ids": [],
|
||||
"upstream_content_digests": []
|
||||
},
|
||||
"quality": {
|
||||
"status": "passed",
|
||||
"checks": [
|
||||
{
|
||||
"check_id": "completeness",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:519b01c60ad85c2b03d6c4f29027fe9b3672a8f6ae3ca8a2d831b281a61c5095"
|
||||
},
|
||||
{
|
||||
"check_id": "duplicate_identity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:7c585c7471ab45886fce037061b5ab0d5f981aede5d5190021963676b0ec0fcc"
|
||||
},
|
||||
{
|
||||
"check_id": "observation_coverage",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:dde5693c3ad79a01a162f9fc5a40c69d3c284965e283c91c5cd456904e0ef737"
|
||||
},
|
||||
{
|
||||
"check_id": "historical_claim_policy",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:f4cf8da48f92e03d75bb28b2569061e649f382ba360d879fef8e41e35169c228"
|
||||
},
|
||||
{
|
||||
"check_id": "range_validity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:6af08355c5e29d846e1c48783dccde791e8a3f48f8416149413217b7c2cbb52a"
|
||||
},
|
||||
{
|
||||
"check_id": "schema_conformance",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:fefdc5b81eba128af92f15a60feb7bb6f40b2e7277bf2beb48c1bb9399ee564b"
|
||||
}
|
||||
]
|
||||
},
|
||||
"qualification": {
|
||||
"status": "qualified",
|
||||
"usage": "retrospective_research",
|
||||
"policy_id": "researchhub.dataset-snapshot.retrospective",
|
||||
"policy_version": "2.0.0",
|
||||
"evaluated_at": "2026-09-08T01:02:00Z",
|
||||
"evidence_digest": "sha256:70c3ddcadf095877f17a1365c8d75a2e5a7078a27722fb8b698e61d350c6bb2d"
|
||||
}
|
||||
},
|
||||
"snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905"
|
||||
}
|
||||
@@ -1,32 +1,103 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
class ModuleSpecTests(unittest.TestCase):
|
||||
def test_module_spec_declares_pure_research_engine_boundary(self) -> None:
|
||||
def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
spec = json.loads((ROOT / "MODULE_SPEC.yaml").read_text(encoding="utf-8"))
|
||||
self.assertEqual(spec["module_id"], "quant_engine")
|
||||
self.assertEqual(spec["authority"]["subject"], spec["module_id"])
|
||||
self.assertEqual(spec["repository"]["type"], "research_engine")
|
||||
self.assertEqual(spec["bounded_context"]["domain"], "quantitative-research-engine")
|
||||
assert spec["module_id"] == "quant_engine"
|
||||
assert spec["authority"]["subject"] == spec["module_id"]
|
||||
assert spec["repository"]["type"] == "research_engine"
|
||||
assert spec["bounded_context"]["domain"] == "quantitative-research-engine"
|
||||
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
|
||||
for term in ("investment advice", "live order", "credentials", "source facts"):
|
||||
self.assertIn(term, prohibited)
|
||||
self.assertEqual(spec["contracts"], {"provides": [], "consumes": []})
|
||||
self.assertEqual(spec["dependencies"], [])
|
||||
self.assertTrue(
|
||||
all(
|
||||
assert term in prohibited
|
||||
assert spec["authority"]["revision"] == 6
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["provides"]
|
||||
} == {
|
||||
("researchhub.factor-definition", "1.0.0"),
|
||||
("researchhub.factor-set-ref", "1.0.0"),
|
||||
("researchhub.backtest-run-ref", "1.0.0"),
|
||||
("researchhub.backtest-evidence-manifest", "1.0.0"),
|
||||
("researchhub.performance-evidence", "1.0.0"),
|
||||
("researchhub.portfolio-decision", "1.0.0"),
|
||||
("researchhub.risk-assessment", "1.0.0"),
|
||||
("researchhub.factor-set-ref", "2.0.0"),
|
||||
("researchhub.backtest-run-ref", "2.0.0"),
|
||||
("researchhub.backtest-evidence-manifest", "2.0.0"),
|
||||
("researchhub.performance-evidence", "2.0.0"),
|
||||
("researchhub.portfolio-target", "2.0.0"),
|
||||
("researchhub.portfolio-decision", "2.0.0"),
|
||||
("researchhub.risk-assessment", "2.0.0"),
|
||||
}
|
||||
expected_paths = {
|
||||
"researchhub.factor-definition": "src/quant_engine/factor_contracts.py",
|
||||
"researchhub.factor-set-ref": "src/quant_engine/factor_contracts.py",
|
||||
"researchhub.backtest-run-ref": "src/quant_engine/governed_pipeline.py",
|
||||
"researchhub.backtest-evidence-manifest": "src/quant_engine/artifact.py",
|
||||
"researchhub.performance-evidence": "src/quant_engine/artifact.py",
|
||||
"researchhub.portfolio-decision": "src/quant_engine/portfolio_risk_contracts.py",
|
||||
"researchhub.risk-assessment": "src/quant_engine/portfolio_risk_contracts.py",
|
||||
}
|
||||
assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"])
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "1.0.0"
|
||||
} == expected_paths
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "2.0.0"
|
||||
} == {
|
||||
"researchhub.factor-set-ref": "src/quant_engine/retrospective_factor_contracts.py",
|
||||
"researchhub.backtest-run-ref": "src/quant_engine/retrospective_backtest_contracts.py",
|
||||
"researchhub.backtest-evidence-manifest": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.performance-evidence": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.portfolio-target": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.portfolio-decision": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.risk-assessment": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
}
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["consumes"]
|
||||
} == {
|
||||
("researchhub.dataset-snapshot", "1.0.0"),
|
||||
("researchhub.data-foundation", "1.0.0"),
|
||||
("researchhub.dataset-snapshot", "2.0.0"),
|
||||
("researchhub.data-foundation", "2.0.0"),
|
||||
}
|
||||
assert all(
|
||||
item["authority"] == "researchhub.data"
|
||||
for item in spec["contracts"]["consumes"]
|
||||
)
|
||||
assert spec["dependencies"] == []
|
||||
capabilities = {item["id"]: item for item in spec["capabilities"]}
|
||||
evidence_contract = capabilities["backtest-evidence-contracts"]
|
||||
assert evidence_contract["status"] == "operational"
|
||||
evidence_summary = evidence_contract["summary"].lower()
|
||||
for term in ("performance-methodology", "without recomputation", "decision authority"):
|
||||
assert term in evidence_summary
|
||||
portfolio_contract = capabilities["portfolio-risk-computation-contracts"]
|
||||
assert portfolio_contract["status"] == "operational"
|
||||
summary = portfolio_contract["summary"].lower()
|
||||
for term in ("receipt", "portfolio decisions", "risk assessments", "without"):
|
||||
assert term in summary
|
||||
retrospective = capabilities["retrospective-computation-contracts"]
|
||||
assert retrospective["status"] == "operational"
|
||||
for term in ("two clocks", "retrospective", "no historical-availability", "no execution"):
|
||||
assert term in retrospective["summary"].lower()
|
||||
for term in ("approval", "maker-checker", "publication", "paper", "live"):
|
||||
assert term in prohibited
|
||||
assert all(
|
||||
command["required"] and not command["network"]
|
||||
for command in spec["verification"]["commands"]
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
test_module_spec_declares_pure_research_engine_boundary()
|
||||
|
||||
@@ -6,10 +6,30 @@ import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
import quant_engine.alpha_factors as alpha_factors_module
|
||||
from quant_engine.factor_contracts import (
|
||||
FactorContractError,
|
||||
FactorInput,
|
||||
ProducerIdentity,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.alpha_factors import (
|
||||
ALPHA158_REGISTRY,
|
||||
ALPHA158_PHASE1_OPERATOR_SPECS,
|
||||
ALPHA158_PHASE2_OPERATOR_SPECS,
|
||||
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE3_FORMULA_SPECS,
|
||||
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE4_FORMULA_SPECS,
|
||||
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE5_FORMULA_SPECS,
|
||||
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE6_FORMULA_SPECS,
|
||||
alpha_001,
|
||||
alpha_002,
|
||||
alpha_003,
|
||||
@@ -170,8 +190,16 @@ from quant_engine.alpha_factors import (
|
||||
alpha_158,
|
||||
evaluate_phase1_operator,
|
||||
evaluate_phase2_operator,
|
||||
evaluate_phase3_formula,
|
||||
evaluate_phase4_formula,
|
||||
evaluate_phase5_formula,
|
||||
evaluate_phase6_formula,
|
||||
list_phase1_operators,
|
||||
list_phase2_operators,
|
||||
list_phase3_formulas,
|
||||
list_phase4_formulas,
|
||||
list_phase5_formulas,
|
||||
list_phase6_formulas,
|
||||
correlation,
|
||||
covariance,
|
||||
decay_linear,
|
||||
@@ -413,6 +441,50 @@ def test_alpha_registry_required_fields():
|
||||
assert required <= set(meta.keys()), f"{alpha_id} missing fields"
|
||||
|
||||
|
||||
def test_alpha_registry_adapts_to_definition_without_copying_formula_or_inputs():
|
||||
factor_input = FactorInput(
|
||||
"market",
|
||||
"sha256:" + "1" * 64,
|
||||
tuple(ALPHA158_REGISTRY["alpha_005"]["inputs"]),
|
||||
)
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "2" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
assert definition.formula == ALPHA158_REGISTRY["alpha_005"]["formula"]
|
||||
assert definition.inputs[0].required_columns == tuple(
|
||||
ALPHA158_REGISTRY["alpha_005"]["inputs"]
|
||||
)
|
||||
|
||||
incomplete = FactorInput("market", "sha256:" + "1" * 64, ("close",))
|
||||
with pytest.raises(FactorContractError, match="exactly correspond"):
|
||||
factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(incomplete,),
|
||||
implementation_digest="sha256:" + "2" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((incomplete,)),
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
|
||||
def test_get_alpha_meta_success():
|
||||
"""已知 alpha_id 返回完整 meta。"""
|
||||
meta = get_alpha_meta("alpha_001")
|
||||
@@ -1463,3 +1535,676 @@ def test_phase2_dispatch_validates_primary_series_and_unused_parameters():
|
||||
evaluate_phase2_operator("rank", values, groups=pd.Series(["x", "x", "x"]))
|
||||
with pytest.raises(ValueError, match="maximum"):
|
||||
evaluate_phase2_operator("product", values, window=253)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 3: versioned alpha001-alpha050 formula contract ──────────
|
||||
|
||||
|
||||
def _phase3_market_inputs() -> dict[str, pd.Series]:
|
||||
positions = np.arange(80, dtype=float)
|
||||
index = pd.RangeIndex(len(positions), name="row")
|
||||
open_ = pd.Series(100.0 + positions * 0.2 + np.sin(positions / 4.0), index=index)
|
||||
close = pd.Series(100.5 + positions * 0.18 + np.cos(positions / 5.0), index=index)
|
||||
high = pd.Series(np.maximum(open_, close) + 1.0, index=index)
|
||||
low = pd.Series(np.minimum(open_, close) - 1.0, index=index)
|
||||
volume = pd.Series(1_000.0 + positions**1.3 + 20.0 * np.sin(positions / 3.0), index=index)
|
||||
vwap = (open_ + close + high + low) / 4.0
|
||||
return {
|
||||
"open": open_,
|
||||
"close": close,
|
||||
"high": high,
|
||||
"low": low,
|
||||
"volume": volume,
|
||||
"vwap": vwap,
|
||||
}
|
||||
|
||||
|
||||
def test_phase3_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(1, 51))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase3_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE3_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE3_FORMULA_SPECS.values()
|
||||
) == {"single": 19, "pair": 26, "triple": 4, "quadruple": 1}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE3_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 == (
|
||||
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE3_FORMULA_SPECS, "alpha_001", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_catalog_freezes_callable_signatures_without_rewriting_formulas():
|
||||
import inspect
|
||||
|
||||
legacy_formula_input_differences = {
|
||||
"alpha_011": ("close", "high", "low"),
|
||||
"alpha_035": ("volume",),
|
||||
"alpha_036": ("close",),
|
||||
"alpha_040": ("high", "low"),
|
||||
"alpha_042": ("close",),
|
||||
"alpha_043": ("volume",),
|
||||
}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == legacy_formula_input_differences.get(
|
||||
alpha_id,
|
||||
signature_inputs,
|
||||
)
|
||||
|
||||
assert ALPHA158_PHASE3_FORMULA_SPECS["alpha_035"]["call_inputs"] == (
|
||||
"close",
|
||||
"volume",
|
||||
)
|
||||
assert ALPHA158_REGISTRY["alpha_035"]["inputs"] == ["volume"]
|
||||
|
||||
|
||||
def test_phase3_dispatch_matches_all_existing_alpha001_alpha050_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase3_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase3_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase3_formula("alpha_051", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
|
||||
evaluate_phase3_formula("alpha_005", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase3_formula(
|
||||
"alpha_005",
|
||||
close=inputs["close"],
|
||||
volume=inputs["volume"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="close must be a pandas Series"):
|
||||
evaluate_phase3_formula( # type: ignore[arg-type]
|
||||
"alpha_005",
|
||||
close=[1.0, 2.0],
|
||||
volume=inputs["volume"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_dispatch_rejects_implicit_series_alignment():
|
||||
inputs = _phase3_market_inputs()
|
||||
misaligned_volume = inputs["volume"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="volume index must align with close"):
|
||||
evaluate_phase3_formula(
|
||||
"alpha_005",
|
||||
close=inputs["close"],
|
||||
volume=misaligned_volume,
|
||||
)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 4: versioned alpha051-alpha100 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase4_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(51, 101))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase4_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE4_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE4_FORMULA_SPECS.values()
|
||||
) == {"single": 1, "pair": 27, "triple": 11, "quadruple": 10, "quintuple": 1}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE4_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 == (
|
||||
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE4_FORMULA_SPECS, "alpha_051", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE4_FORMULA_SPECS["alpha_055"]["call_inputs"] == (
|
||||
"open",
|
||||
"high",
|
||||
"low",
|
||||
"volume",
|
||||
"close",
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_dispatch_matches_all_existing_alpha051_alpha100_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase4_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase4_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase4_formula("alpha_050", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*low"):
|
||||
evaluate_phase4_formula("alpha_051", high=inputs["high"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase4_formula(
|
||||
"alpha_051",
|
||||
high=inputs["high"],
|
||||
low=inputs["low"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="high must be a pandas Series"):
|
||||
evaluate_phase4_formula( # type: ignore[arg-type]
|
||||
"alpha_051",
|
||||
high=[1.0, 2.0],
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_dispatch_rejects_implicit_series_alignment():
|
||||
inputs = _phase3_market_inputs()
|
||||
misaligned_low = inputs["low"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="low index must align with high"):
|
||||
evaluate_phase4_formula(
|
||||
"alpha_051",
|
||||
high=inputs["high"],
|
||||
low=misaligned_low,
|
||||
)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 5: versioned alpha101-alpha150 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase5_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(101, 151))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase5_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE5_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE5_FORMULA_SPECS.values()
|
||||
) == {"pair": 33, "triple": 14, "quadruple": 3}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE5_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 == (
|
||||
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE5_FORMULA_SPECS, "alpha_101", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"] == (
|
||||
"close",
|
||||
"high",
|
||||
"low",
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_dispatch_matches_all_existing_alpha101_alpha150_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase5_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase5_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase5_formula("alpha_100", close=inputs["close"])
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase5_formula("alpha_151", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*low"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
low=inputs["low"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="high must be a pandas Series"):
|
||||
evaluate_phase5_formula( # type: ignore[arg-type]
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=[1.0, 2.0],
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_dispatch_rejects_length_and_index_alignment_errors():
|
||||
inputs = _phase3_market_inputs()
|
||||
shorter_low = inputs["low"].iloc[:-1]
|
||||
misaligned_high = inputs["high"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="low length must match close"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
low=shorter_low,
|
||||
)
|
||||
with pytest.raises(ValueError, match="high index must align with close"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=misaligned_high,
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_contract_is_publicly_exported():
|
||||
assert {
|
||||
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE5_FORMULA_SPECS",
|
||||
"list_phase5_formulas",
|
||||
"evaluate_phase5_formula",
|
||||
} <= set(alpha_factors_module.__all__)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 6: versioned alpha151-alpha158 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase6_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(151, 159))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase6_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE6_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE6_FORMULA_SPECS.values()
|
||||
) == {"pair": 6, "triple": 2}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE6_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 == (
|
||||
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE6_FORMULA_SPECS, "alpha_151", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE6_FORMULA_SPECS["alpha_158"]["call_inputs"] == (
|
||||
"high",
|
||||
"low",
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_dispatch_matches_all_existing_alpha151_alpha158_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase6_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase6_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase6_formula("alpha_150", close=inputs["close"])
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase6_formula("alpha_159", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
|
||||
evaluate_phase6_formula("alpha_151", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=inputs["volume"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="volume must be a pandas Series"):
|
||||
evaluate_phase6_formula( # type: ignore[arg-type]
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=[1.0, 2.0],
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_dispatch_rejects_length_and_index_alignment_errors():
|
||||
inputs = _phase3_market_inputs()
|
||||
shorter_volume = inputs["volume"].iloc[:-1]
|
||||
misaligned_low = inputs["low"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="volume length must match close"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=shorter_volume,
|
||||
)
|
||||
with pytest.raises(ValueError, match="low index must align with high"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_158",
|
||||
high=inputs["high"],
|
||||
low=misaligned_low,
|
||||
volume=inputs["volume"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_preserves_complete_alpha001_alpha158_formula_body_fingerprint():
|
||||
import ast
|
||||
import hashlib
|
||||
import inspect
|
||||
import json
|
||||
import textwrap
|
||||
|
||||
fingerprints = {}
|
||||
for number in range(1, 159):
|
||||
alpha_id = f"alpha_{number:03d}"
|
||||
source = textwrap.dedent(inspect.getsource(getattr(alpha_factors_module, alpha_id)))
|
||||
node = ast.parse(source).body[0]
|
||||
assert isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef))
|
||||
body = ast.dump(
|
||||
ast.Module(body=node.body, type_ignores=[]),
|
||||
include_attributes=False,
|
||||
)
|
||||
fingerprints[alpha_id] = hashlib.sha256(body.encode()).hexdigest()
|
||||
|
||||
encoded = json.dumps(
|
||||
fingerprints,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == (
|
||||
"f3ae807983cf8ca083d0b924c3807ffd84a62a5bd354f9efb1a1a16ca40c7da6"
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_contract_is_publicly_exported():
|
||||
assert {
|
||||
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE6_FORMULA_SPECS",
|
||||
"list_phase6_formulas",
|
||||
"evaluate_phase6_formula",
|
||||
} <= set(alpha_factors_module.__all__)
|
||||
|
||||
@@ -0,0 +1,773 @@
|
||||
"""Backtest run-reference and closed-evidence contract conformance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
BacktestEvidenceManifest,
|
||||
EvidenceQualification,
|
||||
RESEARCH_ARTIFACT_SCHEMA_VERSION,
|
||||
ResearchRunArtifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_legacy_backtest_evidence_manifest,
|
||||
build_research_run_artifact,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
InputBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
ProducerIdentity,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.governed_pipeline import (
|
||||
BacktestContractError,
|
||||
BacktestContractErrorCode,
|
||||
BacktestRun,
|
||||
BacktestRunRef,
|
||||
)
|
||||
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
BACKTEST_FIXTURE = ROOT / "tests" / "fixtures" / "backtest-evidence-v1.golden.json"
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _accepted_authorities(
|
||||
*,
|
||||
factor_evaluation_at: str = "2026-01-03T11:00:00Z",
|
||||
factor_computed_at: str = "2026-01-03T10:15:00Z",
|
||||
factor_artifact_available_at: str = "2026-01-03T10:20:00Z",
|
||||
factor_availability_mode: AvailabilityMode = AvailabilityMode.AS_AVAILABLE,
|
||||
) -> tuple[
|
||||
DatasetSnapshotEnvelope,
|
||||
DataFoundationEnvelope,
|
||||
FactorSetRef,
|
||||
]:
|
||||
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
|
||||
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
artifact_ref = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
selected_view_ref_ids=(VIEW_REF_ID,),
|
||||
input_bindings=(
|
||||
InputBinding(definition.definition_id, "market", VIEW_REF_ID, VIEW_SCHEMA_DIGEST),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
output_quality=OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
output_coverage=OutputCoverage(
|
||||
"complete", 1, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64
|
||||
),
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
output_artifact_ref=artifact_ref,
|
||||
availability_mode=factor_availability_mode,
|
||||
evaluation_at=factor_evaluation_at,
|
||||
computed_at=factor_computed_at,
|
||||
artifact_available_at=factor_artifact_available_at,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
actor=ActorIdentity("service", "factor_worker_v1"),
|
||||
correlation_id="research_run_001",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
evidence_scope="synthetic_fixture",
|
||||
decision_eligible=False,
|
||||
)
|
||||
return snapshot, foundation, factor_set
|
||||
|
||||
|
||||
def _config_digest(parameters: dict[str, object] | None = None) -> str:
|
||||
encoded = json.dumps(
|
||||
PARAMETERS if parameters is None else parameters,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
return _sha256(encoded)
|
||||
|
||||
|
||||
def _run_ref(**overrides: Any) -> BacktestRunRef:
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
arguments: dict[str, Any] = {
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"factor_set": factor_set,
|
||||
"universe_digest": "sha256:" + "5" * 64,
|
||||
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
|
||||
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": "sha256:" + "6" * 64,
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": "sha256:" + "7" * 64,
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": "sha256:" + "8" * 64,
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": "sha256:" + "9" * 64,
|
||||
"configuration_digest": _config_digest(),
|
||||
"evaluation_at": "2026-01-08T01:00:00Z",
|
||||
"computed_at": "2026-01-08T02:00:00Z",
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return BacktestRunRef.create(**arguments)
|
||||
|
||||
|
||||
def _backtest_result() -> FactorBacktestResult:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
return run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _artifact(run_ref: BacktestRunRef, *, run_id: str | None = None) -> ResearchRunArtifact:
|
||||
result = _backtest_result()
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id=run_ref.run_id if run_id is None else run_id,
|
||||
strategy_id=run_ref.strategy_id,
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version=run_ref.strategy_version,
|
||||
engine_version="1.2.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
data_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=PARAMETERS,
|
||||
benchmark_id="000300.SH",
|
||||
benchmark_returns=benchmark,
|
||||
)
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[BacktestContractError],
|
||||
code: BacktestContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_backtest_run_ref_is_deterministic_and_binds_only_opaque_authorities() -> None:
|
||||
first = _run_ref()
|
||||
second = _run_ref()
|
||||
|
||||
assert first == second
|
||||
assert first.run_id.startswith("rhbacktestrunv1:sha256:")
|
||||
assert first.replay_spec_digest.startswith("sha256:")
|
||||
assert first.dataset_snapshot_id.startswith("rhdsv1:sha256:")
|
||||
assert first.foundation_id.startswith("rhdfv1:sha256:")
|
||||
assert first.factor_set_id.startswith("rhfactorsetv1:sha256:")
|
||||
assert first.trading_calendar_revision_ids == (CALENDAR_REVISION_ID,)
|
||||
assert first.corporate_action_revision_ids == (ACTION_REVISION_ID,)
|
||||
assert first.replay_parent_run_id is None
|
||||
assert first.replay_attempt == 0
|
||||
assert first.replay_ancestor_run_ids == ()
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
assert BacktestRunRef.from_dict(
|
||||
first.to_dict(),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
) == first
|
||||
forbidden = ("latest", "locator", "uri", "credential", "provider", "broker")
|
||||
assert not any(token in first.to_json().lower() for token in forbidden)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "value"),
|
||||
[
|
||||
("universe_digest", "sha256:" + "a" * 64),
|
||||
("strategy_digest", "sha256:" + "b" * 64),
|
||||
("execution_model_digest", "sha256:" + "c" * 64),
|
||||
("cost_model_digest", "sha256:" + "e" * 64),
|
||||
("random_seed", 8),
|
||||
("code_revision", "e" * 40),
|
||||
("environment_lock_digest", "sha256:" + "f" * 64),
|
||||
("configuration_digest", "sha256:" + "0" * 64),
|
||||
("evaluation_at", "2026-01-08T01:00:01Z"),
|
||||
("computed_at", "2026-01-08T02:00:01Z"),
|
||||
],
|
||||
)
|
||||
def test_every_governed_run_input_mutation_changes_run_identity(
|
||||
field: str,
|
||||
value: object,
|
||||
) -> None:
|
||||
assert _run_ref(**{field: value}).run_id != _run_ref().run_id
|
||||
|
||||
|
||||
def test_backtest_run_ref_rejects_unclosed_upstream_and_unsafe_scalars() -> None:
|
||||
with pytest.raises(BacktestContractError) as wrong_calendar:
|
||||
_run_ref(trading_calendar_revision_ids=())
|
||||
_assert_error(
|
||||
wrong_calendar,
|
||||
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.trading_calendar_revision_ids",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as wrong_action:
|
||||
_run_ref(corporate_action_revision_ids=())
|
||||
_assert_error(
|
||||
wrong_action,
|
||||
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.corporate_action_revision_ids",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as bool_seed:
|
||||
_run_ref(random_seed=True)
|
||||
_assert_error(bool_seed, BacktestContractErrorCode.TYPE_ERROR, "$.random_seed")
|
||||
with pytest.raises(BacktestContractError) as bad_revision:
|
||||
_run_ref(code_revision="abc")
|
||||
_assert_error(bad_revision, BacktestContractErrorCode.INVALID_FORMAT, "$.code_revision")
|
||||
with pytest.raises(BacktestContractError) as bad_digest:
|
||||
_run_ref(universe_digest="5" * 64)
|
||||
_assert_error(bad_digest, BacktestContractErrorCode.INVALID_FORMAT, "$.universe_digest")
|
||||
with pytest.raises(BacktestContractError) as lookahead:
|
||||
_run_ref(computed_at="2026-01-08T00:59:59Z")
|
||||
_assert_error(lookahead, BacktestContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
|
||||
with pytest.raises(BacktestContractError) as factor_type:
|
||||
_run_ref(factor_set="rhfactorsetv1:sha256:" + "0" * 64)
|
||||
_assert_error(factor_type, BacktestContractErrorCode.TYPE_ERROR, "$.factor_set")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("factor_times", "expected_path"),
|
||||
[
|
||||
(
|
||||
{
|
||||
"factor_evaluation_at": "2026-01-08T01:00:01Z",
|
||||
"factor_computed_at": "2026-01-08T00:59:59Z",
|
||||
"factor_artifact_available_at": "2026-01-08T01:00:00Z",
|
||||
},
|
||||
"$.evaluation_at",
|
||||
),
|
||||
(
|
||||
{
|
||||
"factor_evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"factor_computed_at": "2026-01-08T01:00:00Z",
|
||||
"factor_artifact_available_at": "2026-01-08T01:00:01Z",
|
||||
"factor_availability_mode": AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
},
|
||||
"$.evaluation_at",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_run_ref_evaluation_closes_factor_pit(
|
||||
factor_times: dict[str, Any],
|
||||
expected_path: str,
|
||||
) -> None:
|
||||
snapshot, foundation, factor_set = _accepted_authorities(**factor_times)
|
||||
|
||||
with pytest.raises(BacktestContractError) as lookahead:
|
||||
_run_ref(
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
)
|
||||
|
||||
_assert_error(
|
||||
lookahead,
|
||||
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
expected_path,
|
||||
)
|
||||
|
||||
|
||||
def test_run_ref_rejects_aliases_locators_unsafe_integers_and_invalid_text() -> None:
|
||||
with pytest.raises(BacktestContractError) as mutable_alias:
|
||||
_run_ref(strategy_id="latest")
|
||||
_assert_error(
|
||||
mutable_alias,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as physical_uri:
|
||||
_run_ref(execution_model_version="s3://model-bucket/current")
|
||||
_assert_error(
|
||||
physical_uri,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.execution_model_version",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as unsafe_seed:
|
||||
_run_ref(random_seed=2**53)
|
||||
_assert_error(
|
||||
unsafe_seed,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.random_seed",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as invalid_unicode:
|
||||
_run_ref(strategy_id="\ud800")
|
||||
_assert_error(
|
||||
invalid_unicode,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
|
||||
run_ref = _run_ref()
|
||||
mixed_keys = run_ref.to_dict()
|
||||
mixed_keys[1] = "not-a-contract-key" # type: ignore[index]
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
with pytest.raises(BacktestContractError) as invalid_key:
|
||||
BacktestRunRef.from_dict(
|
||||
mixed_keys,
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
)
|
||||
_assert_error(invalid_key, BacktestContractErrorCode.TYPE_ERROR, "$")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"physical_id",
|
||||
["db.table", "source_alpha", "wind.model", "qtdb_view", "bloomberg-signal"],
|
||||
)
|
||||
def test_run_ref_rejects_physical_terms_in_logical_ids(physical_id: str) -> None:
|
||||
with pytest.raises(BacktestContractError) as physical:
|
||||
_run_ref(strategy_id=physical_id)
|
||||
_assert_error(
|
||||
physical,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("version", ["1.0.0-.", "1.0.0-foo..bar", "1.0.0-01"])
|
||||
def test_run_ref_requires_strict_semver_prerelease_identifiers(version: str) -> None:
|
||||
with pytest.raises(BacktestContractError) as invalid:
|
||||
_run_ref(strategy_version=version)
|
||||
_assert_error(
|
||||
invalid,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_version",
|
||||
)
|
||||
|
||||
assert _run_ref(strategy_version="1.0.0-alpha.1").strategy_version == "1.0.0-alpha.1"
|
||||
|
||||
|
||||
def test_replay_lineage_is_acyclic_and_cannot_claim_changed_inputs() -> None:
|
||||
parent = _run_ref()
|
||||
replay = _run_ref(
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="deterministic_reproduction",
|
||||
replay_attempt=1,
|
||||
)
|
||||
|
||||
assert replay.run_id != parent.run_id
|
||||
assert replay.replay_spec_digest == parent.replay_spec_digest
|
||||
assert replay.replay_parent_run_id == parent.run_id
|
||||
assert replay.replay_ancestor_run_ids == (parent.run_id,)
|
||||
|
||||
with pytest.raises(BacktestContractError) as changed_input:
|
||||
_run_ref(
|
||||
universe_digest="sha256:" + "a" * 64,
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="changed_universe",
|
||||
replay_attempt=1,
|
||||
)
|
||||
_assert_error(
|
||||
changed_input,
|
||||
BacktestContractErrorCode.LINEAGE_VIOLATION,
|
||||
"$.replay_spec_digest",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as skipped_attempt:
|
||||
_run_ref(
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="skipped_attempt",
|
||||
replay_attempt=2,
|
||||
)
|
||||
_assert_error(
|
||||
skipped_attempt,
|
||||
BacktestContractErrorCode.LINEAGE_VIOLATION,
|
||||
"$.replay_attempt",
|
||||
)
|
||||
|
||||
|
||||
def test_offline_research_manifest_closes_exact_existing_evidence_mapping() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
first = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
second = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
|
||||
assert first == second
|
||||
assert first.manifest_id.startswith("rhbacktestevidencev1:sha256:")
|
||||
assert first.run_id == run_ref.run_id
|
||||
assert first.profile == "offline_research_v1"
|
||||
assert first.qualification is EvidenceQualification.CONTRACT_QUALIFIED
|
||||
mapping = {
|
||||
item.category: tuple(table.logical_name for table in item.tables)
|
||||
for item in first.evidence
|
||||
}
|
||||
assert mapping == {
|
||||
"run": ("run",),
|
||||
"signal": ("signals",),
|
||||
"fill": ("trades",),
|
||||
"position_nav": ("positions", "nav"),
|
||||
"performance": ("performance",),
|
||||
"attribution": ("attribution", "attribution_daily"),
|
||||
"risk_snapshot": ("risk",),
|
||||
"replay": (),
|
||||
}
|
||||
assert "order" not in mapping
|
||||
assert "rejection" not in mapping
|
||||
risk = next(item for item in first.evidence if item.category == "risk_snapshot")
|
||||
assert risk.tables[0].row_count == 0
|
||||
assert risk.tables[0].schema_digest.startswith("sha256:")
|
||||
|
||||
changed_performance = artifact.performance
|
||||
changed_performance.loc[0, "n_days"] += 1
|
||||
changed_artifact = replace(artifact, _performance=changed_performance)
|
||||
changed = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
assert changed.manifest_id != first.manifest_id
|
||||
assert run_ref.run_id == first.run_id == changed.run_id
|
||||
|
||||
|
||||
def test_manifest_rejects_missing_mismatched_or_duplicate_evidence() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
with pytest.raises(BacktestContractError) as wrong_run:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
_artifact(run_ref, run_id="different-run"),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
wrong_run,
|
||||
BacktestContractErrorCode.IDENTITY_MISMATCH,
|
||||
"$.artifact.tables.run.run_id",
|
||||
)
|
||||
missing_signals = replace(artifact, _signals=None) # type: ignore[arg-type]
|
||||
with pytest.raises(BacktestContractError) as missing_table:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
missing_signals,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
missing_table,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.artifact.tables.signals",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as digest_mismatch:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
expected_table_digests={"performance": "sha256:" + "0" * 64},
|
||||
)
|
||||
_assert_error(
|
||||
digest_mismatch,
|
||||
BacktestContractErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.artifact.tables.performance.content_digest",
|
||||
)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
duplicate = manifest.to_dict()
|
||||
duplicate["evidence"].append(copy.deepcopy(duplicate["evidence"][0]))
|
||||
with pytest.raises(BacktestContractError) as duplicate_category:
|
||||
BacktestEvidenceManifest.from_dict(
|
||||
duplicate,
|
||||
backtest_run_ref=run_ref,
|
||||
artifact=artifact,
|
||||
)
|
||||
_assert_error(
|
||||
duplicate_category,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.evidence[8].category",
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_binds_supported_schema_and_has_collision_free_cell_encoding() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
|
||||
with pytest.raises(BacktestContractError) as unsupported_schema:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, schema_version="999.0.0"),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsupported_schema,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.artifact.schema_version",
|
||||
)
|
||||
|
||||
identities: set[str] = set()
|
||||
for value in (float("nan"), float("inf"), float("-inf")):
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "alpha"] = value
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=performance),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
assert manifest.artifact_schema_version == RESEARCH_ARTIFACT_SCHEMA_VERSION
|
||||
identities.add(manifest.manifest_id)
|
||||
assert len(identities) == 3
|
||||
|
||||
content_digests: set[str] = set()
|
||||
for value in (float("nan"), {"non_finite_float": "nan"}):
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, "alpha"] = value
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=performance),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
performance_entry = next(
|
||||
entry for entry in manifest.evidence if entry.category == "performance"
|
||||
)
|
||||
content_digests.add(performance_entry.tables[0].content_digest)
|
||||
assert len(content_digests) == 2
|
||||
|
||||
unsupported = artifact.performance.astype(object)
|
||||
unsupported.loc[0, "alpha"] = object()
|
||||
with pytest.raises(BacktestContractError) as unsupported_cell:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=unsupported),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsupported_cell,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.artifact.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
invalid_nested_key = artifact.performance.astype(object)
|
||||
invalid_nested_key.at[0, "alpha"] = {"\ud800": "value"}
|
||||
with pytest.raises(BacktestContractError) as invalid_utf8:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=invalid_nested_key),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
invalid_utf8,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.artifact.tables.performance.rows[0].alpha.keys",
|
||||
)
|
||||
|
||||
unsafe_integer = artifact.performance.astype(object)
|
||||
unsafe_integer.loc[0, "alpha"] = 10**5000
|
||||
with pytest.raises(BacktestContractError) as unsafe_cell:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=unsafe_integer),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsafe_cell,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.artifact.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
|
||||
def test_artifact_canonical_content_has_typed_collision_free_cell_encoding() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
|
||||
content_hashes: set[str] = set()
|
||||
for value in (
|
||||
float("nan"),
|
||||
float("inf"),
|
||||
float("-inf"),
|
||||
{"non_finite_float": "nan"},
|
||||
):
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, "alpha"] = value
|
||||
mutated = replace(artifact, _performance=performance)
|
||||
content_hashes.add(mutated.content_sha256)
|
||||
assert "non_finite_float" in mutated.canonical_json()
|
||||
assert len(content_hashes) == 4
|
||||
|
||||
unsupported = artifact.performance.astype(object)
|
||||
unsupported.loc[0, "alpha"] = object()
|
||||
with pytest.raises(BacktestContractError) as unsupported_cell:
|
||||
replace(artifact, _performance=unsupported).canonical_json()
|
||||
_assert_error(
|
||||
unsupported_cell,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
|
||||
def test_legacy_bridge_is_explicit_and_cannot_be_contract_qualified() -> None:
|
||||
run_ref = _run_ref()
|
||||
legacy_run = BacktestRun(
|
||||
run_id="legacy-run-001",
|
||||
dataset_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
factor_version_id="alpha_005@1.0.0",
|
||||
strategy_version_id="alpha-top1@1.0.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
config_hash=_config_digest().removeprefix("sha256:"),
|
||||
created_at=datetime(2026, 1, 8, 2, 0, tzinfo=UTC),
|
||||
)
|
||||
artifact = _artifact(run_ref, run_id=legacy_run.run_id)
|
||||
manifest = build_legacy_backtest_evidence_manifest(
|
||||
legacy_run,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
|
||||
assert manifest.qualification is EvidenceQualification.LEGACY_EXPLORATORY
|
||||
assert manifest.run_id == legacy_run.run_id
|
||||
assert manifest.backtest_run_ref is None
|
||||
assert manifest.to_dict()["run_reference"]["kind"] == "legacy_backtest_run"
|
||||
assert BacktestEvidenceManifest.from_dict(
|
||||
manifest.to_dict(),
|
||||
artifact=artifact,
|
||||
) == manifest
|
||||
with pytest.raises(BacktestContractError) as implicit_promotion:
|
||||
build_backtest_evidence_manifest( # type: ignore[arg-type]
|
||||
legacy_run,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
_assert_error(
|
||||
implicit_promotion,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.backtest_run_ref",
|
||||
)
|
||||
|
||||
|
||||
def test_golden_contract_and_architecture_boundary() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
golden = json.loads(BACKTEST_FIXTURE.read_text(encoding="utf-8"))
|
||||
|
||||
table_digests = {
|
||||
table.logical_name: table.content_digest
|
||||
for item in manifest.evidence
|
||||
for table in item.tables
|
||||
}
|
||||
assert golden == {
|
||||
"run_id": run_ref.run_id,
|
||||
"replay_spec_digest": run_ref.replay_spec_digest,
|
||||
"manifest_id": manifest.manifest_id,
|
||||
"evidence_digest": manifest.evidence_digest,
|
||||
"table_content_digests": table_digests,
|
||||
}
|
||||
governed_source = (ROOT / "src" / "quant_engine" / "governed_pipeline.py").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
artifact_source = (ROOT / "src" / "quant_engine" / "artifact.py").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
assert "from quant_engine.artifact" not in governed_source
|
||||
assert "BacktestRunRef" in governed_source
|
||||
assert "BacktestEvidenceManifest" not in governed_source
|
||||
assert "BacktestEvidenceManifest" in artifact_source
|
||||
assert not (ROOT / "src" / "quant_engine" / "backtest_contracts.py").exists()
|
||||
@@ -0,0 +1,938 @@
|
||||
"""Versioned factor-definition and factor-set contract conformance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import FrozenInstanceError
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
ContractErrorCode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorContractError,
|
||||
FactorDefinition,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
HistoricalAvailability,
|
||||
InputBinding,
|
||||
LegacyFactorBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
TypedParameter,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
validate_factor_catalog,
|
||||
)
|
||||
from quant_engine.governed_pipeline import (
|
||||
FactorVersion,
|
||||
bind_legacy_factor,
|
||||
project_legacy_factor,
|
||||
)
|
||||
|
||||
|
||||
FIXTURE_PATH = Path(__file__).parent / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
|
||||
|
||||
def _golden() -> dict[str, Any]:
|
||||
loaded = json.loads(FIXTURE_PATH.read_text(encoding="utf-8"))
|
||||
assert isinstance(loaded, dict)
|
||||
return loaded
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _reidentify(item: dict[str, Any], field: str, prefix: str) -> None:
|
||||
payload = {key: value for key, value in item.items() if key != field}
|
||||
item[field] = f"{prefix}{hashlib.sha256(canonical_json_bytes(payload)).hexdigest()}"
|
||||
|
||||
|
||||
def _snapshot_and_foundation(
|
||||
fixture: dict[str, Any] | None = None,
|
||||
) -> tuple[DatasetSnapshotEnvelope, DataFoundationEnvelope]:
|
||||
source = _golden() if fixture is None else fixture
|
||||
return (
|
||||
DatasetSnapshotEnvelope.from_dict(source["dataset_snapshot"]),
|
||||
DataFoundationEnvelope.from_dict(source["data_foundation"]),
|
||||
)
|
||||
|
||||
|
||||
def _definition(
|
||||
*,
|
||||
inputs: tuple[FactorInput, ...] | None = None,
|
||||
**overrides: Any,
|
||||
) -> FactorDefinition:
|
||||
factor_inputs = inputs or (
|
||||
FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume")),
|
||||
)
|
||||
arguments: dict[str, Any] = {
|
||||
"factor_id": "alpha_005",
|
||||
"version": "1.0.0",
|
||||
"formula": "correlation(close, volume, 10)",
|
||||
"parameters": {},
|
||||
"implementation_digest": "sha256:" + "1" * 64,
|
||||
"input_schema_digest": factor_input_schema_digest(factor_inputs),
|
||||
"inputs": factor_inputs,
|
||||
"valid_from": "2026-01-01T00:00:00.000000Z",
|
||||
"valid_until": "2027-01-01T00:00:00Z",
|
||||
"warmup_sessions": 10,
|
||||
"lag_sessions": 1,
|
||||
"producer": ProducerIdentity("quant_engine", "1.0.0"),
|
||||
"code_revision": "c" * 40,
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return FactorDefinition.create(**arguments)
|
||||
|
||||
|
||||
def _golden_definition() -> FactorDefinition:
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
return factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
|
||||
def _factor_set_arguments(
|
||||
*,
|
||||
fixture: dict[str, Any] | None = None,
|
||||
snapshot: DatasetSnapshotEnvelope | None = None,
|
||||
foundation: DataFoundationEnvelope | None = None,
|
||||
definition: FactorDefinition | None = None,
|
||||
) -> dict[str, Any]:
|
||||
source = _golden() if fixture is None else fixture
|
||||
if snapshot is None or foundation is None:
|
||||
parsed_snapshot, parsed_foundation = _snapshot_and_foundation(source)
|
||||
snapshot = snapshot or parsed_snapshot
|
||||
foundation = foundation or parsed_foundation
|
||||
selected_definition = definition or _golden_definition()
|
||||
output_schema_bytes = canonical_json_bytes(source["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(source["output_content"])
|
||||
artifact = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
return {
|
||||
"definitions": (selected_definition,),
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"selected_view_ref_ids": (VIEW_REF_ID,),
|
||||
"input_bindings": (
|
||||
InputBinding(
|
||||
selected_definition.definition_id,
|
||||
"market",
|
||||
VIEW_REF_ID,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
"view_availability": (
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
"output_quality": OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
"output_coverage": OutputCoverage(
|
||||
"complete",
|
||||
1,
|
||||
1,
|
||||
"row",
|
||||
"alpha_005.cn_a",
|
||||
"sha256:" + "4" * 64,
|
||||
),
|
||||
"output_schema_bytes": output_schema_bytes,
|
||||
"output_content_bytes": output_content_bytes,
|
||||
"output_artifact_ref": artifact,
|
||||
"availability_mode": AvailabilityMode.AS_AVAILABLE,
|
||||
"evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"computed_at": "2026-01-03T10:15:00Z",
|
||||
"artifact_available_at": "2026-01-03T10:20:00Z",
|
||||
"producer": ProducerIdentity("quant_engine", "1.0.0"),
|
||||
"code_revision": "c" * 40,
|
||||
"actor": ActorIdentity("service", "factor_worker_v1"),
|
||||
"correlation_id": "research_run_001",
|
||||
"causation": Causation("foundation", foundation.foundation_id),
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def _factor_set(**overrides: Any) -> FactorSetRef:
|
||||
arguments = _factor_set_arguments()
|
||||
arguments.update(overrides)
|
||||
return FactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[FactorContractError],
|
||||
code: ContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def _mutate_artifact_schema_binding(value: dict[str, Any]) -> None:
|
||||
artifact = value["output_artifact_ref"]
|
||||
artifact["schema_digest"] = "sha256:" + "0" * 64
|
||||
_reidentify(artifact, "artifact_id", "rhfactoroutputv1:sha256:")
|
||||
|
||||
|
||||
def test_golden_contracts_are_content_addressed_round_trippable_and_deeply_immutable() -> None:
|
||||
fixture = _golden()
|
||||
original_snapshot = copy.deepcopy(fixture["dataset_snapshot"])
|
||||
original_foundation = copy.deepcopy(fixture["data_foundation"])
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
definition = _golden_definition()
|
||||
factor_set = FactorSetRef.create(**_factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition))
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id="factor:demo-momentum",
|
||||
legacy_version="1.0.0",
|
||||
legacy_definition_sha256="b" * 64,
|
||||
legacy_dataset_schema_version="1.0.0",
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
|
||||
assert snapshot.pit_cutoff == "2026-01-02T07:01:00Z"
|
||||
assert foundation.pit_cutoff == factor_set.pit_cutoff == "2026-01-03T00:00:00Z"
|
||||
assert snapshot.pit_cutoff != foundation.pit_cutoff
|
||||
assert definition.definition_id == fixture["expected"]["definition_id"]
|
||||
assert definition.input_schema_digest == fixture["expected"]["input_schema_digest"]
|
||||
assert factor_set.factor_set_id == fixture["expected"]["factor_set_id"]
|
||||
assert factor_set.output_artifact_ref.artifact_id == fixture["expected"]["output_artifact_id"]
|
||||
assert binding.binding_id == fixture["expected"]["legacy_binding_id"]
|
||||
assert not definition.to_json().endswith("\n")
|
||||
assert not factor_set.to_json().endswith("\n")
|
||||
assert FactorDefinition.from_json(definition.to_json()) == definition
|
||||
|
||||
reparsed = FactorSetRef.from_json(
|
||||
factor_set.to_json(),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
output_schema_bytes=canonical_json_bytes(fixture["output_schema"]),
|
||||
output_content_bytes=canonical_json_bytes(fixture["output_content"]),
|
||||
)
|
||||
reference_only = FactorSetRef.from_json(
|
||||
factor_set.to_json(),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
assert reparsed.factor_set_id == factor_set.factor_set_id
|
||||
assert reparsed.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
assert reference_only.payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
|
||||
fixture["dataset_snapshot"]["descriptor"]["dataset"]["dimensions"].append("forbidden")
|
||||
fixture["data_foundation"]["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
|
||||
assert snapshot.to_dict() == original_snapshot
|
||||
assert foundation.to_dict() == original_foundation
|
||||
returned = snapshot.to_dict()
|
||||
returned["descriptor"]["dataset"]["dimensions"].append("also_forbidden")
|
||||
assert snapshot.to_dict() == original_snapshot
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
snapshot.snapshot_id = "rhdsv1:sha256:" + "0" * 64 # type: ignore[misc]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("variant", ["whitespace", "key_order"])
|
||||
def test_contract_decoders_reject_non_canonical_json(variant: str) -> None:
|
||||
fixture = _golden()
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
definition = _golden_definition()
|
||||
factor_set = FactorSetRef.create(
|
||||
**_factor_set_arguments(
|
||||
fixture=fixture,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
definition=definition,
|
||||
)
|
||||
)
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id="factor:demo-momentum",
|
||||
legacy_version="1.0.0",
|
||||
legacy_definition_sha256="b" * 64,
|
||||
legacy_dataset_schema_version="1.0.0",
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
|
||||
def non_canonical(value: str) -> str:
|
||||
if variant == "whitespace":
|
||||
return value + "\n"
|
||||
loaded = json.loads(value)
|
||||
reversed_items = dict(reversed(tuple(loaded.items())))
|
||||
return json.dumps(reversed_items, ensure_ascii=False, separators=(",", ":"))
|
||||
|
||||
decoders = (
|
||||
lambda value: FactorDefinition.from_json(value),
|
||||
lambda value: FactorSetRef.from_json(
|
||||
value,
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
),
|
||||
lambda value: LegacyFactorBinding.from_json(value, definition=definition),
|
||||
)
|
||||
for decoder, encoded in zip(
|
||||
decoders,
|
||||
(definition.to_json(), factor_set.to_json(), binding.to_json()),
|
||||
strict=True,
|
||||
):
|
||||
with pytest.raises(FactorContractError) as exc_info:
|
||||
decoder(non_canonical(encoded))
|
||||
assert exc_info.value.code is ContractErrorCode.INVALID_FORMAT
|
||||
assert exc_info.value.path == "$"
|
||||
|
||||
|
||||
def test_factor_definition_identity_is_order_independent_where_semantics_are_unordered() -> None:
|
||||
first_input = FactorInput("prices", "sha256:" + "6" * 64, ("close",))
|
||||
second_input = FactorInput("volumes", "sha256:" + "7" * 64, ("volume",))
|
||||
inputs = (first_input, second_input)
|
||||
parameters_a = {
|
||||
"window": TypedParameter("integer", 10),
|
||||
"weights": TypedParameter("json", {"fast": [1, 2], "slow": [3, 4]}),
|
||||
}
|
||||
parameters_b = {
|
||||
"weights": TypedParameter("json", {"slow": [3, 4], "fast": [1, 2]}),
|
||||
"window": TypedParameter("integer", 10),
|
||||
}
|
||||
first = _definition(
|
||||
inputs=inputs,
|
||||
parameters=parameters_a,
|
||||
input_schema_digest=factor_input_schema_digest(inputs),
|
||||
)
|
||||
second = _definition(
|
||||
inputs=tuple(reversed(inputs)),
|
||||
parameters=parameters_b,
|
||||
input_schema_digest=factor_input_schema_digest(tuple(reversed(inputs))),
|
||||
)
|
||||
assert first.definition_id == second.definition_id
|
||||
assert first.to_json() == second.to_json()
|
||||
|
||||
semantic_changes = (
|
||||
_definition(factor_id="alpha_006"),
|
||||
_definition(version="1.0.1"),
|
||||
_definition(formula="correlation(close, volume, 11)"),
|
||||
_definition(parameters={"window": TypedParameter("integer", 10)}),
|
||||
_definition(implementation_digest="sha256:" + "9" * 64),
|
||||
_definition(valid_until="2027-01-02T00:00:00Z"),
|
||||
_definition(warmup_sessions=11),
|
||||
_definition(lag_sessions=2),
|
||||
_definition(producer=ProducerIdentity("quant_engine", "1.0.1")),
|
||||
_definition(code_revision="d" * 40),
|
||||
)
|
||||
assert all(changed.definition_id != _golden_definition().definition_id for changed in semantic_changes)
|
||||
assert len({changed.definition_id for changed in semantic_changes}) == len(semantic_changes)
|
||||
|
||||
|
||||
def test_parameter_types_decimal_profile_and_detached_nested_values_are_strict() -> None:
|
||||
nested = {"ordered": [1, {"flag": True}]}
|
||||
parameter = TypedParameter("json", nested)
|
||||
nested["ordered"].append(2)
|
||||
definition = _definition(parameters={"payload": parameter})
|
||||
assert definition.to_dict()["parameters"]["payload"]["value"] == {
|
||||
"ordered": [1, {"flag": True}]
|
||||
}
|
||||
integer_definition = _definition(parameters={"value": TypedParameter("integer", 1)})
|
||||
string_definition = _definition(parameters={"value": TypedParameter("string", "1")})
|
||||
assert integer_definition.definition_id != string_definition.definition_id
|
||||
|
||||
for parameter_type, value, code in (
|
||||
("decimal", "1.0", ContractErrorCode.INVALID_FORMAT),
|
||||
("decimal", "1e3", ContractErrorCode.INVALID_FORMAT),
|
||||
("decimal", "-0", ContractErrorCode.INVALID_FORMAT),
|
||||
("integer", True, ContractErrorCode.TYPE_ERROR),
|
||||
("json", 1.5, ContractErrorCode.TYPE_ERROR),
|
||||
("json", {"é": "bad-key"}, ContractErrorCode.INVALID_FORMAT),
|
||||
("json", 9_007_199_254_740_992, ContractErrorCode.INVALID_VALUE),
|
||||
):
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
TypedParameter(parameter_type, value)
|
||||
assert error.value.code is code
|
||||
assert TypedParameter("decimal", "10.25").to_dict()["value"] == "10.25"
|
||||
|
||||
|
||||
def test_catalog_rejects_duplicate_and_overlapping_logical_validity_but_allows_adjacency() -> None:
|
||||
base = _golden_definition()
|
||||
adjacent = _definition(valid_from="2027-01-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
|
||||
assert len(validate_factor_catalog((adjacent, base))) == 2
|
||||
with pytest.raises(FactorContractError) as duplicate:
|
||||
validate_factor_catalog((base, base))
|
||||
_assert_error(duplicate, ContractErrorCode.INVALID_VALUE, "$.definitions")
|
||||
overlapping = _definition(valid_from="2026-06-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
|
||||
with pytest.raises(FactorContractError) as overlap:
|
||||
validate_factor_catalog((base, overlapping))
|
||||
_assert_error(overlap, ContractErrorCode.TIME_ORDER_VIOLATION, "$.definitions")
|
||||
|
||||
|
||||
def test_upstream_contracts_reject_unknown_fields_identity_forgery_and_unqualified_input() -> None:
|
||||
unknown = _golden()["dataset_snapshot"]
|
||||
unknown["provider"] = "forbidden"
|
||||
with pytest.raises(FactorContractError) as unknown_error:
|
||||
DatasetSnapshotEnvelope.from_dict(unknown)
|
||||
_assert_error(unknown_error, ContractErrorCode.UNKNOWN_FIELD, "$.provider")
|
||||
|
||||
forged = _golden()["data_foundation"]
|
||||
forged["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
|
||||
with pytest.raises(FactorContractError) as forged_error:
|
||||
DataFoundationEnvelope.from_dict(forged)
|
||||
assert forged_error.value.code is ContractErrorCode.IDENTITY_MISMATCH
|
||||
assert forged_error.value.path.endswith("view_ref_id")
|
||||
|
||||
rejected_source = _golden()
|
||||
rejected_source["dataset_snapshot"]["descriptor"]["qualification"]["status"] = "rejected"
|
||||
_reidentify(rejected_source["dataset_snapshot"], "snapshot_id", "rhdsv1:sha256:")
|
||||
rejected_snapshot = DatasetSnapshotEnvelope.from_dict(rejected_source["dataset_snapshot"])
|
||||
_, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as rejected_error:
|
||||
FactorSetRef.create(
|
||||
**_factor_set_arguments(snapshot=rejected_snapshot, foundation=foundation)
|
||||
)
|
||||
_assert_error(
|
||||
rejected_error,
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
"$.dataset_snapshot.descriptor.qualification",
|
||||
)
|
||||
|
||||
|
||||
def test_foundation_rejects_future_knowledge_and_per_view_calendar_borrowing() -> None:
|
||||
future = _golden()["data_foundation"]
|
||||
action = future["corporate_action_revisions"][0]
|
||||
old_action_id = action["action_revision_id"]
|
||||
action["knowledge_time"] = "2026-01-03T00:00:01Z"
|
||||
_reidentify(action, "action_revision_id", "rhcav1:sha256:")
|
||||
future["standardized_views"][0]["corporate_action_revision_ids"] = [action["action_revision_id"]]
|
||||
lineage = next(item for item in future["revision_lineage"] if item["revision_id"] == old_action_id)
|
||||
lineage["revision_id"] = action["action_revision_id"]
|
||||
lineage["knowledge_time"] = action["knowledge_time"]
|
||||
_reidentify(future["standardized_views"][0], "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(future, "foundation_id", "rhdfv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as future_error:
|
||||
DataFoundationEnvelope.from_dict(future)
|
||||
_assert_error(
|
||||
future_error,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
"$.revision_lineage.knowledge_time",
|
||||
)
|
||||
|
||||
uncovered = _golden()["data_foundation"]
|
||||
original_route_id = uncovered["instrument_routes"][0]["route_revision_id"]
|
||||
second_calendar = copy.deepcopy(uncovered["trading_calendar_revisions"][0])
|
||||
second_calendar["calendar_id"] = "rhcalendar:99990000111122223333444455556666"
|
||||
_reidentify(second_calendar, "calendar_revision_id", "rhcalv1:sha256:")
|
||||
uncovered["trading_calendar_revisions"].append(second_calendar)
|
||||
route = uncovered["instrument_routes"][0]
|
||||
route["calendar_id"] = second_calendar["calendar_id"]
|
||||
_reidentify(route, "route_revision_id", "rhroutev1:sha256:")
|
||||
route_lineage = next(item for item in uncovered["revision_lineage"] if item["revision_id"] == original_route_id)
|
||||
route_lineage["revision_id"] = route["route_revision_id"]
|
||||
uncovered["revision_lineage"].append(
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": second_calendar["calendar_revision_id"],
|
||||
"revision_number": 1,
|
||||
"knowledge_time": second_calendar["knowledge_time"],
|
||||
"evidence_digest": second_calendar["evidence_digest"],
|
||||
}
|
||||
)
|
||||
view = uncovered["standardized_views"][0]
|
||||
view["instrument_route_revision_ids"] = [route["route_revision_id"]]
|
||||
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(uncovered, "foundation_id", "rhdfv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as calendar_error:
|
||||
DataFoundationEnvelope.from_dict(uncovered)
|
||||
assert calendar_error.value.code is ContractErrorCode.INPUT_CLOSURE_VIOLATION
|
||||
assert "selected route calendar" in calendar_error.value.detail
|
||||
|
||||
|
||||
def _replay_fixture() -> dict[str, Any]:
|
||||
fixture = _golden()
|
||||
snapshot = fixture["dataset_snapshot"]
|
||||
snapshot["descriptor"]["published_at"] = "2026-01-04T00:00:00Z"
|
||||
_reidentify(snapshot, "snapshot_id", "rhdsv1:sha256:")
|
||||
foundation = fixture["data_foundation"]
|
||||
foundation["dataset_snapshot_id"] = snapshot["snapshot_id"]
|
||||
for view in foundation["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot["snapshot_id"]
|
||||
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
|
||||
return fixture
|
||||
|
||||
|
||||
def test_as_available_and_retrospective_replay_keep_distinct_time_claims() -> None:
|
||||
as_available = _factor_set()
|
||||
assert as_available.historical_availability is HistoricalAvailability.DECLARED_AS_AVAILABLE
|
||||
|
||||
replay_source = _replay_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(replay_source)
|
||||
replay_view_id = next(iter(foundation.views))
|
||||
arguments = _factor_set_arguments(
|
||||
fixture=replay_source,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
arguments.update(
|
||||
selected_view_ref_ids=(replay_view_id,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
replay_view_id,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(replay_view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
computed_at="2026-01-04T00:20:00Z",
|
||||
artifact_available_at="2026-01-04T00:25:00Z",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
replay = FactorSetRef.create(**arguments)
|
||||
assert replay.evaluation_at == "2026-01-03T11:00:00Z"
|
||||
assert replay.computed_at == "2026-01-04T00:20:00Z"
|
||||
assert replay.historical_availability is HistoricalAvailability.NOT_ESTABLISHED
|
||||
|
||||
replay_source_args = _factor_set_arguments(
|
||||
fixture=replay_source,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
replay_source_args.update(
|
||||
selected_view_ref_ids=(replay_view_id,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
replay_source_args["definitions"][0].definition_id,
|
||||
"market",
|
||||
replay_view_id,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(replay_view_id, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
with pytest.raises(FactorContractError) as late_publication:
|
||||
FactorSetRef.create(**replay_source_args)
|
||||
_assert_error(
|
||||
late_publication,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
"$.dataset_snapshot.descriptor.published_at",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("overrides", "path"),
|
||||
[
|
||||
({"view_availability": (ViewAvailability(VIEW_REF_ID, "2026-01-03T00:00:01Z", "sha256:" + "2" * 64),)}, "$.view_availability"),
|
||||
({"computed_at": "2026-01-02T23:40:00Z"}, "$.computed_at"),
|
||||
({"artifact_available_at": "2026-01-03T10:14:00Z"}, "$.artifact_available_at"),
|
||||
({"artifact_available_at": "2026-01-03T11:00:01Z"}, "$.artifact_available_at"),
|
||||
({"evaluation_at": "2026-01-03T11:00:00"}, "$.evaluation_at"),
|
||||
],
|
||||
)
|
||||
def test_as_available_time_failures_are_typed(overrides: dict[str, Any], path: str) -> None:
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
_factor_set(**overrides)
|
||||
assert error.value.code in {
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
}
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_replay_rejects_backdating_and_historical_availability_promotion() -> None:
|
||||
source = _replay_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(source)
|
||||
view_id = next(iter(foundation.views))
|
||||
arguments = _factor_set_arguments(fixture=source, snapshot=snapshot, foundation=foundation)
|
||||
definition = arguments["definitions"][0]
|
||||
arguments.update(
|
||||
selected_view_ref_ids=(view_id,),
|
||||
input_bindings=(InputBinding(definition.definition_id, "market", view_id, VIEW_SCHEMA_DIGEST),),
|
||||
view_availability=(ViewAvailability(view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),),
|
||||
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
computed_at="2026-01-04T00:20:00Z",
|
||||
artifact_available_at="2026-01-04T00:25:00Z",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
replay = FactorSetRef.create(**arguments)
|
||||
promoted = replay.to_dict()
|
||||
promoted["historical_availability"] = "declared_as_available"
|
||||
_reidentify(promoted, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as promotion_error:
|
||||
FactorSetRef.from_dict(
|
||||
promoted,
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
promotion_error,
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
"$.historical_availability",
|
||||
)
|
||||
arguments["computed_at"] = "2026-01-03T11:30:00Z"
|
||||
with pytest.raises(FactorContractError) as backdated_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(backdated_error, ContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
|
||||
|
||||
|
||||
def _multi_view_fixture() -> tuple[dict[str, Any], str]:
|
||||
fixture = _golden()
|
||||
foundation = fixture["data_foundation"]
|
||||
second = copy.deepcopy(foundation["standardized_views"][0])
|
||||
second["view_id"] = "rhview:11111111222222223333333344444444"
|
||||
second["schema_digest"] = "sha256:" + "6" * 64
|
||||
second["content_digest"] = "sha256:" + "7" * 64
|
||||
second["transformation_digest"] = "sha256:" + "8" * 64
|
||||
_reidentify(second, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
foundation["standardized_views"].append(second)
|
||||
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
|
||||
return fixture, second["view_ref_id"]
|
||||
|
||||
|
||||
def test_multi_input_mapping_requires_exact_consumption_closure_and_is_order_independent() -> None:
|
||||
fixture, second_view_id = _multi_view_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
inputs = (
|
||||
FactorInput("prices", VIEW_SCHEMA_DIGEST, ("close",)),
|
||||
FactorInput("volumes", "sha256:" + "6" * 64, ("volume",)),
|
||||
)
|
||||
definition = _definition(
|
||||
inputs=inputs,
|
||||
formula="correlation(close, volume, 10)",
|
||||
input_schema_digest=factor_input_schema_digest(inputs),
|
||||
)
|
||||
first_binding = InputBinding(definition.definition_id, "prices", VIEW_REF_ID, VIEW_SCHEMA_DIGEST)
|
||||
second_binding = InputBinding(definition.definition_id, "volumes", second_view_id, "sha256:" + "6" * 64)
|
||||
first_availability = ViewAvailability(VIEW_REF_ID, "2026-01-02T23:40:00Z", "sha256:" + "2" * 64)
|
||||
second_availability = ViewAvailability(second_view_id, "2026-01-02T23:50:00Z", "sha256:" + "6" * 64)
|
||||
base = _factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition)
|
||||
base.update(
|
||||
selected_view_ref_ids=(VIEW_REF_ID, second_view_id),
|
||||
input_bindings=(first_binding, second_binding),
|
||||
view_availability=(first_availability, second_availability),
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
first = FactorSetRef.create(**base)
|
||||
reordered = dict(base)
|
||||
reordered.update(
|
||||
selected_view_ref_ids=(second_view_id, VIEW_REF_ID),
|
||||
input_bindings=(second_binding, first_binding),
|
||||
view_availability=(second_availability, first_availability),
|
||||
)
|
||||
assert FactorSetRef.create(**reordered).factor_set_id == first.factor_set_id
|
||||
|
||||
for invalid_bindings, invalid_views in (
|
||||
((first_binding,), (VIEW_REF_ID, second_view_id)),
|
||||
((first_binding, second_binding), (VIEW_REF_ID,)),
|
||||
((first_binding, second_binding), (VIEW_REF_ID, second_view_id, VIEW_REF_ID)),
|
||||
):
|
||||
invalid = dict(base)
|
||||
invalid.update(input_bindings=invalid_bindings, selected_view_ref_ids=invalid_views)
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.create(**invalid)
|
||||
assert error.value.code in {
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
ContractErrorCode.INVALID_VALUE,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("mutate", "code", "path"),
|
||||
[
|
||||
(lambda value: value["producer"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.producer.id"),
|
||||
(lambda value: value["producer"].pop("version"), ContractErrorCode.MISSING_FIELD, "$.producer.version"),
|
||||
(lambda value: value["producer"].update(id="other_engine"), ContractErrorCode.LINEAGE_VIOLATION, "$.producer.id"),
|
||||
(lambda value: value["producer"].update(version="latest"), ContractErrorCode.INVALID_FORMAT, "$.producer.version"),
|
||||
(lambda value: value.update(code_revision="bad"), ContractErrorCode.INVALID_FORMAT, "$.code_revision"),
|
||||
(lambda value: value["actor"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.actor.kind"),
|
||||
(lambda value: value["actor"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.actor.id"),
|
||||
(lambda value: value["actor"].update(kind="robot"), ContractErrorCode.INVALID_VALUE, "$.actor.kind"),
|
||||
(lambda value: value["actor"].update(id="latest"), ContractErrorCode.INVALID_VALUE, "$.actor.id"),
|
||||
(lambda value: value.pop("correlation_id"), ContractErrorCode.MISSING_FIELD, "$.correlation_id"),
|
||||
(lambda value: value.update(correlation_id="latest"), ContractErrorCode.INVALID_VALUE, "$.correlation_id"),
|
||||
(lambda value: value["causation"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.causation.kind"),
|
||||
(lambda value: value["causation"].update(kind="run"), ContractErrorCode.INVALID_VALUE, "$.causation.kind"),
|
||||
(lambda value: value["causation"].update(id="rhdfv1:sha256:" + "0" * 64), ContractErrorCode.LINEAGE_VIOLATION, "$.causation.id"),
|
||||
(lambda value: value.pop("output_artifact_ref"), ContractErrorCode.MISSING_FIELD, "$.output_artifact_ref"),
|
||||
(lambda value: value["output_artifact_ref"].update(artifact_id="rhfactoroutputv1:sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.output_artifact_ref.artifact_id"),
|
||||
(_mutate_artifact_schema_binding, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref"),
|
||||
(lambda value: value.update(availability_mode="implicit_fallback"), ContractErrorCode.INVALID_VALUE, "$.availability_mode"),
|
||||
(lambda value: value.pop("computed_at"), ContractErrorCode.MISSING_FIELD, "$.computed_at"),
|
||||
(lambda value: value.update(decision_eligible=True), ContractErrorCode.READINESS_ESCALATION, "$.decision_eligible"),
|
||||
(lambda value: value.update(evidence_scope="real_data"), ContractErrorCode.READINESS_ESCALATION, "$.evidence_scope"),
|
||||
(lambda value: value["upstream_evidence"].update(qualification_evidence_digest="sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.upstream_evidence"),
|
||||
],
|
||||
)
|
||||
def test_lineage_artifact_and_readiness_fields_have_independent_typed_negatives(
|
||||
mutate: Callable[[dict[str, Any]], Any],
|
||||
code: ContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
factor_set = _factor_set()
|
||||
value = factor_set.to_dict()
|
||||
mutate(value)
|
||||
if "factor_set_id" in value:
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(error, code, path)
|
||||
|
||||
|
||||
def test_output_schema_content_bytes_cannot_be_swapped_or_forged() -> None:
|
||||
factor_set = _factor_set()
|
||||
fixture = _golden()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
with pytest.raises(FactorContractError) as swapped:
|
||||
FactorSetRef.from_dict(
|
||||
factor_set.to_dict(),
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
output_schema_bytes=content_bytes,
|
||||
output_content_bytes=schema_bytes,
|
||||
)
|
||||
_assert_error(swapped, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref")
|
||||
with pytest.raises(FactorContractError) as noncanonical:
|
||||
FactorSetRef.create(
|
||||
**{
|
||||
**_factor_set_arguments(),
|
||||
"output_schema_bytes": json.dumps(fixture["output_schema"], indent=2).encode(),
|
||||
}
|
||||
)
|
||||
_assert_error(noncanonical, ContractErrorCode.INVALID_FORMAT, "$.output_schema_bytes")
|
||||
|
||||
|
||||
def test_unsuccessful_output_quality_or_coverage_cannot_form_a_factor_set() -> None:
|
||||
with pytest.raises(FactorContractError) as failed_quality:
|
||||
_factor_set(
|
||||
output_quality=OutputQuality(
|
||||
"failed",
|
||||
(OutputQualityCheck("finite_values", "failed", "sha256:" + "3" * 64),),
|
||||
)
|
||||
)
|
||||
_assert_error(failed_quality, ContractErrorCode.INVALID_VALUE, "$.output_quality")
|
||||
|
||||
for coverage in (
|
||||
OutputCoverage("incomplete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
|
||||
OutputCoverage("complete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
|
||||
):
|
||||
with pytest.raises(FactorContractError) as incomplete:
|
||||
_factor_set(output_coverage=coverage)
|
||||
_assert_error(incomplete, ContractErrorCode.INVALID_VALUE, "$.output_coverage")
|
||||
|
||||
|
||||
def test_external_snapshot_definition_and_view_references_cannot_be_substituted() -> None:
|
||||
factor_set = _factor_set()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
value = factor_set.to_dict()
|
||||
value["dataset_snapshot_id"] = "rhdsv1:sha256:" + "0" * 64
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as snapshot_error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
snapshot_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.dataset_snapshot_id",
|
||||
)
|
||||
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = ["rhfactorv1:sha256:" + "0" * 64]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as definition_error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
definition_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.definition_ids",
|
||||
)
|
||||
|
||||
arguments = _factor_set_arguments()
|
||||
arguments["selected_view_ref_ids"] = ("rhviewrefv1:sha256:" + "0" * 64,)
|
||||
with pytest.raises(FactorContractError) as view_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(
|
||||
view_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.selected_view_ref_ids",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"invalid_definition_id",
|
||||
[
|
||||
{"unexpected": "object"},
|
||||
["array"],
|
||||
42,
|
||||
True,
|
||||
None,
|
||||
],
|
||||
)
|
||||
def test_factor_set_ref_definition_ids_reject_non_string_types(
|
||||
invalid_definition_id: Any,
|
||||
) -> None:
|
||||
factor_set = _factor_set()
|
||||
definition = _golden_definition()
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = [definition.definition_id, invalid_definition_id]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_json(
|
||||
canonical_json_bytes(value),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
|
||||
_assert_error(error, ContractErrorCode.TYPE_ERROR, "$.definition_ids[1]")
|
||||
|
||||
|
||||
def test_factor_set_ref_definition_ids_still_reject_duplicate_strings() -> None:
|
||||
factor_set = _factor_set()
|
||||
definition = _golden_definition()
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = [definition.definition_id, definition.definition_id]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_json(
|
||||
canonical_json_bytes(value),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
|
||||
_assert_error(error, ContractErrorCode.INVALID_VALUE, "$.definition_ids")
|
||||
|
||||
|
||||
def test_factor_set_parent_requires_exact_identity_and_correlation() -> None:
|
||||
parent = _factor_set()
|
||||
child_arguments = _factor_set_arguments()
|
||||
child_arguments.update(
|
||||
output_content_bytes=canonical_json_bytes({"rows": [{"value": "0.250"}]}),
|
||||
causation=Causation("factor_set", parent.factor_set_id),
|
||||
parent=parent,
|
||||
)
|
||||
child_arguments["output_artifact_ref"] = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(child_arguments["output_schema_bytes"]),
|
||||
content_digest=_sha256(child_arguments["output_content_bytes"]),
|
||||
)
|
||||
child = FactorSetRef.create(**child_arguments)
|
||||
assert child.causation.id == parent.factor_set_id
|
||||
missing_parent = child.to_dict()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as missing_error:
|
||||
FactorSetRef.from_dict(
|
||||
missing_parent,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(missing_error, ContractErrorCode.LINEAGE_VIOLATION, "$.causation")
|
||||
wrong_correlation = dict(child_arguments)
|
||||
wrong_correlation["correlation_id"] = "different_run"
|
||||
with pytest.raises(FactorContractError) as correlation_error:
|
||||
FactorSetRef.create(**wrong_correlation)
|
||||
_assert_error(correlation_error, ContractErrorCode.LINEAGE_VIOLATION, "$.correlation_id")
|
||||
|
||||
|
||||
def test_legacy_bridge_is_explicit_lossy_and_preserves_all_four_historical_fields() -> None:
|
||||
definition = _golden_definition()
|
||||
legacy = FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="1.0.0",
|
||||
)
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id=legacy.factor_id,
|
||||
legacy_version=legacy.version,
|
||||
legacy_definition_sha256=legacy.definition_sha256,
|
||||
legacy_dataset_schema_version=legacy.dataset_schema_version,
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
assert bind_legacy_factor(legacy, definition, binding) is definition
|
||||
assert project_legacy_factor(definition, binding) == legacy
|
||||
assert legacy.version_id == "factor:demo-momentum@1.0.0"
|
||||
assert legacy.definition_sha256 != definition.definition_id.rsplit(":", maxsplit=1)[-1]
|
||||
assert LegacyFactorBinding.from_json(binding.to_json(), definition=definition) == binding
|
||||
|
||||
mismatched = FactorVersion(
|
||||
factor_id="factor:different",
|
||||
version=legacy.version,
|
||||
definition_sha256=legacy.definition_sha256,
|
||||
dataset_schema_version=legacy.dataset_schema_version,
|
||||
)
|
||||
with pytest.raises(FactorContractError) as mismatch_error:
|
||||
bind_legacy_factor(mismatched, definition, binding)
|
||||
_assert_error(mismatch_error, ContractErrorCode.LEGACY_BINDING_MISMATCH, "$.binding")
|
||||
|
||||
|
||||
def test_bare_legacy_factor_or_id_cannot_enter_factor_set_contract() -> None:
|
||||
legacy = FactorVersion("factor:demo-momentum", "1.0.0", "b" * 64, "1.0.0")
|
||||
arguments = _factor_set_arguments()
|
||||
arguments["definitions"] = (legacy,)
|
||||
with pytest.raises(FactorContractError) as legacy_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(legacy_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
|
||||
arguments["definitions"] = (legacy.version_id,)
|
||||
with pytest.raises(FactorContractError) as id_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(id_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
|
||||
@@ -0,0 +1,338 @@
|
||||
"""Governed Personal Quant OS vertical-slice contracts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.governed_pipeline import (
|
||||
DatasetSnapshot,
|
||||
FactorVersion,
|
||||
PaperOrderIntent,
|
||||
RiskDecisionStatus,
|
||||
RiskPolicy,
|
||||
StrategyStage,
|
||||
StrategyVersion,
|
||||
create_paper_order_intent,
|
||||
run_governed_factor_slice,
|
||||
)
|
||||
|
||||
|
||||
def _calendar() -> pd.DatetimeIndex:
|
||||
return pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
|
||||
|
||||
def _scores() -> pd.DataFrame:
|
||||
dates = _calendar()
|
||||
return pd.DataFrame(
|
||||
{"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]},
|
||||
index=dates[:2],
|
||||
)
|
||||
|
||||
|
||||
def _prices() -> tuple[pd.DataFrame, pd.DataFrame]:
|
||||
dates = _calendar()
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 10.2, 10.4], "B": [20.0, 20.0, 20.5, 21.0], "C": [30.0, 30.0, 30.0, 30.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = opens * 1.01
|
||||
return opens, closes
|
||||
|
||||
|
||||
def _snapshot() -> DatasetSnapshot:
|
||||
return DatasetSnapshot(
|
||||
snapshot_id="dataset:cn-a-daily-20260108-v1",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 7, tzinfo=UTC),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 8, 5, tzinfo=UTC),
|
||||
)
|
||||
|
||||
|
||||
def _factor() -> FactorVersion:
|
||||
return FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="1.0.0",
|
||||
)
|
||||
|
||||
|
||||
def _strategy() -> StrategyVersion:
|
||||
return StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id="factor:demo-momentum@1.0.0",
|
||||
stage=StrategyStage.APPROVED,
|
||||
)
|
||||
|
||||
|
||||
def _execution_config() -> ExecutionConfig:
|
||||
return ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
|
||||
def test_governed_slice_is_reproducible_and_creates_only_paper_intent() -> None:
|
||||
opens, closes = _prices()
|
||||
created_at = datetime(2026, 1, 9, 1, tzinfo=UTC)
|
||||
policy = RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
)
|
||||
|
||||
result = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=policy,
|
||||
code_revision="c" * 40,
|
||||
created_at=created_at,
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
assert result.backtest_run.dataset_snapshot_id == _snapshot().snapshot_id
|
||||
assert result.backtest_run.factor_version_id == _factor().version_id
|
||||
assert result.backtest_run.strategy_version_id == _strategy().version_id
|
||||
assert result.backtest_run.code_revision == "c" * 40
|
||||
assert len(result.backtest_run.config_hash) == 64
|
||||
assert result.portfolio_target.backtest_run_id == result.backtest_run.run_id
|
||||
assert result.risk_decision.status is RiskDecisionStatus.APPROVED
|
||||
assert result.risk_decision.portfolio_target_id == result.portfolio_target.target_id
|
||||
assert result.order_intent is not None
|
||||
assert result.order_intent.environment == "paper"
|
||||
assert result.order_intent.risk_decision_id == result.risk_decision.decision_id
|
||||
assert result.order_intent.portfolio_target_id == result.portfolio_target.target_id
|
||||
assert result.factor_version.version_id == "factor:demo-momentum@1.0.0"
|
||||
assert result.factor_version.definition_sha256 == "b" * 64
|
||||
assert result.strategy_version.version_id == "strategy:demo-top2@1.0.0"
|
||||
assert result.backtest_run.run_id == (
|
||||
"backtest-run:e74403571f6a73c98b380220748957a422518c0bed883fabc4b93ebe13f05a37"
|
||||
)
|
||||
assert result.backtest_run.config_hash == (
|
||||
"40a3c804a2dc940161a626d1a5d25817c13463e685005e37fc48c41d1e20b87b"
|
||||
)
|
||||
assert result.portfolio_target.target_id == (
|
||||
"portfolio-target:ab2d398489aa9a292ee155a1098e9340beac0a924874cdaeb5b7b4d379ac9ce8"
|
||||
)
|
||||
assert result.risk_decision.decision_id == (
|
||||
"risk-decision:95924926bd327e44beeb15a63f47d14c5e77b97533fc3227fba2eda68e9b423d"
|
||||
)
|
||||
assert result.order_intent.intent_id == (
|
||||
"order-intent:73c349086c2c05b68424ace9286896eb21507b9080da5ff9b96b58c88d8f6ac1"
|
||||
)
|
||||
|
||||
repeated = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=policy,
|
||||
code_revision="c" * 40,
|
||||
created_at=created_at,
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
assert repeated.backtest_run.run_id == result.backtest_run.run_id
|
||||
assert repeated.portfolio_target.target_id == result.portfolio_target.target_id
|
||||
assert repeated.risk_decision.decision_id == result.risk_decision.decision_id
|
||||
assert repeated.order_intent == result.order_intent
|
||||
|
||||
|
||||
def test_risk_rejection_blocks_order_intent() -> None:
|
||||
opens, closes = _prices()
|
||||
result = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:no-concentration@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.4,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
assert result.risk_decision.status is RiskDecisionStatus.REJECTED
|
||||
assert any("single-asset weight" in reason for reason in result.risk_decision.reasons)
|
||||
assert result.order_intent is None
|
||||
with pytest.raises(ValueError, match="approved risk decision"):
|
||||
create_paper_order_intent(result.portfolio_target, result.risk_decision)
|
||||
with pytest.raises(ValueError, match="approved risk decision"):
|
||||
PaperOrderIntent(result.portfolio_target, result.risk_decision)
|
||||
|
||||
|
||||
def test_dataset_snapshot_requires_point_in_time_ordering_and_aware_times() -> None:
|
||||
with pytest.raises(ValueError, match="timezone-aware"):
|
||||
DatasetSnapshot(
|
||||
snapshot_id="dataset:invalid",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 7),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="effective_at <= available_at <= ingested_at"):
|
||||
DatasetSnapshot(
|
||||
snapshot_id="dataset:invalid",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 10, tzinfo=UTC),
|
||||
)
|
||||
|
||||
|
||||
def test_strategy_factor_lineage_must_match() -> None:
|
||||
opens, closes = _prices()
|
||||
mismatched = StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id="factor:other@1.0.0",
|
||||
stage=StrategyStage.APPROVED,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="factor lineage"):
|
||||
run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=mismatched,
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
|
||||
def test_governed_slice_requires_matching_schema_and_snapshot_available_by_run_time() -> None:
|
||||
opens, closes = _prices()
|
||||
common = {
|
||||
"factor_scores": _scores(),
|
||||
"execution_prices": opens,
|
||||
"valuation_prices": closes,
|
||||
"strategy_version": _strategy(),
|
||||
"risk_policy": RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
"code_revision": "c" * 40,
|
||||
"top_k": 2,
|
||||
"execution_price_field": "open",
|
||||
"valuation_price_field": "close",
|
||||
"execution_config": _execution_config(),
|
||||
}
|
||||
|
||||
with pytest.raises(ValueError, match="dataset schema"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="2.0.0",
|
||||
),
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
**common,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="available before the research run"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
created_at=datetime(2026, 1, 8, 7, 30, tzinfo=UTC),
|
||||
**common,
|
||||
)
|
||||
|
||||
future_scores = _scores()
|
||||
future_scores.index = pd.date_range("2026-01-12", periods=2, freq="B")
|
||||
with pytest.raises(ValueError, match="future decision dates"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
factor_scores=future_scores,
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
strategy_version=common["strategy_version"],
|
||||
risk_policy=common["risk_policy"],
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
|
||||
def test_paper_intent_requires_approved_strategy_stage() -> None:
|
||||
opens, closes = _prices()
|
||||
validated = StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id=_factor().version_id,
|
||||
stage=StrategyStage.VALIDATED,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="Approved or Paper"):
|
||||
run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=validated,
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
@@ -0,0 +1,727 @@
|
||||
"""Closed performance-evidence contract conformance tests."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
PERFORMANCE_EVIDENCE_SCHEMA_VERSION,
|
||||
PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
PERFORMANCE_METHODOLOGY_ID,
|
||||
BacktestEvidenceManifest,
|
||||
EvidenceQualification,
|
||||
PerformanceEvidenceError,
|
||||
PerformanceEvidenceErrorCode,
|
||||
PerformanceEvidenceV1,
|
||||
PerformanceMetricAvailability,
|
||||
ResearchRunArtifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_performance_evidence,
|
||||
build_research_run_artifact,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
InputBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
ProducerIdentity,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.governed_pipeline import BacktestRunRef
|
||||
from quant_engine.metrics import TRADING_DAYS_PER_YEAR, benchmark_summary, summary
|
||||
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
PERFORMANCE_FIXTURE = (
|
||||
ROOT / "tests" / "fixtures" / "performance-evidence-v1.golden.json"
|
||||
)
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _accepted_authorities() -> tuple[
|
||||
DatasetSnapshotEnvelope,
|
||||
DataFoundationEnvelope,
|
||||
FactorSetRef,
|
||||
]:
|
||||
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
|
||||
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
artifact_ref = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
selected_view_ref_ids=(VIEW_REF_ID,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
definition.definition_id,
|
||||
"market",
|
||||
VIEW_REF_ID,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
output_quality=OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
output_coverage=OutputCoverage(
|
||||
"complete",
|
||||
1,
|
||||
1,
|
||||
"row",
|
||||
"alpha_005.cn_a",
|
||||
"sha256:" + "4" * 64,
|
||||
),
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
output_artifact_ref=artifact_ref,
|
||||
availability_mode=AvailabilityMode.AS_AVAILABLE,
|
||||
evaluation_at="2026-01-03T11:00:00Z",
|
||||
computed_at="2026-01-03T10:15:00Z",
|
||||
artifact_available_at="2026-01-03T10:20:00Z",
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
actor=ActorIdentity("service", "factor_worker_v1"),
|
||||
correlation_id="research_run_001",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
evidence_scope="synthetic_fixture",
|
||||
decision_eligible=False,
|
||||
)
|
||||
return snapshot, foundation, factor_set
|
||||
|
||||
|
||||
def _configuration_digest() -> str:
|
||||
return _sha256(
|
||||
json.dumps(
|
||||
PARAMETERS,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
)
|
||||
|
||||
|
||||
def _run_ref(**overrides: Any) -> BacktestRunRef:
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
arguments: dict[str, Any] = {
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"factor_set": factor_set,
|
||||
"universe_digest": "sha256:" + "5" * 64,
|
||||
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
|
||||
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": "sha256:" + "6" * 64,
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": "sha256:" + "7" * 64,
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": "sha256:" + "8" * 64,
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": "sha256:" + "9" * 64,
|
||||
"configuration_digest": _configuration_digest(),
|
||||
"evaluation_at": "2026-01-08T01:00:00Z",
|
||||
"computed_at": "2026-01-08T02:00:00Z",
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return BacktestRunRef.create(**arguments)
|
||||
|
||||
|
||||
def _backtest_result() -> FactorBacktestResult:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
return run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _artifact(
|
||||
run_ref: BacktestRunRef,
|
||||
benchmark_kind: str,
|
||||
) -> tuple[ResearchRunArtifact, FactorBacktestResult]:
|
||||
result = _backtest_result()
|
||||
benchmark_id: str | None
|
||||
benchmark_returns: pd.Series | None
|
||||
if benchmark_kind == "absent":
|
||||
benchmark_id = None
|
||||
benchmark_returns = None
|
||||
elif benchmark_kind == "estimable":
|
||||
benchmark_id = "000300.SH"
|
||||
benchmark_returns = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
elif benchmark_kind == "zero_active_variance":
|
||||
benchmark_id = "000300.SH"
|
||||
benchmark_returns = result.returns.rename("benchmark_return")
|
||||
elif benchmark_kind == "zero_benchmark_variance":
|
||||
benchmark_id = "000300.SH"
|
||||
benchmark_returns = pd.Series(
|
||||
np.zeros(len(result.returns)),
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
else:
|
||||
raise AssertionError(f"unknown benchmark_kind: {benchmark_kind}")
|
||||
artifact = build_research_run_artifact(
|
||||
result,
|
||||
run_id=run_ref.run_id,
|
||||
strategy_id=run_ref.strategy_id,
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version=run_ref.strategy_version,
|
||||
engine_version="1.2.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
data_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=PARAMETERS,
|
||||
benchmark_id=benchmark_id,
|
||||
benchmark_returns=benchmark_returns,
|
||||
)
|
||||
return artifact, result
|
||||
|
||||
|
||||
def _case(
|
||||
benchmark_kind: str,
|
||||
) -> tuple[PerformanceEvidenceV1, ResearchRunArtifact, BacktestRunRef, BacktestEvidenceManifest]:
|
||||
run_ref = _run_ref()
|
||||
artifact, _ = _artifact(run_ref, benchmark_kind)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
return (
|
||||
build_performance_evidence(artifact, run_ref, manifest),
|
||||
artifact,
|
||||
run_ref,
|
||||
manifest,
|
||||
)
|
||||
|
||||
|
||||
def _metric_map(evidence: PerformanceEvidenceV1) -> dict[str, Any]:
|
||||
return {metric.key: metric for metric in evidence.metrics}
|
||||
|
||||
|
||||
def _mutate_frozen(value: Any, field: str, replacement: object) -> Any:
|
||||
changed = copy.copy(value)
|
||||
object.__setattr__(changed, field, replacement)
|
||||
return changed
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[PerformanceEvidenceError],
|
||||
code: PerformanceEvidenceErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_present_evidence_is_deterministic_content_addressed_and_three_party_closed() -> None:
|
||||
first, artifact, run_ref, manifest = _case("estimable")
|
||||
second = build_performance_evidence(artifact, run_ref, manifest)
|
||||
|
||||
assert first == second
|
||||
assert first.schema_version == PERFORMANCE_EVIDENCE_SCHEMA_VERSION
|
||||
assert first.performance_evidence_id.startswith("rhperformanceevidencev1:sha256:")
|
||||
assert first.document_sha256.startswith("sha256:")
|
||||
assert first.authority == "quant_engine"
|
||||
assert first.scope == "offline_research_only"
|
||||
assert first.run_id == first.backtest_run_ref_id == run_ref.run_id == manifest.run_id
|
||||
assert first.backtest_evidence_manifest_id == manifest.manifest_id
|
||||
assert first.backtest_evidence_manifest_evidence_digest == manifest.evidence_digest
|
||||
assert first.backtest_evidence_qualification == "contract_qualified"
|
||||
assert first.research_artifact_content_digest == f"sha256:{artifact.content_sha256}"
|
||||
assert first.performance_table_logical_name == "performance"
|
||||
assert first.performance_table_row_count == 1
|
||||
assert first.performance_row_digest.startswith("sha256:")
|
||||
assert first.benchmark_series_digest is not None
|
||||
assert first.canonical_bytes() == first.to_json().encode("utf-8")
|
||||
assert not first.canonical_bytes().endswith(b"\n")
|
||||
document_payload = first.to_dict()
|
||||
document_payload.pop("document_sha256")
|
||||
expected_document = json.dumps(
|
||||
document_payload,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
assert _sha256(expected_document) == first.document_sha256
|
||||
assert PerformanceEvidenceV1.from_dict(
|
||||
first.to_dict(),
|
||||
artifact=artifact,
|
||||
run_ref=run_ref,
|
||||
evidence_manifest=manifest,
|
||||
) == first
|
||||
|
||||
|
||||
def test_methodology_and_metrics_bind_the_actual_artifact_builder_path() -> None:
|
||||
evidence, artifact, _, _ = _case("estimable")
|
||||
result = _backtest_result()
|
||||
expected_absolute = summary(result.returns, rf=0.0)
|
||||
benchmark = artifact.nav.set_index("trade_date")["benchmark_return"]
|
||||
benchmark.index = result.returns.index
|
||||
expected_relative = benchmark_summary(
|
||||
result.returns,
|
||||
benchmark,
|
||||
risk_free_daily=0.0,
|
||||
annualization=TRADING_DAYS_PER_YEAR,
|
||||
)
|
||||
metrics = _metric_map(evidence)
|
||||
|
||||
assert evidence.methodology.methodology_id == PERFORMANCE_METHODOLOGY_ID
|
||||
assert evidence.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID
|
||||
assert evidence.methodology.return_type == "simple"
|
||||
assert evidence.methodology.source_frequency == "1d"
|
||||
assert evidence.methodology.periods_per_year == TRADING_DAYS_PER_YEAR == 252
|
||||
assert evidence.methodology.annual_risk_free == 0.0
|
||||
assert evidence.methodology.benchmark_risk_free_daily == 0.0
|
||||
assert evidence.methodology.benchmark_alignment == "exact_session_index"
|
||||
assert metrics["annualized_return"].value == pytest.approx(
|
||||
expected_absolute["ann_return"]
|
||||
)
|
||||
assert metrics["sharpe_ratio"].value == pytest.approx(expected_absolute["sharpe"])
|
||||
assert metrics["tracking_error"].value == pytest.approx(
|
||||
expected_relative["tracking_error"]
|
||||
)
|
||||
assert metrics["alpha"].value == pytest.approx(expected_relative["alpha"])
|
||||
assert all(metric.methodology_id == PERFORMANCE_METHODOLOGY_ID for metric in metrics.values())
|
||||
assert all(metric.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID for metric in metrics.values())
|
||||
|
||||
|
||||
def test_relative_metric_availability_is_closed_for_present_absent_and_unestimable() -> None:
|
||||
present, *_ = _case("estimable")
|
||||
absent, *_ = _case("absent")
|
||||
zero_active, *_ = _case("zero_active_variance")
|
||||
zero_benchmark, *_ = _case("zero_benchmark_variance")
|
||||
|
||||
present_metrics = _metric_map(present)
|
||||
assert all(
|
||||
present_metrics[key].availability is PerformanceMetricAvailability.AVAILABLE
|
||||
for key in ("tracking_error", "information_ratio", "alpha", "beta")
|
||||
)
|
||||
absent_metrics = _metric_map(absent)
|
||||
assert absent.benchmark_series_digest is None
|
||||
assert absent.benchmark_id == ""
|
||||
assert absent.benchmark_alignment_policy == "none"
|
||||
assert all(
|
||||
absent_metrics[key].value is None
|
||||
and absent_metrics[key].availability
|
||||
is PerformanceMetricAvailability.BENCHMARK_ABSENT
|
||||
for key in ("tracking_error", "information_ratio", "alpha", "beta")
|
||||
)
|
||||
zero_active_metrics = _metric_map(zero_active)
|
||||
assert zero_active_metrics["tracking_error"].value == pytest.approx(0.0)
|
||||
assert (
|
||||
zero_active_metrics["information_ratio"].availability
|
||||
is PerformanceMetricAvailability.NOT_ESTIMABLE_ACTIVE_VARIANCE
|
||||
)
|
||||
assert zero_active_metrics["information_ratio"].value is None
|
||||
zero_benchmark_metrics = _metric_map(zero_benchmark)
|
||||
assert np.isfinite(zero_benchmark_metrics["tracking_error"].value)
|
||||
for key in ("alpha", "beta"):
|
||||
assert zero_benchmark_metrics[key].value is None
|
||||
assert (
|
||||
zero_benchmark_metrics[key].availability
|
||||
is PerformanceMetricAvailability.NOT_ESTIMABLE_BENCHMARK_VARIANCE
|
||||
)
|
||||
|
||||
|
||||
def test_golden_covers_present_absent_and_both_unestimable_states() -> None:
|
||||
expected = {
|
||||
"schema_version": 1,
|
||||
"source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c",
|
||||
"source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603",
|
||||
"cases": {
|
||||
name: _case(name)[0].to_dict()
|
||||
for name in (
|
||||
"estimable",
|
||||
"zero_active_variance",
|
||||
"zero_benchmark_variance",
|
||||
"absent",
|
||||
)
|
||||
},
|
||||
}
|
||||
assert json.loads(PERFORMANCE_FIXTURE.read_text(encoding="utf-8")) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("owner", "field", "replacement", "code", "path"),
|
||||
[
|
||||
(
|
||||
"run_ref",
|
||||
"run_id",
|
||||
"rhbacktestrunv1:sha256:" + "0" * 64,
|
||||
PerformanceEvidenceErrorCode.IDENTITY_MISMATCH,
|
||||
"$.backtest_run_ref.run_id",
|
||||
),
|
||||
(
|
||||
"manifest",
|
||||
"manifest_id",
|
||||
"rhbacktestevidencev1:sha256:" + "0" * 64,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.backtest_evidence_manifest.manifest_id",
|
||||
),
|
||||
(
|
||||
"manifest",
|
||||
"qualification",
|
||||
EvidenceQualification.EXPLORATORY,
|
||||
PerformanceEvidenceErrorCode.AUTHORITY_REJECTED,
|
||||
"$.backtest_evidence_manifest.qualification",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_owner_identity_and_authority_mismatches_fail_closed(
|
||||
owner: str,
|
||||
field: str,
|
||||
replacement: object,
|
||||
code: PerformanceEvidenceErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
_, artifact, run_ref, manifest = _case("estimable")
|
||||
changed_run_ref = _mutate_frozen(run_ref, field, replacement) if owner == "run_ref" else run_ref
|
||||
changed_manifest = (
|
||||
_mutate_frozen(manifest, field, replacement) if owner == "manifest" else manifest
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
build_performance_evidence(artifact, changed_run_ref, changed_manifest)
|
||||
_assert_error(rejected, code, path)
|
||||
|
||||
|
||||
def test_performance_table_row_and_benchmark_digest_mismatches_fail_closed() -> None:
|
||||
evidence, artifact, run_ref, manifest = _case("estimable")
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "n_days"] += 1
|
||||
changed_artifact = replace(artifact, _performance=performance)
|
||||
with pytest.raises(PerformanceEvidenceError) as table_mismatch:
|
||||
build_performance_evidence(changed_artifact, run_ref, manifest)
|
||||
_assert_error(
|
||||
table_mismatch,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.artifact.tables.performance.content_digest",
|
||||
)
|
||||
|
||||
payload = evidence.to_dict()
|
||||
payload["benchmark_series_digest"] = "sha256:" + "0" * 64
|
||||
with pytest.raises(PerformanceEvidenceError) as benchmark_mismatch:
|
||||
PerformanceEvidenceV1.from_dict(
|
||||
payload,
|
||||
artifact=artifact,
|
||||
run_ref=run_ref,
|
||||
evidence_manifest=manifest,
|
||||
)
|
||||
_assert_error(
|
||||
benchmark_mismatch,
|
||||
PerformanceEvidenceErrorCode.BENCHMARK_INVALID,
|
||||
"$.benchmark_series_digest",
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_closure_covers_non_performance_artifact_tables() -> None:
|
||||
_, artifact, run_ref, manifest = _case("estimable")
|
||||
nav = artifact.nav
|
||||
nav.loc[0, "nav"] += 0.01
|
||||
changed_artifact = replace(artifact, _nav=nav)
|
||||
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
build_performance_evidence(changed_artifact, run_ref, manifest)
|
||||
|
||||
_assert_error(
|
||||
rejected,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.artifact.tables.nav.content_digest",
|
||||
)
|
||||
|
||||
|
||||
def test_row_and_benchmark_mutations_change_their_digests_and_document_identity() -> None:
|
||||
original, artifact, run_ref, _ = _case("estimable")
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "sharpe"] += 0.01
|
||||
changed_performance_artifact = replace(artifact, _performance=performance)
|
||||
changed_performance_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_performance_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
changed_performance = build_performance_evidence(
|
||||
changed_performance_artifact,
|
||||
run_ref,
|
||||
changed_performance_manifest,
|
||||
)
|
||||
assert changed_performance.performance_row_digest != original.performance_row_digest
|
||||
assert changed_performance.performance_evidence_id != original.performance_evidence_id
|
||||
|
||||
nav = artifact.nav
|
||||
nav.loc[0, "benchmark_nav"] += 0.01
|
||||
changed_benchmark_artifact = replace(artifact, _nav=nav)
|
||||
changed_benchmark_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_benchmark_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
changed_benchmark = build_performance_evidence(
|
||||
changed_benchmark_artifact,
|
||||
run_ref,
|
||||
changed_benchmark_manifest,
|
||||
)
|
||||
assert changed_benchmark.benchmark_series_digest != original.benchmark_series_digest
|
||||
assert changed_benchmark.performance_row_digest == original.performance_row_digest
|
||||
assert changed_benchmark.performance_evidence_id != original.performance_evidence_id
|
||||
|
||||
|
||||
def test_relative_metric_null_reasons_cannot_be_invented() -> None:
|
||||
_, artifact, run_ref, _ = _case("estimable")
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "alpha"] = float("nan")
|
||||
changed_artifact = replace(artifact, _performance=performance)
|
||||
changed_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as false_alpha_domain:
|
||||
build_performance_evidence(changed_artifact, run_ref, changed_manifest)
|
||||
_assert_error(
|
||||
false_alpha_domain,
|
||||
PerformanceEvidenceErrorCode.METRIC_INVALID,
|
||||
"$.metrics.alpha.value",
|
||||
)
|
||||
|
||||
_, absent_artifact, absent_run_ref, _ = _case("absent")
|
||||
absent_performance = absent_artifact.performance
|
||||
absent_performance.loc[0, "tracking_error"] = 0.0
|
||||
changed_absent = replace(absent_artifact, _performance=absent_performance)
|
||||
changed_absent_manifest = build_backtest_evidence_manifest(
|
||||
absent_run_ref,
|
||||
changed_absent,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as false_absence:
|
||||
build_performance_evidence(
|
||||
changed_absent,
|
||||
absent_run_ref,
|
||||
changed_absent_manifest,
|
||||
)
|
||||
_assert_error(
|
||||
false_absence,
|
||||
PerformanceEvidenceErrorCode.BENCHMARK_INVALID,
|
||||
"$.metrics.tracking_error.availability",
|
||||
)
|
||||
|
||||
|
||||
def test_artifact_builder_enforces_strict_benchmark_session_alignment() -> None:
|
||||
run_ref = _run_ref()
|
||||
result = _backtest_result()
|
||||
misaligned = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index.shift(1, freq="B"),
|
||||
)
|
||||
with pytest.raises(ValueError, match="matching indexes"):
|
||||
build_research_run_artifact(
|
||||
result,
|
||||
run_id=run_ref.run_id,
|
||||
strategy_id=run_ref.strategy_id,
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version=run_ref.strategy_version,
|
||||
engine_version="1.2.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
data_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=PARAMETERS,
|
||||
benchmark_id="000300.SH",
|
||||
benchmark_returns=misaligned,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("column", "value", "path"),
|
||||
[
|
||||
("total_ret", -1.01, "$.metrics.total_return.value"),
|
||||
("ann_ret", -1.01, "$.metrics.annualized_return.value"),
|
||||
("ann_volatility", -0.01, "$.metrics.annualized_volatility.value"),
|
||||
("max_dd", 0.01, "$.metrics.maximum_drawdown.value"),
|
||||
("win_rate", 1.01, "$.metrics.win_rate.value"),
|
||||
("tracking_error", -0.01, "$.metrics.tracking_error.value"),
|
||||
("n_trades", True, "$.metrics.trade_count.value"),
|
||||
],
|
||||
)
|
||||
def test_metric_domains_reject_invalid_source_values(
|
||||
column: str,
|
||||
value: object,
|
||||
path: str,
|
||||
) -> None:
|
||||
_, artifact, run_ref, _ = _case("estimable")
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, column] = value
|
||||
changed_artifact = replace(artifact, _performance=performance)
|
||||
changed_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
build_performance_evidence(changed_artifact, run_ref, changed_manifest)
|
||||
_assert_error(rejected, PerformanceEvidenceErrorCode.METRIC_INVALID, path)
|
||||
|
||||
|
||||
def test_closed_parser_rejects_unknown_non_ascii_non_finite_bool_and_unsafe_integer() -> None:
|
||||
evidence, artifact, run_ref, manifest = _case("estimable")
|
||||
|
||||
mutations: list[tuple[dict[str, Any], PerformanceEvidenceErrorCode, str]] = []
|
||||
unknown = evidence.to_dict()
|
||||
unknown["unexpected"] = "value"
|
||||
mutations.append((unknown, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.unexpected"))
|
||||
non_ascii = evidence.to_dict()
|
||||
non_ascii["métric"] = "value"
|
||||
mutations.append((non_ascii, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.métric"))
|
||||
non_finite = evidence.to_dict()
|
||||
non_finite["metrics"][0]["value"] = float("inf")
|
||||
mutations.append(
|
||||
(non_finite, PerformanceEvidenceErrorCode.METRIC_INVALID, "$.metrics[0].value")
|
||||
)
|
||||
bool_number = evidence.to_dict()
|
||||
bool_number["methodology"]["periods_per_year"] = True
|
||||
mutations.append(
|
||||
(
|
||||
bool_number,
|
||||
PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH,
|
||||
"$.methodology.periods_per_year",
|
||||
)
|
||||
)
|
||||
unsafe = evidence.to_dict()
|
||||
unsafe["performance_table_row_count"] = 2**53
|
||||
mutations.append(
|
||||
(
|
||||
unsafe,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.performance_table_row_count",
|
||||
)
|
||||
)
|
||||
|
||||
for payload, code, path in mutations:
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
PerformanceEvidenceV1.from_dict(
|
||||
payload,
|
||||
artifact=artifact,
|
||||
run_ref=run_ref,
|
||||
evidence_manifest=manifest,
|
||||
)
|
||||
_assert_error(rejected, code, path)
|
||||
|
||||
|
||||
def test_public_mapping_has_no_raw_inputs_storage_or_runtime_authority() -> None:
|
||||
evidence, *_ = _case("estimable")
|
||||
payload = evidence.to_dict()
|
||||
serialized = evidence.to_json().lower()
|
||||
forbidden_keys = {
|
||||
"parameters",
|
||||
"params_json",
|
||||
"returns",
|
||||
"nav",
|
||||
"benchmark_series",
|
||||
"table_bytes",
|
||||
"locator",
|
||||
"uri",
|
||||
"credential",
|
||||
"decision_eligible",
|
||||
"publication_eligible",
|
||||
"paper_trading",
|
||||
"live_trading",
|
||||
"investment_advice",
|
||||
}
|
||||
|
||||
def keys(value: object) -> set[str]:
|
||||
if isinstance(value, dict):
|
||||
return set(value) | {key for item in value.values() for key in keys(item)}
|
||||
if isinstance(value, list):
|
||||
return {key for item in value for key in keys(item)}
|
||||
return set()
|
||||
|
||||
assert not (keys(payload) & forbidden_keys)
|
||||
for token in ("postgres://", "mysql://", "s3://", "credential", "broker"):
|
||||
assert token not in serialized
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,311 @@
|
||||
"""Fresh synthetic artifact assembly; no reuse or relabelling of real outputs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
EvidenceQualification,
|
||||
PerformanceEvidenceError,
|
||||
ResearchRunArtifact,
|
||||
build_research_run_artifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_performance_evidence,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.research_pipeline import run_factor_backtest_research
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
build_retrospective_performance_evidence,
|
||||
RetrospectiveBacktestEvidenceManifest,
|
||||
RetrospectivePerformanceEvidence,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import identify, replace_at
|
||||
|
||||
CONTRACT_ERRORS = (FactorContractError, BacktestContractError, PerformanceEvidenceError)
|
||||
|
||||
|
||||
def synthetic_artifact(run: RetrospectiveBacktestRunRef) -> ResearchRunArtifact:
|
||||
# Artifact-envelope tests, not an end-to-end proof of factor/source authenticity.
|
||||
# The existing financial methods receive new, in-memory synthetic matrices.
|
||||
dates = pd.date_range("2018-01-02", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"SIM0": [2.0, 0.0], "SIM1": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"SIM0": [10.0, 10.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 20.0, 21.0]}, index=dates
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"SIM0": [10.0, 12.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 18.0, 21.0]}, index=dates
|
||||
)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0
|
||||
),
|
||||
)
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02], index=result.returns.index, name="benchmark_return"
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id=run.run_id,
|
||||
strategy_id=run.strategy_id,
|
||||
strategy_name="Synthetic Top 1",
|
||||
strategy_version=run.strategy_version,
|
||||
engine_version="0.1.0",
|
||||
code_revision=run.code_revision,
|
||||
data_snapshot_id=run.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at=run.evaluation_at,
|
||||
finished_at=run.computed_at,
|
||||
parameters={"lag_sessions": 1, "top_k": 1},
|
||||
benchmark_id="synthetic.benchmark",
|
||||
benchmark_returns=benchmark,
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_closes_all_nine_existing_tables_without_changing_their_schema() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
wire = manifest.to_dict()
|
||||
assert wire["schema_version"] == "2.0.0"
|
||||
assert wire["artifact_schema_version"] == "1.1.0"
|
||||
assert wire["run_id"] == run.run_id
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["execution_validation"] == "not_validated"
|
||||
assert wire["decision_eligible"] is False
|
||||
assert manifest.manifest_id.startswith("rhbacktestevidencev2:sha256:")
|
||||
assert len({table.logical_name for item in manifest.evidence for table in item.tables}) == 9
|
||||
|
||||
|
||||
def test_performance_v2_keeps_existing_metric_methods_and_binds_all_upstreams() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
wire = evidence.to_dict()
|
||||
assert wire["schema_version"] == "researchhub.performance-evidence.v2"
|
||||
assert wire["methodology_id"] == "researchhub.quant-performance-methodology.v1"
|
||||
assert wire["metric_schema_id"] == "researchhub.quant-performance-metrics.v1"
|
||||
assert wire["research_artifact_schema_version"] == "1.1.0"
|
||||
assert wire["backtest_run_ref_id"] == run.run_id
|
||||
assert wire["backtest_evidence_manifest_id"] == manifest.manifest_id
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["start_date"] == "2018-01-02"
|
||||
assert wire["end_date"] == "2018-01-05"
|
||||
assert wire["artifact_available_at"] == "2026-09-08T01:11:00Z"
|
||||
assert evidence.run_id == run.run_id
|
||||
assert evidence.performance_evidence_id.startswith("rhperformancev2:sha256:")
|
||||
for metric in evidence.metrics:
|
||||
if metric.value is not None:
|
||||
assert metric.value == artifact.performance.iloc[0][metric.source_column]
|
||||
assert (
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
evidence.to_json(), artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
== evidence
|
||||
)
|
||||
assert (
|
||||
RetrospectiveBacktestEvidenceManifest.from_json(
|
||||
manifest.to_json(), artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
== manifest
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("run_id", "rhbacktestrunv2:sha256:" + "0" * 64),
|
||||
("profile", "offline_research_v1"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("evidence_scope", "real_data"),
|
||||
("artifact_available_at", "2026-09-08T01:09:00Z"),
|
||||
("artifact_schema_version", "2.0.0"),
|
||||
("qualification", "legacy_exploratory"),
|
||||
("evidence_digest", "sha256:" + "0" * 64),
|
||||
("evidence.0.tables.0.row_count", True),
|
||||
("evidence.0.tables.0.content_digest", "sha256:" + "0" * 64),
|
||||
("run_reference.value.factor_set_id", "rhfactorsetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_reidentified_claims_without_actual_table_closure(
|
||||
path: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
row = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "manifest_id", "rhbacktestevidencev2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
row, artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("table", "column", "value"),
|
||||
[
|
||||
("run", "data_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("run", "config_hash", "0" * 64),
|
||||
("run", "code_revision", "0" * 40),
|
||||
("run", "started_at", "2018-01-02T07:00:00Z"),
|
||||
("run", "finished_at", "2026-09-08T01:12:00Z"),
|
||||
("signals", "asset_id", "/private/data.csv"),
|
||||
("nav", "run_id", "old.run"),
|
||||
("performance", "run_id", "old.run"),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_artifact_identity_time_or_private_data_mismatch(
|
||||
table: str, column: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
frame = getattr(artifact, table)
|
||||
frame.loc[frame.index[0], column] = value
|
||||
forged = replace(artifact, **{"_" + table: frame})
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run, forged, artifact_available_at="2026-09-08T01:13:00Z"
|
||||
)
|
||||
|
||||
|
||||
def test_each_table_is_reconciled_and_legacy_apis_cannot_accept_v2() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
for item in manifest.evidence:
|
||||
for table in item.tables:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
expected_table_digests={table.logical_name: "sha256:" + "0" * 64},
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_performance_evidence(artifact, run, manifest)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
qualification=EvidenceQualification.LEGACY_EXPLORATORY,
|
||||
)
|
||||
|
||||
|
||||
def seal_performance(row: dict[str, Any]) -> None:
|
||||
def sha(document: Any) -> str:
|
||||
return (
|
||||
"sha256:"
|
||||
+ hashlib.sha256(
|
||||
json.dumps(
|
||||
document,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode()
|
||||
).hexdigest()
|
||||
)
|
||||
|
||||
row.pop("document_sha256", None)
|
||||
row.pop("performance_evidence_id", None)
|
||||
row["performance_evidence_id"] = "rhperformancev2:" + sha(row)
|
||||
row["document_sha256"] = sha(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "researchhub.performance-evidence.v1"),
|
||||
("scope", "live"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("evidence_scope", "real_data"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("backtest_run_ref_document_sha256", "sha256:" + "0" * 64),
|
||||
("backtest_evidence_manifest_id", "rhbacktestevidencev2:sha256:" + "0" * 64),
|
||||
("methodology.periods_per_year", 365),
|
||||
("metric_schema_id", "new.metric"),
|
||||
("metrics.0.value", 0.0),
|
||||
("metrics.0.nullable", True),
|
||||
("start_date", "2017-01-01"),
|
||||
("artifact_available_at", "2018-01-02T07:00:00Z"),
|
||||
],
|
||||
)
|
||||
def test_performance_never_accepts_reidentified_changed_facts(path: str, value: Any) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
row = build_retrospective_performance_evidence(artifact, run, manifest).to_dict()
|
||||
replace_at(row, path, value)
|
||||
seal_performance(row)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_dict(
|
||||
row, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
|
||||
|
||||
def test_performance_has_immutable_finite_canonical_payload_and_current_tables() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
assert evidence.document_sha256.startswith("sha256:")
|
||||
exported = evidence.to_dict()
|
||||
exported["metrics"][0]["value"] = 9.0
|
||||
assert evidence.to_dict()["metrics"][0]["value"] != 9.0
|
||||
for data in (
|
||||
evidence.to_json() + "\n",
|
||||
'{"schema_version":"x",' + evidence.to_json()[1:],
|
||||
"null",
|
||||
"{bad",
|
||||
):
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
data, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
frame = artifact.performance
|
||||
frame.loc[0, "total_ret"] = 0.0
|
||||
forged = replace(artifact, _performance=frame)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_performance_evidence(forged, run, manifest)
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Offline synthetic v2 backtest evidence and replay boundaries."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import FactorContractError, PayloadValidation
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef
|
||||
from test_retrospective_data_contracts import digest, identify, replace_at
|
||||
from test_retrospective_factor_contracts import factor_arguments, decoding_arguments
|
||||
|
||||
|
||||
def run_arguments() -> dict[str, Any]:
|
||||
arguments = factor_arguments()
|
||||
factor = RetrospectiveFactorSetRef.create(**arguments)
|
||||
view = next(iter(arguments["foundation"].views.values()))
|
||||
return {
|
||||
"dataset_snapshot": arguments["dataset_snapshot"],
|
||||
"foundation": arguments["foundation"],
|
||||
"factor_set": factor,
|
||||
"universe_digest": digest({"synthetic_universe": 2}),
|
||||
"trading_calendar_revision_ids": view.trading_calendar_revision_ids,
|
||||
"corporate_action_revision_ids": view.corporate_action_revision_ids,
|
||||
"strategy_id": "synthetic.top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": digest({"synthetic_strategy": "top1"}),
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": digest({"synthetic_execution": 1}),
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": digest({"synthetic_cost": 1}),
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": digest({"synthetic_lock": 1}),
|
||||
"configuration_digest": digest({"lag_sessions": 1, "top_k": 1}),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
}
|
||||
|
||||
|
||||
def run_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("dataset_snapshot", "foundation", "factor_set")}
|
||||
|
||||
|
||||
def test_run_identity_closes_observation_inputs_and_preserves_configurations() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
document = run.to_dict()
|
||||
assert document["schema_version"] == "2.0.0"
|
||||
assert run.run_id.startswith("rhbacktestrunv2:sha256:")
|
||||
assert run.dataset_snapshot_id == arguments["dataset_snapshot"].snapshot_id
|
||||
assert run.foundation_id == arguments["foundation"].foundation_id
|
||||
assert run.factor_set_id == arguments["factor_set"].factor_set_id
|
||||
assert document["usage"] == "retrospective_research"
|
||||
assert document["historical_availability"] == "not_established"
|
||||
assert document["decision_eligible"] is False
|
||||
assert document["execution_validation"] == "not_validated"
|
||||
assert document["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert document["replay_attempt"] == 0
|
||||
assert RetrospectiveBacktestRunRef.from_json(run.to_json(), **run_context(arguments)) == run
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("historical_availability", "established"),
|
||||
("usage", "as_available"),
|
||||
("execution_validation", "validated"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("dataset_content_digest", "sha256:" + "0" * 64),
|
||||
("foundation_digest", "sha256:" + "0" * 64),
|
||||
("factor_set_digest", "sha256:" + "0" * 64),
|
||||
("factor_output_content_digest", "sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("evidence_scope", "real_data"),
|
||||
("trading_calendar_revision_ids", []),
|
||||
("corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("evaluation_at", "2026-09-08T01:07:00Z"),
|
||||
("computed_at", "2026-09-08T01:08:00Z"),
|
||||
("computed_at", "2026-09-08T01:10:00.0000001Z"),
|
||||
("random_seed", True),
|
||||
("strategy_version", "latest"),
|
||||
("configuration_digest", "../private/a"),
|
||||
("code_revision", "unknown"),
|
||||
("replay_attempt", 1),
|
||||
("replay_reason", "retry"),
|
||||
("replay_spec_digest", "sha256:" + "0" * 64),
|
||||
("replay_ancestor_run_ids", ["rhbacktestrunv2:sha256:" + "0" * 64]),
|
||||
],
|
||||
)
|
||||
def test_reidentified_run_must_match_exact_context(path: str, value: Any) -> None:
|
||||
arguments = run_arguments()
|
||||
row = RetrospectiveBacktestRunRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "run_id", "rhbacktestrunv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.from_dict(row, **run_context(arguments))
|
||||
|
||||
|
||||
def test_replay_keeps_input_spec_but_requires_new_actual_attempt_times() -> None:
|
||||
arguments = run_arguments()
|
||||
root = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
replay_args = {
|
||||
**arguments,
|
||||
"parent": root,
|
||||
"replay_reason": "synthetic.retry",
|
||||
"replay_attempt": 1,
|
||||
"evaluation_at": "2026-09-08T01:12:00Z",
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
replay = RetrospectiveBacktestRunRef.create(**replay_args)
|
||||
assert replay.replay_spec_digest == root.replay_spec_digest
|
||||
assert replay.run_id != root.run_id
|
||||
assert replay.replay_ancestor_run_ids == (root.run_id,)
|
||||
assert replay.evaluation_at != root.evaluation_at
|
||||
assert (
|
||||
RetrospectiveBacktestRunRef.from_json(
|
||||
replay.to_json(), **run_context(arguments), parent=root
|
||||
)
|
||||
== replay
|
||||
)
|
||||
for changes in (
|
||||
{"random_seed": 9},
|
||||
{"configuration_digest": digest({"different_configuration": 1})},
|
||||
{"evaluation_at": root.evaluation_at},
|
||||
{"replay_attempt": 2},
|
||||
{"replay_reason": None},
|
||||
{"parent": None},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**replay_args, **changes})
|
||||
|
||||
|
||||
def test_reference_only_factors_can_be_read_but_not_used_to_create_new_runs() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
factor = arguments["factor_set"]
|
||||
reference = RetrospectiveFactorSetRef.from_dict(
|
||||
factor.to_dict(),
|
||||
definitions=factor._definitions,
|
||||
dataset_snapshot=arguments["dataset_snapshot"],
|
||||
foundation=arguments["foundation"],
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**arguments, "factor_set": reference})
|
||||
restored = RetrospectiveBacktestRunRef.from_dict(
|
||||
run.to_dict(), **{**run_context(arguments), "factor_set": reference}
|
||||
)
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
with pytest.raises(FactorContractError):
|
||||
restored.require_inputs_revalidated()
|
||||
run.require_inputs_revalidated()
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Frozen synthetic interoperability vector, not end-to-end source provenance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from quant_engine.artifact import _evidence_frame_records
|
||||
from quant_engine.retrospective_artifact_contracts import build_retrospective_performance_evidence
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_factor_contracts import factor_arguments
|
||||
from test_retrospective_portfolio_risk_contracts import portfolio_arguments, risk_arguments
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
VECTOR = ROOT / "tests" / "fixtures" / "retrospective-computation-v2.golden.json"
|
||||
|
||||
|
||||
def build_vector() -> dict[str, Any]:
|
||||
portfolio = portfolio_arguments()
|
||||
risk = risk_arguments(portfolio)
|
||||
run = portfolio["backtest_run_ref"]
|
||||
manifest = portfolio["manifest"]
|
||||
artifact = manifest._artifact
|
||||
factor = factor_arguments()
|
||||
return {
|
||||
"fixture_kind": "synthetic_retrospective_contract_vector",
|
||||
"artifact_data_provenance": "envelope_test_only_not_end_to_end",
|
||||
"source_authenticity": "not_established",
|
||||
"factor_definitions": [item.to_dict() for item in factor["definitions"]],
|
||||
"dataset_chunks": factor["dataset_chunks"],
|
||||
"resolved_view_schema": {"synthetic_schema": "neutral_close_v2"},
|
||||
"factor_output_schema": json.loads(factor["output_schema_bytes"]),
|
||||
"factor_output_records": json.loads(factor["output_content_bytes"]),
|
||||
"factor_set": run._factor_set.to_dict(),
|
||||
"backtest_run_ref": run.to_dict(),
|
||||
"artifact_tables": {
|
||||
name: _evidence_frame_records(frame, name)
|
||||
for name, frame in artifact.table_frames().items()
|
||||
},
|
||||
"backtest_evidence_manifest": manifest.to_dict(),
|
||||
"performance_evidence": build_retrospective_performance_evidence(
|
||||
artifact, run, manifest
|
||||
).to_dict(),
|
||||
"portfolio_target": portfolio["target"].to_dict(),
|
||||
"portfolio_decision": risk["portfolio_decision"].to_dict(),
|
||||
"risk_assessment": assess_retrospective_portfolio_risk(**risk).to_dict(),
|
||||
"covariance_matrix": risk["covariance"].covariance.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
def test_synthetic_vector_matches_all_current_owner_serializers() -> None:
|
||||
expected = VECTOR.read_text(encoding="utf-8")
|
||||
actual = (
|
||||
json.dumps(build_vector(), ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False)
|
||||
+ "\n"
|
||||
)
|
||||
assert actual == expected
|
||||
|
||||
|
||||
def test_v2_modules_do_not_import_data_owners_publishers_or_execution_authority() -> None:
|
||||
modules = sorted((ROOT / "src" / "quant_engine").glob("retrospective_*_contracts.py"))
|
||||
assert len(modules) == 5
|
||||
for path in modules:
|
||||
tree = ast.parse(path.read_text(encoding="utf-8"))
|
||||
imports = {
|
||||
alias.name
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Import)
|
||||
for alias in node.names
|
||||
} | {node.module or "" for node in ast.walk(tree) if isinstance(node, ast.ImportFrom)}
|
||||
assert not any(
|
||||
name.startswith(("research_results", "research_platform", "edb_data_core"))
|
||||
for name in imports
|
||||
)
|
||||
called = {
|
||||
node.func.id
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
|
||||
}
|
||||
assert not called & {
|
||||
"create_paper_order_intent",
|
||||
"run_governed_factor_slice",
|
||||
"evaluate_portfolio_risk",
|
||||
}
|
||||
@@ -0,0 +1,636 @@
|
||||
"""QE-owned decoding tests against RP's synthetic public v2 golden vectors."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from copy import deepcopy
|
||||
from dataclasses import FrozenInstanceError
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorContractError,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
|
||||
def golden(kind: str) -> dict[str, Any]:
|
||||
return json.loads((FIXTURES / f"retrospective-{kind}-v2.golden.json").read_text())
|
||||
|
||||
|
||||
def digest(value: Any) -> str:
|
||||
return "sha256:" + hashlib.sha256(canonical_json_bytes(value)).hexdigest()
|
||||
|
||||
|
||||
def identify(value: dict[str, Any], field: str, prefix: str) -> None:
|
||||
value[field] = prefix + digest({key: item for key, item in value.items() if key != field})
|
||||
|
||||
|
||||
def records() -> list[dict[str, Any]]:
|
||||
return [
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "1" * 32,
|
||||
"metric": "close",
|
||||
"value": "101.25",
|
||||
},
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "2" * 32,
|
||||
"metric": "close",
|
||||
"value": "87.50",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def bind_records(source: dict[str, Any], chunks: list[list[dict[str, Any]]]) -> None:
|
||||
def records_digest(rows: list[dict[str, Any]]) -> str:
|
||||
data = b"[" + b",".join(sorted(canonical_json_bytes(row) for row in rows)) + b"]"
|
||||
return "sha256:" + hashlib.sha256(data).hexdigest()
|
||||
|
||||
manifest = {
|
||||
"record_count": sum(len(rows) for rows in chunks),
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": index,
|
||||
"content_digest": records_digest(rows),
|
||||
"record_count": len(rows),
|
||||
}
|
||||
for index, rows in enumerate(chunks)
|
||||
],
|
||||
}
|
||||
source["descriptor"]["content"].update(
|
||||
{
|
||||
"record_count": manifest["record_count"],
|
||||
"logical_manifest": manifest,
|
||||
"manifest_digest": digest(manifest),
|
||||
"content_digest": records_digest([row for chunk in chunks for row in chunk]),
|
||||
}
|
||||
)
|
||||
source["descriptor"]["observation_manifest"]["batches"] = [
|
||||
{
|
||||
**chunk,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_receipt": index}),
|
||||
}
|
||||
for index, chunk in enumerate(manifest["chunks"])
|
||||
]
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
|
||||
|
||||
def replace_at(source: dict[str, Any], path: str, value: Any) -> None:
|
||||
target: Any = source
|
||||
keys = path.split(".")
|
||||
for key in keys[:-1]:
|
||||
target = target[int(key)] if isinstance(target, list) else target[key]
|
||||
target[int(keys[-1]) if isinstance(target, list) else keys[-1]] = value
|
||||
|
||||
|
||||
COLLECTIONS = (
|
||||
(
|
||||
"instrument_routes",
|
||||
"route_revision_id",
|
||||
"rhroutev2:",
|
||||
"instrument_route",
|
||||
"instrument_route_revision_ids",
|
||||
),
|
||||
(
|
||||
"trading_calendar_revisions",
|
||||
"calendar_revision_id",
|
||||
"rhcalv2:",
|
||||
"trading_calendar",
|
||||
"trading_calendar_revision_ids",
|
||||
),
|
||||
(
|
||||
"corporate_action_revisions",
|
||||
"action_revision_id",
|
||||
"rhcav2:",
|
||||
"corporate_action",
|
||||
"corporate_action_revision_ids",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def seal_foundation(source: dict[str, Any], *, rebuild_lineage: bool = True) -> None:
|
||||
lineage = []
|
||||
for name, key, prefix, kind, view_key in COLLECTIONS:
|
||||
replacements = {}
|
||||
for row in sorted(source[name], key=lambda row: row["observation_sequence"]):
|
||||
old = row[key]
|
||||
if "supersedes_observation_id" in row:
|
||||
row["supersedes_observation_id"] = replacements.get(
|
||||
row["supersedes_observation_id"], row["supersedes_observation_id"]
|
||||
)
|
||||
identify(row, key, prefix)
|
||||
replacements[old] = row[key]
|
||||
lineage.append(
|
||||
{
|
||||
"revision_kind": kind,
|
||||
"revision_id": row[key],
|
||||
**{
|
||||
field: row[field]
|
||||
for field in (
|
||||
"observation_sequence",
|
||||
"observed_by",
|
||||
"earliest_external_knowledge",
|
||||
"history_completeness",
|
||||
"evidence_digest",
|
||||
"supersedes_observation_id",
|
||||
)
|
||||
if field in row
|
||||
},
|
||||
}
|
||||
)
|
||||
for view in source["standardized_views"]:
|
||||
view[view_key] = [replacements.get(item, item) for item in view[view_key]]
|
||||
if rebuild_lineage:
|
||||
source["observation_lineage"] = lineage
|
||||
for view in source["standardized_views"]:
|
||||
identify(view, "view_ref_id", "rhviewrefv2:")
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
|
||||
|
||||
def parse_foundation(source: dict[str, Any]) -> RetrospectiveFoundationEnvelope:
|
||||
return RetrospectiveFoundationEnvelope.from_dict(
|
||||
source, snapshot=RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
)
|
||||
|
||||
|
||||
def test_public_snapshot_golden_is_an_explicit_observation_contract() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert snapshot.to_dict() == source
|
||||
assert snapshot.snapshot_id == source["snapshot_id"]
|
||||
assert snapshot.observation_cutoff == "2026-09-08T01:01:00Z"
|
||||
assert snapshot.evidence_scope == "synthetic_fixture"
|
||||
assert snapshot.earliest_external_knowledge == {"status": "unknown"}
|
||||
assert not hasattr(snapshot, "pit_cutoff")
|
||||
assert not hasattr(snapshot, "knowledge_time")
|
||||
snapshot.require_qualified()
|
||||
assert RetrospectiveSnapshotEnvelope.from_json(snapshot.to_json()) == snapshot
|
||||
|
||||
|
||||
def test_public_foundation_golden_binds_exact_typed_snapshot() -> None:
|
||||
source = golden("data-foundation")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert foundation.to_dict() == source
|
||||
assert foundation.dataset_snapshot_id == snapshot.snapshot_id
|
||||
assert foundation.observation_cutoff == snapshot.observation_cutoff
|
||||
assert foundation.evidence_scope == snapshot.evidence_scope
|
||||
assert foundation.real_data_validation_status == "not_validated"
|
||||
assert not hasattr(foundation, "pit_cutoff")
|
||||
assert (
|
||||
RetrospectiveFoundationEnvelope.from_json(foundation.to_json(), snapshot=snapshot)
|
||||
== foundation
|
||||
)
|
||||
|
||||
|
||||
def test_empty_logical_dimension_is_rejected_after_rebinding_all_bytes() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["instrument_id"] = ""
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError, match="dimension"):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_boolean_lineage_sequence_does_not_equal_integer_one() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["observation_lineage"][0]["observation_sequence"] = True
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("descriptor.time_semantics.pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.knowledge_time", {"start_inclusive": "2018-01-02T07:00:00Z"}),
|
||||
("descriptor.time_semantics.historical_availability", "established"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T01:01:00.0000001Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T09:01:00+08:00"),
|
||||
(
|
||||
"descriptor.time_semantics.earliest_external_knowledge",
|
||||
{"status": "unknown", "evidence_digest": "sha256:" + "0" * 64},
|
||||
),
|
||||
("descriptor.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("descriptor.published_at", "2026-02-30T00:00:00Z"),
|
||||
("descriptor.published_at", "2026-09-08T01:00:00Z"),
|
||||
("descriptor.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.qualification.usage", "as_available"),
|
||||
("descriptor.qualification.policy_version", "1.0.0"),
|
||||
("descriptor.quality.checks.0.severity", "advisory"),
|
||||
("descriptor.quality.checks.0.status", "failed"),
|
||||
("descriptor.quality.checks.0.check_id", "schema_conformance"),
|
||||
("descriptor.quality.status", "failed"),
|
||||
("descriptor.content.record_count", True),
|
||||
("descriptor.content.record_count", 9007199254740992),
|
||||
("descriptor.content.record_count", 2.0),
|
||||
("descriptor.content.content_digest", "bad"),
|
||||
("descriptor.content.manifest_digest", "sha256:" + "0" * 64),
|
||||
("descriptor.observation_manifest.batches", []),
|
||||
("descriptor.observation_manifest.batches.0.record_count", 1),
|
||||
("descriptor.observation_manifest.batches.0.chunk_index", True),
|
||||
("descriptor.observation_manifest.batches.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("descriptor.observation_manifest.batches.0.observation_kind", "first_published_at"),
|
||||
("descriptor.lineage.transformation.id", "rhtransform:private"),
|
||||
("descriptor.dataset.dimensions", ["instrument_id", "knowledge_time"]),
|
||||
("descriptor.dataset.dataset_id", "rhdataset:macroeconomic:" + "4" * 32),
|
||||
],
|
||||
)
|
||||
def test_snapshot_rejects_reidentified_invalid_declarations(path: str, value: Any) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
replace_at(source, path, value)
|
||||
# Noncanonical numbers are rejected before identity formation.
|
||||
if type(value) is not float and value != 9007199254740992:
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_snapshot_materialized_records_bind_the_public_golden_and_chunks() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
snapshot.verify_materialized_records([records()])
|
||||
snapshot.verify_materialized_records([list(reversed(records()))])
|
||||
chunks = [[records()[0]], [records()[1]]]
|
||||
bind_records(source, chunks)
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records(chunks)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records(chunks)
|
||||
rows = records()
|
||||
rows[0]["value"] = "0"
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"mutation", ["duplicate", "legacy", "range", "location", "missing_effective"]
|
||||
)
|
||||
def test_materialized_bad_records_rejected_even_with_matching_digests(mutation: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
if mutation == "duplicate":
|
||||
rows.append(deepcopy(rows[0]))
|
||||
elif mutation == "legacy":
|
||||
rows[0]["knowledge_time"] = "2026-09-08T01:00:00Z"
|
||||
elif mutation == "range":
|
||||
rows[0]["effective_time"] = "2018-01-01T07:00:00Z"
|
||||
elif mutation == "location":
|
||||
rows[0]["value"] = "/private/records.csv"
|
||||
else:
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_unknown_or_evidenced_knowledge_never_changes_usage() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"] = {
|
||||
"status": "evidenced",
|
||||
"range": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z",
|
||||
},
|
||||
"evidence_digest": digest({"synthetic_earliest": True}),
|
||||
}
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
parsed = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert (
|
||||
parsed.to_dict()["descriptor"]["time_semantics"]["historical_availability"]
|
||||
== "not_established"
|
||||
)
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"]["range"][
|
||||
"end_inclusive"
|
||||
] = "2026-09-08T01:01:00Z"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_rejected_snapshot_remains_readable_but_cannot_support_foundation() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["qualification"]["status"] = "rejected"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.require_qualified()
|
||||
foundation = golden("data-foundation")
|
||||
foundation["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in foundation["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(foundation, snapshot=snapshot)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2026-09-08T01:00:00Z"),
|
||||
("published_at", "2026-09-08T01:03:00Z"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "established"),
|
||||
("instrument_routes.0.observation_sequence", 2),
|
||||
("instrument_routes.0.supersedes_observation_id", "rhroutev2:sha256:" + "0" * 64),
|
||||
("instrument_routes.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("instrument_routes.0.history_completeness", "complete"),
|
||||
(
|
||||
"instrument_routes.0.earliest_external_knowledge",
|
||||
{"status": "unknown", "earliest_at": "2018-01-01T00:00:00Z"},
|
||||
),
|
||||
("instrument_routes.0.instrument_type", "index"),
|
||||
("instrument_routes.0.symbol", "WIND.TEST"),
|
||||
("instrument_routes.0.symbol", "A" * 33),
|
||||
("instrument_routes.0.calendar_id", "rhcalendar:" + "0" * 32),
|
||||
("trading_calendar_revisions.0.status", "closed"),
|
||||
("trading_calendar_revisions.0.sessions", []),
|
||||
("trading_calendar_revisions.0.sessions.0.closes_at", "2018-01-02T00:00:00Z"),
|
||||
("trading_calendar_revisions.0.session_date", "2018-02-30"),
|
||||
("standardized_views.0.instrument_route_revision_ids", []),
|
||||
("standardized_views.0.trading_calendar_revision_ids", []),
|
||||
("standardized_views.0.corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("standardized_views.0.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:02:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:06:00Z"),
|
||||
("standardized_views.0.usage", "as_available"),
|
||||
("corporate_action_coverage", []),
|
||||
("corporate_action_coverage.0.instrument_id", "rhinstrument:" + "0" * 32),
|
||||
("corporate_action_coverage.0.effective_time.end_inclusive", "2018-01-01T07:00:00Z"),
|
||||
("corporate_action_coverage.0.effective_time.start_inclusive", "2018-01-02T07:00:01Z"),
|
||||
("corporate_action_coverage.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("corporate_action_coverage.0.evidence_digests", []),
|
||||
("readiness.evidence_scope", "real_data"),
|
||||
("readiness.contract_validation.evidence_digests", []),
|
||||
(
|
||||
"readiness.real_data_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.production_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.live_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_foundation_rejects_semantic_forgery_after_reidentification(path: str, value: Any) -> None:
|
||||
source = golden("data-foundation")
|
||||
replace_at(source, path, value)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_v1_and_v2_never_coerce_each_other() -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
DatasetSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
with pytest.raises(FactorContractError):
|
||||
DataFoundationEnvelope.from_dict(golden("data-foundation"))
|
||||
old = json.loads((FIXTURES / "factor-contracts-v1.golden.json").read_text())
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(old["dataset_snapshot"])
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(old["data_foundation"])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"),
|
||||
snapshot=DatasetSnapshotEnvelope.from_dict(old["dataset_snapshot"]),
|
||||
)
|
||||
|
||||
|
||||
def test_deep_immutability_and_strict_canonical_json() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
source["descriptor"]["quality"]["status"] = "failed"
|
||||
snapshot.to_dict()["descriptor"]["qualification"]["status"] = "rejected"
|
||||
snapshot.require_qualified()
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
snapshot._payload = {}
|
||||
with pytest.raises(TypeError):
|
||||
snapshot.earliest_external_knowledge["status"] = "evidenced"
|
||||
foundation = parse_foundation(golden("data-foundation"))
|
||||
with pytest.raises(TypeError):
|
||||
foundation.views["new"] = next(iter(foundation.views.values()))
|
||||
for decoder, document in (
|
||||
(RetrospectiveSnapshotEnvelope.from_json, golden("dataset-snapshot")),
|
||||
(
|
||||
lambda value: RetrospectiveFoundationEnvelope.from_json(value, snapshot=snapshot),
|
||||
golden("data-foundation"),
|
||||
),
|
||||
):
|
||||
wire = canonical_json_bytes(document)
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(wire + b"\n")
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(b'{"schema_version":"2.0.0",' + wire[1:])
|
||||
|
||||
|
||||
def test_macro_period_is_not_inferred_as_an_effective_instant() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["dataset"].update(
|
||||
dataset_id="rhdataset:macroeconomic:" + "4" * 32,
|
||||
dataset_kind="macroeconomic",
|
||||
dimensions=["series_id", "observation_period"],
|
||||
)
|
||||
rows = [
|
||||
{
|
||||
"series_id": "cpi",
|
||||
"observation_period": "2018-01",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"value": "2.1",
|
||||
}
|
||||
]
|
||||
bind_records(source, [rows])
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def with_successor() -> dict[str, Any]:
|
||||
source = golden("data-foundation")
|
||||
previous = source["instrument_routes"][0]
|
||||
successor = deepcopy(previous)
|
||||
successor.update(
|
||||
observation_sequence=2,
|
||||
observed_by="2026-09-08T01:00:30Z",
|
||||
symbol="SIM0B",
|
||||
supersedes_observation_id=previous["route_revision_id"],
|
||||
)
|
||||
identify(successor, "route_revision_id", "rhroutev2:")
|
||||
source["instrument_routes"].append(successor)
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].append(
|
||||
successor["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
return source
|
||||
|
||||
|
||||
def test_retained_successor_requires_parent_time_and_view_ancestry() -> None:
|
||||
source = with_successor()
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.foundation_id == source["foundation_id"]
|
||||
assert parsed.published_at.isoformat() == "2026-09-08T01:05:00+00:00"
|
||||
assert parsed.contract_evidence_digests
|
||||
assert len(next(iter(parsed.views.values())).instrument_route_revision_ids) == 3
|
||||
for mutation in (
|
||||
"missing_parent",
|
||||
"equal_time",
|
||||
"omitted_ancestor",
|
||||
"duplicate_sequence",
|
||||
"wrong_lineage",
|
||||
):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "missing_parent":
|
||||
forged["instrument_routes"][-1]["supersedes_observation_id"] = (
|
||||
"rhroutev2:sha256:" + "0" * 64
|
||||
)
|
||||
elif mutation == "equal_time":
|
||||
forged["instrument_routes"][-1]["observed_by"] = forged["instrument_routes"][0][
|
||||
"observed_by"
|
||||
]
|
||||
elif mutation == "omitted_ancestor":
|
||||
forged["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
forged["instrument_routes"][0]["route_revision_id"]
|
||||
)
|
||||
elif mutation == "duplicate_sequence":
|
||||
forged["instrument_routes"][-1]["observation_sequence"] = 1
|
||||
else:
|
||||
forged["observation_lineage"][0]["evidence_digest"] = "sha256:" + "0" * 64
|
||||
seal_foundation(forged, rebuild_lineage=mutation != "wrong_lineage")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(forged)
|
||||
|
||||
|
||||
def test_index_and_closed_calendar_are_explicit_non_execution_data() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["instrument_routes"][0].update(asset_class="index", instrument_type="index")
|
||||
source["trading_calendar_revisions"][0].update(status="closed", sessions=[])
|
||||
seal_foundation(source)
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.to_dict()["instrument_routes"][0]["instrument_type"] == "index"
|
||||
assert parsed.to_dict()["readiness"]["live_validation"]["status"] == "not_validated"
|
||||
|
||||
|
||||
def test_real_scope_is_still_a_declaration_with_separate_evidence_and_coverage() -> None:
|
||||
snapshot_source = golden("dataset-snapshot")
|
||||
snapshot_source["evidence_scope"] = "real_data"
|
||||
identify(snapshot_source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_source)
|
||||
source = golden("data-foundation")
|
||||
source["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in source["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
source["readiness"]["evidence_scope"] = "real_data"
|
||||
source["readiness"]["real_data_validation"] = {
|
||||
"status": "validated",
|
||||
"evidence_digests": [digest({"synthetic_real_claim_test": 1})],
|
||||
}
|
||||
seal_foundation(source)
|
||||
parsed = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert parsed.real_data_validation_status == "validated" # NOT actual real-data evidence.
|
||||
for mutation in ("coverage", "reuse"):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "coverage":
|
||||
forged["corporate_action_coverage"][0].update(
|
||||
status="not_validated", evidence_digests=[]
|
||||
)
|
||||
else:
|
||||
forged["readiness"]["real_data_validation"] = deepcopy(
|
||||
forged["readiness"]["contract_validation"]
|
||||
)
|
||||
seal_foundation(forged)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(forged, snapshot=snapshot)
|
||||
synthetic = golden("data-foundation")
|
||||
synthetic["corporate_action_coverage"][0].update(status="not_validated", evidence_digests=[])
|
||||
seal_foundation(synthetic)
|
||||
assert parse_foundation(synthetic).real_data_validation_status == "not_validated"
|
||||
|
||||
|
||||
def test_action_must_belong_to_view_selected_instrument() -> None:
|
||||
source = golden("data-foundation")
|
||||
route = source["instrument_routes"][0]
|
||||
action = {
|
||||
"action_id": "rhaction:" + "7" * 32,
|
||||
"instrument_id": route["instrument_id"],
|
||||
"observation_sequence": 1,
|
||||
"observed_by": route["observed_by"],
|
||||
"earliest_external_knowledge": {
|
||||
"status": "evidenced",
|
||||
"earliest_at": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_action_earliest": 1}),
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": digest({"synthetic_action": 1}),
|
||||
"action_type": "cash_dividend",
|
||||
"status": "confirmed",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"terms_digest": digest({"synthetic_terms": 1}),
|
||||
}
|
||||
identify(action, "action_revision_id", "rhcav2:")
|
||||
source["corporate_action_revisions"] = [action]
|
||||
source["standardized_views"][0]["corporate_action_revision_ids"] = [
|
||||
action["action_revision_id"]
|
||||
]
|
||||
seal_foundation(source)
|
||||
assert len(parse_foundation(source).to_dict()["corporate_action_revisions"]) == 1
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
route["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_views_cannot_borrow_calendar_selection_from_each_other() -> None:
|
||||
source = golden("data-foundation")
|
||||
calendar = deepcopy(source["trading_calendar_revisions"][0])
|
||||
calendar["calendar_id"] = "rhcalendar:" + "9" * 32
|
||||
identify(calendar, "calendar_revision_id", "rhcalv2:")
|
||||
source["trading_calendar_revisions"].append(calendar)
|
||||
view = deepcopy(source["standardized_views"][0])
|
||||
view["view_id"] = "rhview:" + "8" * 32
|
||||
view["trading_calendar_revision_ids"] = [calendar["calendar_revision_id"]]
|
||||
source["standardized_views"].append(view)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"location", ["/private/a", "./a", "../a", "C:\\data\\a", "\\\\host\\a", "s3://private/a"]
|
||||
)
|
||||
def test_materialized_values_cannot_carry_physical_locations(location: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["value"] = location
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
@@ -0,0 +1,367 @@
|
||||
"""Synthetic v2 computation boundaries; never source authentication."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
FactorContractError,
|
||||
FactorDefinition,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
canonical_json_bytes,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
from quant_engine.retrospective_factor_contracts import (
|
||||
ResolvedRetrospectiveView,
|
||||
RetrospectiveCausation,
|
||||
RetrospectiveFactorSetRef,
|
||||
RetrospectiveInputBinding,
|
||||
RetrospectiveViewAvailability,
|
||||
)
|
||||
from test_retrospective_data_contracts import (
|
||||
digest,
|
||||
golden,
|
||||
identify,
|
||||
records,
|
||||
replace_at,
|
||||
seal_foundation,
|
||||
)
|
||||
|
||||
|
||||
def factor_arguments() -> dict[str, Any]:
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"), snapshot=snapshot
|
||||
)
|
||||
view = next(iter(foundation.views.values()))
|
||||
factor_inputs = (FactorInput("market", view.schema_digest, ("value",)),)
|
||||
definition = FactorDefinition.create(
|
||||
factor_id="neutral_close",
|
||||
version="1.0.0",
|
||||
formula="value",
|
||||
parameters={},
|
||||
implementation_digest=digest({"synthetic_formula": "identity"}),
|
||||
input_schema_digest=factor_input_schema_digest(factor_inputs),
|
||||
inputs=factor_inputs,
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=0,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "0.1.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
schema = {"fields": ["instrument_id", "value"]}
|
||||
output = [{"instrument_id": row["instrument_id"], "value": row["value"]} for row in records()]
|
||||
return {
|
||||
"definitions": (definition,),
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"selected_view_ref_ids": (view.view_ref_id,),
|
||||
"input_bindings": (
|
||||
RetrospectiveInputBinding(
|
||||
definition.definition_id, "market", view.view_ref_id, view.schema_digest
|
||||
),
|
||||
),
|
||||
"view_availability": (
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view_receipt": 1})
|
||||
),
|
||||
),
|
||||
"dataset_chunks": [records()],
|
||||
"resolved_views": (
|
||||
ResolvedRetrospectiveView(
|
||||
view.view_ref_id,
|
||||
canonical_json_bytes({"synthetic_schema": "neutral_close_v2"}),
|
||||
canonical_json_bytes(sorted(records(), key=canonical_json_bytes)),
|
||||
),
|
||||
),
|
||||
"output_quality": OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", digest({"synthetic_quality": 1})),),
|
||||
),
|
||||
"output_coverage": OutputCoverage(
|
||||
"complete", 2, 2, "records", "synthetic_market", digest({"synthetic_coverage": 1})
|
||||
),
|
||||
"output_schema_bytes": canonical_json_bytes(schema),
|
||||
"output_content_bytes": canonical_json_bytes(output),
|
||||
"output_artifact_ref": OutputArtifactRef.create(
|
||||
schema_digest=digest(schema), content_digest=digest(output)
|
||||
),
|
||||
"evaluation_at": "2026-09-08T01:06:00Z",
|
||||
"computed_at": "2026-09-08T01:07:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:08:00Z",
|
||||
"producer": ProducerIdentity("quant_engine", "0.1.0"),
|
||||
"code_revision": "d" * 40,
|
||||
"actor": ActorIdentity("service", "synthetic.research"),
|
||||
"correlation_id": "synthetic.retrospective",
|
||||
"causation": RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def decoding_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("definitions", "dataset_snapshot", "foundation")}
|
||||
|
||||
|
||||
def test_factor_result_closes_v2_inputs_and_preserves_v1_definition() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
wire = result.to_dict()
|
||||
assert result.schema_version == "2.0.0"
|
||||
assert result.factor_set_id.startswith("rhfactorsetv2:sha256:")
|
||||
assert result.definition_ids == (arguments["definitions"][0].definition_id,)
|
||||
assert result.definition_ids[0].startswith("rhfactorv1:")
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["availability_mode"] == "retrospective_replay"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert wire["decision_eligible"] is False
|
||||
assert "pit_cutoff" not in wire
|
||||
assert result.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
assert result.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
restored = RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json(), **decoding_arguments(arguments)
|
||||
)
|
||||
assert restored.to_dict() == wire
|
||||
assert restored.payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("contract_name", "researchhub.dataset-snapshot"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("foundation_id", "rhdfv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("selected_view_ref_ids", []),
|
||||
("selected_view_ref_ids", ["rhviewrefv2:sha256:" + "0" * 64]),
|
||||
("definition_ids", ["rhfactorv1:sha256:" + "0" * 64]),
|
||||
("input_bindings", []),
|
||||
("input_bindings.0.input_name", "volume"),
|
||||
("input_bindings.0.schema_digest", "sha256:" + "0" * 64),
|
||||
("input_bindings.0.view_ref_id", "rhviewrefv2:sha256:" + "0" * 64),
|
||||
("view_availability", []),
|
||||
("view_availability.0.available_at", "2026-09-08T01:03:30Z"),
|
||||
("view_availability.0.available_at", "2026-09-08T01:04:00.0000001Z"),
|
||||
("upstream_evidence.quality.checks.0.status", "failed"),
|
||||
("upstream_evidence.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("upstream_evidence.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("evidence_scope", "real_data"),
|
||||
("output_quality.status", "failed"),
|
||||
("output_quality.checks.0.status", "failed"),
|
||||
("output_coverage.status", "incomplete"),
|
||||
("output_coverage.observed_count", 1),
|
||||
("output_schema_digest", "sha256:" + "0" * 64),
|
||||
("output_artifact_ref.artifact_id", "rhfactoroutputv1:sha256:" + "0" * 64),
|
||||
("availability_mode", "as_available"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "declared_as_available"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("evaluation_at", "2018-01-02T07:00:00Z"),
|
||||
("evaluation_at", "2026-09-08T01:04:00Z"),
|
||||
("computed_at", "2026-09-08T01:05:00Z"),
|
||||
("artifact_available_at", "2026-09-08T01:06:00Z"),
|
||||
("producer.id", "research_platform"),
|
||||
("code_revision", "unknown"),
|
||||
("actor.id", "https://private/a"),
|
||||
("causation.id", "rhdfv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_factor_rejects_reidentified_semantic_forgery(path: str, value: Any) -> None:
|
||||
arguments = factor_arguments()
|
||||
row = RetrospectiveFactorSetRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "factor_set_id", "rhfactorsetv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(row, **decoding_arguments(arguments))
|
||||
|
||||
|
||||
def test_payload_validation_is_never_inherited_from_serialization() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
kwargs = decoding_arguments(arguments)
|
||||
reference = RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs)
|
||||
with pytest.raises(FactorContractError):
|
||||
reference.require_payloads_revalidated()
|
||||
checked = RetrospectiveFactorSetRef.from_dict(
|
||||
result.to_dict(),
|
||||
**kwargs,
|
||||
**{
|
||||
key: arguments[key]
|
||||
for key in (
|
||||
"output_schema_bytes",
|
||||
"output_content_bytes",
|
||||
"dataset_chunks",
|
||||
"resolved_views",
|
||||
)
|
||||
},
|
||||
)
|
||||
checked.require_payloads_revalidated()
|
||||
assert checked == result
|
||||
for extra in (
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"]},
|
||||
{"dataset_chunks": arguments["dataset_chunks"]},
|
||||
{"resolved_views": arguments["resolved_views"]},
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"], "output_content_bytes": b"[]"},
|
||||
{"dataset_chunks": arguments["dataset_chunks"], "resolved_views": []},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs, **extra)
|
||||
|
||||
|
||||
def test_create_verifies_actual_snapshot_and_each_resolved_view() -> None:
|
||||
for mutation in (
|
||||
"content",
|
||||
"schema",
|
||||
"snapshot",
|
||||
"duplicate_view",
|
||||
"noncanonical",
|
||||
"unknown_view",
|
||||
):
|
||||
arguments = factor_arguments()
|
||||
view = arguments["resolved_views"][0]
|
||||
if mutation == "content":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, view.schema_bytes, b"[]"),
|
||||
)
|
||||
elif mutation == "schema":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, b"{}", view.content_bytes),
|
||||
)
|
||||
elif mutation == "snapshot":
|
||||
arguments["dataset_chunks"][0][0]["value"] = "0"
|
||||
elif mutation == "duplicate_view":
|
||||
arguments["resolved_views"] = (view, view)
|
||||
elif mutation == "unknown_view":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(
|
||||
"rhviewrefv2:sha256:" + "0" * 64, view.schema_bytes, view.content_bytes
|
||||
),
|
||||
)
|
||||
else:
|
||||
arguments["output_content_bytes"] += b"\n"
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_parent_requires_exact_correlation_scope_and_actual_availability() -> None:
|
||||
arguments = factor_arguments()
|
||||
parent = RetrospectiveFactorSetRef.create(**arguments)
|
||||
child_args = {
|
||||
**arguments,
|
||||
"parent": parent,
|
||||
"causation": RetrospectiveCausation("factor_set", parent.factor_set_id),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:11:00Z",
|
||||
}
|
||||
child = RetrospectiveFactorSetRef.create(**child_args)
|
||||
assert child.factor_set_id != parent.factor_set_id
|
||||
assert (
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
child.to_json(), **decoding_arguments(arguments), parent=parent
|
||||
)
|
||||
== child
|
||||
)
|
||||
for changes in (
|
||||
{"parent": None},
|
||||
{"correlation_id": "different.correlation"},
|
||||
{"causation": RetrospectiveCausation("factor_set", "rhfactorsetv2:sha256:" + "0" * 64)},
|
||||
{"evaluation_at": "2026-09-08T01:07:59Z"},
|
||||
{"causation": arguments["causation"]},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**{**child_args, **changes})
|
||||
|
||||
|
||||
def test_definition_validity_is_checked_at_actual_evaluation() -> None:
|
||||
arguments = factor_arguments()
|
||||
arguments.update(
|
||||
evaluation_at="2027-01-01T00:00:00Z",
|
||||
computed_at="2027-01-01T00:01:00Z",
|
||||
artifact_available_at="2027-01-01T00:02:00Z",
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_factor_contract_is_immutable_and_v1_does_not_accept_it() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
exported = result.to_dict()
|
||||
exported["upstream_evidence"]["quality"]["status"] = "failed"
|
||||
assert result.upstream_evidence["quality"]["status"] == "passed"
|
||||
with pytest.raises(TypeError):
|
||||
result.upstream_evidence["quality"]["status"] = "failed"
|
||||
with pytest.raises(FactorContractError):
|
||||
FactorSetRef.from_dict(result.to_dict(), **decoding_arguments(arguments))
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json() + "\n", **decoding_arguments(arguments)
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
"rhviewrefv1:sha256:" + "0" * 64,
|
||||
"sha256:" + "0" * 64,
|
||||
)
|
||||
|
||||
|
||||
def test_missing_synthetic_or_real_readiness_cannot_be_relabelled() -> None:
|
||||
arguments = factor_arguments()
|
||||
snapshot_row = arguments["dataset_snapshot"].to_dict()
|
||||
snapshot_row["evidence_scope"] = "real_data"
|
||||
identify(snapshot_row, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_row)
|
||||
foundation_row = arguments["foundation"].to_dict()
|
||||
foundation_row["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
foundation_row["readiness"]["evidence_scope"] = "real_data"
|
||||
for view in foundation_row["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation_row)
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation_row, snapshot=snapshot)
|
||||
view = next(iter(foundation.views.values()))
|
||||
arguments.update(
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
evidence_scope="real_data",
|
||||
selected_view_ref_ids=(view.view_ref_id,),
|
||||
input_bindings=(
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
view.view_ref_id,
|
||||
view.schema_digest,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view": 1})
|
||||
),
|
||||
),
|
||||
causation=RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="real-data"):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
@@ -0,0 +1,655 @@
|
||||
"""New synthetic S4 evidence; historical valuation is not actual availability."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import FrozenInstanceError, replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
PortfolioRiskContractError,
|
||||
RiskAssessmentStatus,
|
||||
RiskFindingCode,
|
||||
)
|
||||
from quant_engine.artifact import EvidenceQualification, PerformanceEvidenceError
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.risk import ComponentRiskResult, CovarianceSnapshot, labeled_component_risk
|
||||
import quant_engine.retrospective_portfolio_risk_contracts as contracts
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
RetrospectivePortfolioDecision,
|
||||
RetrospectivePortfolioTarget,
|
||||
RetrospectiveRiskAssessment,
|
||||
build_retrospective_portfolio_decision,
|
||||
compute_retrospective_portfolio_receipt_digests,
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_artifact_contracts import synthetic_artifact
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import digest, replace_at
|
||||
|
||||
ASSETS = ("rhinstrument:" + "1" * 32, "rhinstrument:" + "2" * 32)
|
||||
CONTRACT_ERRORS = (
|
||||
FactorContractError,
|
||||
PortfolioRiskContractError,
|
||||
BacktestContractError,
|
||||
PerformanceEvidenceError,
|
||||
)
|
||||
|
||||
|
||||
def portfolio_arguments() -> dict[str, Any]:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
target = RetrospectivePortfolioTarget.create(
|
||||
backtest_run_id=run.run_id,
|
||||
dataset_snapshot_id=run.dataset_snapshot_id,
|
||||
weights={ASSETS[0]: 0.6, ASSETS[1]: 0.4},
|
||||
effective_at="2018-01-05T07:00:00Z",
|
||||
created_at="2026-09-08T01:12:00Z",
|
||||
)
|
||||
return {
|
||||
"backtest_run_ref": run,
|
||||
"manifest": manifest,
|
||||
"target": target,
|
||||
"objective_name": "synthetic_allocation",
|
||||
"objective_version": "1.0.0",
|
||||
"objective_digest": digest({"synthetic_objective": 1}),
|
||||
"model_name": "bounded_weights",
|
||||
"model_version": "1.0.0",
|
||||
"model_digest": digest({"synthetic_model": 1}),
|
||||
"expected_return_digest": digest({"synthetic_returns": 1}),
|
||||
"covariance_digest": "sha256:" + "a" * 64,
|
||||
"scenario_digest": digest({"synthetic_scenario": 1}),
|
||||
"constraints": ConstraintSetV1(
|
||||
gross_exposure_max=1.0,
|
||||
net_exposure_min=1.0,
|
||||
net_exposure_max=1.0,
|
||||
single_asset_min=0.2,
|
||||
single_asset_max=0.7,
|
||||
position_count_max=2,
|
||||
turnover_max=0.2,
|
||||
),
|
||||
"freshness_policy": FreshnessPolicy(
|
||||
max_manifest_age_seconds=3600, max_covariance_age_days=0
|
||||
),
|
||||
"prior_weights": {ASSETS[0]: 0.5, ASSETS[1]: 0.5},
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
|
||||
|
||||
def portfolio_receipt(arguments: dict[str, Any], **changes: Any) -> ComputationReceipt:
|
||||
values = compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key not in {"computed_at", "receipt"}}
|
||||
)
|
||||
return ComputationReceipt(
|
||||
**{
|
||||
"algorithm": "bounded_weights",
|
||||
"algorithm_version": "1.0.0",
|
||||
"implementation_digest": digest({"synthetic_implementation": 1}),
|
||||
"parameter_digest": digest({"synthetic_parameters": 1}),
|
||||
"input_digest": values["input_digest"],
|
||||
"constraint_digest": values["constraint_digest"],
|
||||
"output_digest": values["output_digest"],
|
||||
"status": "completed",
|
||||
"solver_required": False,
|
||||
"solver_name": None,
|
||||
"solver_version": None,
|
||||
"solver_config_digest": None,
|
||||
"iterations": None,
|
||||
"objective_value": None,
|
||||
"max_constraint_residual": values["max_constraint_residual"],
|
||||
"tolerance": 1e-12,
|
||||
"computed_at": arguments["computed_at"],
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_target_separates_historical_effective_time_from_actual_creation() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
target = arguments["target"]
|
||||
assert target.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert target.created_at == "2026-09-08T01:12:00Z"
|
||||
assert target.target_id.startswith("rhportfoliotargetv2:sha256:")
|
||||
assert target.to_dict()["usage"] == "retrospective_research"
|
||||
|
||||
|
||||
def test_portfolio_decision_preserves_constraints_and_actual_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
assert decision.decision_id.startswith("rhportfoliodecisionv2:sha256:")
|
||||
assert decision.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert decision.created_at == "2026-09-08T01:12:00Z"
|
||||
assert decision.computed_at == "2026-09-08T01:13:00Z"
|
||||
assert decision.gross_exposure == 1.0
|
||||
assert decision.position_count == 2
|
||||
assert decision.to_dict()["decision_eligible"] is False
|
||||
|
||||
|
||||
def covariance(arguments: dict[str, Any], **changes: Any) -> CovarianceSnapshot:
|
||||
return CovarianceSnapshot(
|
||||
**{
|
||||
"snapshot_id": "covariance:synthetic-retrospective",
|
||||
"as_of_date": "2018-01-05",
|
||||
"covariance": pd.DataFrame([[0.04, 0.01], [0.01, 0.09]], index=ASSETS, columns=ASSETS),
|
||||
"return_frequency": "1d",
|
||||
"periods_per_year": 252,
|
||||
"method": "provided",
|
||||
"window_start_date": "2018-01-02",
|
||||
"window_end_date": "2018-01-05",
|
||||
"observations": 4,
|
||||
"lookback_sessions": 4,
|
||||
"missing_policy": "complete_case",
|
||||
"data_snapshot_id": arguments["backtest_run_ref"].dataset_snapshot_id,
|
||||
"input_sha256": "a" * 64,
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def risk_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
return {
|
||||
"portfolio_decision": decision,
|
||||
"backtest_run_ref": arguments["backtest_run_ref"],
|
||||
"manifest": arguments["manifest"],
|
||||
"covariance": covariance(arguments),
|
||||
"risk_model_name": "euler_volatility",
|
||||
"risk_model_version": "1.0.0",
|
||||
"risk_model_digest": digest({"synthetic_risk_model": 1}),
|
||||
"risk_budget": {ASSETS[0]: 0.8, ASSETS[1]: 0.8},
|
||||
"portfolio_volatility_limit": 10.0,
|
||||
"groups": {ASSETS[0]: "equity", ASSETS[1]: "fixed_income"},
|
||||
"computed_at": "2026-09-08T01:14:00Z",
|
||||
}
|
||||
|
||||
|
||||
def test_risk_uses_historical_business_age_and_actual_computation_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.assessment_id.startswith("rhriskassessmentv2:sha256:")
|
||||
assert result.qualified is True
|
||||
assert result.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert result.computed_at == "2026-09-08T01:14:00Z"
|
||||
assert result.to_dict()["decision_eligible"] is False
|
||||
assert result.to_dict()["execution_validation"] == "not_validated"
|
||||
assert sum(result.percentage_risk.values()) == pytest.approx(1.0)
|
||||
assert sum(result.component_risk.values()) == pytest.approx(result.portfolio_volatility)
|
||||
|
||||
|
||||
def test_new_risk_computation_cannot_reuse_stale_actual_manifest_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments["computed_at"] = "2026-09-08T02:11:01Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
def target_with(arguments: dict[str, Any], **changes: Any) -> RetrospectivePortfolioTarget:
|
||||
row = arguments["target"].to_dict()
|
||||
return RetrospectivePortfolioTarget.create(
|
||||
**{
|
||||
key: value
|
||||
for key, value in {**row, **changes}.items()
|
||||
if key
|
||||
in {"backtest_run_id", "dataset_snapshot_id", "weights", "effective_at", "created_at"}
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def decision_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("backtest_run_ref", "manifest", "target")}
|
||||
|
||||
|
||||
def assessment_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
key: arguments[key]
|
||||
for key in ("portfolio_decision", "backtest_run_ref", "manifest", "covariance")
|
||||
}
|
||||
|
||||
|
||||
def reidentify(row: dict[str, Any], field: str, prefix: str) -> None:
|
||||
row.pop(field, None)
|
||||
encoded = json.dumps(
|
||||
row, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False
|
||||
)
|
||||
row[field] = prefix + "sha256:" + hashlib.sha256(encoded.encode()).hexdigest()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"parser",
|
||||
[RetrospectivePortfolioTarget, RetrospectivePortfolioDecision, RetrospectiveRiskAssessment],
|
||||
)
|
||||
def test_json_syntax_failures_use_typed_contract_errors(parser: Any) -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
parser.from_json(b"{")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"method": "alternate_estimator"},
|
||||
{"window_start_date": "2018-01-03"},
|
||||
{"window_end_date": "2018-01-04"},
|
||||
{"observations": 3},
|
||||
{"lookback_sessions": 5},
|
||||
{"missing_policy": "alternate_missing_policy"},
|
||||
],
|
||||
)
|
||||
def test_covariance_estimation_context_is_bound_into_the_result_identity(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
original = assess_retrospective_portfolio_risk(**arguments)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
changed = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert changed.assessment_id != original.assessment_id
|
||||
|
||||
|
||||
def test_canonical_roundtrips_and_immutable_results() -> None:
|
||||
base = portfolio_arguments()
|
||||
target = base["target"]
|
||||
assert RetrospectivePortfolioTarget.from_json(target.to_json().encode()) == target
|
||||
decision = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base))
|
||||
assert (
|
||||
RetrospectivePortfolioDecision.from_json(decision.to_json(), **decision_context(base))
|
||||
== decision
|
||||
)
|
||||
arguments = risk_arguments(base)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert (
|
||||
RetrospectiveRiskAssessment.from_json(
|
||||
result.to_json().encode(), **assessment_context(arguments)
|
||||
)
|
||||
== result
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
target.weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
target.created_at = "2018-01-05T07:00:00Z"
|
||||
with pytest.raises(TypeError):
|
||||
decision.target_weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(TypeError):
|
||||
result.component_risk[ASSETS[0]] = 0.1
|
||||
detached = result.to_dict()
|
||||
detached["component_risk"][ASSETS[0]] = 0.1
|
||||
assert detached != result.to_dict()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"weights": {}},
|
||||
{"weights": {"SIM0": 1.0}},
|
||||
{"weights": {ASSETS[0]: float("nan")}},
|
||||
{"weights": {ASSETS[0]: True}},
|
||||
{"backtest_run_id": "rhbacktestrunv1:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhds:sha256:" + "0" * 64},
|
||||
{"effective_at": "2026-09-09T01:00:00Z"},
|
||||
{"created_at": "2026-09-08T01:12:00.1234567Z"},
|
||||
{"effective_at": "2018-01-05T15:00:00+08:00"},
|
||||
],
|
||||
)
|
||||
def test_target_rejects_legacy_ambiguous_and_nonfinite_inputs(change: dict[str, Any]) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
target_with(portfolio_arguments(), **change)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("usage", "live"),
|
||||
("historical_availability", "established"),
|
||||
("schema_version", "1.0.0"),
|
||||
("extra", True),
|
||||
("target_id", "rhportfoliotargetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_target_rejects_wire_mutations(path: str, value: Any) -> None:
|
||||
row = portfolio_arguments()["target"].to_dict()
|
||||
row[path] = value
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_dict(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field", ["input_digest", "constraint_digest", "output_digest"])
|
||||
def test_receipt_digests_are_recomputed(field: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments, **{field: "sha256:" + "0" * 64})
|
||||
with pytest.raises(FactorContractError, match="independently recomputed"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", ["failed", "fallback"])
|
||||
def test_failed_or_fallback_solver_cannot_form_a_decision(status: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status=status,
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="failed/fallback"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"backtest_run_id": "rhbacktestrunv2:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"weights": {"rhinstrument:" + "f" * 32: 1.0}},
|
||||
{"created_at": "2026-09-08T01:10:00Z"},
|
||||
{"created_at": "2026-09-08T01:14:00Z"},
|
||||
],
|
||||
)
|
||||
def test_decision_closes_target_identity_assets_and_actual_time(change: dict[str, Any]) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
arguments["target"] = target_with(arguments, **change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_actual_manifest_freshness_boundary_and_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00Z"
|
||||
assert (
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
).computed_at
|
||||
== arguments["computed_at"]
|
||||
)
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00.000001Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=portfolio_receipt(arguments))
|
||||
arguments["computed_at"] = "2026-09-08T01:13:00Z"
|
||||
with pytest.raises(FactorContractError, match="receipt actual time"):
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments, computed_at="2026-09-08T01:13:01Z")
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_tables_and_qualification_are_revalidated_at_s4_boundary() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
manifest = arguments["manifest"]
|
||||
artifact = manifest._artifact
|
||||
arguments["manifest"] = build_retrospective_backtest_evidence_manifest(
|
||||
arguments["backtest_run_ref"],
|
||||
artifact,
|
||||
artifact_available_at=manifest.artifact_available_at,
|
||||
qualification=EvidenceQualification.EXPLORATORY,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="contract-qualified"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments["manifest"] = manifest
|
||||
# Public access is an isolated copy. Simulate corruption of the retained bytes,
|
||||
# beyond that normal interface, to exercise the consumer's independent recheck.
|
||||
artifact._performance.loc[0, "n_days"] += 1
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_constraint_residuals_and_prior_assets_cannot_be_bypassed() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["constraints"] = ConstraintSetV1(gross_exposure_max=0.9)
|
||||
# A solver may report convergence within its tolerance; actual contract constraints still bind.
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status="converged",
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
tolerance=0.2,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="violates supported constraints"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments = portfolio_arguments()
|
||||
arguments["prior_weights"] = {"rhinstrument:" + "f" * 32: 0.5}
|
||||
with pytest.raises(FactorContractError, match="prior assets"):
|
||||
compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key != "computed_at"}
|
||||
)
|
||||
arguments["prior_weights"] = None
|
||||
with pytest.raises(PortfolioRiskContractError, match="prior"):
|
||||
portfolio_receipt(arguments)
|
||||
|
||||
|
||||
def test_optional_prior_budget_limit_and_groups_have_explicit_empty_semantics() -> None:
|
||||
base = portfolio_arguments()
|
||||
base["constraints"] = ConstraintSetV1(gross_exposure_max=1.0)
|
||||
base["prior_weights"] = None
|
||||
arguments = risk_arguments(base)
|
||||
arguments.update(risk_budget=None, portfolio_volatility_limit=None, groups=None)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.qualified is True
|
||||
assert result.risk_budget == {}
|
||||
assert result.group_exposure == {}
|
||||
assert result.groups is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("gross_exposure", True),
|
||||
("position_count", 2.0),
|
||||
("target_weights." + ASSETS[0], 0.5),
|
||||
("schema_version", "1.0.0"),
|
||||
("observation_cutoff", "2018-01-05T07:00:00Z"),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_decision_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
base = portfolio_arguments()
|
||||
row = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base)).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "decision_id", "rhportfoliodecisionv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioDecision.from_dict(row, **decision_context(base))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"as_of_date": "2018-01-06"},
|
||||
{"as_of_date": "2018-01-04", "window_end_date": "2018-01-04"},
|
||||
{"window_start_date": None, "window_end_date": None},
|
||||
{"data_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"input_sha256": "b" * 64},
|
||||
],
|
||||
)
|
||||
def test_covariance_business_time_bounds_and_source_binding(change: dict[str, Any]) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
with pytest.raises(FactorContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,index,columns",
|
||||
[
|
||||
([[float("nan"), 0.0], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.1], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], ASSETS[0]), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], "unknown"), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], ASSETS, (ASSETS[0], "unknown")),
|
||||
],
|
||||
)
|
||||
def test_covariance_structure_is_checked_before_computation(
|
||||
matrix: Any, index: Any, columns: Any
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=index, columns=columns)
|
||||
)
|
||||
with pytest.raises(PortfolioRiskContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"risk_budget": {ASSETS[0]: -0.1}},
|
||||
{"risk_budget": {"unknown": 0.1}},
|
||||
{"portfolio_volatility_limit": -0.1},
|
||||
{"groups": {ASSETS[0]: "equity"}},
|
||||
{"groups": []},
|
||||
{"risk_model_version": "latest"},
|
||||
{"risk_model_name": "/private/model"},
|
||||
{"computed_at": "2026-09-08T01:12:59Z"},
|
||||
{"portfolio_decision": object()},
|
||||
{"covariance": object()},
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_invalid_models_budgets_clocks_and_untyped_inputs(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,finding",
|
||||
[
|
||||
([[1.0, 2.0], [2.0, 1.0]], RiskFindingCode.COVARIANCE_NOT_PSD),
|
||||
([[0.0, 0.0], [0.0, 0.0]], RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE),
|
||||
],
|
||||
)
|
||||
def test_numerical_unavailability_is_not_qualification(
|
||||
matrix: Any, finding: RiskFindingCode
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=ASSETS, columns=ASSETS)
|
||||
)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.qualified is False
|
||||
assert result.findings == (finding,)
|
||||
assert result.portfolio_volatility is None
|
||||
|
||||
|
||||
def test_risk_uses_the_existing_numeric_implementation_exactly_once(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
calls = []
|
||||
|
||||
def recorded(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
calls.append((weights, matrix))
|
||||
return labeled_component_risk(weights, matrix)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", recorded)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert len(calls) == 1
|
||||
expected = labeled_component_risk(*calls[0])
|
||||
assert result.component_risk == expected.component.to_dict()
|
||||
assert result.portfolio_volatility == expected.portfolio_volatility
|
||||
|
||||
|
||||
def test_unknown_numeric_failures_are_sanitized(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def failed(*args: Any) -> ComponentRiskResult:
|
||||
raise ValueError("synthetic internal detail")
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", failed)
|
||||
with pytest.raises(PortfolioRiskContractError, match="risk computation failed") as error:
|
||||
assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert "internal detail" not in str(error.value)
|
||||
|
||||
|
||||
def test_nonclosed_decomposition_is_unavailable(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def nonclosed(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
output = labeled_component_risk(weights, matrix)
|
||||
return replace(output, component=output.component * 0.5)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", nonclosed)
|
||||
result = assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.findings == (RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED,)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change", [{"portfolio_volatility_limit": 0.0}, {"risk_budget": {ASSETS[0]: 0.0}}]
|
||||
)
|
||||
def test_budget_breach_keeps_ready_but_unqualified_evidence(change: dict[str, Any]) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.READY
|
||||
assert result.qualified is False
|
||||
assert result.findings == (RiskFindingCode.RISK_BUDGET_BREACH,)
|
||||
assert result.decision_eligible is False
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("qualified", 1),
|
||||
("portfolio_volatility", 1.0),
|
||||
("component_risk." + ASSETS[0], 1.0),
|
||||
("schema_version", "1.0.0"),
|
||||
("covariance_matrix_digest", "sha256:" + "0" * 64),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
row = assess_retrospective_portfolio_risk(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "assessment_id", "rhriskassessmentv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveRiskAssessment.from_dict(row, **assessment_context(arguments))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw",
|
||||
[
|
||||
b'{"x":1,"x":2}',
|
||||
b'{ "x":1}',
|
||||
b"[]",
|
||||
b'{"x":NaN}',
|
||||
b'{"x":Infinity}',
|
||||
b'{"x":9007199254740992}',
|
||||
1,
|
||||
],
|
||||
)
|
||||
def test_json_profiles_reject_ambiguous_nonfinite_and_noncanonical_input(raw: Any) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_json(raw)
|
||||
Reference in New Issue
Block a user