Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3374884f57 |
+2
-29
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"module_id": "quant_engine",
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 6, "effective_from": "2026-09-08T19:33:40+08:00"},
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 1, "effective_from": "2026-08-20T00:00:00+08:00"},
|
||||
"repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"},
|
||||
"bounded_context": {
|
||||
"domain": "quantitative-research-engine",
|
||||
@@ -11,7 +11,6 @@
|
||||
"Submitting live orders, routing trades, managing brokerage accounts, or claiming transaction execution",
|
||||
"Owning market-data source facts, research-result publication, or platform presentation state",
|
||||
"Loading provider credentials, brokerage credentials, or production secrets",
|
||||
"Granting portfolio approval, maker-checker decisions, publication eligibility, paper execution, or live execution authority",
|
||||
"Changing financial model semantics through module metadata"
|
||||
]
|
||||
},
|
||||
@@ -19,39 +18,13 @@
|
||||
{"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"},
|
||||
{"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"},
|
||||
{"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"},
|
||||
{"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact and performance-methodology evidence without recomputation, persistence, or decision authority.", "status": "operational"},
|
||||
{"id": "portfolio-risk-computation-contracts", "summary": "Verify deterministic portfolio-computation receipts and expose S3-bound portfolio decisions and risk assessments without adding algorithms or execution authority.", "status": "operational"},
|
||||
{"id": "retrospective-computation-contracts", "summary": "Decode observation-aware v2 data and expose explicit retrospective factor, backtest, portfolio and risk contracts with two clocks, no historical-availability claim and no execution authority.", "status": "operational"},
|
||||
{"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"}
|
||||
],
|
||||
"data": {"owns": [
|
||||
{"asset_id": "quantitative-model-implementations", "kind": "model", "classification": "internal"},
|
||||
{"asset_id": "simulation-and-metric-results", "kind": "artifact", "classification": "confidential"}
|
||||
]},
|
||||
"contracts": {
|
||||
"provides": [
|
||||
{"contract_id": "researchhub.factor-definition", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
|
||||
{"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-run-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/governed_pipeline.py"},
|
||||
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
|
||||
{"contract_id": "researchhub.performance-evidence", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
|
||||
{"contract_id": "researchhub.portfolio-decision", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.factor-set-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_factor_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-run-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_backtest_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
|
||||
{"contract_id": "researchhub.performance-evidence", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
|
||||
{"contract_id": "researchhub.portfolio-target", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.portfolio-decision", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"}
|
||||
],
|
||||
"consumes": [
|
||||
{"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"},
|
||||
{"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"},
|
||||
{"contract_id": "researchhub.dataset-snapshot", "version": "2.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_retrospective_envelope_and_materialized_chunks"},
|
||||
{"contract_id": "researchhub.data-foundation", "version": "2.0.0", "authority": "researchhub.data", "admission": "observation_bound_selected_views_and_materialized_bytes"}
|
||||
]
|
||||
},
|
||||
"contracts": {"provides": [], "consumes": []},
|
||||
"dependencies": [],
|
||||
"agent_context": {
|
||||
"default_entrypoints": [
|
||||
|
||||
@@ -19,21 +19,15 @@
|
||||
## 模块
|
||||
|
||||
- `alpha_factors` — 158 alpha 公式 + 24 基础算子(移植自 qlib alpha158)
|
||||
- `factor_contracts` — `FactorDefinition` / `FactorSetRef` v1 纯计算合同、严格 PIT/availability 输入准入与显式 legacy 投影
|
||||
- `execution` — A 股长仓执行仿真(成本/滑点/现金约束)+ 稀疏调仓/完整交易日 Ledger + 可投影成交与 NAV 审计;T+1、涨跌停、成交量与价差提供独立约束函数
|
||||
- `indicators` — 50+ 技术指标(MACD / KDJ / 布林 / ATR / ADX / 等)
|
||||
- `data_adapter` — 桥接 qtdb_pro 长表与新模块(rename / long-wide / 复权 / vwap 代理)
|
||||
- `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark)
|
||||
- `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表
|
||||
- `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排)
|
||||
- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef`
|
||||
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest`
|
||||
- `portfolio_risk_contracts` — S3 证据闭合的 `PortfolioDecision` / `RiskAssessment` v1;独立复核 freshness、约束与 computation receipt,并复用既有标签安全风险分解
|
||||
- `retrospective_*_contracts` — 未发布的显式 v2 回顾性合同:区分历史业务日期与实际可得/计算时间,保留 v1 和现有金融公式,不授予历史可得性、发布或执行权限;见 [v2 接口说明](docs/RETROSPECTIVE_COMPUTATION_V2.md)
|
||||
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表与 manifest
|
||||
- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计
|
||||
- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效
|
||||
- `strategy_research` / `strategy_optimizer` / `trade_pairing` / `strategy_artifact` — 七策略候选、下一日开盘执行、统一账本、FIFO双边成本、有界真实优化和完整策略报告工件;见 [策略合同](docs/strategy-research.md)
|
||||
- `factor_diagnostics` — 候选0.1.0:完整键配对、逐日IC/RankIC、样本与未定义值、显式日历前瞻标签;见 [诊断合同](docs/factor-diagnostics.md)
|
||||
- `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test)
|
||||
- `portfolio_decomp` — 组合分解(risk_parity / mean_variance / 因子归因)
|
||||
- `risk` — ndarray 低层风险公式 + 标签安全、可分组的 Euler 成分风险分解
|
||||
@@ -61,9 +55,6 @@ pytest # 单元测试
|
||||
pytest --cov=src # 覆盖率
|
||||
mypy --strict src/ # 类型检查
|
||||
ruff check src/ tests/ # lint
|
||||
|
||||
# 无网络、无数据库、无券商的架构烟测
|
||||
uv run python -m quant_engine.governed_pipeline
|
||||
```
|
||||
|
||||
## 使用
|
||||
@@ -200,220 +191,6 @@ print(backtest.stats())
|
||||
print(backtest.benchmark_report())
|
||||
```
|
||||
|
||||
## 因子/特征合同 v1
|
||||
|
||||
`quant_engine.factor_contracts` 提供 `researchhub.factor-definition` 与
|
||||
`researchhub.factor-set-ref` `1.0.0`。合同使用受限 canonical JSON:只接受 ASCII
|
||||
lower-snake-case object key、UTF-8 string、bool/null 和 safe integer;小数参数必须用显式
|
||||
canonical decimal string。定义、输入映射、上游证据、输出 schema/content 和 lineage 的任一
|
||||
语义变化都会产生新 identity。
|
||||
|
||||
创建 `FactorSetRef` 必须提供完整且可重算 identity 的 `DatasetSnapshotEnvelope` 与
|
||||
`DataFoundationEnvelope`,不能用 ID 字符串或布尔值代替资格证明。每个因子输入都要映射到一个
|
||||
实际选中的 `StandardizedViewRef`,schema 必须同时匹配定义和 view;未消费、缺失、重复或跨
|
||||
snapshot/Foundation/PIT 的 view 都会失败关闭。snapshot PIT 可以早于 Foundation/view PIT,
|
||||
但始终满足 knowledge ≤ snapshot PIT ≤ Foundation/view/FactorSet PIT ≤ evaluation。
|
||||
|
||||
```python
|
||||
from quant_engine.factor_contracts import (
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorDefinition,
|
||||
FactorSetRef,
|
||||
)
|
||||
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(dataset_snapshot_v1)
|
||||
foundation = DataFoundationEnvelope.from_dict(data_foundation_v1)
|
||||
|
||||
# definition 必须是完整的 FactorDefinition;FactorSetRef.create 还要求显式 input bindings、
|
||||
# view availability、output quality/coverage、canonical output bytes 和 immutable artifact ref。
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
**explicit_factor_set_evidence,
|
||||
)
|
||||
```
|
||||
|
||||
`availability_mode="as_available"` 声明 source/view 和计算产物在历史 evaluation 前实际可用;
|
||||
`"retrospective_replay"` 保留历史 evaluation,但要求真实 publication/view creation、compute 和
|
||||
artifact 时间位于之后,并固定 `historical_availability="not_established"`。两种模式都不会授予
|
||||
decision、real-data、production、paper 或 live readiness。
|
||||
|
||||
旧 `governed_pipeline.FactorVersion` 的四字段构造器、`factor_id@version`、run/target/risk/order
|
||||
identity 均保持不变。迁移只能通过 content-addressed `LegacyFactorBinding`,再显式调用
|
||||
`bind_legacy_factor()` 或 `project_legacy_factor()`;后者是有损投影,不表示旧 digest 与新定义
|
||||
digest 等价,也不会把旧 run 静默升级为新合同。
|
||||
|
||||
## 回测引用与证据合同 v1
|
||||
|
||||
`quant_engine.governed_pipeline.BacktestRunRef` 是合格回测运行身份的唯一权威。`run_id` 只由
|
||||
已验收的 Dataset Snapshot / Data Foundation / `FactorSetRef` 身份、universe、日历与公司行动
|
||||
祖先、策略、执行/成本模型、严格整数 seed、完整代码提交、环境锁、配置、时间和重放血缘决定;
|
||||
它不包含任何输出摘要。重放必须绑定直接父运行、连续 attempt 和不变的
|
||||
`replay_spec_digest`,输入漂移或血缘环会失败关闭。
|
||||
|
||||
`quant_engine.artifact.BacktestEvidenceManifest` 只摘要 `ResearchRunArtifact` 已有的九张事实表。
|
||||
固定 `offline_research_v1` 映射为 `run`、`signal`、`fill`、`position_nav`、`performance`、
|
||||
`attribution`、`risk_snapshot` 与 `replay`;每张表都保留列模式摘要、行数和内容摘要,空 risk
|
||||
表也必须有稳定 schema。`manifest_id` 由完整 RunRef 与输出证据决定,因此结果变化不会反向改变
|
||||
`run_id`。当前 artifact 不拥有订单或拒绝事实,所以此画像明确不声明 `order` / `rejection`。
|
||||
|
||||
旧 `BacktestRun` 只能通过 `build_legacy_backtest_evidence_manifest()` 显式映射为
|
||||
`LEGACY_EXPLORATORY`;不能隐式提升为 `CONTRACT_QUALIFIED`。所有资格均只描述离线证据闭合,
|
||||
不表示投资有效、组合获批、Paper、生产或实盘就绪。
|
||||
|
||||
## 绩效证据与方法论合同 v1
|
||||
|
||||
`quant_engine.artifact.PerformanceEvidenceV1` 在现有计算和事实表之外增加一层只读、内容寻址的
|
||||
owner 证据。`build_performance_evidence()` 只接受同一运行的完整 `ResearchRunArtifact`、
|
||||
`BacktestRunRef` 与 `CONTRACT_QUALIFIED BacktestEvidenceManifest`;它核对全部 artifact 表、
|
||||
performance 表和唯一行摘要,并绑定 artifact、row 与严格对齐 benchmark series 的独立摘要。
|
||||
生产 builder 不重算、填补、重命名或覆盖任何绩效值。
|
||||
|
||||
方法论固定为日简单收益、252 期年化、绝对指标年化无风险利率 `0.0`、benchmark 日无风险利率
|
||||
`0.0`,以及 benchmark 存在时的 `exact_session_index`。相对指标使用封闭 availability:无基准为
|
||||
`benchmark_absent`;active variance、benchmark variance 或 alpha 几何年化域不足时分别使用
|
||||
对应 `not_estimable_*` 原因。benchmark 存在时 tracking error 始终必须是有限非负值;null 不会
|
||||
被转成零。
|
||||
|
||||
```python
|
||||
from quant_engine.artifact import build_performance_evidence
|
||||
|
||||
performance_evidence = build_performance_evidence(
|
||||
artifact,
|
||||
backtest_run_ref,
|
||||
backtest_evidence_manifest,
|
||||
)
|
||||
canonical_bytes = performance_evidence.canonical_bytes()
|
||||
```
|
||||
|
||||
该合同范围固定为 `offline_research_only`。它不授予排名、推荐、决策、发布、论文、Paper、生产、
|
||||
实盘、交易或投资建议权限,也不包含原始参数、returns、NAV、benchmark series、表字节、存储
|
||||
locator、URI 或凭证。
|
||||
|
||||
## 组合决策与风险评估合同 v1
|
||||
|
||||
`quant_engine.portfolio_risk_contracts` 是现有计算 owner 外围的薄合同层。创建
|
||||
`PortfolioDecision` 必须同时提供完整 `BacktestRunRef`、嵌入同一 RunRef 的
|
||||
`CONTRACT_QUALIFIED` 非 legacy `BacktestEvidenceManifest`、现有 `PortfolioTarget`、
|
||||
`FreshnessPolicy`、`ConstraintSetV1` 与 `ComputationReceipt`。适配器会从权威输入独立重算
|
||||
receipt 的 input/constraint/output digest、敞口、持仓数和 L1 turnover 残差;receipt 自报
|
||||
成功、fallback 或放宽 tolerance 均不能替代复核。
|
||||
|
||||
```python
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
assess_portfolio_risk,
|
||||
build_portfolio_decision,
|
||||
compute_portfolio_receipt_digests,
|
||||
)
|
||||
|
||||
freshness = FreshnessPolicy(
|
||||
max_manifest_age_seconds=3600,
|
||||
max_covariance_age_days=5,
|
||||
)
|
||||
constraints = ConstraintSetV1(
|
||||
gross_exposure_max=1.0,
|
||||
single_asset_max=0.10,
|
||||
position_count_max=20,
|
||||
turnover_max=0.30,
|
||||
)
|
||||
|
||||
# 生产者先形成公开 canonical digest;decision 构建时仍会独立重算。
|
||||
expected = compute_portfolio_receipt_digests(
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
target=portfolio_target,
|
||||
objective_name="long_only_allocation",
|
||||
objective_version="1.0.0",
|
||||
objective_digest=objective_digest,
|
||||
model_name="factor_weighting",
|
||||
model_version="1.0.0",
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
|
||||
receipt = ComputationReceipt(
|
||||
algorithm="factor_weighting",
|
||||
algorithm_version="1.0.0",
|
||||
implementation_digest=implementation_digest,
|
||||
parameter_digest=parameter_digest,
|
||||
input_digest=expected["input_digest"],
|
||||
constraint_digest=expected["constraint_digest"],
|
||||
output_digest=expected["output_digest"],
|
||||
status="completed",
|
||||
solver_required=False,
|
||||
solver_name=None,
|
||||
solver_version=None,
|
||||
solver_config_digest=None,
|
||||
iterations=None,
|
||||
objective_value=None,
|
||||
max_constraint_residual=expected["max_constraint_residual"],
|
||||
tolerance=1e-12,
|
||||
computed_at=computed_at,
|
||||
)
|
||||
|
||||
decision = build_portfolio_decision(
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
target=portfolio_target,
|
||||
objective_name="long_only_allocation",
|
||||
objective_version="1.0.0",
|
||||
objective_digest=objective_digest,
|
||||
model_name="factor_weighting",
|
||||
model_version="1.0.0",
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness,
|
||||
receipt=receipt,
|
||||
computed_at=computed_at,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
|
||||
assessment = assess_portfolio_risk(
|
||||
portfolio_decision=decision,
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
covariance=covariance_snapshot,
|
||||
risk_model_name="euler_volatility",
|
||||
risk_model_version="1.0.0",
|
||||
risk_model_digest=risk_model_digest,
|
||||
)
|
||||
```
|
||||
|
||||
`source_universe_digest` 保留 S3 研究 universe 身份,`portfolio_asset_set_digest` 只描述实际
|
||||
目标资产标签;二者不会互相冒充成员证明。风险评估在任何数值计算前要求 covariance、target、
|
||||
RunRef 的 dataset identity 三方一致,并且只调用一次现有 `labeled_component_risk()`。合同中的
|
||||
`qualified` 仅表示 S4.1 计算证据闭合,不授予 maker-checker、发布、订单、Paper、生产或实盘权限。
|
||||
|
||||
## 治理垂直切片
|
||||
|
||||
`governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。
|
||||
调用方必须显式提供 `DatasetSnapshot`、`FactorVersion`、`StrategyVersion`、代码提交和
|
||||
`RiskPolicy`。模块只会生成 `environment="paper"` 的订单意图,不连接数据库、数据供应商或
|
||||
券商;风险决策为拒绝时,订单意图固定为空,直接调用创建函数也会失败关闭。
|
||||
|
||||
该切片对应 ResearchHub 架构的首个可执行验收链路:
|
||||
|
||||
```text
|
||||
DatasetSnapshot → FactorVersion → StrategyVersion → BacktestRun
|
||||
→ PortfolioTarget → RiskDecision → PaperOrderIntent
|
||||
```
|
||||
|
||||
平台总架构、五仓职责和十二层能力映射仍以 `research_platform/docs/architecture/` 为权威;
|
||||
本仓只拥有纯计算与离线模拟合同。
|
||||
|
||||
## 与 research_results 的关系
|
||||
|
||||
`research_results` 依赖 `quant_engine`(通过 re-export 保持向后兼容):
|
||||
|
||||
@@ -1,145 +0,0 @@
|
||||
# Retrospective computation contracts v2 (unreleased)
|
||||
|
||||
This pure, storage-neutral compatibility path consumes the separate data-contract
|
||||
major 2.0.0. It does not migrate, reinterpret or relax the accepted v1 contracts.
|
||||
No financial formula, execution simulation, dependency lock, production database,
|
||||
publisher or live/paper-order interface changes here. Package version is unchanged;
|
||||
the new contract major is not a package release or deployment.
|
||||
|
||||
## Explicit public boundaries
|
||||
|
||||
| Module | Public types/builders | Changed wire identity |
|
||||
| --- | --- | --- |
|
||||
| `retrospective_data_contracts` | `RetrospectiveSnapshotEnvelope`, `RetrospectiveFoundationEnvelope` | `rhdsv2`, `rhdfv2`; consume RP-owned 2.0.0 data semantics |
|
||||
| `retrospective_factor_contracts` | `RetrospectiveFactorSetRef`, typed input/view/causation bindings, `ResolvedRetrospectiveView` | `rhfactorsetv2` |
|
||||
| `retrospective_backtest_contracts` | `RetrospectiveBacktestRunRef` | `rhbacktestrunv2` |
|
||||
| `retrospective_artifact_contracts` | `RetrospectiveBacktestEvidenceManifest`, `RetrospectivePerformanceEvidence` and their builders | `rhbacktestevidencev2`, `rhperformancev2` |
|
||||
| `retrospective_portfolio_risk_contracts` | `RetrospectivePortfolioTarget`, `RetrospectivePortfolioDecision`, `RetrospectiveRiskAssessment`; receipt-digest, decision and assessment builders | `rhportfoliotargetv2`, `rhportfoliodecisionv2`, `rhriskassessmentv2` |
|
||||
|
||||
These are separate types and domain-separated content identities. There is no
|
||||
automatic v1-to-v2 cast. Unknown schema versions and fields are rejected. The
|
||||
performance wire keeps its named schema `researchhub.performance-evidence.v2`;
|
||||
the other new computation contracts use `schema_version: 2.0.0`.
|
||||
|
||||
FactorDefinition, factor-output byte references, output quality/coverage,
|
||||
ConstraintSetV1, FreshnessPolicy, ComputationReceipt, CovarianceSnapshot, financial
|
||||
algorithms, performance metric/methodology IDs and the nine ResearchRunArtifact
|
||||
tables keep their existing semantics. The table schema remains **1.1.0**. Reusing
|
||||
these neutral primitives does not make a new-major upstream reference v1-compatible.
|
||||
|
||||
## Two clocks, not backdated evidence
|
||||
|
||||
Every new result fixes `usage=retrospective_research` and
|
||||
`historical_availability=not_established`. A business date describes the historical
|
||||
period being researched. Observation, publication, evaluation, artifact availability,
|
||||
target creation and computation describe actual events, and must not be backdated.
|
||||
Public v2 instants require UTC `Z` with at most six fractional digits.
|
||||
|
||||
`observation_cutoff` and chunk `observed_by` are upper-bound observations. They are
|
||||
not the earliest public knowledge time or a PIT cutoff. Unknown earliest knowledge
|
||||
stays unknown; a supplied knowledge-evidence digest is not authenticated by parsing.
|
||||
Foundation observation sequences describe retained revisions, not complete original
|
||||
history. Selected view routes, calendars, corporate-action coverage and lineage
|
||||
must close exactly within the supplied Foundation.
|
||||
|
||||
Required actual order for factor/backtest evidence is:
|
||||
|
||||
1. Foundation publication <= factor evaluation <= factor computation <= factor availability.
|
||||
2. Factor availability <= backtest evaluation <= artifact start <= artifact finish
|
||||
<= backtest computation <= artifact availability.
|
||||
3. Artifact availability <= target creation <= portfolio computation <= risk computation.
|
||||
|
||||
RetrospectivePortfolioTarget has a historical `effective_at` and a distinct actual
|
||||
`created_at`. PortfolioDecision carries both plus actual `computed_at`. Covariance
|
||||
window end <= covariance as-of date <= the historical effective date; covariance
|
||||
maximum age is measured against that historical date. Manifest maximum age is
|
||||
measured against **actual** portfolio and risk computation separately. Passing one
|
||||
age check cannot substitute for the other. Generic v1 receipt timestamps retain
|
||||
their original normalization; binding compares parsed actual instants.
|
||||
|
||||
## Materialized bytes and reference-only reads
|
||||
|
||||
Snapshot decoding checks structure, all six blocking-quality declarations,
|
||||
qualification/time ordering, observation receipts and identities.
|
||||
`verify_materialized_records` additionally checks supplied chunks, per-chunk and
|
||||
aggregate content, counts, dimensions, effective ranges and macro effective instants.
|
||||
Provider/physical paths are forbidden in public metadata and materialized records.
|
||||
|
||||
Factor creation requires actual snapshot chunks, selected view schema/content bytes,
|
||||
and factor-output schema/content bytes. Definition inputs, view availability,
|
||||
Foundation ancestry and computed digests must close. Reference-only deserialization
|
||||
is allowed for display/inspection, but input/output validation flags are derived from
|
||||
supplied bytes, are not serialized claims, and must be re-established for new
|
||||
computation. Backtest creation requires a factor whose payloads were revalidated.
|
||||
Reference decoding cannot turn an unverified factor into an admitted compute input.
|
||||
|
||||
Backtest manifest decoding rebuilds evidence from the supplied typed run and all
|
||||
nine actual artifact tables. It checks table/run/config/strategy bindings and time
|
||||
ordering. Portfolio composition revalidates those retained tables again, rather
|
||||
than trusting a serialized manifest or mutable Python context. A table digest proves
|
||||
content binding, not that those tables were produced by the claimed computation.
|
||||
|
||||
All content-addressed IDs exclude their own ID field and bind the remainder of the
|
||||
closed payload. Data/factor/backtest/manifest JSON retains the strict data profile
|
||||
(no JSON floating-point numbers; financial record decimals are strings). Performance
|
||||
and S4 preserve the existing finite numeric JSON profile: finite floats, safe ints,
|
||||
exact booleans, sorted keys, compact separators, UTF-8. Duplicate keys, NaN,
|
||||
Infinity, noncanonical JSON and extra fields are rejected. Wire revalidation uses
|
||||
type-sensitive comparisons, including `true` versus `1`. Serializers return
|
||||
detached copies; internal public maps are immutable.
|
||||
|
||||
## Replay, receipts and risk
|
||||
|
||||
Backtest v2 replay specification binds immutable input identities, selected calendar
|
||||
and actions, strategy/execution/cost versions and digests, configuration, code,
|
||||
environment lock and random seed. It excludes **both actual evaluation and actual
|
||||
computation time**. These actual times remain in each run's identity. A replay must
|
||||
retain the same replay specification, append its unique full ancestry, increment
|
||||
attempt by one, and have parent computation < new actual evaluation <= computation.
|
||||
This explicit new-major rule allows a later genuine replay without pretending its
|
||||
evaluation happened at the parent's clock time.
|
||||
|
||||
Portfolio computation-input v2 binds the full run and manifest document digests,
|
||||
new target (including both clocks), objective/model versions and digests, declared
|
||||
expected returns/covariance/scenario inputs, freshness policy and prior weights.
|
||||
The receipt separately binds that input, constraints and recomputed outputs/residuals.
|
||||
Targets and prior holdings must use selected logical instrument IDs, not ad-hoc
|
||||
symbol matches. Failed/fallback receipts and any actual constraint residual are
|
||||
rejected, even if a solver declares convergence within a permissive tolerance.
|
||||
|
||||
Risk uses the existing labelled Euler decomposition exactly once. Its result binds
|
||||
the supplied matrix content plus covariance method, bounded estimation window,
|
||||
observation count, lookback, missing policy, annualization, source dataset/input,
|
||||
model/budgets/groups and actual computation time. It checks exact labels, finite
|
||||
symmetry, covariance-source binding and both freshness clocks. Non-PSD,
|
||||
non-positive portfolio variance or non-closed contributions produce an unavailable,
|
||||
unqualified result. A budget breach is a ready but unqualified calculation result.
|
||||
`qualified=true` means only that these calculation checks passed. Every result
|
||||
remains `decision_eligible=false`, `execution_validation=not_validated`; no portfolio
|
||||
approval, maker-checker, publication, paper or live permission is granted here.
|
||||
|
||||
## Trust, ownership and test evidence
|
||||
|
||||
Pure builders accept declarations. Hashes, typed objects, model names, successful
|
||||
constraint checks and synthetic fixtures do **not** authenticate data or compute
|
||||
producers. Trusted owner-version bindings and receipt/qualification/view/clock
|
||||
admission ports remain mandatory. RP owns governance and presentation; Research
|
||||
Results owns publication. QE supplies validated calculation facts only.
|
||||
|
||||
The two data fixtures are public RP candidate vectors from
|
||||
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (PR #100). EDB producer candidate
|
||||
`88433dfc9d865ef782465498cdf9454c73920abd` (PR #13) is not a runtime dependency or
|
||||
accepted owner binding. Acceptance/review gates remain separate from local tests.
|
||||
|
||||
`tests/fixtures/retrospective-computation-v2.golden.json` freezes newly constructed
|
||||
synthetic factor/run/manifest/performance/portfolio/risk payloads and their inputs.
|
||||
Its artifact matrices are separate synthetic envelope-test inputs: the one-day
|
||||
public data fixture is **not** claimed to have produced the four-day artifact.
|
||||
The vector is not an end-to-end data/computation provenance proof or real-data run.
|
||||
Its deterministic IDs are contract-regression evidence, not admitted source facts.
|
||||
|
||||
Focused tests cover v2 goldens, mutation and strict JSON, bytes versus references,
|
||||
two-clock freshness, replay ancestry, exact table bindings, constraints, receipts,
|
||||
covariance provenance and numerical findings. Existing v1 tests must also pass.
|
||||
Rollback is disabling the explicit v2 entry path while retaining v1 and original
|
||||
immutable results; never retag old results or silently downgrade failed v2 admission.
|
||||
@@ -1,37 +0,0 @@
|
||||
# Keyed factor diagnostics, candidate API 0.1.0
|
||||
|
||||
`quant_engine.factor_diagnostics` is a pure calculation module for caller-supplied observations. It does not change the older `factor_library.ic_summary` API or the governed factor-set contracts. Its calculation/API version does not establish a production algorithm, source qualification, historical availability or execution eligibility.
|
||||
|
||||
## Inputs and outputs
|
||||
|
||||
Factor values are a DataFrame indexed by unique `(date, asset)` keys, with unique factor columns. Dates must be a naive DatetimeIndex of midnight session labels; strings and timezone-aware/intraday timestamps are rejected. Order may vary and is normalized without modifying inputs. Identifiers must be nonempty printable strings. Values are real finite numbers or missing (`None`, `pd.NA`, NaN); booleans, numeric strings, infinities and duplicate keys are rejected.
|
||||
|
||||
`daily_ic(factors, returns, min_pairs=3, min_days=2)` aligns the forward-return Series by complete keys inside each date. The factor keys define the observed universe. Extra return keys are ignored and counted, while absent return keys remain missing. Every observed factor/date survives, including zero valid pairs. Output includes daily Pearson and average-tie Spearman values, actual pair/observation/value counts and separate reasons. Returns must already be labels for the intended forward interval; this function does not infer their unit, calendar, origin or availability.
|
||||
|
||||
Daily summaries weight each valid date equally. Mean and sample standard deviation (`ddof=1`) use valid daily ICs, not the number of securities. IR is unannualized `mean/std`; the descriptive IID statistic is `IR*sqrt(valid_days)`, with two-sided Student-t p and `df=valid_days-1`. Pearson and RankIC have separate valid/missing day counts. No-valid-day and insufficient-day results remain explicit. Constant series have standard deviation zero and undefined ratios.
|
||||
|
||||
The absolute standard-deviation resolution for ratio statistics is `32*float64 epsilon` (about 7.11e-15) because IC is bounded to [-1,1]. Nonconstant dispersion at or below that resolution retains its mean, observed standard deviation and counts, but IR/t/p are null with `below_resolution`. This candidate numerical policy prevents floating-point noise between equivalent cross sections from becoming extreme significance. It is not a statistical materiality threshold. Serial correlation, overlapping forward intervals and effective sample size are **not corrected**; t/p do not establish inferential validity or decision admission.
|
||||
|
||||
`correlation_matrix(factors, method='pearson', min_pairs=3)` pools complete `(date,asset)` pairs for each cell. It is not the mean of daily cross-sectional correlations: dates with more pairs contribute more observations. Every cell reports pair count and status. Empty, short or constant diagonals are null, not identity values. Pairwise deletion can produce a non-positive-semidefinite matrix; this is not an admitted risk/covariance matrix. Pearson translates before scaling to preserve representable small differences near a large offset, with scale-first fallback only if subtraction overflows. Spearman ranks original paired observations to avoid creating ties through underflow.
|
||||
|
||||
## Explicit forward-return intervals
|
||||
|
||||
`forward_returns(prices, sessions=..., entry_lag_sessions=..., holding_sessions=..., price_field=..., price_basis=...)` accepts a keyed price Series and an explicit, unique, increasing session calendar. Lag must be an integer >=0 and holding an integer >=1; booleans are rejected. The returned `ForwardReturns` object owns a return Series and interval DataFrame for every supplied session and observed asset, plus method metadata.
|
||||
|
||||
For signal session `t`, entry is session `t+lag`, exit is `t+lag+holding`, and the label is `P_exit/P_entry-1`. Endpoint prices must be positive when present and comparable under the caller-declared field/basis. Missing endpoint prices yield `missing_price`; calendar-tail insufficiency yields `insufficient_calendar`; nonfinite arithmetic yields `numerical_failure`. No filling, per-asset dropna calendar, next-available-price jump or daily-return summation is used. Intermediate prices are not required for this endpoint ratio. A lag of zero describes a same-session price basis and does not mean a closing signal can trade at the same close.
|
||||
|
||||
## Reuse decision and verification
|
||||
|
||||
需求:按完整证券/日期键提供逐日IC、相关矩阵与显式前瞻标签,保留样本及未定义原因。
|
||||
|
||||
已有方案:`factor_library.ic_summary(periods=(1,))`的单截面Pearson和`spearman_ic`;旧多周期rolling-sum不符合显式端点区间,旧摘要也不代替逐日统计。
|
||||
|
||||
候选开源方案:不需要,已有pandas/numpy/scipy及核心函数足够,无新增依赖。
|
||||
|
||||
推荐方案:在核心中二次封装现有单截面相关函数,增加观察键、分组、计数与区间合同。
|
||||
|
||||
原因:保持计算归属quant_engine,平台只做输入/输出适配;旧API保持兼容。
|
||||
|
||||
风险:浮点分辨率、缺失机制、序列依赖和PIT均须显式记录,纯合成通过不能升级正式资格。
|
||||
|
||||
Tests include manually checkable daily `[1,-0.5,0]` ICs with three effective dates, pairwise missingness, empty/constant results, average ranks, extreme numeric ranges, affine-equivalent daily ICs, large-offset Pearson precision, and explicit-calendar endpoint labels. Existing factor-library tests remain unchanged. No database, provider, production recomputation or ETL is part of this contract.
|
||||
@@ -1,78 +0,0 @@
|
||||
# Seven-strategy candidate research
|
||||
|
||||
The core owns the seven example strategies, sequential decisions, shared execution ledger, FIFO trade pairing and bounded exhaustive optimization. The platform supplies one selected asset's OHLC and displays results. The reserved asset identity `CASH` cannot be used as a security, preventing collision with projected cash positions. These pure functions perform no source reads, persistence, publication or order routing. `decision_eligible` remains false. Caller dates do not establish an exchange calendar or historical data availability.
|
||||
|
||||
## Reuse and scope
|
||||
|
||||
Requirement: reproduce the seven platform examples with explicit causal timing, cash/cost accounting, benchmark status and testable results.
|
||||
|
||||
Existing capabilities: platform strategy definitions, core `simulate_daily_ledger_with_audit`, cash-constrained fills and `metrics.summary`/`benchmark_summary`. The shared ledger is reused, including fee calculation. The new policy input adapts stateful decisions to that ledger; there is no second accounting engine. FIFO pairing adds the previously missing entry-cost and completed-lot interpretation. No external package or new dependency is needed.
|
||||
|
||||
Financial mathematical behavior is L3. The candidate is isolated to the original core branch/PR. It neither changes formal source admission nor asserts exchange lot sizes, tradability, T+1, price-limit, volume or point-in-time coverage. Those require qualified caller facts and further integration. Existing independent execution-constraint helpers are not silently enabled by this research interface.
|
||||
|
||||
## Input and timing contract
|
||||
|
||||
`run_strategy_research(strategy, bars, *, asset, params=None, initial_cash=..., commission=..., stamp_duty=..., benchmark=None, min_trade_amount=0)` consumes a single asset's complete daily OHLC DataFrame.
|
||||
|
||||
- Its naive daily DatetimeIndex is unique, sorted and within 1900–2100. There are 2–5000 observations. OHLC must be real finite positive values with valid bounds. Missing high/low/open, strings, booleans, duplicate dates and malformed bars fail. Prices are never synthesized from close.
|
||||
- Parameters merge the seven original defaults before strict validation. Periods are native integers from 1 to 500; optional ATR period also allows zero. Fast is less than slow; RSI oversold is below overbought. Allocation is between zero and one. Multipliers are bounded by 100; positive multipliers cannot be zero, while DualThrust coefficients may be zero.
|
||||
- Initial capital is positive, finite and at most 1e12. Explicit fee inputs are fractions: 0.01 means 1%. Commission and additional sell fee are nonnegative and their sum cannot exceed one. Defaults preserve implementation values and are not current-market tax assertions. Fractional shares follow the pre-existing research ledger. Costs may reduce a full allocation to a cash-constrained partial fill.
|
||||
- All input, benchmark and history checks precede ledger execution. Insufficient initialization history is an error. Optional longer ATR and exit lookbacks do not delay an otherwise valid entry signal; their own conditions wait for their own available history.
|
||||
- At each supplied session's open, the ledger processes the previous close's pending target. It then values actual holdings at the current close. The policy sees a separate copy of actual post-fill holdings and cash. `None` means no order; it does not liquidate or rebalance existing holdings.
|
||||
- A close signal schedules only the next supplied session's open. A final-session signal records `no_next_session` with no execution date. There is no same-close fallback. A zero allocation while already flat is `no_change`; an unfilled exit that retains holdings is `not_filled`.
|
||||
|
||||
The shared ledger's optional `decision_policy` cannot be combined with a fixed target schedule. Its execution-price calendar must cover the full valuation calendar. Static schedule behavior remains supported. A caller policy can itself misuse future information; the supplied seven policies use only causal windows. Future-perturbation tests establish that implementation property, not real-source PIT qualification.
|
||||
|
||||
## Strategy definitions
|
||||
|
||||
| Name | Entry while flat | Exit while held | Defaults |
|
||||
|---|---|---|---|
|
||||
| BuyAndHold | First supplied close schedules the allocation once | No automatic exit | buy_pct=.95 |
|
||||
| SmaCross | Fast SMA crosses strictly above slow SMA | Fast SMA crosses strictly below slow SMA | fast=5, slow=20 |
|
||||
| MACross | Same SMA cross definition | SMA cross down or optional trailing ATR stop | fast=10, slow=30, atr_period=0, atr_mult=2 |
|
||||
| RSI | Wilder RSI strictly below oversold | Strictly above overbought | period=14, oversold=30, overbought=70 |
|
||||
| BollingerBreakout | Close strictly above current-window mean plus population standard deviation times multiplier | Close strictly below current-window mean | period=20, std_mult=2 |
|
||||
| DualThrust | Close strictly above current open plus k1 times prior-window HH−LL | Close strictly below current open minus k2 times prior-window HH−LL | period=5, k1=.5, k2=.5 |
|
||||
| TurtleBreakout | Close strictly above the preceding entry-window high | Close strictly below the preceding exit-window low | entry_period=20, exit_period=10 |
|
||||
|
||||
MACross remains a dual-moving-average example; it is not renamed MACD. ATR is the arithmetic mean of true ranges over its explicit window. Zero disables ATR. After an actual entry the historical reference starts at its execution open, then tracks observed closes while held. At each later close, the stop is the greater of the prior stop and historical peak minus the previous session's ATR times the multiplier. Close at or below that stop schedules the next open exit. The current close does not construct a stop that is then impossibly compared with itself. Stops reset only when actual holdings become flat. If both exit conditions occur together, the recorded reason is `atr_stop`.
|
||||
|
||||
RSI is 50 when average gain and loss are both zero, 100 when only loss is zero, and otherwise follows Wilder smoothing. Bollinger uses population standard deviation, with each observed window independently scaled and deviations translated before scaling. This prevents tiny-price squared variance underflow and huge-price overflow without using a future/global scale. Nonfinite indicators after warmup fail explicitly. DualThrust deliberately preserves the existing example's HH−LL variant; it does not silently replace it with another range definition.
|
||||
|
||||
## Accounting, completed trades and metrics
|
||||
|
||||
Every fill is charged once by the shared ledger before that session's final NAV. Multi-asset same-session fills, final-session fills and both sides of rotation retain individual costs. Tiny fractional residual holdings are preserved; an explicit full exit consumes the exact held quantity, avoiding a rounded notional leaving a phantom lot.
|
||||
|
||||
`pair_ledger_trades` matches actual buy and sell fills FIFO per asset, using their net cash flows. Buy cost includes entry fees/slippage; net sell proceeds include exit fees/slippage. A match records allocated entry cost, exit proceeds and net PnL. A trade for win-rate purposes is one fully closed entry lot, even if exited in pieces. Partial exits contribute realized PnL but do not enter the completed-trade denominator. No completed lots yields `None`, not zero. An actual small residual is not treated as closed by relative tolerance. For a final asset fill followed by an exact flat ledger position, FIFO consumes every entry lot only when total quantity agrees within accumulated ULP resolution; this reconciles multi-entry subtraction rounding. The last matched lot receives the remaining net proceeds so the sell cash flow is conserved. Winning lots require PnL above 32 float64 ULPs at the cash magnitude; raw PnL is retained.
|
||||
|
||||
Daily returns and fees come from the shared ledger. Total return is final NAV divided by initial capital minus one. Complete loss is valid zero NAV, with no invented recovery; positive NAV whose return cannot be represented is rejected. Nonfinite returns are checked before general metrics, preventing generic cleaning from dropping an observation.
|
||||
|
||||
Other absolute measures reuse `metrics.summary`: 252 supplied trading sessions per year, CAGR-based annual return and Sharpe numerator, sample daily volatility, initial-capital-aware drawdown and daily-positive-return frequency. The latter is named `daily_win_rate`, distinct from FIFO `trade_win_rate`. Sharpe with zero volatility, Calmar with zero drawdown, Sortino with no downside and trade win rate with no closed lots are `None` with explicit reasons. Nonfinite derived metric outputs are marked unavailable; no default score is substituted.
|
||||
|
||||
## Benchmark and optimization
|
||||
|
||||
`BenchmarkInput` has four caller-reported states. `not_requested` contains no data; `empty` contains an empty Series. `present` requires a complete positive-price Series with exactly the valuation dates; missing/extra/duplicate dates fail rather than inner join. `source_error` fails before any strategy execution. Benchmark normalization and returns must be representable, finite and compatible with positive prices; non-first missing returns are never filled as zero. Output preserves empty versus not-requested status. Relative metrics reuse the existing strict `benchmark_summary`; constant-benchmark regressions remain unavailable.
|
||||
|
||||
`optimize_strategy_research` validates the entire grid before any trial. It allows at most four axes, ten candidates per axis and 100 combinations. Empty axes, unknown keys, duplicate values, non-native numbers, invalid relationships and any insufficient history fail the whole request. The size bound precedes Cartesian expansion. Every trial runs a fresh real seven-strategy pipeline and ledger with explicit fees. Targets are `total_return`, `sharpe_ratio` or `calmar_ratio`. An unavailable objective fails the ranking instead of silently skipping a candidate. Stable descending sorting preserves canonical axis order and supplied candidate order for ties; zero and negative finite scores remain valid.
|
||||
|
||||
## Strategy artifact projection
|
||||
|
||||
`strategy_artifact.build_strategy_research_artifact` accepts an actual `StrategyResearchResult` or `StrategyOptimizationResult` and returns the existing schema1.1.0 `ResearchRunArtifact`. The calculation result retains detached OHLC and benchmark-return views and explicit cost inputs; optimization retains a detached grid snapshot. The builder projects the selected result's ledger, actual fees, next-session signal links, close-marked holdings and existing metrics. It does not run a second strategy, accounting system or performance formula.
|
||||
|
||||
`params_json.strategy_report` uses `researchhub.strategy-research.v1`. It covers every causal signal, including `no_next_session`, unfilled targets and partial fills; complete FIFO matches, closed/open lots and net PnL; daily versus completed-trade win rates; benchmark status; exact parameters/costs; and all ranked candidate summaries. A grid artifact stores the selected first-ranked ledger plus every candidate's parameters, objective score, metrics and missing reasons, up to100 trials. It does not repeat100 full ledgers. Report contents are covered by the artifact's canonical content digest. User metadata cannot overwrite this report or its performance explanation.
|
||||
|
||||
The legacy factor-signal table remains empty: strategy signals have no factor score. The report explicitly locates the real strategy signals and marks factor attribution and covariance risk as not computed. Actual fills link to report signal IDs. A zero-value portfolio retains its actual zero values and an undefined weight, with the exact dates and reason recorded; it is not assigned an invented zero or full-cash weight. Non-numerical performance values carry explicit reasons keyed to their fact-table columns, and win-rate basis is completed trades. Requested empty benchmark data retains its identity and empty status, while unrequested data has no identity.
|
||||
|
||||
This is a storage-neutral candidate. The platform's isolated durable-file adapter must validate this report and expose undefined weights before enabling this path. This projection does not validate a production database schema, execute SQL, admit real data, or grant decision eligibility. Existing factor-artifact consumers remain unchanged.
|
||||
|
||||
## Verification scope
|
||||
|
||||
The missing public modules and sequential policy first failed actual tests. Further actual RED→GREEN regressions cover unfilled exits, legitimate zero NAV, independent ATR/exit windows, benchmark underflow, nonfinite portfolio returns, tiny residual holdings, Bollinger scale invariance, undefined Sortino, premature FIFO closure, full-exit quantity rounding and phantom residues after three accumulated entries at prices 3, 11 and 13.
|
||||
|
||||
Final focused verification is 93 new tests plus 85 existing execution tests, 178 passing in one run. The complete core suite then passed 1186 tests in 15.42 seconds, including governance, existing execution, pipeline, metrics and artifact contracts; 1168 existing-style pandas deprecation warnings remain. The new tests include seven default strategies, six separate trade/cash/NAV hand calculations plus BuyAndHold costs, a triggerable historical ATR stop, future perturbation for all seven, final signals without a next session, FIFO partial exits and same-session fees, and 100 actual optimization trials. Final targeted Ruff passed for all nine affected Python files, and strict typing passed for the four new source modules. Central clean-candidate validation and delivery state are separate receipts.
|
||||
|
||||
Two reused read-only reviewers independently exercised causal prefixes and cash/FIFO examples and found substantive defects subsequently turned into permanent regressions. The final independent FIFO review passed after reproducing three-lot full exits, retaining real tiny balances and reconciling fee-bearing add/partial-exit/full-exit PnL with final NAV. Saved revision and central delivery status are separate evidence. These tests do not establish real data coverage, source/PIT qualification, platform task publication or formal production availability.
|
||||
|
||||
The reviewed implementation was saved and pushed as `7d3e840483d7f3b5d7b2d987c95fb79a5dc4a63b` to the existing PR #21 (https://gitea.puyuanfh.cn/ageorge156/quant_engine/pulls/21), advancing the original 3c97102f candidate. This receipt-only checkpoint does not change the tested code. One explicit central candidate refresh and `ship --ready --confirm-l3` is the next boundary; no prior CI wait was polled or resumed. Push is a save checkpoint, and does not establish merge or source admission.
|
||||
|
||||
Artifact increment verification: 26 new artifact tests plus81 strategy/optimizer and7 existing factor-artifact tests passed together (114). The final complete core suite passed1212 tests in15.59s with1170 pandas deprecation warnings. Four-file Ruff and three-module strict typing passed. The read-only reviewer confirmed the reserved-CASH rejection and fact-column explanation keys after actual regressions. The earlier5192370 central receipt remains CI pending without polling; this new increment is saved and gated separately on the same delivery.
|
||||
@@ -1,39 +0,0 @@
|
||||
# Quant OS factor diagnostics and strategy core increments
|
||||
|
||||
## 2026-10-04 strategy artifact increment
|
||||
|
||||
The existing delivery now adds a storage-neutral strategy artifact projection required by platform Draft #102. The earlier central boundary for candidate5192370a9ecfeca09ed39ee9ca3583060d8777d9 returned `waiting_on_dependency`, CI pending, receipt e28c09224685c20569a8a6f627a2c34b6a1c69f3dc43eadabe918c7458476e45. That candidate and the older3c97102f receipt remain preserved and have not been polled or resumed. This is a new independently completed code increment on the same branch, PR #21, lifecycle and sole writer.
|
||||
|
||||
The result keeps detached input/grid snapshots. The new builder reuses schema1.1.0 and the original ledger, projecting real fills, NAV, positions and metrics while covering all strategy signals, FIFO pairing and up to100 ranked trials in `strategy_report`. It does not fabricate factor scores, attribution or covariance risk. Zero-NAV weight is undefined with a reason. A security cannot claim reserved CASH identity. Production schema/storage and source admission remain outside this pure candidate; platform isolated validation and display still need integration.
|
||||
|
||||
Actual missing-module RED preceded implementation. Independent review found a reproducible cash-identity collision (run/optimizer RED) and fact-column explanation naming inconsistency (RED); both are fixed with permanent regressions and final read-only PASS. Final focused run114 passed (26 new artifact tests,81 strategy/optimizer tests,7 existing factor-artifact tests). Final whole-core run1212 passed in15.59s with1170 pandas deprecation warnings. Targeted four-file Ruff and three-source strict mypy passed. Prior financial calculations and their accepted test evidence are not recomputed by the projection. The fixed49 framework/source and delivery route remain unchanged, resume/review critical gpt-6-astra/xhigh was reused/validated, runtime observation remains unknown.
|
||||
|
||||
Next: save this completed increment and enter one central candidate refresh/Ready gate, then bind its saved revision in the platform isolated task/artifact path. The broad platform delivery remains WIP and formal run/optimize remains gated.
|
||||
|
||||
## 2026-10-04 seven-strategy core increment
|
||||
|
||||
The same owner, lifecycle, branch and PR #21 now include the independent seven-strategy calculation increment needed by platform Draft #102. The original 3c97102f CI-wait receipt has not been queried, resumed or treated as resolved. New code is being reviewed as a new candidate; previous fixed-archive platform consumers remain bound to their saved revisions.
|
||||
|
||||
The active source binding is the clean pinned49d0fc5929653a3a98a0edcf1da630237dcec770. Its core feature entry was reused; the delivery operation returned §16/§21 and those ranges were read. Resume/review parameters critical gpt-6-astra/xhigh passed; actual runtime remains unknown. Root remains sole code/Git writer. The business stage source is absent; no foreign stage ledger or task is borrowed.
|
||||
|
||||
Pure OHLC research now uses the existing daily ledger with a post-close policy and next-open execution. Seven signal definitions, explicit historical ATR stop, source/benchmark states, fee/cash timing, completed-lot FIFO accounting and a maximum100-combination real optimizer are implemented. Financial scope is L3; no source access, DB, migration, live execution, service reload or platform production admission is included. Precise methods, numeric guards and current evidence are in docs/strategy-research.md.
|
||||
|
||||
Actual tests first failed for missing APIs. Review and additional boundary tests reproduced and closed the documented unfilled-exit, numeric, fractional-holding and FIFO defects. Final focused run178 passed (93 new plus85 existing execution), then the entire core suite1186 passed in15.42s with1168 pandas deprecation warnings. Final nine-file Ruff and four-module mypy passed. Both reused reviewers passed their final relevant scopes; the final FIFO review independently reconciled fee-bearing add/partial-exit/full-exit PnL with final NAV and retained real tiny balances. The pure-library increment is code complete; clean-candidate save/delivery receipts follow at the final boundary. The broader platform/M0–M5 scope remains unfinished. Commit-budget continuation: this same delivery now includes an independently tested seven-strategy calculation increment plus necessary saved-revision receipts; preserve reviewed history rather than split or rewrite the delivery. Next: complete this new core candidate's applicable review and one central delivery boundary, then bind the saved source in the platform candidate task/artifact path without opening formal run/optimize.
|
||||
|
||||
The reviewed implementation was saved and pushed as `7d3e840483d7f3b5d7b2d987c95fb79a5dc4a63b` to the existing PR #21 (https://gitea.puyuanfh.cn/ageorge156/quant_engine/pulls/21), advancing the original 3c97102f candidate. This receipt-only checkpoint does not change the tested code. One explicit central candidate refresh and `ship --ready --confirm-l3` is the next boundary; no prior CI wait was polled or resumed. Push is a save checkpoint, and does not establish merge or source admission.
|
||||
|
||||
## Previous factor-diagnostics increment
|
||||
|
||||
Delivery: quant-os-factor-diagnostics-20261004; branch codex/quant-os-factor-diagnostics-20261004. Base is remote-confirmed origin/main 861c1e97a8bf1c3e962c4cd1ee88ef58e6b9ddd5. This is the independent core-repository increment consumed by the existing platform delivery/Draft #102, not a new platform branch or chat. Root in chat 01a0bc8f-dcaf-7452-9ab3-3215df6dfa97 is the sole core code/Git writer; reviewers are read-only. Primary, merged retrospective-v2 and old Alpha158 worktrees and their user files remain untouched.
|
||||
|
||||
The fixed framework source is 16e96351fbc5bd5a918c7deff556490bbc44fb99. Valid source cleanliness/version evidence was reused; actual core feature/worktree router ranges and applicable AGENTS were read. Resume/review model-policy critical gpt-6-astra/xhigh passed; actual runtime remains unknown. The core has no stage ledger at its declared path, so no foreign stage state is borrowed. Lifecycle created this task after policy-check; old completed trees were retained for ignored local data, no old lease/state was rewritten. The new .venv was created through lifecycle run with the existing frozen lock and dev extras, offline from local caches.
|
||||
|
||||
Scope is L3 mathematical behavior, pure memory/caller data. The new candidate module provides daily IC/RankIC summaries, keyed pooled correlation matrices and explicit-calendar forward-return labels. Production algorithm/source/PIT/execution qualifications remain unestablished and decision_eligible=false. Existing factor-library and governed factor-set contracts are unchanged. Reuse rationale and exact numerical/statistical assumptions are in docs/factor-diagnostics.md.
|
||||
|
||||
Evidence so far: initial47 tests failed for the missing API; first implementation46 passed and one strict floating-zero assertion failed, corrected to a stated1e-15 tolerance without clipping values. A genuine extreme-rank scaling regression was reproduced and fixed by ranking original values. An affine-equivalent IC case reproduced enormous IR/t from rounding dispersion, fixed with the public32eps resolution policy. Independent reviewer found two representable-offset Pearson errors; both were reproduced and fixed by translate-before-scale. Latest54 new plus16 old tests=70 passed; final whole-core run1093 passed in12.82s with1166 existing pandas deprecation warnings. Targeted Ruff and final module-only mypy passed. Reviewer rechecked the two numeric fixes with tiny independent samples, no new blocker.
|
||||
|
||||
The implementation was saved and pushed as 69383a30b079790018edb25679be08460a5ab17f in Draft PR #21 (https://gitea.puyuanfh.cn/ageorge156/quant_engine/pulls/21). The existing platform delivery now consumes that exact prepared archive through an unwired thin adapter. Its 25 tests call this real core; 9 isolated HTTP tests and 21 actual-component tests pass. The local browser displayed daily IC [1, -0.5, approximately 0], mean 1/6, actual daily pairs, pooled pairs, constant/missing reasons and holding-period valid days [3, 2, 1]; source failure cleared old results and explicit retry restored them. The prepared-report boundary rejects conflicting counts, intervals and method claims without recalculating statistics. Platform changes remain its separate, unmerged delivery; this is evidence for a consumer, not combined production acceptance.
|
||||
|
||||
The bounded pure-library increment is code complete and reviewed. The next operation is the central delivery gate for this core PR; its receipt, not this handoff, establishes Ready/CI/merge/main acceptance and cleanup facts. After successful core delivery, the platform will bind the accepted core source revision while retaining candidate and unestablished qualifications. Main full-suite evidence remains the unchanged 1093-pass candidate run above until the gate establishes its own validation receipt.
|
||||
|
||||
No source/provider/NAS query, production ETL, migration, release, deployment or trade occurred in this increment. Production data/source/PIT/algorithm admission is outside this pure-library delivery and remains explicitly unestablished. The broader platform/M0–M5 delivery remains unfinished. Reuse this same lifecycle/branch/PR; a push alone is not completion.
|
||||
@@ -1,88 +0,0 @@
|
||||
# Quant Engine retrospective v2 compatibility
|
||||
|
||||
Scope: implement the user-authorized retrospective v2 compatibility without changing
|
||||
v1 semantics, financial algorithms, original results, production databases, deployment
|
||||
or trading. No claim of complete Quant OS delivery or real-data qualification.
|
||||
|
||||
Branch: `codex/research-quant-os-retrospective-contract-v2-20260908`.
|
||||
Declared base: accepted `main@68dd68392a26251391fbdae40c22eee370adb56e`.
|
||||
One isolated delivery worktree; the old primary checkout is preserved. This is not
|
||||
reactivation of an old registered stage or creation of a new stage ledger.
|
||||
|
||||
## Dependency baseline
|
||||
|
||||
Public RP data-contract candidate: PR #100, initial schemas/goldens at
|
||||
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (review/acceptance pending).
|
||||
EDB mapping candidate: PR #13, initial implementation `88433df`, local full
|
||||
validation passed. Neither candidate is silently treated as accepted owner evidence.
|
||||
The shared public major is 2.0.0; preserve the accepted v1 paths independently.
|
||||
|
||||
Order: public data contracts -> EDB mapping/Foundation -> Quant Engine typed
|
||||
factor/backtest/portfolio/risk -> RP governance -> Research Results -> RP read.
|
||||
Accepted owner-version bindings and runtime admission must still close every boundary.
|
||||
|
||||
## Internal reuse decision
|
||||
|
||||
Need: carry observation-aware inputs and retrospective-only claims through computation.
|
||||
Existing: strict canonical JSON, immutable envelopes, factor definitions, input/output
|
||||
closure, numerical algorithms, governed backtest and portfolio/risk contracts.
|
||||
External candidates: not needed; this is project-owned semantics, not a missing library.
|
||||
Approach: reuse those primitives and algorithms; introduce explicit new-major wrappers
|
||||
only where upstream identity, time or usage semantics change.
|
||||
Risk: reusing the v1 decoder or coercing observed-by into knowledge/PIT would make a
|
||||
false historical claim. Unknown versions and unsupported usages must fail closed.
|
||||
|
||||
## Implemented, not yet accepted or released
|
||||
|
||||
Five separate v2 modules now implement immutable DatasetSnapshot/Foundation decoding
|
||||
and materialized-content verification, FactorSet with explicit v2 nested bindings,
|
||||
BacktestRunRef and replay ancestry, nine-table BacktestEvidenceManifest,
|
||||
PerformanceEvidence, PortfolioTarget/Decision and RiskAssessment. The metadata
|
||||
registers the new major alongside every existing v1 entry. See
|
||||
`docs/RETROSPECTIVE_COMPUTATION_V2.md` for normative clocks, JSON profiles, input
|
||||
closure, replay and owner-port boundaries.
|
||||
|
||||
Factor definitions, generic output/receipt/constraint/covariance primitives and
|
||||
financial implementations are reused without semantic edits. Table schema remains
|
||||
1.1.0; v1 business source, v1 goldens, `pyproject.toml`, `uv.lock` and `ci-profile.yml`
|
||||
are unchanged. Package version remains unreleased. Only module metadata, its exact
|
||||
inventory test and README gain v2 alongside the new files.
|
||||
|
||||
The frozen synthetic computation vector includes fresh factor/backtest/manifest/
|
||||
performance/target/portfolio/risk documents and synthetic artifact tables. It is
|
||||
explicitly **envelope-only**, not an end-to-end claim that the one-day data fixture
|
||||
produced the four-day synthetic financial artifact. No old real run was rerun,
|
||||
retagged or backdated.
|
||||
|
||||
## Local verification (2026-09-08)
|
||||
|
||||
- Full repository unit suite: **1039 passed**, 1166 warnings, 31.50 seconds.
|
||||
- S4 focused new + unchanged v1 contracts: **120 passed**; new S4 332 statements,
|
||||
20 branches, 100% measured coverage. Coverage is not source authentication or
|
||||
proof of complete business semantics.
|
||||
- All five new source modules passed mypy; all six new test modules, five new
|
||||
sources and the updated metadata test passed Ruff.
|
||||
- The combined synthetic vector and metadata smoke checks: **3 passed**.
|
||||
- Actual negative tests reproduced and fixed missing covariance-estimation context
|
||||
in result identity, untyped malformed-JSON errors, and risk-time stale-manifest
|
||||
reuse. Other modules' earlier RED/GREEN evidence remains part of the same turn.
|
||||
|
||||
The full suite was run directly against the frozen local environment. This is not
|
||||
the same claim as remote CI or central ship acceptance; the unchanged declared CI
|
||||
profile is `lite` with the module-metadata smoke command. Central validation and
|
||||
Draft PR creation follow the implementation commit. No Ready, merge, accepted
|
||||
upstream binding or independent-review pass is claimed here.
|
||||
|
||||
Actual computation/admission times are distinct from simulated business dates. New
|
||||
formal outputs cannot inherit the old run's producer identity or be backdated to it.
|
||||
Real receipt/qualification/view/clock ports remain mandatory; typed objects and hashes
|
||||
are not source authentication. The optional independent reviewer delegation is still
|
||||
awaiting the already-requested user choice.
|
||||
|
||||
Next: preserve the candidate for review, then carry explicit v2 facts through
|
||||
RP governance -> Research Results publication -> RP read compatibility. Bind final
|
||||
accepted upstream versions only when actual acceptance evidence exists. The entire
|
||||
Quant OS goal is not complete at this intermediate owner unit.
|
||||
|
||||
Rollback: disable the explicit v2 path and retain v1 plus immutable artifacts; never
|
||||
retag v2 into v1 or silently use synthetic evidence for real admission.
|
||||
@@ -15,9 +15,7 @@ v1.2.0 Phase 0:5 个基础算子 + 5 个 alpha 公式(alpha001–alpha005)
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Callable, Mapping
|
||||
from types import MappingProxyType
|
||||
from typing import Any, cast
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
@@ -211,367 +209,6 @@ def indneutralize(series: pd.Series, groups: pd.Series) -> pd.Series:
|
||||
return series - series.groupby(groups).transform("mean")
|
||||
|
||||
|
||||
# ── Phase 1 operator contract ──────────────────────────
|
||||
|
||||
# This is deliberately a small, stable surface for downstream research
|
||||
# orchestration. The full alpha158 formula catalogue can continue to grow,
|
||||
# while callers use one validated dispatch entry point for the first ten
|
||||
# deterministic building blocks.
|
||||
ALPHA158_PHASE1_MAX_WINDOW = 252
|
||||
|
||||
ALPHA158_PHASE1_OPERATOR_SPECS: dict[str, dict[str, Any]] = {
|
||||
"rank": {
|
||||
"name": "rank",
|
||||
"formula": "rank(series)",
|
||||
"inputs": ["series"],
|
||||
"windowed": False,
|
||||
},
|
||||
"delta": {
|
||||
"name": "delta",
|
||||
"formula": "delta(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_mean": {
|
||||
"name": "ts_mean",
|
||||
"formula": "ts_mean(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_std": {
|
||||
"name": "ts_std",
|
||||
"formula": "ts_std(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_rank": {
|
||||
"name": "ts_rank",
|
||||
"formula": "ts_rank(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"correlation": {
|
||||
"name": "correlation",
|
||||
"formula": "correlation(series, secondary, window)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_min": {
|
||||
"name": "ts_min",
|
||||
"formula": "ts_min(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_max": {
|
||||
"name": "ts_max",
|
||||
"formula": "ts_max(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_sum": {
|
||||
"name": "ts_sum",
|
||||
"formula": "ts_sum(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"decay_linear": {
|
||||
"name": "decay_linear",
|
||||
"formula": "decay_linear(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
}
|
||||
|
||||
_PHASE1_OPERATOR_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"rank": rank,
|
||||
"delta": delta,
|
||||
"ts_mean": ts_mean,
|
||||
"ts_std": ts_std,
|
||||
"ts_rank": ts_rank,
|
||||
"correlation": correlation,
|
||||
"ts_min": ts_min,
|
||||
"ts_max": ts_max,
|
||||
"ts_sum": ts_sum,
|
||||
"decay_linear": decay_linear,
|
||||
}
|
||||
|
||||
|
||||
def list_phase1_operators() -> tuple[str, ...]:
|
||||
"""Return the deterministic Phase 1 operator names in stable order."""
|
||||
return tuple(ALPHA158_PHASE1_OPERATOR_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase1_operator(
|
||||
name: str,
|
||||
series: pd.Series,
|
||||
secondary: pd.Series | None = None,
|
||||
*,
|
||||
window: int | None = None,
|
||||
) -> pd.Series:
|
||||
"""Evaluate one of the ten Phase 1 operators with a validated contract.
|
||||
|
||||
``window`` is required for time-series operators and forbidden for the
|
||||
cross-sectional ``rank`` operator. Binary ``correlation`` also requires
|
||||
a same-index secondary series so that callers cannot silently introduce
|
||||
alignment-dependent results.
|
||||
"""
|
||||
if name not in ALPHA158_PHASE1_OPERATOR_SPECS:
|
||||
raise KeyError(f"operator {name!r} not registered")
|
||||
|
||||
is_windowed = bool(ALPHA158_PHASE1_OPERATOR_SPECS[name]["windowed"])
|
||||
if is_windowed:
|
||||
if isinstance(window, bool) or not isinstance(window, int) or window <= 0:
|
||||
raise ValueError(f"window must be a positive integer for {name}")
|
||||
if window > ALPHA158_PHASE1_MAX_WINDOW:
|
||||
raise ValueError(
|
||||
f"window exceeds maximum supported value {ALPHA158_PHASE1_MAX_WINDOW} for {name}"
|
||||
)
|
||||
if not is_windowed and window is not None:
|
||||
raise ValueError(f"window is not supported for {name}")
|
||||
|
||||
if name == "correlation":
|
||||
if secondary is None:
|
||||
raise ValueError("secondary is required for correlation")
|
||||
if not series.index.equals(secondary.index):
|
||||
raise ValueError("secondary index must align with series")
|
||||
return correlation(series, secondary, window) # type: ignore[arg-type]
|
||||
|
||||
if secondary is not None:
|
||||
raise ValueError(f"secondary is not supported for {name}")
|
||||
|
||||
operator = _PHASE1_OPERATOR_FUNCTIONS[name]
|
||||
if name == "rank":
|
||||
return operator(series)
|
||||
return operator(series, window)
|
||||
|
||||
|
||||
# ── Phase 2 cumulative operator contract ──────────────────────────────
|
||||
|
||||
# Phase 2 is cumulative: downstream callers can upgrade to one dispatch
|
||||
# surface covering every existing alpha158 building block, while Phase 1
|
||||
# names, metadata, ordering, and evaluation remain unchanged.
|
||||
ALPHA158_PHASE2_MAX_WINDOW = ALPHA158_PHASE1_MAX_WINDOW
|
||||
|
||||
ALPHA158_PHASE2_OPERATOR_SPECS: dict[str, dict[str, Any]] = {
|
||||
name: {
|
||||
**spec,
|
||||
"parameters": ["window"] if bool(spec["windowed"]) else [],
|
||||
}
|
||||
for name, spec in ALPHA158_PHASE1_OPERATOR_SPECS.items()
|
||||
}
|
||||
ALPHA158_PHASE2_OPERATOR_SPECS.update(
|
||||
{
|
||||
"ts_argmin": {
|
||||
"name": "ts_argmin",
|
||||
"formula": "ts_argmin(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_argmax": {
|
||||
"name": "ts_argmax",
|
||||
"formula": "ts_argmax(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"product": {
|
||||
"name": "product",
|
||||
"formula": "product(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"returns": {
|
||||
"name": "returns",
|
||||
"formula": "returns(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"scale": {
|
||||
"name": "scale",
|
||||
"formula": "scale(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"signed_power": {
|
||||
"name": "signed_power",
|
||||
"formula": "signed_power(series, exponent)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["exponent"],
|
||||
"windowed": False,
|
||||
},
|
||||
"stddev": {
|
||||
"name": "stddev",
|
||||
"formula": "stddev(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"covariance": {
|
||||
"name": "covariance",
|
||||
"formula": "covariance(series, secondary, window)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"log": {
|
||||
"name": "log",
|
||||
"formula": "log(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"abs_series": {
|
||||
"name": "abs_series",
|
||||
"formula": "abs_series(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"sign": {
|
||||
"name": "sign",
|
||||
"formula": "sign(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"max_pair": {
|
||||
"name": "max_pair",
|
||||
"formula": "max_pair(series, secondary)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"min_pair": {
|
||||
"name": "min_pair",
|
||||
"formula": "min_pair(series, secondary)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"indneutralize": {
|
||||
"name": "indneutralize",
|
||||
"formula": "indneutralize(series, groups)",
|
||||
"inputs": ["series", "groups"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
}
|
||||
)
|
||||
|
||||
_PHASE2_OPERATOR_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
**_PHASE1_OPERATOR_FUNCTIONS,
|
||||
"ts_argmin": ts_argmin,
|
||||
"ts_argmax": ts_argmax,
|
||||
"product": product,
|
||||
"returns": returns,
|
||||
"scale": scale,
|
||||
"signed_power": signed_power,
|
||||
"stddev": stddev,
|
||||
"covariance": covariance,
|
||||
"log": log,
|
||||
"abs_series": abs_series,
|
||||
"sign": sign,
|
||||
"max_pair": max_pair,
|
||||
"min_pair": min_pair,
|
||||
"indneutralize": indneutralize,
|
||||
}
|
||||
|
||||
_PHASE2_WINDOWED_OPERATORS = frozenset(
|
||||
name for name, spec in ALPHA158_PHASE2_OPERATOR_SPECS.items() if bool(spec["windowed"])
|
||||
)
|
||||
_PHASE2_BINARY_OPERATORS = frozenset({"correlation", "covariance", "max_pair", "min_pair"})
|
||||
|
||||
|
||||
def list_phase2_operators() -> tuple[str, ...]:
|
||||
"""Return all Phase 2 operator names in stable cumulative order."""
|
||||
return tuple(ALPHA158_PHASE2_OPERATOR_SPECS)
|
||||
|
||||
|
||||
def _validate_phase2_window(name: str, window: int | None) -> int:
|
||||
if isinstance(window, bool) or not isinstance(window, int) or window <= 0:
|
||||
raise ValueError(f"window must be a positive integer for {name}")
|
||||
if window > ALPHA158_PHASE2_MAX_WINDOW:
|
||||
raise ValueError(
|
||||
f"window exceeds maximum supported value {ALPHA158_PHASE2_MAX_WINDOW} for {name}"
|
||||
)
|
||||
return window
|
||||
|
||||
|
||||
def evaluate_phase2_operator(
|
||||
name: str,
|
||||
series: pd.Series,
|
||||
secondary: pd.Series | None = None,
|
||||
*,
|
||||
window: int | None = None,
|
||||
exponent: float | None = None,
|
||||
groups: pd.Series | None = None,
|
||||
) -> pd.Series:
|
||||
"""Evaluate any existing alpha158 building block through a strict contract.
|
||||
|
||||
Phase 2 rejects implicit alignment, missing required arguments, unused
|
||||
arguments, unbounded windows, and non-finite exponents before dispatch.
|
||||
"""
|
||||
if name not in ALPHA158_PHASE2_OPERATOR_SPECS:
|
||||
raise KeyError(f"operator {name!r} not registered")
|
||||
if not isinstance(series, pd.Series):
|
||||
raise TypeError("series must be a pandas Series")
|
||||
|
||||
validated_window: int | None = None
|
||||
if name in _PHASE2_WINDOWED_OPERATORS:
|
||||
validated_window = _validate_phase2_window(name, window)
|
||||
elif window is not None:
|
||||
raise ValueError(f"window is not supported for {name}")
|
||||
|
||||
if name in _PHASE2_BINARY_OPERATORS:
|
||||
if secondary is None:
|
||||
raise ValueError(f"secondary is required for {name}")
|
||||
if not isinstance(secondary, pd.Series):
|
||||
raise TypeError("secondary must be a pandas Series")
|
||||
if not series.index.equals(secondary.index):
|
||||
raise ValueError("secondary index must align with series")
|
||||
elif secondary is not None:
|
||||
raise ValueError(f"secondary is not supported for {name}")
|
||||
|
||||
validated_exponent: float | None = None
|
||||
if name == "signed_power":
|
||||
if (
|
||||
isinstance(exponent, bool)
|
||||
or not isinstance(exponent, (int, float))
|
||||
or not np.isfinite(exponent)
|
||||
):
|
||||
raise ValueError("exponent must be a finite number for signed_power")
|
||||
validated_exponent = float(exponent)
|
||||
elif exponent is not None:
|
||||
raise ValueError(f"exponent is not supported for {name}")
|
||||
|
||||
if name == "indneutralize":
|
||||
if groups is None:
|
||||
raise ValueError("groups is required for indneutralize")
|
||||
if not isinstance(groups, pd.Series):
|
||||
raise TypeError("groups must be a pandas Series")
|
||||
if not series.index.equals(groups.index):
|
||||
raise ValueError("groups index must align with series")
|
||||
elif groups is not None:
|
||||
raise ValueError(f"groups is not supported for {name}")
|
||||
|
||||
operator = _PHASE2_OPERATOR_FUNCTIONS[name]
|
||||
if name == "signed_power":
|
||||
return operator(series, validated_exponent)
|
||||
if name == "indneutralize":
|
||||
return operator(series, groups)
|
||||
if name in {"correlation", "covariance"}:
|
||||
return operator(series, secondary, validated_window)
|
||||
if name in {"max_pair", "min_pair"}:
|
||||
return operator(series, secondary)
|
||||
if validated_window is not None:
|
||||
return operator(series, validated_window)
|
||||
return operator(series)
|
||||
|
||||
|
||||
# ── 组合算子(alpha158 公式样本) ─────────────────────────
|
||||
|
||||
|
||||
@@ -3093,541 +2730,6 @@ def parse_alpha_formula(formula_str: str) -> dict[str, Any]:
|
||||
return parsed
|
||||
|
||||
|
||||
# ── Phase 3 formula contract: frozen alpha001-alpha050 surface ──────────────
|
||||
|
||||
# Formula functions remain the implementation source of truth. This contract
|
||||
# freezes their callable surface separately from formula dependencies so that
|
||||
# historical compatibility-only arguments remain explicit without rewriting
|
||||
# formulas or changing direct-call APIs.
|
||||
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 = (
|
||||
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
|
||||
)
|
||||
|
||||
_PHASE3_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_001": alpha_001,
|
||||
"alpha_002": alpha_002,
|
||||
"alpha_003": alpha_003,
|
||||
"alpha_004": alpha_004,
|
||||
"alpha_005": alpha_005,
|
||||
"alpha_006": alpha_006,
|
||||
"alpha_007": alpha_007,
|
||||
"alpha_008": alpha_008,
|
||||
"alpha_009": alpha_009,
|
||||
"alpha_010": alpha_010,
|
||||
"alpha_011": alpha_011,
|
||||
"alpha_012": alpha_012,
|
||||
"alpha_013": alpha_013,
|
||||
"alpha_014": alpha_014,
|
||||
"alpha_015": alpha_015,
|
||||
"alpha_016": alpha_016,
|
||||
"alpha_017": alpha_017,
|
||||
"alpha_018": alpha_018,
|
||||
"alpha_019": alpha_019,
|
||||
"alpha_020": alpha_020,
|
||||
"alpha_021": alpha_021,
|
||||
"alpha_022": alpha_022,
|
||||
"alpha_023": alpha_023,
|
||||
"alpha_024": alpha_024,
|
||||
"alpha_025": alpha_025,
|
||||
"alpha_026": alpha_026,
|
||||
"alpha_027": alpha_027,
|
||||
"alpha_028": alpha_028,
|
||||
"alpha_029": alpha_029,
|
||||
"alpha_030": alpha_030,
|
||||
"alpha_031": alpha_031,
|
||||
"alpha_032": alpha_032,
|
||||
"alpha_033": alpha_033,
|
||||
"alpha_034": alpha_034,
|
||||
"alpha_035": alpha_035,
|
||||
"alpha_036": alpha_036,
|
||||
"alpha_037": alpha_037,
|
||||
"alpha_038": alpha_038,
|
||||
"alpha_039": alpha_039,
|
||||
"alpha_040": alpha_040,
|
||||
"alpha_041": alpha_041,
|
||||
"alpha_042": alpha_042,
|
||||
"alpha_043": alpha_043,
|
||||
"alpha_044": alpha_044,
|
||||
"alpha_045": alpha_045,
|
||||
"alpha_046": alpha_046,
|
||||
"alpha_047": alpha_047,
|
||||
"alpha_048": alpha_048,
|
||||
"alpha_049": alpha_049,
|
||||
"alpha_050": alpha_050,
|
||||
}
|
||||
|
||||
_PHASE3_FORMULA_INPUT_OVERRIDES: dict[str, list[str]] = {
|
||||
"alpha_011": ["close", "high", "low"],
|
||||
"alpha_035": ["volume"],
|
||||
"alpha_036": ["close"],
|
||||
"alpha_040": ["high", "low"],
|
||||
"alpha_042": ["close"],
|
||||
"alpha_043": ["volume"],
|
||||
}
|
||||
|
||||
_PHASE3_INPUT_CATEGORIES = {
|
||||
1: "single",
|
||||
2: "pair",
|
||||
3: "triple",
|
||||
4: "quadruple",
|
||||
}
|
||||
|
||||
|
||||
def _phase3_call_inputs(function: Callable[..., pd.Series]) -> list[str]:
|
||||
import inspect
|
||||
|
||||
parameters = list(inspect.signature(function).parameters.values())
|
||||
if any(
|
||||
parameter.kind is not inspect.Parameter.POSITIONAL_OR_KEYWORD
|
||||
or parameter.default is not inspect.Parameter.empty
|
||||
for parameter in parameters
|
||||
):
|
||||
raise RuntimeError(f"unsupported formula signature for {function.__name__}")
|
||||
return ["open" if parameter.name == "open_" else parameter.name for parameter in parameters]
|
||||
|
||||
|
||||
def _phase3_string_list(meta: dict[str, Any], field: str, alpha_id: str) -> list[str]:
|
||||
value = meta[field]
|
||||
if not isinstance(value, list) or not all(isinstance(item, str) for item in value):
|
||||
raise RuntimeError(f"{field} must be a list of strings for {alpha_id}")
|
||||
return list(value)
|
||||
|
||||
|
||||
def _build_phase3_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE3_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _PHASE3_FORMULA_INPUT_OVERRIDES.get(alpha_id, call_inputs)
|
||||
input_category = _PHASE3_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": list(formula_inputs),
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
def _freeze_phase3_formula_specs(
|
||||
specs: dict[str, dict[str, Any]],
|
||||
) -> Mapping[str, Mapping[str, Any]]:
|
||||
frozen_specs: dict[str, Mapping[str, Any]] = {}
|
||||
for alpha_id, spec in specs.items():
|
||||
frozen_specs[alpha_id] = MappingProxyType(
|
||||
{field: tuple(value) if isinstance(value, list) else value for field, value in spec.items()}
|
||||
)
|
||||
return MappingProxyType(frozen_specs)
|
||||
|
||||
|
||||
ALPHA158_PHASE3_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase3_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase3_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha001-alpha050 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE3_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase3_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 3 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE3_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE3_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE3_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 4 formula contract: frozen alpha051-alpha100 surface ──────────────
|
||||
|
||||
# Phase 4 extends the versioned formula contract without mutating the Phase 3
|
||||
# catalogue, digest, dispatch surface, or the existing formula functions.
|
||||
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 = (
|
||||
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
|
||||
)
|
||||
|
||||
_PHASE4_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_051": alpha_051,
|
||||
"alpha_052": alpha_052,
|
||||
"alpha_053": alpha_053,
|
||||
"alpha_054": alpha_054,
|
||||
"alpha_055": alpha_055,
|
||||
"alpha_056": alpha_056,
|
||||
"alpha_057": alpha_057,
|
||||
"alpha_058": alpha_058,
|
||||
"alpha_059": alpha_059,
|
||||
"alpha_060": alpha_060,
|
||||
"alpha_061": alpha_061,
|
||||
"alpha_062": alpha_062,
|
||||
"alpha_063": alpha_063,
|
||||
"alpha_064": alpha_064,
|
||||
"alpha_065": alpha_065,
|
||||
"alpha_066": alpha_066,
|
||||
"alpha_067": alpha_067,
|
||||
"alpha_068": alpha_068,
|
||||
"alpha_069": alpha_069,
|
||||
"alpha_070": alpha_070,
|
||||
"alpha_071": alpha_071,
|
||||
"alpha_072": alpha_072,
|
||||
"alpha_073": alpha_073,
|
||||
"alpha_074": alpha_074,
|
||||
"alpha_075": alpha_075,
|
||||
"alpha_076": alpha_076,
|
||||
"alpha_077": alpha_077,
|
||||
"alpha_078": alpha_078,
|
||||
"alpha_079": alpha_079,
|
||||
"alpha_080": alpha_080,
|
||||
"alpha_081": alpha_081,
|
||||
"alpha_082": alpha_082,
|
||||
"alpha_083": alpha_083,
|
||||
"alpha_084": alpha_084,
|
||||
"alpha_085": alpha_085,
|
||||
"alpha_086": alpha_086,
|
||||
"alpha_087": alpha_087,
|
||||
"alpha_088": alpha_088,
|
||||
"alpha_089": alpha_089,
|
||||
"alpha_090": alpha_090,
|
||||
"alpha_091": alpha_091,
|
||||
"alpha_092": alpha_092,
|
||||
"alpha_093": alpha_093,
|
||||
"alpha_094": alpha_094,
|
||||
"alpha_095": alpha_095,
|
||||
"alpha_096": alpha_096,
|
||||
"alpha_097": alpha_097,
|
||||
"alpha_098": alpha_098,
|
||||
"alpha_099": alpha_099,
|
||||
"alpha_100": alpha_100,
|
||||
}
|
||||
|
||||
_PHASE4_INPUT_CATEGORIES = {
|
||||
1: "single",
|
||||
2: "pair",
|
||||
3: "triple",
|
||||
4: "quadruple",
|
||||
5: "quintuple",
|
||||
}
|
||||
|
||||
|
||||
def _build_phase4_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE4_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE4_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase4_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase4_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha051-alpha100 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE4_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase4_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 4 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE4_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE4_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE4_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 5 formula contract: frozen alpha101-alpha150 surface ──────────────
|
||||
|
||||
# Phase 5 extends the versioned formula contract without mutating any earlier
|
||||
# catalogue, digest, dispatch surface, or existing formula implementation.
|
||||
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 = (
|
||||
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
|
||||
)
|
||||
|
||||
_PHASE5_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_101": alpha_101,
|
||||
"alpha_102": alpha_102,
|
||||
"alpha_103": alpha_103,
|
||||
"alpha_104": alpha_104,
|
||||
"alpha_105": alpha_105,
|
||||
"alpha_106": alpha_106,
|
||||
"alpha_107": alpha_107,
|
||||
"alpha_108": alpha_108,
|
||||
"alpha_109": alpha_109,
|
||||
"alpha_110": alpha_110,
|
||||
"alpha_111": alpha_111,
|
||||
"alpha_112": alpha_112,
|
||||
"alpha_113": alpha_113,
|
||||
"alpha_114": alpha_114,
|
||||
"alpha_115": alpha_115,
|
||||
"alpha_116": alpha_116,
|
||||
"alpha_117": alpha_117,
|
||||
"alpha_118": alpha_118,
|
||||
"alpha_119": alpha_119,
|
||||
"alpha_120": alpha_120,
|
||||
"alpha_121": alpha_121,
|
||||
"alpha_122": alpha_122,
|
||||
"alpha_123": alpha_123,
|
||||
"alpha_124": alpha_124,
|
||||
"alpha_125": alpha_125,
|
||||
"alpha_126": alpha_126,
|
||||
"alpha_127": alpha_127,
|
||||
"alpha_128": alpha_128,
|
||||
"alpha_129": alpha_129,
|
||||
"alpha_130": alpha_130,
|
||||
"alpha_131": alpha_131,
|
||||
"alpha_132": alpha_132,
|
||||
"alpha_133": alpha_133,
|
||||
"alpha_134": alpha_134,
|
||||
"alpha_135": alpha_135,
|
||||
"alpha_136": alpha_136,
|
||||
"alpha_137": alpha_137,
|
||||
"alpha_138": alpha_138,
|
||||
"alpha_139": alpha_139,
|
||||
"alpha_140": alpha_140,
|
||||
"alpha_141": alpha_141,
|
||||
"alpha_142": alpha_142,
|
||||
"alpha_143": alpha_143,
|
||||
"alpha_144": alpha_144,
|
||||
"alpha_145": alpha_145,
|
||||
"alpha_146": alpha_146,
|
||||
"alpha_147": alpha_147,
|
||||
"alpha_148": alpha_148,
|
||||
"alpha_149": alpha_149,
|
||||
"alpha_150": alpha_150,
|
||||
}
|
||||
|
||||
|
||||
def _build_phase5_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE5_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE5_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase5_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase5_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha101-alpha150 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE5_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase5_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 5 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE5_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE5_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if len(inputs[field]) != len(primary):
|
||||
raise ValueError(f"{field} length must match {primary_field}")
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE5_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 6 formula contract: frozen alpha151-alpha158 surface ──────────────
|
||||
|
||||
# Phase 6 completes the versioned formula contract without mutating any
|
||||
# earlier catalogue, digest, dispatch surface, or formula implementation.
|
||||
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 = (
|
||||
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
|
||||
)
|
||||
|
||||
_PHASE6_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_151": alpha_151,
|
||||
"alpha_152": alpha_152,
|
||||
"alpha_153": alpha_153,
|
||||
"alpha_154": alpha_154,
|
||||
"alpha_155": alpha_155,
|
||||
"alpha_156": alpha_156,
|
||||
"alpha_157": alpha_157,
|
||||
"alpha_158": alpha_158,
|
||||
}
|
||||
|
||||
|
||||
def _build_phase6_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE6_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE6_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase6_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase6_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha151-alpha158 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE6_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase6_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 6 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE6_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE6_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if len(inputs[field]) != len(primary):
|
||||
raise ValueError(f"{field} length must match {primary_field}")
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE6_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
__all__ = [
|
||||
"rank",
|
||||
"delta",
|
||||
@@ -3653,34 +2755,6 @@ __all__ = [
|
||||
"max_pair",
|
||||
"min_pair",
|
||||
"indneutralize",
|
||||
"ALPHA158_PHASE1_MAX_WINDOW",
|
||||
"ALPHA158_PHASE1_OPERATOR_SPECS",
|
||||
"list_phase1_operators",
|
||||
"evaluate_phase1_operator",
|
||||
"ALPHA158_PHASE2_MAX_WINDOW",
|
||||
"ALPHA158_PHASE2_OPERATOR_SPECS",
|
||||
"list_phase2_operators",
|
||||
"evaluate_phase2_operator",
|
||||
"ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE3_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE3_FORMULA_SPECS",
|
||||
"list_phase3_formulas",
|
||||
"evaluate_phase3_formula",
|
||||
"ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE4_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE4_FORMULA_SPECS",
|
||||
"list_phase4_formulas",
|
||||
"evaluate_phase4_formula",
|
||||
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE5_FORMULA_SPECS",
|
||||
"list_phase5_formulas",
|
||||
"evaluate_phase5_formula",
|
||||
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE6_FORMULA_SPECS",
|
||||
"list_phase6_formulas",
|
||||
"evaluate_phase6_formula",
|
||||
"alpha_001",
|
||||
"alpha_002",
|
||||
"alpha_003",
|
||||
|
||||
+19
-2368
File diff suppressed because it is too large
Load Diff
@@ -22,7 +22,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from collections.abc import Callable, Mapping
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, replace
|
||||
from typing import Any
|
||||
|
||||
@@ -614,21 +614,16 @@ def _rebalance_at_prices(
|
||||
filled: list[ExecutionResult] = []
|
||||
for raw_execution in sell_executions:
|
||||
price = prices[raw_execution.stock_code]
|
||||
held = holdings.get(raw_execution.stock_code, 0.0)
|
||||
# A full exit consumes the exact held quantity. Dividing a rounded
|
||||
# weight-derived notional back by price can otherwise leave a phantom lot.
|
||||
quantity = (held if effective_targets[raw_execution.stock_code] == 0
|
||||
else abs(raw_execution.target_value) / price)
|
||||
quantity = abs(raw_execution.target_value) / price
|
||||
execution = replace(
|
||||
raw_execution,
|
||||
side="sell",
|
||||
quantity=quantity,
|
||||
price=price,
|
||||
)
|
||||
held = holdings.get(execution.stock_code, 0.0)
|
||||
holdings[execution.stock_code] = max(0.0, held - quantity)
|
||||
# Fractional research holdings can be tiny shares with substantial value.
|
||||
# Only a requested full exit or an exact zero removes the position.
|
||||
if effective_targets[execution.stock_code] == 0 or holdings[execution.stock_code] == 0:
|
||||
if holdings[execution.stock_code] < 1e-6:
|
||||
del holdings[execution.stock_code]
|
||||
cash += execution.net_cash_flow
|
||||
filled.append(execution)
|
||||
@@ -711,23 +706,10 @@ def _simulate_daily_ledger(
|
||||
valuation_price_history: list[tuple[str, dict[str, float]]],
|
||||
initial_cash: float,
|
||||
config: ExecutionConfig,
|
||||
decision_policy: Callable[[DailyPosition], Mapping[str, float] | None] | None = None,
|
||||
) -> ExecutionSimulationResult:
|
||||
if decision_policy is not None:
|
||||
if target_weights_history:
|
||||
raise ValueError("decision_policy cannot be combined with a fixed schedule")
|
||||
if not callable(decision_policy):
|
||||
raise ValueError("decision_policy must be callable")
|
||||
if [date for date, _ in execution_price_history] != [date for date, _ in valuation_price_history]:
|
||||
raise ValueError("policy execution prices must cover the complete valuation calendar")
|
||||
# Empty targets here validate the full price calendar only. Policy targets
|
||||
# are produced after a close and consumed at the following session's open.
|
||||
validation_targets = [(date, {}) for date, _ in execution_price_history]
|
||||
else:
|
||||
validation_targets = target_weights_history
|
||||
targets_by_date, execution_prices_by_date, valuation_history = (
|
||||
_validate_sparse_daily_histories(
|
||||
validation_targets,
|
||||
target_weights_history,
|
||||
execution_price_history,
|
||||
valuation_price_history,
|
||||
)
|
||||
@@ -737,9 +719,8 @@ def _simulate_daily_ledger(
|
||||
positions: list[DailyPosition] = []
|
||||
daily_executions: list[DailyExecution] = []
|
||||
|
||||
pending_targets: dict[str, float] | None = None
|
||||
for date, valuation_prices in valuation_history:
|
||||
targets = pending_targets if decision_policy is not None else targets_by_date.get(date)
|
||||
targets = targets_by_date.get(date)
|
||||
if targets is None:
|
||||
executions: tuple[ExecutionResult, ...] = ()
|
||||
nav_before = 0.0
|
||||
@@ -783,11 +764,6 @@ def _simulate_daily_ledger(
|
||||
rebalance_triggered=rebalance_triggered,
|
||||
)
|
||||
)
|
||||
if decision_policy is not None:
|
||||
# The policy receives its own snapshot, never live holdings or a
|
||||
# snapshot already stored in the result. None means no order.
|
||||
proposed = decision_policy(DailyPosition(date, cash, dict(holdings), portfolio_value))
|
||||
pending_targets = None if proposed is None else _validate_target_weights(date, proposed)
|
||||
|
||||
return ExecutionSimulationResult(
|
||||
initial_cash=initial_cash,
|
||||
@@ -802,15 +778,11 @@ def simulate_daily_ledger_with_audit(
|
||||
valuation_price_history: list[tuple[str, dict[str, float]]],
|
||||
initial_cash: float,
|
||||
config: ExecutionConfig | None = None,
|
||||
*,
|
||||
decision_policy: Callable[[DailyPosition], Mapping[str, float] | None] | None = None,
|
||||
) -> ExecutionSimulationResult:
|
||||
"""以稀疏调仓和完整日历运行成交后持仓 Ledger。
|
||||
|
||||
执行价只用于调仓日现金与股数变化,估值价用于每个交易日日末 NAV;二者
|
||||
显式分离,从而支持“下一日 open 成交、同日 close 估值”的无前视研究。
|
||||
可选 decision_policy 在收盘估值后接收实际持仓副本,仅为下一日生成目标;
|
||||
此模式须传空固定目标和完整开盘价日历。最后日决定不会执行。
|
||||
"""
|
||||
if not math.isfinite(initial_cash) or initial_cash <= 0:
|
||||
raise ValueError(f"initial_cash must be positive and finite, got {initial_cash}")
|
||||
@@ -820,7 +792,6 @@ def simulate_daily_ledger_with_audit(
|
||||
valuation_price_history,
|
||||
initial_cash,
|
||||
ExecutionConfig() if config is None else config,
|
||||
decision_policy,
|
||||
)
|
||||
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,325 +0,0 @@
|
||||
"""Pure, observation-keyed factor diagnostics on caller-supplied data.
|
||||
|
||||
These low-level candidate calculations neither fetch sources nor grant data,
|
||||
historical-availability, production-algorithm or execution qualification. The
|
||||
legacy factor_library API remains unchanged. IC summaries here aggregate daily
|
||||
cross sections, never pooled asset/day observations. All dates are naive session
|
||||
labels, not information-availability timestamps.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal
|
||||
import math
|
||||
from numbers import Real
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from scipy import stats
|
||||
|
||||
from quant_engine.factor_library import ic_summary, spearman_ic
|
||||
|
||||
FACTOR_DIAGNOSTICS_VERSION = "0.1.0"
|
||||
# IC is bounded to [-1, 1]. Below this absolute float64 resolution, dispersion
|
||||
# cannot reliably distinguish equivalent cross sections from rounding noise.
|
||||
MINIMUM_IC_STD_FOR_RATIOS = 32 * np.finfo(np.float64).eps
|
||||
|
||||
|
||||
def _qualification() -> dict[str, Any]:
|
||||
return {
|
||||
"contract_version": FACTOR_DIAGNOSTICS_VERSION,
|
||||
"production_algorithm_version": None,
|
||||
"source_admission": "not_established",
|
||||
"historical_availability": "not_established",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def _integer(value: int, minimum: int, name: str) -> None:
|
||||
if type(value) is not int or value < minimum:
|
||||
raise ValueError(f"{name} must be an integer >= {minimum}")
|
||||
|
||||
|
||||
def _label(value: str) -> bool:
|
||||
return (isinstance(value, str) and 0 < len(value) <= 128
|
||||
and value == value.strip() and all(char.isprintable() for char in value))
|
||||
|
||||
|
||||
def _dates(index: pd.Index) -> None:
|
||||
if (not isinstance(index, pd.DatetimeIndex) or index.tz is not None
|
||||
or index.hasnans or not index.equals(index.normalize())):
|
||||
raise ValueError("Expected naive midnight session dates")
|
||||
|
||||
|
||||
def _keys(index: pd.Index) -> None:
|
||||
if (not isinstance(index, pd.MultiIndex) or list(index.names) != ["date", "asset"]
|
||||
or not index.is_unique):
|
||||
raise ValueError("Expected unique (date, asset) observation keys")
|
||||
_dates(index.get_level_values("date"))
|
||||
if any(not _label(asset) for asset in index.get_level_values("asset")):
|
||||
raise ValueError("Asset identifiers must be nonempty strings")
|
||||
|
||||
|
||||
def _numbers(series: pd.Series) -> pd.Series:
|
||||
numbers = []
|
||||
for value in series:
|
||||
if value is None or value is pd.NA:
|
||||
numbers.append(math.nan)
|
||||
continue
|
||||
if isinstance(value, (bool, np.bool_)) or not isinstance(value, (Real, Decimal)):
|
||||
raise ValueError("Values must be real numbers or missing, without coercion")
|
||||
try:
|
||||
number = float(value)
|
||||
except (OverflowError, ValueError) as exc:
|
||||
raise ValueError("Value cannot be represented as a finite number") from exc
|
||||
if math.isinf(number):
|
||||
raise ValueError("Infinite observations are not permitted")
|
||||
numbers.append(number)
|
||||
return pd.Series(numbers, index=series.index, name=series.name, dtype=float)
|
||||
|
||||
|
||||
def _panel(frame: pd.DataFrame) -> pd.DataFrame:
|
||||
if not isinstance(frame, pd.DataFrame):
|
||||
raise ValueError("Expected a factor DataFrame")
|
||||
_keys(frame.index)
|
||||
if not frame.columns.is_unique or any(not _label(name) for name in frame.columns):
|
||||
raise ValueError("Factor identifiers must be unique nonempty strings")
|
||||
result = pd.DataFrame(index=frame.index)
|
||||
for name in frame.columns:
|
||||
result[name] = _numbers(frame[name])
|
||||
return result.sort_index()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _Correlation:
|
||||
value: float | None
|
||||
n_pairs: int
|
||||
status: str
|
||||
|
||||
|
||||
def _center_scale(series: pd.Series) -> pd.Series:
|
||||
# Translate first to preserve distinguishable low bits beside a large offset.
|
||||
# Opposite extreme endpoints may overflow subtraction; only then scale first.
|
||||
with np.errstate(all="ignore"):
|
||||
shifted = series - series.iloc[0]
|
||||
if not bool(np.isfinite(shifted).all()):
|
||||
bounded = series / series.abs().max()
|
||||
shifted = bounded - bounded.iloc[0]
|
||||
return shifted / shifted.abs().max()
|
||||
|
||||
|
||||
def _correlation(left: pd.Series, right: pd.Series, method: str, minimum: int) -> _Correlation:
|
||||
# Callers already bind the observation domain. Concat still aligns complete
|
||||
# keys rather than independently deleting missing left and right values.
|
||||
paired = pd.concat([left.rename("left"), right.rename("right")], axis=1).dropna()
|
||||
count = len(paired)
|
||||
if count < minimum:
|
||||
return _Correlation(None, count, "no_pairs" if count == 0 else "insufficient_pairs")
|
||||
constant_left = bool((paired["left"] == paired["left"].iloc[0]).all())
|
||||
constant_right = bool((paired["right"] == paired["right"].iloc[0]).all())
|
||||
if constant_left or constant_right:
|
||||
reason = ("constant_both" if constant_left and constant_right else
|
||||
"constant_left" if constant_left else "constant_right")
|
||||
return _Correlation(None, count, reason)
|
||||
# Pearson needs bounded magnitudes to avoid covariance overflow. Rank the
|
||||
# original observations: scaling could underflow distinct tiny values into
|
||||
# artificial ties next to a very large outlier.
|
||||
with np.errstate(all="ignore"):
|
||||
if method == "pearson":
|
||||
scaled_left = _center_scale(paired["left"])
|
||||
scaled_right = _center_scale(paired["right"])
|
||||
value = float(ic_summary(scaled_left, scaled_right, periods=(1,),
|
||||
method="pearson").loc[1, "ic_mean"])
|
||||
else:
|
||||
value = float(spearman_ic(paired["left"], paired["right"]))
|
||||
if not math.isfinite(value) or abs(value) > 1 + 1e-12:
|
||||
return _Correlation(None, count, "numerical_failure")
|
||||
return _Correlation(max(-1.0, min(1.0, value)), count, "ok")
|
||||
|
||||
|
||||
def _summary(values: list[float | None], minimum: int) -> dict[str, Any]:
|
||||
available = [value for value in values if value is not None]
|
||||
count = len(available)
|
||||
result: dict[str, Any] = {
|
||||
"valid_days": count, "missing_days": len(values) - count,
|
||||
"mean": None, "std": None, "ir": None, "t": None, "p": None,
|
||||
"status": "no_valid_days" if count == 0 else "insufficient_days",
|
||||
}
|
||||
if count == 0:
|
||||
return result
|
||||
constant = all(value == available[0] for value in available)
|
||||
mean = available[0] if constant else math.fsum(available) / count
|
||||
result["mean"] = mean
|
||||
if count < minimum:
|
||||
return result
|
||||
std = 0.0 if constant else float(np.std(available, ddof=1))
|
||||
result["std"] = std
|
||||
if constant:
|
||||
result["status"] = "constant_values"
|
||||
return result
|
||||
if not math.isfinite(std) or std <= 0:
|
||||
result["std"] = None
|
||||
result["status"] = "numerical_failure"
|
||||
return result
|
||||
if std <= MINIMUM_IC_STD_FOR_RATIOS:
|
||||
result["status"] = "below_resolution"
|
||||
return result
|
||||
ir = mean / std
|
||||
t_value = ir * math.sqrt(count)
|
||||
p_value = float(2 * stats.t.sf(abs(t_value), df=count - 1))
|
||||
if not all(math.isfinite(value) for value in (ir, t_value, p_value)):
|
||||
result["status"] = "numerical_failure"
|
||||
return result
|
||||
result.update(ir=ir, t=t_value, p=p_value, status="ok")
|
||||
return result
|
||||
|
||||
|
||||
def daily_ic(
|
||||
factors: pd.DataFrame, returns: pd.Series, *, min_pairs: int = 3, min_days: int = 2,
|
||||
) -> dict[str, Any]:
|
||||
"""Daily Pearson/average-tie RankIC and unannualized daily-IC summaries.
|
||||
|
||||
Factors define the observation domain. Extra return keys are ignored and
|
||||
counted; missing return keys remain unavailable. Every factor/date is kept,
|
||||
including zero-pair dates. IID t/p are descriptive only: serial correlation
|
||||
and overlapping holding intervals are not corrected or admitted.
|
||||
"""
|
||||
_integer(min_pairs, 3, "min_pairs")
|
||||
_integer(min_days, 2, "min_days")
|
||||
frame = _panel(factors)
|
||||
if not isinstance(returns, pd.Series):
|
||||
raise ValueError("Expected a forward-return Series")
|
||||
_keys(returns.index)
|
||||
clean_returns = _numbers(returns)
|
||||
aligned = clean_returns.reindex(frame.index)
|
||||
rows: list[dict[str, Any]] = []
|
||||
summaries = []
|
||||
for factor_id in frame.columns:
|
||||
points = []
|
||||
for day, left in frame[factor_id].groupby(level="date", sort=True):
|
||||
right = aligned.reindex(left.index)
|
||||
pearson = _correlation(left, right, "pearson", min_pairs)
|
||||
rank = _correlation(left, right, "spearman", min_pairs)
|
||||
point = {
|
||||
"date": day.strftime("%Y-%m-%d"), "factor_id": factor_id,
|
||||
"ic": pearson.value, "rank_ic": rank.value, "n_pairs": pearson.n_pairs,
|
||||
"n_observations": len(left), "n_factor": int(left.notna().sum()),
|
||||
"n_return": int(right.notna().sum()),
|
||||
"ic_status": pearson.status, "rank_ic_status": rank.status,
|
||||
}
|
||||
rows.append(point)
|
||||
points.append(point)
|
||||
summary: dict[str, Any] = {"factor_id": factor_id, "n_days": len(points)}
|
||||
for field in ("ic", "rank_ic"):
|
||||
summary.update({f"{field}_{key}": value for key, value in
|
||||
_summary([point[field] for point in points], min_days).items()})
|
||||
summaries.append(summary)
|
||||
return {
|
||||
**_qualification(), "rows": rows, "summary": summaries,
|
||||
"method": {"aggregation": "equal_weight_daily_cross_sections", "min_pairs": min_pairs,
|
||||
"min_days": min_days, "standard_deviation_ddof": 1, "ir_annualized": False,
|
||||
"minimum_ic_std_for_ratios": float(MINIMUM_IC_STD_FOR_RATIOS),
|
||||
"ic_std_resolution_policy": "absolute_32_float64_eps",
|
||||
"rank_ties": "average", "t_method": "naive_iid_unadjusted",
|
||||
"pearson_normalization": "translate_then_scale_with_overflow_fallback",
|
||||
"p_method": "two_sided_student_t", "t_degrees_of_freedom": "valid_days - 1",
|
||||
"serial_correlation_adjusted": False, "holding_overlap_adjusted": False,
|
||||
"observation_domain": "factor_keys", "missing_policy": "pairwise_complete_keys",
|
||||
"extra_return_keys": len(returns.index.difference(frame.index))},
|
||||
}
|
||||
|
||||
|
||||
def correlation_matrix(
|
||||
factors: pd.DataFrame, *, method: str = "pearson", min_pairs: int = 3,
|
||||
) -> dict[str, Any]:
|
||||
"""Pool keyed asset/session pairs, not an average of daily correlations.
|
||||
|
||||
Larger cross sections contribute more pairs. Pairwise deletion may produce
|
||||
a non-PSD matrix; this output is not a covariance/risk-matrix contract.
|
||||
"""
|
||||
_integer(min_pairs, 3, "min_pairs")
|
||||
if method not in ("pearson", "spearman"):
|
||||
raise ValueError("method must be pearson or spearman")
|
||||
frame = _panel(factors)
|
||||
names = list(frame.columns)
|
||||
estimates = [[_correlation(frame[left], frame[right], method, min_pairs)
|
||||
for right in names] for left in names]
|
||||
return {
|
||||
**_qualification(), "factors": names, "n_observations": len(frame),
|
||||
"matrix": [[estimate.value for estimate in row] for row in estimates],
|
||||
"n_pairs": [[estimate.n_pairs for estimate in row] for row in estimates],
|
||||
"status": [[estimate.status for estimate in row] for row in estimates],
|
||||
"method": {"correlation": method, "aggregation": "pooled_asset_session_pairwise",
|
||||
"pearson_normalization": "translate_then_scale_with_overflow_fallback",
|
||||
"min_pairs": min_pairs, "missing_policy": "pairwise_complete_keys",
|
||||
"rank_ties": "average", "positive_semidefinite_guaranteed": False},
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ForwardReturns:
|
||||
"""Owned output frames; frozen attributes do not make pandas objects immutable."""
|
||||
returns: pd.Series
|
||||
intervals: pd.DataFrame
|
||||
metadata: dict[str, Any]
|
||||
|
||||
|
||||
def forward_returns(
|
||||
prices: pd.Series, *, sessions: pd.DatetimeIndex, entry_lag_sessions: int,
|
||||
holding_sessions: int, price_field: str, price_basis: str,
|
||||
) -> ForwardReturns:
|
||||
"""Label P[t+lag+holding]/P[t+lag]-1 on an explicit session calendar.
|
||||
|
||||
Requires comparable endpoint prices, not every intermediate price. Missing
|
||||
keys/prices remain missing on the supplied calendar. lag=0 is a same-session
|
||||
price basis, not a claim that a closing signal can execute at that close.
|
||||
"""
|
||||
_integer(entry_lag_sessions, 0, "entry_lag_sessions")
|
||||
_integer(holding_sessions, 1, "holding_sessions")
|
||||
if not _label(price_field) or not _label(price_basis):
|
||||
raise ValueError("Explicit price field and comparable-price basis are required")
|
||||
_dates(sessions)
|
||||
if not sessions.is_unique or not sessions.is_monotonic_increasing:
|
||||
raise ValueError("Calendar sessions must be unique and increasing")
|
||||
if not isinstance(prices, pd.Series):
|
||||
raise ValueError("Expected a keyed price Series")
|
||||
_keys(prices.index)
|
||||
values = _numbers(prices)
|
||||
if len(prices.index.get_level_values("date").difference(sessions)):
|
||||
raise ValueError("Price observation outside the supplied calendar")
|
||||
if bool((values.dropna() <= 0).any()):
|
||||
raise ValueError("Endpoint prices must be positive when present")
|
||||
assets = sorted(prices.index.get_level_values("asset").unique())
|
||||
index = pd.MultiIndex.from_product([sessions, assets], names=["date", "asset"])
|
||||
grid = values.reindex(index)
|
||||
output, intervals = [], []
|
||||
for position, signal_date in enumerate(sessions):
|
||||
entry_position = position + entry_lag_sessions
|
||||
exit_position = entry_position + holding_sessions
|
||||
entry_date = sessions[entry_position] if entry_position < len(sessions) else None
|
||||
exit_date = sessions[exit_position] if exit_position < len(sessions) else None
|
||||
for asset in assets:
|
||||
value, status = math.nan, "insufficient_calendar"
|
||||
if entry_date is not None and exit_date is not None:
|
||||
entry, exit_price = grid.loc[(entry_date, asset)], grid.loc[(exit_date, asset)]
|
||||
status = "missing_price"
|
||||
if pd.notna(entry) and pd.notna(exit_price):
|
||||
with np.errstate(over="ignore", invalid="ignore"):
|
||||
value = float(exit_price / entry - 1)
|
||||
status = "ok" if math.isfinite(value) else "numerical_failure"
|
||||
if status != "ok":
|
||||
value = math.nan
|
||||
output.append(value)
|
||||
intervals.append({"signal_date": signal_date.strftime("%Y-%m-%d"),
|
||||
"entry_date": entry_date.strftime("%Y-%m-%d") if entry_date is not None else None,
|
||||
"exit_date": exit_date.strftime("%Y-%m-%d") if exit_date is not None else None,
|
||||
"status": status})
|
||||
metadata = {**_qualification(), "formula": "exit_price / entry_price - 1", "unit": "ratio",
|
||||
"entry_lag_sessions": entry_lag_sessions, "holding_sessions": holding_sessions,
|
||||
"price_field": price_field, "price_basis": price_basis,
|
||||
"price_coverage": "endpoints_only", "calendar": "explicit_caller_sessions",
|
||||
"missing_policy": "no_fill_no_session_skipping", "execution_eligibility": "not_established"}
|
||||
return ForwardReturns(pd.Series(output, index=index, dtype=float, name="forward_return"),
|
||||
pd.DataFrame(intervals, index=index,
|
||||
columns=["signal_date", "entry_date", "exit_date", "status"]), metadata)
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,489 +0,0 @@
|
||||
"""Retrospective-only evidence wrappers over the unchanged research fact tables."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Self, cast
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from quant_engine.artifact import (
|
||||
BacktestEvidenceEntry,
|
||||
EvidenceQualification,
|
||||
ResearchRunArtifact,
|
||||
PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
PERFORMANCE_METHODOLOGY_ID,
|
||||
PerformanceMethodology,
|
||||
PerformanceMetric,
|
||||
_PERFORMANCE_SOURCE_COLUMNS,
|
||||
_absolute_performance_metrics,
|
||||
_benchmark_context,
|
||||
_count_performance_metrics,
|
||||
_performance_canonical_bytes,
|
||||
_performance_compare,
|
||||
_performance_date,
|
||||
_performance_digest,
|
||||
_performance_methodology,
|
||||
_performance_text,
|
||||
_performance_validate_tree,
|
||||
_relative_performance_metrics,
|
||||
_artifact_frames,
|
||||
_evidence_entries,
|
||||
_evidence_frame_records,
|
||||
_manifest_instant,
|
||||
_run_row,
|
||||
_table_evidence,
|
||||
_validate_table_run_ids,
|
||||
)
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
FactorContractError,
|
||||
_assert_canonical_profile,
|
||||
_content_address,
|
||||
_digest_bytes,
|
||||
_duplicate_key_pairs,
|
||||
_freeze_json,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_data_contracts import _check, _public, _shape
|
||||
|
||||
|
||||
def _validated_run(run: Any) -> RetrospectiveBacktestRunRef:
|
||||
_check(
|
||||
type(run) is RetrospectiveBacktestRunRef,
|
||||
"$.run_ref",
|
||||
"explicit v2 run reference required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
factor = run._factor_set
|
||||
return RetrospectiveBacktestRunRef.from_dict(
|
||||
run.to_dict(),
|
||||
dataset_snapshot=factor._dataset_snapshot,
|
||||
foundation=factor._foundation,
|
||||
factor_set=factor,
|
||||
parent=run._parent,
|
||||
)
|
||||
|
||||
|
||||
def _validated_frames(
|
||||
artifact: ResearchRunArtifact, run: RetrospectiveBacktestRunRef
|
||||
) -> dict[str, pd.DataFrame]:
|
||||
frames = _artifact_frames(artifact)
|
||||
_validate_table_run_ids(frames, run.run_id)
|
||||
row = _run_row(frames)
|
||||
expected = {
|
||||
"run_id": run.run_id,
|
||||
"data_snapshot_id": run.dataset_snapshot_id,
|
||||
"strategy_id": run.strategy_id,
|
||||
"strategy_version": run.strategy_version,
|
||||
"code_revision": run.code_revision,
|
||||
"config_hash": run.configuration_digest.removeprefix("sha256:"),
|
||||
"schema_version": artifact.schema_version,
|
||||
}
|
||||
_check(
|
||||
set(expected) | {"started_at", "finished_at"} <= set(row.index),
|
||||
"$.artifact.tables.run",
|
||||
"run schema fields missing",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for key, value in expected.items():
|
||||
_check(
|
||||
type(row[key]) is str and row[key] == value,
|
||||
f"$.artifact.tables.run.{key}",
|
||||
"artifact does not bind exact v2 run",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
# The unchanged artifact 1.1 timestamp profile admits offsets; public v2
|
||||
# envelope times remain strict UTC. No knowledge-time inference is performed.
|
||||
_, started = _manifest_instant(row["started_at"], "$.artifact.tables.run.started_at")
|
||||
_, finished = _manifest_instant(row["finished_at"], "$.artifact.tables.run.finished_at")
|
||||
_check(
|
||||
_parse_utc(run.evaluation_at, "$.run_ref.evaluation_at")
|
||||
<= started
|
||||
<= finished
|
||||
<= _parse_utc(run.computed_at, "$.run_ref.computed_at"),
|
||||
"$.artifact.tables.run",
|
||||
"actual evaluation <= start <= finish <= computed required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
for name, frame in frames.items():
|
||||
_public(_evidence_frame_records(frame, name), f"$.artifact.tables.{name}")
|
||||
return frames
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveBacktestEvidenceManifest:
|
||||
"""Exact artifact closure, not authenticity, historical or execution authority."""
|
||||
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
manifest_id: str
|
||||
run_id: str
|
||||
profile: str
|
||||
artifact_schema_version: str
|
||||
artifact_available_at: str
|
||||
qualification: EvidenceQualification
|
||||
evidence_digest: str
|
||||
evidence: tuple[BacktestEvidenceEntry, ...]
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
observation_cutoff: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_artifact: ResearchRunArtifact = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
artifact: ResearchRunArtifact,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
) -> Self:
|
||||
_assert_canonical_profile(value)
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version manifest_id run_id profile artifact_schema_version artifact_available_at qualification "
|
||||
"run_reference evidence_digest evidence evidence_scope usage historical_availability observation_cutoff decision_eligible execution_validation",
|
||||
)
|
||||
_check(
|
||||
type(row["qualification"]) is str
|
||||
and row["qualification"] in {"exploratory", "contract_qualified"},
|
||||
"$.qualification",
|
||||
"explicit non-legacy contract qualification required",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
rebuilt = build_retrospective_backtest_evidence_manifest(
|
||||
backtest_run_ref,
|
||||
artifact,
|
||||
artifact_available_at=row["artifact_available_at"],
|
||||
qualification=EvidenceQualification(row["qualification"]),
|
||||
)
|
||||
_check(
|
||||
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
|
||||
"$",
|
||||
"manifest differs from actual run/table closure",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_backtest_evidence_manifest(
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
artifact: ResearchRunArtifact,
|
||||
*,
|
||||
artifact_available_at: str,
|
||||
qualification: EvidenceQualification = EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
expected_table_digests: Mapping[str, str] | None = None,
|
||||
) -> RetrospectiveBacktestEvidenceManifest:
|
||||
"""Close new in-memory artifact bytes; never promote an old exploratory run."""
|
||||
run = _validated_run(backtest_run_ref)
|
||||
_check(
|
||||
type(qualification) is EvidenceQualification
|
||||
and qualification is not EvidenceQualification.LEGACY_EXPLORATORY,
|
||||
"$.qualification",
|
||||
"legacy evidence cannot enter the v2 path",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
available = _parse_utc(artifact_available_at, "$.artifact_available_at")
|
||||
_check(
|
||||
_parse_utc(run.computed_at, "$.run_ref.computed_at") <= available,
|
||||
"$.artifact_available_at",
|
||||
"artifact precedes actual computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
frames = _validated_frames(artifact, run)
|
||||
summaries = _table_evidence(frames, expected_table_digests)
|
||||
reference: dict[str, object] = {"kind": "backtest_run_ref", "value": run.to_dict()}
|
||||
# Table categories and canonical content hashing have not changed semantics.
|
||||
evidence = _evidence_entries(summaries, reference, legacy=False)
|
||||
evidence_digest = _digest_bytes(canonical_json_bytes([item.to_dict() for item in evidence]))
|
||||
payload = {
|
||||
"contract_name": "researchhub.backtest-evidence-manifest",
|
||||
"schema_version": "2.0.0",
|
||||
"run_id": run.run_id,
|
||||
"profile": "offline_research_retrospective_v2",
|
||||
"artifact_schema_version": artifact.schema_version,
|
||||
"artifact_available_at": artifact_available_at,
|
||||
"qualification": qualification.value,
|
||||
"run_reference": reference,
|
||||
"evidence_digest": evidence_digest,
|
||||
"evidence": [item.to_dict() for item in evidence],
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
payload["manifest_id"] = _content_address(
|
||||
payload, "manifest_id", "rhbacktestevidencev2:sha256:"
|
||||
)
|
||||
instance = object.__new__(RetrospectiveBacktestEvidenceManifest)
|
||||
values = {
|
||||
**payload,
|
||||
"qualification": qualification,
|
||||
"evidence": evidence,
|
||||
"backtest_run_ref": run,
|
||||
"_payload": _freeze_json(payload),
|
||||
"_artifact": artifact,
|
||||
}
|
||||
del values["run_reference"]
|
||||
for name, value in values.items():
|
||||
object.__setattr__(instance, name, value)
|
||||
return instance
|
||||
|
||||
|
||||
def _freeze_numeric_evidence(value: Any) -> Any:
|
||||
"""Freeze the existing finite-number metric profile, not the data JSON profile."""
|
||||
if type(value) is dict:
|
||||
return MappingProxyType(
|
||||
{key: _freeze_numeric_evidence(item) for key, item in value.items()}
|
||||
)
|
||||
if type(value) is list:
|
||||
return tuple(_freeze_numeric_evidence(item) for item in value)
|
||||
return value
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePerformanceEvidence:
|
||||
"""New upstream/time identity; unchanged finite-number metric/methodology v1."""
|
||||
|
||||
methodology: PerformanceMethodology
|
||||
metrics: tuple[PerformanceMetric, ...]
|
||||
_payload: Mapping[str, Any] = field(repr=False)
|
||||
|
||||
@property
|
||||
def performance_evidence_id(self) -> str:
|
||||
return cast(str, self._payload["performance_evidence_id"])
|
||||
|
||||
@property
|
||||
def document_sha256(self) -> str:
|
||||
return cast(str, self._payload["document_sha256"])
|
||||
|
||||
@property
|
||||
def run_id(self) -> str:
|
||||
return cast(str, self._payload["run_id"])
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def canonical_bytes(self) -> bytes:
|
||||
return _performance_canonical_bytes(self.to_dict())
|
||||
|
||||
def to_json(self) -> str:
|
||||
return self.canonical_bytes().decode("utf-8")
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
artifact: ResearchRunArtifact,
|
||||
run_ref: RetrospectiveBacktestRunRef,
|
||||
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
rebuilt = build_retrospective_performance_evidence(artifact, run_ref, evidence_manifest)
|
||||
_performance_compare(value, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
_check(
|
||||
type(value) in {str, bytes},
|
||||
"$",
|
||||
"canonical JSON text/bytes required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
raw = value.encode("utf-8") if isinstance(value, str) else value
|
||||
try:
|
||||
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
|
||||
except (UnicodeDecodeError, json.JSONDecodeError) as error:
|
||||
raise FactorContractError(
|
||||
ContractErrorCode.INVALID_FORMAT, "$", "invalid performance evidence JSON"
|
||||
) from error
|
||||
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
|
||||
_check(
|
||||
_performance_canonical_bytes(document) == raw,
|
||||
"$",
|
||||
"canonical finite-number JSON required",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
return cls.from_dict(document, **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_performance_evidence(
|
||||
artifact: ResearchRunArtifact,
|
||||
run_ref: RetrospectiveBacktestRunRef,
|
||||
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
) -> RetrospectivePerformanceEvidence:
|
||||
"""Bind current tables and existing methodology; no performance recalculation."""
|
||||
run = _validated_run(run_ref)
|
||||
_check(
|
||||
type(evidence_manifest) is RetrospectiveBacktestEvidenceManifest,
|
||||
"$.evidence_manifest",
|
||||
"explicit v2 manifest required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
evidence_manifest.to_dict(), artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
frames = _validated_frames(artifact, run)
|
||||
performance = frames["performance"]
|
||||
_check(
|
||||
len(performance) == 1
|
||||
and tuple(str(column) for column in performance.columns) == _PERFORMANCE_SOURCE_COLUMNS,
|
||||
"$.artifact.tables.performance",
|
||||
"one row in the unchanged closed performance schema required",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
performance_row = performance.iloc[0]
|
||||
run_row = _run_row(frames)
|
||||
frequency = _performance_text(run_row["frequency"], "$.artifact.tables.run.frequency")
|
||||
_check(
|
||||
frequency == "1d",
|
||||
"$.artifact.tables.run.frequency",
|
||||
"only existing daily methodology is supported",
|
||||
)
|
||||
calendar = _performance_text(run_row["calendar"], "$.artifact.tables.run.calendar")
|
||||
timezone = _performance_text(run_row["timezone"], "$.artifact.tables.run.timezone")
|
||||
start_date = _performance_date(run_row["start_date"], "$.artifact.tables.run.start_date")
|
||||
end_date = _performance_date(run_row["end_date"], "$.artifact.tables.run.end_date")
|
||||
nav = frames["nav"]
|
||||
_check(
|
||||
not nav.empty,
|
||||
"$.artifact.tables.nav",
|
||||
"NAV observation window required",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
_performance_date(nav.iloc[0]["trade_date"], "$.artifact.tables.nav.start") == start_date
|
||||
and _performance_date(nav.iloc[-1]["trade_date"], "$.artifact.tables.nav.end") == end_date,
|
||||
"$.artifact.tables.nav",
|
||||
"observation window differs from artifact dates",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
benchmark_digest, active_std, benchmark_variance, alpha_domain_unestimable = _benchmark_context(
|
||||
frames, run_row, performance_row
|
||||
)
|
||||
metrics = (
|
||||
*_absolute_performance_metrics(performance_row),
|
||||
*_relative_performance_metrics(
|
||||
performance_row,
|
||||
benchmark_present=benchmark_digest is not None,
|
||||
active_std=active_std,
|
||||
benchmark_variance=benchmark_variance,
|
||||
alpha_domain_unestimable=alpha_domain_unestimable,
|
||||
),
|
||||
*_count_performance_metrics(performance_row),
|
||||
)
|
||||
normalized_row: dict[str, object] = {metric.source_column: metric.value for metric in metrics}
|
||||
normalized_row["run_id"] = run.run_id
|
||||
row_digest = _performance_digest(
|
||||
{"columns": list(_PERFORMANCE_SOURCE_COLUMNS), "row": normalized_row}
|
||||
)
|
||||
alignment = cast(str, run_row["benchmark_alignment_policy"])
|
||||
methodology = _performance_methodology(
|
||||
frequency=frequency, alignment=alignment, code_revision=run.code_revision
|
||||
)
|
||||
performance_table = next(
|
||||
table
|
||||
for entry in manifest.evidence
|
||||
for table in entry.tables
|
||||
if table.logical_name == "performance"
|
||||
)
|
||||
run_document = run.to_dict()
|
||||
payload: dict[str, Any] = {
|
||||
"schema_version": "researchhub.performance-evidence.v2",
|
||||
"authority": "quant_engine",
|
||||
"scope": "offline_retrospective_research_only",
|
||||
"run_id": run.run_id,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
"backtest_run_ref_id": run.run_id,
|
||||
"backtest_run_ref_document_sha256": _digest_bytes(canonical_json_bytes(run_document)),
|
||||
"backtest_evidence_manifest_id": manifest.manifest_id,
|
||||
"backtest_evidence_manifest_document_sha256": _digest_bytes(
|
||||
canonical_json_bytes(manifest.to_dict())
|
||||
),
|
||||
"backtest_evidence_manifest_evidence_digest": manifest.evidence_digest,
|
||||
"backtest_evidence_qualification": manifest.qualification.value,
|
||||
"research_artifact_schema_version": artifact.schema_version,
|
||||
"research_artifact_content_digest": "sha256:" + artifact.content_sha256,
|
||||
"artifact_available_at": manifest.artifact_available_at,
|
||||
"computed_at": run.computed_at,
|
||||
"performance_table_logical_name": performance_table.logical_name,
|
||||
"performance_table_row_count": performance_table.row_count,
|
||||
"performance_table_schema_digest": performance_table.schema_digest,
|
||||
"performance_table_content_digest": performance_table.content_digest,
|
||||
"performance_row_digest": row_digest,
|
||||
"benchmark_series_digest": benchmark_digest,
|
||||
"methodology_id": PERFORMANCE_METHODOLOGY_ID,
|
||||
"metric_schema_id": PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
**{
|
||||
key: run_document[key]
|
||||
for key in (
|
||||
"dataset_snapshot_id",
|
||||
"dataset_content_digest",
|
||||
"dataset_manifest_digest",
|
||||
"foundation_id",
|
||||
"foundation_digest",
|
||||
"factor_set_id",
|
||||
"factor_set_digest",
|
||||
"factor_output_content_digest",
|
||||
"strategy_id",
|
||||
"strategy_version",
|
||||
"strategy_digest",
|
||||
"execution_model_version",
|
||||
"execution_model_digest",
|
||||
"cost_model_version",
|
||||
"cost_model_digest",
|
||||
"code_revision",
|
||||
"environment_lock_digest",
|
||||
"configuration_digest",
|
||||
)
|
||||
},
|
||||
"frequency": frequency,
|
||||
"calendar": calendar,
|
||||
"timezone": timezone,
|
||||
"benchmark_id": run_row["benchmark_id"],
|
||||
"benchmark_alignment_policy": alignment,
|
||||
"start_date": start_date,
|
||||
"end_date": end_date,
|
||||
"methodology": methodology.to_dict(),
|
||||
"metrics": [metric.to_dict() for metric in metrics],
|
||||
}
|
||||
payload["performance_evidence_id"] = "rhperformancev2:" + _performance_digest(payload)
|
||||
payload["document_sha256"] = _performance_digest(payload)
|
||||
instance = object.__new__(RetrospectivePerformanceEvidence)
|
||||
object.__setattr__(instance, "_payload", _freeze_numeric_evidence(payload))
|
||||
object.__setattr__(instance, "methodology", methodology)
|
||||
object.__setattr__(instance, "metrics", metrics)
|
||||
return instance
|
||||
@@ -1,422 +0,0 @@
|
||||
"""Explicit retrospective v2 run identities and offline artifact evidence."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Self, cast
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
PayloadValidation,
|
||||
_assert_canonical_profile,
|
||||
_content_address,
|
||||
_digest,
|
||||
_digest_bytes,
|
||||
_freeze_json,
|
||||
_git_revision,
|
||||
_logical_id,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_safe_integer,
|
||||
_semver,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
_IDS,
|
||||
_check,
|
||||
_public,
|
||||
_shape,
|
||||
_strings,
|
||||
)
|
||||
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef, _context
|
||||
|
||||
_CONFIG_FIELDS = (
|
||||
"universe_digest",
|
||||
"strategy_id",
|
||||
"strategy_version",
|
||||
"strategy_digest",
|
||||
"execution_model_version",
|
||||
"execution_model_digest",
|
||||
"cost_model_version",
|
||||
"cost_model_digest",
|
||||
"random_seed",
|
||||
"code_revision",
|
||||
"environment_lock_digest",
|
||||
"configuration_digest",
|
||||
)
|
||||
_RUN_FIELDS = (
|
||||
"contract_name schema_version run_id dataset_snapshot_id dataset_content_digest dataset_manifest_digest "
|
||||
"foundation_id foundation_digest factor_set_id factor_set_digest factor_output_content_digest "
|
||||
"observation_cutoff evidence_scope usage historical_availability decision_eligible execution_validation "
|
||||
"universe_digest trading_calendar_revision_ids trading_calendar_digest corporate_action_revision_ids corporate_action_digest "
|
||||
"strategy_id strategy_version strategy_digest execution_model_version execution_model_digest cost_model_version cost_model_digest "
|
||||
"random_seed code_revision environment_lock_digest configuration_digest evaluation_at computed_at replay_spec_digest "
|
||||
"replay_parent_run_id replay_reason replay_attempt replay_ancestor_run_ids"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveBacktestRunRef:
|
||||
"""New-major deterministic-input identity with separate actual attempt times."""
|
||||
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
run_id: str
|
||||
dataset_snapshot_id: str
|
||||
dataset_content_digest: str
|
||||
dataset_manifest_digest: str
|
||||
foundation_id: str
|
||||
foundation_digest: str
|
||||
factor_set_id: str
|
||||
factor_set_digest: str
|
||||
factor_output_content_digest: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
universe_digest: str
|
||||
trading_calendar_revision_ids: tuple[str, ...]
|
||||
trading_calendar_digest: str
|
||||
corporate_action_revision_ids: tuple[str, ...]
|
||||
corporate_action_digest: str
|
||||
strategy_id: str
|
||||
strategy_version: str
|
||||
strategy_digest: str
|
||||
execution_model_version: str
|
||||
execution_model_digest: str
|
||||
cost_model_version: str
|
||||
cost_model_digest: str
|
||||
random_seed: int
|
||||
code_revision: str
|
||||
environment_lock_digest: str
|
||||
configuration_digest: str
|
||||
evaluation_at: str
|
||||
computed_at: str
|
||||
replay_spec_digest: str
|
||||
replay_parent_run_id: str | None
|
||||
replay_reason: str | None
|
||||
replay_attempt: int
|
||||
replay_ancestor_run_ids: tuple[str, ...]
|
||||
input_payload_validation: PayloadValidation = field(compare=False)
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_factor_set: RetrospectiveFactorSetRef = field(repr=False, compare=False)
|
||||
_parent: RetrospectiveBacktestRunRef | None = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
factor_set: RetrospectiveFactorSetRef,
|
||||
universe_digest: str,
|
||||
trading_calendar_revision_ids: Sequence[str],
|
||||
corporate_action_revision_ids: Sequence[str],
|
||||
strategy_id: str,
|
||||
strategy_version: str,
|
||||
strategy_digest: str,
|
||||
execution_model_version: str,
|
||||
execution_model_digest: str,
|
||||
cost_model_version: str,
|
||||
cost_model_digest: str,
|
||||
random_seed: int,
|
||||
code_revision: str,
|
||||
environment_lock_digest: str,
|
||||
configuration_digest: str,
|
||||
evaluation_at: str,
|
||||
computed_at: str,
|
||||
parent: RetrospectiveBacktestRunRef | None = None,
|
||||
replay_reason: str | None = None,
|
||||
replay_attempt: int = 0,
|
||||
) -> Self:
|
||||
_check(
|
||||
type(factor_set) is RetrospectiveFactorSetRef,
|
||||
"$.factor_set",
|
||||
"explicit v2 factor result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
factor_set.require_payloads_revalidated()
|
||||
return cls._build(
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
trading_calendar_revision_ids=trading_calendar_revision_ids,
|
||||
corporate_action_revision_ids=corporate_action_revision_ids,
|
||||
configuration={
|
||||
"universe_digest": universe_digest,
|
||||
"strategy_id": strategy_id,
|
||||
"strategy_version": strategy_version,
|
||||
"strategy_digest": strategy_digest,
|
||||
"execution_model_version": execution_model_version,
|
||||
"execution_model_digest": execution_model_digest,
|
||||
"cost_model_version": cost_model_version,
|
||||
"cost_model_digest": cost_model_digest,
|
||||
"random_seed": random_seed,
|
||||
"code_revision": code_revision,
|
||||
"environment_lock_digest": environment_lock_digest,
|
||||
"configuration_digest": configuration_digest,
|
||||
},
|
||||
evaluation_at=evaluation_at,
|
||||
computed_at=computed_at,
|
||||
parent=parent,
|
||||
replay_reason=replay_reason,
|
||||
replay_attempt=replay_attempt,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def _build(
|
||||
cls,
|
||||
*,
|
||||
dataset_snapshot: Any,
|
||||
foundation: Any,
|
||||
factor_set: Any,
|
||||
trading_calendar_revision_ids: Any,
|
||||
corporate_action_revision_ids: Any,
|
||||
configuration: dict[str, Any],
|
||||
evaluation_at: Any,
|
||||
computed_at: Any,
|
||||
parent: RetrospectiveBacktestRunRef | None,
|
||||
replay_reason: Any,
|
||||
replay_attempt: Any,
|
||||
) -> Self:
|
||||
_check(
|
||||
type(factor_set) is RetrospectiveFactorSetRef,
|
||||
"$.factor_set",
|
||||
"explicit v2 factor result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
definitions, snapshot, foundation = _context(
|
||||
factor_set._definitions, dataset_snapshot, foundation
|
||||
)
|
||||
# Reconstruct the serialized factor boundary against the exact supplied inputs.
|
||||
checked_factor = RetrospectiveFactorSetRef.from_dict(
|
||||
factor_set.to_dict(),
|
||||
definitions=definitions,
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
parent=factor_set._parent,
|
||||
)
|
||||
closures: dict[str, tuple[str, ...]] = {}
|
||||
for field_name, supplied, kind in (
|
||||
("trading_calendar_revision_ids", trading_calendar_revision_ids, "calendar_revision"),
|
||||
("corporate_action_revision_ids", corporate_action_revision_ids, "action_revision"),
|
||||
):
|
||||
_check(
|
||||
type(supplied) in {tuple, list},
|
||||
f"$.{field_name}",
|
||||
"list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
supplied_ids = tuple(
|
||||
sorted(
|
||||
_strings(
|
||||
list(supplied),
|
||||
f"$.{field_name}",
|
||||
_IDS[kind],
|
||||
1 if kind == "calendar_revision" else 0,
|
||||
)
|
||||
)
|
||||
)
|
||||
expected_ids = tuple(
|
||||
sorted(
|
||||
{
|
||||
identity
|
||||
for view_id in checked_factor.selected_view_ref_ids
|
||||
for identity in getattr(foundation.views[view_id], field_name)
|
||||
}
|
||||
)
|
||||
)
|
||||
_check(
|
||||
supplied_ids == expected_ids,
|
||||
f"$.{field_name}",
|
||||
"exact selected observation ancestry required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
closures[field_name] = supplied_ids
|
||||
_shape(configuration, "$.configuration", " ".join(_CONFIG_FIELDS))
|
||||
for name, value in configuration.items():
|
||||
if name.endswith("_digest"):
|
||||
_digest(value, f"$.{name}")
|
||||
elif name.endswith("_version"):
|
||||
_semver(value, f"$.{name}")
|
||||
elif name == "random_seed":
|
||||
_safe_integer(value, f"$.{name}", minimum=0)
|
||||
elif name == "code_revision":
|
||||
_git_revision(value, f"$.{name}")
|
||||
else:
|
||||
_logical_id(value, f"$.{name}")
|
||||
_public(configuration, "$.configuration")
|
||||
evaluation = _parse_utc(evaluation_at, "$.evaluation_at")
|
||||
computed = _parse_utc(computed_at, "$.computed_at")
|
||||
_check(
|
||||
_parse_utc(checked_factor.artifact_available_at, "$.factor_set.artifact_available_at")
|
||||
<= evaluation
|
||||
<= computed,
|
||||
"$.computed_at",
|
||||
"factor availability <= actual evaluation <= computation required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
content = snapshot.to_dict()["descriptor"]["content"]
|
||||
spec: dict[str, Any] = {
|
||||
"dataset_snapshot_id": snapshot.snapshot_id,
|
||||
"dataset_content_digest": content["content_digest"],
|
||||
"dataset_manifest_digest": content["manifest_digest"],
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"foundation_digest": foundation.foundation_id.removeprefix("rhdfv2:"),
|
||||
"factor_set_id": checked_factor.factor_set_id,
|
||||
"factor_set_digest": checked_factor.factor_set_id.removeprefix("rhfactorsetv2:"),
|
||||
"factor_output_content_digest": checked_factor.output_content_digest,
|
||||
"observation_cutoff": foundation.observation_cutoff,
|
||||
"evidence_scope": checked_factor.evidence_scope,
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
**configuration,
|
||||
}
|
||||
for field_name, identities in closures.items():
|
||||
spec[field_name] = list(identities)
|
||||
digest_field = (
|
||||
"trading_calendar_digest"
|
||||
if field_name == "trading_calendar_revision_ids"
|
||||
else "corporate_action_digest"
|
||||
)
|
||||
spec[digest_field] = _digest_bytes(canonical_json_bytes(list(identities)))
|
||||
# v2 replay specification excludes BOTH actual attempt times. They remain in
|
||||
# run_id, so a replay never backdates evaluation to manufacture equality.
|
||||
replay_spec_digest = _digest_bytes(canonical_json_bytes(spec))
|
||||
replay_count = _safe_integer(replay_attempt, "$.replay_attempt", minimum=0)
|
||||
if parent is None:
|
||||
_check(
|
||||
replay_reason is None and replay_count == 0,
|
||||
"$.replay_attempt",
|
||||
"root must use zero attempt and no reason",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
parent_id = None
|
||||
ancestors: tuple[str, ...] = ()
|
||||
else:
|
||||
_check(
|
||||
type(parent) is RetrospectiveBacktestRunRef,
|
||||
"$.parent",
|
||||
"exact v2 run parent required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_logical_id(replay_reason, "$.replay_reason")
|
||||
_check(
|
||||
replay_count == parent.replay_attempt + 1
|
||||
and replay_spec_digest == parent.replay_spec_digest,
|
||||
"$.replay_spec_digest",
|
||||
"replay requires unchanged inputs and the next attempt",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(parent.computed_at, "$.parent.computed_at") < evaluation <= computed,
|
||||
"$.evaluation_at",
|
||||
"new actual attempt must follow parent computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
parent_id = parent.run_id
|
||||
ancestors = (*parent.replay_ancestor_run_ids, parent_id)
|
||||
_check(
|
||||
len(ancestors) == len(set(ancestors)),
|
||||
"$.replay_ancestor_run_ids",
|
||||
"replay cycle",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.backtest-run-ref",
|
||||
"schema_version": "2.0.0",
|
||||
**spec,
|
||||
"evaluation_at": evaluation_at,
|
||||
"computed_at": computed_at,
|
||||
"replay_spec_digest": replay_spec_digest,
|
||||
"replay_parent_run_id": parent_id,
|
||||
"replay_reason": replay_reason,
|
||||
"replay_attempt": replay_count,
|
||||
"replay_ancestor_run_ids": list(ancestors),
|
||||
}
|
||||
payload["run_id"] = _content_address(payload, "run_id", "rhbacktestrunv2:sha256:")
|
||||
_check(
|
||||
payload["run_id"] not in ancestors,
|
||||
"$.run_id",
|
||||
"self-parent cycle",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
verified = (
|
||||
factor_set.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
and factor_set.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
)
|
||||
instance = object.__new__(cls)
|
||||
for name, value in {
|
||||
**payload,
|
||||
**closures,
|
||||
"replay_ancestor_run_ids": ancestors,
|
||||
"input_payload_validation": PayloadValidation.PAYLOAD_REVALIDATED
|
||||
if verified
|
||||
else PayloadValidation.REFERENCE_ONLY,
|
||||
"_payload": _freeze_json(payload),
|
||||
"_factor_set": factor_set,
|
||||
"_parent": parent,
|
||||
}.items():
|
||||
object.__setattr__(instance, name, value)
|
||||
return instance
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
def require_inputs_revalidated(self) -> None:
|
||||
_check(
|
||||
self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
|
||||
"$.input_payload_validation",
|
||||
"reference-only factors cannot admit a new computation",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
factor_set: RetrospectiveFactorSetRef,
|
||||
parent: RetrospectiveBacktestRunRef | None = None,
|
||||
) -> Self:
|
||||
_assert_canonical_profile(value)
|
||||
_public(value)
|
||||
row = _shape(value, "$", _RUN_FIELDS)
|
||||
rebuilt = cls._build(
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
trading_calendar_revision_ids=row["trading_calendar_revision_ids"],
|
||||
corporate_action_revision_ids=row["corporate_action_revision_ids"],
|
||||
configuration={key: row[key] for key in _CONFIG_FIELDS},
|
||||
evaluation_at=row["evaluation_at"],
|
||||
computed_at=row["computed_at"],
|
||||
parent=parent,
|
||||
replay_reason=row["replay_reason"],
|
||||
replay_attempt=row["replay_attempt"],
|
||||
)
|
||||
_check(
|
||||
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
|
||||
"$",
|
||||
"serialized run differs from exact v2 input/configuration/lineage closure",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
return rebuilt
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,721 +0,0 @@
|
||||
"""Observation-aware factor results; no historical, governance or execution grant."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Self, cast
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
ContractErrorCode,
|
||||
FactorDefinition,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
_DEFINITION_ID,
|
||||
_FIELD_NAME,
|
||||
_array,
|
||||
_assert_canonical_profile,
|
||||
_canonical_evidence_bytes,
|
||||
_content_address,
|
||||
_digest,
|
||||
_digest_bytes,
|
||||
_freeze_json,
|
||||
_git_revision,
|
||||
_logical_id,
|
||||
_parse_json_object,
|
||||
_parse_utc,
|
||||
_string,
|
||||
_thaw_json,
|
||||
canonical_json,
|
||||
canonical_json_bytes,
|
||||
validate_factor_catalog,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
_IDS,
|
||||
_check,
|
||||
_choice,
|
||||
_public,
|
||||
_restrictions,
|
||||
_shape,
|
||||
_strings,
|
||||
)
|
||||
|
||||
_FACTOR_SET_ID = re.compile(r"^rhfactorsetv2:sha256:[0-9a-f]{64}$")
|
||||
|
||||
|
||||
def _instant_text(value: Any, path: str) -> str:
|
||||
_parse_utc(value, path)
|
||||
return cast(str, value)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveInputBinding:
|
||||
definition_id: str
|
||||
input_name: str
|
||||
view_ref_id: str
|
||||
schema_digest: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.definition_id, "$.input_bindings[].definition_id", _DEFINITION_ID)
|
||||
_string(self.input_name, "$.input_bindings[].input_name", _FIELD_NAME)
|
||||
_string(self.view_ref_id, "$.input_bindings[].view_ref_id", _IDS["view_ref"])
|
||||
_digest(self.schema_digest, "$.input_bindings[].schema_digest")
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"definition_id": self.definition_id,
|
||||
"input_name": self.input_name,
|
||||
"view_ref_id": self.view_ref_id,
|
||||
"schema_digest": self.schema_digest,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any, path: str = "$.input_bindings[]") -> Self:
|
||||
row = _shape(value, path, "definition_id input_name view_ref_id schema_digest")
|
||||
return cls(**row)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveViewAvailability:
|
||||
view_ref_id: str
|
||||
available_at: str
|
||||
evidence_digest: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.view_ref_id, "$.view_availability[].view_ref_id", _IDS["view_ref"])
|
||||
_instant_text(self.available_at, "$.view_availability[].available_at")
|
||||
_digest(self.evidence_digest, "$.view_availability[].evidence_digest")
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"view_ref_id": self.view_ref_id,
|
||||
"available_at": self.available_at,
|
||||
"evidence_digest": self.evidence_digest,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any, path: str = "$.view_availability[]") -> Self:
|
||||
return cls(**_shape(value, path, "view_ref_id available_at evidence_digest"))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class RetrospectiveCausation:
|
||||
kind: str
|
||||
id: str
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
kind = _choice(self.kind, "$.causation.kind", {"foundation", "factor_set"})
|
||||
_string(
|
||||
self.id,
|
||||
"$.causation.id",
|
||||
_IDS["foundation"] if kind == "foundation" else _FACTOR_SET_ID,
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {"kind": self.kind, "id": self.id}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any) -> Self:
|
||||
return cls(**_shape(value, "$.causation", "kind id"))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ResolvedRetrospectiveView:
|
||||
"""In-memory logical bytes; no locator, source authentication or transformation claim."""
|
||||
|
||||
view_ref_id: str
|
||||
schema_bytes: bytes
|
||||
content_bytes: bytes
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
_string(self.view_ref_id, "$.resolved_views[].view_ref_id", _IDS["view_ref"])
|
||||
for key, value in (
|
||||
("schema_bytes", self.schema_bytes),
|
||||
("content_bytes", self.content_bytes),
|
||||
):
|
||||
_canonical_evidence_bytes(value, f"$.resolved_views[].{key}")
|
||||
_public(json.loads(value), f"$.resolved_views[].{key}")
|
||||
|
||||
|
||||
def _typed(values: Any, expected: type[Any], path: str) -> tuple[Any, ...]:
|
||||
_check(
|
||||
type(values) in {tuple, list},
|
||||
path,
|
||||
"typed list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
all(type(value) is expected for value in values),
|
||||
path,
|
||||
f"{expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
return tuple(values)
|
||||
|
||||
|
||||
def _context(
|
||||
definitions: Sequence[FactorDefinition],
|
||||
snapshot: Any,
|
||||
foundation: Any,
|
||||
) -> tuple[
|
||||
tuple[FactorDefinition, ...], RetrospectiveSnapshotEnvelope, RetrospectiveFoundationEnvelope
|
||||
]:
|
||||
_check(
|
||||
type(snapshot) is RetrospectiveSnapshotEnvelope,
|
||||
"$.dataset_snapshot",
|
||||
"explicit v2 snapshot required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
type(foundation) is RetrospectiveFoundationEnvelope,
|
||||
"$.foundation",
|
||||
"explicit v2 foundation required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot.to_dict())
|
||||
snapshot.require_qualified()
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation.to_dict(), snapshot=snapshot)
|
||||
supplied = _typed(definitions, FactorDefinition, "$.definitions")
|
||||
# Definitions stay v1, but are parsed again so mutable/caller summaries are not authority.
|
||||
normalized = validate_factor_catalog(
|
||||
tuple(FactorDefinition.from_dict(item.to_dict()) for item in supplied)
|
||||
)
|
||||
return normalized, snapshot, foundation
|
||||
|
||||
|
||||
def _upstream(
|
||||
snapshot: RetrospectiveSnapshotEnvelope, foundation: RetrospectiveFoundationEnvelope
|
||||
) -> dict[str, Any]:
|
||||
descriptor = snapshot.to_dict()["descriptor"]
|
||||
return {
|
||||
"dataset_snapshot_id": snapshot.snapshot_id,
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"evidence_scope": snapshot.evidence_scope,
|
||||
"content_digest": descriptor["content"]["content_digest"],
|
||||
"manifest_digest": descriptor["content"]["manifest_digest"],
|
||||
"observation_manifest_digest": _digest_bytes(
|
||||
canonical_json_bytes(descriptor["observation_manifest"])
|
||||
),
|
||||
"time_semantics": descriptor["time_semantics"],
|
||||
"quality": descriptor["quality"],
|
||||
"qualification": descriptor["qualification"],
|
||||
"foundation_readiness": foundation.to_dict()["readiness"],
|
||||
}
|
||||
|
||||
|
||||
def _input_payloads(
|
||||
snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
selected: tuple[str, ...],
|
||||
dataset_chunks: Any,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView] | None,
|
||||
) -> PayloadValidation:
|
||||
_check(
|
||||
(dataset_chunks is None) == (resolved_views is None),
|
||||
"$.input_payloads",
|
||||
"snapshot chunks and resolved views must be supplied together",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
if dataset_chunks is None:
|
||||
return PayloadValidation.REFERENCE_ONLY
|
||||
snapshot.verify_materialized_records(dataset_chunks)
|
||||
views = _typed(resolved_views, ResolvedRetrospectiveView, "$.resolved_views")
|
||||
view_ids = [view.view_ref_id for view in views]
|
||||
_check(
|
||||
len(view_ids) == len(selected) and set(view_ids) == set(selected),
|
||||
"$.resolved_views",
|
||||
"resolved view closure mismatch",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for item in views:
|
||||
declared = foundation.views[item.view_ref_id]
|
||||
# Recheck canonical bytes even for caller-constructed typed payloads.
|
||||
schema = _canonical_evidence_bytes(item.schema_bytes, "$.resolved_views[].schema_bytes")
|
||||
content = _canonical_evidence_bytes(item.content_bytes, "$.resolved_views[].content_bytes")
|
||||
_public(json.loads(schema))
|
||||
_public(json.loads(content))
|
||||
_check(
|
||||
_digest_bytes(schema) == declared.schema_digest
|
||||
and _digest_bytes(content) == declared.content_digest,
|
||||
"$.resolved_views",
|
||||
"view bytes do not match Foundation",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
return PayloadValidation.PAYLOAD_REVALIDATED
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveFactorSetRef:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
factor_set_id: str
|
||||
definition_ids: tuple[str, ...]
|
||||
dataset_snapshot_id: str
|
||||
foundation_id: str
|
||||
observation_cutoff: str
|
||||
selected_view_ref_ids: tuple[str, ...]
|
||||
input_bindings: tuple[RetrospectiveInputBinding, ...]
|
||||
view_availability: tuple[RetrospectiveViewAvailability, ...]
|
||||
upstream_evidence: Mapping[str, Any]
|
||||
output_quality: OutputQuality
|
||||
output_coverage: OutputCoverage
|
||||
output_schema_digest: str
|
||||
output_content_digest: str
|
||||
output_artifact_ref: OutputArtifactRef
|
||||
availability_mode: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
evaluation_at: str
|
||||
computed_at: str
|
||||
artifact_available_at: str
|
||||
producer: ProducerIdentity
|
||||
code_revision: str
|
||||
actor: ActorIdentity
|
||||
correlation_id: str
|
||||
causation: RetrospectiveCausation
|
||||
evidence_scope: str
|
||||
decision_eligible: bool
|
||||
payload_validation: PayloadValidation = field(compare=False)
|
||||
input_payload_validation: PayloadValidation = field(compare=False)
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_definitions: tuple[FactorDefinition, ...] = field(repr=False, compare=False)
|
||||
_dataset_snapshot: RetrospectiveSnapshotEnvelope = field(repr=False, compare=False)
|
||||
_foundation: RetrospectiveFoundationEnvelope = field(repr=False, compare=False)
|
||||
_parent: RetrospectiveFactorSetRef | None = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
definitions: Sequence[FactorDefinition],
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
selected_view_ref_ids: Sequence[str],
|
||||
input_bindings: Sequence[RetrospectiveInputBinding],
|
||||
view_availability: Sequence[RetrospectiveViewAvailability],
|
||||
dataset_chunks: Any,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView],
|
||||
output_quality: OutputQuality,
|
||||
output_coverage: OutputCoverage,
|
||||
output_schema_bytes: bytes,
|
||||
output_content_bytes: bytes,
|
||||
output_artifact_ref: OutputArtifactRef,
|
||||
evaluation_at: str,
|
||||
computed_at: str,
|
||||
artifact_available_at: str,
|
||||
producer: ProducerIdentity,
|
||||
code_revision: str,
|
||||
actor: ActorIdentity,
|
||||
correlation_id: str,
|
||||
causation: RetrospectiveCausation,
|
||||
evidence_scope: str,
|
||||
decision_eligible: bool,
|
||||
parent: RetrospectiveFactorSetRef | None = None,
|
||||
) -> Self:
|
||||
definitions, dataset_snapshot, foundation = _context(
|
||||
definitions, dataset_snapshot, foundation
|
||||
)
|
||||
_check(
|
||||
type(selected_view_ref_ids) in {list, tuple},
|
||||
"$.selected_view_ref_ids",
|
||||
"list/tuple required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
selected = sorted(
|
||||
_strings(list(selected_view_ref_ids), "$.selected_view_ref_ids", _IDS["view_ref"], 1)
|
||||
)
|
||||
bindings = sorted(
|
||||
_typed(input_bindings, RetrospectiveInputBinding, "$.input_bindings"),
|
||||
key=lambda item: (item.definition_id, item.input_name),
|
||||
)
|
||||
availability = sorted(
|
||||
_typed(view_availability, RetrospectiveViewAvailability, "$.view_availability"),
|
||||
key=lambda item: item.view_ref_id,
|
||||
)
|
||||
for value, expected, path in (
|
||||
(output_quality, OutputQuality, "$.output_quality"),
|
||||
(output_coverage, OutputCoverage, "$.output_coverage"),
|
||||
(output_artifact_ref, OutputArtifactRef, "$.output_artifact_ref"),
|
||||
(producer, ProducerIdentity, "$.producer"),
|
||||
(actor, ActorIdentity, "$.actor"),
|
||||
(causation, RetrospectiveCausation, "$.causation"),
|
||||
):
|
||||
_check(
|
||||
type(value) is expected,
|
||||
path,
|
||||
f"{expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
schema_digest = _digest_bytes(
|
||||
_canonical_evidence_bytes(output_schema_bytes, "$.output_schema_bytes")
|
||||
)
|
||||
content_digest = _digest_bytes(
|
||||
_canonical_evidence_bytes(output_content_bytes, "$.output_content_bytes")
|
||||
)
|
||||
document = {
|
||||
"contract_name": "researchhub.factor-set-ref",
|
||||
"schema_version": "2.0.0",
|
||||
"definition_ids": [definition.definition_id for definition in definitions],
|
||||
"dataset_snapshot_id": dataset_snapshot.snapshot_id,
|
||||
"foundation_id": foundation.foundation_id,
|
||||
"observation_cutoff": foundation.observation_cutoff,
|
||||
"selected_view_ref_ids": selected,
|
||||
"input_bindings": [item.to_dict() for item in bindings],
|
||||
"view_availability": [item.to_dict() for item in availability],
|
||||
"upstream_evidence": _upstream(dataset_snapshot, foundation),
|
||||
"output_quality": output_quality.to_dict(),
|
||||
"output_coverage": output_coverage.to_dict(),
|
||||
"output_schema_digest": schema_digest,
|
||||
"output_content_digest": content_digest,
|
||||
"output_artifact_ref": output_artifact_ref.to_dict(),
|
||||
"availability_mode": "retrospective_replay",
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"evaluation_at": evaluation_at,
|
||||
"computed_at": computed_at,
|
||||
"artifact_available_at": artifact_available_at,
|
||||
"producer": producer.to_dict(),
|
||||
"code_revision": code_revision,
|
||||
"actor": actor.to_dict(),
|
||||
"correlation_id": correlation_id,
|
||||
"causation": causation.to_dict(),
|
||||
"evidence_scope": evidence_scope,
|
||||
"decision_eligible": decision_eligible,
|
||||
}
|
||||
document["factor_set_id"] = _content_address(
|
||||
document, "factor_set_id", "rhfactorsetv2:sha256:"
|
||||
)
|
||||
result = cls.from_dict(
|
||||
document,
|
||||
definitions=definitions,
|
||||
dataset_snapshot=dataset_snapshot,
|
||||
foundation=foundation,
|
||||
parent=parent,
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
dataset_chunks=dataset_chunks,
|
||||
resolved_views=resolved_views,
|
||||
)
|
||||
result.require_payloads_revalidated()
|
||||
return result
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
definitions: Sequence[FactorDefinition],
|
||||
dataset_snapshot: RetrospectiveSnapshotEnvelope,
|
||||
foundation: RetrospectiveFoundationEnvelope,
|
||||
parent: RetrospectiveFactorSetRef | None = None,
|
||||
output_schema_bytes: bytes | None = None,
|
||||
output_content_bytes: bytes | None = None,
|
||||
dataset_chunks: Any = None,
|
||||
resolved_views: Sequence[ResolvedRetrospectiveView] | None = None,
|
||||
) -> Self:
|
||||
definitions, dataset_snapshot, foundation = _context(
|
||||
definitions, dataset_snapshot, foundation
|
||||
)
|
||||
_assert_canonical_profile(value)
|
||||
_public(value)
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version factor_set_id definition_ids dataset_snapshot_id foundation_id observation_cutoff "
|
||||
"selected_view_ref_ids input_bindings view_availability upstream_evidence output_quality output_coverage "
|
||||
"output_schema_digest output_content_digest output_artifact_ref availability_mode usage historical_availability "
|
||||
"evaluation_at computed_at artifact_available_at producer code_revision actor correlation_id causation evidence_scope decision_eligible",
|
||||
)
|
||||
_choice(row["contract_name"], "$.contract_name", {"researchhub.factor-set-ref"})
|
||||
_choice(row["schema_version"], "$.schema_version", {"2.0.0"})
|
||||
_choice(row["availability_mode"], "$.availability_mode", {"retrospective_replay"})
|
||||
_restrictions(row, "$")
|
||||
_check(
|
||||
type(row["decision_eligible"]) is bool and not row["decision_eligible"],
|
||||
"$.decision_eligible",
|
||||
"computation is never decision eligible",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
_check(
|
||||
row["dataset_snapshot_id"] == dataset_snapshot.snapshot_id
|
||||
and row["foundation_id"] == foundation.foundation_id
|
||||
and row["observation_cutoff"] == foundation.observation_cutoff,
|
||||
"$.foundation_id",
|
||||
"exact snapshot/foundation/cutoff required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
definition_ids = _strings(row["definition_ids"], "$.definition_ids", _DEFINITION_ID, 1)
|
||||
_check(
|
||||
definition_ids == tuple(item.definition_id for item in definitions),
|
||||
"$.definition_ids",
|
||||
"normalized exact definitions required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
selected = _strings(
|
||||
row["selected_view_ref_ids"], "$.selected_view_ref_ids", _IDS["view_ref"], 1
|
||||
)
|
||||
_check(
|
||||
tuple(sorted(selected)) == selected and set(selected) <= foundation.views.keys(),
|
||||
"$.selected_view_ref_ids",
|
||||
"unknown/unnormalized selected views",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
bindings = tuple(
|
||||
RetrospectiveInputBinding.from_dict(item)
|
||||
for item in _array(row["input_bindings"], "$.input_bindings", minimum=1, unique=True)
|
||||
)
|
||||
keys = [(item.definition_id, item.input_name) for item in bindings]
|
||||
expected = {
|
||||
(item.definition_id, input_spec.input_name): input_spec
|
||||
for item in definitions
|
||||
for input_spec in item.inputs
|
||||
}
|
||||
_check(
|
||||
len(keys) == len(expected) and set(keys) == expected.keys() and keys == sorted(keys),
|
||||
"$.input_bindings",
|
||||
"exact normalized factor input closure required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for binding in bindings:
|
||||
_check(
|
||||
binding.view_ref_id in selected,
|
||||
"$.input_bindings",
|
||||
"unselected view",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
binding.schema_digest
|
||||
== expected[(binding.definition_id, binding.input_name)].schema_digest
|
||||
== foundation.views[binding.view_ref_id].schema_digest,
|
||||
"$.input_bindings",
|
||||
"schema mismatch",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
{item.view_ref_id for item in bindings} == set(selected),
|
||||
"$.selected_view_ref_ids",
|
||||
"unused selected view",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
availability = tuple(
|
||||
RetrospectiveViewAvailability.from_dict(item)
|
||||
for item in _array(
|
||||
row["view_availability"], "$.view_availability", minimum=1, unique=True
|
||||
)
|
||||
)
|
||||
_check(
|
||||
tuple(item.view_ref_id for item in availability) == selected,
|
||||
"$.view_availability",
|
||||
"exact normalized selected view availability required",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
for item in availability:
|
||||
_check(
|
||||
item.available_at == foundation.views[item.view_ref_id].available_at,
|
||||
"$.view_availability",
|
||||
"availability must equal its Foundation fact",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
upstream = _upstream(dataset_snapshot, foundation)
|
||||
_check(
|
||||
canonical_json_bytes(row["upstream_evidence"]) == canonical_json_bytes(upstream),
|
||||
"$.upstream_evidence",
|
||||
"upstream evidence differs from complete input envelopes",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
row["evidence_scope"] == dataset_snapshot.evidence_scope == foundation.evidence_scope,
|
||||
"$.evidence_scope",
|
||||
"scope must equal both inputs",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
if row["evidence_scope"] == "real_data":
|
||||
_check(
|
||||
foundation.real_data_validation_status == "validated",
|
||||
"$.evidence_scope",
|
||||
"real-data Foundation validation required",
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
)
|
||||
quality = OutputQuality.from_dict(row["output_quality"])
|
||||
coverage = OutputCoverage.from_dict(row["output_coverage"])
|
||||
_check(
|
||||
quality.status == "passed" and all(item.status == "passed" for item in quality.checks),
|
||||
"$.output_quality",
|
||||
"all output checks must pass",
|
||||
)
|
||||
_check(
|
||||
coverage.status == "complete" and coverage.observed_count == coverage.expected_count,
|
||||
"$.output_coverage",
|
||||
"complete output coverage required",
|
||||
)
|
||||
artifact = OutputArtifactRef.from_dict(row["output_artifact_ref"])
|
||||
schema_digest = _digest(row["output_schema_digest"], "$.output_schema_digest")
|
||||
content_digest = _digest(row["output_content_digest"], "$.output_content_digest")
|
||||
_check(
|
||||
artifact.schema_digest == schema_digest and artifact.content_digest == content_digest,
|
||||
"$.output_artifact_ref",
|
||||
"output artifact mismatch",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
(output_schema_bytes is None) == (output_content_bytes is None),
|
||||
"$.output_artifact_ref",
|
||||
"both output payloads required together",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
validation = PayloadValidation.REFERENCE_ONLY
|
||||
if output_schema_bytes is not None and output_content_bytes is not None:
|
||||
for data, expected_digest, path in (
|
||||
(output_schema_bytes, schema_digest, "$.output_schema_bytes"),
|
||||
(output_content_bytes, content_digest, "$.output_content_bytes"),
|
||||
):
|
||||
canonical = _canonical_evidence_bytes(data, path)
|
||||
_public(json.loads(canonical), path)
|
||||
_check(
|
||||
_digest_bytes(canonical) == expected_digest,
|
||||
path,
|
||||
"output bytes mismatch",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
validation = PayloadValidation.PAYLOAD_REVALIDATED
|
||||
input_validation = _input_payloads(
|
||||
dataset_snapshot, foundation, selected, dataset_chunks, resolved_views
|
||||
)
|
||||
evaluation = _parse_utc(row["evaluation_at"], "$.evaluation_at")
|
||||
computed = _parse_utc(row["computed_at"], "$.computed_at")
|
||||
available = _parse_utc(row["artifact_available_at"], "$.artifact_available_at")
|
||||
_check(
|
||||
foundation.published_at <= evaluation <= computed <= available,
|
||||
"$.computed_at",
|
||||
"input publication <= actual evaluation <= computation <= artifact required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
for definition in definitions:
|
||||
_check(
|
||||
_parse_utc(definition.valid_from, "$.definitions[].valid_from")
|
||||
<= evaluation
|
||||
< _parse_utc(definition.valid_until, "$.definitions[].valid_until"),
|
||||
"$.definitions",
|
||||
"factor definition is not valid at actual evaluation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
producer = ProducerIdentity.from_dict(row["producer"])
|
||||
_check(
|
||||
producer.id == "quant_engine",
|
||||
"$.producer.id",
|
||||
"computation owner must be quant_engine",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_git_revision(row["code_revision"], "$.code_revision")
|
||||
actor = ActorIdentity.from_dict(row["actor"])
|
||||
correlation = _logical_id(row["correlation_id"], "$.correlation_id")
|
||||
cause = RetrospectiveCausation.from_dict(row["causation"])
|
||||
for name, parsed in (
|
||||
("output_quality", quality),
|
||||
("output_coverage", coverage),
|
||||
("output_artifact_ref", artifact),
|
||||
("producer", producer),
|
||||
("actor", actor),
|
||||
("causation", cause),
|
||||
):
|
||||
_check(
|
||||
canonical_json_bytes(row[name]) == canonical_json_bytes(parsed.to_dict()),
|
||||
f"$.{name}",
|
||||
"nested contract is not normalized",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
if cause.kind == "foundation":
|
||||
_check(
|
||||
cause.id == foundation.foundation_id and parent is None,
|
||||
"$.causation",
|
||||
"exact Foundation cause required",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
else:
|
||||
_check(
|
||||
type(parent) is RetrospectiveFactorSetRef,
|
||||
"$.causation",
|
||||
"exact v2 parent object required",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
assert parent is not None
|
||||
_check(
|
||||
cause.id == parent.factor_set_id
|
||||
and correlation == parent.correlation_id
|
||||
and row["evidence_scope"] == parent.evidence_scope,
|
||||
"$.causation",
|
||||
"parent identity/correlation/scope mismatch",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(parent.artifact_available_at, "$.parent.artifact_available_at")
|
||||
<= evaluation,
|
||||
"$.causation",
|
||||
"parent artifact postdates child evaluation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
factor_set_id = _string(row["factor_set_id"], "$.factor_set_id", _FACTOR_SET_ID)
|
||||
_check(
|
||||
factor_set_id == _content_address(row, "factor_set_id", "rhfactorsetv2:sha256:"),
|
||||
"$.factor_set_id",
|
||||
"factor result identity mismatch",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
cause.id != factor_set_id,
|
||||
"$.causation",
|
||||
"self parent is forbidden",
|
||||
ContractErrorCode.LINEAGE_VIOLATION,
|
||||
)
|
||||
instance = object.__new__(cls)
|
||||
values = {
|
||||
**row,
|
||||
"definition_ids": definition_ids,
|
||||
"selected_view_ref_ids": selected,
|
||||
"input_bindings": bindings,
|
||||
"view_availability": availability,
|
||||
"upstream_evidence": _freeze_json(upstream),
|
||||
"output_quality": quality,
|
||||
"output_coverage": coverage,
|
||||
"output_artifact_ref": artifact,
|
||||
"producer": producer,
|
||||
"actor": actor,
|
||||
"causation": cause,
|
||||
"payload_validation": validation,
|
||||
"input_payload_validation": input_validation,
|
||||
"_payload": _freeze_json(row),
|
||||
"_definitions": definitions,
|
||||
"_dataset_snapshot": dataset_snapshot,
|
||||
"_foundation": foundation,
|
||||
"_parent": parent,
|
||||
}
|
||||
for name, item in values.items():
|
||||
object.__setattr__(instance, name, item)
|
||||
return instance
|
||||
|
||||
def require_payloads_revalidated(self) -> None:
|
||||
_check(
|
||||
self.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
and self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
|
||||
"$.payload_validation",
|
||||
"reference-only data is not computation admission",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
|
||||
@@ -1,995 +0,0 @@
|
||||
"""Retrospective-only portfolio/risk evidence with separate business/actual clocks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Self, TypedDict, cast
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from quant_engine.artifact import (
|
||||
EvidenceQualification,
|
||||
_performance_compare,
|
||||
_performance_validate_tree,
|
||||
)
|
||||
from quant_engine.factor_contracts import (
|
||||
ContractErrorCode,
|
||||
FactorContractError,
|
||||
_duplicate_key_pairs,
|
||||
_parse_utc,
|
||||
_string,
|
||||
_thaw_json,
|
||||
)
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
ReceiptStatus,
|
||||
PortfolioRiskContractError,
|
||||
PortfolioRiskContractErrorCode,
|
||||
RiskAssessmentStatus,
|
||||
RiskFindingCode,
|
||||
_CLOSURE_ATOL,
|
||||
_CLOSURE_RTOL,
|
||||
_finite_number,
|
||||
_series_mapping,
|
||||
_validate_covariance_structure,
|
||||
_canonical_json,
|
||||
_constraint_metrics,
|
||||
_constraint_residuals,
|
||||
_digest,
|
||||
_document_sha256,
|
||||
_immutable_float_mapping,
|
||||
_mapping_dict,
|
||||
_payload_digest,
|
||||
_semver,
|
||||
_text,
|
||||
)
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
RetrospectiveBacktestEvidenceManifest,
|
||||
_freeze_numeric_evidence,
|
||||
_validated_run,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_data_contracts import _IDS, _check, _public, _shape
|
||||
from quant_engine.risk import CovarianceSnapshot, labeled_component_risk
|
||||
|
||||
_RUN_ID = re.compile(r"^rhbacktestrunv2:sha256:[0-9a-f]{64}$")
|
||||
|
||||
|
||||
def _json_object(value: str | bytes) -> dict[str, Any]:
|
||||
_check(
|
||||
type(value) in {str, bytes},
|
||||
"$",
|
||||
"canonical JSON text/bytes required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
try:
|
||||
raw = value.encode("utf-8") if isinstance(value, str) else value
|
||||
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
|
||||
except (json.JSONDecodeError, UnicodeError) as error:
|
||||
raise FactorContractError(
|
||||
ContractErrorCode.INVALID_FORMAT, "$", "valid UTF-8 JSON required"
|
||||
) from error
|
||||
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
|
||||
_performance_validate_tree(document, "$")
|
||||
_check(
|
||||
_canonical_json(document).encode() == raw,
|
||||
"$",
|
||||
"canonical numeric JSON required",
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
)
|
||||
return cast(dict[str, Any], document)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePortfolioTarget:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
target_id: str
|
||||
backtest_run_id: str
|
||||
dataset_snapshot_id: str
|
||||
weights: Mapping[str, float]
|
||||
effective_at: str
|
||||
created_at: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
|
||||
@classmethod
|
||||
def create(
|
||||
cls,
|
||||
*,
|
||||
backtest_run_id: str,
|
||||
dataset_snapshot_id: str,
|
||||
weights: Mapping[str, float],
|
||||
effective_at: str,
|
||||
created_at: str,
|
||||
) -> Self:
|
||||
_string(backtest_run_id, "$.backtest_run_id", _RUN_ID)
|
||||
_string(dataset_snapshot_id, "$.dataset_snapshot_id", _IDS["snapshot"])
|
||||
normalized = _immutable_float_mapping(weights, "$.weights")
|
||||
_check(bool(normalized), "$.weights", "non-empty target asset set required")
|
||||
for instrument in normalized:
|
||||
_string(instrument, "$.weights.keys", _IDS["instrument"])
|
||||
effective = _parse_utc(effective_at, "$.effective_at")
|
||||
created = _parse_utc(created_at, "$.created_at")
|
||||
_check(
|
||||
effective <= created,
|
||||
"$.effective_at",
|
||||
"historical effective time exceeds actual creation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.portfolio-target",
|
||||
"schema_version": "2.0.0",
|
||||
"backtest_run_id": backtest_run_id,
|
||||
"dataset_snapshot_id": dataset_snapshot_id,
|
||||
"weights": _mapping_dict(normalized),
|
||||
"effective_at": effective_at,
|
||||
"created_at": created_at,
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
}
|
||||
_public(payload)
|
||||
payload["target_id"] = "rhportfoliotargetv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(cls)
|
||||
for key, value in {
|
||||
**payload,
|
||||
"weights": normalized,
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
}.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, value: Any) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
"contract_name schema_version target_id backtest_run_id dataset_snapshot_id weights effective_at created_at usage historical_availability",
|
||||
)
|
||||
rebuilt = cls.create(
|
||||
**{
|
||||
key: row[key]
|
||||
for key in (
|
||||
"backtest_run_id",
|
||||
"dataset_snapshot_id",
|
||||
"weights",
|
||||
"effective_at",
|
||||
"created_at",
|
||||
)
|
||||
}
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return rebuilt
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes) -> Self:
|
||||
return cls.from_dict(_json_object(value))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _PortfolioInputs:
|
||||
run: RetrospectiveBacktestRunRef
|
||||
manifest: RetrospectiveBacktestEvidenceManifest
|
||||
target: RetrospectivePortfolioTarget
|
||||
constraints: ConstraintSetV1
|
||||
freshness: FreshnessPolicy
|
||||
weights: Mapping[str, float]
|
||||
prior: Mapping[str, float] | None
|
||||
metrics: dict[str, float | int | None]
|
||||
residuals: dict[str, float]
|
||||
input_payload: dict[str, object]
|
||||
|
||||
|
||||
def _material(
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
objective_name: str,
|
||||
objective_version: str,
|
||||
objective_digest: str,
|
||||
model_name: str,
|
||||
model_version: str,
|
||||
model_digest: str,
|
||||
expected_return_digest: str,
|
||||
covariance_digest: str,
|
||||
scenario_digest: str,
|
||||
constraints: ConstraintSetV1,
|
||||
freshness_policy: FreshnessPolicy,
|
||||
prior_weights: Mapping[str, float] | None = None,
|
||||
) -> _PortfolioInputs:
|
||||
run = _validated_run(backtest_run_ref)
|
||||
for item, expected, path in (
|
||||
(manifest, RetrospectiveBacktestEvidenceManifest, "$.manifest"),
|
||||
(target, RetrospectivePortfolioTarget, "$.target"),
|
||||
(constraints, ConstraintSetV1, "$.constraints"),
|
||||
(freshness_policy, FreshnessPolicy, "$.freshness_policy"),
|
||||
):
|
||||
_check(
|
||||
type(item) is expected,
|
||||
path,
|
||||
f"explicit {expected.__name__} required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
checked_manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
manifest.to_dict(), artifact=manifest._artifact, backtest_run_ref=run
|
||||
)
|
||||
checked_target = RetrospectivePortfolioTarget.from_dict(target.to_dict())
|
||||
_check(
|
||||
checked_manifest.qualification is EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
"$.manifest.qualification",
|
||||
"contract-qualified retrospective S3 required",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
_check(
|
||||
checked_target.backtest_run_id == run.run_id
|
||||
and checked_target.dataset_snapshot_id == run.dataset_snapshot_id,
|
||||
"$.target",
|
||||
"target and S3 identities differ",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
foundation = run._factor_set._foundation
|
||||
selected_routes = {
|
||||
identity
|
||||
for view_id in run._factor_set.selected_view_ref_ids
|
||||
for identity in foundation.views[view_id].instrument_route_revision_ids
|
||||
}
|
||||
selected_instruments = {
|
||||
row["instrument_id"]
|
||||
for row in foundation.to_dict()["instrument_routes"]
|
||||
if row["route_revision_id"] in selected_routes
|
||||
}
|
||||
_check(
|
||||
set(checked_target.weights) <= selected_instruments,
|
||||
"$.target.weights",
|
||||
"target assets must be selected logical instruments",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
constraints = ConstraintSetV1.from_dict(constraints.to_dict())
|
||||
freshness_policy = FreshnessPolicy.from_dict(freshness_policy.to_dict())
|
||||
prior = (
|
||||
None
|
||||
if prior_weights is None
|
||||
else _immutable_float_mapping(prior_weights, "$.prior_weights")
|
||||
)
|
||||
if prior is not None:
|
||||
_check(
|
||||
set(prior) <= selected_instruments,
|
||||
"$.prior_weights",
|
||||
"prior assets outside selected instruments",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
weights = checked_target.weights
|
||||
metrics = _constraint_metrics(weights, prior)
|
||||
residuals = _constraint_residuals(constraints, weights, metrics)
|
||||
payload: dict[str, object] = {
|
||||
"contract_name": "researchhub.portfolio-computation-input",
|
||||
"schema_version": "2.0.0",
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"run_ref_document_sha256": _document_sha256(run.to_json()),
|
||||
"manifest_document_sha256": _document_sha256(checked_manifest.to_json()),
|
||||
"portfolio_target": checked_target.to_dict(),
|
||||
"objective": {
|
||||
"name": _text(objective_name, "$.objective_name"),
|
||||
"version": _semver(objective_version, "$.objective_version"),
|
||||
"digest": _digest(objective_digest, "$.objective_digest"),
|
||||
},
|
||||
"model": {
|
||||
"name": _text(model_name, "$.model_name"),
|
||||
"version": _semver(model_version, "$.model_version"),
|
||||
"digest": _digest(model_digest, "$.model_digest"),
|
||||
},
|
||||
"expected_return_digest": _digest(expected_return_digest, "$.expected_return_digest"),
|
||||
"covariance_digest": _digest(covariance_digest, "$.covariance_digest"),
|
||||
"scenario_digest": _digest(scenario_digest, "$.scenario_digest"),
|
||||
"freshness_policy_digest": _payload_digest(freshness_policy.to_dict()),
|
||||
"prior_weights": None if prior is None else _mapping_dict(prior),
|
||||
}
|
||||
_public(payload)
|
||||
return _PortfolioInputs(
|
||||
run,
|
||||
checked_manifest,
|
||||
checked_target,
|
||||
constraints,
|
||||
freshness_policy,
|
||||
weights,
|
||||
prior,
|
||||
metrics,
|
||||
residuals,
|
||||
payload,
|
||||
)
|
||||
|
||||
|
||||
def _receipt_digests(inputs: _PortfolioInputs) -> dict[str, str | float]:
|
||||
return {
|
||||
"input_digest": _payload_digest(inputs.input_payload),
|
||||
"constraint_digest": _payload_digest(inputs.constraints.to_dict()),
|
||||
"output_digest": _payload_digest(
|
||||
{
|
||||
"weights": _mapping_dict(inputs.weights),
|
||||
"metrics": inputs.metrics,
|
||||
"constraint_residuals": inputs.residuals,
|
||||
}
|
||||
),
|
||||
"max_constraint_residual": max(inputs.residuals.values(), default=0.0),
|
||||
}
|
||||
|
||||
|
||||
def compute_retrospective_portfolio_receipt_digests(**kwargs: Any) -> Mapping[str, str | float]:
|
||||
"""Recompute receipt claims; the returned digests are not producer authentication."""
|
||||
return MappingProxyType(_receipt_digests(_material(**kwargs)))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectivePortfolioDecision:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
decision_id: str
|
||||
run_id: str
|
||||
manifest_id: str
|
||||
evidence_digest: str
|
||||
dataset_snapshot_id: str
|
||||
run_ref_document_sha256: str
|
||||
manifest_document_sha256: str
|
||||
source_universe_digest: str
|
||||
portfolio_asset_set_digest: str
|
||||
target_id: str
|
||||
target_weights: Mapping[str, float]
|
||||
prior_weights: Mapping[str, float] | None
|
||||
objective_name: str
|
||||
objective_version: str
|
||||
objective_digest: str
|
||||
model_name: str
|
||||
model_version: str
|
||||
model_digest: str
|
||||
expected_return_digest: str
|
||||
covariance_digest: str
|
||||
scenario_digest: str
|
||||
constraints: ConstraintSetV1
|
||||
freshness_policy: FreshnessPolicy
|
||||
receipt: ComputationReceipt
|
||||
gross_exposure: float
|
||||
net_exposure: float
|
||||
turnover_l1: float | None
|
||||
position_count: int
|
||||
constraint_residuals: Mapping[str, float]
|
||||
output_digest: str
|
||||
effective_at: str
|
||||
created_at: str
|
||||
computed_at: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
_target: RetrospectivePortfolioTarget = field(repr=False, compare=False)
|
||||
_run: RetrospectiveBacktestRunRef = field(repr=False, compare=False)
|
||||
_manifest: RetrospectiveBacktestEvidenceManifest = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
# Rebuild from independent typed inputs, not from a self-approved target in the wire.
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
|
||||
)
|
||||
arguments = {
|
||||
key: row[key]
|
||||
for key in (
|
||||
"objective_name",
|
||||
"objective_version",
|
||||
"objective_digest",
|
||||
"model_name",
|
||||
"model_version",
|
||||
"model_digest",
|
||||
"expected_return_digest",
|
||||
"covariance_digest",
|
||||
"scenario_digest",
|
||||
"computed_at",
|
||||
"prior_weights",
|
||||
)
|
||||
}
|
||||
rebuilt = build_retrospective_portfolio_decision(
|
||||
**arguments,
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=target,
|
||||
constraints=ConstraintSetV1.from_dict(row["constraints"]),
|
||||
freshness_policy=FreshnessPolicy.from_dict(row["freshness_policy"]),
|
||||
receipt=ComputationReceipt.from_dict(row["receipt"]),
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_json_object(value), **kwargs)
|
||||
|
||||
|
||||
def build_retrospective_portfolio_decision(
|
||||
*,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
target: RetrospectivePortfolioTarget,
|
||||
objective_name: str,
|
||||
objective_version: str,
|
||||
objective_digest: str,
|
||||
model_name: str,
|
||||
model_version: str,
|
||||
model_digest: str,
|
||||
expected_return_digest: str,
|
||||
covariance_digest: str,
|
||||
scenario_digest: str,
|
||||
constraints: ConstraintSetV1,
|
||||
freshness_policy: FreshnessPolicy,
|
||||
receipt: ComputationReceipt,
|
||||
computed_at: str,
|
||||
prior_weights: Mapping[str, float] | None = None,
|
||||
) -> RetrospectivePortfolioDecision:
|
||||
"""Verify the existing constraints and receipt, with two explicitly different clocks."""
|
||||
_check(
|
||||
type(receipt) is ComputationReceipt,
|
||||
"$.receipt",
|
||||
"typed computation receipt required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
receipt = ComputationReceipt.from_dict(receipt.to_dict())
|
||||
inputs = _material(
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=target,
|
||||
objective_name=objective_name,
|
||||
objective_version=objective_version,
|
||||
objective_digest=objective_digest,
|
||||
model_name=model_name,
|
||||
model_version=model_version,
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness_policy,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
run, manifest, target = inputs.run, inputs.manifest, inputs.target
|
||||
computed = _parse_utc(computed_at, "$.computed_at")
|
||||
created = _parse_utc(target.created_at, "$.target.created_at")
|
||||
available = _parse_utc(manifest.artifact_available_at, "$.manifest.artifact_available_at")
|
||||
_check(
|
||||
available <= created <= computed,
|
||||
"$.target.created_at",
|
||||
"artifact availability <= actual target creation <= computation required",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
_parse_utc(receipt.computed_at, "$.receipt.computed_at") == computed,
|
||||
"$.receipt.computed_at",
|
||||
"receipt actual time differs from computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
(computed - available).total_seconds() <= inputs.freshness.max_manifest_age_seconds,
|
||||
"$.manifest.artifact_available_at",
|
||||
"manifest is stale at actual computation",
|
||||
)
|
||||
_check(
|
||||
receipt.status not in {ReceiptStatus.FAILED, ReceiptStatus.FALLBACK},
|
||||
"$.receipt.status",
|
||||
"failed/fallback computation cannot form a result",
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
)
|
||||
digests = _receipt_digests(inputs)
|
||||
for key, expected in digests.items():
|
||||
_check(
|
||||
getattr(receipt, key) == expected,
|
||||
f"$.receipt.{key}",
|
||||
"receipt differs from independently recomputed evidence",
|
||||
ContractErrorCode.ARTIFACT_MISMATCH,
|
||||
)
|
||||
_check(
|
||||
digests["max_constraint_residual"] == 0.0,
|
||||
"$.constraints",
|
||||
"target violates supported constraints",
|
||||
)
|
||||
payload = {
|
||||
"contract_name": "researchhub.portfolio-decision",
|
||||
"schema_version": "2.0.0",
|
||||
"run_id": run.run_id,
|
||||
"manifest_id": manifest.manifest_id,
|
||||
"evidence_digest": manifest.evidence_digest,
|
||||
"dataset_snapshot_id": run.dataset_snapshot_id,
|
||||
"run_ref_document_sha256": _document_sha256(run.to_json()),
|
||||
"manifest_document_sha256": _document_sha256(manifest.to_json()),
|
||||
"source_universe_digest": run.universe_digest,
|
||||
"portfolio_asset_set_digest": _payload_digest(sorted(inputs.weights)),
|
||||
"target_id": target.target_id,
|
||||
"target_weights": _mapping_dict(inputs.weights),
|
||||
"prior_weights": None if inputs.prior is None else _mapping_dict(inputs.prior),
|
||||
"objective_name": objective_name,
|
||||
"objective_version": objective_version,
|
||||
"objective_digest": objective_digest,
|
||||
"model_name": model_name,
|
||||
"model_version": model_version,
|
||||
"model_digest": model_digest,
|
||||
"expected_return_digest": expected_return_digest,
|
||||
"covariance_digest": covariance_digest,
|
||||
"scenario_digest": scenario_digest,
|
||||
"constraints": inputs.constraints.to_dict(),
|
||||
"freshness_policy": inputs.freshness.to_dict(),
|
||||
"receipt": receipt.to_dict(),
|
||||
**inputs.metrics,
|
||||
"constraint_residuals": inputs.residuals,
|
||||
"output_digest": digests["output_digest"],
|
||||
"effective_at": target.effective_at,
|
||||
"created_at": target.created_at,
|
||||
"computed_at": computed_at,
|
||||
"observation_cutoff": run.observation_cutoff,
|
||||
"evidence_scope": run.evidence_scope,
|
||||
"usage": run.usage,
|
||||
"historical_availability": run.historical_availability,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
_public(payload)
|
||||
payload["decision_id"] = "rhportfoliodecisionv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(RetrospectivePortfolioDecision)
|
||||
values = {
|
||||
**payload,
|
||||
"target_weights": inputs.weights,
|
||||
"prior_weights": inputs.prior,
|
||||
"constraints": inputs.constraints,
|
||||
"freshness_policy": inputs.freshness,
|
||||
"receipt": receipt,
|
||||
"constraint_residuals": MappingProxyType(inputs.residuals),
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
"_target": target,
|
||||
"_run": run,
|
||||
"_manifest": manifest,
|
||||
}
|
||||
for key, value in values.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False)
|
||||
class RetrospectiveRiskAssessment:
|
||||
contract_name: str
|
||||
schema_version: str
|
||||
assessment_id: str
|
||||
decision_id: str
|
||||
run_id: str
|
||||
manifest_id: str
|
||||
dataset_snapshot_id: str
|
||||
covariance_data_snapshot_id: str
|
||||
covariance_snapshot_id: str
|
||||
covariance_as_of_date: str
|
||||
covariance_method: str
|
||||
covariance_window_start_date: str
|
||||
covariance_window_end_date: str
|
||||
covariance_observations: int | None
|
||||
covariance_lookback_sessions: int | None
|
||||
covariance_missing_policy: str
|
||||
covariance_input_digest: str
|
||||
covariance_matrix_digest: str
|
||||
return_frequency: str
|
||||
periods_per_year: int
|
||||
risk_model_name: str
|
||||
risk_model_version: str
|
||||
risk_model_digest: str
|
||||
freshness_policy_digest: str
|
||||
scenario_digest: str
|
||||
portfolio_volatility_limit: float | None
|
||||
risk_budget: Mapping[str, float]
|
||||
groups: Mapping[str, str] | None
|
||||
marginal_risk: Mapping[str, float]
|
||||
component_risk: Mapping[str, float]
|
||||
percentage_risk: Mapping[str, float]
|
||||
portfolio_volatility: float | None
|
||||
group_exposure: Mapping[str, float]
|
||||
findings: tuple[RiskFindingCode, ...]
|
||||
status: RiskAssessmentStatus
|
||||
qualified: bool
|
||||
effective_at: str
|
||||
computed_at: str
|
||||
observation_cutoff: str
|
||||
evidence_scope: str
|
||||
usage: str
|
||||
historical_availability: str
|
||||
decision_eligible: bool
|
||||
execution_validation: str
|
||||
_payload: Mapping[str, Any] = field(repr=False, compare=False)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return cast(dict[str, Any], _thaw_json(self._payload))
|
||||
|
||||
def to_json(self) -> str:
|
||||
return _canonical_json(self.to_dict())
|
||||
|
||||
@classmethod
|
||||
def from_dict(
|
||||
cls,
|
||||
value: Any,
|
||||
*,
|
||||
portfolio_decision: RetrospectivePortfolioDecision,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
covariance: CovarianceSnapshot,
|
||||
) -> Self:
|
||||
_performance_validate_tree(value, "$")
|
||||
row = _shape(
|
||||
value,
|
||||
"$",
|
||||
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
|
||||
)
|
||||
rebuilt = assess_retrospective_portfolio_risk(
|
||||
portfolio_decision=portfolio_decision,
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
covariance=covariance,
|
||||
**{
|
||||
key: row[key]
|
||||
for key in (
|
||||
"risk_model_name",
|
||||
"risk_model_version",
|
||||
"risk_model_digest",
|
||||
"risk_budget",
|
||||
"portfolio_volatility_limit",
|
||||
"groups",
|
||||
"computed_at",
|
||||
)
|
||||
},
|
||||
)
|
||||
_performance_compare(row, rebuilt.to_dict(), "$")
|
||||
return cast(Self, rebuilt)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
|
||||
return cls.from_dict(_json_object(value), **kwargs)
|
||||
|
||||
|
||||
class _RiskContext(TypedDict):
|
||||
decision: RetrospectivePortfolioDecision
|
||||
covariance: CovarianceSnapshot
|
||||
matrix_digest: str
|
||||
risk_model_name: str
|
||||
risk_model_version: str
|
||||
risk_model_digest: str
|
||||
portfolio_volatility_limit: float | None
|
||||
risk_budget: Mapping[str, float]
|
||||
groups: Mapping[str, str] | None
|
||||
computed_at: str
|
||||
|
||||
|
||||
def _risk_result(
|
||||
*,
|
||||
decision: RetrospectivePortfolioDecision,
|
||||
covariance: CovarianceSnapshot,
|
||||
matrix_digest: str,
|
||||
risk_model_name: str,
|
||||
risk_model_version: str,
|
||||
risk_model_digest: str,
|
||||
portfolio_volatility_limit: float | None,
|
||||
risk_budget: Mapping[str, float],
|
||||
groups: Mapping[str, str] | None,
|
||||
marginal: Mapping[str, float],
|
||||
component: Mapping[str, float],
|
||||
percentage: Mapping[str, float],
|
||||
volatility: float | None,
|
||||
grouped: Mapping[str, float],
|
||||
findings: tuple[RiskFindingCode, ...],
|
||||
status: RiskAssessmentStatus,
|
||||
qualified: bool,
|
||||
computed_at: str,
|
||||
) -> RetrospectiveRiskAssessment:
|
||||
assert covariance.window_start_date is not None
|
||||
assert covariance.window_end_date is not None
|
||||
payload = {
|
||||
"contract_name": "researchhub.risk-assessment",
|
||||
"schema_version": "2.0.0",
|
||||
"decision_id": decision.decision_id,
|
||||
"run_id": decision.run_id,
|
||||
"manifest_id": decision.manifest_id,
|
||||
"dataset_snapshot_id": decision.dataset_snapshot_id,
|
||||
"covariance_data_snapshot_id": covariance.data_snapshot_id,
|
||||
"covariance_snapshot_id": covariance.snapshot_id,
|
||||
"covariance_as_of_date": covariance.as_of_date.isoformat(),
|
||||
"covariance_method": covariance.method,
|
||||
"covariance_window_start_date": covariance.window_start_date.isoformat(),
|
||||
"covariance_window_end_date": covariance.window_end_date.isoformat(),
|
||||
"covariance_observations": covariance.observations,
|
||||
"covariance_lookback_sessions": covariance.lookback_sessions,
|
||||
"covariance_missing_policy": covariance.missing_policy,
|
||||
"covariance_input_digest": "sha256:" + covariance.input_sha256,
|
||||
"covariance_matrix_digest": matrix_digest,
|
||||
"return_frequency": covariance.return_frequency,
|
||||
"periods_per_year": covariance.periods_per_year,
|
||||
"risk_model_name": risk_model_name,
|
||||
"risk_model_version": risk_model_version,
|
||||
"risk_model_digest": risk_model_digest,
|
||||
"freshness_policy_digest": _payload_digest(decision.freshness_policy.to_dict()),
|
||||
"scenario_digest": decision.scenario_digest,
|
||||
"portfolio_volatility_limit": portfolio_volatility_limit,
|
||||
"risk_budget": _mapping_dict(risk_budget),
|
||||
"groups": None if groups is None else dict(groups),
|
||||
"marginal_risk": _mapping_dict(marginal),
|
||||
"component_risk": _mapping_dict(component),
|
||||
"percentage_risk": _mapping_dict(percentage),
|
||||
"portfolio_volatility": volatility,
|
||||
"group_exposure": _mapping_dict(grouped),
|
||||
"findings": [finding.value for finding in findings],
|
||||
"status": status.value,
|
||||
"qualified": qualified,
|
||||
"effective_at": decision.effective_at,
|
||||
"computed_at": computed_at,
|
||||
"observation_cutoff": decision.observation_cutoff,
|
||||
"evidence_scope": decision.evidence_scope,
|
||||
"usage": decision.usage,
|
||||
"historical_availability": decision.historical_availability,
|
||||
"decision_eligible": False,
|
||||
"execution_validation": "not_validated",
|
||||
}
|
||||
_public(payload)
|
||||
payload["assessment_id"] = "rhriskassessmentv2:" + _payload_digest(payload)
|
||||
instance = object.__new__(RetrospectiveRiskAssessment)
|
||||
values = {
|
||||
**payload,
|
||||
"risk_budget": risk_budget,
|
||||
"groups": groups,
|
||||
"marginal_risk": marginal,
|
||||
"component_risk": component,
|
||||
"percentage_risk": percentage,
|
||||
"group_exposure": grouped,
|
||||
"findings": findings,
|
||||
"status": status,
|
||||
"_payload": _freeze_numeric_evidence(payload),
|
||||
}
|
||||
for key, value in values.items():
|
||||
object.__setattr__(instance, key, value)
|
||||
return instance
|
||||
|
||||
|
||||
def assess_retrospective_portfolio_risk(
|
||||
*,
|
||||
portfolio_decision: RetrospectivePortfolioDecision,
|
||||
backtest_run_ref: RetrospectiveBacktestRunRef,
|
||||
manifest: RetrospectiveBacktestEvidenceManifest,
|
||||
covariance: CovarianceSnapshot,
|
||||
risk_model_name: str,
|
||||
risk_model_version: str,
|
||||
risk_model_digest: str,
|
||||
computed_at: str,
|
||||
risk_budget: Mapping[str, float] | None = None,
|
||||
portfolio_volatility_limit: float | None = None,
|
||||
groups: Mapping[str, str] | None = None,
|
||||
) -> RetrospectiveRiskAssessment:
|
||||
"""Use the existing Euler decomposition once; distinguish the two freshness clocks."""
|
||||
_check(
|
||||
type(portfolio_decision) is RetrospectivePortfolioDecision,
|
||||
"$.portfolio_decision",
|
||||
"explicit v2 portfolio result required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
_check(
|
||||
type(covariance) is CovarianceSnapshot,
|
||||
"$.covariance",
|
||||
"typed covariance required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
decision = RetrospectivePortfolioDecision.from_dict(
|
||||
portfolio_decision.to_dict(),
|
||||
backtest_run_ref=backtest_run_ref,
|
||||
manifest=manifest,
|
||||
target=portfolio_decision._target,
|
||||
)
|
||||
actual_computed = _parse_utc(computed_at, "$.computed_at")
|
||||
_check(
|
||||
_parse_utc(decision.computed_at, "$.portfolio_decision.computed_at") <= actual_computed,
|
||||
"$.computed_at",
|
||||
"risk computation precedes portfolio computation",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
manifest_age = (
|
||||
actual_computed
|
||||
- _parse_utc(decision._manifest.artifact_available_at, "$.manifest.artifact_available_at")
|
||||
).total_seconds()
|
||||
_check(
|
||||
0 <= manifest_age <= decision.freshness_policy.max_manifest_age_seconds,
|
||||
"$.manifest.artifact_available_at",
|
||||
"manifest is stale at actual risk computation",
|
||||
)
|
||||
_check(
|
||||
covariance.data_snapshot_id == decision.dataset_snapshot_id,
|
||||
"$.covariance.data_snapshot_id",
|
||||
"covariance and decision data differ",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
business_date = _parse_utc(decision.effective_at, "$.portfolio_decision.effective_at").date()
|
||||
_check(
|
||||
covariance.window_start_date is not None and covariance.window_end_date is not None,
|
||||
"$.covariance",
|
||||
"bounded covariance window required",
|
||||
)
|
||||
assert covariance.window_start_date is not None
|
||||
assert covariance.window_end_date is not None
|
||||
_check(
|
||||
covariance.window_start_date
|
||||
<= covariance.window_end_date
|
||||
<= covariance.as_of_date
|
||||
<= business_date,
|
||||
"$.covariance.as_of_date",
|
||||
"covariance business dates exceed the historical target date",
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
)
|
||||
_check(
|
||||
(business_date - covariance.as_of_date).days
|
||||
<= decision.freshness_policy.max_covariance_age_days,
|
||||
"$.covariance.as_of_date",
|
||||
"covariance is stale at historical target date",
|
||||
)
|
||||
_check(
|
||||
_digest("sha256:" + covariance.input_sha256, "$.covariance.input_sha256")
|
||||
== decision.covariance_digest,
|
||||
"$.covariance.input_sha256",
|
||||
"covariance input differs from portfolio receipt",
|
||||
ContractErrorCode.IDENTITY_MISMATCH,
|
||||
)
|
||||
name = _text(risk_model_name, "$.risk_model_name")
|
||||
version = _semver(risk_model_version, "$.risk_model_version")
|
||||
model_digest = _digest(risk_model_digest, "$.risk_model_digest")
|
||||
limit = (
|
||||
None
|
||||
if portfolio_volatility_limit is None
|
||||
else _finite_number(
|
||||
portfolio_volatility_limit, "$.portfolio_volatility_limit", non_negative=True
|
||||
)
|
||||
)
|
||||
budget: Mapping[str, float] = (
|
||||
MappingProxyType({})
|
||||
if risk_budget is None
|
||||
else _immutable_float_mapping(risk_budget, "$.risk_budget")
|
||||
)
|
||||
_check(
|
||||
all(value >= 0 for value in budget.values())
|
||||
and set(budget) <= decision.target_weights.keys(),
|
||||
"$.risk_budget",
|
||||
"risk budgets must be non-negative and use target labels",
|
||||
)
|
||||
normalized_groups = None
|
||||
if groups is not None:
|
||||
_check(
|
||||
isinstance(groups, Mapping),
|
||||
"$.groups",
|
||||
"mapping required",
|
||||
ContractErrorCode.TYPE_ERROR,
|
||||
)
|
||||
group_values = {
|
||||
_text(key, "$.groups.keys"): _text(value, "$.groups.values")
|
||||
for key, value in groups.items()
|
||||
}
|
||||
_check(
|
||||
set(group_values) == decision.target_weights.keys(),
|
||||
"$.groups",
|
||||
"groups must label every target exactly once",
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
)
|
||||
normalized_groups = MappingProxyType(dict(sorted(group_values.items())))
|
||||
aligned = _validate_covariance_structure(decision.target_weights, covariance)
|
||||
matrix_digest = _payload_digest(
|
||||
{
|
||||
"assets": sorted(decision.target_weights),
|
||||
"matrix": aligned.to_numpy(dtype=float).tolist(),
|
||||
}
|
||||
)
|
||||
arguments: _RiskContext = {
|
||||
"decision": decision,
|
||||
"covariance": covariance,
|
||||
"matrix_digest": matrix_digest,
|
||||
"risk_model_name": name,
|
||||
"risk_model_version": version,
|
||||
"risk_model_digest": model_digest,
|
||||
"portfolio_volatility_limit": limit,
|
||||
"risk_budget": budget,
|
||||
"groups": normalized_groups,
|
||||
"computed_at": computed_at,
|
||||
}
|
||||
empty: Mapping[str, float] = MappingProxyType({})
|
||||
|
||||
def unavailable(finding: RiskFindingCode) -> RetrospectiveRiskAssessment:
|
||||
return _risk_result(
|
||||
**arguments,
|
||||
marginal=empty,
|
||||
component=empty,
|
||||
percentage=empty,
|
||||
volatility=None,
|
||||
grouped=empty,
|
||||
findings=(finding,),
|
||||
status=RiskAssessmentStatus.UNAVAILABLE,
|
||||
qualified=False,
|
||||
)
|
||||
|
||||
weights = pd.Series(_mapping_dict(decision.target_weights), dtype=float, name="weight")
|
||||
try:
|
||||
decomposition = labeled_component_risk(weights, aligned * covariance.periods_per_year)
|
||||
except ValueError as error:
|
||||
finding = {
|
||||
"covariance must be positive semidefinite": RiskFindingCode.COVARIANCE_NOT_PSD,
|
||||
"weights and covariance must produce positive portfolio variance": RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE,
|
||||
}.get(str(error))
|
||||
if finding is None:
|
||||
raise PortfolioRiskContractError(
|
||||
PortfolioRiskContractErrorCode.COMPUTATION_FAILURE,
|
||||
"$.covariance",
|
||||
"risk computation failed",
|
||||
) from error
|
||||
return unavailable(finding)
|
||||
marginal = _series_mapping(decomposition.marginal)
|
||||
component = _series_mapping(decomposition.component)
|
||||
percentage = _series_mapping(decomposition.percentage)
|
||||
volatility = _finite_number(
|
||||
decomposition.portfolio_volatility, "$.risk_output.portfolio_volatility", non_negative=True
|
||||
)
|
||||
if not (
|
||||
set(marginal) == set(component) == set(percentage) == decision.target_weights.keys()
|
||||
and math.isclose(
|
||||
sum(component.values()), volatility, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
|
||||
)
|
||||
and math.isclose(
|
||||
sum(percentage.values()), 1.0, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
|
||||
)
|
||||
):
|
||||
return unavailable(RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED)
|
||||
grouped = (
|
||||
empty
|
||||
if normalized_groups is None
|
||||
else _series_mapping(
|
||||
decomposition.grouped_component(pd.Series(dict(normalized_groups), dtype="object"))
|
||||
)
|
||||
)
|
||||
breached = (limit is not None and volatility > limit + _CLOSURE_ATOL) or any(
|
||||
percentage[label] > maximum + _CLOSURE_ATOL for label, maximum in budget.items()
|
||||
)
|
||||
return _risk_result(
|
||||
**arguments,
|
||||
marginal=marginal,
|
||||
component=component,
|
||||
percentage=percentage,
|
||||
volatility=volatility,
|
||||
grouped=grouped,
|
||||
findings=(RiskFindingCode.RISK_BUDGET_BREACH,) if breached else (),
|
||||
status=RiskAssessmentStatus.READY,
|
||||
qualified=not breached,
|
||||
)
|
||||
@@ -1,354 +0,0 @@
|
||||
"""Storage-neutral projection of seven-strategy research and bounded grid rankings.
|
||||
|
||||
All accounting and metrics are owned by the completed core result. Strategy
|
||||
signals have no factor scores, so their full causal record lives in the covered
|
||||
report, not the legacy factor-signal table. No source or decision admission is
|
||||
granted by this projection.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import asdict
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from .artifact import (
|
||||
RESEARCH_ARTIFACT_SCHEMA_VERSION,
|
||||
RISK_COLUMNS,
|
||||
ResearchRunArtifact,
|
||||
_aware_timestamp,
|
||||
_canonical_mapping_json,
|
||||
_required_text,
|
||||
)
|
||||
from .strategy_optimizer import StrategyOptimizationResult
|
||||
from .strategy_research import StrategyResearchResult
|
||||
|
||||
STRATEGY_REPORT_SCHEMA = "researchhub.strategy-research.v1"
|
||||
_PERFORMANCE = {
|
||||
"total_ret": "total_return",
|
||||
"ann_ret": "ann_return",
|
||||
"ann_volatility": "ann_volatility",
|
||||
"sharpe": "sharpe",
|
||||
"sortino": "sortino",
|
||||
"max_dd": "max_drawdown",
|
||||
"calmar": "calmar",
|
||||
"win_rate": "trade_win_rate",
|
||||
}
|
||||
_RELATIVE = {
|
||||
"tracking_error": "tracking_error",
|
||||
"ir": "information_ratio",
|
||||
"alpha": "alpha",
|
||||
"beta": "beta",
|
||||
}
|
||||
_SIGNAL_COLUMNS = [
|
||||
"run_id",
|
||||
"signal_date",
|
||||
"execution_date",
|
||||
"asset_id",
|
||||
"factor_score",
|
||||
"target_weight",
|
||||
]
|
||||
_POSITION_COLUMNS = [
|
||||
"run_id",
|
||||
"trade_date",
|
||||
"asset_id",
|
||||
"asset_type",
|
||||
"quantity",
|
||||
"mark_price",
|
||||
"market_value",
|
||||
"weight",
|
||||
]
|
||||
|
||||
|
||||
def _report(
|
||||
result: StrategyResearchResult,
|
||||
run_id: str,
|
||||
optimization: StrategyOptimizationResult | None,
|
||||
) -> dict[str, object]:
|
||||
ranking: dict[str, object] | None = None
|
||||
if optimization is not None:
|
||||
ranking = {
|
||||
"objective": optimization.objective,
|
||||
"grid": optimization.param_grid,
|
||||
"trial_count": len(optimization.trials),
|
||||
"selected_rank": 1,
|
||||
"trials": [
|
||||
{
|
||||
"rank": rank,
|
||||
"parameters": dict(trial.parameters),
|
||||
"score": trial.score,
|
||||
"metrics": dict(trial.result.metrics),
|
||||
"metric_unavailable": dict(trial.result.metric_unavailable),
|
||||
}
|
||||
for rank, trial in enumerate(optimization.trials, start=1)
|
||||
],
|
||||
}
|
||||
return {
|
||||
"schema_version": STRATEGY_REPORT_SCHEMA,
|
||||
"strategy": result.strategy,
|
||||
"asset": result.asset,
|
||||
"parameters": dict(result.parameters),
|
||||
"costs": dict(result.cost_parameters),
|
||||
"execution": {
|
||||
"signal_observation": "after_close",
|
||||
"fill_price": "next_session_open",
|
||||
"valuation_price": "session_close",
|
||||
"lag_sessions": 1,
|
||||
"quantity_basis": "fractional_research",
|
||||
},
|
||||
"signals": [
|
||||
asdict(signal)
|
||||
| {
|
||||
"asset_id": result.asset,
|
||||
"signal_id": f"{run_id}:signal:{signal.decision_date}",
|
||||
}
|
||||
for signal in result.signals
|
||||
],
|
||||
"trade_pairing": asdict(result.pairing),
|
||||
"metrics": dict(result.metrics),
|
||||
"metric_unavailable": dict(result.metric_unavailable),
|
||||
"benchmark": {"status": result.benchmark_status, "metrics": result.benchmark_metrics},
|
||||
"optimization": ranking,
|
||||
"projections": {
|
||||
"signals": "strategy_report.signals",
|
||||
"attribution": "not_computed",
|
||||
"risk": "not_computed",
|
||||
"position_weight_basis": "market_value_over_nav",
|
||||
"undefined_weight_reason": "zero_portfolio_value",
|
||||
"undefined_weight_dates": [
|
||||
position.date
|
||||
for position in result.ledger.positions
|
||||
if position.portfolio_value == 0
|
||||
],
|
||||
},
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def _performance(
|
||||
result: StrategyResearchResult,
|
||||
run_id: str,
|
||||
) -> tuple[pd.DataFrame, dict[str, object]]:
|
||||
values: dict[str, object] = {"run_id": run_id}
|
||||
reasons: dict[str, str] = {}
|
||||
for target, source in _PERFORMANCE.items():
|
||||
value = result.metrics[source]
|
||||
values[target] = value
|
||||
if value is None:
|
||||
reasons[target] = result.metric_unavailable[source]
|
||||
for target, source in _RELATIVE.items():
|
||||
value = None if result.benchmark_metrics is None else result.benchmark_metrics[source]
|
||||
values[target] = value
|
||||
if value is None:
|
||||
reasons[target] = (
|
||||
"benchmark_" + result.benchmark_status
|
||||
if result.benchmark_metrics is None
|
||||
else "benchmark_metric_undefined"
|
||||
)
|
||||
values["n_trades"] = len(result.ledger.trades_frame)
|
||||
values["n_days"] = len(result.ledger.positions)
|
||||
return pd.DataFrame([values]), {
|
||||
"unavailable_reasons": reasons,
|
||||
"win_rate_basis": "completed_trades",
|
||||
}
|
||||
|
||||
|
||||
def _nav(result: StrategyResearchResult, run_id: str) -> pd.DataFrame:
|
||||
frame = result.ledger.ledger_frame
|
||||
frame.insert(0, "run_id", run_id)
|
||||
frame["trade_date"] = pd.to_datetime(frame["trade_date"]).dt.date
|
||||
frame["total_cost"] = [
|
||||
sum(fill.total_cost for fill in day.executions) for day in result.ledger.daily_executions
|
||||
]
|
||||
benchmark_returns = result.benchmark_returns
|
||||
if result.benchmark_nav is None or benchmark_returns is None:
|
||||
frame["benchmark_nav"] = None
|
||||
frame["benchmark_return"] = None
|
||||
frame["excess_ret"] = None
|
||||
else:
|
||||
frame["benchmark_nav"] = result.benchmark_nav.to_numpy(copy=True)
|
||||
frame["benchmark_return"] = benchmark_returns.to_numpy(copy=True)
|
||||
frame["excess_ret"] = result.ledger.daily_returns.to_numpy() - benchmark_returns.to_numpy()
|
||||
return frame
|
||||
|
||||
|
||||
def _trades(result: StrategyResearchResult, run_id: str) -> pd.DataFrame:
|
||||
frame = result.ledger.trades_frame
|
||||
frame.insert(0, "run_id", run_id)
|
||||
frame.insert(
|
||||
1, "trade_id", [f"{run_id}:{sequence:08d}" for sequence in range(1, len(frame) + 1)]
|
||||
)
|
||||
signal_by_execution = {
|
||||
signal.execution_date: signal.decision_date
|
||||
for signal in result.signals
|
||||
if signal.execution_date is not None
|
||||
}
|
||||
frame["signal_id"] = [
|
||||
f"{run_id}:signal:{signal_by_execution[day]}" for day in frame["trade_date"]
|
||||
]
|
||||
frame["trade_date"] = pd.to_datetime(frame["trade_date"]).dt.date
|
||||
frame["total_cost"] = frame["fee"] + frame["slippage"]
|
||||
return frame
|
||||
|
||||
|
||||
def _positions(result: StrategyResearchResult, run_id: str) -> pd.DataFrame:
|
||||
rows = []
|
||||
observed = result.bars
|
||||
for day, position in zip(observed.index, result.ledger.positions, strict=True):
|
||||
for asset, quantity in position.holdings.items():
|
||||
if asset != result.asset:
|
||||
raise ValueError("Strategy ledger contains an unexpected asset")
|
||||
price = float(observed.at[day, "close"])
|
||||
market_value = quantity * price
|
||||
rows.append(
|
||||
{
|
||||
"run_id": run_id,
|
||||
"trade_date": day.date(),
|
||||
"asset_id": asset,
|
||||
"asset_type": "security",
|
||||
"quantity": quantity,
|
||||
"mark_price": price,
|
||||
"market_value": market_value,
|
||||
"weight": market_value / position.portfolio_value
|
||||
if position.portfolio_value
|
||||
else None,
|
||||
}
|
||||
)
|
||||
rows.append(
|
||||
{
|
||||
"run_id": run_id,
|
||||
"trade_date": day.date(),
|
||||
"asset_id": "CASH",
|
||||
"asset_type": "cash",
|
||||
"quantity": position.cash,
|
||||
"mark_price": 1.0,
|
||||
"market_value": position.cash,
|
||||
"weight": position.cash / position.portfolio_value
|
||||
if position.portfolio_value
|
||||
else None,
|
||||
}
|
||||
)
|
||||
return pd.DataFrame(rows, columns=_POSITION_COLUMNS)
|
||||
|
||||
|
||||
def build_strategy_research_artifact(
|
||||
result: StrategyResearchResult | StrategyOptimizationResult,
|
||||
*,
|
||||
run_id: str,
|
||||
strategy_id: str,
|
||||
strategy_name: str,
|
||||
strategy_version: str,
|
||||
engine_version: str,
|
||||
code_revision: str,
|
||||
data_snapshot_id: str,
|
||||
calendar: str,
|
||||
timezone: str,
|
||||
started_at: str | pd.Timestamp,
|
||||
finished_at: str | pd.Timestamp,
|
||||
parameters: Mapping[str, object],
|
||||
benchmark_id: str | None = None,
|
||||
) -> ResearchRunArtifact:
|
||||
"""Snapshot the chosen real ledger plus all bounded trial summaries.
|
||||
|
||||
``parameters`` carries task/source identity supplied by the caller. The core
|
||||
reserves its report and metric explanation; callers cannot substitute them.
|
||||
Empty requested benchmark data remains distinguishable from no request.
|
||||
"""
|
||||
optimization = result if type(result) is StrategyOptimizationResult else None
|
||||
if optimization is not None:
|
||||
if not 1 <= len(optimization.trials) <= 100:
|
||||
raise ValueError("A bounded nonempty optimization result is required")
|
||||
selected = optimization.trials[0].result
|
||||
elif type(result) is StrategyResearchResult:
|
||||
selected = result
|
||||
else:
|
||||
raise TypeError("result must be strategy research or optimization")
|
||||
if selected.decision_eligible or (optimization is not None and optimization.decision_eligible):
|
||||
raise ValueError("Strategy research cannot grant decision eligibility")
|
||||
if selected.asset == "CASH":
|
||||
raise ValueError("Strategy asset cannot use the reserved CASH identity")
|
||||
normalized_run_id = _required_text(run_id, "run_id", max_length=128)
|
||||
metadata = {
|
||||
name: _required_text(value, name)
|
||||
for name, value in {
|
||||
"strategy_id": strategy_id,
|
||||
"strategy_name": strategy_name,
|
||||
"strategy_version": strategy_version,
|
||||
"engine_version": engine_version,
|
||||
"code_revision": code_revision,
|
||||
"data_snapshot_id": data_snapshot_id,
|
||||
"calendar": calendar,
|
||||
"timezone": timezone,
|
||||
}.items()
|
||||
}
|
||||
if not isinstance(parameters, Mapping):
|
||||
raise TypeError("parameters must be a mapping")
|
||||
if set(parameters) & {"strategy_report", "performance_interpretation"}:
|
||||
raise ValueError("Strategy report and performance interpretation are reserved")
|
||||
started, finished = (
|
||||
_aware_timestamp(started_at, "started_at"),
|
||||
_aware_timestamp(finished_at, "finished_at"),
|
||||
)
|
||||
if finished < started:
|
||||
raise ValueError("finished_at must not precede started_at")
|
||||
if (selected.benchmark_status == "not_requested") != (benchmark_id is None):
|
||||
raise ValueError(
|
||||
"Requested benchmark requires its identity; unrequested benchmark cannot have one"
|
||||
)
|
||||
benchmark = "" if benchmark_id is None else _required_text(benchmark_id, "benchmark_id")
|
||||
performance, interpretation = _performance(selected, normalized_run_id)
|
||||
params_json = _canonical_mapping_json(
|
||||
dict(parameters)
|
||||
| {
|
||||
"strategy_report": _report(selected, normalized_run_id, optimization),
|
||||
"performance_interpretation": interpretation,
|
||||
}
|
||||
)
|
||||
dates = selected.bars.index
|
||||
run = pd.DataFrame(
|
||||
[
|
||||
metadata
|
||||
| {
|
||||
"schema_version": RESEARCH_ARTIFACT_SCHEMA_VERSION,
|
||||
"run_id": normalized_run_id,
|
||||
"config_hash": hashlib.sha256(params_json.encode("utf-8")).hexdigest(),
|
||||
"benchmark_id": benchmark,
|
||||
"benchmark_alignment_policy": "exact_session_index"
|
||||
if selected.benchmark_status == "present"
|
||||
else "none",
|
||||
"frequency": "1d",
|
||||
"initial_capital": selected.ledger.initial_cash,
|
||||
"start_date": dates[0].date(),
|
||||
"end_date": dates[-1].date(),
|
||||
"status": "success",
|
||||
"started_at": started,
|
||||
"finished_at": finished,
|
||||
"params_json": params_json,
|
||||
}
|
||||
]
|
||||
)
|
||||
return ResearchRunArtifact(
|
||||
schema_version=RESEARCH_ARTIFACT_SCHEMA_VERSION,
|
||||
_run=run,
|
||||
_signals=pd.DataFrame(columns=_SIGNAL_COLUMNS),
|
||||
_nav=_nav(selected, normalized_run_id),
|
||||
_trades=_trades(selected, normalized_run_id),
|
||||
_positions=_positions(selected, normalized_run_id),
|
||||
_attribution=pd.DataFrame(
|
||||
columns=["run_id", "trade_date", "asset_id", "overnight", "intraday", "asset_total"]
|
||||
),
|
||||
_attribution_daily=pd.DataFrame(
|
||||
columns=[
|
||||
"run_id",
|
||||
"trade_date",
|
||||
"transaction_cost",
|
||||
"explained_return",
|
||||
"residual",
|
||||
"total_return",
|
||||
]
|
||||
),
|
||||
_risk=pd.DataFrame(columns=RISK_COLUMNS),
|
||||
_performance=performance,
|
||||
)
|
||||
@@ -1,140 +0,0 @@
|
||||
"""Bounded contracts for the seven migrated example strategies, without I/O."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from itertools import product
|
||||
|
||||
type Number = int | float
|
||||
type Parameters = dict[str, Number]
|
||||
MAX_OBSERVATIONS = 5000
|
||||
MAX_COMBINATIONS = 100
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _Parameter:
|
||||
default: Number
|
||||
minimum: Number
|
||||
maximum: Number
|
||||
integer: bool = False
|
||||
strict_minimum: bool = False
|
||||
|
||||
|
||||
def _period(default: int, minimum: int = 1) -> _Parameter:
|
||||
return _Parameter(default, minimum, 500, integer=True)
|
||||
|
||||
|
||||
_DEFINITIONS: dict[str, dict[str, _Parameter]] = {
|
||||
"BuyAndHold": {"buy_pct": _Parameter(0.95, 0, 1)},
|
||||
"SmaCross": {"fast": _period(5), "slow": _period(20)},
|
||||
"MACross": {
|
||||
"fast": _period(10),
|
||||
"slow": _period(30),
|
||||
"atr_period": _period(0, 0),
|
||||
"atr_mult": _Parameter(2.0, 0, 100, strict_minimum=True),
|
||||
},
|
||||
"RSI": {
|
||||
"period": _period(14),
|
||||
"oversold": _Parameter(30, 0, 100),
|
||||
"overbought": _Parameter(70, 0, 100),
|
||||
},
|
||||
"BollingerBreakout": {
|
||||
"period": _period(20),
|
||||
"std_mult": _Parameter(2.0, 0, 100, strict_minimum=True),
|
||||
},
|
||||
"DualThrust": {
|
||||
"period": _period(5),
|
||||
"k1": _Parameter(0.5, 0, 100),
|
||||
"k2": _Parameter(0.5, 0, 100),
|
||||
},
|
||||
"TurtleBreakout": {"entry_period": _period(20), "exit_period": _period(10)},
|
||||
}
|
||||
STRATEGIES = tuple(_DEFINITIONS)
|
||||
|
||||
|
||||
def finite_number(value: object, name: str) -> Number:
|
||||
if type(value) not in (int, float):
|
||||
raise ValueError(f"{name} must be a native finite number")
|
||||
assert isinstance(value, (int, float))
|
||||
try:
|
||||
valid = math.isfinite(value)
|
||||
except OverflowError:
|
||||
valid = False
|
||||
if not valid:
|
||||
raise ValueError(f"{name} must be a native finite number")
|
||||
return value
|
||||
|
||||
|
||||
def strategy_parameters(name: str, overrides: Parameters | None = None) -> Parameters:
|
||||
if type(name) is not str or name not in _DEFINITIONS:
|
||||
raise ValueError("Unknown built-in strategy")
|
||||
definition = _DEFINITIONS[name]
|
||||
if overrides is None:
|
||||
overrides = {}
|
||||
if type(overrides) is not dict or any(
|
||||
type(key) is not str or key not in definition for key in overrides
|
||||
):
|
||||
raise ValueError("Unknown strategy parameter")
|
||||
values = {key: spec.default for key, spec in definition.items()} | overrides
|
||||
for key, value in values.items():
|
||||
spec = definition[key]
|
||||
finite_number(value, key)
|
||||
if spec.integer and type(value) is not int:
|
||||
raise ValueError(f"{key} must be an integer")
|
||||
if value > spec.maximum or (
|
||||
value <= spec.minimum if spec.strict_minimum else value < spec.minimum
|
||||
):
|
||||
raise ValueError(f"{key} outside supported bounds")
|
||||
if name in ("SmaCross", "MACross") and values["fast"] >= values["slow"]:
|
||||
raise ValueError("fast must be less than slow")
|
||||
if name == "RSI" and values["oversold"] >= values["overbought"]:
|
||||
raise ValueError("oversold must be less than overbought")
|
||||
return values
|
||||
|
||||
|
||||
def required_history(name: str, params: Parameters) -> int:
|
||||
if name in ("SmaCross", "MACross"):
|
||||
return int(max(params["slow"], params.get("atr_period", 0))) + 1
|
||||
if name in ("RSI", "BollingerBreakout", "DualThrust"):
|
||||
return int(params["period"]) + 1
|
||||
if name == "TurtleBreakout":
|
||||
return int(max(params["entry_period"], params["exit_period"])) + 1
|
||||
return 2
|
||||
|
||||
|
||||
def validate_money(initial_cash: Number, commission: Number, stamp_duty: Number) -> None:
|
||||
for name, value in (
|
||||
("initial_cash", initial_cash),
|
||||
("commission", commission),
|
||||
("stamp_duty", stamp_duty),
|
||||
):
|
||||
finite_number(value, name)
|
||||
if not 0 < initial_cash <= 1e12:
|
||||
raise ValueError("initial_cash outside supported bounds")
|
||||
if not (0 <= commission <= 1 and 0 <= stamp_duty <= 1 and commission + stamp_duty <= 1):
|
||||
raise ValueError("Fee fractions and combined sell fee must be between zero and one")
|
||||
|
||||
|
||||
def parameter_grid(name: str, grid: dict[str, list[Number]]) -> tuple[Parameters, ...]:
|
||||
defaults = strategy_parameters(name)
|
||||
if type(grid) is not dict or not 1 <= len(grid) <= 4:
|
||||
raise ValueError("A bounded nonempty grid is required")
|
||||
count = 1
|
||||
for key, candidates in grid.items():
|
||||
if type(key) is not str or key not in defaults:
|
||||
raise ValueError("Unknown grid parameter")
|
||||
if type(candidates) is not list or not 1 <= len(candidates) <= 10:
|
||||
raise ValueError("A bounded nonempty candidate list is required")
|
||||
count *= len(candidates)
|
||||
if count > MAX_COMBINATIONS:
|
||||
raise ValueError("Grid combination limit exceeded")
|
||||
for value in candidates:
|
||||
finite_number(value, key)
|
||||
if len(set(candidates)) != len(candidates):
|
||||
raise ValueError("Duplicate grid candidates")
|
||||
keys = [key for key in defaults if key in grid]
|
||||
return tuple(
|
||||
strategy_parameters(name, dict(zip(keys, values, strict=True)))
|
||||
for values in product(*(grid[key] for key in keys))
|
||||
)
|
||||
@@ -1,109 +0,0 @@
|
||||
"""Bounded exhaustive research over the real seven-strategy core pipeline."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from .strategy_contracts import (
|
||||
Parameters,
|
||||
finite_number,
|
||||
parameter_grid,
|
||||
required_history,
|
||||
validate_money,
|
||||
)
|
||||
from .strategy_research import (
|
||||
BenchmarkInput,
|
||||
StrategyResearchResult,
|
||||
_benchmark,
|
||||
run_strategy_research,
|
||||
validate_strategy_bars,
|
||||
)
|
||||
|
||||
_OBJECTIVES = {"sharpe_ratio": "sharpe", "total_return": "total_return", "calmar_ratio": "calmar"}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StrategyTrial:
|
||||
parameters: Parameters
|
||||
score: float
|
||||
result: StrategyResearchResult
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StrategyOptimizationResult:
|
||||
strategy: str
|
||||
objective: str
|
||||
trials: tuple[StrategyTrial, ...]
|
||||
_param_grid: dict[str, list[int | float]]
|
||||
decision_eligible: bool = False
|
||||
|
||||
@property
|
||||
def param_grid(self) -> dict[str, list[int | float]]:
|
||||
return {key: list(values) for key, values in self._param_grid.items()}
|
||||
|
||||
|
||||
def optimize_strategy_research(
|
||||
strategy: str,
|
||||
bars: pd.DataFrame,
|
||||
*,
|
||||
asset: str,
|
||||
param_grid: dict[str, list[int | float]],
|
||||
objective: str = "sharpe_ratio",
|
||||
initial_cash: float = 1_000_000,
|
||||
commission: float = 0.0003,
|
||||
stamp_duty: float = 0.001,
|
||||
benchmark: BenchmarkInput | None = None,
|
||||
min_trade_amount: float = 0,
|
||||
) -> StrategyOptimizationResult:
|
||||
"""Validate every candidate first; an unavailable objective fails the ranking.
|
||||
|
||||
This has no source fetch, publication, task state or model-selection claim.
|
||||
Ties retain canonical axis/candidate order. Every trial owns a fresh ledger.
|
||||
"""
|
||||
if type(objective) is not str or objective not in _OBJECTIVES:
|
||||
raise ValueError("Unsupported optimization objective")
|
||||
validate_money(initial_cash, commission, stamp_duty)
|
||||
finite_number(min_trade_amount, "min_trade_amount")
|
||||
if min_trade_amount < 0:
|
||||
raise ValueError("min_trade_amount must be nonnegative")
|
||||
if (
|
||||
type(asset) is not str
|
||||
or not asset
|
||||
or asset == "CASH"
|
||||
or asset.strip() != asset
|
||||
or len(asset) > 100
|
||||
):
|
||||
raise ValueError("An explicit bounded asset key is required")
|
||||
combinations = parameter_grid(strategy, param_grid)
|
||||
observed = validate_strategy_bars(bars)
|
||||
for parameters in combinations:
|
||||
if len(observed) < required_history(strategy, parameters):
|
||||
raise ValueError("Insufficient strategy history for all grid candidates")
|
||||
_benchmark(benchmark, observed.index)
|
||||
trials = []
|
||||
for parameters in combinations:
|
||||
result = run_strategy_research(
|
||||
strategy,
|
||||
observed,
|
||||
asset=asset,
|
||||
params=parameters,
|
||||
initial_cash=initial_cash,
|
||||
commission=commission,
|
||||
stamp_duty=stamp_duty,
|
||||
benchmark=benchmark,
|
||||
min_trade_amount=min_trade_amount,
|
||||
)
|
||||
value = result.metrics[_OBJECTIVES[objective]]
|
||||
if value is None or not math.isfinite(value):
|
||||
raise ValueError(f"Optimization objective {objective} is unavailable for {parameters}")
|
||||
trials.append(StrategyTrial(dict(parameters), value, result))
|
||||
trials.sort(key=lambda trial: trial.score, reverse=True)
|
||||
return StrategyOptimizationResult(
|
||||
strategy,
|
||||
objective,
|
||||
tuple(trials),
|
||||
{key: list(values) for key, values in param_grid.items()},
|
||||
)
|
||||
@@ -1,472 +0,0 @@
|
||||
"""Seven long-only examples on the shared ledger, using caller-supplied OHLC.
|
||||
|
||||
Signals are observed after close, filled at the next supplied session's open,
|
||||
and marked at that day's close. Fractional quantities follow the shared research
|
||||
ledger. This is not exchange lot sizing, market/source admission or live execution.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal
|
||||
from numbers import Real
|
||||
from typing import Literal
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from .execution import (
|
||||
DailyPosition,
|
||||
ExecutionConfig,
|
||||
ExecutionSimulationResult,
|
||||
simulate_daily_ledger_with_audit,
|
||||
)
|
||||
from .metrics import benchmark_summary, summary
|
||||
from .strategy_contracts import (
|
||||
MAX_OBSERVATIONS,
|
||||
Parameters,
|
||||
finite_number,
|
||||
required_history,
|
||||
strategy_parameters,
|
||||
validate_money,
|
||||
)
|
||||
from .trade_pairing import TradePairing, pair_ledger_trades
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BenchmarkInput:
|
||||
status: Literal["not_requested", "present", "empty", "source_error"] = "not_requested"
|
||||
closes: pd.Series | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StrategySignal:
|
||||
decision_date: str
|
||||
execution_date: str | None
|
||||
target_weight: float
|
||||
reason: str
|
||||
status: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StrategyResearchResult:
|
||||
strategy: str
|
||||
parameters: Parameters
|
||||
ledger: ExecutionSimulationResult
|
||||
signals: tuple[StrategySignal, ...]
|
||||
pairing: TradePairing
|
||||
metrics: dict[str, float | None]
|
||||
metric_unavailable: dict[str, str]
|
||||
benchmark_status: str
|
||||
benchmark_nav: pd.Series | None
|
||||
benchmark_metrics: dict[str, float | None] | None
|
||||
asset: str
|
||||
_bars: pd.DataFrame
|
||||
_benchmark_returns: pd.Series | None
|
||||
cost_parameters: dict[str, float]
|
||||
decision_eligible: bool = False
|
||||
|
||||
@property
|
||||
def bars(self) -> pd.DataFrame:
|
||||
return self._bars.copy(deep=True)
|
||||
|
||||
@property
|
||||
def benchmark_returns(self) -> pd.Series | None:
|
||||
return None if self._benchmark_returns is None else self._benchmark_returns.copy(deep=True)
|
||||
|
||||
|
||||
def _calendar(index: pd.Index) -> pd.DatetimeIndex:
|
||||
if (
|
||||
not isinstance(index, pd.DatetimeIndex)
|
||||
or index.tz is not None
|
||||
or index.hasnans
|
||||
or index.has_duplicates
|
||||
or not index.is_monotonic_increasing
|
||||
or not index.equals(index.normalize())
|
||||
or (len(index) and (index.min().year < 1900 or index.max().year > 2100))
|
||||
):
|
||||
raise ValueError("A unique ordered naive daily DatetimeIndex is required")
|
||||
return index.copy()
|
||||
|
||||
|
||||
def _prices(values: list[object], name: str) -> list[float]:
|
||||
result = []
|
||||
for value in values:
|
||||
if isinstance(value, (bool, np.bool_)) or not isinstance(value, (Real, Decimal)):
|
||||
raise ValueError(f"{name} requires finite positive numeric prices")
|
||||
try:
|
||||
price = float(value)
|
||||
except (OverflowError, ValueError) as error:
|
||||
raise ValueError(f"{name} requires finite positive numeric prices") from error
|
||||
if not math.isfinite(price) or price <= 0:
|
||||
raise ValueError(f"{name} requires finite positive numeric prices")
|
||||
result.append(price)
|
||||
return result
|
||||
|
||||
|
||||
def validate_strategy_bars(bars: pd.DataFrame) -> pd.DataFrame:
|
||||
if (
|
||||
not isinstance(bars, pd.DataFrame)
|
||||
or bars.columns.has_duplicates
|
||||
or not {"open", "high", "low", "close"}.issubset(bars.columns)
|
||||
or not 2 <= len(bars) <= MAX_OBSERVATIONS
|
||||
):
|
||||
raise ValueError("Complete OHLC and 2 to 5000 observed sessions are required")
|
||||
index = _calendar(bars.index)
|
||||
result = pd.DataFrame(
|
||||
{
|
||||
column: _prices(bars[column].tolist(), column)
|
||||
for column in ("open", "high", "low", "close")
|
||||
},
|
||||
index=index,
|
||||
)
|
||||
if (result["high"] < result[["open", "close", "low"]].max(axis=1)).any() or (
|
||||
result["low"] > result[["open", "close", "high"]].min(axis=1)
|
||||
).any():
|
||||
raise ValueError("Invalid OHLC bounds")
|
||||
return result
|
||||
|
||||
|
||||
def _benchmark(
|
||||
observation: BenchmarkInput | None, calendar: pd.DatetimeIndex
|
||||
) -> tuple[str, pd.Series | None]:
|
||||
if observation is None:
|
||||
observation = BenchmarkInput()
|
||||
if type(observation) is not BenchmarkInput:
|
||||
raise ValueError("An explicit benchmark observation is required")
|
||||
status, closes = observation.status, observation.closes
|
||||
if status == "source_error":
|
||||
raise ValueError("Requested benchmark source failed")
|
||||
if status == "not_requested":
|
||||
if closes is not None:
|
||||
raise ValueError("Unrequested benchmark cannot contain prices")
|
||||
return status, None
|
||||
if status == "empty":
|
||||
if not isinstance(closes, pd.Series) or len(closes):
|
||||
raise ValueError("Empty benchmark must contain an empty series")
|
||||
return status, None
|
||||
if status != "present" or not isinstance(closes, pd.Series):
|
||||
raise ValueError("Invalid benchmark status or prices")
|
||||
index = _calendar(closes.index)
|
||||
if not index.equals(calendar):
|
||||
raise ValueError("Benchmark dates must exactly match the valuation calendar")
|
||||
prices = pd.Series(_prices(closes.tolist(), "benchmark"), index=calendar)
|
||||
with np.errstate(over="ignore", under="ignore", invalid="ignore", divide="ignore"):
|
||||
normalized = prices / prices.iloc[0]
|
||||
returns = prices.pct_change(fill_method=None)
|
||||
returns.iloc[0] = 0.0
|
||||
if (
|
||||
not np.isfinite(normalized.to_numpy()).all()
|
||||
or (normalized <= 0).any()
|
||||
or not np.isfinite(returns.to_numpy()).all()
|
||||
or (returns <= -1).any()
|
||||
):
|
||||
raise ValueError("Benchmark numeric range cannot represent positive prices and returns")
|
||||
return status, prices
|
||||
|
||||
|
||||
def _positive_mean(values: np.ndarray) -> float:
|
||||
scale = float(np.max(values))
|
||||
return scale * float(np.mean(values / scale)) if scale else 0.0
|
||||
|
||||
|
||||
def _bollinger_windows(
|
||||
close: pd.Series, period: int, multiple: float
|
||||
) -> tuple[np.ndarray, np.ndarray]:
|
||||
"""Scale each observed window, never a future/global price range.
|
||||
|
||||
Translate before scaling deviations so variance neither squares huge prices
|
||||
nor underflows tiny ones. The mean is scaled independently to avoid summation
|
||||
overflow and cancellation against a large translation anchor.
|
||||
"""
|
||||
values = close.to_numpy()
|
||||
middle, upper = np.full(len(values), np.nan), np.full(len(values), np.nan)
|
||||
for index in range(period - 1, len(values)):
|
||||
window = values[index - period + 1 : index + 1]
|
||||
mean = _positive_mean(window)
|
||||
deviations = window - window[0]
|
||||
spread = float(np.max(np.abs(deviations)))
|
||||
std = spread * float(np.std(deviations / spread, ddof=0)) if spread else 0.0
|
||||
if spread and std == 0:
|
||||
raise ValueError("Bollinger deviation exceeded numeric resolution")
|
||||
middle[index], upper[index] = mean, mean + multiple * std
|
||||
return middle, upper
|
||||
|
||||
|
||||
def _indicators(name: str, bars: pd.DataFrame, params: Parameters) -> dict[str, np.ndarray]:
|
||||
close, high, low = bars["close"], bars["high"], bars["low"]
|
||||
result: dict[str, np.ndarray] = {}
|
||||
if name in ("SmaCross", "MACross"):
|
||||
for label in ("fast", "slow"):
|
||||
result[label] = close.rolling(int(params[label])).mean().to_numpy()
|
||||
period = int(params.get("atr_period", 0))
|
||||
if period:
|
||||
# Arithmetic rolling ATR, matching the migrated strategy definition.
|
||||
tr = pd.concat(
|
||||
(high - low, (high - close.shift()).abs(), (low - close.shift()).abs()), axis=1
|
||||
).max(axis=1)
|
||||
result["atr"] = tr.rolling(period).mean().to_numpy()
|
||||
elif name == "RSI":
|
||||
period = int(params["period"])
|
||||
changes = np.diff(close.to_numpy())
|
||||
gains, losses = np.maximum(changes, 0), np.maximum(-changes, 0)
|
||||
gain, loss = _positive_mean(gains[:period]), _positive_mean(losses[:period])
|
||||
values = np.full(len(close), np.nan)
|
||||
for index in range(period, len(close)):
|
||||
if index > period:
|
||||
gain = gain * (1 - 1 / period) + float(gains[index - 1]) / period
|
||||
loss = loss * (1 - 1 / period) + float(losses[index - 1]) / period
|
||||
values[index] = (
|
||||
50 if gain == loss == 0 else 100 if loss == 0 else 100 - 100 / (1 + gain / loss)
|
||||
)
|
||||
result["rsi"] = values
|
||||
elif name == "BollingerBreakout":
|
||||
result["middle"], result["upper"] = _bollinger_windows(
|
||||
close, int(params["period"]), float(params["std_mult"])
|
||||
)
|
||||
elif name == "DualThrust":
|
||||
period = int(params["period"])
|
||||
# Preserve this example's documented HH-LL range, not another variant.
|
||||
width = high.shift().rolling(period).max() - low.shift().rolling(period).min()
|
||||
result["upper"] = (bars["open"] + float(params["k1"]) * width).to_numpy()
|
||||
result["lower"] = (bars["open"] - float(params["k2"]) * width).to_numpy()
|
||||
elif name == "TurtleBreakout":
|
||||
result["upper"] = high.shift().rolling(int(params["entry_period"])).max().to_numpy()
|
||||
result["lower"] = low.shift().rolling(int(params["exit_period"])).min().to_numpy()
|
||||
for label, values in result.items():
|
||||
if name in ("SmaCross", "MACross"):
|
||||
warmup = int(params["atr_period"] if label == "atr" else params[label]) - 1
|
||||
elif name == "TurtleBreakout":
|
||||
warmup = int(params["entry_period"] if label == "upper" else params["exit_period"])
|
||||
else:
|
||||
warmup = int(params["period"]) - (1 if name == "BollingerBreakout" else 0)
|
||||
if not np.isfinite(values[warmup:]).all():
|
||||
raise ValueError("Strategy indicator exceeded numeric range")
|
||||
return result
|
||||
|
||||
|
||||
class _Policy:
|
||||
def __init__(self, name: str, bars: pd.DataFrame, params: Parameters, asset: str):
|
||||
self.name, self.bars, self.params, self.asset = name, bars, params, asset
|
||||
self.dates = bars.index.strftime("%Y-%m-%d").tolist()
|
||||
self.indicators = _indicators(name, bars, params)
|
||||
self.index = 0
|
||||
self.was_held = False
|
||||
self.peak = 0.0
|
||||
self.stop: float | None = None
|
||||
self.signals: list[StrategySignal] = []
|
||||
|
||||
def __call__(self, position: DailyPosition) -> dict[str, float] | None:
|
||||
index = self.index
|
||||
self.index += 1
|
||||
held = position.holdings.get(self.asset, 0) > 0
|
||||
if held and not self.was_held:
|
||||
self.peak = float(self.bars["open"].iloc[index])
|
||||
self.stop = None
|
||||
if not held:
|
||||
self.peak, self.stop = 0.0, None
|
||||
self.was_held = held
|
||||
weight, reason = self._decide(index, held)
|
||||
if weight is None:
|
||||
return None
|
||||
next_date = self.dates[index + 1] if index + 1 < len(self.dates) else None
|
||||
self.signals.append(
|
||||
StrategySignal(
|
||||
position.date,
|
||||
next_date,
|
||||
weight,
|
||||
reason,
|
||||
"pending" if next_date else "no_next_session",
|
||||
)
|
||||
)
|
||||
return {self.asset: weight}
|
||||
|
||||
def _decide(self, index: int, held: bool) -> tuple[float | None, str]:
|
||||
name, p, values = self.name, self.params, self.indicators
|
||||
close = float(self.bars["close"].iloc[index])
|
||||
if name == "BuyAndHold":
|
||||
return (float(p["buy_pct"]), "initial_allocation") if index == 0 else (None, "")
|
||||
if name in ("SmaCross", "MACross"):
|
||||
start = int(p["slow"])
|
||||
elif name == "TurtleBreakout":
|
||||
start = int(p["exit_period"] if held else p["entry_period"])
|
||||
else:
|
||||
start = int(p["period"])
|
||||
if index < start:
|
||||
return None, ""
|
||||
enter, leave, reason = False, False, "signal_exit"
|
||||
if name in ("SmaCross", "MACross"):
|
||||
fast, slow = values["fast"], values["slow"]
|
||||
enter = bool(fast[index - 1] <= slow[index - 1] and fast[index] > slow[index])
|
||||
leave = bool(fast[index - 1] >= slow[index - 1] and fast[index] < slow[index])
|
||||
reason = "ma_cross_down"
|
||||
if name == "MACross" and held and "atr" in values:
|
||||
prior_atr = float(values["atr"][index - 1])
|
||||
if math.isfinite(prior_atr):
|
||||
historical_stop = self.peak - float(p["atr_mult"]) * prior_atr
|
||||
self.stop = (
|
||||
historical_stop if self.stop is None else max(self.stop, historical_stop)
|
||||
)
|
||||
if close <= self.stop:
|
||||
leave, reason = True, "atr_stop"
|
||||
self.peak = max(self.peak, close)
|
||||
elif name == "RSI":
|
||||
enter, leave = (
|
||||
values["rsi"][index] < p["oversold"],
|
||||
values["rsi"][index] > p["overbought"],
|
||||
)
|
||||
else:
|
||||
enter = close > values["upper"][index]
|
||||
lower = values["middle"] if name == "BollingerBreakout" else values["lower"]
|
||||
leave = close < lower[index]
|
||||
if held and leave:
|
||||
return 0.0, reason
|
||||
if not held and enter:
|
||||
return 1.0, "signal_entry"
|
||||
return None, ""
|
||||
|
||||
|
||||
def _metrics(
|
||||
ledger: ExecutionSimulationResult, pairing: TradePairing
|
||||
) -> tuple[dict[str, float | None], dict[str, str]]:
|
||||
returns = ledger.daily_returns
|
||||
if (
|
||||
not np.isfinite(returns.to_numpy()).all()
|
||||
or ((ledger.nav_series > 0) & (returns <= -1)).any()
|
||||
):
|
||||
raise ValueError("Portfolio returns must be finite and within representable numeric range")
|
||||
with np.errstate(over="ignore", invalid="ignore", divide="ignore"):
|
||||
values = dict(summary(returns))
|
||||
values["daily_win_rate"] = values.pop("win_rate")
|
||||
values["total_return"] = ledger.final_portfolio_value / ledger.initial_cash - 1
|
||||
metrics: dict[str, float | None] = {
|
||||
key: float(value) if math.isfinite(value) else None for key, value in values.items()
|
||||
}
|
||||
unavailable = {key: "numeric_range" for key, value in metrics.items() if value is None}
|
||||
if values["ann_volatility"] == 0:
|
||||
metrics["sharpe"] = None
|
||||
unavailable["sharpe"] = "zero_volatility"
|
||||
if values["max_drawdown"] == 0:
|
||||
metrics["calmar"] = None
|
||||
unavailable["calmar"] = "zero_drawdown"
|
||||
if not (returns < 0).any():
|
||||
metrics["sortino"] = None
|
||||
unavailable["sortino"] = "no_downside_deviation"
|
||||
metrics["trade_win_rate"] = pairing.win_rate
|
||||
metrics["closed_trades"] = float(len(pairing.closed_lots))
|
||||
metrics["realized_net_pnl"] = pairing.realized_net_pnl
|
||||
if pairing.win_rate is None:
|
||||
unavailable["trade_win_rate"] = "no_closed_lots"
|
||||
return metrics, unavailable
|
||||
|
||||
|
||||
def run_strategy_research(
|
||||
strategy: str,
|
||||
bars: pd.DataFrame,
|
||||
*,
|
||||
asset: str,
|
||||
params: Parameters | None = None,
|
||||
initial_cash: float = 1_000_000,
|
||||
commission: float = 0.0003,
|
||||
stamp_duty: float = 0.001,
|
||||
benchmark: BenchmarkInput | None = None,
|
||||
min_trade_amount: float = 0,
|
||||
) -> StrategyResearchResult:
|
||||
"""Pure candidate research. Fees are explicit fractions, not current tax claims."""
|
||||
parameters = strategy_parameters(strategy, params)
|
||||
validate_money(initial_cash, commission, stamp_duty)
|
||||
finite_number(min_trade_amount, "min_trade_amount")
|
||||
if min_trade_amount < 0:
|
||||
raise ValueError("min_trade_amount must be nonnegative")
|
||||
if (
|
||||
type(asset) is not str
|
||||
or not asset
|
||||
or asset == "CASH"
|
||||
or asset.strip() != asset
|
||||
or len(asset) > 100
|
||||
):
|
||||
raise ValueError("An explicit bounded asset key is required")
|
||||
observed = validate_strategy_bars(bars)
|
||||
if len(observed) < required_history(strategy, parameters):
|
||||
raise ValueError("Insufficient strategy history")
|
||||
benchmark_status, benchmark_close = _benchmark(benchmark, observed.index)
|
||||
policy = _Policy(strategy, observed, parameters, asset)
|
||||
dates = policy.dates
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[],
|
||||
list(zip(dates, ({asset: price} for price in observed["open"]), strict=True)),
|
||||
list(zip(dates, ({asset: price} for price in observed["close"]), strict=True)),
|
||||
initial_cash,
|
||||
ExecutionConfig(
|
||||
commission_bps=commission * 10000,
|
||||
stamp_tax_bps=stamp_duty * 10000,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=min_trade_amount,
|
||||
),
|
||||
decision_policy=policy,
|
||||
)
|
||||
if not np.isfinite(ledger.nav_series.to_numpy()).all() or (ledger.nav_series < 0).any():
|
||||
raise ValueError("Ledger NAV must remain nonnegative and finite")
|
||||
days = {day.date: day for day in ledger.daily_executions}
|
||||
positions = {position.date: position for position in ledger.positions}
|
||||
signals = []
|
||||
for signal in policy.signals:
|
||||
status = signal.status
|
||||
if signal.execution_date is not None:
|
||||
fills = [fill for fill in days[signal.execution_date].executions if fill.quantity > 0]
|
||||
if fills:
|
||||
status = (
|
||||
"partial_fill" if any(fill.partial_fill_pct < 1 for fill in fills) else "filled"
|
||||
)
|
||||
else:
|
||||
flat_target_met = (
|
||||
signal.target_weight == 0
|
||||
and positions[signal.execution_date].holdings.get(asset, 0) == 0
|
||||
)
|
||||
status = "no_change" if flat_target_met else "not_filled"
|
||||
signals.append(
|
||||
StrategySignal(
|
||||
signal.decision_date,
|
||||
signal.execution_date,
|
||||
signal.target_weight,
|
||||
signal.reason,
|
||||
status,
|
||||
)
|
||||
)
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
metrics, unavailable = _metrics(ledger, pairing)
|
||||
benchmark_nav, benchmark_metrics, benchmark_returns = None, None, None
|
||||
if benchmark_close is not None:
|
||||
benchmark_nav = benchmark_close / benchmark_close.iloc[0]
|
||||
benchmark_returns = benchmark_close.pct_change(fill_method=None)
|
||||
benchmark_returns.iloc[0] = 0.0
|
||||
benchmark_returns.index = ledger.daily_returns.index
|
||||
with np.errstate(over="ignore", invalid="ignore", divide="ignore"):
|
||||
relative = benchmark_summary(ledger.daily_returns, benchmark_returns)
|
||||
benchmark_metrics = {
|
||||
key: float(value) if math.isfinite(value) else None for key, value in relative.items()
|
||||
}
|
||||
benchmark_metrics["total_return"] = float(benchmark_nav.iloc[-1] - 1)
|
||||
return StrategyResearchResult(
|
||||
strategy,
|
||||
dict(parameters),
|
||||
ledger,
|
||||
tuple(signals),
|
||||
pairing,
|
||||
metrics,
|
||||
unavailable,
|
||||
benchmark_status,
|
||||
benchmark_nav,
|
||||
benchmark_metrics,
|
||||
asset,
|
||||
observed.copy(deep=True),
|
||||
benchmark_returns,
|
||||
{
|
||||
"initial_cash": initial_cash,
|
||||
"commission": commission,
|
||||
"stamp_duty": stamp_duty,
|
||||
"min_trade_amount": min_trade_amount,
|
||||
"slippage_bps": 0,
|
||||
},
|
||||
)
|
||||
@@ -1,170 +0,0 @@
|
||||
"""FIFO long-only lot pairing from actual fills and their net cash flows.
|
||||
|
||||
An entry lot is one trade for win-rate purposes, and counts only once fully closed.
|
||||
Partial exits contribute realized PnL but do not count as completed trades. Costs
|
||||
include entry/exit fees and slippage exactly once through execution cash flows.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from collections import defaultdict, deque
|
||||
from dataclasses import dataclass
|
||||
|
||||
from .execution import ExecutionSimulationResult
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TradeMatch:
|
||||
asset: str
|
||||
buy_date: str
|
||||
sell_date: str
|
||||
quantity: float
|
||||
cost: float
|
||||
net_proceeds: float
|
||||
net_pnl: float
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ClosedLot:
|
||||
asset: str
|
||||
buy_date: str
|
||||
sell_date: str
|
||||
quantity: float
|
||||
cost: float
|
||||
net_proceeds: float
|
||||
net_pnl: float
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class OpenLot:
|
||||
asset: str
|
||||
buy_date: str
|
||||
quantity: float
|
||||
remaining_cost: float
|
||||
realized_net_pnl: float
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TradePairing:
|
||||
matches: tuple[TradeMatch, ...]
|
||||
closed_lots: tuple[ClosedLot, ...]
|
||||
open_lots: tuple[OpenLot, ...]
|
||||
realized_net_pnl: float
|
||||
win_rate: float | None
|
||||
|
||||
|
||||
@dataclass
|
||||
class _Lot:
|
||||
asset: str
|
||||
date: str
|
||||
quantity: float
|
||||
remaining: float
|
||||
cost: float
|
||||
proceeds: float = 0.0
|
||||
matched_cost: float = 0.0
|
||||
|
||||
|
||||
def pair_ledger_trades(ledger: ExecutionSimulationResult) -> TradePairing:
|
||||
lots: dict[str, deque[_Lot]] = defaultdict(deque)
|
||||
matches: list[TradeMatch] = []
|
||||
closed: list[ClosedLot] = []
|
||||
positions = {position.date: position for position in ledger.positions}
|
||||
if len(positions) != len(ledger.positions):
|
||||
raise ValueError("Duplicate ledger position dates")
|
||||
for day in ledger.daily_executions:
|
||||
if day.date not in positions:
|
||||
raise ValueError("Execution date has no ledger position")
|
||||
last_fill = {
|
||||
fill.stock_code: index
|
||||
for index, fill in enumerate(day.executions)
|
||||
if fill.quantity > 0
|
||||
}
|
||||
for index, fill in enumerate(day.executions):
|
||||
quantity, cash = fill.quantity, fill.net_cash_flow
|
||||
if not math.isfinite(quantity) or quantity < 0 or not math.isfinite(cash):
|
||||
raise ValueError("Invalid fill quantity or cash flow")
|
||||
if quantity == 0:
|
||||
if cash != 0:
|
||||
raise ValueError("Unfilled order cannot move cash")
|
||||
continue
|
||||
if fill.side == "buy":
|
||||
if cash >= 0:
|
||||
raise ValueError("Buy cash flow must be negative")
|
||||
lots[fill.stock_code].append(
|
||||
_Lot(fill.stock_code, day.date, quantity, quantity, -cash)
|
||||
)
|
||||
continue
|
||||
if fill.side != "sell" or cash < 0:
|
||||
raise ValueError("Invalid long-only sell fill")
|
||||
remaining = quantity
|
||||
queue = lots[fill.stock_code]
|
||||
available = math.fsum(lot.remaining for lot in queue)
|
||||
quantity_resolution = 8 * math.fsum(
|
||||
[math.ulp(quantity), math.ulp(available)]
|
||||
+ [math.ulp(lot.remaining) for lot in queue]
|
||||
)
|
||||
if remaining - available > quantity_resolution:
|
||||
raise ValueError("Sell quantity exceeds FIFO inventory")
|
||||
# A final fill followed by an actual flat ledger position proves a
|
||||
# full exit. Reconcile only accumulated floating-point rounding here;
|
||||
# any real positive holding keeps the exact partial-lot behavior.
|
||||
full_exit = (
|
||||
index == last_fill[fill.stock_code]
|
||||
and positions[day.date].holdings.get(fill.stock_code, 0.0) == 0
|
||||
)
|
||||
if full_exit and abs(quantity - available) > quantity_resolution:
|
||||
raise ValueError("Flat ledger position conflicts with FIFO inventory")
|
||||
allocated_proceeds: list[float] = []
|
||||
while queue and (remaining > 0 or full_exit):
|
||||
lot = queue[0]
|
||||
take = lot.remaining if full_exit else min(remaining, lot.remaining)
|
||||
full_lot = take == lot.remaining
|
||||
final_match = len(queue) == 1 if full_exit else take == remaining
|
||||
cost = lot.cost - lot.matched_cost if full_lot else lot.cost * (take / lot.quantity)
|
||||
proceeds = (
|
||||
cash - math.fsum(allocated_proceeds)
|
||||
if final_match
|
||||
else cash * (take / quantity)
|
||||
)
|
||||
allocated_proceeds.append(proceeds)
|
||||
matches.append(
|
||||
TradeMatch(lot.asset, lot.date, day.date, take, cost, proceeds, proceeds - cost)
|
||||
)
|
||||
lot.proceeds += proceeds
|
||||
lot.matched_cost += cost
|
||||
lot.remaining -= take
|
||||
remaining -= take
|
||||
if full_lot:
|
||||
closed.append(
|
||||
ClosedLot(
|
||||
lot.asset,
|
||||
lot.date,
|
||||
day.date,
|
||||
lot.quantity,
|
||||
lot.cost,
|
||||
lot.proceeds,
|
||||
lot.proceeds - lot.cost,
|
||||
)
|
||||
)
|
||||
lots[fill.stock_code].popleft()
|
||||
opened = tuple(
|
||||
OpenLot(
|
||||
lot.asset,
|
||||
lot.date,
|
||||
lot.remaining,
|
||||
lot.cost - lot.matched_cost,
|
||||
lot.proceeds - lot.matched_cost,
|
||||
)
|
||||
for queue in lots.values()
|
||||
for lot in queue
|
||||
)
|
||||
# A representational residual is not a winning trade; raw PnL is retained.
|
||||
wins = sum(lot.net_pnl > 32 * math.ulp(max(lot.cost, lot.net_proceeds, 1.0)) for lot in closed)
|
||||
return TradePairing(
|
||||
tuple(matches),
|
||||
tuple(closed),
|
||||
opened,
|
||||
math.fsum(match.net_pnl for match in matches),
|
||||
wins / len(closed) if closed else None,
|
||||
)
|
||||
-17
@@ -1,17 +0,0 @@
|
||||
{
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"replay_spec_digest": "sha256:20f07fcb526bc38b4ba3d71d6c3b00a8c63ad96cab0fecd869326503ab42d98b",
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117",
|
||||
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"table_content_digests": {
|
||||
"run": "sha256:7d947ec93f714641669cbb14bd69dd8cf30387aabb7f3a086918fc2e878ab05c",
|
||||
"signals": "sha256:72dc15064cc45d7c51d2dd4b8c3a6d8c7d4155232d5d70d3c9e7697fab70ce50",
|
||||
"trades": "sha256:2b8b9321e7993941ac486cf50cab5b4c6425b2571a4701f0eb02273ec9a96c53",
|
||||
"positions": "sha256:b456a48fab51742b05084ca6dcaf01c03dfe5b39ad71215ada070e9b1f59f7ea",
|
||||
"nav": "sha256:25649efce860b76410f87dbd36c81085887620086dc0b48b07099918f65d9c78",
|
||||
"performance": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
|
||||
"attribution": "sha256:ad0f12f668d0a2d9ae5b3636d29989ab4bafcfb95f5de77abeb52a6d9e95d366",
|
||||
"attribution_daily": "sha256:d4459ad650f88871d7b1e40392033037b5818ef92fdf1ee021868929fc1455a5",
|
||||
"risk": "sha256:4816dd4812b5ff2e97e74bf34ca2221bfde675b387a96e277683557f0ad7d975"
|
||||
}
|
||||
}
|
||||
-206
@@ -1,206 +0,0 @@
|
||||
{
|
||||
"dataset_snapshot": {
|
||||
"contract_name": "researchhub.dataset-snapshot",
|
||||
"schema_version": "1.0.0",
|
||||
"snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"descriptor": {
|
||||
"dataset": {
|
||||
"dataset_id": "rhdataset:market:0123456789abcdef0123456789abcdef",
|
||||
"dataset_kind": "market",
|
||||
"record_schema_version": "1.0.0",
|
||||
"dimensions": ["instrument_id", "effective_time"]
|
||||
},
|
||||
"published_at": "2026-01-02T07:05:00Z",
|
||||
"time_semantics": {
|
||||
"effective_time": {
|
||||
"start_inclusive": "2026-01-02T07:00:00Z",
|
||||
"end_inclusive": "2026-01-02T07:00:00Z"
|
||||
},
|
||||
"knowledge_time": {
|
||||
"start_inclusive": "2026-01-02T07:01:00Z",
|
||||
"end_inclusive": "2026-01-02T07:01:00Z"
|
||||
},
|
||||
"pit_cutoff": "2026-01-02T07:01:00Z"
|
||||
},
|
||||
"content": {
|
||||
"digest_algorithm": "sha256",
|
||||
"canonicalization": "RFC8785",
|
||||
"record_order": "canonical-record-byte-order",
|
||||
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"logical_manifest": {
|
||||
"record_count": 2,
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"record_count": 2
|
||||
}
|
||||
]
|
||||
},
|
||||
"manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"record_count": 2
|
||||
},
|
||||
"lineage": {
|
||||
"publisher": {"id": "researchhub.data", "version": "1.0.0"},
|
||||
"transformation": {
|
||||
"id": "rhtransform:00112233445566778899aabbccddeeff",
|
||||
"version": "1.0.0"
|
||||
},
|
||||
"upstream_snapshot_ids": [],
|
||||
"upstream_content_digests": []
|
||||
},
|
||||
"quality": {
|
||||
"status": "passed",
|
||||
"checks": [
|
||||
{
|
||||
"check_id": "completeness",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:876fc2fcc6414ddc3f824a47f475d34c82a53d2bda5dc72a234a3f3164e8e2ec"
|
||||
},
|
||||
{
|
||||
"check_id": "pit_time_integrity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:90a6cc46b9f2ab317a1c6d14dc173784e19b5621cd7fe956e8f41338cbdc5944"
|
||||
}
|
||||
]
|
||||
},
|
||||
"qualification": {
|
||||
"status": "qualified",
|
||||
"policy_id": "researchhub.dataset-snapshot.pit",
|
||||
"policy_version": "1.0.0",
|
||||
"evaluated_at": "2026-01-02T07:04:00Z",
|
||||
"evidence_digest": "sha256:e192462f9022f2b477f73cdbe9e6c9f891ebcfdc2b4ed4f8ddd7b1ff107ee6a6"
|
||||
}
|
||||
}
|
||||
},
|
||||
"data_foundation": {
|
||||
"contract_name": "researchhub.data-foundation",
|
||||
"schema_version": "1.0.0",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"pit_cutoff": "2026-01-03T00:00:00Z",
|
||||
"instrument_routes": [
|
||||
{
|
||||
"route_revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"revision_number": 1,
|
||||
"symbol": "600000",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:11112222333344445555666677778888",
|
||||
"effective_from": "2020-01-01T00:00:00Z",
|
||||
"knowledge_time": "2026-01-01T07:00:00Z",
|
||||
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
}
|
||||
],
|
||||
"trading_calendar_revisions": [
|
||||
{
|
||||
"calendar_revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
|
||||
"calendar_id": "rhcalendar:11112222333344445555666677778888",
|
||||
"session_date": "2026-01-02",
|
||||
"revision_number": 1,
|
||||
"status": "open",
|
||||
"sessions": [
|
||||
{"opens_at": "2026-01-02T01:30:00Z", "closes_at": "2026-01-02T07:00:00Z"}
|
||||
],
|
||||
"knowledge_time": "2026-01-01T08:00:00Z",
|
||||
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
}
|
||||
],
|
||||
"corporate_action_revisions": [
|
||||
{
|
||||
"action_revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
|
||||
"action_id": "rhaction:99998888777766665555444433332222",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"revision_number": 1,
|
||||
"action_type": "cash_dividend",
|
||||
"status": "confirmed",
|
||||
"effective_time": "2026-01-02T00:00:00Z",
|
||||
"knowledge_time": "2026-01-01T09:00:00Z",
|
||||
"terms_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
|
||||
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
|
||||
}
|
||||
],
|
||||
"standardized_views": [
|
||||
{
|
||||
"view_ref_id": "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c",
|
||||
"view_id": "rhview:abcdef0123456789abcdef0123456789",
|
||||
"view_version": "1.0.0",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"pit_cutoff": "2026-01-03T00:00:00Z",
|
||||
"schema_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
|
||||
"content_digest": "sha256:123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef0",
|
||||
"transformation_digest": "sha256:23456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef01",
|
||||
"instrument_route_revision_ids": [
|
||||
"rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86"
|
||||
],
|
||||
"trading_calendar_revision_ids": [
|
||||
"rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
],
|
||||
"corporate_action_revision_ids": [
|
||||
"rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
]
|
||||
}
|
||||
],
|
||||
"revision_lineage": [
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T07:00:00Z",
|
||||
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
},
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T08:00:00Z",
|
||||
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
},
|
||||
{
|
||||
"revision_kind": "corporate_action",
|
||||
"revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T09:00:00Z",
|
||||
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
|
||||
}
|
||||
],
|
||||
"readiness": {
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"contract_validation": {
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee"
|
||||
]
|
||||
},
|
||||
"real_data_validation": {"status": "not_validated", "evidence_digests": []},
|
||||
"production_validation": {"status": "not_validated", "evidence_digests": []},
|
||||
"live_validation": {"status": "not_validated", "evidence_digests": []}
|
||||
}
|
||||
},
|
||||
"output_schema": {
|
||||
"columns": ["evaluation_at", "factor_id", "instrument_id", "value"],
|
||||
"schema_version": "1.0.0"
|
||||
},
|
||||
"output_content": {
|
||||
"rows": [
|
||||
{
|
||||
"evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"factor_id": "alpha_005",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"value": "0.125"
|
||||
}
|
||||
]
|
||||
},
|
||||
"expected": {
|
||||
"definition_id": "rhfactorv1:sha256:978fb8000d318373844a5e044ca14bf377e01ebe8d85964b826ecd2af9085ce9",
|
||||
"input_schema_digest": "sha256:4501aeab99b4bcc25a1b8813ebe197fb498053fd710d73746bf20fc8eeb4bfa7",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"output_artifact_id": "rhfactoroutputv1:sha256:a4803b5ff66d12d3a0e7e8e5b8cca953cbc137e2bf41514b7c4e5f05da5ee68b",
|
||||
"legacy_binding_id": "rhlegacyfactorv1:sha256:541bc5a9469f9c8e4c2d696a9972fc5f2e6e2bef218b86f728823994b915dede"
|
||||
}
|
||||
}
|
||||
-871
@@ -1,871 +0,0 @@
|
||||
{
|
||||
"cases": {
|
||||
"absent": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:b7e9139e021de3122bee9376a8c8387fca3db75fe2a698adccf33471caf971d2",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:fae26a93d754e98f437bf9e4b635fc1cdc4f85d004a4bde396823b52e2115de5",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:3f67fcad23c75684ffeb325d8405b2f81139a732e237d30fbb30d23f91e32726",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "none",
|
||||
"benchmark_id": "",
|
||||
"benchmark_series_digest": null,
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:729852af307fd4a94ac566ae45df2e57b3d45c4fbdf45820351b5ef19cdf0ad3",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "none",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "benchmark_absent",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:d735abe76c28db429e107cb5ba6a1929c2c0f2084a6c4ac41b38ffbc02b2d2bd",
|
||||
"performance_row_digest": "sha256:4e6329b0db4731513fe793491a163dd358e3a6f7baa84d3c94a4d4d2355c7a7d",
|
||||
"performance_table_content_digest": "sha256:074b8511ff9a2154235e4dde8bac300658cb69f900e49f332d61560b82be9af5",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:ad85ee1317bd8ce5fbef8fddc268b91be8678701ba5b8fbf6d61bf670f918bf9",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
},
|
||||
"estimable": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:e2e0e63fb4c733d2dba9a290511b6c8b0132bfe877dfe89ee7e614b74b87bc51",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "exact_session_index",
|
||||
"benchmark_id": "000300.SH",
|
||||
"benchmark_series_digest": "sha256:1c564e4d12af0d59b1fd707d7816da2895238969d9207470984db63cfeac423a",
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:a6a1a3313f511505c0cf57fddeebdf20a3b0f83a2958f31332d5597d172165e7",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "exact_session_index",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.3032171729991897
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": 22.801264912443322
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 123663320625.66454
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": 3.2500000000000013
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:f1f72c34d7427bd0d2e273a66b9677d112aa51c2417ea1cda8245e51e334e4f6",
|
||||
"performance_row_digest": "sha256:bad2b506679a6ca13d80ab00859a442ba692f3872d42b0b45bbc3b1d6ffe72a4",
|
||||
"performance_table_content_digest": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:6ec566d73da3966d3d5f946313e2b0181993374c9f4ceb4b56e165cc6793d382",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
},
|
||||
"zero_active_variance": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:8a13212575154d97f85011ed54ed3bb588f73a691926b3ca03b360d563e9fc29",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:8c0c2052ab93aa920254bc7f4d89435d839b7150ecaaaa798703e89ddf876ec1",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:699ad35ee30a294082c607c40617097aa90d5d58343e7d8c8185d8a53f52596c",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "exact_session_index",
|
||||
"benchmark_id": "000300.SH",
|
||||
"benchmark_series_digest": "sha256:7668b77ac689cd542005fae42cb2abd48fdf772d2f982086dca409606653e6ac",
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:44a04be0105f29e1cbf8d2430b6b0eb084853d428bc2ccd5834dfdbcf9482bd7",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "exact_session_index",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "not_estimable_active_variance",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": 1.0000000000000002
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:e6730d25d90c85a6370adb72bc45c1bf505235f77b72521dd8197898ca369fe5",
|
||||
"performance_row_digest": "sha256:6528c47fa9416e350aa5010477dbd8c83a35cecf87aa5bc413cec95a9765114d",
|
||||
"performance_table_content_digest": "sha256:7d57a966431c68093a9f6193ae62f79a63d33832f11dd68524ed9bcd2cb6257c",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:b17ab9e158a7ceeb5107f6fe8a61f332009c314186f3436d4750aa44d6ca3856",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
},
|
||||
"zero_benchmark_variance": {
|
||||
"artifact_available_at": "2026-01-08T02:05:00Z",
|
||||
"authority": "quant_engine",
|
||||
"backtest_evidence_manifest_document_sha256": "sha256:2389b804f5d8d20a37b31f596b52892484777c5da799dd68865a082ee90e6e5b",
|
||||
"backtest_evidence_manifest_evidence_digest": "sha256:54bb315bc1940ef6df79999cf03ebb8d10defaf526bf7802d63da33f612520ec",
|
||||
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:ad9559e1f8feca28bb310765216a975935f05f51057139c2d02fd2fe7dfe501e",
|
||||
"backtest_evidence_qualification": "contract_qualified",
|
||||
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"benchmark_alignment_policy": "exact_session_index",
|
||||
"benchmark_id": "000300.SH",
|
||||
"benchmark_series_digest": "sha256:d0f1e954d796e95254bcccc893d6e3a8f9d541c3028b88b2c0b24c1658bb2408",
|
||||
"calendar": "CN-A",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
|
||||
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
|
||||
"cost_model_version": "1.0.0",
|
||||
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"document_sha256": "sha256:077f3ec84802aff2f046365a3a5f61f475b1ad63ec05e96742844bc531bcb551",
|
||||
"end_date": "2026-01-08",
|
||||
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
|
||||
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
|
||||
"execution_model_version": "1.0.0",
|
||||
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
|
||||
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"frequency": "1d",
|
||||
"methodology": {
|
||||
"alpha": "daily_ols_intercept_geometric_annualization",
|
||||
"annual_risk_free": 0.0,
|
||||
"annualized_return": "geometric_compound",
|
||||
"annualized_volatility": "sample_std_sqrt_periods",
|
||||
"benchmark_alignment": "exact_session_index",
|
||||
"benchmark_risk_free_daily": 0.0,
|
||||
"beta": "sample_covariance_over_sample_variance",
|
||||
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
|
||||
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
|
||||
"implementation_module": "quant_engine.metrics",
|
||||
"implementation_version": "researchhub.quant-performance-methodology.v1",
|
||||
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
|
||||
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"periods_per_year": 252,
|
||||
"return_type": "simple",
|
||||
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
|
||||
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
|
||||
"source_frequency": "1d",
|
||||
"total_return": "final_nav_minus_one",
|
||||
"tracking_error": "sample_std_active_return_sqrt_periods",
|
||||
"win_rate": "positive_daily_return_count_over_observation_count"
|
||||
},
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"metrics": [
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "total_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "total_ret",
|
||||
"unit": "ratio",
|
||||
"value": 0.575
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_return",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_ret",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 2683336646708.1
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "annualized_volatility",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "ann_volatility",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sharpe_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sharpe",
|
||||
"unit": "ratio",
|
||||
"value": 1931820803008.3313
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "sortino_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "sortino",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "maximum_drawdown",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "max_dd",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "calmar_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "calmar",
|
||||
"unit": "ratio",
|
||||
"value": 0.0
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "win_rate",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "win_rate",
|
||||
"unit": "ratio",
|
||||
"value": 0.75
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "tracking_error",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "tracking_error",
|
||||
"unit": "ratio_per_year",
|
||||
"value": 1.38901943830891
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "information_ratio",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "ir",
|
||||
"unit": "ratio",
|
||||
"value": 22.299903907544408
|
||||
},
|
||||
{
|
||||
"availability": "not_estimable_benchmark_variance",
|
||||
"key": "alpha",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "alpha",
|
||||
"unit": "ratio_per_year",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "not_estimable_benchmark_variance",
|
||||
"key": "beta",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": true,
|
||||
"source_column": "beta",
|
||||
"unit": "ratio",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "trade_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_trades",
|
||||
"unit": "count",
|
||||
"value": 3
|
||||
},
|
||||
{
|
||||
"availability": "available",
|
||||
"key": "day_count",
|
||||
"methodology_id": "researchhub.quant-performance-methodology.v1",
|
||||
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
|
||||
"nullable": false,
|
||||
"source_column": "n_days",
|
||||
"unit": "count",
|
||||
"value": 4
|
||||
}
|
||||
],
|
||||
"performance_evidence_id": "rhperformanceevidencev1:sha256:86133a6b3c3091477cdc4618ebc294e671c9ba0a3ef5015a068f92bb575729b9",
|
||||
"performance_row_digest": "sha256:90c9aa2f4168c15ff9ac5d4da2a35c3e915bb4e04736e5a440ea80c787a99553",
|
||||
"performance_table_content_digest": "sha256:1570b64d83ad2a849607e6f5203f5ec2f0291a8d6cbb00b8edfe0ae030b7967d",
|
||||
"performance_table_logical_name": "performance",
|
||||
"performance_table_row_count": 1,
|
||||
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
|
||||
"research_artifact_content_digest": "sha256:fed7a28888a6e98ccce78e7d84f22d4e3636e928485861af7a97f90eddc91f94",
|
||||
"research_artifact_schema_version": "1.1.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"schema_version": "researchhub.performance-evidence.v1",
|
||||
"scope": "offline_research_only",
|
||||
"start_date": "2026-01-05",
|
||||
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"timezone": "Asia/Shanghai"
|
||||
}
|
||||
},
|
||||
"schema_version": 1,
|
||||
"source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c",
|
||||
"source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603"
|
||||
}
|
||||
@@ -1,129 +0,0 @@
|
||||
{
|
||||
"portfolio_decision": {
|
||||
"computed_at": "2026-01-08T03:01:00Z",
|
||||
"constraint_residuals": {
|
||||
"gross_exposure_max": 0.0,
|
||||
"net_exposure_max": 0.0,
|
||||
"net_exposure_min": 0.0,
|
||||
"position_count_max": 0.0,
|
||||
"single_asset_max": 0.0,
|
||||
"single_asset_min": 0.0,
|
||||
"turnover_max": 0.0
|
||||
},
|
||||
"constraints": {
|
||||
"gross_exposure_max": 1.0,
|
||||
"net_exposure_max": 1.0,
|
||||
"net_exposure_min": 1.0,
|
||||
"position_count_max": 2,
|
||||
"schema_version": "1.0.0",
|
||||
"single_asset_max": 0.7,
|
||||
"single_asset_min": 0.2,
|
||||
"turnover_max": 0.2
|
||||
},
|
||||
"contract_name": "researchhub.portfolio-decision",
|
||||
"covariance_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
|
||||
"effective_at": "2026-01-08T03:00:00Z",
|
||||
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"expected_return_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd",
|
||||
"freshness_policy": {
|
||||
"max_covariance_age_days": 0,
|
||||
"max_manifest_age_seconds": 3600,
|
||||
"schema_version": "1.0.0"
|
||||
},
|
||||
"gross_exposure": 1.0,
|
||||
"manifest_document_sha256": "fbf54218f770528978f9ccd35577e1ab00877143ea397a576e071e75f0afbab0",
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
|
||||
"model_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
|
||||
"model_name": "deterministic_weights",
|
||||
"model_version": "1.0.0",
|
||||
"net_exposure": 1.0,
|
||||
"objective_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
|
||||
"objective_name": "long_only_allocation",
|
||||
"objective_version": "1.0.0",
|
||||
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
|
||||
"portfolio_asset_set_digest": "sha256:b64e3448a83a5b86466465080361c1a7e1157a27ddccd4b68069cb18caffb74a",
|
||||
"position_count": 2,
|
||||
"prior_weights": {
|
||||
"A": 0.5,
|
||||
"B": 0.5
|
||||
},
|
||||
"receipt": {
|
||||
"algorithm": "bounded_allocation",
|
||||
"algorithm_version": "1.0.0",
|
||||
"computed_at": "2026-01-08T03:01:00Z",
|
||||
"constraint_digest": "sha256:34df0e5c00f503748ff936f9cd415a8947169181f8ca0917bce97dd606e08e94",
|
||||
"implementation_digest": "sha256:ffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffff",
|
||||
"input_digest": "sha256:ebd8ff957115e1adfd84eafa5cea49470356194d747b64ee5897858f9dd067b7",
|
||||
"iterations": null,
|
||||
"max_constraint_residual": 0.0,
|
||||
"objective_value": null,
|
||||
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
|
||||
"parameter_digest": "sha256:0000000000000000000000000000000000000000000000000000000000000000",
|
||||
"schema_version": "1.0.0",
|
||||
"solver_config_digest": null,
|
||||
"solver_name": null,
|
||||
"solver_required": false,
|
||||
"solver_version": null,
|
||||
"status": "completed",
|
||||
"tolerance": 1e-12
|
||||
},
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"run_ref_document_sha256": "6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
|
||||
"schema_version": "1.0.0",
|
||||
"source_universe_digest": "sha256:5555555555555555555555555555555555555555555555555555555555555555",
|
||||
"target_id": "portfolio-target:synthetic-v1",
|
||||
"target_weights": {
|
||||
"A": 0.6,
|
||||
"B": 0.4
|
||||
},
|
||||
"turnover_l1": 0.19999999999999996
|
||||
},
|
||||
"risk_assessment": {
|
||||
"assessment_id": "rhriskassessmentv1:sha256:dbc38825cffcf6d95bd0216d22dbba4e0d4a1359ca5924a3ee529b99a7d78b6d",
|
||||
"component_risk": {
|
||||
"A": 1.4549226783578566,
|
||||
"B": 1.4549226783578568
|
||||
},
|
||||
"contract_name": "researchhub.risk-assessment",
|
||||
"covariance_as_of_date": "2026-01-08",
|
||||
"covariance_data_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"covariance_input_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
|
||||
"covariance_snapshot_id": "covariance:synthetic-v1",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
|
||||
"findings": [],
|
||||
"freshness_policy_digest": "sha256:833f58f4d1056f0450f7369fd4edd70534cc74e9e55cd60f05b2b3d7a2979763",
|
||||
"group_exposure": {
|
||||
"equity": 1.4549226783578566,
|
||||
"fixed_income": 1.4549226783578568
|
||||
},
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
|
||||
"marginal_risk": {
|
||||
"A": 2.424871130596428,
|
||||
"B": 3.637306695894642
|
||||
},
|
||||
"percentage_risk": {
|
||||
"A": 0.49999999999999983,
|
||||
"B": 0.49999999999999994
|
||||
},
|
||||
"periods_per_year": 252,
|
||||
"portfolio_volatility": 2.909845356715714,
|
||||
"portfolio_volatility_limit": 10.0,
|
||||
"qualified": true,
|
||||
"return_frequency": "1d",
|
||||
"risk_budget": {
|
||||
"A": 0.8,
|
||||
"B": 0.8
|
||||
},
|
||||
"risk_model_digest": "sha256:2222222222222222222222222222222222222222222222222222222222222222",
|
||||
"risk_model_name": "euler_volatility",
|
||||
"risk_model_version": "1.0.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
|
||||
"schema_version": "1.0.0",
|
||||
"status": "ready"
|
||||
}
|
||||
}
|
||||
-1215
File diff suppressed because it is too large
Load Diff
@@ -1,175 +0,0 @@
|
||||
{
|
||||
"contract_name": "researchhub.data-foundation",
|
||||
"schema_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"published_at": "2026-09-08T01:05:00Z",
|
||||
"instrument_routes": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"symbol": "SIM0",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb",
|
||||
"route_revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9"
|
||||
},
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"symbol": "SIM1",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"effective_from": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799",
|
||||
"route_revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
}
|
||||
],
|
||||
"trading_calendar_revisions": [
|
||||
{
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"calendar_id": "rhcalendar:33333333333333333333333333333333",
|
||||
"session_date": "2018-01-02",
|
||||
"status": "open",
|
||||
"sessions": [
|
||||
{
|
||||
"opens_at": "2018-01-02T01:30:00Z",
|
||||
"closes_at": "2018-01-02T07:00:00Z"
|
||||
}
|
||||
],
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a",
|
||||
"calendar_revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
}
|
||||
],
|
||||
"corporate_action_revisions": [],
|
||||
"standardized_views": [
|
||||
{
|
||||
"view_id": "rhview:66666666666666666666666666666666",
|
||||
"view_version": "2.0.0",
|
||||
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"transformation_digest": "sha256:b2376dbe4424b38e5b27874c998f082aecc086179f611896ebaa0fdeb5362c68",
|
||||
"instrument_route_revision_ids": [
|
||||
"rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
|
||||
],
|
||||
"trading_calendar_revision_ids": [
|
||||
"rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
|
||||
],
|
||||
"corporate_action_revision_ids": [],
|
||||
"usage": "retrospective_research",
|
||||
"historical_availability": "not_established",
|
||||
"available_at": "2026-09-08T01:04:00Z",
|
||||
"view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309"
|
||||
}
|
||||
],
|
||||
"observation_lineage": [
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb"
|
||||
},
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799"
|
||||
},
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25",
|
||||
"observation_sequence": 1,
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a"
|
||||
}
|
||||
],
|
||||
"corporate_action_coverage": [
|
||||
{
|
||||
"instrument_id": "rhinstrument:11111111111111111111111111111111",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:233cc2968985e953e59e408b6619a837a792ae17f8afba32261f861de8fe9c7c"
|
||||
]
|
||||
},
|
||||
{
|
||||
"instrument_id": "rhinstrument:22222222222222222222222222222222",
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:21f2594fc42e46168a65d7cac7e9b52aeabf24ecc5fb409530c07ad56ed3986c"
|
||||
]
|
||||
}
|
||||
],
|
||||
"readiness": {
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"contract_validation": {
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:62f2abbbb79f4ff9164e10cc4b21df37f0964224a4c546f978b97794241ff5b3"
|
||||
]
|
||||
},
|
||||
"real_data_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"production_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
},
|
||||
"live_validation": {
|
||||
"status": "not_validated",
|
||||
"evidence_digests": []
|
||||
}
|
||||
},
|
||||
"foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260"
|
||||
}
|
||||
@@ -1,120 +0,0 @@
|
||||
{
|
||||
"contract_name": "researchhub.dataset-snapshot",
|
||||
"schema_version": "2.0.0",
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"descriptor": {
|
||||
"dataset": {
|
||||
"dataset_id": "rhdataset:market:44444444444444444444444444444444",
|
||||
"dataset_kind": "market",
|
||||
"record_schema_version": "2.0.0",
|
||||
"dimensions": [
|
||||
"instrument_id",
|
||||
"effective_time"
|
||||
]
|
||||
},
|
||||
"published_at": "2026-09-08T01:03:00Z",
|
||||
"time_semantics": {
|
||||
"effective_time": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z"
|
||||
},
|
||||
"observation_cutoff": "2026-09-08T01:01:00Z",
|
||||
"earliest_external_knowledge": {
|
||||
"status": "unknown"
|
||||
},
|
||||
"historical_availability": "not_established"
|
||||
},
|
||||
"content": {
|
||||
"digest_algorithm": "sha256",
|
||||
"canonicalization": "RFC8785",
|
||||
"record_order": "canonical-record-byte-order",
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"logical_manifest": {
|
||||
"record_count": 2,
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2
|
||||
}
|
||||
]
|
||||
},
|
||||
"manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7",
|
||||
"record_count": 2
|
||||
},
|
||||
"observation_manifest": {
|
||||
"batches": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
|
||||
"record_count": 2,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": "sha256:6f35f19a2ede0610d461b458a0f2cbc96f40cee6e90fa6592fd66daf617d3ec0"
|
||||
}
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"publisher": {
|
||||
"id": "researchhub.data",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"transformation": {
|
||||
"id": "rhtransform:55555555555555555555555555555555",
|
||||
"version": "2.0.0"
|
||||
},
|
||||
"upstream_snapshot_ids": [],
|
||||
"upstream_content_digests": []
|
||||
},
|
||||
"quality": {
|
||||
"status": "passed",
|
||||
"checks": [
|
||||
{
|
||||
"check_id": "completeness",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:519b01c60ad85c2b03d6c4f29027fe9b3672a8f6ae3ca8a2d831b281a61c5095"
|
||||
},
|
||||
{
|
||||
"check_id": "duplicate_identity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:7c585c7471ab45886fce037061b5ab0d5f981aede5d5190021963676b0ec0fcc"
|
||||
},
|
||||
{
|
||||
"check_id": "observation_coverage",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:dde5693c3ad79a01a162f9fc5a40c69d3c284965e283c91c5cd456904e0ef737"
|
||||
},
|
||||
{
|
||||
"check_id": "historical_claim_policy",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:f4cf8da48f92e03d75bb28b2569061e649f382ba360d879fef8e41e35169c228"
|
||||
},
|
||||
{
|
||||
"check_id": "range_validity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:6af08355c5e29d846e1c48783dccde791e8a3f48f8416149413217b7c2cbb52a"
|
||||
},
|
||||
{
|
||||
"check_id": "schema_conformance",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:fefdc5b81eba128af92f15a60feb7bb6f40b2e7277bf2beb48c1bb9399ee564b"
|
||||
}
|
||||
]
|
||||
},
|
||||
"qualification": {
|
||||
"status": "qualified",
|
||||
"usage": "retrospective_research",
|
||||
"policy_id": "researchhub.dataset-snapshot.retrospective",
|
||||
"policy_version": "2.0.0",
|
||||
"evaluated_at": "2026-09-08T01:02:00Z",
|
||||
"evidence_digest": "sha256:70c3ddcadf095877f17a1365c8d75a2e5a7078a27722fb8b698e61d350c6bb2d"
|
||||
}
|
||||
},
|
||||
"snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905"
|
||||
}
|
||||
@@ -1,103 +1,32 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
spec = json.loads((ROOT / "MODULE_SPEC.yaml").read_text(encoding="utf-8"))
|
||||
assert spec["module_id"] == "quant_engine"
|
||||
assert spec["authority"]["subject"] == spec["module_id"]
|
||||
assert spec["repository"]["type"] == "research_engine"
|
||||
assert spec["bounded_context"]["domain"] == "quantitative-research-engine"
|
||||
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
|
||||
for term in ("investment advice", "live order", "credentials", "source facts"):
|
||||
assert term in prohibited
|
||||
assert spec["authority"]["revision"] == 6
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["provides"]
|
||||
} == {
|
||||
("researchhub.factor-definition", "1.0.0"),
|
||||
("researchhub.factor-set-ref", "1.0.0"),
|
||||
("researchhub.backtest-run-ref", "1.0.0"),
|
||||
("researchhub.backtest-evidence-manifest", "1.0.0"),
|
||||
("researchhub.performance-evidence", "1.0.0"),
|
||||
("researchhub.portfolio-decision", "1.0.0"),
|
||||
("researchhub.risk-assessment", "1.0.0"),
|
||||
("researchhub.factor-set-ref", "2.0.0"),
|
||||
("researchhub.backtest-run-ref", "2.0.0"),
|
||||
("researchhub.backtest-evidence-manifest", "2.0.0"),
|
||||
("researchhub.performance-evidence", "2.0.0"),
|
||||
("researchhub.portfolio-target", "2.0.0"),
|
||||
("researchhub.portfolio-decision", "2.0.0"),
|
||||
("researchhub.risk-assessment", "2.0.0"),
|
||||
}
|
||||
expected_paths = {
|
||||
"researchhub.factor-definition": "src/quant_engine/factor_contracts.py",
|
||||
"researchhub.factor-set-ref": "src/quant_engine/factor_contracts.py",
|
||||
"researchhub.backtest-run-ref": "src/quant_engine/governed_pipeline.py",
|
||||
"researchhub.backtest-evidence-manifest": "src/quant_engine/artifact.py",
|
||||
"researchhub.performance-evidence": "src/quant_engine/artifact.py",
|
||||
"researchhub.portfolio-decision": "src/quant_engine/portfolio_risk_contracts.py",
|
||||
"researchhub.risk-assessment": "src/quant_engine/portfolio_risk_contracts.py",
|
||||
}
|
||||
assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"])
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "1.0.0"
|
||||
} == expected_paths
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
if item["version"] == "2.0.0"
|
||||
} == {
|
||||
"researchhub.factor-set-ref": "src/quant_engine/retrospective_factor_contracts.py",
|
||||
"researchhub.backtest-run-ref": "src/quant_engine/retrospective_backtest_contracts.py",
|
||||
"researchhub.backtest-evidence-manifest": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.performance-evidence": "src/quant_engine/retrospective_artifact_contracts.py",
|
||||
"researchhub.portfolio-target": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.portfolio-decision": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
"researchhub.risk-assessment": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
|
||||
}
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["consumes"]
|
||||
} == {
|
||||
("researchhub.dataset-snapshot", "1.0.0"),
|
||||
("researchhub.data-foundation", "1.0.0"),
|
||||
("researchhub.dataset-snapshot", "2.0.0"),
|
||||
("researchhub.data-foundation", "2.0.0"),
|
||||
}
|
||||
assert all(
|
||||
item["authority"] == "researchhub.data"
|
||||
for item in spec["contracts"]["consumes"]
|
||||
)
|
||||
assert spec["dependencies"] == []
|
||||
capabilities = {item["id"]: item for item in spec["capabilities"]}
|
||||
evidence_contract = capabilities["backtest-evidence-contracts"]
|
||||
assert evidence_contract["status"] == "operational"
|
||||
evidence_summary = evidence_contract["summary"].lower()
|
||||
for term in ("performance-methodology", "without recomputation", "decision authority"):
|
||||
assert term in evidence_summary
|
||||
portfolio_contract = capabilities["portfolio-risk-computation-contracts"]
|
||||
assert portfolio_contract["status"] == "operational"
|
||||
summary = portfolio_contract["summary"].lower()
|
||||
for term in ("receipt", "portfolio decisions", "risk assessments", "without"):
|
||||
assert term in summary
|
||||
retrospective = capabilities["retrospective-computation-contracts"]
|
||||
assert retrospective["status"] == "operational"
|
||||
for term in ("two clocks", "retrospective", "no historical-availability", "no execution"):
|
||||
assert term in retrospective["summary"].lower()
|
||||
for term in ("approval", "maker-checker", "publication", "paper", "live"):
|
||||
assert term in prohibited
|
||||
assert all(
|
||||
command["required"] and not command["network"]
|
||||
for command in spec["verification"]["commands"]
|
||||
)
|
||||
class ModuleSpecTests(unittest.TestCase):
|
||||
def test_module_spec_declares_pure_research_engine_boundary(self) -> None:
|
||||
spec = json.loads((ROOT / "MODULE_SPEC.yaml").read_text(encoding="utf-8"))
|
||||
self.assertEqual(spec["module_id"], "quant_engine")
|
||||
self.assertEqual(spec["authority"]["subject"], spec["module_id"])
|
||||
self.assertEqual(spec["repository"]["type"], "research_engine")
|
||||
self.assertEqual(spec["bounded_context"]["domain"], "quantitative-research-engine")
|
||||
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
|
||||
for term in ("investment advice", "live order", "credentials", "source facts"):
|
||||
self.assertIn(term, prohibited)
|
||||
self.assertEqual(spec["contracts"], {"provides": [], "consumes": []})
|
||||
self.assertEqual(spec["dependencies"], [])
|
||||
self.assertTrue(
|
||||
all(
|
||||
command["required"] and not command["network"]
|
||||
for command in spec["verification"]["commands"]
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
test_module_spec_declares_pure_research_engine_boundary()
|
||||
unittest.main()
|
||||
|
||||
@@ -6,30 +6,8 @@ import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
import quant_engine.alpha_factors as alpha_factors_module
|
||||
from quant_engine.factor_contracts import (
|
||||
FactorContractError,
|
||||
FactorInput,
|
||||
ProducerIdentity,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.alpha_factors import (
|
||||
ALPHA158_REGISTRY,
|
||||
ALPHA158_PHASE1_OPERATOR_SPECS,
|
||||
ALPHA158_PHASE2_OPERATOR_SPECS,
|
||||
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE3_FORMULA_SPECS,
|
||||
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE4_FORMULA_SPECS,
|
||||
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE5_FORMULA_SPECS,
|
||||
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE6_FORMULA_SPECS,
|
||||
alpha_001,
|
||||
alpha_002,
|
||||
alpha_003,
|
||||
@@ -188,18 +166,6 @@ from quant_engine.alpha_factors import (
|
||||
alpha_156,
|
||||
alpha_157,
|
||||
alpha_158,
|
||||
evaluate_phase1_operator,
|
||||
evaluate_phase2_operator,
|
||||
evaluate_phase3_formula,
|
||||
evaluate_phase4_formula,
|
||||
evaluate_phase5_formula,
|
||||
evaluate_phase6_formula,
|
||||
list_phase1_operators,
|
||||
list_phase2_operators,
|
||||
list_phase3_formulas,
|
||||
list_phase4_formulas,
|
||||
list_phase5_formulas,
|
||||
list_phase6_formulas,
|
||||
correlation,
|
||||
covariance,
|
||||
decay_linear,
|
||||
@@ -441,50 +407,6 @@ def test_alpha_registry_required_fields():
|
||||
assert required <= set(meta.keys()), f"{alpha_id} missing fields"
|
||||
|
||||
|
||||
def test_alpha_registry_adapts_to_definition_without_copying_formula_or_inputs():
|
||||
factor_input = FactorInput(
|
||||
"market",
|
||||
"sha256:" + "1" * 64,
|
||||
tuple(ALPHA158_REGISTRY["alpha_005"]["inputs"]),
|
||||
)
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "2" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
assert definition.formula == ALPHA158_REGISTRY["alpha_005"]["formula"]
|
||||
assert definition.inputs[0].required_columns == tuple(
|
||||
ALPHA158_REGISTRY["alpha_005"]["inputs"]
|
||||
)
|
||||
|
||||
incomplete = FactorInput("market", "sha256:" + "1" * 64, ("close",))
|
||||
with pytest.raises(FactorContractError, match="exactly correspond"):
|
||||
factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(incomplete,),
|
||||
implementation_digest="sha256:" + "2" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((incomplete,)),
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
|
||||
def test_get_alpha_meta_success():
|
||||
"""已知 alpha_id 返回完整 meta。"""
|
||||
meta = get_alpha_meta("alpha_001")
|
||||
@@ -1310,901 +1232,3 @@ def test_parse_alpha_formula_round_trip_jsonb():
|
||||
serialized = json.dumps(parsed)
|
||||
assert isinstance(serialized, str)
|
||||
assert "ts_rank" in serialized
|
||||
|
||||
|
||||
# ── v1.2.0 Phase 1: deterministic operator dispatch contract ──────────────
|
||||
|
||||
|
||||
def test_phase1_operator_catalog_is_explicit_and_serializable():
|
||||
"""Phase 1 exposes a stable, JSON-friendly catalog for downstream callers."""
|
||||
import json
|
||||
|
||||
expected = {
|
||||
"rank",
|
||||
"delta",
|
||||
"ts_mean",
|
||||
"ts_std",
|
||||
"ts_rank",
|
||||
"correlation",
|
||||
"ts_min",
|
||||
"ts_max",
|
||||
"ts_sum",
|
||||
"decay_linear",
|
||||
}
|
||||
assert set(list_phase1_operators()) == expected
|
||||
assert set(ALPHA158_PHASE1_OPERATOR_SPECS) == expected
|
||||
json.dumps(ALPHA158_PHASE1_OPERATOR_SPECS)
|
||||
for name, spec in ALPHA158_PHASE1_OPERATOR_SPECS.items():
|
||||
assert spec["name"] == name
|
||||
assert isinstance(spec["inputs"], list)
|
||||
assert isinstance(spec["formula"], str)
|
||||
|
||||
|
||||
def test_phase1_unary_operators_preserve_index_and_are_deterministic():
|
||||
values = pd.Series([1.0, 2.0, 3.0, 4.0], index=["a", "b", "c", "d"])
|
||||
|
||||
first = evaluate_phase1_operator("rank", values)
|
||||
second = evaluate_phase1_operator("rank", values)
|
||||
|
||||
pd.testing.assert_series_equal(first, second)
|
||||
assert first.index.equals(values.index)
|
||||
assert first.iloc[-1] == pytest.approx(1.0)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("name", "window"),
|
||||
[
|
||||
("delta", 2),
|
||||
("ts_mean", 2),
|
||||
("ts_std", 2),
|
||||
("ts_rank", 2),
|
||||
("ts_min", 2),
|
||||
("ts_max", 2),
|
||||
("ts_sum", 2),
|
||||
("decay_linear", 2),
|
||||
],
|
||||
)
|
||||
def test_phase1_windowed_operators_require_explicit_window(name: str, window: int):
|
||||
values = pd.Series([1.0, 2.0, 3.0, 4.0])
|
||||
|
||||
result = evaluate_phase1_operator(name, values, window=window)
|
||||
|
||||
assert result.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase1_operator(name, values)
|
||||
with pytest.raises(ValueError, match="positive integer"):
|
||||
evaluate_phase1_operator(name, values, window=1.5) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_phase1_binary_correlation_requires_aligned_secondary_input():
|
||||
values = pd.Series([1.0, 2.0, 3.0, 4.0])
|
||||
other = pd.Series([4.0, 3.0, 2.0, 1.0])
|
||||
|
||||
result = evaluate_phase1_operator("correlation", values, other, window=2)
|
||||
|
||||
assert result.iloc[-1] == pytest.approx(-1.0)
|
||||
with pytest.raises(ValueError, match="secondary"):
|
||||
evaluate_phase1_operator("correlation", values, window=2)
|
||||
|
||||
|
||||
def test_phase1_dispatch_rejects_unknown_or_unused_arguments():
|
||||
values = pd.Series([1.0, 2.0, 3.0])
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase1_operator("unknown", values)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase1_operator("rank", values, window=2)
|
||||
with pytest.raises(ValueError, match="secondary"):
|
||||
evaluate_phase1_operator("rank", values, values)
|
||||
|
||||
|
||||
def test_phase1_dispatch_rejects_window_above_supported_limit():
|
||||
values = pd.Series([1.0, 2.0, 3.0])
|
||||
|
||||
with pytest.raises(ValueError, match="maximum"):
|
||||
evaluate_phase1_operator("ts_mean", values, window=2**63)
|
||||
|
||||
|
||||
# ── v1.2.0 Phase 2: cumulative deterministic operator contract ─────────────
|
||||
|
||||
|
||||
def test_phase2_operator_catalog_is_cumulative_stable_and_serializable():
|
||||
"""Phase 2 exposes all existing building blocks without changing Phase 1."""
|
||||
import json
|
||||
|
||||
phase1 = list_phase1_operators()
|
||||
expected_phase2 = (
|
||||
*phase1,
|
||||
"ts_argmin",
|
||||
"ts_argmax",
|
||||
"product",
|
||||
"returns",
|
||||
"scale",
|
||||
"signed_power",
|
||||
"stddev",
|
||||
"covariance",
|
||||
"log",
|
||||
"abs_series",
|
||||
"sign",
|
||||
"max_pair",
|
||||
"min_pair",
|
||||
"indneutralize",
|
||||
)
|
||||
|
||||
assert list_phase2_operators() == expected_phase2
|
||||
assert tuple(ALPHA158_PHASE2_OPERATOR_SPECS) == expected_phase2
|
||||
assert tuple(ALPHA158_PHASE1_OPERATOR_SPECS) == phase1
|
||||
json.dumps(ALPHA158_PHASE2_OPERATOR_SPECS)
|
||||
for name, spec in ALPHA158_PHASE2_OPERATOR_SPECS.items():
|
||||
assert spec["name"] == name
|
||||
assert isinstance(spec["inputs"], list)
|
||||
assert isinstance(spec["parameters"], list)
|
||||
assert isinstance(spec["formula"], str)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", ["ts_argmin", "ts_argmax", "product", "stddev"])
|
||||
def test_phase2_windowed_unary_dispatch_is_deterministic(name: str):
|
||||
values = pd.Series([3.0, 1.0, 4.0, 2.0], index=["a", "b", "c", "d"])
|
||||
|
||||
first = evaluate_phase2_operator(name, values, window=3)
|
||||
second = evaluate_phase2_operator(name, values, window=3)
|
||||
|
||||
pd.testing.assert_series_equal(first, second)
|
||||
assert first.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase2_operator(name, values)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", ["returns", "scale", "log", "abs_series", "sign"])
|
||||
def test_phase2_unary_dispatch_rejects_unused_arguments(name: str):
|
||||
values = pd.Series([1.0, 2.0, 4.0], index=["a", "b", "c"])
|
||||
|
||||
result = evaluate_phase2_operator(name, values)
|
||||
|
||||
assert result.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase2_operator(name, values, window=2)
|
||||
with pytest.raises(ValueError, match="secondary"):
|
||||
evaluate_phase2_operator(name, values, secondary=values)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("name", "window"),
|
||||
[("correlation", 2), ("covariance", 2), ("max_pair", None), ("min_pair", None)],
|
||||
)
|
||||
def test_phase2_binary_dispatch_requires_aligned_secondary(name: str, window: int | None):
|
||||
values = pd.Series([1.0, 2.0, 3.0], index=["a", "b", "c"])
|
||||
secondary = pd.Series([3.0, 2.0, 1.0], index=values.index)
|
||||
|
||||
result = evaluate_phase2_operator(name, values, secondary=secondary, window=window)
|
||||
|
||||
assert result.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="secondary is required"):
|
||||
evaluate_phase2_operator(name, values, window=window)
|
||||
with pytest.raises(ValueError, match="secondary index"):
|
||||
evaluate_phase2_operator(
|
||||
name,
|
||||
values,
|
||||
secondary=secondary.rename(index={"c": "z"}),
|
||||
window=window,
|
||||
)
|
||||
|
||||
|
||||
def test_phase2_signed_power_requires_finite_numeric_exponent():
|
||||
values = pd.Series([-4.0, 0.0, 9.0])
|
||||
|
||||
result = evaluate_phase2_operator("signed_power", values, exponent=0.5)
|
||||
|
||||
pd.testing.assert_series_equal(result, pd.Series([-2.0, 0.0, 3.0]))
|
||||
for exponent in (None, True, float("inf"), float("nan"), "2"):
|
||||
with pytest.raises(ValueError, match="exponent"):
|
||||
evaluate_phase2_operator( # type: ignore[arg-type]
|
||||
"signed_power",
|
||||
values,
|
||||
exponent=exponent,
|
||||
)
|
||||
|
||||
|
||||
def test_phase2_indneutralize_requires_aligned_groups():
|
||||
values = pd.Series([1.0, 3.0, 10.0, 14.0], index=["a", "b", "c", "d"])
|
||||
groups = pd.Series(["x", "x", "y", "y"], index=values.index)
|
||||
|
||||
result = evaluate_phase2_operator("indneutralize", values, groups=groups)
|
||||
|
||||
pd.testing.assert_series_equal(result, pd.Series([-1.0, 1.0, -2.0, 2.0], index=values.index))
|
||||
with pytest.raises(ValueError, match="groups is required"):
|
||||
evaluate_phase2_operator("indneutralize", values)
|
||||
with pytest.raises(ValueError, match="groups index"):
|
||||
evaluate_phase2_operator(
|
||||
"indneutralize",
|
||||
values,
|
||||
groups=groups.rename(index={"d": "z"}),
|
||||
)
|
||||
|
||||
|
||||
def test_phase2_dispatch_validates_primary_series_and_unused_parameters():
|
||||
values = pd.Series([1.0, 2.0, 3.0])
|
||||
|
||||
with pytest.raises(TypeError, match="series must be a pandas Series"):
|
||||
evaluate_phase2_operator("rank", [1.0, 2.0, 3.0]) # type: ignore[arg-type]
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase2_operator("unknown", values)
|
||||
with pytest.raises(ValueError, match="exponent"):
|
||||
evaluate_phase2_operator("rank", values, exponent=2.0)
|
||||
with pytest.raises(ValueError, match="groups"):
|
||||
evaluate_phase2_operator("rank", values, groups=pd.Series(["x", "x", "x"]))
|
||||
with pytest.raises(ValueError, match="maximum"):
|
||||
evaluate_phase2_operator("product", values, window=253)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 3: versioned alpha001-alpha050 formula contract ──────────
|
||||
|
||||
|
||||
def _phase3_market_inputs() -> dict[str, pd.Series]:
|
||||
positions = np.arange(80, dtype=float)
|
||||
index = pd.RangeIndex(len(positions), name="row")
|
||||
open_ = pd.Series(100.0 + positions * 0.2 + np.sin(positions / 4.0), index=index)
|
||||
close = pd.Series(100.5 + positions * 0.18 + np.cos(positions / 5.0), index=index)
|
||||
high = pd.Series(np.maximum(open_, close) + 1.0, index=index)
|
||||
low = pd.Series(np.minimum(open_, close) - 1.0, index=index)
|
||||
volume = pd.Series(1_000.0 + positions**1.3 + 20.0 * np.sin(positions / 3.0), index=index)
|
||||
vwap = (open_ + close + high + low) / 4.0
|
||||
return {
|
||||
"open": open_,
|
||||
"close": close,
|
||||
"high": high,
|
||||
"low": low,
|
||||
"volume": volume,
|
||||
"vwap": vwap,
|
||||
}
|
||||
|
||||
|
||||
def test_phase3_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(1, 51))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase3_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE3_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE3_FORMULA_SPECS.values()
|
||||
) == {"single": 19, "pair": 26, "triple": 4, "quadruple": 1}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE3_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 == (
|
||||
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE3_FORMULA_SPECS, "alpha_001", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_catalog_freezes_callable_signatures_without_rewriting_formulas():
|
||||
import inspect
|
||||
|
||||
legacy_formula_input_differences = {
|
||||
"alpha_011": ("close", "high", "low"),
|
||||
"alpha_035": ("volume",),
|
||||
"alpha_036": ("close",),
|
||||
"alpha_040": ("high", "low"),
|
||||
"alpha_042": ("close",),
|
||||
"alpha_043": ("volume",),
|
||||
}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == legacy_formula_input_differences.get(
|
||||
alpha_id,
|
||||
signature_inputs,
|
||||
)
|
||||
|
||||
assert ALPHA158_PHASE3_FORMULA_SPECS["alpha_035"]["call_inputs"] == (
|
||||
"close",
|
||||
"volume",
|
||||
)
|
||||
assert ALPHA158_REGISTRY["alpha_035"]["inputs"] == ["volume"]
|
||||
|
||||
|
||||
def test_phase3_dispatch_matches_all_existing_alpha001_alpha050_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase3_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase3_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase3_formula("alpha_051", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
|
||||
evaluate_phase3_formula("alpha_005", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase3_formula(
|
||||
"alpha_005",
|
||||
close=inputs["close"],
|
||||
volume=inputs["volume"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="close must be a pandas Series"):
|
||||
evaluate_phase3_formula( # type: ignore[arg-type]
|
||||
"alpha_005",
|
||||
close=[1.0, 2.0],
|
||||
volume=inputs["volume"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_dispatch_rejects_implicit_series_alignment():
|
||||
inputs = _phase3_market_inputs()
|
||||
misaligned_volume = inputs["volume"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="volume index must align with close"):
|
||||
evaluate_phase3_formula(
|
||||
"alpha_005",
|
||||
close=inputs["close"],
|
||||
volume=misaligned_volume,
|
||||
)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 4: versioned alpha051-alpha100 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase4_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(51, 101))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase4_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE4_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE4_FORMULA_SPECS.values()
|
||||
) == {"single": 1, "pair": 27, "triple": 11, "quadruple": 10, "quintuple": 1}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE4_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 == (
|
||||
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE4_FORMULA_SPECS, "alpha_051", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE4_FORMULA_SPECS["alpha_055"]["call_inputs"] == (
|
||||
"open",
|
||||
"high",
|
||||
"low",
|
||||
"volume",
|
||||
"close",
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_dispatch_matches_all_existing_alpha051_alpha100_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase4_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase4_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase4_formula("alpha_050", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*low"):
|
||||
evaluate_phase4_formula("alpha_051", high=inputs["high"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase4_formula(
|
||||
"alpha_051",
|
||||
high=inputs["high"],
|
||||
low=inputs["low"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="high must be a pandas Series"):
|
||||
evaluate_phase4_formula( # type: ignore[arg-type]
|
||||
"alpha_051",
|
||||
high=[1.0, 2.0],
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_dispatch_rejects_implicit_series_alignment():
|
||||
inputs = _phase3_market_inputs()
|
||||
misaligned_low = inputs["low"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="low index must align with high"):
|
||||
evaluate_phase4_formula(
|
||||
"alpha_051",
|
||||
high=inputs["high"],
|
||||
low=misaligned_low,
|
||||
)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 5: versioned alpha101-alpha150 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase5_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(101, 151))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase5_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE5_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE5_FORMULA_SPECS.values()
|
||||
) == {"pair": 33, "triple": 14, "quadruple": 3}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE5_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 == (
|
||||
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE5_FORMULA_SPECS, "alpha_101", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"] == (
|
||||
"close",
|
||||
"high",
|
||||
"low",
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_dispatch_matches_all_existing_alpha101_alpha150_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase5_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase5_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase5_formula("alpha_100", close=inputs["close"])
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase5_formula("alpha_151", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*low"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
low=inputs["low"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="high must be a pandas Series"):
|
||||
evaluate_phase5_formula( # type: ignore[arg-type]
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=[1.0, 2.0],
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_dispatch_rejects_length_and_index_alignment_errors():
|
||||
inputs = _phase3_market_inputs()
|
||||
shorter_low = inputs["low"].iloc[:-1]
|
||||
misaligned_high = inputs["high"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="low length must match close"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
low=shorter_low,
|
||||
)
|
||||
with pytest.raises(ValueError, match="high index must align with close"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=misaligned_high,
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_contract_is_publicly_exported():
|
||||
assert {
|
||||
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE5_FORMULA_SPECS",
|
||||
"list_phase5_formulas",
|
||||
"evaluate_phase5_formula",
|
||||
} <= set(alpha_factors_module.__all__)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 6: versioned alpha151-alpha158 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase6_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(151, 159))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase6_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE6_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE6_FORMULA_SPECS.values()
|
||||
) == {"pair": 6, "triple": 2}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE6_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 == (
|
||||
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE6_FORMULA_SPECS, "alpha_151", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE6_FORMULA_SPECS["alpha_158"]["call_inputs"] == (
|
||||
"high",
|
||||
"low",
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_dispatch_matches_all_existing_alpha151_alpha158_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase6_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase6_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase6_formula("alpha_150", close=inputs["close"])
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase6_formula("alpha_159", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
|
||||
evaluate_phase6_formula("alpha_151", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=inputs["volume"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="volume must be a pandas Series"):
|
||||
evaluate_phase6_formula( # type: ignore[arg-type]
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=[1.0, 2.0],
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_dispatch_rejects_length_and_index_alignment_errors():
|
||||
inputs = _phase3_market_inputs()
|
||||
shorter_volume = inputs["volume"].iloc[:-1]
|
||||
misaligned_low = inputs["low"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="volume length must match close"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=shorter_volume,
|
||||
)
|
||||
with pytest.raises(ValueError, match="low index must align with high"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_158",
|
||||
high=inputs["high"],
|
||||
low=misaligned_low,
|
||||
volume=inputs["volume"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_preserves_complete_alpha001_alpha158_formula_body_fingerprint():
|
||||
import ast
|
||||
import hashlib
|
||||
import inspect
|
||||
import json
|
||||
import textwrap
|
||||
|
||||
fingerprints = {}
|
||||
for number in range(1, 159):
|
||||
alpha_id = f"alpha_{number:03d}"
|
||||
source = textwrap.dedent(inspect.getsource(getattr(alpha_factors_module, alpha_id)))
|
||||
node = ast.parse(source).body[0]
|
||||
assert isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef))
|
||||
body = ast.dump(
|
||||
ast.Module(body=node.body, type_ignores=[]),
|
||||
include_attributes=False,
|
||||
)
|
||||
fingerprints[alpha_id] = hashlib.sha256(body.encode()).hexdigest()
|
||||
|
||||
encoded = json.dumps(
|
||||
fingerprints,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == (
|
||||
"f3ae807983cf8ca083d0b924c3807ffd84a62a5bd354f9efb1a1a16ca40c7da6"
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_contract_is_publicly_exported():
|
||||
assert {
|
||||
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE6_FORMULA_SPECS",
|
||||
"list_phase6_formulas",
|
||||
"evaluate_phase6_formula",
|
||||
} <= set(alpha_factors_module.__all__)
|
||||
|
||||
@@ -1,773 +0,0 @@
|
||||
"""Backtest run-reference and closed-evidence contract conformance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
BacktestEvidenceManifest,
|
||||
EvidenceQualification,
|
||||
RESEARCH_ARTIFACT_SCHEMA_VERSION,
|
||||
ResearchRunArtifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_legacy_backtest_evidence_manifest,
|
||||
build_research_run_artifact,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
InputBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
ProducerIdentity,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.governed_pipeline import (
|
||||
BacktestContractError,
|
||||
BacktestContractErrorCode,
|
||||
BacktestRun,
|
||||
BacktestRunRef,
|
||||
)
|
||||
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
BACKTEST_FIXTURE = ROOT / "tests" / "fixtures" / "backtest-evidence-v1.golden.json"
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _accepted_authorities(
|
||||
*,
|
||||
factor_evaluation_at: str = "2026-01-03T11:00:00Z",
|
||||
factor_computed_at: str = "2026-01-03T10:15:00Z",
|
||||
factor_artifact_available_at: str = "2026-01-03T10:20:00Z",
|
||||
factor_availability_mode: AvailabilityMode = AvailabilityMode.AS_AVAILABLE,
|
||||
) -> tuple[
|
||||
DatasetSnapshotEnvelope,
|
||||
DataFoundationEnvelope,
|
||||
FactorSetRef,
|
||||
]:
|
||||
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
|
||||
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
artifact_ref = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
selected_view_ref_ids=(VIEW_REF_ID,),
|
||||
input_bindings=(
|
||||
InputBinding(definition.definition_id, "market", VIEW_REF_ID, VIEW_SCHEMA_DIGEST),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
output_quality=OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
output_coverage=OutputCoverage(
|
||||
"complete", 1, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64
|
||||
),
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
output_artifact_ref=artifact_ref,
|
||||
availability_mode=factor_availability_mode,
|
||||
evaluation_at=factor_evaluation_at,
|
||||
computed_at=factor_computed_at,
|
||||
artifact_available_at=factor_artifact_available_at,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
actor=ActorIdentity("service", "factor_worker_v1"),
|
||||
correlation_id="research_run_001",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
evidence_scope="synthetic_fixture",
|
||||
decision_eligible=False,
|
||||
)
|
||||
return snapshot, foundation, factor_set
|
||||
|
||||
|
||||
def _config_digest(parameters: dict[str, object] | None = None) -> str:
|
||||
encoded = json.dumps(
|
||||
PARAMETERS if parameters is None else parameters,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
return _sha256(encoded)
|
||||
|
||||
|
||||
def _run_ref(**overrides: Any) -> BacktestRunRef:
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
arguments: dict[str, Any] = {
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"factor_set": factor_set,
|
||||
"universe_digest": "sha256:" + "5" * 64,
|
||||
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
|
||||
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": "sha256:" + "6" * 64,
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": "sha256:" + "7" * 64,
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": "sha256:" + "8" * 64,
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": "sha256:" + "9" * 64,
|
||||
"configuration_digest": _config_digest(),
|
||||
"evaluation_at": "2026-01-08T01:00:00Z",
|
||||
"computed_at": "2026-01-08T02:00:00Z",
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return BacktestRunRef.create(**arguments)
|
||||
|
||||
|
||||
def _backtest_result() -> FactorBacktestResult:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
return run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _artifact(run_ref: BacktestRunRef, *, run_id: str | None = None) -> ResearchRunArtifact:
|
||||
result = _backtest_result()
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id=run_ref.run_id if run_id is None else run_id,
|
||||
strategy_id=run_ref.strategy_id,
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version=run_ref.strategy_version,
|
||||
engine_version="1.2.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
data_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=PARAMETERS,
|
||||
benchmark_id="000300.SH",
|
||||
benchmark_returns=benchmark,
|
||||
)
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[BacktestContractError],
|
||||
code: BacktestContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_backtest_run_ref_is_deterministic_and_binds_only_opaque_authorities() -> None:
|
||||
first = _run_ref()
|
||||
second = _run_ref()
|
||||
|
||||
assert first == second
|
||||
assert first.run_id.startswith("rhbacktestrunv1:sha256:")
|
||||
assert first.replay_spec_digest.startswith("sha256:")
|
||||
assert first.dataset_snapshot_id.startswith("rhdsv1:sha256:")
|
||||
assert first.foundation_id.startswith("rhdfv1:sha256:")
|
||||
assert first.factor_set_id.startswith("rhfactorsetv1:sha256:")
|
||||
assert first.trading_calendar_revision_ids == (CALENDAR_REVISION_ID,)
|
||||
assert first.corporate_action_revision_ids == (ACTION_REVISION_ID,)
|
||||
assert first.replay_parent_run_id is None
|
||||
assert first.replay_attempt == 0
|
||||
assert first.replay_ancestor_run_ids == ()
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
assert BacktestRunRef.from_dict(
|
||||
first.to_dict(),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
) == first
|
||||
forbidden = ("latest", "locator", "uri", "credential", "provider", "broker")
|
||||
assert not any(token in first.to_json().lower() for token in forbidden)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "value"),
|
||||
[
|
||||
("universe_digest", "sha256:" + "a" * 64),
|
||||
("strategy_digest", "sha256:" + "b" * 64),
|
||||
("execution_model_digest", "sha256:" + "c" * 64),
|
||||
("cost_model_digest", "sha256:" + "e" * 64),
|
||||
("random_seed", 8),
|
||||
("code_revision", "e" * 40),
|
||||
("environment_lock_digest", "sha256:" + "f" * 64),
|
||||
("configuration_digest", "sha256:" + "0" * 64),
|
||||
("evaluation_at", "2026-01-08T01:00:01Z"),
|
||||
("computed_at", "2026-01-08T02:00:01Z"),
|
||||
],
|
||||
)
|
||||
def test_every_governed_run_input_mutation_changes_run_identity(
|
||||
field: str,
|
||||
value: object,
|
||||
) -> None:
|
||||
assert _run_ref(**{field: value}).run_id != _run_ref().run_id
|
||||
|
||||
|
||||
def test_backtest_run_ref_rejects_unclosed_upstream_and_unsafe_scalars() -> None:
|
||||
with pytest.raises(BacktestContractError) as wrong_calendar:
|
||||
_run_ref(trading_calendar_revision_ids=())
|
||||
_assert_error(
|
||||
wrong_calendar,
|
||||
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.trading_calendar_revision_ids",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as wrong_action:
|
||||
_run_ref(corporate_action_revision_ids=())
|
||||
_assert_error(
|
||||
wrong_action,
|
||||
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.corporate_action_revision_ids",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as bool_seed:
|
||||
_run_ref(random_seed=True)
|
||||
_assert_error(bool_seed, BacktestContractErrorCode.TYPE_ERROR, "$.random_seed")
|
||||
with pytest.raises(BacktestContractError) as bad_revision:
|
||||
_run_ref(code_revision="abc")
|
||||
_assert_error(bad_revision, BacktestContractErrorCode.INVALID_FORMAT, "$.code_revision")
|
||||
with pytest.raises(BacktestContractError) as bad_digest:
|
||||
_run_ref(universe_digest="5" * 64)
|
||||
_assert_error(bad_digest, BacktestContractErrorCode.INVALID_FORMAT, "$.universe_digest")
|
||||
with pytest.raises(BacktestContractError) as lookahead:
|
||||
_run_ref(computed_at="2026-01-08T00:59:59Z")
|
||||
_assert_error(lookahead, BacktestContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
|
||||
with pytest.raises(BacktestContractError) as factor_type:
|
||||
_run_ref(factor_set="rhfactorsetv1:sha256:" + "0" * 64)
|
||||
_assert_error(factor_type, BacktestContractErrorCode.TYPE_ERROR, "$.factor_set")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("factor_times", "expected_path"),
|
||||
[
|
||||
(
|
||||
{
|
||||
"factor_evaluation_at": "2026-01-08T01:00:01Z",
|
||||
"factor_computed_at": "2026-01-08T00:59:59Z",
|
||||
"factor_artifact_available_at": "2026-01-08T01:00:00Z",
|
||||
},
|
||||
"$.evaluation_at",
|
||||
),
|
||||
(
|
||||
{
|
||||
"factor_evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"factor_computed_at": "2026-01-08T01:00:00Z",
|
||||
"factor_artifact_available_at": "2026-01-08T01:00:01Z",
|
||||
"factor_availability_mode": AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
},
|
||||
"$.evaluation_at",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_run_ref_evaluation_closes_factor_pit(
|
||||
factor_times: dict[str, Any],
|
||||
expected_path: str,
|
||||
) -> None:
|
||||
snapshot, foundation, factor_set = _accepted_authorities(**factor_times)
|
||||
|
||||
with pytest.raises(BacktestContractError) as lookahead:
|
||||
_run_ref(
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
)
|
||||
|
||||
_assert_error(
|
||||
lookahead,
|
||||
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
expected_path,
|
||||
)
|
||||
|
||||
|
||||
def test_run_ref_rejects_aliases_locators_unsafe_integers_and_invalid_text() -> None:
|
||||
with pytest.raises(BacktestContractError) as mutable_alias:
|
||||
_run_ref(strategy_id="latest")
|
||||
_assert_error(
|
||||
mutable_alias,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as physical_uri:
|
||||
_run_ref(execution_model_version="s3://model-bucket/current")
|
||||
_assert_error(
|
||||
physical_uri,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.execution_model_version",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as unsafe_seed:
|
||||
_run_ref(random_seed=2**53)
|
||||
_assert_error(
|
||||
unsafe_seed,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.random_seed",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as invalid_unicode:
|
||||
_run_ref(strategy_id="\ud800")
|
||||
_assert_error(
|
||||
invalid_unicode,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
|
||||
run_ref = _run_ref()
|
||||
mixed_keys = run_ref.to_dict()
|
||||
mixed_keys[1] = "not-a-contract-key" # type: ignore[index]
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
with pytest.raises(BacktestContractError) as invalid_key:
|
||||
BacktestRunRef.from_dict(
|
||||
mixed_keys,
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
)
|
||||
_assert_error(invalid_key, BacktestContractErrorCode.TYPE_ERROR, "$")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"physical_id",
|
||||
["db.table", "source_alpha", "wind.model", "qtdb_view", "bloomberg-signal"],
|
||||
)
|
||||
def test_run_ref_rejects_physical_terms_in_logical_ids(physical_id: str) -> None:
|
||||
with pytest.raises(BacktestContractError) as physical:
|
||||
_run_ref(strategy_id=physical_id)
|
||||
_assert_error(
|
||||
physical,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("version", ["1.0.0-.", "1.0.0-foo..bar", "1.0.0-01"])
|
||||
def test_run_ref_requires_strict_semver_prerelease_identifiers(version: str) -> None:
|
||||
with pytest.raises(BacktestContractError) as invalid:
|
||||
_run_ref(strategy_version=version)
|
||||
_assert_error(
|
||||
invalid,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_version",
|
||||
)
|
||||
|
||||
assert _run_ref(strategy_version="1.0.0-alpha.1").strategy_version == "1.0.0-alpha.1"
|
||||
|
||||
|
||||
def test_replay_lineage_is_acyclic_and_cannot_claim_changed_inputs() -> None:
|
||||
parent = _run_ref()
|
||||
replay = _run_ref(
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="deterministic_reproduction",
|
||||
replay_attempt=1,
|
||||
)
|
||||
|
||||
assert replay.run_id != parent.run_id
|
||||
assert replay.replay_spec_digest == parent.replay_spec_digest
|
||||
assert replay.replay_parent_run_id == parent.run_id
|
||||
assert replay.replay_ancestor_run_ids == (parent.run_id,)
|
||||
|
||||
with pytest.raises(BacktestContractError) as changed_input:
|
||||
_run_ref(
|
||||
universe_digest="sha256:" + "a" * 64,
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="changed_universe",
|
||||
replay_attempt=1,
|
||||
)
|
||||
_assert_error(
|
||||
changed_input,
|
||||
BacktestContractErrorCode.LINEAGE_VIOLATION,
|
||||
"$.replay_spec_digest",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as skipped_attempt:
|
||||
_run_ref(
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="skipped_attempt",
|
||||
replay_attempt=2,
|
||||
)
|
||||
_assert_error(
|
||||
skipped_attempt,
|
||||
BacktestContractErrorCode.LINEAGE_VIOLATION,
|
||||
"$.replay_attempt",
|
||||
)
|
||||
|
||||
|
||||
def test_offline_research_manifest_closes_exact_existing_evidence_mapping() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
first = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
second = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
|
||||
assert first == second
|
||||
assert first.manifest_id.startswith("rhbacktestevidencev1:sha256:")
|
||||
assert first.run_id == run_ref.run_id
|
||||
assert first.profile == "offline_research_v1"
|
||||
assert first.qualification is EvidenceQualification.CONTRACT_QUALIFIED
|
||||
mapping = {
|
||||
item.category: tuple(table.logical_name for table in item.tables)
|
||||
for item in first.evidence
|
||||
}
|
||||
assert mapping == {
|
||||
"run": ("run",),
|
||||
"signal": ("signals",),
|
||||
"fill": ("trades",),
|
||||
"position_nav": ("positions", "nav"),
|
||||
"performance": ("performance",),
|
||||
"attribution": ("attribution", "attribution_daily"),
|
||||
"risk_snapshot": ("risk",),
|
||||
"replay": (),
|
||||
}
|
||||
assert "order" not in mapping
|
||||
assert "rejection" not in mapping
|
||||
risk = next(item for item in first.evidence if item.category == "risk_snapshot")
|
||||
assert risk.tables[0].row_count == 0
|
||||
assert risk.tables[0].schema_digest.startswith("sha256:")
|
||||
|
||||
changed_performance = artifact.performance
|
||||
changed_performance.loc[0, "n_days"] += 1
|
||||
changed_artifact = replace(artifact, _performance=changed_performance)
|
||||
changed = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
assert changed.manifest_id != first.manifest_id
|
||||
assert run_ref.run_id == first.run_id == changed.run_id
|
||||
|
||||
|
||||
def test_manifest_rejects_missing_mismatched_or_duplicate_evidence() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
with pytest.raises(BacktestContractError) as wrong_run:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
_artifact(run_ref, run_id="different-run"),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
wrong_run,
|
||||
BacktestContractErrorCode.IDENTITY_MISMATCH,
|
||||
"$.artifact.tables.run.run_id",
|
||||
)
|
||||
missing_signals = replace(artifact, _signals=None) # type: ignore[arg-type]
|
||||
with pytest.raises(BacktestContractError) as missing_table:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
missing_signals,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
missing_table,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.artifact.tables.signals",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as digest_mismatch:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
expected_table_digests={"performance": "sha256:" + "0" * 64},
|
||||
)
|
||||
_assert_error(
|
||||
digest_mismatch,
|
||||
BacktestContractErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.artifact.tables.performance.content_digest",
|
||||
)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
duplicate = manifest.to_dict()
|
||||
duplicate["evidence"].append(copy.deepcopy(duplicate["evidence"][0]))
|
||||
with pytest.raises(BacktestContractError) as duplicate_category:
|
||||
BacktestEvidenceManifest.from_dict(
|
||||
duplicate,
|
||||
backtest_run_ref=run_ref,
|
||||
artifact=artifact,
|
||||
)
|
||||
_assert_error(
|
||||
duplicate_category,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.evidence[8].category",
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_binds_supported_schema_and_has_collision_free_cell_encoding() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
|
||||
with pytest.raises(BacktestContractError) as unsupported_schema:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, schema_version="999.0.0"),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsupported_schema,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.artifact.schema_version",
|
||||
)
|
||||
|
||||
identities: set[str] = set()
|
||||
for value in (float("nan"), float("inf"), float("-inf")):
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "alpha"] = value
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=performance),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
assert manifest.artifact_schema_version == RESEARCH_ARTIFACT_SCHEMA_VERSION
|
||||
identities.add(manifest.manifest_id)
|
||||
assert len(identities) == 3
|
||||
|
||||
content_digests: set[str] = set()
|
||||
for value in (float("nan"), {"non_finite_float": "nan"}):
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, "alpha"] = value
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=performance),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
performance_entry = next(
|
||||
entry for entry in manifest.evidence if entry.category == "performance"
|
||||
)
|
||||
content_digests.add(performance_entry.tables[0].content_digest)
|
||||
assert len(content_digests) == 2
|
||||
|
||||
unsupported = artifact.performance.astype(object)
|
||||
unsupported.loc[0, "alpha"] = object()
|
||||
with pytest.raises(BacktestContractError) as unsupported_cell:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=unsupported),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsupported_cell,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.artifact.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
invalid_nested_key = artifact.performance.astype(object)
|
||||
invalid_nested_key.at[0, "alpha"] = {"\ud800": "value"}
|
||||
with pytest.raises(BacktestContractError) as invalid_utf8:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=invalid_nested_key),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
invalid_utf8,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.artifact.tables.performance.rows[0].alpha.keys",
|
||||
)
|
||||
|
||||
unsafe_integer = artifact.performance.astype(object)
|
||||
unsafe_integer.loc[0, "alpha"] = 10**5000
|
||||
with pytest.raises(BacktestContractError) as unsafe_cell:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=unsafe_integer),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsafe_cell,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.artifact.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
|
||||
def test_artifact_canonical_content_has_typed_collision_free_cell_encoding() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
|
||||
content_hashes: set[str] = set()
|
||||
for value in (
|
||||
float("nan"),
|
||||
float("inf"),
|
||||
float("-inf"),
|
||||
{"non_finite_float": "nan"},
|
||||
):
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, "alpha"] = value
|
||||
mutated = replace(artifact, _performance=performance)
|
||||
content_hashes.add(mutated.content_sha256)
|
||||
assert "non_finite_float" in mutated.canonical_json()
|
||||
assert len(content_hashes) == 4
|
||||
|
||||
unsupported = artifact.performance.astype(object)
|
||||
unsupported.loc[0, "alpha"] = object()
|
||||
with pytest.raises(BacktestContractError) as unsupported_cell:
|
||||
replace(artifact, _performance=unsupported).canonical_json()
|
||||
_assert_error(
|
||||
unsupported_cell,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
|
||||
def test_legacy_bridge_is_explicit_and_cannot_be_contract_qualified() -> None:
|
||||
run_ref = _run_ref()
|
||||
legacy_run = BacktestRun(
|
||||
run_id="legacy-run-001",
|
||||
dataset_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
factor_version_id="alpha_005@1.0.0",
|
||||
strategy_version_id="alpha-top1@1.0.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
config_hash=_config_digest().removeprefix("sha256:"),
|
||||
created_at=datetime(2026, 1, 8, 2, 0, tzinfo=UTC),
|
||||
)
|
||||
artifact = _artifact(run_ref, run_id=legacy_run.run_id)
|
||||
manifest = build_legacy_backtest_evidence_manifest(
|
||||
legacy_run,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
|
||||
assert manifest.qualification is EvidenceQualification.LEGACY_EXPLORATORY
|
||||
assert manifest.run_id == legacy_run.run_id
|
||||
assert manifest.backtest_run_ref is None
|
||||
assert manifest.to_dict()["run_reference"]["kind"] == "legacy_backtest_run"
|
||||
assert BacktestEvidenceManifest.from_dict(
|
||||
manifest.to_dict(),
|
||||
artifact=artifact,
|
||||
) == manifest
|
||||
with pytest.raises(BacktestContractError) as implicit_promotion:
|
||||
build_backtest_evidence_manifest( # type: ignore[arg-type]
|
||||
legacy_run,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
_assert_error(
|
||||
implicit_promotion,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.backtest_run_ref",
|
||||
)
|
||||
|
||||
|
||||
def test_golden_contract_and_architecture_boundary() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
golden = json.loads(BACKTEST_FIXTURE.read_text(encoding="utf-8"))
|
||||
|
||||
table_digests = {
|
||||
table.logical_name: table.content_digest
|
||||
for item in manifest.evidence
|
||||
for table in item.tables
|
||||
}
|
||||
assert golden == {
|
||||
"run_id": run_ref.run_id,
|
||||
"replay_spec_digest": run_ref.replay_spec_digest,
|
||||
"manifest_id": manifest.manifest_id,
|
||||
"evidence_digest": manifest.evidence_digest,
|
||||
"table_content_digests": table_digests,
|
||||
}
|
||||
governed_source = (ROOT / "src" / "quant_engine" / "governed_pipeline.py").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
artifact_source = (ROOT / "src" / "quant_engine" / "artifact.py").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
assert "from quant_engine.artifact" not in governed_source
|
||||
assert "BacktestRunRef" in governed_source
|
||||
assert "BacktestEvidenceManifest" not in governed_source
|
||||
assert "BacktestEvidenceManifest" in artifact_source
|
||||
assert not (ROOT / "src" / "quant_engine" / "backtest_contracts.py").exists()
|
||||
@@ -1,58 +0,0 @@
|
||||
"""Policies observe actual post-fill holdings and only schedule the next open."""
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.execution import ExecutionConfig, simulate_daily_ledger_with_audit
|
||||
|
||||
|
||||
def test_policy_next_open_actual_holdings_and_immutable_past_snapshots():
|
||||
seen = []
|
||||
|
||||
def decide(position):
|
||||
seen.append((position.date, dict(position.holdings), position.cash))
|
||||
position.holdings.clear()
|
||||
return {"A": 0.5} if position.date == "d1" else None
|
||||
|
||||
result = simulate_daily_ledger_with_audit(
|
||||
[],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 20}), ("d3", {"A": 30})],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 25}), ("d3", {"A": 40})],
|
||||
1000,
|
||||
ExecutionConfig(commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0),
|
||||
decision_policy=decide,
|
||||
)
|
||||
assert seen[0][1] == {}
|
||||
assert seen[1][1] == {"A": 25}
|
||||
assert result.nav_series.tolist() == [1000, 1125, 1500]
|
||||
assert result.positions[1].holdings == {"A": 25}
|
||||
assert len(result.trades_frame) == 1
|
||||
|
||||
|
||||
def test_rejected_entry_does_not_create_a_position_for_policy():
|
||||
holdings = []
|
||||
|
||||
def decide(position):
|
||||
holdings.append(dict(position.holdings))
|
||||
return {"A": 1} if position.date == "d1" else None
|
||||
|
||||
result = simulate_daily_ledger_with_audit(
|
||||
[],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 10})],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 10})],
|
||||
1000,
|
||||
ExecutionConfig(min_trade_amount=2000),
|
||||
decision_policy=decide,
|
||||
)
|
||||
assert holdings == [{}, {}]
|
||||
assert result.trades_frame.empty
|
||||
|
||||
|
||||
def test_policy_and_fixed_schedule_cannot_be_mixed():
|
||||
with pytest.raises(ValueError, match="fixed"):
|
||||
simulate_daily_ledger_with_audit(
|
||||
[("d1", {"A": 1})],
|
||||
[("d1", {"A": 10})],
|
||||
[("d1", {"A": 10})],
|
||||
1000,
|
||||
decision_policy=lambda p: None,
|
||||
)
|
||||
@@ -1,938 +0,0 @@
|
||||
"""Versioned factor-definition and factor-set contract conformance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import FrozenInstanceError
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
ContractErrorCode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorContractError,
|
||||
FactorDefinition,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
HistoricalAvailability,
|
||||
InputBinding,
|
||||
LegacyFactorBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
TypedParameter,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
validate_factor_catalog,
|
||||
)
|
||||
from quant_engine.governed_pipeline import (
|
||||
FactorVersion,
|
||||
bind_legacy_factor,
|
||||
project_legacy_factor,
|
||||
)
|
||||
|
||||
|
||||
FIXTURE_PATH = Path(__file__).parent / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
|
||||
|
||||
def _golden() -> dict[str, Any]:
|
||||
loaded = json.loads(FIXTURE_PATH.read_text(encoding="utf-8"))
|
||||
assert isinstance(loaded, dict)
|
||||
return loaded
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _reidentify(item: dict[str, Any], field: str, prefix: str) -> None:
|
||||
payload = {key: value for key, value in item.items() if key != field}
|
||||
item[field] = f"{prefix}{hashlib.sha256(canonical_json_bytes(payload)).hexdigest()}"
|
||||
|
||||
|
||||
def _snapshot_and_foundation(
|
||||
fixture: dict[str, Any] | None = None,
|
||||
) -> tuple[DatasetSnapshotEnvelope, DataFoundationEnvelope]:
|
||||
source = _golden() if fixture is None else fixture
|
||||
return (
|
||||
DatasetSnapshotEnvelope.from_dict(source["dataset_snapshot"]),
|
||||
DataFoundationEnvelope.from_dict(source["data_foundation"]),
|
||||
)
|
||||
|
||||
|
||||
def _definition(
|
||||
*,
|
||||
inputs: tuple[FactorInput, ...] | None = None,
|
||||
**overrides: Any,
|
||||
) -> FactorDefinition:
|
||||
factor_inputs = inputs or (
|
||||
FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume")),
|
||||
)
|
||||
arguments: dict[str, Any] = {
|
||||
"factor_id": "alpha_005",
|
||||
"version": "1.0.0",
|
||||
"formula": "correlation(close, volume, 10)",
|
||||
"parameters": {},
|
||||
"implementation_digest": "sha256:" + "1" * 64,
|
||||
"input_schema_digest": factor_input_schema_digest(factor_inputs),
|
||||
"inputs": factor_inputs,
|
||||
"valid_from": "2026-01-01T00:00:00.000000Z",
|
||||
"valid_until": "2027-01-01T00:00:00Z",
|
||||
"warmup_sessions": 10,
|
||||
"lag_sessions": 1,
|
||||
"producer": ProducerIdentity("quant_engine", "1.0.0"),
|
||||
"code_revision": "c" * 40,
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return FactorDefinition.create(**arguments)
|
||||
|
||||
|
||||
def _golden_definition() -> FactorDefinition:
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
return factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
|
||||
def _factor_set_arguments(
|
||||
*,
|
||||
fixture: dict[str, Any] | None = None,
|
||||
snapshot: DatasetSnapshotEnvelope | None = None,
|
||||
foundation: DataFoundationEnvelope | None = None,
|
||||
definition: FactorDefinition | None = None,
|
||||
) -> dict[str, Any]:
|
||||
source = _golden() if fixture is None else fixture
|
||||
if snapshot is None or foundation is None:
|
||||
parsed_snapshot, parsed_foundation = _snapshot_and_foundation(source)
|
||||
snapshot = snapshot or parsed_snapshot
|
||||
foundation = foundation or parsed_foundation
|
||||
selected_definition = definition or _golden_definition()
|
||||
output_schema_bytes = canonical_json_bytes(source["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(source["output_content"])
|
||||
artifact = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
return {
|
||||
"definitions": (selected_definition,),
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"selected_view_ref_ids": (VIEW_REF_ID,),
|
||||
"input_bindings": (
|
||||
InputBinding(
|
||||
selected_definition.definition_id,
|
||||
"market",
|
||||
VIEW_REF_ID,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
"view_availability": (
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
"output_quality": OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
"output_coverage": OutputCoverage(
|
||||
"complete",
|
||||
1,
|
||||
1,
|
||||
"row",
|
||||
"alpha_005.cn_a",
|
||||
"sha256:" + "4" * 64,
|
||||
),
|
||||
"output_schema_bytes": output_schema_bytes,
|
||||
"output_content_bytes": output_content_bytes,
|
||||
"output_artifact_ref": artifact,
|
||||
"availability_mode": AvailabilityMode.AS_AVAILABLE,
|
||||
"evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"computed_at": "2026-01-03T10:15:00Z",
|
||||
"artifact_available_at": "2026-01-03T10:20:00Z",
|
||||
"producer": ProducerIdentity("quant_engine", "1.0.0"),
|
||||
"code_revision": "c" * 40,
|
||||
"actor": ActorIdentity("service", "factor_worker_v1"),
|
||||
"correlation_id": "research_run_001",
|
||||
"causation": Causation("foundation", foundation.foundation_id),
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def _factor_set(**overrides: Any) -> FactorSetRef:
|
||||
arguments = _factor_set_arguments()
|
||||
arguments.update(overrides)
|
||||
return FactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[FactorContractError],
|
||||
code: ContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def _mutate_artifact_schema_binding(value: dict[str, Any]) -> None:
|
||||
artifact = value["output_artifact_ref"]
|
||||
artifact["schema_digest"] = "sha256:" + "0" * 64
|
||||
_reidentify(artifact, "artifact_id", "rhfactoroutputv1:sha256:")
|
||||
|
||||
|
||||
def test_golden_contracts_are_content_addressed_round_trippable_and_deeply_immutable() -> None:
|
||||
fixture = _golden()
|
||||
original_snapshot = copy.deepcopy(fixture["dataset_snapshot"])
|
||||
original_foundation = copy.deepcopy(fixture["data_foundation"])
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
definition = _golden_definition()
|
||||
factor_set = FactorSetRef.create(**_factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition))
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id="factor:demo-momentum",
|
||||
legacy_version="1.0.0",
|
||||
legacy_definition_sha256="b" * 64,
|
||||
legacy_dataset_schema_version="1.0.0",
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
|
||||
assert snapshot.pit_cutoff == "2026-01-02T07:01:00Z"
|
||||
assert foundation.pit_cutoff == factor_set.pit_cutoff == "2026-01-03T00:00:00Z"
|
||||
assert snapshot.pit_cutoff != foundation.pit_cutoff
|
||||
assert definition.definition_id == fixture["expected"]["definition_id"]
|
||||
assert definition.input_schema_digest == fixture["expected"]["input_schema_digest"]
|
||||
assert factor_set.factor_set_id == fixture["expected"]["factor_set_id"]
|
||||
assert factor_set.output_artifact_ref.artifact_id == fixture["expected"]["output_artifact_id"]
|
||||
assert binding.binding_id == fixture["expected"]["legacy_binding_id"]
|
||||
assert not definition.to_json().endswith("\n")
|
||||
assert not factor_set.to_json().endswith("\n")
|
||||
assert FactorDefinition.from_json(definition.to_json()) == definition
|
||||
|
||||
reparsed = FactorSetRef.from_json(
|
||||
factor_set.to_json(),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
output_schema_bytes=canonical_json_bytes(fixture["output_schema"]),
|
||||
output_content_bytes=canonical_json_bytes(fixture["output_content"]),
|
||||
)
|
||||
reference_only = FactorSetRef.from_json(
|
||||
factor_set.to_json(),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
assert reparsed.factor_set_id == factor_set.factor_set_id
|
||||
assert reparsed.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
assert reference_only.payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
|
||||
fixture["dataset_snapshot"]["descriptor"]["dataset"]["dimensions"].append("forbidden")
|
||||
fixture["data_foundation"]["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
|
||||
assert snapshot.to_dict() == original_snapshot
|
||||
assert foundation.to_dict() == original_foundation
|
||||
returned = snapshot.to_dict()
|
||||
returned["descriptor"]["dataset"]["dimensions"].append("also_forbidden")
|
||||
assert snapshot.to_dict() == original_snapshot
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
snapshot.snapshot_id = "rhdsv1:sha256:" + "0" * 64 # type: ignore[misc]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("variant", ["whitespace", "key_order"])
|
||||
def test_contract_decoders_reject_non_canonical_json(variant: str) -> None:
|
||||
fixture = _golden()
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
definition = _golden_definition()
|
||||
factor_set = FactorSetRef.create(
|
||||
**_factor_set_arguments(
|
||||
fixture=fixture,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
definition=definition,
|
||||
)
|
||||
)
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id="factor:demo-momentum",
|
||||
legacy_version="1.0.0",
|
||||
legacy_definition_sha256="b" * 64,
|
||||
legacy_dataset_schema_version="1.0.0",
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
|
||||
def non_canonical(value: str) -> str:
|
||||
if variant == "whitespace":
|
||||
return value + "\n"
|
||||
loaded = json.loads(value)
|
||||
reversed_items = dict(reversed(tuple(loaded.items())))
|
||||
return json.dumps(reversed_items, ensure_ascii=False, separators=(",", ":"))
|
||||
|
||||
decoders = (
|
||||
lambda value: FactorDefinition.from_json(value),
|
||||
lambda value: FactorSetRef.from_json(
|
||||
value,
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
),
|
||||
lambda value: LegacyFactorBinding.from_json(value, definition=definition),
|
||||
)
|
||||
for decoder, encoded in zip(
|
||||
decoders,
|
||||
(definition.to_json(), factor_set.to_json(), binding.to_json()),
|
||||
strict=True,
|
||||
):
|
||||
with pytest.raises(FactorContractError) as exc_info:
|
||||
decoder(non_canonical(encoded))
|
||||
assert exc_info.value.code is ContractErrorCode.INVALID_FORMAT
|
||||
assert exc_info.value.path == "$"
|
||||
|
||||
|
||||
def test_factor_definition_identity_is_order_independent_where_semantics_are_unordered() -> None:
|
||||
first_input = FactorInput("prices", "sha256:" + "6" * 64, ("close",))
|
||||
second_input = FactorInput("volumes", "sha256:" + "7" * 64, ("volume",))
|
||||
inputs = (first_input, second_input)
|
||||
parameters_a = {
|
||||
"window": TypedParameter("integer", 10),
|
||||
"weights": TypedParameter("json", {"fast": [1, 2], "slow": [3, 4]}),
|
||||
}
|
||||
parameters_b = {
|
||||
"weights": TypedParameter("json", {"slow": [3, 4], "fast": [1, 2]}),
|
||||
"window": TypedParameter("integer", 10),
|
||||
}
|
||||
first = _definition(
|
||||
inputs=inputs,
|
||||
parameters=parameters_a,
|
||||
input_schema_digest=factor_input_schema_digest(inputs),
|
||||
)
|
||||
second = _definition(
|
||||
inputs=tuple(reversed(inputs)),
|
||||
parameters=parameters_b,
|
||||
input_schema_digest=factor_input_schema_digest(tuple(reversed(inputs))),
|
||||
)
|
||||
assert first.definition_id == second.definition_id
|
||||
assert first.to_json() == second.to_json()
|
||||
|
||||
semantic_changes = (
|
||||
_definition(factor_id="alpha_006"),
|
||||
_definition(version="1.0.1"),
|
||||
_definition(formula="correlation(close, volume, 11)"),
|
||||
_definition(parameters={"window": TypedParameter("integer", 10)}),
|
||||
_definition(implementation_digest="sha256:" + "9" * 64),
|
||||
_definition(valid_until="2027-01-02T00:00:00Z"),
|
||||
_definition(warmup_sessions=11),
|
||||
_definition(lag_sessions=2),
|
||||
_definition(producer=ProducerIdentity("quant_engine", "1.0.1")),
|
||||
_definition(code_revision="d" * 40),
|
||||
)
|
||||
assert all(changed.definition_id != _golden_definition().definition_id for changed in semantic_changes)
|
||||
assert len({changed.definition_id for changed in semantic_changes}) == len(semantic_changes)
|
||||
|
||||
|
||||
def test_parameter_types_decimal_profile_and_detached_nested_values_are_strict() -> None:
|
||||
nested = {"ordered": [1, {"flag": True}]}
|
||||
parameter = TypedParameter("json", nested)
|
||||
nested["ordered"].append(2)
|
||||
definition = _definition(parameters={"payload": parameter})
|
||||
assert definition.to_dict()["parameters"]["payload"]["value"] == {
|
||||
"ordered": [1, {"flag": True}]
|
||||
}
|
||||
integer_definition = _definition(parameters={"value": TypedParameter("integer", 1)})
|
||||
string_definition = _definition(parameters={"value": TypedParameter("string", "1")})
|
||||
assert integer_definition.definition_id != string_definition.definition_id
|
||||
|
||||
for parameter_type, value, code in (
|
||||
("decimal", "1.0", ContractErrorCode.INVALID_FORMAT),
|
||||
("decimal", "1e3", ContractErrorCode.INVALID_FORMAT),
|
||||
("decimal", "-0", ContractErrorCode.INVALID_FORMAT),
|
||||
("integer", True, ContractErrorCode.TYPE_ERROR),
|
||||
("json", 1.5, ContractErrorCode.TYPE_ERROR),
|
||||
("json", {"é": "bad-key"}, ContractErrorCode.INVALID_FORMAT),
|
||||
("json", 9_007_199_254_740_992, ContractErrorCode.INVALID_VALUE),
|
||||
):
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
TypedParameter(parameter_type, value)
|
||||
assert error.value.code is code
|
||||
assert TypedParameter("decimal", "10.25").to_dict()["value"] == "10.25"
|
||||
|
||||
|
||||
def test_catalog_rejects_duplicate_and_overlapping_logical_validity_but_allows_adjacency() -> None:
|
||||
base = _golden_definition()
|
||||
adjacent = _definition(valid_from="2027-01-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
|
||||
assert len(validate_factor_catalog((adjacent, base))) == 2
|
||||
with pytest.raises(FactorContractError) as duplicate:
|
||||
validate_factor_catalog((base, base))
|
||||
_assert_error(duplicate, ContractErrorCode.INVALID_VALUE, "$.definitions")
|
||||
overlapping = _definition(valid_from="2026-06-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
|
||||
with pytest.raises(FactorContractError) as overlap:
|
||||
validate_factor_catalog((base, overlapping))
|
||||
_assert_error(overlap, ContractErrorCode.TIME_ORDER_VIOLATION, "$.definitions")
|
||||
|
||||
|
||||
def test_upstream_contracts_reject_unknown_fields_identity_forgery_and_unqualified_input() -> None:
|
||||
unknown = _golden()["dataset_snapshot"]
|
||||
unknown["provider"] = "forbidden"
|
||||
with pytest.raises(FactorContractError) as unknown_error:
|
||||
DatasetSnapshotEnvelope.from_dict(unknown)
|
||||
_assert_error(unknown_error, ContractErrorCode.UNKNOWN_FIELD, "$.provider")
|
||||
|
||||
forged = _golden()["data_foundation"]
|
||||
forged["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
|
||||
with pytest.raises(FactorContractError) as forged_error:
|
||||
DataFoundationEnvelope.from_dict(forged)
|
||||
assert forged_error.value.code is ContractErrorCode.IDENTITY_MISMATCH
|
||||
assert forged_error.value.path.endswith("view_ref_id")
|
||||
|
||||
rejected_source = _golden()
|
||||
rejected_source["dataset_snapshot"]["descriptor"]["qualification"]["status"] = "rejected"
|
||||
_reidentify(rejected_source["dataset_snapshot"], "snapshot_id", "rhdsv1:sha256:")
|
||||
rejected_snapshot = DatasetSnapshotEnvelope.from_dict(rejected_source["dataset_snapshot"])
|
||||
_, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as rejected_error:
|
||||
FactorSetRef.create(
|
||||
**_factor_set_arguments(snapshot=rejected_snapshot, foundation=foundation)
|
||||
)
|
||||
_assert_error(
|
||||
rejected_error,
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
"$.dataset_snapshot.descriptor.qualification",
|
||||
)
|
||||
|
||||
|
||||
def test_foundation_rejects_future_knowledge_and_per_view_calendar_borrowing() -> None:
|
||||
future = _golden()["data_foundation"]
|
||||
action = future["corporate_action_revisions"][0]
|
||||
old_action_id = action["action_revision_id"]
|
||||
action["knowledge_time"] = "2026-01-03T00:00:01Z"
|
||||
_reidentify(action, "action_revision_id", "rhcav1:sha256:")
|
||||
future["standardized_views"][0]["corporate_action_revision_ids"] = [action["action_revision_id"]]
|
||||
lineage = next(item for item in future["revision_lineage"] if item["revision_id"] == old_action_id)
|
||||
lineage["revision_id"] = action["action_revision_id"]
|
||||
lineage["knowledge_time"] = action["knowledge_time"]
|
||||
_reidentify(future["standardized_views"][0], "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(future, "foundation_id", "rhdfv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as future_error:
|
||||
DataFoundationEnvelope.from_dict(future)
|
||||
_assert_error(
|
||||
future_error,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
"$.revision_lineage.knowledge_time",
|
||||
)
|
||||
|
||||
uncovered = _golden()["data_foundation"]
|
||||
original_route_id = uncovered["instrument_routes"][0]["route_revision_id"]
|
||||
second_calendar = copy.deepcopy(uncovered["trading_calendar_revisions"][0])
|
||||
second_calendar["calendar_id"] = "rhcalendar:99990000111122223333444455556666"
|
||||
_reidentify(second_calendar, "calendar_revision_id", "rhcalv1:sha256:")
|
||||
uncovered["trading_calendar_revisions"].append(second_calendar)
|
||||
route = uncovered["instrument_routes"][0]
|
||||
route["calendar_id"] = second_calendar["calendar_id"]
|
||||
_reidentify(route, "route_revision_id", "rhroutev1:sha256:")
|
||||
route_lineage = next(item for item in uncovered["revision_lineage"] if item["revision_id"] == original_route_id)
|
||||
route_lineage["revision_id"] = route["route_revision_id"]
|
||||
uncovered["revision_lineage"].append(
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": second_calendar["calendar_revision_id"],
|
||||
"revision_number": 1,
|
||||
"knowledge_time": second_calendar["knowledge_time"],
|
||||
"evidence_digest": second_calendar["evidence_digest"],
|
||||
}
|
||||
)
|
||||
view = uncovered["standardized_views"][0]
|
||||
view["instrument_route_revision_ids"] = [route["route_revision_id"]]
|
||||
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(uncovered, "foundation_id", "rhdfv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as calendar_error:
|
||||
DataFoundationEnvelope.from_dict(uncovered)
|
||||
assert calendar_error.value.code is ContractErrorCode.INPUT_CLOSURE_VIOLATION
|
||||
assert "selected route calendar" in calendar_error.value.detail
|
||||
|
||||
|
||||
def _replay_fixture() -> dict[str, Any]:
|
||||
fixture = _golden()
|
||||
snapshot = fixture["dataset_snapshot"]
|
||||
snapshot["descriptor"]["published_at"] = "2026-01-04T00:00:00Z"
|
||||
_reidentify(snapshot, "snapshot_id", "rhdsv1:sha256:")
|
||||
foundation = fixture["data_foundation"]
|
||||
foundation["dataset_snapshot_id"] = snapshot["snapshot_id"]
|
||||
for view in foundation["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot["snapshot_id"]
|
||||
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
|
||||
return fixture
|
||||
|
||||
|
||||
def test_as_available_and_retrospective_replay_keep_distinct_time_claims() -> None:
|
||||
as_available = _factor_set()
|
||||
assert as_available.historical_availability is HistoricalAvailability.DECLARED_AS_AVAILABLE
|
||||
|
||||
replay_source = _replay_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(replay_source)
|
||||
replay_view_id = next(iter(foundation.views))
|
||||
arguments = _factor_set_arguments(
|
||||
fixture=replay_source,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
arguments.update(
|
||||
selected_view_ref_ids=(replay_view_id,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
replay_view_id,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(replay_view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
computed_at="2026-01-04T00:20:00Z",
|
||||
artifact_available_at="2026-01-04T00:25:00Z",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
replay = FactorSetRef.create(**arguments)
|
||||
assert replay.evaluation_at == "2026-01-03T11:00:00Z"
|
||||
assert replay.computed_at == "2026-01-04T00:20:00Z"
|
||||
assert replay.historical_availability is HistoricalAvailability.NOT_ESTABLISHED
|
||||
|
||||
replay_source_args = _factor_set_arguments(
|
||||
fixture=replay_source,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
replay_source_args.update(
|
||||
selected_view_ref_ids=(replay_view_id,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
replay_source_args["definitions"][0].definition_id,
|
||||
"market",
|
||||
replay_view_id,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(replay_view_id, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
with pytest.raises(FactorContractError) as late_publication:
|
||||
FactorSetRef.create(**replay_source_args)
|
||||
_assert_error(
|
||||
late_publication,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
"$.dataset_snapshot.descriptor.published_at",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("overrides", "path"),
|
||||
[
|
||||
({"view_availability": (ViewAvailability(VIEW_REF_ID, "2026-01-03T00:00:01Z", "sha256:" + "2" * 64),)}, "$.view_availability"),
|
||||
({"computed_at": "2026-01-02T23:40:00Z"}, "$.computed_at"),
|
||||
({"artifact_available_at": "2026-01-03T10:14:00Z"}, "$.artifact_available_at"),
|
||||
({"artifact_available_at": "2026-01-03T11:00:01Z"}, "$.artifact_available_at"),
|
||||
({"evaluation_at": "2026-01-03T11:00:00"}, "$.evaluation_at"),
|
||||
],
|
||||
)
|
||||
def test_as_available_time_failures_are_typed(overrides: dict[str, Any], path: str) -> None:
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
_factor_set(**overrides)
|
||||
assert error.value.code in {
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
}
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_replay_rejects_backdating_and_historical_availability_promotion() -> None:
|
||||
source = _replay_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(source)
|
||||
view_id = next(iter(foundation.views))
|
||||
arguments = _factor_set_arguments(fixture=source, snapshot=snapshot, foundation=foundation)
|
||||
definition = arguments["definitions"][0]
|
||||
arguments.update(
|
||||
selected_view_ref_ids=(view_id,),
|
||||
input_bindings=(InputBinding(definition.definition_id, "market", view_id, VIEW_SCHEMA_DIGEST),),
|
||||
view_availability=(ViewAvailability(view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),),
|
||||
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
computed_at="2026-01-04T00:20:00Z",
|
||||
artifact_available_at="2026-01-04T00:25:00Z",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
replay = FactorSetRef.create(**arguments)
|
||||
promoted = replay.to_dict()
|
||||
promoted["historical_availability"] = "declared_as_available"
|
||||
_reidentify(promoted, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as promotion_error:
|
||||
FactorSetRef.from_dict(
|
||||
promoted,
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
promotion_error,
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
"$.historical_availability",
|
||||
)
|
||||
arguments["computed_at"] = "2026-01-03T11:30:00Z"
|
||||
with pytest.raises(FactorContractError) as backdated_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(backdated_error, ContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
|
||||
|
||||
|
||||
def _multi_view_fixture() -> tuple[dict[str, Any], str]:
|
||||
fixture = _golden()
|
||||
foundation = fixture["data_foundation"]
|
||||
second = copy.deepcopy(foundation["standardized_views"][0])
|
||||
second["view_id"] = "rhview:11111111222222223333333344444444"
|
||||
second["schema_digest"] = "sha256:" + "6" * 64
|
||||
second["content_digest"] = "sha256:" + "7" * 64
|
||||
second["transformation_digest"] = "sha256:" + "8" * 64
|
||||
_reidentify(second, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
foundation["standardized_views"].append(second)
|
||||
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
|
||||
return fixture, second["view_ref_id"]
|
||||
|
||||
|
||||
def test_multi_input_mapping_requires_exact_consumption_closure_and_is_order_independent() -> None:
|
||||
fixture, second_view_id = _multi_view_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
inputs = (
|
||||
FactorInput("prices", VIEW_SCHEMA_DIGEST, ("close",)),
|
||||
FactorInput("volumes", "sha256:" + "6" * 64, ("volume",)),
|
||||
)
|
||||
definition = _definition(
|
||||
inputs=inputs,
|
||||
formula="correlation(close, volume, 10)",
|
||||
input_schema_digest=factor_input_schema_digest(inputs),
|
||||
)
|
||||
first_binding = InputBinding(definition.definition_id, "prices", VIEW_REF_ID, VIEW_SCHEMA_DIGEST)
|
||||
second_binding = InputBinding(definition.definition_id, "volumes", second_view_id, "sha256:" + "6" * 64)
|
||||
first_availability = ViewAvailability(VIEW_REF_ID, "2026-01-02T23:40:00Z", "sha256:" + "2" * 64)
|
||||
second_availability = ViewAvailability(second_view_id, "2026-01-02T23:50:00Z", "sha256:" + "6" * 64)
|
||||
base = _factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition)
|
||||
base.update(
|
||||
selected_view_ref_ids=(VIEW_REF_ID, second_view_id),
|
||||
input_bindings=(first_binding, second_binding),
|
||||
view_availability=(first_availability, second_availability),
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
first = FactorSetRef.create(**base)
|
||||
reordered = dict(base)
|
||||
reordered.update(
|
||||
selected_view_ref_ids=(second_view_id, VIEW_REF_ID),
|
||||
input_bindings=(second_binding, first_binding),
|
||||
view_availability=(second_availability, first_availability),
|
||||
)
|
||||
assert FactorSetRef.create(**reordered).factor_set_id == first.factor_set_id
|
||||
|
||||
for invalid_bindings, invalid_views in (
|
||||
((first_binding,), (VIEW_REF_ID, second_view_id)),
|
||||
((first_binding, second_binding), (VIEW_REF_ID,)),
|
||||
((first_binding, second_binding), (VIEW_REF_ID, second_view_id, VIEW_REF_ID)),
|
||||
):
|
||||
invalid = dict(base)
|
||||
invalid.update(input_bindings=invalid_bindings, selected_view_ref_ids=invalid_views)
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.create(**invalid)
|
||||
assert error.value.code in {
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
ContractErrorCode.INVALID_VALUE,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("mutate", "code", "path"),
|
||||
[
|
||||
(lambda value: value["producer"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.producer.id"),
|
||||
(lambda value: value["producer"].pop("version"), ContractErrorCode.MISSING_FIELD, "$.producer.version"),
|
||||
(lambda value: value["producer"].update(id="other_engine"), ContractErrorCode.LINEAGE_VIOLATION, "$.producer.id"),
|
||||
(lambda value: value["producer"].update(version="latest"), ContractErrorCode.INVALID_FORMAT, "$.producer.version"),
|
||||
(lambda value: value.update(code_revision="bad"), ContractErrorCode.INVALID_FORMAT, "$.code_revision"),
|
||||
(lambda value: value["actor"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.actor.kind"),
|
||||
(lambda value: value["actor"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.actor.id"),
|
||||
(lambda value: value["actor"].update(kind="robot"), ContractErrorCode.INVALID_VALUE, "$.actor.kind"),
|
||||
(lambda value: value["actor"].update(id="latest"), ContractErrorCode.INVALID_VALUE, "$.actor.id"),
|
||||
(lambda value: value.pop("correlation_id"), ContractErrorCode.MISSING_FIELD, "$.correlation_id"),
|
||||
(lambda value: value.update(correlation_id="latest"), ContractErrorCode.INVALID_VALUE, "$.correlation_id"),
|
||||
(lambda value: value["causation"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.causation.kind"),
|
||||
(lambda value: value["causation"].update(kind="run"), ContractErrorCode.INVALID_VALUE, "$.causation.kind"),
|
||||
(lambda value: value["causation"].update(id="rhdfv1:sha256:" + "0" * 64), ContractErrorCode.LINEAGE_VIOLATION, "$.causation.id"),
|
||||
(lambda value: value.pop("output_artifact_ref"), ContractErrorCode.MISSING_FIELD, "$.output_artifact_ref"),
|
||||
(lambda value: value["output_artifact_ref"].update(artifact_id="rhfactoroutputv1:sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.output_artifact_ref.artifact_id"),
|
||||
(_mutate_artifact_schema_binding, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref"),
|
||||
(lambda value: value.update(availability_mode="implicit_fallback"), ContractErrorCode.INVALID_VALUE, "$.availability_mode"),
|
||||
(lambda value: value.pop("computed_at"), ContractErrorCode.MISSING_FIELD, "$.computed_at"),
|
||||
(lambda value: value.update(decision_eligible=True), ContractErrorCode.READINESS_ESCALATION, "$.decision_eligible"),
|
||||
(lambda value: value.update(evidence_scope="real_data"), ContractErrorCode.READINESS_ESCALATION, "$.evidence_scope"),
|
||||
(lambda value: value["upstream_evidence"].update(qualification_evidence_digest="sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.upstream_evidence"),
|
||||
],
|
||||
)
|
||||
def test_lineage_artifact_and_readiness_fields_have_independent_typed_negatives(
|
||||
mutate: Callable[[dict[str, Any]], Any],
|
||||
code: ContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
factor_set = _factor_set()
|
||||
value = factor_set.to_dict()
|
||||
mutate(value)
|
||||
if "factor_set_id" in value:
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(error, code, path)
|
||||
|
||||
|
||||
def test_output_schema_content_bytes_cannot_be_swapped_or_forged() -> None:
|
||||
factor_set = _factor_set()
|
||||
fixture = _golden()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
with pytest.raises(FactorContractError) as swapped:
|
||||
FactorSetRef.from_dict(
|
||||
factor_set.to_dict(),
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
output_schema_bytes=content_bytes,
|
||||
output_content_bytes=schema_bytes,
|
||||
)
|
||||
_assert_error(swapped, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref")
|
||||
with pytest.raises(FactorContractError) as noncanonical:
|
||||
FactorSetRef.create(
|
||||
**{
|
||||
**_factor_set_arguments(),
|
||||
"output_schema_bytes": json.dumps(fixture["output_schema"], indent=2).encode(),
|
||||
}
|
||||
)
|
||||
_assert_error(noncanonical, ContractErrorCode.INVALID_FORMAT, "$.output_schema_bytes")
|
||||
|
||||
|
||||
def test_unsuccessful_output_quality_or_coverage_cannot_form_a_factor_set() -> None:
|
||||
with pytest.raises(FactorContractError) as failed_quality:
|
||||
_factor_set(
|
||||
output_quality=OutputQuality(
|
||||
"failed",
|
||||
(OutputQualityCheck("finite_values", "failed", "sha256:" + "3" * 64),),
|
||||
)
|
||||
)
|
||||
_assert_error(failed_quality, ContractErrorCode.INVALID_VALUE, "$.output_quality")
|
||||
|
||||
for coverage in (
|
||||
OutputCoverage("incomplete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
|
||||
OutputCoverage("complete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
|
||||
):
|
||||
with pytest.raises(FactorContractError) as incomplete:
|
||||
_factor_set(output_coverage=coverage)
|
||||
_assert_error(incomplete, ContractErrorCode.INVALID_VALUE, "$.output_coverage")
|
||||
|
||||
|
||||
def test_external_snapshot_definition_and_view_references_cannot_be_substituted() -> None:
|
||||
factor_set = _factor_set()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
value = factor_set.to_dict()
|
||||
value["dataset_snapshot_id"] = "rhdsv1:sha256:" + "0" * 64
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as snapshot_error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
snapshot_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.dataset_snapshot_id",
|
||||
)
|
||||
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = ["rhfactorv1:sha256:" + "0" * 64]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as definition_error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
definition_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.definition_ids",
|
||||
)
|
||||
|
||||
arguments = _factor_set_arguments()
|
||||
arguments["selected_view_ref_ids"] = ("rhviewrefv1:sha256:" + "0" * 64,)
|
||||
with pytest.raises(FactorContractError) as view_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(
|
||||
view_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.selected_view_ref_ids",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"invalid_definition_id",
|
||||
[
|
||||
{"unexpected": "object"},
|
||||
["array"],
|
||||
42,
|
||||
True,
|
||||
None,
|
||||
],
|
||||
)
|
||||
def test_factor_set_ref_definition_ids_reject_non_string_types(
|
||||
invalid_definition_id: Any,
|
||||
) -> None:
|
||||
factor_set = _factor_set()
|
||||
definition = _golden_definition()
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = [definition.definition_id, invalid_definition_id]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_json(
|
||||
canonical_json_bytes(value),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
|
||||
_assert_error(error, ContractErrorCode.TYPE_ERROR, "$.definition_ids[1]")
|
||||
|
||||
|
||||
def test_factor_set_ref_definition_ids_still_reject_duplicate_strings() -> None:
|
||||
factor_set = _factor_set()
|
||||
definition = _golden_definition()
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = [definition.definition_id, definition.definition_id]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_json(
|
||||
canonical_json_bytes(value),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
|
||||
_assert_error(error, ContractErrorCode.INVALID_VALUE, "$.definition_ids")
|
||||
|
||||
|
||||
def test_factor_set_parent_requires_exact_identity_and_correlation() -> None:
|
||||
parent = _factor_set()
|
||||
child_arguments = _factor_set_arguments()
|
||||
child_arguments.update(
|
||||
output_content_bytes=canonical_json_bytes({"rows": [{"value": "0.250"}]}),
|
||||
causation=Causation("factor_set", parent.factor_set_id),
|
||||
parent=parent,
|
||||
)
|
||||
child_arguments["output_artifact_ref"] = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(child_arguments["output_schema_bytes"]),
|
||||
content_digest=_sha256(child_arguments["output_content_bytes"]),
|
||||
)
|
||||
child = FactorSetRef.create(**child_arguments)
|
||||
assert child.causation.id == parent.factor_set_id
|
||||
missing_parent = child.to_dict()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as missing_error:
|
||||
FactorSetRef.from_dict(
|
||||
missing_parent,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(missing_error, ContractErrorCode.LINEAGE_VIOLATION, "$.causation")
|
||||
wrong_correlation = dict(child_arguments)
|
||||
wrong_correlation["correlation_id"] = "different_run"
|
||||
with pytest.raises(FactorContractError) as correlation_error:
|
||||
FactorSetRef.create(**wrong_correlation)
|
||||
_assert_error(correlation_error, ContractErrorCode.LINEAGE_VIOLATION, "$.correlation_id")
|
||||
|
||||
|
||||
def test_legacy_bridge_is_explicit_lossy_and_preserves_all_four_historical_fields() -> None:
|
||||
definition = _golden_definition()
|
||||
legacy = FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="1.0.0",
|
||||
)
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id=legacy.factor_id,
|
||||
legacy_version=legacy.version,
|
||||
legacy_definition_sha256=legacy.definition_sha256,
|
||||
legacy_dataset_schema_version=legacy.dataset_schema_version,
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
assert bind_legacy_factor(legacy, definition, binding) is definition
|
||||
assert project_legacy_factor(definition, binding) == legacy
|
||||
assert legacy.version_id == "factor:demo-momentum@1.0.0"
|
||||
assert legacy.definition_sha256 != definition.definition_id.rsplit(":", maxsplit=1)[-1]
|
||||
assert LegacyFactorBinding.from_json(binding.to_json(), definition=definition) == binding
|
||||
|
||||
mismatched = FactorVersion(
|
||||
factor_id="factor:different",
|
||||
version=legacy.version,
|
||||
definition_sha256=legacy.definition_sha256,
|
||||
dataset_schema_version=legacy.dataset_schema_version,
|
||||
)
|
||||
with pytest.raises(FactorContractError) as mismatch_error:
|
||||
bind_legacy_factor(mismatched, definition, binding)
|
||||
_assert_error(mismatch_error, ContractErrorCode.LEGACY_BINDING_MISMATCH, "$.binding")
|
||||
|
||||
|
||||
def test_bare_legacy_factor_or_id_cannot_enter_factor_set_contract() -> None:
|
||||
legacy = FactorVersion("factor:demo-momentum", "1.0.0", "b" * 64, "1.0.0")
|
||||
arguments = _factor_set_arguments()
|
||||
arguments["definitions"] = (legacy,)
|
||||
with pytest.raises(FactorContractError) as legacy_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(legacy_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
|
||||
arguments["definitions"] = (legacy.version_id,)
|
||||
with pytest.raises(FactorContractError) as id_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(id_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
|
||||
@@ -1,318 +0,0 @@
|
||||
"""Deterministic diagnostics contracts with independently checkable samples."""
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
from math import sqrt
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
|
||||
def api():
|
||||
return importlib.import_module("quant_engine.factor_diagnostics")
|
||||
|
||||
|
||||
def panel(values, *, days=1, assets=None):
|
||||
assets = assets or ["A", "B", "C"]
|
||||
index = pd.MultiIndex.from_product(
|
||||
[pd.date_range("2026-09-21", periods=days), assets], names=["date", "asset"]
|
||||
)
|
||||
return pd.DataFrame(values, index=index)
|
||||
|
||||
|
||||
def test_pairwise_matrix_keeps_complete_security_date_pairs():
|
||||
frame = panel({"pe": [None, 0, 1, 2, 3, 4, 5, 6],
|
||||
"pb": [1000, None, 1, 2, 3, 4, 5, 6]}, assets=list("ABCDEFGH"))
|
||||
original = frame.copy()
|
||||
result = api().correlation_matrix(frame)
|
||||
assert result["matrix"][0][1] == pytest.approx(1)
|
||||
assert result["n_pairs"] == [[7, 6], [6, 7]]
|
||||
assert result["status"][0][1] == "ok"
|
||||
assert result["method"]["aggregation"] == "pooled_asset_session_pairwise"
|
||||
assert result["method"]["positive_semidefinite_guaranteed"] is False
|
||||
pd.testing.assert_frame_equal(frame, original)
|
||||
|
||||
|
||||
def test_matrix_empty_constant_and_short_diagonals_are_undefined():
|
||||
frame = panel({"empty": [None] * 3, "constant": [0.05] * 3, "short": [1, 2, None]})
|
||||
result = api().correlation_matrix(frame)
|
||||
assert result["matrix"] == [[None] * 3 for _ in range(3)]
|
||||
assert result["n_pairs"][0][0] == 0
|
||||
assert result["status"][0][0] == "no_pairs"
|
||||
assert result["status"][1][1] == "constant_both"
|
||||
assert result["status"][2][2] == "insufficient_pairs"
|
||||
|
||||
|
||||
def test_rank_correlation_uses_average_ties_after_pairing():
|
||||
frame = panel({"f": [1, 1, 2, None], "r": [1, 2, 3, 999]}, assets=list("ABCD"))
|
||||
result = api().correlation_matrix(frame, method="spearman")
|
||||
assert result["matrix"][0][1] == pytest.approx(sqrt(3) / 2)
|
||||
assert result["n_pairs"][0][1] == 3
|
||||
|
||||
|
||||
def test_daily_ic_is_not_one_pooled_correlation():
|
||||
frame = panel({"f": [1, 2, 3, 101, 102, 103]}, days=2)
|
||||
returns = pd.Series([3, 2, 1, 101, 102, 103], index=frame.index)
|
||||
result = api().daily_ic(frame, returns)
|
||||
assert [r["ic"] for r in result["rows"]] == pytest.approx([-1, 1])
|
||||
assert result["summary"][0]["ic_mean"] == pytest.approx(0)
|
||||
assert result["summary"][0]["ic_std"] == pytest.approx(sqrt(2))
|
||||
assert result["summary"][0]["ic_ir"] == 0
|
||||
assert result["summary"][0]["ic_t"] == 0
|
||||
assert result["method"]["ir_annualized"] is False
|
||||
assert result["method"]["t_method"] == "naive_iid_unadjusted"
|
||||
assert result["method"]["serial_correlation_adjusted"] is False
|
||||
|
||||
|
||||
def test_daily_summary_matches_three_known_days_and_keeps_zero():
|
||||
frame = panel({"f": [1, 2, 3] * 3}, days=3)
|
||||
returns = pd.Series([3, 2, 1, 1, 0, 1, 1, 2, 3], index=frame.index)
|
||||
result = api().daily_ic(frame, returns)
|
||||
assert [r["ic"] for r in result["rows"]] == pytest.approx([-1, 0, 1])
|
||||
assert [r["rank_ic"] for r in result["rows"]] == pytest.approx([-1, 0, 1])
|
||||
summary = result["summary"][0]
|
||||
assert (summary["n_days"], summary["ic_valid_days"], summary["rank_ic_valid_days"]) == (3, 3, 3)
|
||||
assert summary["ic_std"] == pytest.approx(1)
|
||||
assert summary["ic_ir"] == pytest.approx(0, abs=1e-15)
|
||||
assert summary["ic_t"] == pytest.approx(0, abs=1e-15)
|
||||
assert summary["rank_ic_ir"] == pytest.approx(0, abs=1e-15)
|
||||
assert summary["rank_ic_t"] == pytest.approx(0, abs=1e-15)
|
||||
|
||||
|
||||
def test_undefined_days_and_factor_observation_domain_survive_alignment():
|
||||
frame = panel({"f": [1, 2, 3, None, None, None, 1, 2, 3]}, days=3)
|
||||
returns = pd.Series([1, 2, 3], index=frame.index[:3])
|
||||
extra = pd.Series([100], index=pd.MultiIndex.from_tuples(
|
||||
[(pd.Timestamp("2026-09-25"), "D")], names=["date", "asset"]))
|
||||
result = api().daily_ic(frame, pd.concat([returns, extra]))
|
||||
assert len(result["rows"]) == 3
|
||||
assert [r["n_pairs"] for r in result["rows"]] == [3, 0, 0]
|
||||
assert [r["n_observations"] for r in result["rows"]] == [3, 3, 3]
|
||||
assert [r["ic_status"] for r in result["rows"]] == ["ok", "no_pairs", "no_pairs"]
|
||||
summary = result["summary"][0]
|
||||
assert summary["ic_valid_days"] == 1
|
||||
assert summary["ic_missing_days"] == 2
|
||||
assert summary["ic_mean"] == pytest.approx(1)
|
||||
assert summary["ic_std"] is summary["ic_ir"] is summary["ic_t"] is None
|
||||
assert summary["ic_status"] == "insufficient_days"
|
||||
assert result["method"]["extra_return_keys"] == 1
|
||||
|
||||
|
||||
def test_daily_summary_of_constant_ic_has_no_ratio_statistics():
|
||||
frame = panel({"f": [1, 2, 3] * 3}, days=3)
|
||||
returns = pd.Series([1, 1, 2] * 3, index=frame.index)
|
||||
summary = api().daily_ic(frame, returns)["summary"][0]
|
||||
assert summary["ic_mean"] == pytest.approx(sqrt(3) / 2)
|
||||
assert summary["ic_std"] == 0
|
||||
assert summary["ic_ir"] is summary["ic_t"] is None
|
||||
assert summary["ic_status"] == "constant_values"
|
||||
assert summary["rank_ic_ir"] is summary["rank_ic_t"] is None
|
||||
|
||||
|
||||
def test_permutation_and_input_copies_do_not_change_diagnostics():
|
||||
frame = panel({"f": [1, 2, 3, 4, 5, 6]}, days=2)
|
||||
returns = pd.Series([1, 3, 2, 4, 6, 5], index=frame.index)
|
||||
original_frame, original_returns = frame.copy(), returns.copy()
|
||||
baseline = api().daily_ic(frame, returns)
|
||||
assert api().daily_ic(frame.iloc[::-1], returns.iloc[[2, 0, 4, 1, 5, 3]]) == baseline
|
||||
pd.testing.assert_frame_equal(frame, original_frame)
|
||||
pd.testing.assert_series_equal(returns, original_returns)
|
||||
|
||||
|
||||
def test_large_finite_inputs_reuse_scaled_correlation_without_overflow():
|
||||
frame = panel({"a": [-1e308, 0, 1e308], "b": [-1e307, 0, 1e307]})
|
||||
assert api().correlation_matrix(frame)["matrix"][0][1] == pytest.approx(1)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value", [True, "1.2", np.inf, -np.inf, object()])
|
||||
def test_diagnostics_reject_invalid_numbers(value):
|
||||
frame = panel({"f": [value, 2, 3]})
|
||||
with pytest.raises(ValueError, match=r"Values|Infinite"):
|
||||
api().correlation_matrix(frame)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("kind", ["duplicate_keys", "bad_names", "bad_date", "timezone", "empty_asset", "duplicate_factors"])
|
||||
def test_diagnostics_reject_ambiguous_keys(kind):
|
||||
frame = panel({"f": [1, 2, 3]})
|
||||
if kind == "duplicate_keys":
|
||||
frame.index = pd.MultiIndex.from_tuples([frame.index[0]] * 3, names=["date", "asset"])
|
||||
elif kind == "bad_names":
|
||||
frame.index.names = ["day", "asset"]
|
||||
elif kind == "bad_date":
|
||||
frame.index = pd.MultiIndex.from_product([["2026-09-21"], list("ABC")], names=["date", "asset"])
|
||||
elif kind == "timezone":
|
||||
frame.index = pd.MultiIndex.from_product([pd.date_range("2026-09-21", periods=1, tz="UTC"), list("ABC")], names=["date", "asset"])
|
||||
elif kind == "empty_asset":
|
||||
frame.index = pd.MultiIndex.from_product([pd.date_range("2026-09-21", periods=1), ["", "B", "C"]], names=["date", "asset"])
|
||||
else:
|
||||
frame = pd.concat([frame, frame], axis=1)
|
||||
with pytest.raises(ValueError, match=r"keys|dates|identifiers"):
|
||||
api().correlation_matrix(frame)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("minimum", [True, 0, 2, 3.1, "3"])
|
||||
def test_minimum_pairs_is_explicit_and_at_least_three(minimum):
|
||||
with pytest.raises(ValueError, match="min_pairs"):
|
||||
api().correlation_matrix(panel({"f": [1, 2, 3]}), min_pairs=minimum)
|
||||
|
||||
|
||||
def test_all_missing_and_empty_inputs_remain_observed_not_identity_or_zero():
|
||||
frame = panel({"f": [None] * 3})
|
||||
result = api().daily_ic(frame, pd.Series([None] * 3, index=frame.index))
|
||||
assert result["summary"][0]["ic_mean"] is None
|
||||
assert result["summary"][0]["ic_status"] == "no_valid_days"
|
||||
empty = frame.iloc[:0]
|
||||
assert api().correlation_matrix(empty)["matrix"] == [[None]]
|
||||
assert api().daily_ic(empty, pd.Series([], index=empty.index, dtype=float))["rows"] == []
|
||||
|
||||
|
||||
def prices(values=(100, 110, 121, 133.1)):
|
||||
sessions = pd.DatetimeIndex(["2026-09-18", "2026-09-21", "2026-09-22", "2026-09-23"])
|
||||
index = pd.MultiIndex.from_product([sessions, ["A"]], names=["date", "asset"])
|
||||
return pd.Series(values, index=index), sessions
|
||||
|
||||
|
||||
def forward(series, sessions, **changes):
|
||||
options = {"entry_lag_sessions": 0, "holding_sessions": 2,
|
||||
"price_field": "close", "price_basis": "synthetic_comparable"}
|
||||
options.update(changes)
|
||||
return api().forward_returns(series, sessions=sessions, **options)
|
||||
|
||||
|
||||
def test_forward_returns_use_explicit_calendar_endpoints_and_compound_price_ratio():
|
||||
series, sessions = prices()
|
||||
result = forward(series, sessions)
|
||||
assert result.returns.iloc[0] == pytest.approx(0.21)
|
||||
assert result.intervals.iloc[0]["entry_date"] == "2026-09-18"
|
||||
assert result.intervals.iloc[0]["exit_date"] == "2026-09-22"
|
||||
assert result.intervals.iloc[-1]["status"] == "insufficient_calendar"
|
||||
assert np.isnan(result.returns.iloc[-1])
|
||||
assert result.metadata["formula"] == "exit_price / entry_price - 1"
|
||||
assert result.metadata["entry_lag_sessions"] == 0
|
||||
assert result.metadata["holding_sessions"] == 2
|
||||
assert result.metadata["execution_eligibility"] == "not_established"
|
||||
assert result.metadata["historical_availability"] == "not_established"
|
||||
|
||||
|
||||
def test_forward_entry_lag_changes_both_endpoints():
|
||||
series, sessions = prices([10, 20, 30, 80])
|
||||
result = forward(series, sessions, entry_lag_sessions=1)
|
||||
assert result.returns.iloc[0] == pytest.approx(3)
|
||||
assert result.intervals.iloc[0]["entry_date"] == "2026-09-21"
|
||||
assert result.intervals.iloc[0]["exit_date"] == "2026-09-23"
|
||||
|
||||
|
||||
def test_forward_does_not_skip_missing_prices_or_derive_calendar_from_rows():
|
||||
series, sessions = prices([100, None, 121, 133.1])
|
||||
result = forward(series.drop(index=(sessions[1], "A")), sessions, holding_sessions=1)
|
||||
assert len(result.returns) == 4
|
||||
assert np.isnan(result.returns.iloc[0])
|
||||
assert np.isnan(result.returns.iloc[1])
|
||||
assert result.intervals.iloc[0]["status"] == "missing_price"
|
||||
assert result.intervals.iloc[0]["exit_date"] == "2026-09-21"
|
||||
assert result.returns.iloc[2] == pytest.approx(0.1)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("options", [{"holding_sessions": 0}, {"holding_sessions": True},
|
||||
{"holding_sessions": 1.5}, {"entry_lag_sessions": -1},
|
||||
{"entry_lag_sessions": False}, {"price_field": ""},
|
||||
{"price_basis": ""}])
|
||||
def test_forward_rejects_ambiguous_interval_parameters(options):
|
||||
series, sessions = prices()
|
||||
with pytest.raises(ValueError, match=r"sessions|price field"):
|
||||
forward(series, sessions, **options)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("kind", ["duplicate", "descending", "timezone", "intraday", "outside"])
|
||||
def test_forward_rejects_invalid_calendar(kind):
|
||||
series, sessions = prices()
|
||||
if kind == "duplicate":
|
||||
sessions = sessions.insert(1, sessions[0])
|
||||
elif kind == "descending":
|
||||
sessions = sessions[::-1]
|
||||
elif kind == "timezone":
|
||||
sessions = sessions.tz_localize("UTC")
|
||||
elif kind == "intraday":
|
||||
sessions = sessions + pd.Timedelta(hours=1)
|
||||
else:
|
||||
sessions = sessions[:-1]
|
||||
with pytest.raises(ValueError, match=r"sessions|dates|calendar"):
|
||||
forward(series, sessions)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("value", [0, -1, True, "2", np.inf])
|
||||
def test_forward_rejects_nonpositive_or_nonfinite_prices(value):
|
||||
series, sessions = prices([value, 110, 121, 133.1])
|
||||
with pytest.raises(ValueError, match=r"positive|Values|Infinite"):
|
||||
forward(series, sessions)
|
||||
|
||||
|
||||
def test_result_version_and_parameters_do_not_grant_source_or_decision_admission():
|
||||
frame = panel({"f": [1, 2, 3]})
|
||||
result = api().daily_ic(frame, pd.Series([1, 2, 3], index=frame.index))
|
||||
assert result["contract_version"] == api().FACTOR_DIAGNOSTICS_VERSION
|
||||
assert result["method"]["min_pairs"] == 3
|
||||
assert result["method"]["min_days"] == 2
|
||||
assert result["production_algorithm_version"] is None
|
||||
assert result["source_admission"] == result["historical_availability"] == "not_established"
|
||||
assert result["decision_eligible"] is False
|
||||
|
||||
|
||||
def test_nonzero_daily_summary_uses_three_days_not_nine_asset_rows():
|
||||
frame = panel({"f": [1, 2, 3] * 3}, days=3)
|
||||
returns = pd.Series([.01, .02, .03, .02, .03, .01, .01, -.02, .01], index=frame.index)
|
||||
result = api().daily_ic(frame, returns)
|
||||
assert [point["ic"] for point in result["rows"]] == pytest.approx([1, -.5, 0], abs=1e-15)
|
||||
summary = result["summary"][0]
|
||||
for prefix in ("ic", "rank_ic"):
|
||||
assert summary[f"{prefix}_valid_days"] == 3
|
||||
assert summary[f"{prefix}_mean"] == pytest.approx(1 / 6)
|
||||
assert summary[f"{prefix}_std"] == pytest.approx(sqrt(7 / 12))
|
||||
assert summary[f"{prefix}_ir"] == pytest.approx(.2182178902359924)
|
||||
assert summary[f"{prefix}_t"] == pytest.approx(.3779644730092272)
|
||||
assert summary[f"{prefix}_p"] == pytest.approx(.7418011102528389)
|
||||
|
||||
|
||||
def test_each_side_has_three_values_but_only_two_common_pairs():
|
||||
frame = panel({"left": [1, 2, 3, None], "right": [None, 2, 3, 4]}, assets=list("ABCD"))
|
||||
result = api().correlation_matrix(frame)
|
||||
assert result["n_pairs"][0][1] == 2
|
||||
assert result["matrix"][0][1] is None
|
||||
assert result["status"][0][1] == "insufficient_pairs"
|
||||
|
||||
|
||||
def test_rank_preserves_distinct_tiny_values_beside_a_large_outlier():
|
||||
frame = panel({"left": [1e-308, 2e-308, 1e308], "right": [1, 2, 3]})
|
||||
assert api().correlation_matrix(frame, method="spearman")["matrix"][0][1] == pytest.approx(1)
|
||||
|
||||
|
||||
def test_daily_rank_ic_preserves_tiny_distinct_observations():
|
||||
frame = panel({"f": [1e-200, 2e-200, 3e-200, 1e308]}, assets=list("ABCD"))
|
||||
returns = pd.Series([4, 3, 2, 1], index=frame.index)
|
||||
result = api().daily_ic(frame, returns)
|
||||
assert result["rows"][0]["rank_ic"] == pytest.approx(-1)
|
||||
assert result["rows"][0]["rank_ic_status"] == "ok"
|
||||
|
||||
|
||||
def test_affine_equivalent_daily_ic_does_not_turn_roundoff_into_extreme_significance():
|
||||
frame = panel({"f": [1, 2, 3, 10, 20, 30, 101, 102, 103]}, days=3)
|
||||
returns = pd.Series([1, 1, 2, 101, 101, 102, 100001, 100001, 100002], index=frame.index)
|
||||
result = api().daily_ic(frame, returns)
|
||||
assert [point["ic"] for point in result["rows"]] == pytest.approx([sqrt(3) / 2] * 3)
|
||||
summary = result["summary"][0]
|
||||
assert summary["ic_mean"] == pytest.approx(sqrt(3) / 2)
|
||||
assert summary["ic_valid_days"] == 3
|
||||
assert summary["ic_ir"] is summary["ic_t"] is summary["ic_p"] is None
|
||||
assert summary["ic_std"] <= result["method"]["minimum_ic_std_for_ratios"]
|
||||
assert summary["ic_status"] in {"constant_values", "below_resolution"}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("offset,step", [(1e16, 2.0), (1e12, np.spacing(1e12))])
|
||||
def test_pearson_preserves_representable_differences_beside_a_large_offset(offset, step):
|
||||
frame = panel({"f": offset + np.arange(5) * step}, assets=list("ABCDE"))
|
||||
returns = pd.Series([2, 5, 1, 4, 3], index=frame.index)
|
||||
matrix = api().correlation_matrix(frame.assign(other=returns))
|
||||
assert matrix["matrix"][0][1] == pytest.approx(.1, abs=1e-14)
|
||||
daily = api().daily_ic(frame, returns)
|
||||
assert daily["rows"][0]["ic"] == pytest.approx(.1, abs=1e-14)
|
||||
@@ -1,338 +0,0 @@
|
||||
"""Governed Personal Quant OS vertical-slice contracts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.governed_pipeline import (
|
||||
DatasetSnapshot,
|
||||
FactorVersion,
|
||||
PaperOrderIntent,
|
||||
RiskDecisionStatus,
|
||||
RiskPolicy,
|
||||
StrategyStage,
|
||||
StrategyVersion,
|
||||
create_paper_order_intent,
|
||||
run_governed_factor_slice,
|
||||
)
|
||||
|
||||
|
||||
def _calendar() -> pd.DatetimeIndex:
|
||||
return pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
|
||||
|
||||
def _scores() -> pd.DataFrame:
|
||||
dates = _calendar()
|
||||
return pd.DataFrame(
|
||||
{"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]},
|
||||
index=dates[:2],
|
||||
)
|
||||
|
||||
|
||||
def _prices() -> tuple[pd.DataFrame, pd.DataFrame]:
|
||||
dates = _calendar()
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 10.2, 10.4], "B": [20.0, 20.0, 20.5, 21.0], "C": [30.0, 30.0, 30.0, 30.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = opens * 1.01
|
||||
return opens, closes
|
||||
|
||||
|
||||
def _snapshot() -> DatasetSnapshot:
|
||||
return DatasetSnapshot(
|
||||
snapshot_id="dataset:cn-a-daily-20260108-v1",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 7, tzinfo=UTC),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 8, 5, tzinfo=UTC),
|
||||
)
|
||||
|
||||
|
||||
def _factor() -> FactorVersion:
|
||||
return FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="1.0.0",
|
||||
)
|
||||
|
||||
|
||||
def _strategy() -> StrategyVersion:
|
||||
return StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id="factor:demo-momentum@1.0.0",
|
||||
stage=StrategyStage.APPROVED,
|
||||
)
|
||||
|
||||
|
||||
def _execution_config() -> ExecutionConfig:
|
||||
return ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
|
||||
def test_governed_slice_is_reproducible_and_creates_only_paper_intent() -> None:
|
||||
opens, closes = _prices()
|
||||
created_at = datetime(2026, 1, 9, 1, tzinfo=UTC)
|
||||
policy = RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
)
|
||||
|
||||
result = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=policy,
|
||||
code_revision="c" * 40,
|
||||
created_at=created_at,
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
assert result.backtest_run.dataset_snapshot_id == _snapshot().snapshot_id
|
||||
assert result.backtest_run.factor_version_id == _factor().version_id
|
||||
assert result.backtest_run.strategy_version_id == _strategy().version_id
|
||||
assert result.backtest_run.code_revision == "c" * 40
|
||||
assert len(result.backtest_run.config_hash) == 64
|
||||
assert result.portfolio_target.backtest_run_id == result.backtest_run.run_id
|
||||
assert result.risk_decision.status is RiskDecisionStatus.APPROVED
|
||||
assert result.risk_decision.portfolio_target_id == result.portfolio_target.target_id
|
||||
assert result.order_intent is not None
|
||||
assert result.order_intent.environment == "paper"
|
||||
assert result.order_intent.risk_decision_id == result.risk_decision.decision_id
|
||||
assert result.order_intent.portfolio_target_id == result.portfolio_target.target_id
|
||||
assert result.factor_version.version_id == "factor:demo-momentum@1.0.0"
|
||||
assert result.factor_version.definition_sha256 == "b" * 64
|
||||
assert result.strategy_version.version_id == "strategy:demo-top2@1.0.0"
|
||||
assert result.backtest_run.run_id == (
|
||||
"backtest-run:e74403571f6a73c98b380220748957a422518c0bed883fabc4b93ebe13f05a37"
|
||||
)
|
||||
assert result.backtest_run.config_hash == (
|
||||
"40a3c804a2dc940161a626d1a5d25817c13463e685005e37fc48c41d1e20b87b"
|
||||
)
|
||||
assert result.portfolio_target.target_id == (
|
||||
"portfolio-target:ab2d398489aa9a292ee155a1098e9340beac0a924874cdaeb5b7b4d379ac9ce8"
|
||||
)
|
||||
assert result.risk_decision.decision_id == (
|
||||
"risk-decision:95924926bd327e44beeb15a63f47d14c5e77b97533fc3227fba2eda68e9b423d"
|
||||
)
|
||||
assert result.order_intent.intent_id == (
|
||||
"order-intent:73c349086c2c05b68424ace9286896eb21507b9080da5ff9b96b58c88d8f6ac1"
|
||||
)
|
||||
|
||||
repeated = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=policy,
|
||||
code_revision="c" * 40,
|
||||
created_at=created_at,
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
assert repeated.backtest_run.run_id == result.backtest_run.run_id
|
||||
assert repeated.portfolio_target.target_id == result.portfolio_target.target_id
|
||||
assert repeated.risk_decision.decision_id == result.risk_decision.decision_id
|
||||
assert repeated.order_intent == result.order_intent
|
||||
|
||||
|
||||
def test_risk_rejection_blocks_order_intent() -> None:
|
||||
opens, closes = _prices()
|
||||
result = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:no-concentration@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.4,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
assert result.risk_decision.status is RiskDecisionStatus.REJECTED
|
||||
assert any("single-asset weight" in reason for reason in result.risk_decision.reasons)
|
||||
assert result.order_intent is None
|
||||
with pytest.raises(ValueError, match="approved risk decision"):
|
||||
create_paper_order_intent(result.portfolio_target, result.risk_decision)
|
||||
with pytest.raises(ValueError, match="approved risk decision"):
|
||||
PaperOrderIntent(result.portfolio_target, result.risk_decision)
|
||||
|
||||
|
||||
def test_dataset_snapshot_requires_point_in_time_ordering_and_aware_times() -> None:
|
||||
with pytest.raises(ValueError, match="timezone-aware"):
|
||||
DatasetSnapshot(
|
||||
snapshot_id="dataset:invalid",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 7),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="effective_at <= available_at <= ingested_at"):
|
||||
DatasetSnapshot(
|
||||
snapshot_id="dataset:invalid",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 10, tzinfo=UTC),
|
||||
)
|
||||
|
||||
|
||||
def test_strategy_factor_lineage_must_match() -> None:
|
||||
opens, closes = _prices()
|
||||
mismatched = StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id="factor:other@1.0.0",
|
||||
stage=StrategyStage.APPROVED,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="factor lineage"):
|
||||
run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=mismatched,
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
|
||||
def test_governed_slice_requires_matching_schema_and_snapshot_available_by_run_time() -> None:
|
||||
opens, closes = _prices()
|
||||
common = {
|
||||
"factor_scores": _scores(),
|
||||
"execution_prices": opens,
|
||||
"valuation_prices": closes,
|
||||
"strategy_version": _strategy(),
|
||||
"risk_policy": RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
"code_revision": "c" * 40,
|
||||
"top_k": 2,
|
||||
"execution_price_field": "open",
|
||||
"valuation_price_field": "close",
|
||||
"execution_config": _execution_config(),
|
||||
}
|
||||
|
||||
with pytest.raises(ValueError, match="dataset schema"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="2.0.0",
|
||||
),
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
**common,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="available before the research run"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
created_at=datetime(2026, 1, 8, 7, 30, tzinfo=UTC),
|
||||
**common,
|
||||
)
|
||||
|
||||
future_scores = _scores()
|
||||
future_scores.index = pd.date_range("2026-01-12", periods=2, freq="B")
|
||||
with pytest.raises(ValueError, match="future decision dates"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
factor_scores=future_scores,
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
strategy_version=common["strategy_version"],
|
||||
risk_policy=common["risk_policy"],
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
|
||||
def test_paper_intent_requires_approved_strategy_stage() -> None:
|
||||
opens, closes = _prices()
|
||||
validated = StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id=_factor().version_id,
|
||||
stage=StrategyStage.VALIDATED,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="Approved or Paper"):
|
||||
run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=validated,
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
@@ -1,727 +0,0 @@
|
||||
"""Closed performance-evidence contract conformance tests."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
PERFORMANCE_EVIDENCE_SCHEMA_VERSION,
|
||||
PERFORMANCE_METRIC_SCHEMA_ID,
|
||||
PERFORMANCE_METHODOLOGY_ID,
|
||||
BacktestEvidenceManifest,
|
||||
EvidenceQualification,
|
||||
PerformanceEvidenceError,
|
||||
PerformanceEvidenceErrorCode,
|
||||
PerformanceEvidenceV1,
|
||||
PerformanceMetricAvailability,
|
||||
ResearchRunArtifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_performance_evidence,
|
||||
build_research_run_artifact,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
InputBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
ProducerIdentity,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.governed_pipeline import BacktestRunRef
|
||||
from quant_engine.metrics import TRADING_DAYS_PER_YEAR, benchmark_summary, summary
|
||||
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
PERFORMANCE_FIXTURE = (
|
||||
ROOT / "tests" / "fixtures" / "performance-evidence-v1.golden.json"
|
||||
)
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _accepted_authorities() -> tuple[
|
||||
DatasetSnapshotEnvelope,
|
||||
DataFoundationEnvelope,
|
||||
FactorSetRef,
|
||||
]:
|
||||
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
|
||||
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
artifact_ref = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
selected_view_ref_ids=(VIEW_REF_ID,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
definition.definition_id,
|
||||
"market",
|
||||
VIEW_REF_ID,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
output_quality=OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
output_coverage=OutputCoverage(
|
||||
"complete",
|
||||
1,
|
||||
1,
|
||||
"row",
|
||||
"alpha_005.cn_a",
|
||||
"sha256:" + "4" * 64,
|
||||
),
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
output_artifact_ref=artifact_ref,
|
||||
availability_mode=AvailabilityMode.AS_AVAILABLE,
|
||||
evaluation_at="2026-01-03T11:00:00Z",
|
||||
computed_at="2026-01-03T10:15:00Z",
|
||||
artifact_available_at="2026-01-03T10:20:00Z",
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
actor=ActorIdentity("service", "factor_worker_v1"),
|
||||
correlation_id="research_run_001",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
evidence_scope="synthetic_fixture",
|
||||
decision_eligible=False,
|
||||
)
|
||||
return snapshot, foundation, factor_set
|
||||
|
||||
|
||||
def _configuration_digest() -> str:
|
||||
return _sha256(
|
||||
json.dumps(
|
||||
PARAMETERS,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
)
|
||||
|
||||
|
||||
def _run_ref(**overrides: Any) -> BacktestRunRef:
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
arguments: dict[str, Any] = {
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"factor_set": factor_set,
|
||||
"universe_digest": "sha256:" + "5" * 64,
|
||||
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
|
||||
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": "sha256:" + "6" * 64,
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": "sha256:" + "7" * 64,
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": "sha256:" + "8" * 64,
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": "sha256:" + "9" * 64,
|
||||
"configuration_digest": _configuration_digest(),
|
||||
"evaluation_at": "2026-01-08T01:00:00Z",
|
||||
"computed_at": "2026-01-08T02:00:00Z",
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return BacktestRunRef.create(**arguments)
|
||||
|
||||
|
||||
def _backtest_result() -> FactorBacktestResult:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
return run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _artifact(
|
||||
run_ref: BacktestRunRef,
|
||||
benchmark_kind: str,
|
||||
) -> tuple[ResearchRunArtifact, FactorBacktestResult]:
|
||||
result = _backtest_result()
|
||||
benchmark_id: str | None
|
||||
benchmark_returns: pd.Series | None
|
||||
if benchmark_kind == "absent":
|
||||
benchmark_id = None
|
||||
benchmark_returns = None
|
||||
elif benchmark_kind == "estimable":
|
||||
benchmark_id = "000300.SH"
|
||||
benchmark_returns = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
elif benchmark_kind == "zero_active_variance":
|
||||
benchmark_id = "000300.SH"
|
||||
benchmark_returns = result.returns.rename("benchmark_return")
|
||||
elif benchmark_kind == "zero_benchmark_variance":
|
||||
benchmark_id = "000300.SH"
|
||||
benchmark_returns = pd.Series(
|
||||
np.zeros(len(result.returns)),
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
else:
|
||||
raise AssertionError(f"unknown benchmark_kind: {benchmark_kind}")
|
||||
artifact = build_research_run_artifact(
|
||||
result,
|
||||
run_id=run_ref.run_id,
|
||||
strategy_id=run_ref.strategy_id,
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version=run_ref.strategy_version,
|
||||
engine_version="1.2.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
data_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=PARAMETERS,
|
||||
benchmark_id=benchmark_id,
|
||||
benchmark_returns=benchmark_returns,
|
||||
)
|
||||
return artifact, result
|
||||
|
||||
|
||||
def _case(
|
||||
benchmark_kind: str,
|
||||
) -> tuple[PerformanceEvidenceV1, ResearchRunArtifact, BacktestRunRef, BacktestEvidenceManifest]:
|
||||
run_ref = _run_ref()
|
||||
artifact, _ = _artifact(run_ref, benchmark_kind)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
return (
|
||||
build_performance_evidence(artifact, run_ref, manifest),
|
||||
artifact,
|
||||
run_ref,
|
||||
manifest,
|
||||
)
|
||||
|
||||
|
||||
def _metric_map(evidence: PerformanceEvidenceV1) -> dict[str, Any]:
|
||||
return {metric.key: metric for metric in evidence.metrics}
|
||||
|
||||
|
||||
def _mutate_frozen(value: Any, field: str, replacement: object) -> Any:
|
||||
changed = copy.copy(value)
|
||||
object.__setattr__(changed, field, replacement)
|
||||
return changed
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[PerformanceEvidenceError],
|
||||
code: PerformanceEvidenceErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_present_evidence_is_deterministic_content_addressed_and_three_party_closed() -> None:
|
||||
first, artifact, run_ref, manifest = _case("estimable")
|
||||
second = build_performance_evidence(artifact, run_ref, manifest)
|
||||
|
||||
assert first == second
|
||||
assert first.schema_version == PERFORMANCE_EVIDENCE_SCHEMA_VERSION
|
||||
assert first.performance_evidence_id.startswith("rhperformanceevidencev1:sha256:")
|
||||
assert first.document_sha256.startswith("sha256:")
|
||||
assert first.authority == "quant_engine"
|
||||
assert first.scope == "offline_research_only"
|
||||
assert first.run_id == first.backtest_run_ref_id == run_ref.run_id == manifest.run_id
|
||||
assert first.backtest_evidence_manifest_id == manifest.manifest_id
|
||||
assert first.backtest_evidence_manifest_evidence_digest == manifest.evidence_digest
|
||||
assert first.backtest_evidence_qualification == "contract_qualified"
|
||||
assert first.research_artifact_content_digest == f"sha256:{artifact.content_sha256}"
|
||||
assert first.performance_table_logical_name == "performance"
|
||||
assert first.performance_table_row_count == 1
|
||||
assert first.performance_row_digest.startswith("sha256:")
|
||||
assert first.benchmark_series_digest is not None
|
||||
assert first.canonical_bytes() == first.to_json().encode("utf-8")
|
||||
assert not first.canonical_bytes().endswith(b"\n")
|
||||
document_payload = first.to_dict()
|
||||
document_payload.pop("document_sha256")
|
||||
expected_document = json.dumps(
|
||||
document_payload,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
assert _sha256(expected_document) == first.document_sha256
|
||||
assert PerformanceEvidenceV1.from_dict(
|
||||
first.to_dict(),
|
||||
artifact=artifact,
|
||||
run_ref=run_ref,
|
||||
evidence_manifest=manifest,
|
||||
) == first
|
||||
|
||||
|
||||
def test_methodology_and_metrics_bind_the_actual_artifact_builder_path() -> None:
|
||||
evidence, artifact, _, _ = _case("estimable")
|
||||
result = _backtest_result()
|
||||
expected_absolute = summary(result.returns, rf=0.0)
|
||||
benchmark = artifact.nav.set_index("trade_date")["benchmark_return"]
|
||||
benchmark.index = result.returns.index
|
||||
expected_relative = benchmark_summary(
|
||||
result.returns,
|
||||
benchmark,
|
||||
risk_free_daily=0.0,
|
||||
annualization=TRADING_DAYS_PER_YEAR,
|
||||
)
|
||||
metrics = _metric_map(evidence)
|
||||
|
||||
assert evidence.methodology.methodology_id == PERFORMANCE_METHODOLOGY_ID
|
||||
assert evidence.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID
|
||||
assert evidence.methodology.return_type == "simple"
|
||||
assert evidence.methodology.source_frequency == "1d"
|
||||
assert evidence.methodology.periods_per_year == TRADING_DAYS_PER_YEAR == 252
|
||||
assert evidence.methodology.annual_risk_free == 0.0
|
||||
assert evidence.methodology.benchmark_risk_free_daily == 0.0
|
||||
assert evidence.methodology.benchmark_alignment == "exact_session_index"
|
||||
assert metrics["annualized_return"].value == pytest.approx(
|
||||
expected_absolute["ann_return"]
|
||||
)
|
||||
assert metrics["sharpe_ratio"].value == pytest.approx(expected_absolute["sharpe"])
|
||||
assert metrics["tracking_error"].value == pytest.approx(
|
||||
expected_relative["tracking_error"]
|
||||
)
|
||||
assert metrics["alpha"].value == pytest.approx(expected_relative["alpha"])
|
||||
assert all(metric.methodology_id == PERFORMANCE_METHODOLOGY_ID for metric in metrics.values())
|
||||
assert all(metric.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID for metric in metrics.values())
|
||||
|
||||
|
||||
def test_relative_metric_availability_is_closed_for_present_absent_and_unestimable() -> None:
|
||||
present, *_ = _case("estimable")
|
||||
absent, *_ = _case("absent")
|
||||
zero_active, *_ = _case("zero_active_variance")
|
||||
zero_benchmark, *_ = _case("zero_benchmark_variance")
|
||||
|
||||
present_metrics = _metric_map(present)
|
||||
assert all(
|
||||
present_metrics[key].availability is PerformanceMetricAvailability.AVAILABLE
|
||||
for key in ("tracking_error", "information_ratio", "alpha", "beta")
|
||||
)
|
||||
absent_metrics = _metric_map(absent)
|
||||
assert absent.benchmark_series_digest is None
|
||||
assert absent.benchmark_id == ""
|
||||
assert absent.benchmark_alignment_policy == "none"
|
||||
assert all(
|
||||
absent_metrics[key].value is None
|
||||
and absent_metrics[key].availability
|
||||
is PerformanceMetricAvailability.BENCHMARK_ABSENT
|
||||
for key in ("tracking_error", "information_ratio", "alpha", "beta")
|
||||
)
|
||||
zero_active_metrics = _metric_map(zero_active)
|
||||
assert zero_active_metrics["tracking_error"].value == pytest.approx(0.0)
|
||||
assert (
|
||||
zero_active_metrics["information_ratio"].availability
|
||||
is PerformanceMetricAvailability.NOT_ESTIMABLE_ACTIVE_VARIANCE
|
||||
)
|
||||
assert zero_active_metrics["information_ratio"].value is None
|
||||
zero_benchmark_metrics = _metric_map(zero_benchmark)
|
||||
assert np.isfinite(zero_benchmark_metrics["tracking_error"].value)
|
||||
for key in ("alpha", "beta"):
|
||||
assert zero_benchmark_metrics[key].value is None
|
||||
assert (
|
||||
zero_benchmark_metrics[key].availability
|
||||
is PerformanceMetricAvailability.NOT_ESTIMABLE_BENCHMARK_VARIANCE
|
||||
)
|
||||
|
||||
|
||||
def test_golden_covers_present_absent_and_both_unestimable_states() -> None:
|
||||
expected = {
|
||||
"schema_version": 1,
|
||||
"source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c",
|
||||
"source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603",
|
||||
"cases": {
|
||||
name: _case(name)[0].to_dict()
|
||||
for name in (
|
||||
"estimable",
|
||||
"zero_active_variance",
|
||||
"zero_benchmark_variance",
|
||||
"absent",
|
||||
)
|
||||
},
|
||||
}
|
||||
assert json.loads(PERFORMANCE_FIXTURE.read_text(encoding="utf-8")) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("owner", "field", "replacement", "code", "path"),
|
||||
[
|
||||
(
|
||||
"run_ref",
|
||||
"run_id",
|
||||
"rhbacktestrunv1:sha256:" + "0" * 64,
|
||||
PerformanceEvidenceErrorCode.IDENTITY_MISMATCH,
|
||||
"$.backtest_run_ref.run_id",
|
||||
),
|
||||
(
|
||||
"manifest",
|
||||
"manifest_id",
|
||||
"rhbacktestevidencev1:sha256:" + "0" * 64,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.backtest_evidence_manifest.manifest_id",
|
||||
),
|
||||
(
|
||||
"manifest",
|
||||
"qualification",
|
||||
EvidenceQualification.EXPLORATORY,
|
||||
PerformanceEvidenceErrorCode.AUTHORITY_REJECTED,
|
||||
"$.backtest_evidence_manifest.qualification",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_owner_identity_and_authority_mismatches_fail_closed(
|
||||
owner: str,
|
||||
field: str,
|
||||
replacement: object,
|
||||
code: PerformanceEvidenceErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
_, artifact, run_ref, manifest = _case("estimable")
|
||||
changed_run_ref = _mutate_frozen(run_ref, field, replacement) if owner == "run_ref" else run_ref
|
||||
changed_manifest = (
|
||||
_mutate_frozen(manifest, field, replacement) if owner == "manifest" else manifest
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
build_performance_evidence(artifact, changed_run_ref, changed_manifest)
|
||||
_assert_error(rejected, code, path)
|
||||
|
||||
|
||||
def test_performance_table_row_and_benchmark_digest_mismatches_fail_closed() -> None:
|
||||
evidence, artifact, run_ref, manifest = _case("estimable")
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "n_days"] += 1
|
||||
changed_artifact = replace(artifact, _performance=performance)
|
||||
with pytest.raises(PerformanceEvidenceError) as table_mismatch:
|
||||
build_performance_evidence(changed_artifact, run_ref, manifest)
|
||||
_assert_error(
|
||||
table_mismatch,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.artifact.tables.performance.content_digest",
|
||||
)
|
||||
|
||||
payload = evidence.to_dict()
|
||||
payload["benchmark_series_digest"] = "sha256:" + "0" * 64
|
||||
with pytest.raises(PerformanceEvidenceError) as benchmark_mismatch:
|
||||
PerformanceEvidenceV1.from_dict(
|
||||
payload,
|
||||
artifact=artifact,
|
||||
run_ref=run_ref,
|
||||
evidence_manifest=manifest,
|
||||
)
|
||||
_assert_error(
|
||||
benchmark_mismatch,
|
||||
PerformanceEvidenceErrorCode.BENCHMARK_INVALID,
|
||||
"$.benchmark_series_digest",
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_closure_covers_non_performance_artifact_tables() -> None:
|
||||
_, artifact, run_ref, manifest = _case("estimable")
|
||||
nav = artifact.nav
|
||||
nav.loc[0, "nav"] += 0.01
|
||||
changed_artifact = replace(artifact, _nav=nav)
|
||||
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
build_performance_evidence(changed_artifact, run_ref, manifest)
|
||||
|
||||
_assert_error(
|
||||
rejected,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.artifact.tables.nav.content_digest",
|
||||
)
|
||||
|
||||
|
||||
def test_row_and_benchmark_mutations_change_their_digests_and_document_identity() -> None:
|
||||
original, artifact, run_ref, _ = _case("estimable")
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "sharpe"] += 0.01
|
||||
changed_performance_artifact = replace(artifact, _performance=performance)
|
||||
changed_performance_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_performance_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
changed_performance = build_performance_evidence(
|
||||
changed_performance_artifact,
|
||||
run_ref,
|
||||
changed_performance_manifest,
|
||||
)
|
||||
assert changed_performance.performance_row_digest != original.performance_row_digest
|
||||
assert changed_performance.performance_evidence_id != original.performance_evidence_id
|
||||
|
||||
nav = artifact.nav
|
||||
nav.loc[0, "benchmark_nav"] += 0.01
|
||||
changed_benchmark_artifact = replace(artifact, _nav=nav)
|
||||
changed_benchmark_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_benchmark_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
changed_benchmark = build_performance_evidence(
|
||||
changed_benchmark_artifact,
|
||||
run_ref,
|
||||
changed_benchmark_manifest,
|
||||
)
|
||||
assert changed_benchmark.benchmark_series_digest != original.benchmark_series_digest
|
||||
assert changed_benchmark.performance_row_digest == original.performance_row_digest
|
||||
assert changed_benchmark.performance_evidence_id != original.performance_evidence_id
|
||||
|
||||
|
||||
def test_relative_metric_null_reasons_cannot_be_invented() -> None:
|
||||
_, artifact, run_ref, _ = _case("estimable")
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "alpha"] = float("nan")
|
||||
changed_artifact = replace(artifact, _performance=performance)
|
||||
changed_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as false_alpha_domain:
|
||||
build_performance_evidence(changed_artifact, run_ref, changed_manifest)
|
||||
_assert_error(
|
||||
false_alpha_domain,
|
||||
PerformanceEvidenceErrorCode.METRIC_INVALID,
|
||||
"$.metrics.alpha.value",
|
||||
)
|
||||
|
||||
_, absent_artifact, absent_run_ref, _ = _case("absent")
|
||||
absent_performance = absent_artifact.performance
|
||||
absent_performance.loc[0, "tracking_error"] = 0.0
|
||||
changed_absent = replace(absent_artifact, _performance=absent_performance)
|
||||
changed_absent_manifest = build_backtest_evidence_manifest(
|
||||
absent_run_ref,
|
||||
changed_absent,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as false_absence:
|
||||
build_performance_evidence(
|
||||
changed_absent,
|
||||
absent_run_ref,
|
||||
changed_absent_manifest,
|
||||
)
|
||||
_assert_error(
|
||||
false_absence,
|
||||
PerformanceEvidenceErrorCode.BENCHMARK_INVALID,
|
||||
"$.metrics.tracking_error.availability",
|
||||
)
|
||||
|
||||
|
||||
def test_artifact_builder_enforces_strict_benchmark_session_alignment() -> None:
|
||||
run_ref = _run_ref()
|
||||
result = _backtest_result()
|
||||
misaligned = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index.shift(1, freq="B"),
|
||||
)
|
||||
with pytest.raises(ValueError, match="matching indexes"):
|
||||
build_research_run_artifact(
|
||||
result,
|
||||
run_id=run_ref.run_id,
|
||||
strategy_id=run_ref.strategy_id,
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version=run_ref.strategy_version,
|
||||
engine_version="1.2.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
data_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=PARAMETERS,
|
||||
benchmark_id="000300.SH",
|
||||
benchmark_returns=misaligned,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("column", "value", "path"),
|
||||
[
|
||||
("total_ret", -1.01, "$.metrics.total_return.value"),
|
||||
("ann_ret", -1.01, "$.metrics.annualized_return.value"),
|
||||
("ann_volatility", -0.01, "$.metrics.annualized_volatility.value"),
|
||||
("max_dd", 0.01, "$.metrics.maximum_drawdown.value"),
|
||||
("win_rate", 1.01, "$.metrics.win_rate.value"),
|
||||
("tracking_error", -0.01, "$.metrics.tracking_error.value"),
|
||||
("n_trades", True, "$.metrics.trade_count.value"),
|
||||
],
|
||||
)
|
||||
def test_metric_domains_reject_invalid_source_values(
|
||||
column: str,
|
||||
value: object,
|
||||
path: str,
|
||||
) -> None:
|
||||
_, artifact, run_ref, _ = _case("estimable")
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, column] = value
|
||||
changed_artifact = replace(artifact, _performance=performance)
|
||||
changed_manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
build_performance_evidence(changed_artifact, run_ref, changed_manifest)
|
||||
_assert_error(rejected, PerformanceEvidenceErrorCode.METRIC_INVALID, path)
|
||||
|
||||
|
||||
def test_closed_parser_rejects_unknown_non_ascii_non_finite_bool_and_unsafe_integer() -> None:
|
||||
evidence, artifact, run_ref, manifest = _case("estimable")
|
||||
|
||||
mutations: list[tuple[dict[str, Any], PerformanceEvidenceErrorCode, str]] = []
|
||||
unknown = evidence.to_dict()
|
||||
unknown["unexpected"] = "value"
|
||||
mutations.append((unknown, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.unexpected"))
|
||||
non_ascii = evidence.to_dict()
|
||||
non_ascii["métric"] = "value"
|
||||
mutations.append((non_ascii, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.métric"))
|
||||
non_finite = evidence.to_dict()
|
||||
non_finite["metrics"][0]["value"] = float("inf")
|
||||
mutations.append(
|
||||
(non_finite, PerformanceEvidenceErrorCode.METRIC_INVALID, "$.metrics[0].value")
|
||||
)
|
||||
bool_number = evidence.to_dict()
|
||||
bool_number["methodology"]["periods_per_year"] = True
|
||||
mutations.append(
|
||||
(
|
||||
bool_number,
|
||||
PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH,
|
||||
"$.methodology.periods_per_year",
|
||||
)
|
||||
)
|
||||
unsafe = evidence.to_dict()
|
||||
unsafe["performance_table_row_count"] = 2**53
|
||||
mutations.append(
|
||||
(
|
||||
unsafe,
|
||||
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.performance_table_row_count",
|
||||
)
|
||||
)
|
||||
|
||||
for payload, code, path in mutations:
|
||||
with pytest.raises(PerformanceEvidenceError) as rejected:
|
||||
PerformanceEvidenceV1.from_dict(
|
||||
payload,
|
||||
artifact=artifact,
|
||||
run_ref=run_ref,
|
||||
evidence_manifest=manifest,
|
||||
)
|
||||
_assert_error(rejected, code, path)
|
||||
|
||||
|
||||
def test_public_mapping_has_no_raw_inputs_storage_or_runtime_authority() -> None:
|
||||
evidence, *_ = _case("estimable")
|
||||
payload = evidence.to_dict()
|
||||
serialized = evidence.to_json().lower()
|
||||
forbidden_keys = {
|
||||
"parameters",
|
||||
"params_json",
|
||||
"returns",
|
||||
"nav",
|
||||
"benchmark_series",
|
||||
"table_bytes",
|
||||
"locator",
|
||||
"uri",
|
||||
"credential",
|
||||
"decision_eligible",
|
||||
"publication_eligible",
|
||||
"paper_trading",
|
||||
"live_trading",
|
||||
"investment_advice",
|
||||
}
|
||||
|
||||
def keys(value: object) -> set[str]:
|
||||
if isinstance(value, dict):
|
||||
return set(value) | {key for item in value.values() for key in keys(item)}
|
||||
if isinstance(value, list):
|
||||
return {key for item in value for key in keys(item)}
|
||||
return set()
|
||||
|
||||
assert not (keys(payload) & forbidden_keys)
|
||||
for token in ("postgres://", "mysql://", "s3://", "credential", "broker"):
|
||||
assert token not in serialized
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,311 +0,0 @@
|
||||
"""Fresh synthetic artifact assembly; no reuse or relabelling of real outputs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
EvidenceQualification,
|
||||
PerformanceEvidenceError,
|
||||
ResearchRunArtifact,
|
||||
build_research_run_artifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_performance_evidence,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.research_pipeline import run_factor_backtest_research
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
build_retrospective_performance_evidence,
|
||||
RetrospectiveBacktestEvidenceManifest,
|
||||
RetrospectivePerformanceEvidence,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import identify, replace_at
|
||||
|
||||
CONTRACT_ERRORS = (FactorContractError, BacktestContractError, PerformanceEvidenceError)
|
||||
|
||||
|
||||
def synthetic_artifact(run: RetrospectiveBacktestRunRef) -> ResearchRunArtifact:
|
||||
# Artifact-envelope tests, not an end-to-end proof of factor/source authenticity.
|
||||
# The existing financial methods receive new, in-memory synthetic matrices.
|
||||
dates = pd.date_range("2018-01-02", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"SIM0": [2.0, 0.0], "SIM1": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"SIM0": [10.0, 10.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 20.0, 21.0]}, index=dates
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"SIM0": [10.0, 12.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 18.0, 21.0]}, index=dates
|
||||
)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0
|
||||
),
|
||||
)
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02], index=result.returns.index, name="benchmark_return"
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id=run.run_id,
|
||||
strategy_id=run.strategy_id,
|
||||
strategy_name="Synthetic Top 1",
|
||||
strategy_version=run.strategy_version,
|
||||
engine_version="0.1.0",
|
||||
code_revision=run.code_revision,
|
||||
data_snapshot_id=run.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at=run.evaluation_at,
|
||||
finished_at=run.computed_at,
|
||||
parameters={"lag_sessions": 1, "top_k": 1},
|
||||
benchmark_id="synthetic.benchmark",
|
||||
benchmark_returns=benchmark,
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_closes_all_nine_existing_tables_without_changing_their_schema() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
wire = manifest.to_dict()
|
||||
assert wire["schema_version"] == "2.0.0"
|
||||
assert wire["artifact_schema_version"] == "1.1.0"
|
||||
assert wire["run_id"] == run.run_id
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["execution_validation"] == "not_validated"
|
||||
assert wire["decision_eligible"] is False
|
||||
assert manifest.manifest_id.startswith("rhbacktestevidencev2:sha256:")
|
||||
assert len({table.logical_name for item in manifest.evidence for table in item.tables}) == 9
|
||||
|
||||
|
||||
def test_performance_v2_keeps_existing_metric_methods_and_binds_all_upstreams() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
wire = evidence.to_dict()
|
||||
assert wire["schema_version"] == "researchhub.performance-evidence.v2"
|
||||
assert wire["methodology_id"] == "researchhub.quant-performance-methodology.v1"
|
||||
assert wire["metric_schema_id"] == "researchhub.quant-performance-metrics.v1"
|
||||
assert wire["research_artifact_schema_version"] == "1.1.0"
|
||||
assert wire["backtest_run_ref_id"] == run.run_id
|
||||
assert wire["backtest_evidence_manifest_id"] == manifest.manifest_id
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["start_date"] == "2018-01-02"
|
||||
assert wire["end_date"] == "2018-01-05"
|
||||
assert wire["artifact_available_at"] == "2026-09-08T01:11:00Z"
|
||||
assert evidence.run_id == run.run_id
|
||||
assert evidence.performance_evidence_id.startswith("rhperformancev2:sha256:")
|
||||
for metric in evidence.metrics:
|
||||
if metric.value is not None:
|
||||
assert metric.value == artifact.performance.iloc[0][metric.source_column]
|
||||
assert (
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
evidence.to_json(), artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
== evidence
|
||||
)
|
||||
assert (
|
||||
RetrospectiveBacktestEvidenceManifest.from_json(
|
||||
manifest.to_json(), artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
== manifest
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("run_id", "rhbacktestrunv2:sha256:" + "0" * 64),
|
||||
("profile", "offline_research_v1"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("evidence_scope", "real_data"),
|
||||
("artifact_available_at", "2026-09-08T01:09:00Z"),
|
||||
("artifact_schema_version", "2.0.0"),
|
||||
("qualification", "legacy_exploratory"),
|
||||
("evidence_digest", "sha256:" + "0" * 64),
|
||||
("evidence.0.tables.0.row_count", True),
|
||||
("evidence.0.tables.0.content_digest", "sha256:" + "0" * 64),
|
||||
("run_reference.value.factor_set_id", "rhfactorsetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_reidentified_claims_without_actual_table_closure(
|
||||
path: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
row = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "manifest_id", "rhbacktestevidencev2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveBacktestEvidenceManifest.from_dict(
|
||||
row, artifact=artifact, backtest_run_ref=run
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("table", "column", "value"),
|
||||
[
|
||||
("run", "data_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("run", "config_hash", "0" * 64),
|
||||
("run", "code_revision", "0" * 40),
|
||||
("run", "started_at", "2018-01-02T07:00:00Z"),
|
||||
("run", "finished_at", "2026-09-08T01:12:00Z"),
|
||||
("signals", "asset_id", "/private/data.csv"),
|
||||
("nav", "run_id", "old.run"),
|
||||
("performance", "run_id", "old.run"),
|
||||
],
|
||||
)
|
||||
def test_manifest_rejects_artifact_identity_time_or_private_data_mismatch(
|
||||
table: str, column: str, value: Any
|
||||
) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
frame = getattr(artifact, table)
|
||||
frame.loc[frame.index[0], column] = value
|
||||
forged = replace(artifact, **{"_" + table: frame})
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run, forged, artifact_available_at="2026-09-08T01:13:00Z"
|
||||
)
|
||||
|
||||
|
||||
def test_each_table_is_reconciled_and_legacy_apis_cannot_accept_v2() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
for item in manifest.evidence:
|
||||
for table in item.tables:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
expected_table_digests={table.logical_name: "sha256:" + "0" * 64},
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_performance_evidence(artifact, run, manifest)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_backtest_evidence_manifest(
|
||||
run,
|
||||
artifact,
|
||||
artifact_available_at="2026-09-08T01:11:00Z",
|
||||
qualification=EvidenceQualification.LEGACY_EXPLORATORY,
|
||||
)
|
||||
|
||||
|
||||
def seal_performance(row: dict[str, Any]) -> None:
|
||||
def sha(document: Any) -> str:
|
||||
return (
|
||||
"sha256:"
|
||||
+ hashlib.sha256(
|
||||
json.dumps(
|
||||
document,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode()
|
||||
).hexdigest()
|
||||
)
|
||||
|
||||
row.pop("document_sha256", None)
|
||||
row.pop("performance_evidence_id", None)
|
||||
row["performance_evidence_id"] = "rhperformancev2:" + sha(row)
|
||||
row["document_sha256"] = sha(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "researchhub.performance-evidence.v1"),
|
||||
("scope", "live"),
|
||||
("historical_availability", "established"),
|
||||
("decision_eligible", True),
|
||||
("evidence_scope", "real_data"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("backtest_run_ref_document_sha256", "sha256:" + "0" * 64),
|
||||
("backtest_evidence_manifest_id", "rhbacktestevidencev2:sha256:" + "0" * 64),
|
||||
("methodology.periods_per_year", 365),
|
||||
("metric_schema_id", "new.metric"),
|
||||
("metrics.0.value", 0.0),
|
||||
("metrics.0.nullable", True),
|
||||
("start_date", "2017-01-01"),
|
||||
("artifact_available_at", "2018-01-02T07:00:00Z"),
|
||||
],
|
||||
)
|
||||
def test_performance_never_accepts_reidentified_changed_facts(path: str, value: Any) -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
row = build_retrospective_performance_evidence(artifact, run, manifest).to_dict()
|
||||
replace_at(row, path, value)
|
||||
seal_performance(row)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_dict(
|
||||
row, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
|
||||
|
||||
def test_performance_has_immutable_finite_canonical_payload_and_current_tables() -> None:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
|
||||
assert evidence.document_sha256.startswith("sha256:")
|
||||
exported = evidence.to_dict()
|
||||
exported["metrics"][0]["value"] = 9.0
|
||||
assert evidence.to_dict()["metrics"][0]["value"] != 9.0
|
||||
for data in (
|
||||
evidence.to_json() + "\n",
|
||||
'{"schema_version":"x",' + evidence.to_json()[1:],
|
||||
"null",
|
||||
"{bad",
|
||||
):
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePerformanceEvidence.from_json(
|
||||
data, artifact=artifact, run_ref=run, evidence_manifest=manifest
|
||||
)
|
||||
frame = artifact.performance
|
||||
frame.loc[0, "total_ret"] = 0.0
|
||||
forged = replace(artifact, _performance=frame)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_performance_evidence(forged, run, manifest)
|
||||
@@ -1,157 +0,0 @@
|
||||
"""Offline synthetic v2 backtest evidence and replay boundaries."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import FactorContractError, PayloadValidation
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef
|
||||
from test_retrospective_data_contracts import digest, identify, replace_at
|
||||
from test_retrospective_factor_contracts import factor_arguments, decoding_arguments
|
||||
|
||||
|
||||
def run_arguments() -> dict[str, Any]:
|
||||
arguments = factor_arguments()
|
||||
factor = RetrospectiveFactorSetRef.create(**arguments)
|
||||
view = next(iter(arguments["foundation"].views.values()))
|
||||
return {
|
||||
"dataset_snapshot": arguments["dataset_snapshot"],
|
||||
"foundation": arguments["foundation"],
|
||||
"factor_set": factor,
|
||||
"universe_digest": digest({"synthetic_universe": 2}),
|
||||
"trading_calendar_revision_ids": view.trading_calendar_revision_ids,
|
||||
"corporate_action_revision_ids": view.corporate_action_revision_ids,
|
||||
"strategy_id": "synthetic.top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": digest({"synthetic_strategy": "top1"}),
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": digest({"synthetic_execution": 1}),
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": digest({"synthetic_cost": 1}),
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": digest({"synthetic_lock": 1}),
|
||||
"configuration_digest": digest({"lag_sessions": 1, "top_k": 1}),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
}
|
||||
|
||||
|
||||
def run_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("dataset_snapshot", "foundation", "factor_set")}
|
||||
|
||||
|
||||
def test_run_identity_closes_observation_inputs_and_preserves_configurations() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
document = run.to_dict()
|
||||
assert document["schema_version"] == "2.0.0"
|
||||
assert run.run_id.startswith("rhbacktestrunv2:sha256:")
|
||||
assert run.dataset_snapshot_id == arguments["dataset_snapshot"].snapshot_id
|
||||
assert run.foundation_id == arguments["foundation"].foundation_id
|
||||
assert run.factor_set_id == arguments["factor_set"].factor_set_id
|
||||
assert document["usage"] == "retrospective_research"
|
||||
assert document["historical_availability"] == "not_established"
|
||||
assert document["decision_eligible"] is False
|
||||
assert document["execution_validation"] == "not_validated"
|
||||
assert document["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert document["replay_attempt"] == 0
|
||||
assert RetrospectiveBacktestRunRef.from_json(run.to_json(), **run_context(arguments)) == run
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("historical_availability", "established"),
|
||||
("usage", "as_available"),
|
||||
("execution_validation", "validated"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("dataset_content_digest", "sha256:" + "0" * 64),
|
||||
("foundation_digest", "sha256:" + "0" * 64),
|
||||
("factor_set_digest", "sha256:" + "0" * 64),
|
||||
("factor_output_content_digest", "sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("evidence_scope", "real_data"),
|
||||
("trading_calendar_revision_ids", []),
|
||||
("corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("evaluation_at", "2026-09-08T01:07:00Z"),
|
||||
("computed_at", "2026-09-08T01:08:00Z"),
|
||||
("computed_at", "2026-09-08T01:10:00.0000001Z"),
|
||||
("random_seed", True),
|
||||
("strategy_version", "latest"),
|
||||
("configuration_digest", "../private/a"),
|
||||
("code_revision", "unknown"),
|
||||
("replay_attempt", 1),
|
||||
("replay_reason", "retry"),
|
||||
("replay_spec_digest", "sha256:" + "0" * 64),
|
||||
("replay_ancestor_run_ids", ["rhbacktestrunv2:sha256:" + "0" * 64]),
|
||||
],
|
||||
)
|
||||
def test_reidentified_run_must_match_exact_context(path: str, value: Any) -> None:
|
||||
arguments = run_arguments()
|
||||
row = RetrospectiveBacktestRunRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "run_id", "rhbacktestrunv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.from_dict(row, **run_context(arguments))
|
||||
|
||||
|
||||
def test_replay_keeps_input_spec_but_requires_new_actual_attempt_times() -> None:
|
||||
arguments = run_arguments()
|
||||
root = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
replay_args = {
|
||||
**arguments,
|
||||
"parent": root,
|
||||
"replay_reason": "synthetic.retry",
|
||||
"replay_attempt": 1,
|
||||
"evaluation_at": "2026-09-08T01:12:00Z",
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
replay = RetrospectiveBacktestRunRef.create(**replay_args)
|
||||
assert replay.replay_spec_digest == root.replay_spec_digest
|
||||
assert replay.run_id != root.run_id
|
||||
assert replay.replay_ancestor_run_ids == (root.run_id,)
|
||||
assert replay.evaluation_at != root.evaluation_at
|
||||
assert (
|
||||
RetrospectiveBacktestRunRef.from_json(
|
||||
replay.to_json(), **run_context(arguments), parent=root
|
||||
)
|
||||
== replay
|
||||
)
|
||||
for changes in (
|
||||
{"random_seed": 9},
|
||||
{"configuration_digest": digest({"different_configuration": 1})},
|
||||
{"evaluation_at": root.evaluation_at},
|
||||
{"replay_attempt": 2},
|
||||
{"replay_reason": None},
|
||||
{"parent": None},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**replay_args, **changes})
|
||||
|
||||
|
||||
def test_reference_only_factors_can_be_read_but_not_used_to_create_new_runs() -> None:
|
||||
arguments = run_arguments()
|
||||
run = RetrospectiveBacktestRunRef.create(**arguments)
|
||||
factor = arguments["factor_set"]
|
||||
reference = RetrospectiveFactorSetRef.from_dict(
|
||||
factor.to_dict(),
|
||||
definitions=factor._definitions,
|
||||
dataset_snapshot=arguments["dataset_snapshot"],
|
||||
foundation=arguments["foundation"],
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveBacktestRunRef.create(**{**arguments, "factor_set": reference})
|
||||
restored = RetrospectiveBacktestRunRef.from_dict(
|
||||
run.to_dict(), **{**run_context(arguments), "factor_set": reference}
|
||||
)
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
with pytest.raises(FactorContractError):
|
||||
restored.require_inputs_revalidated()
|
||||
run.require_inputs_revalidated()
|
||||
@@ -1,88 +0,0 @@
|
||||
"""Frozen synthetic interoperability vector, not end-to-end source provenance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from quant_engine.artifact import _evidence_frame_records
|
||||
from quant_engine.retrospective_artifact_contracts import build_retrospective_performance_evidence
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_factor_contracts import factor_arguments
|
||||
from test_retrospective_portfolio_risk_contracts import portfolio_arguments, risk_arguments
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
VECTOR = ROOT / "tests" / "fixtures" / "retrospective-computation-v2.golden.json"
|
||||
|
||||
|
||||
def build_vector() -> dict[str, Any]:
|
||||
portfolio = portfolio_arguments()
|
||||
risk = risk_arguments(portfolio)
|
||||
run = portfolio["backtest_run_ref"]
|
||||
manifest = portfolio["manifest"]
|
||||
artifact = manifest._artifact
|
||||
factor = factor_arguments()
|
||||
return {
|
||||
"fixture_kind": "synthetic_retrospective_contract_vector",
|
||||
"artifact_data_provenance": "envelope_test_only_not_end_to_end",
|
||||
"source_authenticity": "not_established",
|
||||
"factor_definitions": [item.to_dict() for item in factor["definitions"]],
|
||||
"dataset_chunks": factor["dataset_chunks"],
|
||||
"resolved_view_schema": {"synthetic_schema": "neutral_close_v2"},
|
||||
"factor_output_schema": json.loads(factor["output_schema_bytes"]),
|
||||
"factor_output_records": json.loads(factor["output_content_bytes"]),
|
||||
"factor_set": run._factor_set.to_dict(),
|
||||
"backtest_run_ref": run.to_dict(),
|
||||
"artifact_tables": {
|
||||
name: _evidence_frame_records(frame, name)
|
||||
for name, frame in artifact.table_frames().items()
|
||||
},
|
||||
"backtest_evidence_manifest": manifest.to_dict(),
|
||||
"performance_evidence": build_retrospective_performance_evidence(
|
||||
artifact, run, manifest
|
||||
).to_dict(),
|
||||
"portfolio_target": portfolio["target"].to_dict(),
|
||||
"portfolio_decision": risk["portfolio_decision"].to_dict(),
|
||||
"risk_assessment": assess_retrospective_portfolio_risk(**risk).to_dict(),
|
||||
"covariance_matrix": risk["covariance"].covariance.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
def test_synthetic_vector_matches_all_current_owner_serializers() -> None:
|
||||
expected = VECTOR.read_text(encoding="utf-8")
|
||||
actual = (
|
||||
json.dumps(build_vector(), ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False)
|
||||
+ "\n"
|
||||
)
|
||||
assert actual == expected
|
||||
|
||||
|
||||
def test_v2_modules_do_not_import_data_owners_publishers_or_execution_authority() -> None:
|
||||
modules = sorted((ROOT / "src" / "quant_engine").glob("retrospective_*_contracts.py"))
|
||||
assert len(modules) == 5
|
||||
for path in modules:
|
||||
tree = ast.parse(path.read_text(encoding="utf-8"))
|
||||
imports = {
|
||||
alias.name
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Import)
|
||||
for alias in node.names
|
||||
} | {node.module or "" for node in ast.walk(tree) if isinstance(node, ast.ImportFrom)}
|
||||
assert not any(
|
||||
name.startswith(("research_results", "research_platform", "edb_data_core"))
|
||||
for name in imports
|
||||
)
|
||||
called = {
|
||||
node.func.id
|
||||
for node in ast.walk(tree)
|
||||
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
|
||||
}
|
||||
assert not called & {
|
||||
"create_paper_order_intent",
|
||||
"run_governed_factor_slice",
|
||||
"evaluate_portfolio_risk",
|
||||
}
|
||||
@@ -1,636 +0,0 @@
|
||||
"""QE-owned decoding tests against RP's synthetic public v2 golden vectors."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from copy import deepcopy
|
||||
from dataclasses import FrozenInstanceError
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorContractError,
|
||||
canonical_json_bytes,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
|
||||
def golden(kind: str) -> dict[str, Any]:
|
||||
return json.loads((FIXTURES / f"retrospective-{kind}-v2.golden.json").read_text())
|
||||
|
||||
|
||||
def digest(value: Any) -> str:
|
||||
return "sha256:" + hashlib.sha256(canonical_json_bytes(value)).hexdigest()
|
||||
|
||||
|
||||
def identify(value: dict[str, Any], field: str, prefix: str) -> None:
|
||||
value[field] = prefix + digest({key: item for key, item in value.items() if key != field})
|
||||
|
||||
|
||||
def records() -> list[dict[str, Any]]:
|
||||
return [
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "1" * 32,
|
||||
"metric": "close",
|
||||
"value": "101.25",
|
||||
},
|
||||
{
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"instrument_id": "rhinstrument:" + "2" * 32,
|
||||
"metric": "close",
|
||||
"value": "87.50",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def bind_records(source: dict[str, Any], chunks: list[list[dict[str, Any]]]) -> None:
|
||||
def records_digest(rows: list[dict[str, Any]]) -> str:
|
||||
data = b"[" + b",".join(sorted(canonical_json_bytes(row) for row in rows)) + b"]"
|
||||
return "sha256:" + hashlib.sha256(data).hexdigest()
|
||||
|
||||
manifest = {
|
||||
"record_count": sum(len(rows) for rows in chunks),
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": index,
|
||||
"content_digest": records_digest(rows),
|
||||
"record_count": len(rows),
|
||||
}
|
||||
for index, rows in enumerate(chunks)
|
||||
],
|
||||
}
|
||||
source["descriptor"]["content"].update(
|
||||
{
|
||||
"record_count": manifest["record_count"],
|
||||
"logical_manifest": manifest,
|
||||
"manifest_digest": digest(manifest),
|
||||
"content_digest": records_digest([row for chunk in chunks for row in chunk]),
|
||||
}
|
||||
)
|
||||
source["descriptor"]["observation_manifest"]["batches"] = [
|
||||
{
|
||||
**chunk,
|
||||
"observation_kind": "observed_by",
|
||||
"observed_by": "2026-09-08T01:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_receipt": index}),
|
||||
}
|
||||
for index, chunk in enumerate(manifest["chunks"])
|
||||
]
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
|
||||
|
||||
def replace_at(source: dict[str, Any], path: str, value: Any) -> None:
|
||||
target: Any = source
|
||||
keys = path.split(".")
|
||||
for key in keys[:-1]:
|
||||
target = target[int(key)] if isinstance(target, list) else target[key]
|
||||
target[int(keys[-1]) if isinstance(target, list) else keys[-1]] = value
|
||||
|
||||
|
||||
COLLECTIONS = (
|
||||
(
|
||||
"instrument_routes",
|
||||
"route_revision_id",
|
||||
"rhroutev2:",
|
||||
"instrument_route",
|
||||
"instrument_route_revision_ids",
|
||||
),
|
||||
(
|
||||
"trading_calendar_revisions",
|
||||
"calendar_revision_id",
|
||||
"rhcalv2:",
|
||||
"trading_calendar",
|
||||
"trading_calendar_revision_ids",
|
||||
),
|
||||
(
|
||||
"corporate_action_revisions",
|
||||
"action_revision_id",
|
||||
"rhcav2:",
|
||||
"corporate_action",
|
||||
"corporate_action_revision_ids",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def seal_foundation(source: dict[str, Any], *, rebuild_lineage: bool = True) -> None:
|
||||
lineage = []
|
||||
for name, key, prefix, kind, view_key in COLLECTIONS:
|
||||
replacements = {}
|
||||
for row in sorted(source[name], key=lambda row: row["observation_sequence"]):
|
||||
old = row[key]
|
||||
if "supersedes_observation_id" in row:
|
||||
row["supersedes_observation_id"] = replacements.get(
|
||||
row["supersedes_observation_id"], row["supersedes_observation_id"]
|
||||
)
|
||||
identify(row, key, prefix)
|
||||
replacements[old] = row[key]
|
||||
lineage.append(
|
||||
{
|
||||
"revision_kind": kind,
|
||||
"revision_id": row[key],
|
||||
**{
|
||||
field: row[field]
|
||||
for field in (
|
||||
"observation_sequence",
|
||||
"observed_by",
|
||||
"earliest_external_knowledge",
|
||||
"history_completeness",
|
||||
"evidence_digest",
|
||||
"supersedes_observation_id",
|
||||
)
|
||||
if field in row
|
||||
},
|
||||
}
|
||||
)
|
||||
for view in source["standardized_views"]:
|
||||
view[view_key] = [replacements.get(item, item) for item in view[view_key]]
|
||||
if rebuild_lineage:
|
||||
source["observation_lineage"] = lineage
|
||||
for view in source["standardized_views"]:
|
||||
identify(view, "view_ref_id", "rhviewrefv2:")
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
|
||||
|
||||
def parse_foundation(source: dict[str, Any]) -> RetrospectiveFoundationEnvelope:
|
||||
return RetrospectiveFoundationEnvelope.from_dict(
|
||||
source, snapshot=RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
)
|
||||
|
||||
|
||||
def test_public_snapshot_golden_is_an_explicit_observation_contract() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert snapshot.to_dict() == source
|
||||
assert snapshot.snapshot_id == source["snapshot_id"]
|
||||
assert snapshot.observation_cutoff == "2026-09-08T01:01:00Z"
|
||||
assert snapshot.evidence_scope == "synthetic_fixture"
|
||||
assert snapshot.earliest_external_knowledge == {"status": "unknown"}
|
||||
assert not hasattr(snapshot, "pit_cutoff")
|
||||
assert not hasattr(snapshot, "knowledge_time")
|
||||
snapshot.require_qualified()
|
||||
assert RetrospectiveSnapshotEnvelope.from_json(snapshot.to_json()) == snapshot
|
||||
|
||||
|
||||
def test_public_foundation_golden_binds_exact_typed_snapshot() -> None:
|
||||
source = golden("data-foundation")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert foundation.to_dict() == source
|
||||
assert foundation.dataset_snapshot_id == snapshot.snapshot_id
|
||||
assert foundation.observation_cutoff == snapshot.observation_cutoff
|
||||
assert foundation.evidence_scope == snapshot.evidence_scope
|
||||
assert foundation.real_data_validation_status == "not_validated"
|
||||
assert not hasattr(foundation, "pit_cutoff")
|
||||
assert (
|
||||
RetrospectiveFoundationEnvelope.from_json(foundation.to_json(), snapshot=snapshot)
|
||||
== foundation
|
||||
)
|
||||
|
||||
|
||||
def test_empty_logical_dimension_is_rejected_after_rebinding_all_bytes() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["instrument_id"] = ""
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError, match="dimension"):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_boolean_lineage_sequence_does_not_equal_integer_one() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["observation_lineage"][0]["observation_sequence"] = True
|
||||
identify(source, "foundation_id", "rhdfv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("descriptor.time_semantics.pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.knowledge_time", {"start_inclusive": "2018-01-02T07:00:00Z"}),
|
||||
("descriptor.time_semantics.historical_availability", "established"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T01:01:00.0000001Z"),
|
||||
("descriptor.time_semantics.observation_cutoff", "2026-09-08T09:01:00+08:00"),
|
||||
(
|
||||
"descriptor.time_semantics.earliest_external_knowledge",
|
||||
{"status": "unknown", "evidence_digest": "sha256:" + "0" * 64},
|
||||
),
|
||||
("descriptor.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("descriptor.published_at", "2026-02-30T00:00:00Z"),
|
||||
("descriptor.published_at", "2026-09-08T01:00:00Z"),
|
||||
("descriptor.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("descriptor.qualification.usage", "as_available"),
|
||||
("descriptor.qualification.policy_version", "1.0.0"),
|
||||
("descriptor.quality.checks.0.severity", "advisory"),
|
||||
("descriptor.quality.checks.0.status", "failed"),
|
||||
("descriptor.quality.checks.0.check_id", "schema_conformance"),
|
||||
("descriptor.quality.status", "failed"),
|
||||
("descriptor.content.record_count", True),
|
||||
("descriptor.content.record_count", 9007199254740992),
|
||||
("descriptor.content.record_count", 2.0),
|
||||
("descriptor.content.content_digest", "bad"),
|
||||
("descriptor.content.manifest_digest", "sha256:" + "0" * 64),
|
||||
("descriptor.observation_manifest.batches", []),
|
||||
("descriptor.observation_manifest.batches.0.record_count", 1),
|
||||
("descriptor.observation_manifest.batches.0.chunk_index", True),
|
||||
("descriptor.observation_manifest.batches.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("descriptor.observation_manifest.batches.0.observation_kind", "first_published_at"),
|
||||
("descriptor.lineage.transformation.id", "rhtransform:private"),
|
||||
("descriptor.dataset.dimensions", ["instrument_id", "knowledge_time"]),
|
||||
("descriptor.dataset.dataset_id", "rhdataset:macroeconomic:" + "4" * 32),
|
||||
],
|
||||
)
|
||||
def test_snapshot_rejects_reidentified_invalid_declarations(path: str, value: Any) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
replace_at(source, path, value)
|
||||
# Noncanonical numbers are rejected before identity formation.
|
||||
if type(value) is not float and value != 9007199254740992:
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_snapshot_materialized_records_bind_the_public_golden_and_chunks() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
snapshot.verify_materialized_records([records()])
|
||||
snapshot.verify_materialized_records([list(reversed(records()))])
|
||||
chunks = [[records()[0]], [records()[1]]]
|
||||
bind_records(source, chunks)
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records(chunks)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records(chunks)
|
||||
rows = records()
|
||||
rows[0]["value"] = "0"
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"mutation", ["duplicate", "legacy", "range", "location", "missing_effective"]
|
||||
)
|
||||
def test_materialized_bad_records_rejected_even_with_matching_digests(mutation: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
if mutation == "duplicate":
|
||||
rows.append(deepcopy(rows[0]))
|
||||
elif mutation == "legacy":
|
||||
rows[0]["knowledge_time"] = "2026-09-08T01:00:00Z"
|
||||
elif mutation == "range":
|
||||
rows[0]["effective_time"] = "2018-01-01T07:00:00Z"
|
||||
elif mutation == "location":
|
||||
rows[0]["value"] = "/private/records.csv"
|
||||
else:
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def test_unknown_or_evidenced_knowledge_never_changes_usage() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"] = {
|
||||
"status": "evidenced",
|
||||
"range": {
|
||||
"start_inclusive": "2018-01-02T07:00:00Z",
|
||||
"end_inclusive": "2018-01-02T07:00:00Z",
|
||||
},
|
||||
"evidence_digest": digest({"synthetic_earliest": True}),
|
||||
}
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
parsed = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
assert (
|
||||
parsed.to_dict()["descriptor"]["time_semantics"]["historical_availability"]
|
||||
== "not_established"
|
||||
)
|
||||
source["descriptor"]["time_semantics"]["earliest_external_knowledge"]["range"][
|
||||
"end_inclusive"
|
||||
] = "2026-09-08T01:01:00Z"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
|
||||
|
||||
def test_rejected_snapshot_remains_readable_but_cannot_support_foundation() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["qualification"]["status"] = "rejected"
|
||||
identify(source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.require_qualified()
|
||||
foundation = golden("data-foundation")
|
||||
foundation["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in foundation["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(foundation, snapshot=snapshot)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2026-09-08T01:00:00Z"),
|
||||
("published_at", "2026-09-08T01:03:00Z"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "established"),
|
||||
("instrument_routes.0.observation_sequence", 2),
|
||||
("instrument_routes.0.supersedes_observation_id", "rhroutev2:sha256:" + "0" * 64),
|
||||
("instrument_routes.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("instrument_routes.0.history_completeness", "complete"),
|
||||
(
|
||||
"instrument_routes.0.earliest_external_knowledge",
|
||||
{"status": "unknown", "earliest_at": "2018-01-01T00:00:00Z"},
|
||||
),
|
||||
("instrument_routes.0.instrument_type", "index"),
|
||||
("instrument_routes.0.symbol", "WIND.TEST"),
|
||||
("instrument_routes.0.symbol", "A" * 33),
|
||||
("instrument_routes.0.calendar_id", "rhcalendar:" + "0" * 32),
|
||||
("trading_calendar_revisions.0.status", "closed"),
|
||||
("trading_calendar_revisions.0.sessions", []),
|
||||
("trading_calendar_revisions.0.sessions.0.closes_at", "2018-01-02T00:00:00Z"),
|
||||
("trading_calendar_revisions.0.session_date", "2018-02-30"),
|
||||
("standardized_views.0.instrument_route_revision_ids", []),
|
||||
("standardized_views.0.trading_calendar_revision_ids", []),
|
||||
("standardized_views.0.corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
|
||||
("standardized_views.0.observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:02:00Z"),
|
||||
("standardized_views.0.available_at", "2026-09-08T01:06:00Z"),
|
||||
("standardized_views.0.usage", "as_available"),
|
||||
("corporate_action_coverage", []),
|
||||
("corporate_action_coverage.0.instrument_id", "rhinstrument:" + "0" * 32),
|
||||
("corporate_action_coverage.0.effective_time.end_inclusive", "2018-01-01T07:00:00Z"),
|
||||
("corporate_action_coverage.0.effective_time.start_inclusive", "2018-01-02T07:00:01Z"),
|
||||
("corporate_action_coverage.0.observed_by", "2026-09-08T01:02:00Z"),
|
||||
("corporate_action_coverage.0.evidence_digests", []),
|
||||
("readiness.evidence_scope", "real_data"),
|
||||
("readiness.contract_validation.evidence_digests", []),
|
||||
(
|
||||
"readiness.real_data_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.production_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
(
|
||||
"readiness.live_validation",
|
||||
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_foundation_rejects_semantic_forgery_after_reidentification(path: str, value: Any) -> None:
|
||||
source = golden("data-foundation")
|
||||
replace_at(source, path, value)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_v1_and_v2_never_coerce_each_other() -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
DatasetSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
with pytest.raises(FactorContractError):
|
||||
DataFoundationEnvelope.from_dict(golden("data-foundation"))
|
||||
old = json.loads((FIXTURES / "factor-contracts-v1.golden.json").read_text())
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(old["dataset_snapshot"])
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(old["data_foundation"])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"),
|
||||
snapshot=DatasetSnapshotEnvelope.from_dict(old["dataset_snapshot"]),
|
||||
)
|
||||
|
||||
|
||||
def test_deep_immutability_and_strict_canonical_json() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
source["descriptor"]["quality"]["status"] = "failed"
|
||||
snapshot.to_dict()["descriptor"]["qualification"]["status"] = "rejected"
|
||||
snapshot.require_qualified()
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
snapshot._payload = {}
|
||||
with pytest.raises(TypeError):
|
||||
snapshot.earliest_external_knowledge["status"] = "evidenced"
|
||||
foundation = parse_foundation(golden("data-foundation"))
|
||||
with pytest.raises(TypeError):
|
||||
foundation.views["new"] = next(iter(foundation.views.values()))
|
||||
for decoder, document in (
|
||||
(RetrospectiveSnapshotEnvelope.from_json, golden("dataset-snapshot")),
|
||||
(
|
||||
lambda value: RetrospectiveFoundationEnvelope.from_json(value, snapshot=snapshot),
|
||||
golden("data-foundation"),
|
||||
),
|
||||
):
|
||||
wire = canonical_json_bytes(document)
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(wire + b"\n")
|
||||
with pytest.raises(FactorContractError):
|
||||
decoder(b'{"schema_version":"2.0.0",' + wire[1:])
|
||||
|
||||
|
||||
def test_macro_period_is_not_inferred_as_an_effective_instant() -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
source["descriptor"]["dataset"].update(
|
||||
dataset_id="rhdataset:macroeconomic:" + "4" * 32,
|
||||
dataset_kind="macroeconomic",
|
||||
dimensions=["series_id", "observation_period"],
|
||||
)
|
||||
rows = [
|
||||
{
|
||||
"series_id": "cpi",
|
||||
"observation_period": "2018-01",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"value": "2.1",
|
||||
}
|
||||
]
|
||||
bind_records(source, [rows])
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
del rows[0]["effective_time"]
|
||||
bind_records(source, [rows])
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
|
||||
|
||||
|
||||
def with_successor() -> dict[str, Any]:
|
||||
source = golden("data-foundation")
|
||||
previous = source["instrument_routes"][0]
|
||||
successor = deepcopy(previous)
|
||||
successor.update(
|
||||
observation_sequence=2,
|
||||
observed_by="2026-09-08T01:00:30Z",
|
||||
symbol="SIM0B",
|
||||
supersedes_observation_id=previous["route_revision_id"],
|
||||
)
|
||||
identify(successor, "route_revision_id", "rhroutev2:")
|
||||
source["instrument_routes"].append(successor)
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].append(
|
||||
successor["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
return source
|
||||
|
||||
|
||||
def test_retained_successor_requires_parent_time_and_view_ancestry() -> None:
|
||||
source = with_successor()
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.foundation_id == source["foundation_id"]
|
||||
assert parsed.published_at.isoformat() == "2026-09-08T01:05:00+00:00"
|
||||
assert parsed.contract_evidence_digests
|
||||
assert len(next(iter(parsed.views.values())).instrument_route_revision_ids) == 3
|
||||
for mutation in (
|
||||
"missing_parent",
|
||||
"equal_time",
|
||||
"omitted_ancestor",
|
||||
"duplicate_sequence",
|
||||
"wrong_lineage",
|
||||
):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "missing_parent":
|
||||
forged["instrument_routes"][-1]["supersedes_observation_id"] = (
|
||||
"rhroutev2:sha256:" + "0" * 64
|
||||
)
|
||||
elif mutation == "equal_time":
|
||||
forged["instrument_routes"][-1]["observed_by"] = forged["instrument_routes"][0][
|
||||
"observed_by"
|
||||
]
|
||||
elif mutation == "omitted_ancestor":
|
||||
forged["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
forged["instrument_routes"][0]["route_revision_id"]
|
||||
)
|
||||
elif mutation == "duplicate_sequence":
|
||||
forged["instrument_routes"][-1]["observation_sequence"] = 1
|
||||
else:
|
||||
forged["observation_lineage"][0]["evidence_digest"] = "sha256:" + "0" * 64
|
||||
seal_foundation(forged, rebuild_lineage=mutation != "wrong_lineage")
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(forged)
|
||||
|
||||
|
||||
def test_index_and_closed_calendar_are_explicit_non_execution_data() -> None:
|
||||
source = golden("data-foundation")
|
||||
source["instrument_routes"][0].update(asset_class="index", instrument_type="index")
|
||||
source["trading_calendar_revisions"][0].update(status="closed", sessions=[])
|
||||
seal_foundation(source)
|
||||
parsed = parse_foundation(source)
|
||||
assert parsed.to_dict()["instrument_routes"][0]["instrument_type"] == "index"
|
||||
assert parsed.to_dict()["readiness"]["live_validation"]["status"] == "not_validated"
|
||||
|
||||
|
||||
def test_real_scope_is_still_a_declaration_with_separate_evidence_and_coverage() -> None:
|
||||
snapshot_source = golden("dataset-snapshot")
|
||||
snapshot_source["evidence_scope"] = "real_data"
|
||||
identify(snapshot_source, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_source)
|
||||
source = golden("data-foundation")
|
||||
source["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
for view in source["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
source["readiness"]["evidence_scope"] = "real_data"
|
||||
source["readiness"]["real_data_validation"] = {
|
||||
"status": "validated",
|
||||
"evidence_digests": [digest({"synthetic_real_claim_test": 1})],
|
||||
}
|
||||
seal_foundation(source)
|
||||
parsed = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
|
||||
assert parsed.real_data_validation_status == "validated" # NOT actual real-data evidence.
|
||||
for mutation in ("coverage", "reuse"):
|
||||
forged = deepcopy(source)
|
||||
if mutation == "coverage":
|
||||
forged["corporate_action_coverage"][0].update(
|
||||
status="not_validated", evidence_digests=[]
|
||||
)
|
||||
else:
|
||||
forged["readiness"]["real_data_validation"] = deepcopy(
|
||||
forged["readiness"]["contract_validation"]
|
||||
)
|
||||
seal_foundation(forged)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFoundationEnvelope.from_dict(forged, snapshot=snapshot)
|
||||
synthetic = golden("data-foundation")
|
||||
synthetic["corporate_action_coverage"][0].update(status="not_validated", evidence_digests=[])
|
||||
seal_foundation(synthetic)
|
||||
assert parse_foundation(synthetic).real_data_validation_status == "not_validated"
|
||||
|
||||
|
||||
def test_action_must_belong_to_view_selected_instrument() -> None:
|
||||
source = golden("data-foundation")
|
||||
route = source["instrument_routes"][0]
|
||||
action = {
|
||||
"action_id": "rhaction:" + "7" * 32,
|
||||
"instrument_id": route["instrument_id"],
|
||||
"observation_sequence": 1,
|
||||
"observed_by": route["observed_by"],
|
||||
"earliest_external_knowledge": {
|
||||
"status": "evidenced",
|
||||
"earliest_at": "2018-01-01T00:00:00Z",
|
||||
"evidence_digest": digest({"synthetic_action_earliest": 1}),
|
||||
},
|
||||
"history_completeness": "not_established",
|
||||
"evidence_digest": digest({"synthetic_action": 1}),
|
||||
"action_type": "cash_dividend",
|
||||
"status": "confirmed",
|
||||
"effective_time": "2018-01-02T07:00:00Z",
|
||||
"terms_digest": digest({"synthetic_terms": 1}),
|
||||
}
|
||||
identify(action, "action_revision_id", "rhcav2:")
|
||||
source["corporate_action_revisions"] = [action]
|
||||
source["standardized_views"][0]["corporate_action_revision_ids"] = [
|
||||
action["action_revision_id"]
|
||||
]
|
||||
seal_foundation(source)
|
||||
assert len(parse_foundation(source).to_dict()["corporate_action_revisions"]) == 1
|
||||
source["standardized_views"][0]["instrument_route_revision_ids"].remove(
|
||||
route["route_revision_id"]
|
||||
)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
def test_views_cannot_borrow_calendar_selection_from_each_other() -> None:
|
||||
source = golden("data-foundation")
|
||||
calendar = deepcopy(source["trading_calendar_revisions"][0])
|
||||
calendar["calendar_id"] = "rhcalendar:" + "9" * 32
|
||||
identify(calendar, "calendar_revision_id", "rhcalv2:")
|
||||
source["trading_calendar_revisions"].append(calendar)
|
||||
view = deepcopy(source["standardized_views"][0])
|
||||
view["view_id"] = "rhview:" + "8" * 32
|
||||
view["trading_calendar_revision_ids"] = [calendar["calendar_revision_id"]]
|
||||
source["standardized_views"].append(view)
|
||||
seal_foundation(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
parse_foundation(source)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"location", ["/private/a", "./a", "../a", "C:\\data\\a", "\\\\host\\a", "s3://private/a"]
|
||||
)
|
||||
def test_materialized_values_cannot_carry_physical_locations(location: str) -> None:
|
||||
source = golden("dataset-snapshot")
|
||||
rows = records()
|
||||
rows[0]["value"] = location
|
||||
bind_records(source, [rows])
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
|
||||
with pytest.raises(FactorContractError):
|
||||
snapshot.verify_materialized_records([rows])
|
||||
@@ -1,367 +0,0 @@
|
||||
"""Synthetic v2 computation boundaries; never source authentication."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from copy import deepcopy
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
FactorContractError,
|
||||
FactorDefinition,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
canonical_json_bytes,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.retrospective_data_contracts import (
|
||||
RetrospectiveFoundationEnvelope,
|
||||
RetrospectiveSnapshotEnvelope,
|
||||
)
|
||||
from quant_engine.retrospective_factor_contracts import (
|
||||
ResolvedRetrospectiveView,
|
||||
RetrospectiveCausation,
|
||||
RetrospectiveFactorSetRef,
|
||||
RetrospectiveInputBinding,
|
||||
RetrospectiveViewAvailability,
|
||||
)
|
||||
from test_retrospective_data_contracts import (
|
||||
digest,
|
||||
golden,
|
||||
identify,
|
||||
records,
|
||||
replace_at,
|
||||
seal_foundation,
|
||||
)
|
||||
|
||||
|
||||
def factor_arguments() -> dict[str, Any]:
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(
|
||||
golden("data-foundation"), snapshot=snapshot
|
||||
)
|
||||
view = next(iter(foundation.views.values()))
|
||||
factor_inputs = (FactorInput("market", view.schema_digest, ("value",)),)
|
||||
definition = FactorDefinition.create(
|
||||
factor_id="neutral_close",
|
||||
version="1.0.0",
|
||||
formula="value",
|
||||
parameters={},
|
||||
implementation_digest=digest({"synthetic_formula": "identity"}),
|
||||
input_schema_digest=factor_input_schema_digest(factor_inputs),
|
||||
inputs=factor_inputs,
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=0,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "0.1.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
schema = {"fields": ["instrument_id", "value"]}
|
||||
output = [{"instrument_id": row["instrument_id"], "value": row["value"]} for row in records()]
|
||||
return {
|
||||
"definitions": (definition,),
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"selected_view_ref_ids": (view.view_ref_id,),
|
||||
"input_bindings": (
|
||||
RetrospectiveInputBinding(
|
||||
definition.definition_id, "market", view.view_ref_id, view.schema_digest
|
||||
),
|
||||
),
|
||||
"view_availability": (
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view_receipt": 1})
|
||||
),
|
||||
),
|
||||
"dataset_chunks": [records()],
|
||||
"resolved_views": (
|
||||
ResolvedRetrospectiveView(
|
||||
view.view_ref_id,
|
||||
canonical_json_bytes({"synthetic_schema": "neutral_close_v2"}),
|
||||
canonical_json_bytes(sorted(records(), key=canonical_json_bytes)),
|
||||
),
|
||||
),
|
||||
"output_quality": OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", digest({"synthetic_quality": 1})),),
|
||||
),
|
||||
"output_coverage": OutputCoverage(
|
||||
"complete", 2, 2, "records", "synthetic_market", digest({"synthetic_coverage": 1})
|
||||
),
|
||||
"output_schema_bytes": canonical_json_bytes(schema),
|
||||
"output_content_bytes": canonical_json_bytes(output),
|
||||
"output_artifact_ref": OutputArtifactRef.create(
|
||||
schema_digest=digest(schema), content_digest=digest(output)
|
||||
),
|
||||
"evaluation_at": "2026-09-08T01:06:00Z",
|
||||
"computed_at": "2026-09-08T01:07:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:08:00Z",
|
||||
"producer": ProducerIdentity("quant_engine", "0.1.0"),
|
||||
"code_revision": "d" * 40,
|
||||
"actor": ActorIdentity("service", "synthetic.research"),
|
||||
"correlation_id": "synthetic.retrospective",
|
||||
"causation": RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def decoding_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("definitions", "dataset_snapshot", "foundation")}
|
||||
|
||||
|
||||
def test_factor_result_closes_v2_inputs_and_preserves_v1_definition() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
wire = result.to_dict()
|
||||
assert result.schema_version == "2.0.0"
|
||||
assert result.factor_set_id.startswith("rhfactorsetv2:sha256:")
|
||||
assert result.definition_ids == (arguments["definitions"][0].definition_id,)
|
||||
assert result.definition_ids[0].startswith("rhfactorv1:")
|
||||
assert wire["usage"] == "retrospective_research"
|
||||
assert wire["availability_mode"] == "retrospective_replay"
|
||||
assert wire["historical_availability"] == "not_established"
|
||||
assert wire["observation_cutoff"] == arguments["foundation"].observation_cutoff
|
||||
assert wire["decision_eligible"] is False
|
||||
assert "pit_cutoff" not in wire
|
||||
assert result.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
assert result.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
restored = RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json(), **decoding_arguments(arguments)
|
||||
)
|
||||
assert restored.to_dict() == wire
|
||||
assert restored.payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "value"),
|
||||
[
|
||||
("schema_version", "1.0.0"),
|
||||
("schema_version", "2.1.0"),
|
||||
("contract_name", "researchhub.dataset-snapshot"),
|
||||
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
|
||||
("foundation_id", "rhdfv2:sha256:" + "0" * 64),
|
||||
("observation_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("pit_cutoff", "2018-01-02T07:00:00Z"),
|
||||
("selected_view_ref_ids", []),
|
||||
("selected_view_ref_ids", ["rhviewrefv2:sha256:" + "0" * 64]),
|
||||
("definition_ids", ["rhfactorv1:sha256:" + "0" * 64]),
|
||||
("input_bindings", []),
|
||||
("input_bindings.0.input_name", "volume"),
|
||||
("input_bindings.0.schema_digest", "sha256:" + "0" * 64),
|
||||
("input_bindings.0.view_ref_id", "rhviewrefv2:sha256:" + "0" * 64),
|
||||
("view_availability", []),
|
||||
("view_availability.0.available_at", "2026-09-08T01:03:30Z"),
|
||||
("view_availability.0.available_at", "2026-09-08T01:04:00.0000001Z"),
|
||||
("upstream_evidence.quality.checks.0.status", "failed"),
|
||||
("upstream_evidence.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
|
||||
("upstream_evidence.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
|
||||
("evidence_scope", "real_data"),
|
||||
("output_quality.status", "failed"),
|
||||
("output_quality.checks.0.status", "failed"),
|
||||
("output_coverage.status", "incomplete"),
|
||||
("output_coverage.observed_count", 1),
|
||||
("output_schema_digest", "sha256:" + "0" * 64),
|
||||
("output_artifact_ref.artifact_id", "rhfactoroutputv1:sha256:" + "0" * 64),
|
||||
("availability_mode", "as_available"),
|
||||
("usage", "paper_trading"),
|
||||
("historical_availability", "declared_as_available"),
|
||||
("decision_eligible", True),
|
||||
("decision_eligible", 0),
|
||||
("evaluation_at", "2018-01-02T07:00:00Z"),
|
||||
("evaluation_at", "2026-09-08T01:04:00Z"),
|
||||
("computed_at", "2026-09-08T01:05:00Z"),
|
||||
("artifact_available_at", "2026-09-08T01:06:00Z"),
|
||||
("producer.id", "research_platform"),
|
||||
("code_revision", "unknown"),
|
||||
("actor.id", "https://private/a"),
|
||||
("causation.id", "rhdfv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_factor_rejects_reidentified_semantic_forgery(path: str, value: Any) -> None:
|
||||
arguments = factor_arguments()
|
||||
row = RetrospectiveFactorSetRef.create(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
identify(row, "factor_set_id", "rhfactorsetv2:")
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(row, **decoding_arguments(arguments))
|
||||
|
||||
|
||||
def test_payload_validation_is_never_inherited_from_serialization() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
kwargs = decoding_arguments(arguments)
|
||||
reference = RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs)
|
||||
with pytest.raises(FactorContractError):
|
||||
reference.require_payloads_revalidated()
|
||||
checked = RetrospectiveFactorSetRef.from_dict(
|
||||
result.to_dict(),
|
||||
**kwargs,
|
||||
**{
|
||||
key: arguments[key]
|
||||
for key in (
|
||||
"output_schema_bytes",
|
||||
"output_content_bytes",
|
||||
"dataset_chunks",
|
||||
"resolved_views",
|
||||
)
|
||||
},
|
||||
)
|
||||
checked.require_payloads_revalidated()
|
||||
assert checked == result
|
||||
for extra in (
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"]},
|
||||
{"dataset_chunks": arguments["dataset_chunks"]},
|
||||
{"resolved_views": arguments["resolved_views"]},
|
||||
{"output_schema_bytes": arguments["output_schema_bytes"], "output_content_bytes": b"[]"},
|
||||
{"dataset_chunks": arguments["dataset_chunks"], "resolved_views": []},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs, **extra)
|
||||
|
||||
|
||||
def test_create_verifies_actual_snapshot_and_each_resolved_view() -> None:
|
||||
for mutation in (
|
||||
"content",
|
||||
"schema",
|
||||
"snapshot",
|
||||
"duplicate_view",
|
||||
"noncanonical",
|
||||
"unknown_view",
|
||||
):
|
||||
arguments = factor_arguments()
|
||||
view = arguments["resolved_views"][0]
|
||||
if mutation == "content":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, view.schema_bytes, b"[]"),
|
||||
)
|
||||
elif mutation == "schema":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(view.view_ref_id, b"{}", view.content_bytes),
|
||||
)
|
||||
elif mutation == "snapshot":
|
||||
arguments["dataset_chunks"][0][0]["value"] = "0"
|
||||
elif mutation == "duplicate_view":
|
||||
arguments["resolved_views"] = (view, view)
|
||||
elif mutation == "unknown_view":
|
||||
arguments["resolved_views"] = (
|
||||
ResolvedRetrospectiveView(
|
||||
"rhviewrefv2:sha256:" + "0" * 64, view.schema_bytes, view.content_bytes
|
||||
),
|
||||
)
|
||||
else:
|
||||
arguments["output_content_bytes"] += b"\n"
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_parent_requires_exact_correlation_scope_and_actual_availability() -> None:
|
||||
arguments = factor_arguments()
|
||||
parent = RetrospectiveFactorSetRef.create(**arguments)
|
||||
child_args = {
|
||||
**arguments,
|
||||
"parent": parent,
|
||||
"causation": RetrospectiveCausation("factor_set", parent.factor_set_id),
|
||||
"evaluation_at": "2026-09-08T01:09:00Z",
|
||||
"computed_at": "2026-09-08T01:10:00Z",
|
||||
"artifact_available_at": "2026-09-08T01:11:00Z",
|
||||
}
|
||||
child = RetrospectiveFactorSetRef.create(**child_args)
|
||||
assert child.factor_set_id != parent.factor_set_id
|
||||
assert (
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
child.to_json(), **decoding_arguments(arguments), parent=parent
|
||||
)
|
||||
== child
|
||||
)
|
||||
for changes in (
|
||||
{"parent": None},
|
||||
{"correlation_id": "different.correlation"},
|
||||
{"causation": RetrospectiveCausation("factor_set", "rhfactorsetv2:sha256:" + "0" * 64)},
|
||||
{"evaluation_at": "2026-09-08T01:07:59Z"},
|
||||
{"causation": arguments["causation"]},
|
||||
):
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**{**child_args, **changes})
|
||||
|
||||
|
||||
def test_definition_validity_is_checked_at_actual_evaluation() -> None:
|
||||
arguments = factor_arguments()
|
||||
arguments.update(
|
||||
evaluation_at="2027-01-01T00:00:00Z",
|
||||
computed_at="2027-01-01T00:01:00Z",
|
||||
artifact_available_at="2027-01-01T00:02:00Z",
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def test_factor_contract_is_immutable_and_v1_does_not_accept_it() -> None:
|
||||
arguments = factor_arguments()
|
||||
result = RetrospectiveFactorSetRef.create(**arguments)
|
||||
exported = result.to_dict()
|
||||
exported["upstream_evidence"]["quality"]["status"] = "failed"
|
||||
assert result.upstream_evidence["quality"]["status"] == "passed"
|
||||
with pytest.raises(TypeError):
|
||||
result.upstream_evidence["quality"]["status"] = "failed"
|
||||
with pytest.raises(FactorContractError):
|
||||
FactorSetRef.from_dict(result.to_dict(), **decoding_arguments(arguments))
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveFactorSetRef.from_json(
|
||||
result.to_json() + "\n", **decoding_arguments(arguments)
|
||||
)
|
||||
with pytest.raises(FactorContractError):
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
"rhviewrefv1:sha256:" + "0" * 64,
|
||||
"sha256:" + "0" * 64,
|
||||
)
|
||||
|
||||
|
||||
def test_missing_synthetic_or_real_readiness_cannot_be_relabelled() -> None:
|
||||
arguments = factor_arguments()
|
||||
snapshot_row = arguments["dataset_snapshot"].to_dict()
|
||||
snapshot_row["evidence_scope"] = "real_data"
|
||||
identify(snapshot_row, "snapshot_id", "rhdsv2:")
|
||||
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_row)
|
||||
foundation_row = arguments["foundation"].to_dict()
|
||||
foundation_row["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
foundation_row["readiness"]["evidence_scope"] = "real_data"
|
||||
for view in foundation_row["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot.snapshot_id
|
||||
seal_foundation(foundation_row)
|
||||
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation_row, snapshot=snapshot)
|
||||
view = next(iter(foundation.views.values()))
|
||||
arguments.update(
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
evidence_scope="real_data",
|
||||
selected_view_ref_ids=(view.view_ref_id,),
|
||||
input_bindings=(
|
||||
RetrospectiveInputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
view.view_ref_id,
|
||||
view.schema_digest,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
RetrospectiveViewAvailability(
|
||||
view.view_ref_id, view.available_at, digest({"synthetic_view": 1})
|
||||
),
|
||||
),
|
||||
causation=RetrospectiveCausation("foundation", foundation.foundation_id),
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="real-data"):
|
||||
RetrospectiveFactorSetRef.create(**arguments)
|
||||
@@ -1,655 +0,0 @@
|
||||
"""New synthetic S4 evidence; historical valuation is not actual availability."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import FrozenInstanceError, replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
PortfolioRiskContractError,
|
||||
RiskAssessmentStatus,
|
||||
RiskFindingCode,
|
||||
)
|
||||
from quant_engine.artifact import EvidenceQualification, PerformanceEvidenceError
|
||||
from quant_engine.factor_contracts import FactorContractError
|
||||
from quant_engine.governed_pipeline import BacktestContractError
|
||||
from quant_engine.risk import ComponentRiskResult, CovarianceSnapshot, labeled_component_risk
|
||||
import quant_engine.retrospective_portfolio_risk_contracts as contracts
|
||||
from quant_engine.retrospective_artifact_contracts import (
|
||||
build_retrospective_backtest_evidence_manifest,
|
||||
)
|
||||
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
|
||||
from quant_engine.retrospective_portfolio_risk_contracts import (
|
||||
RetrospectivePortfolioDecision,
|
||||
RetrospectivePortfolioTarget,
|
||||
RetrospectiveRiskAssessment,
|
||||
build_retrospective_portfolio_decision,
|
||||
compute_retrospective_portfolio_receipt_digests,
|
||||
assess_retrospective_portfolio_risk,
|
||||
)
|
||||
from test_retrospective_artifact_contracts import synthetic_artifact
|
||||
from test_retrospective_backtest_contracts import run_arguments
|
||||
from test_retrospective_data_contracts import digest, replace_at
|
||||
|
||||
ASSETS = ("rhinstrument:" + "1" * 32, "rhinstrument:" + "2" * 32)
|
||||
CONTRACT_ERRORS = (
|
||||
FactorContractError,
|
||||
PortfolioRiskContractError,
|
||||
BacktestContractError,
|
||||
PerformanceEvidenceError,
|
||||
)
|
||||
|
||||
|
||||
def portfolio_arguments() -> dict[str, Any]:
|
||||
run = RetrospectiveBacktestRunRef.create(**run_arguments())
|
||||
artifact = synthetic_artifact(run)
|
||||
manifest = build_retrospective_backtest_evidence_manifest(
|
||||
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
|
||||
)
|
||||
target = RetrospectivePortfolioTarget.create(
|
||||
backtest_run_id=run.run_id,
|
||||
dataset_snapshot_id=run.dataset_snapshot_id,
|
||||
weights={ASSETS[0]: 0.6, ASSETS[1]: 0.4},
|
||||
effective_at="2018-01-05T07:00:00Z",
|
||||
created_at="2026-09-08T01:12:00Z",
|
||||
)
|
||||
return {
|
||||
"backtest_run_ref": run,
|
||||
"manifest": manifest,
|
||||
"target": target,
|
||||
"objective_name": "synthetic_allocation",
|
||||
"objective_version": "1.0.0",
|
||||
"objective_digest": digest({"synthetic_objective": 1}),
|
||||
"model_name": "bounded_weights",
|
||||
"model_version": "1.0.0",
|
||||
"model_digest": digest({"synthetic_model": 1}),
|
||||
"expected_return_digest": digest({"synthetic_returns": 1}),
|
||||
"covariance_digest": "sha256:" + "a" * 64,
|
||||
"scenario_digest": digest({"synthetic_scenario": 1}),
|
||||
"constraints": ConstraintSetV1(
|
||||
gross_exposure_max=1.0,
|
||||
net_exposure_min=1.0,
|
||||
net_exposure_max=1.0,
|
||||
single_asset_min=0.2,
|
||||
single_asset_max=0.7,
|
||||
position_count_max=2,
|
||||
turnover_max=0.2,
|
||||
),
|
||||
"freshness_policy": FreshnessPolicy(
|
||||
max_manifest_age_seconds=3600, max_covariance_age_days=0
|
||||
),
|
||||
"prior_weights": {ASSETS[0]: 0.5, ASSETS[1]: 0.5},
|
||||
"computed_at": "2026-09-08T01:13:00Z",
|
||||
}
|
||||
|
||||
|
||||
def portfolio_receipt(arguments: dict[str, Any], **changes: Any) -> ComputationReceipt:
|
||||
values = compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key not in {"computed_at", "receipt"}}
|
||||
)
|
||||
return ComputationReceipt(
|
||||
**{
|
||||
"algorithm": "bounded_weights",
|
||||
"algorithm_version": "1.0.0",
|
||||
"implementation_digest": digest({"synthetic_implementation": 1}),
|
||||
"parameter_digest": digest({"synthetic_parameters": 1}),
|
||||
"input_digest": values["input_digest"],
|
||||
"constraint_digest": values["constraint_digest"],
|
||||
"output_digest": values["output_digest"],
|
||||
"status": "completed",
|
||||
"solver_required": False,
|
||||
"solver_name": None,
|
||||
"solver_version": None,
|
||||
"solver_config_digest": None,
|
||||
"iterations": None,
|
||||
"objective_value": None,
|
||||
"max_constraint_residual": values["max_constraint_residual"],
|
||||
"tolerance": 1e-12,
|
||||
"computed_at": arguments["computed_at"],
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_target_separates_historical_effective_time_from_actual_creation() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
target = arguments["target"]
|
||||
assert target.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert target.created_at == "2026-09-08T01:12:00Z"
|
||||
assert target.target_id.startswith("rhportfoliotargetv2:sha256:")
|
||||
assert target.to_dict()["usage"] == "retrospective_research"
|
||||
|
||||
|
||||
def test_portfolio_decision_preserves_constraints_and_actual_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
assert decision.decision_id.startswith("rhportfoliodecisionv2:sha256:")
|
||||
assert decision.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert decision.created_at == "2026-09-08T01:12:00Z"
|
||||
assert decision.computed_at == "2026-09-08T01:13:00Z"
|
||||
assert decision.gross_exposure == 1.0
|
||||
assert decision.position_count == 2
|
||||
assert decision.to_dict()["decision_eligible"] is False
|
||||
|
||||
|
||||
def covariance(arguments: dict[str, Any], **changes: Any) -> CovarianceSnapshot:
|
||||
return CovarianceSnapshot(
|
||||
**{
|
||||
"snapshot_id": "covariance:synthetic-retrospective",
|
||||
"as_of_date": "2018-01-05",
|
||||
"covariance": pd.DataFrame([[0.04, 0.01], [0.01, 0.09]], index=ASSETS, columns=ASSETS),
|
||||
"return_frequency": "1d",
|
||||
"periods_per_year": 252,
|
||||
"method": "provided",
|
||||
"window_start_date": "2018-01-02",
|
||||
"window_end_date": "2018-01-05",
|
||||
"observations": 4,
|
||||
"lookback_sessions": 4,
|
||||
"missing_policy": "complete_case",
|
||||
"data_snapshot_id": arguments["backtest_run_ref"].dataset_snapshot_id,
|
||||
"input_sha256": "a" * 64,
|
||||
**changes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def risk_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
decision = build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
)
|
||||
return {
|
||||
"portfolio_decision": decision,
|
||||
"backtest_run_ref": arguments["backtest_run_ref"],
|
||||
"manifest": arguments["manifest"],
|
||||
"covariance": covariance(arguments),
|
||||
"risk_model_name": "euler_volatility",
|
||||
"risk_model_version": "1.0.0",
|
||||
"risk_model_digest": digest({"synthetic_risk_model": 1}),
|
||||
"risk_budget": {ASSETS[0]: 0.8, ASSETS[1]: 0.8},
|
||||
"portfolio_volatility_limit": 10.0,
|
||||
"groups": {ASSETS[0]: "equity", ASSETS[1]: "fixed_income"},
|
||||
"computed_at": "2026-09-08T01:14:00Z",
|
||||
}
|
||||
|
||||
|
||||
def test_risk_uses_historical_business_age_and_actual_computation_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.assessment_id.startswith("rhriskassessmentv2:sha256:")
|
||||
assert result.qualified is True
|
||||
assert result.effective_at == "2018-01-05T07:00:00Z"
|
||||
assert result.computed_at == "2026-09-08T01:14:00Z"
|
||||
assert result.to_dict()["decision_eligible"] is False
|
||||
assert result.to_dict()["execution_validation"] == "not_validated"
|
||||
assert sum(result.percentage_risk.values()) == pytest.approx(1.0)
|
||||
assert sum(result.component_risk.values()) == pytest.approx(result.portfolio_volatility)
|
||||
|
||||
|
||||
def test_new_risk_computation_cannot_reuse_stale_actual_manifest_time() -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments["computed_at"] = "2026-09-08T02:11:01Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
def target_with(arguments: dict[str, Any], **changes: Any) -> RetrospectivePortfolioTarget:
|
||||
row = arguments["target"].to_dict()
|
||||
return RetrospectivePortfolioTarget.create(
|
||||
**{
|
||||
key: value
|
||||
for key, value in {**row, **changes}.items()
|
||||
if key
|
||||
in {"backtest_run_id", "dataset_snapshot_id", "weights", "effective_at", "created_at"}
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def decision_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {key: arguments[key] for key in ("backtest_run_ref", "manifest", "target")}
|
||||
|
||||
|
||||
def assessment_context(arguments: dict[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
key: arguments[key]
|
||||
for key in ("portfolio_decision", "backtest_run_ref", "manifest", "covariance")
|
||||
}
|
||||
|
||||
|
||||
def reidentify(row: dict[str, Any], field: str, prefix: str) -> None:
|
||||
row.pop(field, None)
|
||||
encoded = json.dumps(
|
||||
row, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False
|
||||
)
|
||||
row[field] = prefix + "sha256:" + hashlib.sha256(encoded.encode()).hexdigest()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"parser",
|
||||
[RetrospectivePortfolioTarget, RetrospectivePortfolioDecision, RetrospectiveRiskAssessment],
|
||||
)
|
||||
def test_json_syntax_failures_use_typed_contract_errors(parser: Any) -> None:
|
||||
with pytest.raises(FactorContractError):
|
||||
parser.from_json(b"{")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"method": "alternate_estimator"},
|
||||
{"window_start_date": "2018-01-03"},
|
||||
{"window_end_date": "2018-01-04"},
|
||||
{"observations": 3},
|
||||
{"lookback_sessions": 5},
|
||||
{"missing_policy": "alternate_missing_policy"},
|
||||
],
|
||||
)
|
||||
def test_covariance_estimation_context_is_bound_into_the_result_identity(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
original = assess_retrospective_portfolio_risk(**arguments)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
changed = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert changed.assessment_id != original.assessment_id
|
||||
|
||||
|
||||
def test_canonical_roundtrips_and_immutable_results() -> None:
|
||||
base = portfolio_arguments()
|
||||
target = base["target"]
|
||||
assert RetrospectivePortfolioTarget.from_json(target.to_json().encode()) == target
|
||||
decision = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base))
|
||||
assert (
|
||||
RetrospectivePortfolioDecision.from_json(decision.to_json(), **decision_context(base))
|
||||
== decision
|
||||
)
|
||||
arguments = risk_arguments(base)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert (
|
||||
RetrospectiveRiskAssessment.from_json(
|
||||
result.to_json().encode(), **assessment_context(arguments)
|
||||
)
|
||||
== result
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
target.weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
target.created_at = "2018-01-05T07:00:00Z"
|
||||
with pytest.raises(TypeError):
|
||||
decision.target_weights[ASSETS[0]] = 0.1
|
||||
with pytest.raises(TypeError):
|
||||
result.component_risk[ASSETS[0]] = 0.1
|
||||
detached = result.to_dict()
|
||||
detached["component_risk"][ASSETS[0]] = 0.1
|
||||
assert detached != result.to_dict()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"weights": {}},
|
||||
{"weights": {"SIM0": 1.0}},
|
||||
{"weights": {ASSETS[0]: float("nan")}},
|
||||
{"weights": {ASSETS[0]: True}},
|
||||
{"backtest_run_id": "rhbacktestrunv1:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhds:sha256:" + "0" * 64},
|
||||
{"effective_at": "2026-09-09T01:00:00Z"},
|
||||
{"created_at": "2026-09-08T01:12:00.1234567Z"},
|
||||
{"effective_at": "2018-01-05T15:00:00+08:00"},
|
||||
],
|
||||
)
|
||||
def test_target_rejects_legacy_ambiguous_and_nonfinite_inputs(change: dict[str, Any]) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
target_with(portfolio_arguments(), **change)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("usage", "live"),
|
||||
("historical_availability", "established"),
|
||||
("schema_version", "1.0.0"),
|
||||
("extra", True),
|
||||
("target_id", "rhportfoliotargetv2:sha256:" + "0" * 64),
|
||||
],
|
||||
)
|
||||
def test_target_rejects_wire_mutations(path: str, value: Any) -> None:
|
||||
row = portfolio_arguments()["target"].to_dict()
|
||||
row[path] = value
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_dict(row)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("field", ["input_digest", "constraint_digest", "output_digest"])
|
||||
def test_receipt_digests_are_recomputed(field: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments, **{field: "sha256:" + "0" * 64})
|
||||
with pytest.raises(FactorContractError, match="independently recomputed"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", ["failed", "fallback"])
|
||||
def test_failed_or_fallback_solver_cannot_form_a_decision(status: str) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status=status,
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="failed/fallback"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"backtest_run_id": "rhbacktestrunv2:sha256:" + "0" * 64},
|
||||
{"dataset_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"weights": {"rhinstrument:" + "f" * 32: 1.0}},
|
||||
{"created_at": "2026-09-08T01:10:00Z"},
|
||||
{"created_at": "2026-09-08T01:14:00Z"},
|
||||
],
|
||||
)
|
||||
def test_decision_closes_target_identity_assets_and_actual_time(change: dict[str, Any]) -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
arguments["target"] = target_with(arguments, **change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_actual_manifest_freshness_boundary_and_receipt_time() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00Z"
|
||||
assert (
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments)
|
||||
).computed_at
|
||||
== arguments["computed_at"]
|
||||
)
|
||||
arguments["computed_at"] = "2026-09-08T02:11:00.000001Z"
|
||||
with pytest.raises(FactorContractError, match="stale"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=portfolio_receipt(arguments))
|
||||
arguments["computed_at"] = "2026-09-08T01:13:00Z"
|
||||
with pytest.raises(FactorContractError, match="receipt actual time"):
|
||||
build_retrospective_portfolio_decision(
|
||||
**arguments, receipt=portfolio_receipt(arguments, computed_at="2026-09-08T01:13:01Z")
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_tables_and_qualification_are_revalidated_at_s4_boundary() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
receipt = portfolio_receipt(arguments)
|
||||
manifest = arguments["manifest"]
|
||||
artifact = manifest._artifact
|
||||
arguments["manifest"] = build_retrospective_backtest_evidence_manifest(
|
||||
arguments["backtest_run_ref"],
|
||||
artifact,
|
||||
artifact_available_at=manifest.artifact_available_at,
|
||||
qualification=EvidenceQualification.EXPLORATORY,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="contract-qualified"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments["manifest"] = manifest
|
||||
# Public access is an isolated copy. Simulate corruption of the retained bytes,
|
||||
# beyond that normal interface, to exercise the consumer's independent recheck.
|
||||
artifact._performance.loc[0, "n_days"] += 1
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
|
||||
|
||||
def test_constraint_residuals_and_prior_assets_cannot_be_bypassed() -> None:
|
||||
arguments = portfolio_arguments()
|
||||
arguments["constraints"] = ConstraintSetV1(gross_exposure_max=0.9)
|
||||
# A solver may report convergence within its tolerance; actual contract constraints still bind.
|
||||
receipt = portfolio_receipt(
|
||||
arguments,
|
||||
status="converged",
|
||||
solver_required=True,
|
||||
solver_name="synthetic_solver",
|
||||
solver_version="1.0.0",
|
||||
solver_config_digest=digest({"synthetic_solver": 1}),
|
||||
iterations=1,
|
||||
objective_value=0.0,
|
||||
tolerance=0.2,
|
||||
)
|
||||
with pytest.raises(FactorContractError, match="violates supported constraints"):
|
||||
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
|
||||
arguments = portfolio_arguments()
|
||||
arguments["prior_weights"] = {"rhinstrument:" + "f" * 32: 0.5}
|
||||
with pytest.raises(FactorContractError, match="prior assets"):
|
||||
compute_retrospective_portfolio_receipt_digests(
|
||||
**{key: value for key, value in arguments.items() if key != "computed_at"}
|
||||
)
|
||||
arguments["prior_weights"] = None
|
||||
with pytest.raises(PortfolioRiskContractError, match="prior"):
|
||||
portfolio_receipt(arguments)
|
||||
|
||||
|
||||
def test_optional_prior_budget_limit_and_groups_have_explicit_empty_semantics() -> None:
|
||||
base = portfolio_arguments()
|
||||
base["constraints"] = ConstraintSetV1(gross_exposure_max=1.0)
|
||||
base["prior_weights"] = None
|
||||
arguments = risk_arguments(base)
|
||||
arguments.update(risk_budget=None, portfolio_volatility_limit=None, groups=None)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.qualified is True
|
||||
assert result.risk_budget == {}
|
||||
assert result.group_exposure == {}
|
||||
assert result.groups is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("gross_exposure", True),
|
||||
("position_count", 2.0),
|
||||
("target_weights." + ASSETS[0], 0.5),
|
||||
("schema_version", "1.0.0"),
|
||||
("observation_cutoff", "2018-01-05T07:00:00Z"),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_decision_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
base = portfolio_arguments()
|
||||
row = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base)).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "decision_id", "rhportfoliodecisionv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioDecision.from_dict(row, **decision_context(base))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"as_of_date": "2018-01-06"},
|
||||
{"as_of_date": "2018-01-04", "window_end_date": "2018-01-04"},
|
||||
{"window_start_date": None, "window_end_date": None},
|
||||
{"data_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
|
||||
{"input_sha256": "b" * 64},
|
||||
],
|
||||
)
|
||||
def test_covariance_business_time_bounds_and_source_binding(change: dict[str, Any]) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(base, **change)
|
||||
with pytest.raises(FactorContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,index,columns",
|
||||
[
|
||||
([[float("nan"), 0.0], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.1], [0.0, 0.1]], ASSETS, ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], ASSETS[0]), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], "unknown"), ASSETS),
|
||||
([[0.1, 0.0], [0.0, 0.1]], ASSETS, (ASSETS[0], "unknown")),
|
||||
],
|
||||
)
|
||||
def test_covariance_structure_is_checked_before_computation(
|
||||
matrix: Any, index: Any, columns: Any
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=index, columns=columns)
|
||||
)
|
||||
with pytest.raises(PortfolioRiskContractError):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change",
|
||||
[
|
||||
{"risk_budget": {ASSETS[0]: -0.1}},
|
||||
{"risk_budget": {"unknown": 0.1}},
|
||||
{"portfolio_volatility_limit": -0.1},
|
||||
{"groups": {ASSETS[0]: "equity"}},
|
||||
{"groups": []},
|
||||
{"risk_model_version": "latest"},
|
||||
{"risk_model_name": "/private/model"},
|
||||
{"computed_at": "2026-09-08T01:12:59Z"},
|
||||
{"portfolio_decision": object()},
|
||||
{"covariance": object()},
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_invalid_models_budgets_clocks_and_untyped_inputs(
|
||||
change: dict[str, Any],
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
assess_retrospective_portfolio_risk(**arguments)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"matrix,finding",
|
||||
[
|
||||
([[1.0, 2.0], [2.0, 1.0]], RiskFindingCode.COVARIANCE_NOT_PSD),
|
||||
([[0.0, 0.0], [0.0, 0.0]], RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE),
|
||||
],
|
||||
)
|
||||
def test_numerical_unavailability_is_not_qualification(
|
||||
matrix: Any, finding: RiskFindingCode
|
||||
) -> None:
|
||||
base = portfolio_arguments()
|
||||
arguments = risk_arguments(base)
|
||||
arguments["covariance"] = covariance(
|
||||
base, covariance=pd.DataFrame(matrix, index=ASSETS, columns=ASSETS)
|
||||
)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.qualified is False
|
||||
assert result.findings == (finding,)
|
||||
assert result.portfolio_volatility is None
|
||||
|
||||
|
||||
def test_risk_uses_the_existing_numeric_implementation_exactly_once(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
calls = []
|
||||
|
||||
def recorded(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
calls.append((weights, matrix))
|
||||
return labeled_component_risk(weights, matrix)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", recorded)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert len(calls) == 1
|
||||
expected = labeled_component_risk(*calls[0])
|
||||
assert result.component_risk == expected.component.to_dict()
|
||||
assert result.portfolio_volatility == expected.portfolio_volatility
|
||||
|
||||
|
||||
def test_unknown_numeric_failures_are_sanitized(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def failed(*args: Any) -> ComponentRiskResult:
|
||||
raise ValueError("synthetic internal detail")
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", failed)
|
||||
with pytest.raises(PortfolioRiskContractError, match="risk computation failed") as error:
|
||||
assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert "internal detail" not in str(error.value)
|
||||
|
||||
|
||||
def test_nonclosed_decomposition_is_unavailable(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def nonclosed(weights: Any, matrix: Any) -> ComponentRiskResult:
|
||||
output = labeled_component_risk(weights, matrix)
|
||||
return replace(output, component=output.component * 0.5)
|
||||
|
||||
monkeypatch.setattr(contracts, "labeled_component_risk", nonclosed)
|
||||
result = assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
|
||||
assert result.status is RiskAssessmentStatus.UNAVAILABLE
|
||||
assert result.findings == (RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED,)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"change", [{"portfolio_volatility_limit": 0.0}, {"risk_budget": {ASSETS[0]: 0.0}}]
|
||||
)
|
||||
def test_budget_breach_keeps_ready_but_unqualified_evidence(change: dict[str, Any]) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
arguments.update(change)
|
||||
result = assess_retrospective_portfolio_risk(**arguments)
|
||||
assert result.status is RiskAssessmentStatus.READY
|
||||
assert result.qualified is False
|
||||
assert result.findings == (RiskFindingCode.RISK_BUDGET_BREACH,)
|
||||
assert result.decision_eligible is False
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"path,value",
|
||||
[
|
||||
("decision_eligible", True),
|
||||
("execution_validation", "validated"),
|
||||
("historical_availability", "established"),
|
||||
("qualified", 1),
|
||||
("portfolio_volatility", 1.0),
|
||||
("component_risk." + ASSETS[0], 1.0),
|
||||
("schema_version", "1.0.0"),
|
||||
("covariance_matrix_digest", "sha256:" + "0" * 64),
|
||||
("extra", True),
|
||||
],
|
||||
)
|
||||
def test_risk_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
|
||||
arguments = risk_arguments(portfolio_arguments())
|
||||
row = assess_retrospective_portfolio_risk(**arguments).to_dict()
|
||||
replace_at(row, path, value)
|
||||
reidentify(row, "assessment_id", "rhriskassessmentv2:")
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectiveRiskAssessment.from_dict(row, **assessment_context(arguments))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"raw",
|
||||
[
|
||||
b'{"x":1,"x":2}',
|
||||
b'{ "x":1}',
|
||||
b"[]",
|
||||
b'{"x":NaN}',
|
||||
b'{"x":Infinity}',
|
||||
b'{"x":9007199254740992}',
|
||||
1,
|
||||
],
|
||||
)
|
||||
def test_json_profiles_reject_ambiguous_nonfinite_and_noncanonical_input(raw: Any) -> None:
|
||||
with pytest.raises(CONTRACT_ERRORS):
|
||||
RetrospectivePortfolioTarget.from_json(raw)
|
||||
@@ -1,293 +0,0 @@
|
||||
"""Strategy reports preserve real ledger facts without factor-score fabrication."""
|
||||
|
||||
import json
|
||||
from dataclasses import asdict
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.strategy_artifact import build_strategy_research_artifact
|
||||
from quant_engine.strategy_optimizer import optimize_strategy_research
|
||||
from quant_engine.strategy_research import BenchmarkInput, run_strategy_research
|
||||
|
||||
|
||||
def bars(closes, opens=None):
|
||||
close = np.asarray(closes, dtype=float)
|
||||
opening = np.asarray(opens if opens is not None else closes, dtype=float)
|
||||
return pd.DataFrame(
|
||||
{
|
||||
"open": opening,
|
||||
"high": np.maximum(close, opening),
|
||||
"low": np.minimum(close, opening),
|
||||
"close": close,
|
||||
},
|
||||
index=pd.date_range("2026-01-01", periods=len(close), freq="B"),
|
||||
)
|
||||
|
||||
|
||||
def run(strategy="BuyAndHold", feed=None, **kwargs):
|
||||
return run_strategy_research(
|
||||
strategy,
|
||||
bars([10, 11, 12, 13]) if feed is None else feed,
|
||||
asset="SYNTHETIC",
|
||||
initial_cash=1000,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
def build(result, **kwargs):
|
||||
metadata = {
|
||||
"run_id": "strategy-run",
|
||||
"strategy_id": "isolated-strategy",
|
||||
"strategy_name": "Synthetic",
|
||||
"strategy_version": "1",
|
||||
"engine_version": "candidate",
|
||||
"code_revision": "candidate",
|
||||
"data_snapshot_id": "synthetic:ohlc-v1",
|
||||
"calendar": "synthetic-sessions",
|
||||
"timezone": "Asia/Shanghai",
|
||||
"started_at": "2026-01-08T10:00:00+08:00",
|
||||
"finished_at": "2026-01-08T10:00:01+08:00",
|
||||
"parameters": {"synthetic": True},
|
||||
}
|
||||
return build_strategy_research_artifact(result, **(metadata | kwargs))
|
||||
|
||||
|
||||
def report(artifact):
|
||||
return json.loads(artifact.run.iloc[0]["params_json"])["strategy_report"]
|
||||
|
||||
|
||||
def test_projects_same_ledger_cash_fees_positions_and_completed_trade_basis():
|
||||
result = run(
|
||||
"SmaCross",
|
||||
bars([10, 8, 12, 6, 14, 5, 12]),
|
||||
params={"fast": 1, "slow": 2},
|
||||
commission=0.01,
|
||||
stamp_duty=0.02,
|
||||
)
|
||||
artifact = build(result)
|
||||
detail = report(artifact)
|
||||
assert artifact.schema_version == "1.1.0"
|
||||
assert artifact.nav.portfolio_value.tolist() == result.ledger.nav_series.tolist()
|
||||
assert artifact.nav.pnl_pct.tolist() == result.ledger.daily_returns.tolist()
|
||||
assert artifact.trades.fee.sum() == pytest.approx(result.ledger.trades_frame.fee.sum())
|
||||
assert artifact.nav.total_cost.sum() == pytest.approx(artifact.trades.total_cost.sum())
|
||||
assert artifact.performance.iloc[0].win_rate == result.pairing.win_rate
|
||||
assert artifact.performance.iloc[0].n_trades == len(result.ledger.trades_frame)
|
||||
assert detail["trade_pairing"] == json.loads(json.dumps(asdict(result.pairing)))
|
||||
assert detail["costs"] == {
|
||||
"initial_cash": 1000,
|
||||
"commission": 0.01,
|
||||
"stamp_duty": 0.02,
|
||||
"min_trade_amount": 0,
|
||||
"slippage_bps": 0,
|
||||
}
|
||||
assert detail["decision_eligible"] is False
|
||||
for day, frame in artifact.positions.groupby("trade_date"):
|
||||
nav = artifact.nav.loc[artifact.nav.trade_date == day].iloc[0]
|
||||
assert frame.market_value.sum() == pytest.approx(nav.portfolio_value)
|
||||
assert frame.weight.sum() == pytest.approx(1)
|
||||
assert artifact.signals.empty
|
||||
assert artifact.attribution.empty
|
||||
assert artifact.risk.empty
|
||||
assert detail["projections"]["signals"] == "strategy_report.signals"
|
||||
assert detail["projections"]["attribution"] == "not_computed"
|
||||
params = json.loads(artifact.run.iloc[0].params_json)
|
||||
assert params["performance_interpretation"]["win_rate_basis"] == "completed_trades"
|
||||
|
||||
|
||||
def test_last_signal_is_not_lost_or_fabricated_as_a_factor_signal():
|
||||
result = run("SmaCross", bars([10, 8, 12]), params={"fast": 1, "slow": 2})
|
||||
artifact = build(result)
|
||||
signal = report(artifact)["signals"][0]
|
||||
assert signal["status"] == "no_next_session"
|
||||
assert signal["execution_date"] is None
|
||||
assert signal["signal_id"] == "strategy-run:signal:2026-01-05"
|
||||
assert artifact.trades.empty
|
||||
assert artifact.signals.empty
|
||||
|
||||
|
||||
def test_signal_ids_join_actual_fills_and_report():
|
||||
artifact = build(run())
|
||||
signals = {item["signal_id"]: item for item in report(artifact)["signals"]}
|
||||
for fill in artifact.trades.to_dict("records"):
|
||||
signal = signals[fill["signal_id"]]
|
||||
assert signal["execution_date"] == fill["trade_date"].isoformat()
|
||||
assert signal["decision_date"] < signal["execution_date"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name",
|
||||
[
|
||||
"BuyAndHold",
|
||||
"SmaCross",
|
||||
"MACross",
|
||||
"RSI",
|
||||
"BollingerBreakout",
|
||||
"DualThrust",
|
||||
"TurtleBreakout",
|
||||
],
|
||||
)
|
||||
def test_all_seven_defaults_have_canonical_serializable_reports(name):
|
||||
values = 10 + np.sin(np.arange(80) / 2) * 2
|
||||
result = run(name, bars(values, np.r_[values[0], values[:-1]]))
|
||||
artifact = build(result)
|
||||
assert report(artifact)["parameters"] == result.parameters
|
||||
assert json.loads(artifact.canonical_json())["schema_version"] == "1.1.0"
|
||||
assert artifact.content_sha256 == build(result).content_sha256
|
||||
assert len(artifact.nav) == 80
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", ["not_requested", "empty", "present"])
|
||||
def test_benchmark_states_and_original_returns_are_preserved(status):
|
||||
feed = bars([10, 11, 12, 13])
|
||||
closes = (
|
||||
pd.Series([20, 22, 21, 23], index=feed.index)
|
||||
if status == "present"
|
||||
else (pd.Series(dtype=float) if status == "empty" else None)
|
||||
)
|
||||
result = run(feed=feed, benchmark=BenchmarkInput(status, closes))
|
||||
artifact = build(result, benchmark_id="SYNTHETIC-BENCH" if status != "not_requested" else None)
|
||||
assert report(artifact)["benchmark"]["status"] == status
|
||||
if status == "present":
|
||||
assert artifact.nav.benchmark_return.tolist() == pytest.approx(
|
||||
[0, 0.1, 21 / 22 - 1, 23 / 21 - 1]
|
||||
)
|
||||
assert artifact.nav.benchmark_nav.tolist() == pytest.approx([1, 1.1, 1.05, 1.15])
|
||||
assert artifact.run.iloc[0].benchmark_alignment_policy == "exact_session_index"
|
||||
else:
|
||||
assert artifact.nav.benchmark_nav.isna().all()
|
||||
assert artifact.run.iloc[0].benchmark_alignment_policy == "none"
|
||||
|
||||
|
||||
def test_zero_nav_preserves_zero_value_and_undefined_weight_with_reason():
|
||||
result = run(
|
||||
"SmaCross",
|
||||
bars([10, 8, 12, 6, 14]),
|
||||
params={"fast": 1, "slow": 2},
|
||||
commission=0,
|
||||
stamp_duty=1,
|
||||
)
|
||||
artifact = build(result)
|
||||
last = artifact.positions.iloc[-1]
|
||||
assert last.market_value == 0
|
||||
assert pd.isna(last.weight)
|
||||
assert artifact.nav.iloc[-1].nav == 0
|
||||
assert report(artifact)["projections"]["undefined_weight_dates"] == ["2026-01-07"]
|
||||
|
||||
|
||||
def test_no_closed_lot_metrics_are_null_with_covered_reason():
|
||||
artifact = build(run())
|
||||
metadata = json.loads(artifact.run.iloc[0].params_json)
|
||||
assert report(artifact)["metrics"]["trade_win_rate"] is None
|
||||
assert (
|
||||
metadata["performance_interpretation"]["unavailable_reasons"]["win_rate"]
|
||||
== "no_closed_lots"
|
||||
)
|
||||
assert pd.isna(artifact.performance.iloc[0].win_rate)
|
||||
|
||||
|
||||
def test_result_snapshots_detach_caller_data_and_returned_views():
|
||||
feed = bars([10, 11, 12, 13])
|
||||
result = run(feed=feed)
|
||||
before = build(result).content_sha256
|
||||
feed.iloc[:] = 999
|
||||
view = result.bars
|
||||
view.iloc[:] = 777
|
||||
artifact = build(result)
|
||||
assert artifact.content_sha256 == before
|
||||
positions = artifact.positions
|
||||
positions["market_value"] = 0
|
||||
assert artifact.content_sha256 == before
|
||||
|
||||
|
||||
def test_grid_artifact_retains_all_ranks_and_selected_ledger_and_detaches_input():
|
||||
grid = {"buy_pct": [0.2, 0.5, 1]}
|
||||
result = optimize_strategy_research(
|
||||
"BuyAndHold",
|
||||
bars([10, 11, 12, 13]),
|
||||
asset="SYNTHETIC",
|
||||
param_grid=grid,
|
||||
objective="total_return",
|
||||
initial_cash=1000,
|
||||
commission=0,
|
||||
stamp_duty=0,
|
||||
)
|
||||
grid["buy_pct"].append(0.9)
|
||||
artifact = build(result)
|
||||
ranking = report(artifact)["optimization"]
|
||||
assert ranking["grid"] == {"buy_pct": [0.2, 0.5, 1]}
|
||||
assert ranking["trial_count"] == 3
|
||||
assert ranking["selected_rank"] == 1
|
||||
assert [trial["rank"] for trial in ranking["trials"]] == [1, 2, 3]
|
||||
assert [trial["score"] for trial in ranking["trials"]] == [
|
||||
trial.score for trial in result.trials
|
||||
]
|
||||
assert ranking["trials"][0]["parameters"] == {"buy_pct": 1}
|
||||
assert (
|
||||
artifact.nav.portfolio_value.tolist() == result.trials[0].result.ledger.nav_series.tolist()
|
||||
)
|
||||
assert all("ledger" not in trial for trial in ranking["trials"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("key", ["strategy_report", "performance_interpretation"])
|
||||
def test_callers_cannot_overwrite_authoritative_report_or_metric_explanation(key):
|
||||
with pytest.raises(ValueError, match="reserved"):
|
||||
build(run(), parameters={key: {"decision_eligible": True}})
|
||||
|
||||
|
||||
def test_report_mutation_changes_canonical_artifact_digest():
|
||||
first = build(run(commission=0))
|
||||
second = build(run(commission=0.01))
|
||||
assert first.content_sha256 != second.content_sha256
|
||||
assert first.run.iloc[0].config_hash != second.run.iloc[0].config_hash
|
||||
|
||||
|
||||
def test_invalid_metadata_fails_before_artifact_creation():
|
||||
with pytest.raises(ValueError, match="finished_at"):
|
||||
build(run(), finished_at="2026-01-07T10:00:00+08:00")
|
||||
with pytest.raises(ValueError, match="benchmark"):
|
||||
build(run(), benchmark_id="FAKE")
|
||||
|
||||
|
||||
def test_missing_relative_metrics_explain_their_fact_column_names():
|
||||
artifact = build(run())
|
||||
reasons = json.loads(artifact.run.iloc[0].params_json)["performance_interpretation"][
|
||||
"unavailable_reasons"
|
||||
]
|
||||
assert reasons["ir"] == "benchmark_not_requested"
|
||||
assert "information_ratio" not in reasons
|
||||
|
||||
|
||||
@pytest.mark.parametrize("producer", ["run", "optimization", "artifact"])
|
||||
def test_reserved_cash_asset_cannot_collide_with_cash_position(producer):
|
||||
from dataclasses import replace
|
||||
|
||||
operation = {
|
||||
"run": lambda: run_strategy_research("BuyAndHold", bars([10, 11]), asset="CASH"),
|
||||
"optimization": lambda: optimize_strategy_research(
|
||||
"BuyAndHold",
|
||||
bars([10, 11]),
|
||||
asset="CASH",
|
||||
param_grid={"buy_pct": [0.5]},
|
||||
objective="total_return",
|
||||
),
|
||||
"artifact": lambda: build(replace(run(params={"buy_pct": 0}), asset="CASH")),
|
||||
}[producer]
|
||||
with pytest.raises(ValueError, match="asset"):
|
||||
operation()
|
||||
|
||||
|
||||
def test_full_100_trial_tied_grid_keeps_complete_stable_ranking():
|
||||
grid = {"k1": [index / 10 for index in range(10)], "k2": [index / 10 for index in range(10)]}
|
||||
result = optimize_strategy_research(
|
||||
"DualThrust", bars([10] * 10), asset="SYNTHETIC", param_grid=grid, objective="total_return"
|
||||
)
|
||||
ranking = report(build(result))["optimization"]
|
||||
assert ranking["trial_count"] == 100
|
||||
assert len(ranking["trials"]) == 100
|
||||
assert [trial["parameters"] for trial in ranking["trials"]] == [
|
||||
trial.parameters for trial in result.trials
|
||||
]
|
||||
assert all(trial["score"] == 0 for trial in ranking["trials"])
|
||||
@@ -1,179 +0,0 @@
|
||||
"""Bounded optimizer runs actual core strategies and rejects ambiguous ranking."""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine import strategy_optimizer as optimizer
|
||||
from quant_engine.strategy_contracts import STRATEGIES, strategy_parameters
|
||||
from quant_engine.strategy_research import BenchmarkInput, run_strategy_research
|
||||
from quant_engine import strategy_contracts
|
||||
|
||||
|
||||
def feed(values=None):
|
||||
values = np.asarray(
|
||||
values if values is not None else 10 + 2 * np.sin(np.arange(80) / 2), dtype=float
|
||||
)
|
||||
return pd.DataFrame(
|
||||
dict.fromkeys(("open", "high", "low", "close"), values),
|
||||
index=pd.date_range("2026-01-01", periods=len(values), freq="B"),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", STRATEGIES)
|
||||
def test_each_default_strategy_uses_the_real_ledger(name):
|
||||
defaults = strategy_parameters(name)
|
||||
key = next(iter(defaults))
|
||||
result = optimizer.optimize_strategy_research(
|
||||
name,
|
||||
feed(),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={key: [defaults[key]]},
|
||||
objective="total_return",
|
||||
initial_cash=1000,
|
||||
commission=0.01,
|
||||
stamp_duty=0.002,
|
||||
)
|
||||
direct = run_strategy_research(
|
||||
name, feed(), asset="SYNTHETIC", initial_cash=1000, commission=0.01, stamp_duty=0.002
|
||||
)
|
||||
assert len(result.trials) == 1
|
||||
assert result.trials[0].result.ledger.positions == direct.ledger.positions
|
||||
assert result.trials[0].result.ledger.daily_executions == direct.ledger.daily_executions
|
||||
assert result.trials[0].score == direct.metrics["total_return"]
|
||||
assert result.decision_eligible is False
|
||||
|
||||
|
||||
def test_real_negative_zero_scores_and_explicit_costs_sort_without_defaults():
|
||||
result = optimizer.optimize_strategy_research(
|
||||
"BuyAndHold",
|
||||
feed([10, 10, 9, 8]),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={"buy_pct": [0.5, 0, 0.25]},
|
||||
objective="total_return",
|
||||
initial_cash=1000,
|
||||
commission=0.01,
|
||||
stamp_duty=0.002,
|
||||
)
|
||||
assert [trial.parameters["buy_pct"] for trial in result.trials] == [0, 0.25, 0.5]
|
||||
assert [trial.score for trial in result.trials] == pytest.approx([0, -0.0525, -0.105])
|
||||
assert [trial.result.ledger.total_costs for trial in result.trials] == pytest.approx(
|
||||
[0, 2.5, 5]
|
||||
)
|
||||
|
||||
|
||||
def test_stable_ties_retain_canonical_axis_and_candidate_order():
|
||||
result = optimizer.optimize_strategy_research(
|
||||
"MACross",
|
||||
feed([10] * 40),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={"atr_period": [2, 0], "fast": [5, 4]},
|
||||
objective="total_return",
|
||||
commission=0,
|
||||
stamp_duty=0,
|
||||
)
|
||||
assert [
|
||||
(trial.parameters["fast"], trial.parameters["atr_period"]) for trial in result.trials
|
||||
] == [(5, 2), (5, 0), (4, 2), (4, 0)]
|
||||
|
||||
|
||||
def test_hundred_combinations_are_unique_actual_results():
|
||||
result = optimizer.optimize_strategy_research(
|
||||
"MACross",
|
||||
feed(),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={"fast": list(range(1, 11)), "slow": list(range(11, 21))},
|
||||
objective="total_return",
|
||||
commission=0,
|
||||
stamp_duty=0,
|
||||
)
|
||||
assert len(result.trials) == 100
|
||||
assert len({tuple(trial.parameters.items()) for trial in result.trials}) == 100
|
||||
assert all(len(trial.result.ledger.positions) == 80 for trial in result.trials)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"grid,objective",
|
||||
[
|
||||
({"fast": []}, "total_return"),
|
||||
({"fast": [5, 40]}, "total_return"),
|
||||
({"fast": [True]}, "total_return"),
|
||||
({"fast": [5]}, "unknown"),
|
||||
({"fast": [5, 5]}, "total_return"),
|
||||
(
|
||||
{"fast": list(range(1, 11)), "slow": list(range(11, 21)), "atr_period": [0, 1]},
|
||||
"total_return",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_invalid_or_partly_invalid_grid_never_starts_a_strategy(monkeypatch, grid, objective):
|
||||
calls = []
|
||||
monkeypatch.setattr(optimizer, "run_strategy_research", lambda *a, **kw: calls.append(1))
|
||||
with pytest.raises(
|
||||
ValueError, match=r"grid|Grid|candidate|number|less than|objective|Duplicate"
|
||||
):
|
||||
optimizer.optimize_strategy_research(
|
||||
"MACross", feed(), asset="SYNTHETIC", param_grid=grid, objective=objective
|
||||
)
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_history_failure_for_one_candidate_prevents_all_runs(monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(optimizer, "run_strategy_research", lambda *a, **kw: calls.append(1))
|
||||
with pytest.raises(ValueError, match=r"history"):
|
||||
optimizer.optimize_strategy_research(
|
||||
"MACross",
|
||||
feed([10] * 40),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={"slow": [30, 100]},
|
||||
objective="total_return",
|
||||
)
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_requested_benchmark_error_prevents_every_run(monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(optimizer, "run_strategy_research", lambda *a, **kw: calls.append(1))
|
||||
with pytest.raises(ValueError, match=r"benchmark source"):
|
||||
optimizer.optimize_strategy_research(
|
||||
"BuyAndHold",
|
||||
feed(),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={"buy_pct": [0.5, 1]},
|
||||
objective="total_return",
|
||||
benchmark=BenchmarkInput("source_error"),
|
||||
)
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_undefined_sharpe_fails_the_ranking_instead_of_winning_as_zero():
|
||||
with pytest.raises(ValueError, match=r"objective.*unavailable"):
|
||||
optimizer.optimize_strategy_research(
|
||||
"BuyAndHold",
|
||||
feed([10] * 4),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={"buy_pct": [0, 0.5]},
|
||||
objective="sharpe_ratio",
|
||||
commission=0,
|
||||
stamp_duty=0,
|
||||
)
|
||||
|
||||
|
||||
def test_large_grid_fails_before_creating_cartesian_product(monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(strategy_contracts, "product", lambda *a, **kw: calls.append("product"))
|
||||
monkeypatch.setattr(optimizer, "run_strategy_research", lambda *a, **kw: calls.append("run"))
|
||||
with pytest.raises(ValueError, match=r"limit"):
|
||||
optimizer.optimize_strategy_research(
|
||||
"MACross",
|
||||
feed(),
|
||||
asset="SYNTHETIC",
|
||||
param_grid={
|
||||
"fast": list(range(1, 11)),
|
||||
"slow": list(range(11, 21)),
|
||||
"atr_period": list(range(10)),
|
||||
"atr_mult": list(range(1, 11)),
|
||||
},
|
||||
)
|
||||
assert calls == []
|
||||
@@ -1,436 +0,0 @@
|
||||
"""Caller-supplied OHLC, causal signals and the real shared daily ledger."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine import strategy_research as research
|
||||
|
||||
|
||||
def bars(closes, opens=None):
|
||||
close = np.asarray(closes, dtype=float)
|
||||
opening = np.asarray(opens if opens is not None else closes, dtype=float)
|
||||
return pd.DataFrame(
|
||||
{
|
||||
"open": opening,
|
||||
"high": np.maximum(close, opening),
|
||||
"low": np.minimum(close, opening),
|
||||
"close": close,
|
||||
},
|
||||
index=pd.date_range("2026-01-01", periods=len(close), freq="B"),
|
||||
)
|
||||
|
||||
|
||||
def run(name, feed, **kwargs):
|
||||
return research.run_strategy_research(
|
||||
name, feed, asset="SYNTHETIC", initial_cash=1000, commission=0, stamp_duty=0, **kwargs
|
||||
)
|
||||
|
||||
|
||||
def test_buy_hold_signal_close_next_open_and_last_close_value():
|
||||
result = research.run_strategy_research(
|
||||
"BuyAndHold",
|
||||
bars([10, 11, 12, 13], [10, 10, 11, 12]),
|
||||
asset="SYNTHETIC",
|
||||
initial_cash=1000,
|
||||
commission=0.01,
|
||||
stamp_duty=0,
|
||||
params={"buy_pct": 1},
|
||||
)
|
||||
assert result.ledger.nav_series.tolist() == pytest.approx(
|
||||
[1000, 1100 / 1.01, 1200 / 1.01, 1300 / 1.01]
|
||||
)
|
||||
assert result.ledger.total_rebalances == 1
|
||||
trade = result.ledger.trades_frame.iloc[0]
|
||||
assert trade["trade_date"] == "2026-01-02"
|
||||
assert trade["price"] == 10
|
||||
assert trade["fee"] == pytest.approx(1000 - 1000 / 1.01)
|
||||
assert result.signals[0].decision_date == "2026-01-01"
|
||||
assert result.signals[0].execution_date == "2026-01-02"
|
||||
assert result.signals[0].status == "partial_fill"
|
||||
assert result.metrics["total_return"] == pytest.approx(1300 / 1.01 / 1000 - 1)
|
||||
assert result.pairing.win_rate is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name,params,closes",
|
||||
[
|
||||
("SmaCross", {"fast": 1, "slow": 2}, [10, 8, 12, 6, 14, 5, 12]),
|
||||
("MACross", {"fast": 1, "slow": 2}, [10, 8, 12, 6, 14, 5, 12]),
|
||||
("RSI", {"period": 2}, [10, 8, 6, 10, 14, 8, 6, 10]),
|
||||
("BollingerBreakout", {"period": 2, "std_mult": 0.5}, [10, 10, 12, 8, 12, 8]),
|
||||
("DualThrust", {"period": 2, "k1": 0.5, "k2": 0.5}, [10, 10, 12, 8, 12, 8]),
|
||||
("TurtleBreakout", {"entry_period": 2, "exit_period": 2}, [10, 10, 12, 8, 12, 8]),
|
||||
],
|
||||
)
|
||||
def test_each_strategy_has_real_entry_exit_and_sparse_execution(name, params, closes):
|
||||
opening = [closes[0], *closes[:-1]]
|
||||
result = run(name, bars(closes, opening), params=params)
|
||||
trades = result.ledger.trades_frame
|
||||
assert trades["side"].iloc[:2].tolist() == ["buy", "sell"]
|
||||
assert result.signals[0].decision_date == "2026-01-05"
|
||||
assert trades["trade_date"].iloc[0] == "2026-01-06"
|
||||
for signal in result.signals:
|
||||
if signal.execution_date:
|
||||
assert signal.execution_date > signal.decision_date
|
||||
assert len(result.ledger.positions) == len(closes)
|
||||
assert all(position.cash >= -1e-9 for position in result.ledger.positions)
|
||||
assert result.pairing.closed_lots
|
||||
|
||||
|
||||
def test_final_day_signal_is_recorded_without_same_close_execution():
|
||||
result = run("SmaCross", bars([10, 8, 12]), params={"fast": 1, "slow": 2})
|
||||
assert result.ledger.trades_frame.empty
|
||||
assert len(result.signals) == 1
|
||||
assert result.signals[0].status == "no_next_session"
|
||||
assert result.signals[0].execution_date is None
|
||||
|
||||
|
||||
def test_atr_stop_uses_prior_peak_and_prior_atr_and_can_trigger():
|
||||
feed = bars([10, 8, 8, 10, 20, 19, 18], [10, 10, 8, 8, 10, 20, 19])
|
||||
params = {"fast": 2, "slow": 3, "atr_period": 1, "atr_mult": 0.1}
|
||||
result = run("MACross", feed, params=params)
|
||||
stop = next(signal for signal in result.signals if signal.reason == "atr_stop")
|
||||
assert stop.decision_date == "2026-01-08"
|
||||
assert stop.execution_date == "2026-01-09"
|
||||
assert result.ledger.trades_frame.iloc[-1]["side"] == "sell"
|
||||
disabled = run("MACross", feed, params=params | {"atr_period": 0})
|
||||
assert all(signal.reason != "atr_stop" for signal in disabled.signals)
|
||||
|
||||
|
||||
def test_flat_rsi_is_neutral_and_does_not_create_artificial_trades():
|
||||
result = run("RSI", bars([10] * 8), params={"period": 2, "oversold": 40, "overbought": 60})
|
||||
assert result.ledger.trades_frame.empty
|
||||
assert result.metrics["sharpe"] is None
|
||||
assert result.metric_unavailable["sharpe"] == "zero_volatility"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name",
|
||||
[
|
||||
"BuyAndHold",
|
||||
"SmaCross",
|
||||
"MACross",
|
||||
"RSI",
|
||||
"BollingerBreakout",
|
||||
"DualThrust",
|
||||
"TurtleBreakout",
|
||||
],
|
||||
)
|
||||
def test_default_parameters_and_future_perturbation_preserve_observed_prefix(name):
|
||||
values = 10 + np.sin(np.arange(80) / 2) * 2
|
||||
original = bars(values, np.r_[values[0], values[:-1]])
|
||||
changed = original.copy()
|
||||
changed.iloc[55:] *= 7
|
||||
first = run(name, original)
|
||||
second = run(name, changed)
|
||||
assert first.ledger.positions[:55] == second.ledger.positions[:55]
|
||||
assert first.ledger.daily_executions[:55] == second.ledger.daily_executions[:55]
|
||||
observed = original.index[53].strftime("%Y-%m-%d")
|
||||
assert [s for s in first.signals if s.decision_date <= observed] == [
|
||||
s for s in second.signals if s.decision_date <= observed
|
||||
]
|
||||
assert first.parameters == research.strategy_parameters(name)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"mutation",
|
||||
[
|
||||
"missing_open",
|
||||
"missing_high",
|
||||
"missing_low",
|
||||
"null",
|
||||
"boolean",
|
||||
"string",
|
||||
"infinite",
|
||||
"bad_bounds",
|
||||
"duplicate",
|
||||
"unsorted",
|
||||
],
|
||||
)
|
||||
def test_invalid_ohlc_fails_before_ledger(monkeypatch, mutation):
|
||||
feed = bars([10] * 40)
|
||||
if mutation.startswith("missing_"):
|
||||
feed = feed.drop(columns=mutation[8:])
|
||||
elif mutation == "duplicate":
|
||||
feed.index = [feed.index[0]] * len(feed)
|
||||
elif mutation == "unsorted":
|
||||
feed = feed.iloc[::-1]
|
||||
elif mutation == "bad_bounds":
|
||||
feed.iloc[0, feed.columns.get_loc("high")] = 9
|
||||
else:
|
||||
feed = feed.astype(object)
|
||||
feed.iloc[0, 0] = {"null": None, "boolean": True, "string": "10", "infinite": float("inf")}[
|
||||
mutation
|
||||
]
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
research, "simulate_daily_ledger_with_audit", lambda *a, **kw: calls.append(1)
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"OHLC|prices|DatetimeIndex"):
|
||||
run("DualThrust", feed)
|
||||
assert calls == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name,params",
|
||||
[
|
||||
("SmaCross", {"fast": True}),
|
||||
("SmaCross", {"fast": 20}),
|
||||
("RSI", {"oversold": 70}),
|
||||
("MACross", {"atr_period": -1}),
|
||||
("TurtleBreakout", {"entry_period": "20"}),
|
||||
("BuyAndHold", {"buy_pct": float("nan")}),
|
||||
("DualThrust", {"unknown": 1}),
|
||||
],
|
||||
)
|
||||
def test_invalid_parameters_before_ledger(monkeypatch, name, params):
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
research, "simulate_daily_ledger_with_audit", lambda *a, **kw: calls.append(1)
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"parameter|finite|less than|bounds|integer"):
|
||||
run(name, bars([10] * 40), params=params)
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_insufficient_history_is_failure_before_ledger(monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
research, "simulate_daily_ledger_with_audit", lambda *a, **kw: calls.append(1)
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"history"):
|
||||
run("MACross", bars([10] * 30))
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_zero_allocation_is_no_change_not_a_filled_position():
|
||||
result = run("BuyAndHold", bars([10, 11, 12]), params={"buy_pct": 0})
|
||||
assert result.ledger.trades_frame.empty
|
||||
assert result.signals[0].status == "no_change"
|
||||
assert result.ledger.nav_series.tolist() == [1000, 1000, 1000]
|
||||
|
||||
|
||||
def test_exit_below_minimum_is_not_filled_and_state_keeps_actual_holdings():
|
||||
result = run(
|
||||
"SmaCross",
|
||||
bars([10, 8, 12, 6, 14, 5, 12], [10, 10, 8, 12, 6, 14, 5]),
|
||||
params={"fast": 1, "slow": 2},
|
||||
min_trade_amount=900,
|
||||
)
|
||||
exit_signal = result.signals[1]
|
||||
assert exit_signal.target_weight == 0
|
||||
assert exit_signal.status == "not_filled"
|
||||
assert result.ledger.positions[4].holdings == {"SYNTHETIC": pytest.approx(1000 / 12)}
|
||||
assert len(result.ledger.trades_frame) == 1
|
||||
|
||||
|
||||
def test_total_loss_from_explicit_full_sell_fee_is_reported_as_zero_nav():
|
||||
result = research.run_strategy_research(
|
||||
"SmaCross",
|
||||
bars([10, 8, 12, 6, 14, 5]),
|
||||
asset="SYNTHETIC",
|
||||
initial_cash=1000,
|
||||
commission=0,
|
||||
stamp_duty=1,
|
||||
params={"fast": 1, "slow": 2},
|
||||
)
|
||||
assert result.ledger.nav_series.iloc[-1] == 0
|
||||
assert result.metrics["total_return"] == -1
|
||||
assert result.pairing.win_rate == 0
|
||||
|
||||
|
||||
def test_turtle_entry_does_not_wait_for_longer_exit_lookback():
|
||||
result = run(
|
||||
"TurtleBreakout",
|
||||
bars([10, 10, 12, 13, 14, 15, 16]),
|
||||
params={"entry_period": 2, "exit_period": 5},
|
||||
)
|
||||
assert result.signals[0].decision_date == "2026-01-05"
|
||||
|
||||
|
||||
def test_ma_entry_does_not_wait_for_optional_atr_calibration():
|
||||
result = run(
|
||||
"MACross", bars([10, 8, 12, 6, 14, 5, 12]), params={"fast": 1, "slow": 2, "atr_period": 5}
|
||||
)
|
||||
assert result.signals[0].decision_date == "2026-01-05"
|
||||
|
||||
|
||||
def test_benchmark_exact_calendar_and_distinct_empty_unrequested_states():
|
||||
feed = bars([10, 10, 10])
|
||||
benchmark = pd.Series([100.0, 90.0, 99.0], index=feed.index)
|
||||
result = run("BuyAndHold", feed, benchmark=research.BenchmarkInput("present", benchmark))
|
||||
assert result.benchmark_status == "present"
|
||||
assert result.benchmark_nav.tolist() == pytest.approx([1, 0.9, 0.99])
|
||||
assert result.benchmark_metrics["total_return"] == pytest.approx(-0.01)
|
||||
empty = run(
|
||||
"BuyAndHold", feed, benchmark=research.BenchmarkInput("empty", pd.Series(dtype=float))
|
||||
)
|
||||
none = run("BuyAndHold", feed)
|
||||
assert empty.benchmark_status == "empty"
|
||||
assert empty.benchmark_nav is None
|
||||
assert none.benchmark_status == "not_requested"
|
||||
assert none.benchmark_nav is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", ["missing_day", "duplicate", "source_error"])
|
||||
def test_bad_benchmark_fails_before_any_strategy_execution(monkeypatch, status):
|
||||
feed = bars([10, 10, 10])
|
||||
series = pd.Series([100.0, 90.0, 99.0], index=feed.index)
|
||||
if status == "missing_day":
|
||||
series = series.iloc[:2]
|
||||
elif status == "duplicate":
|
||||
series.index = [feed.index[0]] * 3
|
||||
observation = research.BenchmarkInput(
|
||||
"source_error" if status == "source_error" else "present",
|
||||
series if status != "source_error" else None,
|
||||
)
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
research, "simulate_daily_ledger_with_audit", lambda *a, **kw: calls.append(1)
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"Benchmark|benchmark|DatetimeIndex"):
|
||||
run("BuyAndHold", feed, benchmark=observation)
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_benchmark_numeric_underflow_is_not_filled_as_zero_return(monkeypatch):
|
||||
feed = bars([10] * 4)
|
||||
benchmark = research.BenchmarkInput(
|
||||
"present", pd.Series([1e300, 1e300, 1e-300, 1e-300], index=feed.index)
|
||||
)
|
||||
calls = []
|
||||
real_ledger = research.simulate_daily_ledger_with_audit
|
||||
|
||||
def observed(*args, **kwargs):
|
||||
calls.append(1)
|
||||
return real_ledger(*args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(research, "simulate_daily_ledger_with_audit", observed)
|
||||
with pytest.raises(ValueError, match=r"benchmark.*numeric|Benchmark.*numeric"):
|
||||
run("BuyAndHold", feed, benchmark=benchmark)
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_nonfinite_portfolio_return_cannot_be_silently_dropped_from_metrics():
|
||||
with pytest.raises(ValueError, match=r"return.*finite|return.*numeric"):
|
||||
run("BuyAndHold", bars([10, 10, 1e-300, 1e300]), params={"buy_pct": 1})
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name,params,closes,expected_nav,expected_cash,prices,quantities,pnl",
|
||||
[
|
||||
(
|
||||
"SmaCross",
|
||||
{"fast": 1, "slow": 2},
|
||||
[10, 8, 12, 6, 14, 5, 12],
|
||||
[1000, 1000, 1000, 500, 500, 1250 / 7, 1250 / 7],
|
||||
[1000, 1000, 1000, 0, 500, 0, 1250 / 7],
|
||||
[12, 6, 14, 5],
|
||||
[250 / 3, 250 / 3, 250 / 7, 250 / 7],
|
||||
-5750 / 7,
|
||||
),
|
||||
(
|
||||
"MACross",
|
||||
{"fast": 1, "slow": 2},
|
||||
[10, 8, 12, 6, 14, 5, 12],
|
||||
[1000, 1000, 1000, 500, 500, 1250 / 7, 1250 / 7],
|
||||
[1000, 1000, 1000, 0, 500, 0, 1250 / 7],
|
||||
[12, 6, 14, 5],
|
||||
[250 / 3, 250 / 3, 250 / 7, 250 / 7],
|
||||
-5750 / 7,
|
||||
),
|
||||
(
|
||||
"RSI",
|
||||
{"period": 2},
|
||||
[10, 8, 6, 10, 14, 8, 6, 10],
|
||||
[1000, 1000, 1000, 5000 / 3, 7000 / 3, 7000 / 3, 7000 / 3, 35000 / 9],
|
||||
[1000, 1000, 1000, 0, 0, 7000 / 3, 7000 / 3, 0],
|
||||
[6, 14, 6],
|
||||
[500 / 3, 500 / 3, 3500 / 9],
|
||||
4000 / 3,
|
||||
),
|
||||
(
|
||||
"BollingerBreakout",
|
||||
{"period": 2, "std_mult": 0.5},
|
||||
[10, 10, 12, 8, 12, 8],
|
||||
[1000, 1000, 1000, 2000 / 3, 2000 / 3, 4000 / 9],
|
||||
[1000, 1000, 1000, 0, 2000 / 3, 0],
|
||||
[12, 8, 12],
|
||||
[250 / 3, 250 / 3, 500 / 9],
|
||||
-1000 / 3,
|
||||
),
|
||||
(
|
||||
"DualThrust",
|
||||
{"period": 2, "k1": 0.5, "k2": 0.5},
|
||||
[10, 10, 12, 8, 12, 8],
|
||||
[1000, 1000, 1000, 2000 / 3, 2000 / 3, 4000 / 9],
|
||||
[1000, 1000, 1000, 0, 2000 / 3, 0],
|
||||
[12, 8, 12],
|
||||
[250 / 3, 250 / 3, 500 / 9],
|
||||
-1000 / 3,
|
||||
),
|
||||
(
|
||||
"TurtleBreakout",
|
||||
{"entry_period": 2, "exit_period": 2},
|
||||
[10, 10, 12, 8, 12, 8],
|
||||
[1000, 1000, 1000, 2000 / 3, 2000 / 3, 2000 / 3],
|
||||
[1000, 1000, 1000, 0, 2000 / 3, 2000 / 3],
|
||||
[12, 8],
|
||||
[250 / 3, 250 / 3],
|
||||
-1000 / 3,
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_hand_calculated_strategy_cash_nav_and_every_fill(
|
||||
name, params, closes, expected_nav, expected_cash, prices, quantities, pnl
|
||||
):
|
||||
# These rational constants were calculated from the expected sparse trades,
|
||||
# independently of the signal and ledger implementation.
|
||||
result = run(name, bars(closes, [closes[0], *closes[:-1]]), params=params)
|
||||
assert result.ledger.nav_series.tolist() == pytest.approx(expected_nav)
|
||||
assert [position.cash for position in result.ledger.positions] == pytest.approx(expected_cash)
|
||||
assert result.ledger.trades_frame["price"].tolist() == prices
|
||||
assert result.ledger.trades_frame["qty"].tolist() == pytest.approx(quantities)
|
||||
assert result.pairing.realized_net_pnl == pytest.approx(pnl)
|
||||
assert result.metrics["total_return"] == pytest.approx(expected_nav[-1] / 1000 - 1)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"field,value",
|
||||
[
|
||||
("initial_cash", True),
|
||||
("initial_cash", 0),
|
||||
("commission", "0.01"),
|
||||
("commission", -1),
|
||||
("stamp_duty", float("nan")),
|
||||
("stamp_duty", 2),
|
||||
],
|
||||
)
|
||||
def test_money_contract_fails_before_ledger(monkeypatch, field, value):
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
research, "simulate_daily_ledger_with_audit", lambda *a, **kw: calls.append(1)
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"number|bounds|fee|Fee"):
|
||||
research.run_strategy_research(
|
||||
"BuyAndHold", bars([10, 10]), asset="SYNTHETIC", **{field: value}
|
||||
)
|
||||
assert calls == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("scale", [1e-200, 1e200])
|
||||
def test_bollinger_signal_is_invariant_to_representable_price_scaling(scale):
|
||||
original = bars([10, 10, 12, 8, 12, 8])
|
||||
normal = run("BollingerBreakout", original, params={"period": 2, "std_mult": 2})
|
||||
scaled = run("BollingerBreakout", original * scale, params={"period": 2, "std_mult": 2})
|
||||
assert scaled.signals == normal.signals
|
||||
assert scaled.ledger.nav_series.tolist() == normal.ledger.nav_series.tolist()
|
||||
|
||||
|
||||
def test_no_downside_sortino_is_explicitly_unavailable():
|
||||
result = run("BuyAndHold", bars([10, 10, 11, 12]), params={"buy_pct": 0.5})
|
||||
assert result.metrics["sortino"] is None
|
||||
assert result.metric_unavailable["sortino"] == "no_downside_deviation"
|
||||
@@ -1,127 +0,0 @@
|
||||
"""Cost-aware FIFO pairing uses actual ledger cash flows, including both fees."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.execution import ExecutionConfig, simulate_daily_ledger_with_audit
|
||||
from quant_engine.trade_pairing import pair_ledger_trades
|
||||
|
||||
|
||||
def test_loss_after_both_fees_is_not_a_win():
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[("d1", {"A": 0.1}), ("d2", {})],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 9})],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 9})],
|
||||
1000,
|
||||
ExecutionConfig(commission_bps=100, stamp_tax_bps=200, slippage_bps=0, min_trade_amount=0),
|
||||
)
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
assert len(pairing.closed_lots) == 1
|
||||
assert pairing.closed_lots[0].quantity == 10
|
||||
assert pairing.closed_lots[0].cost == 101
|
||||
assert pairing.closed_lots[0].net_proceeds == pytest.approx(87.3)
|
||||
assert pairing.realized_net_pnl == pytest.approx(-13.7)
|
||||
assert pairing.win_rate == 0
|
||||
assert pairing.open_lots == ()
|
||||
assert ledger.nav_series.tolist() == pytest.approx([999, 986.3])
|
||||
|
||||
|
||||
def test_last_day_multiple_fills_each_pay_once_and_match_nav():
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[("d2", {"A": 0.25, "B": 0.5})],
|
||||
[("d2", {"A": 10, "B": 20})],
|
||||
[("d1", {"A": 10, "B": 20}), ("d2", {"A": 10, "B": 20})],
|
||||
1000,
|
||||
ExecutionConfig(commission_bps=100, stamp_tax_bps=200, slippage_bps=0, min_trade_amount=0),
|
||||
)
|
||||
assert ledger.nav_series.tolist() == pytest.approx([1000, 992.5])
|
||||
assert ledger.trades_frame["fee"].tolist() == [2.5, 5.0]
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
assert len(pairing.open_lots) == 2
|
||||
assert pairing.closed_lots == ()
|
||||
assert pairing.realized_net_pnl == 0
|
||||
assert pairing.win_rate is None
|
||||
|
||||
|
||||
def test_partial_fifo_sales_allocate_entry_cost_and_keep_unclosed_lot_out_of_win_rate():
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[("d1", {"A": 0.2}), ("d2", {"A": 0.1}), ("d3", {})],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 10}), ("d3", {"A": 10})],
|
||||
[("d1", {"A": 10}), ("d2", {"A": 10}), ("d3", {"A": 10})],
|
||||
1000,
|
||||
ExecutionConfig(commission_bps=100, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0),
|
||||
)
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
assert len(pairing.matches) == 2
|
||||
assert len(pairing.closed_lots) == 1
|
||||
assert pairing.closed_lots[0].quantity == 20
|
||||
assert pairing.closed_lots[0].cost == 202
|
||||
assert pairing.closed_lots[0].net_proceeds == pytest.approx(198)
|
||||
assert pairing.realized_net_pnl == pytest.approx(-4)
|
||||
assert pairing.win_rate == 0
|
||||
|
||||
|
||||
def test_same_day_sell_and_buy_are_different_lots_with_no_duplicate_fees():
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[("d1", {"A": 0.5}), ("d2", {"B": 0.5})],
|
||||
[("d1", {"A": 10, "B": 10}), ("d2", {"A": 10, "B": 10})],
|
||||
[("d1", {"A": 10, "B": 10}), ("d2", {"A": 10, "B": 10})],
|
||||
1000,
|
||||
ExecutionConfig(commission_bps=100, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0),
|
||||
)
|
||||
assert ledger.nav_series.tolist() == pytest.approx([995, 985.025])
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
assert pairing.closed_lots[0].asset == "A"
|
||||
assert pairing.closed_lots[0].net_pnl == pytest.approx(-10)
|
||||
assert pairing.open_lots[0].asset == "B"
|
||||
|
||||
|
||||
def test_small_fractional_holding_is_not_destroyed_after_partial_sale():
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[("d1", {"A": 0.5}), ("d2", {"A": 0.25})],
|
||||
[("d1", {"A": 1e9}), ("d2", {"A": 1e9})],
|
||||
[("d1", {"A": 1e9}), ("d2", {"A": 1e9})],
|
||||
1000,
|
||||
ExecutionConfig(commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0),
|
||||
)
|
||||
assert ledger.nav_series.tolist() == [1000, 1000]
|
||||
assert ledger.positions[-1].holdings["A"] == 2.5e-7
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
assert pairing.open_lots[0].quantity == 2.5e-7
|
||||
assert pairing.open_lots[0].remaining_cost == 250
|
||||
|
||||
|
||||
def test_real_tiny_remaining_lot_is_not_treated_as_a_completed_trade():
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[("d1", {"A": 1}), ("d2", {"A": 1e-13})],
|
||||
[("d1", {"A": 1}), ("d2", {"A": 1})],
|
||||
[("d1", {"A": 1}), ("d2", {"A": 1})],
|
||||
1e12,
|
||||
ExecutionConfig(commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0),
|
||||
)
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
assert pairing.closed_lots == ()
|
||||
assert pairing.win_rate is None
|
||||
assert pairing.open_lots[0].quantity == ledger.positions[-1].holdings["A"]
|
||||
assert pairing.open_lots[0].remaining_cost == pytest.approx(ledger.positions[-1].holdings["A"])
|
||||
|
||||
|
||||
@pytest.mark.parametrize("price", [3, 11, 13])
|
||||
def test_complete_exit_closes_all_accumulated_lots_without_rounding_residue(price):
|
||||
prices = [(date, {"A": price}) for date in ("d1", "d2", "d3", "d4")]
|
||||
ledger = simulate_daily_ledger_with_audit(
|
||||
[("d1", {"A": 0.1}), ("d2", {"A": 0.2}), ("d3", {"A": 0.3}), ("d4", {})],
|
||||
prices,
|
||||
prices,
|
||||
1000,
|
||||
ExecutionConfig(commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0),
|
||||
)
|
||||
pairing = pair_ledger_trades(ledger)
|
||||
assert ledger.positions[-1].holdings == {}
|
||||
assert pairing.open_lots == ()
|
||||
assert len(pairing.closed_lots) == 3
|
||||
assert sum(lot.cost for lot in pairing.closed_lots) == pytest.approx(300)
|
||||
assert sum(match.net_proceeds for match in pairing.matches) == pytest.approx(300)
|
||||
assert pairing.realized_net_pnl == pytest.approx(0)
|
||||
assert pairing.win_rate == 0
|
||||
Reference in New Issue
Block a user