Compare commits

..
Author SHA1 Message Date
ao gong 2b0e8ee637 feat: add explicit retrospective v2 computation contracts
CI / lite (pull_request) Successful in 9s
2026-09-08 19:40:19 +08:00
ageorge156 68dd68392a test(performance): enforce semantic artifact authority boundary (#19)
CI / lite (push) Successful in 9s
2026-09-01 23:12:10 +08:00
ageorge156 a724e1e57a feat: add portfolio risk computation contracts (#18)
CI / lite (push) Successful in 9s
2026-09-01 14:08:23 +08:00
ageorge156 78d65b4db0 fix: close final backtest contract boundaries (#17)
CI / lite (push) Successful in 8s
2026-09-01 12:40:04 +08:00
ageorge156 598c2b92a2 fix: type-check factor set definition ids (#16)
CI / lite (push) Successful in 8s
2026-09-01 11:10:18 +08:00
ageorge156 62ed09842d fix: enforce canonical factor contract json (#15)
CI / lite (push) Successful in 7s
2026-09-01 08:36:41 +08:00
ageorge156 e782e223f7 feat(governance): add risk-gated paper research slice (#14)
CI / lite (push) Successful in 8s
2026-08-30 16:14:34 +08:00
ageorge156 015c1a3602 Merge remote-tracking branch 'origin/main' into codex/research-alpha158-phase6-formula-con (#13)
CI / lite (push) Successful in 7s
2026-08-28 18:33:31 +08:00
ageorge156 2bc8aea435 Merge remote-tracking branch 'origin/main' into codex/research-alpha158-phase5-formula-con (#12)
CI / lite (push) Successful in 7s
2026-08-28 18:32:18 +08:00
ageorge156 03e38d5123 Merge remote-tracking branch 'origin/main' into codex/research-alpha158-phase4-formula-con (#11)
CI / lite (push) Successful in 7s
2026-08-28 18:30:57 +08:00
ageorge156 90a43adda2 Merge remote-tracking branch 'origin/main' into codex/research-alpha158-phase3-formula-con (#10)
CI / lite (push) Successful in 8s
2026-08-28 18:29:47 +08:00
ageorge156 e72fe0a8d1 Merge remote-tracking branch 'origin/main' into codex/research-alpha158-phase2-20260827 (#9)
CI / lite (push) Successful in 7s
2026-08-28 18:28:22 +08:00
ageorge156 fd3014c286 fix: bound phase1 operator windows (#8)
CI / lite (push) Successful in 8s
2026-08-28 18:26:14 +08:00
ageorge156 38a984b245 feat(quant): consolidate research artifact contract (#7)
CI / lite (push) Successful in 10s
2026-08-26 20:55:09 +08:00
ageorge156 8a30bf5ebc fix(ci): verify the unified quant runtime (#6)
CI / lite (push) Successful in 10s
2026-08-24 20:56:36 +08:00
ageorge156 60027c8d3e fix(ci): publish quant engine lite verification (#1)
CI / lite (push) Successful in 3s
2026-08-20 21:36:32 +08:00
63 changed files with 28394 additions and 231 deletions
+15 -2
View File
@@ -16,6 +16,8 @@ permissions:
jobs:
lite:
runs-on: ubuntu-latest
env:
UV_PYTHON_DOWNLOADS: never
steps:
- uses: actions/checkout@524e936cd9e579adf00e308bfdf971aebc7de09e
with:
@@ -26,7 +28,18 @@ jobs:
if git ls-files .DS_Store | grep -q .; then echo "跟踪 .DS_Store"; exit 1; fi
if git grep -n -I -E 'sk-[A-Za-z0-9]{20,}|AKIA[0-9A-Z]{16}|ghp_[A-Za-z0-9]{36}|xox[baprs]-[A-Za-z0-9-]{10,}' HEAD | grep -q .; then echo "检出疑似凭证"; exit 1; fi
echo "Gitea 合规校验通过"
- name: 验证并同步共享运行时
run: |
test "$(python3 --version)" = "Python 3.13.15"
test "$(uv --version | cut -d' ' -f1-2)" = "uv 0.12.3"
uv sync --locked --extra dev
uv run --locked --no-sync python -c 'import sys; assert sys.version_info[:2] == (3, 13)'
- name: 架构模块契约测试
run: python3 tests/governance/test_module_spec.py
run: |
uv run --locked --no-sync python tests/governance/test_module_spec.py
uv run --locked --no-sync python tests/governance/test_ci_contract.py
- name: Syntax check
run: git ls-files -z '*.py' | xargs -0 python3 -m py_compile
run: git ls-files -z '*.py' | xargs -0 uv run --locked --no-sync python -m py_compile
+1
View File
@@ -0,0 +1 @@
3.13
+29 -2
View File
@@ -1,7 +1,7 @@
{
"schema_version": 1,
"module_id": "quant_engine",
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 1, "effective_from": "2026-08-20T00:00:00+08:00"},
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 6, "effective_from": "2026-09-08T19:33:40+08:00"},
"repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"},
"bounded_context": {
"domain": "quantitative-research-engine",
@@ -11,6 +11,7 @@
"Submitting live orders, routing trades, managing brokerage accounts, or claiming transaction execution",
"Owning market-data source facts, research-result publication, or platform presentation state",
"Loading provider credentials, brokerage credentials, or production secrets",
"Granting portfolio approval, maker-checker decisions, publication eligibility, paper execution, or live execution authority",
"Changing financial model semantics through module metadata"
]
},
@@ -18,13 +19,39 @@
{"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"},
{"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"},
{"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"},
{"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact and performance-methodology evidence without recomputation, persistence, or decision authority.", "status": "operational"},
{"id": "portfolio-risk-computation-contracts", "summary": "Verify deterministic portfolio-computation receipts and expose S3-bound portfolio decisions and risk assessments without adding algorithms or execution authority.", "status": "operational"},
{"id": "retrospective-computation-contracts", "summary": "Decode observation-aware v2 data and expose explicit retrospective factor, backtest, portfolio and risk contracts with two clocks, no historical-availability claim and no execution authority.", "status": "operational"},
{"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"}
],
"data": {"owns": [
{"asset_id": "quantitative-model-implementations", "kind": "model", "classification": "internal"},
{"asset_id": "simulation-and-metric-results", "kind": "artifact", "classification": "confidential"}
]},
"contracts": {"provides": [], "consumes": []},
"contracts": {
"provides": [
{"contract_id": "researchhub.factor-definition", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
{"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
{"contract_id": "researchhub.backtest-run-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/governed_pipeline.py"},
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
{"contract_id": "researchhub.performance-evidence", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
{"contract_id": "researchhub.portfolio-decision", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
{"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
{"contract_id": "researchhub.factor-set-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_factor_contracts.py"},
{"contract_id": "researchhub.backtest-run-ref", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_backtest_contracts.py"},
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
{"contract_id": "researchhub.performance-evidence", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_artifact_contracts.py"},
{"contract_id": "researchhub.portfolio-target", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
{"contract_id": "researchhub.portfolio-decision", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"},
{"contract_id": "researchhub.risk-assessment", "version": "2.0.0", "authority": "quant_engine", "path": "src/quant_engine/retrospective_portfolio_risk_contracts.py"}
],
"consumes": [
{"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"},
{"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"},
{"contract_id": "researchhub.dataset-snapshot", "version": "2.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_retrospective_envelope_and_materialized_chunks"},
{"contract_id": "researchhub.data-foundation", "version": "2.0.0", "authority": "researchhub.data", "admission": "observation_bound_selected_views_and_materialized_bytes"}
]
},
"dependencies": [],
"agent_context": {
"default_entrypoints": [
+345 -7
View File
@@ -10,7 +10,7 @@
| 仓库 | 角色 |
|---|---|
| `quant_engine` | **纯回测核心**(alpha + execution + indicators + data_adapter + backtest + metrics) |
| `quant_engine` | **纯研究核心**(alpha + execution + ledger + attribution + risk + metrics) |
| `research_results` | 业务集成(47 个 proj 调度 + 注册 + 平台对接) |
| `tushare2db_pro_aoge` | 数据层(行情 ELT) |
| `research_platform` | 展示层(FastAPI + Next.js) |
@@ -19,14 +19,22 @@
## 模块
- `alpha_factors` — 158 alpha 公式 + 24 基础算子(移植自 qlib alpha158)
- `execution` — 执行仿真(成本/滑点/T+1/涨跌停/部分成交/价差)+ 多日 NAV + PnL 拆解(借鉴 hikyuu 部件化思想)
- `factor_contracts` — `FactorDefinition` / `FactorSetRef` v1 纯计算合同、严格 PIT/availability 输入准入与显式 legacy 投影
- `execution` — A 股长仓执行仿真(成本/滑点/现金约束)+ 稀疏调仓/完整交易日 Ledger + 可投影成交与 NAV 审计;T+1、涨跌停、成交量与价差提供独立约束函数
- `indicators` — 50+ 技术指标(MACD / KDJ / 布林 / ATR / ADX / 等)
- `data_adapter` — 桥接 qtdb_pro 长表与新模块(rename / long-wide / 复权 / vwap 代理)
- `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark)
- `metrics` — 绩效(年化收益 / 波动率 / Sharpe / 最大回撤 / Calmar)
- `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表
- `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排)
- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef`
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest`
- `portfolio_risk_contracts` — S3 证据闭合的 `PortfolioDecision` / `RiskAssessment` v1;独立复核 freshness、约束与 computation receipt,并复用既有标签安全风险分解
- `retrospective_*_contracts` — 未发布的显式 v2 回顾性合同:区分历史业务日期与实际可得/计算时间,保留 v1 和现有金融公式,不授予历史可得性、发布或执行权限;见 [v2 接口说明](docs/RETROSPECTIVE_COMPUTATION_V2.md)
- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计
- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效
- `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test)
- `portfolio_decomp` — 组合分解(risk_parity / mean_variance / 因子归因)
- `risk` — 风险指标(边际 / 风险贡献)
- `risk` — ndarray 低层风险公式 + 标签安全、可分组的 Euler 成分风险分解
- `perf_stats` — 详细绩效(与 metrics 并存)
- `logging` — 统一 logger(标准库 + 可选 loguru)
@@ -51,6 +59,9 @@ pytest # 单元测试
pytest --cov=src # 覆盖率
mypy --strict src/ # 类型检查
ruff check src/ tests/ # lint
# 无网络、无数据库、无券商的架构烟测
uv run python -m quant_engine.governed_pipeline
```
## 使用
@@ -58,8 +69,13 @@ ruff check src/ tests/ # lint
```python
from quant_engine.alpha_factors import alpha_001, alpha_005, ALPHA158_REGISTRY
from quant_engine.execution import (
ExecutionConfig, simulate_with_daily_data, compute_realized_pnl,
ExecutionConfig, simulate_daily_ledger_with_audit,
simulate_multi_day_with_audit, simulate_with_daily_data,
)
from quant_engine.research_pipeline import (
run_factor_backtest_research, run_factor_execution_research,
)
from quant_engine.backtest import run_weight_backtest
from quant_engine.indicators import macd, bollinger, kdj
from quant_engine.data_adapter import (
long_to_wide, wide_to_long, rename_tushare_columns,
@@ -70,10 +86,332 @@ from quant_engine.data_adapter import (
# 端到端:qtdb_pro 长表 → 适配 → alpha158 → execution
df = load_qtdb_daily(["000001.SZ"], "2024-01-01", with_adj=True)
prices, volumes = prepare_execution_inputs(df)
result = simulate_with_daily_data(prices, initial_cash=1_000_000.0)
close_prices, volumes = prepare_execution_inputs(df)
open_prices, _ = prepare_execution_inputs(df, price_col="open")
result = simulate_with_daily_data(close_prices, initial_cash=1_000_000.0)
# 已正确滞后的目标权重 → 现金约束执行 → 唯一来源的成交/拒绝/日末持仓/NAV
execution = simulate_multi_day_with_audit(
target_weights_history=[
("2024-01-02", {"000001.SZ": 1.0}),
("2024-01-03", {"000001.SZ": 1.0}),
],
price_history=[
("2024-01-02", {"000001.SZ": 10.0}),
("2024-01-03", {"000001.SZ": 10.5}),
],
initial_cash=1_000_000.0,
config=ExecutionConfig(),
)
print(execution.nav_series)
print(execution.daily_executions)
# 多期因子分数(必须是 point-in-time 数据)→ Top-K → 下一交易日 open 执行
factor_execution = run_factor_execution_research(
factor_scores,
top_k=20,
execution_prices=open_prices,
execution_price_field="open",
initial_cash=1_000_000.0,
)
# 推荐研究入口:同一交易日历上显式区分 open 成交和 close 估值。
# 因子日保持现金,下一交易日成交后的真实持仓才参与当日收盘收益。
factor_backtest = run_factor_backtest_research(
factor_scores,
top_k=20,
execution_prices=open_prices,
valuation_prices=close_prices,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000_000.0,
config=ExecutionConfig(),
)
print(factor_backtest.nav)
print(factor_backtest.returns)
print(factor_backtest.stats())
print(factor_backtest.execution.ledger_frame)
print(factor_backtest.execution.trades_frame)
print(factor_backtest.position_weights) # 实际日末资产权重
print(factor_backtest.cash_weights)
# 所有分析都以实际成交后的 Ledger 为事实源,不直接使用目标权重伪造结果。
attribution = factor_backtest.return_attribution()
print(attribution.asset_contributions)
print(attribution.transaction_cost)
print(attribution.residual) # 应接近 0;否则说明贡献未闭合到账本收益
# benchmark_returns 必须与成本后 factor_backtest.returns 使用完全相同的日期索引。
print(factor_backtest.benchmark_stats(benchmark_returns))
# 下游稳定交付:显式提供代码版本、数据快照和时区,不在核心层写数据库。
from quant_engine.artifact import build_research_run_artifact
from quant_engine.data_adapter import prepare_asset_return_snapshot
from quant_engine.risk import estimate_covariance_snapshot
risk_date = factor_backtest.position_weights.index[-1].date()
market_snapshot = prepare_asset_return_snapshot(
qtdb_daily_long,
source="qtdb_pro.hq_daily",
source_snapshot_id="<upstream-ingestion-snapshot-id>",
adjustment="qfq",
)
risk_snapshot = estimate_covariance_snapshot(
market_snapshot.returns,
as_of_date=risk_date,
lookback_sessions=252,
min_observations=120,
data_snapshot_id=market_snapshot.data_snapshot_id,
)
artifact = build_research_run_artifact(
factor_backtest,
run_id="research-run-001",
strategy_id="alpha-top20",
strategy_name="Alpha Top 20",
strategy_version="1.0.0",
engine_version="1.2.0",
code_revision="<git-sha>",
data_snapshot_id=market_snapshot.data_snapshot_id,
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-08-21T10:00:00+08:00",
finished_at="2026-08-21T10:01:00+08:00",
parameters={"top_k": 20, "lag_sessions": 1},
benchmark_id="000300.SH",
benchmark_returns=benchmark_returns,
risk_snapshots={risk_date: risk_snapshot},
)
print(artifact.manifest())
# run_weight_backtest 是低层算子:只接受收益区间开始前已经生效的持仓权重。
# 不要把 signal-date 的 factor_scores/decision_weights 直接传给它。
backtest = run_weight_backtest(
weights=effective_holding_weights,
stock_returns=daily_returns,
initial_capital=1_000_000.0,
benchmark_nav=benchmark_nav,
)
print(factor_execution.schedule.signal_to_execution)
print(factor_execution.execution.daily_executions)
print(backtest.stats())
print(backtest.benchmark_report())
```
## 因子/特征合同 v1
`quant_engine.factor_contracts` 提供 `researchhub.factor-definition` 与
`researchhub.factor-set-ref` `1.0.0`。合同使用受限 canonical JSON:只接受 ASCII
lower-snake-case object key、UTF-8 string、bool/null 和 safe integer;小数参数必须用显式
canonical decimal string。定义、输入映射、上游证据、输出 schema/content 和 lineage 的任一
语义变化都会产生新 identity。
创建 `FactorSetRef` 必须提供完整且可重算 identity 的 `DatasetSnapshotEnvelope` 与
`DataFoundationEnvelope`,不能用 ID 字符串或布尔值代替资格证明。每个因子输入都要映射到一个
实际选中的 `StandardizedViewRef`,schema 必须同时匹配定义和 view;未消费、缺失、重复或跨
snapshot/Foundation/PIT 的 view 都会失败关闭。snapshot PIT 可以早于 Foundation/view PIT,
但始终满足 knowledge ≤ snapshot PIT ≤ Foundation/view/FactorSet PIT ≤ evaluation。
```python
from quant_engine.factor_contracts import (
DataFoundationEnvelope,
DatasetSnapshotEnvelope,
FactorDefinition,
FactorSetRef,
)
snapshot = DatasetSnapshotEnvelope.from_dict(dataset_snapshot_v1)
foundation = DataFoundationEnvelope.from_dict(data_foundation_v1)
# definition 必须是完整的 FactorDefinition;FactorSetRef.create 还要求显式 input bindings、
# view availability、output quality/coverage、canonical output bytes 和 immutable artifact ref。
factor_set = FactorSetRef.create(
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
**explicit_factor_set_evidence,
)
```
`availability_mode="as_available"` 声明 source/view 和计算产物在历史 evaluation 前实际可用;
`"retrospective_replay"` 保留历史 evaluation,但要求真实 publication/view creation、compute 和
artifact 时间位于之后,并固定 `historical_availability="not_established"`。两种模式都不会授予
decision、real-data、production、paper 或 live readiness。
旧 `governed_pipeline.FactorVersion` 的四字段构造器、`factor_id@version`、run/target/risk/order
identity 均保持不变。迁移只能通过 content-addressed `LegacyFactorBinding`,再显式调用
`bind_legacy_factor()` 或 `project_legacy_factor()`;后者是有损投影,不表示旧 digest 与新定义
digest 等价,也不会把旧 run 静默升级为新合同。
## 回测引用与证据合同 v1
`quant_engine.governed_pipeline.BacktestRunRef` 是合格回测运行身份的唯一权威。`run_id` 只由
已验收的 Dataset Snapshot / Data Foundation / `FactorSetRef` 身份、universe、日历与公司行动
祖先、策略、执行/成本模型、严格整数 seed、完整代码提交、环境锁、配置、时间和重放血缘决定;
它不包含任何输出摘要。重放必须绑定直接父运行、连续 attempt 和不变的
`replay_spec_digest`,输入漂移或血缘环会失败关闭。
`quant_engine.artifact.BacktestEvidenceManifest` 只摘要 `ResearchRunArtifact` 已有的九张事实表。
固定 `offline_research_v1` 映射为 `run`、`signal`、`fill`、`position_nav`、`performance`、
`attribution`、`risk_snapshot` 与 `replay`;每张表都保留列模式摘要、行数和内容摘要,空 risk
表也必须有稳定 schema。`manifest_id` 由完整 RunRef 与输出证据决定,因此结果变化不会反向改变
`run_id`。当前 artifact 不拥有订单或拒绝事实,所以此画像明确不声明 `order` / `rejection`。
旧 `BacktestRun` 只能通过 `build_legacy_backtest_evidence_manifest()` 显式映射为
`LEGACY_EXPLORATORY`;不能隐式提升为 `CONTRACT_QUALIFIED`。所有资格均只描述离线证据闭合,
不表示投资有效、组合获批、Paper、生产或实盘就绪。
## 绩效证据与方法论合同 v1
`quant_engine.artifact.PerformanceEvidenceV1` 在现有计算和事实表之外增加一层只读、内容寻址的
owner 证据。`build_performance_evidence()` 只接受同一运行的完整 `ResearchRunArtifact`、
`BacktestRunRef` 与 `CONTRACT_QUALIFIED BacktestEvidenceManifest`;它核对全部 artifact 表、
performance 表和唯一行摘要,并绑定 artifact、row 与严格对齐 benchmark series 的独立摘要。
生产 builder 不重算、填补、重命名或覆盖任何绩效值。
方法论固定为日简单收益、252 期年化、绝对指标年化无风险利率 `0.0`、benchmark 日无风险利率
`0.0`,以及 benchmark 存在时的 `exact_session_index`。相对指标使用封闭 availability:无基准为
`benchmark_absent`;active variance、benchmark variance 或 alpha 几何年化域不足时分别使用
对应 `not_estimable_*` 原因。benchmark 存在时 tracking error 始终必须是有限非负值;null 不会
被转成零。
```python
from quant_engine.artifact import build_performance_evidence
performance_evidence = build_performance_evidence(
artifact,
backtest_run_ref,
backtest_evidence_manifest,
)
canonical_bytes = performance_evidence.canonical_bytes()
```
该合同范围固定为 `offline_research_only`。它不授予排名、推荐、决策、发布、论文、Paper、生产、
实盘、交易或投资建议权限,也不包含原始参数、returns、NAV、benchmark series、表字节、存储
locator、URI 或凭证。
## 组合决策与风险评估合同 v1
`quant_engine.portfolio_risk_contracts` 是现有计算 owner 外围的薄合同层。创建
`PortfolioDecision` 必须同时提供完整 `BacktestRunRef`、嵌入同一 RunRef 的
`CONTRACT_QUALIFIED` 非 legacy `BacktestEvidenceManifest`、现有 `PortfolioTarget`、
`FreshnessPolicy`、`ConstraintSetV1` 与 `ComputationReceipt`。适配器会从权威输入独立重算
receipt 的 input/constraint/output digest、敞口、持仓数和 L1 turnover 残差;receipt 自报
成功、fallback 或放宽 tolerance 均不能替代复核。
```python
from quant_engine.portfolio_risk_contracts import (
ComputationReceipt,
ConstraintSetV1,
FreshnessPolicy,
assess_portfolio_risk,
build_portfolio_decision,
compute_portfolio_receipt_digests,
)
freshness = FreshnessPolicy(
max_manifest_age_seconds=3600,
max_covariance_age_days=5,
)
constraints = ConstraintSetV1(
gross_exposure_max=1.0,
single_asset_max=0.10,
position_count_max=20,
turnover_max=0.30,
)
# 生产者先形成公开 canonical digest;decision 构建时仍会独立重算。
expected = compute_portfolio_receipt_digests(
backtest_run_ref=run_ref,
manifest=evidence_manifest,
target=portfolio_target,
objective_name="long_only_allocation",
objective_version="1.0.0",
objective_digest=objective_digest,
model_name="factor_weighting",
model_version="1.0.0",
model_digest=model_digest,
expected_return_digest=expected_return_digest,
covariance_digest=covariance_digest,
scenario_digest=scenario_digest,
constraints=constraints,
freshness_policy=freshness,
prior_weights=prior_weights,
)
receipt = ComputationReceipt(
algorithm="factor_weighting",
algorithm_version="1.0.0",
implementation_digest=implementation_digest,
parameter_digest=parameter_digest,
input_digest=expected["input_digest"],
constraint_digest=expected["constraint_digest"],
output_digest=expected["output_digest"],
status="completed",
solver_required=False,
solver_name=None,
solver_version=None,
solver_config_digest=None,
iterations=None,
objective_value=None,
max_constraint_residual=expected["max_constraint_residual"],
tolerance=1e-12,
computed_at=computed_at,
)
decision = build_portfolio_decision(
backtest_run_ref=run_ref,
manifest=evidence_manifest,
target=portfolio_target,
objective_name="long_only_allocation",
objective_version="1.0.0",
objective_digest=objective_digest,
model_name="factor_weighting",
model_version="1.0.0",
model_digest=model_digest,
expected_return_digest=expected_return_digest,
covariance_digest=covariance_digest,
scenario_digest=scenario_digest,
constraints=constraints,
freshness_policy=freshness,
receipt=receipt,
computed_at=computed_at,
prior_weights=prior_weights,
)
assessment = assess_portfolio_risk(
portfolio_decision=decision,
backtest_run_ref=run_ref,
manifest=evidence_manifest,
covariance=covariance_snapshot,
risk_model_name="euler_volatility",
risk_model_version="1.0.0",
risk_model_digest=risk_model_digest,
)
```
`source_universe_digest` 保留 S3 研究 universe 身份,`portfolio_asset_set_digest` 只描述实际
目标资产标签;二者不会互相冒充成员证明。风险评估在任何数值计算前要求 covariance、target、
RunRef 的 dataset identity 三方一致,并且只调用一次现有 `labeled_component_risk()`。合同中的
`qualified` 仅表示 S4.1 计算证据闭合,不授予 maker-checker、发布、订单、Paper、生产或实盘权限。
## 治理垂直切片
`governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。
调用方必须显式提供 `DatasetSnapshot`、`FactorVersion`、`StrategyVersion`、代码提交和
`RiskPolicy`。模块只会生成 `environment="paper"` 的订单意图,不连接数据库、数据供应商或
券商;风险决策为拒绝时,订单意图固定为空,直接调用创建函数也会失败关闭。
该切片对应 ResearchHub 架构的首个可执行验收链路:
```text
DatasetSnapshot → FactorVersion → StrategyVersion → BacktestRun
→ PortfolioTarget → RiskDecision → PaperOrderIntent
```
平台总架构、五仓职责和十二层能力映射仍以 `research_platform/docs/architecture/` 为权威;
本仓只拥有纯计算与离线模拟合同。
## 与 research_results 的关系
`research_results` 依赖 `quant_engine`(通过 re-export 保持向后兼容):
+9
View File
@@ -0,0 +1,9 @@
profile: lite
runtime_contract: v1
language: python
python_version: "3.13"
python_manager: uv
python_root: "."
local_test_command: "python3 tests/governance/test_module_spec.py"
requires_database: false
integration_profile: none
+64
View File
@@ -0,0 +1,64 @@
# Open-source design references
本项目采用“借鉴稳定语义、保留轻量实现”的策略。引入新量化能力前先检查成熟
开源案例;除非维护成本和许可证收益明确优于本地小型实现,否则不增加框架级依赖。
## 2026-08-21:成交后归因与相对绩效
| 项目 | 借鉴内容 | 当前决策 |
|---|---|---|
| [Qlib](https://github.com/microsoft/qlib) | 信号时间与交易时间分离、成本前后超额收益分开报告 | 借鉴语义;不引入完整框架 |
| [Zipline](https://github.com/quantopian/zipline) | Ledger / transaction / portfolio value 状态模型 | 以现有 `ExecutionSimulationResult` 承担事实源 |
| [empyrical](https://github.com/quantopian/empyrical) | beta 协方差口径、alpha 几何年化、年化因子 | 移植小型公式;不增加老旧运行时依赖 |
| [Riskfolio-Lib](https://github.com/dcajasn/Riskfolio-Lib) | Euler component risk 与分组/因子风险贡献 | 只实现当前需要的 pandas/numpy 标签安全封装 |
| [PyPortfolioOpt](https://github.com/PyPortfolio/PyPortfolioOpt) | 协方差估计与优化器解耦 | 留作未来风险模型适配器参考 |
当前核心不新增依赖。逐日收益归因必须从实际换仓前后持仓、成交记录、执行价和
收盘估值推导;因子分数与目标权重只是意图,不能作为成交后归因事实源。
## 2026-08-21:研究运行工件
- 借鉴 [Qlib Recorder / RecordTemplate](https://github.com/microsoft/qlib/blob/main/qlib/workflow/record_temp.py)
将 signal、portfolio analysis 和 risk analysis 分成稳定事实,但不引入 Qlib 运行时;
- 借鉴 [MLflow Tracking](https://mlflow.org/docs/latest/tracking/) 的 run / params /
metrics / artifacts 分层,但 MLflow 只保留为未来可选 exporter;
- HTML、PNG 和 tearsheet 是可再生展示物,不能替代 NAV、成交、持仓、归因和绩效事实。
因此 `ResearchRunArtifact` 使用显式 `schema_version`、`config_hash`、代码版本和数据
快照身份,并提供确定性 JSON / SHA-256 manifest;核心层仍不写数据库或 artifact store。
schema `1.1.0` 将 Qlib 的独立 risk-analysis artifact 思路与 Riskfolio-Lib 的 Euler
component-risk 语义结合,但只保留本项目需要的轻量合同:协方差快照必须声明
`snapshot_id`、`as_of_date`、收益频率和年化期数;风险从成交后的实际日末持仓计算,
component risk 闭合到年化组合波动,percentage contribution 闭合到 1。未来日期、资产
标签不完整和零方差组合都直接失败,不以默认值伪造结果。
## 2026-08-21:协方差快照估计
| 项目 | 借鉴内容 | 当前决策 |
|---|---|---|
| [PyPortfolioOpt risk models](https://github.com/PyPortfolio/PyPortfolioOpt/blob/main/pypfopt/risk_models.py) | 将收益输入、协方差估计器和组合优化解耦;sample / EWM / shrinkage 使用统一标签输出 | 借鉴可替换估计器边界,不引入完整包 |
| [scikit-learn covariance](https://github.com/scikit-learn/scikit-learn/blob/main/sklearn/covariance/_shrunk_covariance.py) | 维护成熟的 Ledoit–Wolf / OAS shrinkage 实现 | 未来作为可选 adapter;不复制统计公式 |
| [Qlib structured risk model](https://github.com/microsoft/qlib/blob/main/qlib/model/riskmodel/structured.py) | PCA/FA 结构化协方差和固定随机状态 | 留作因子风险模型阶段,不进入当前 baseline |
当前 `estimate_covariance_snapshot` 只编排 pandas 的 sample covariance:先按 `as_of_date`
截断,再取固定 session 窗口,使用 complete-case 行并拒绝历史不足;禁止 pandas 默认的
pairwise 样本集合产生含义不一致的矩阵。snapshot ID 对窗口数据、缺失掩码、上游数据
快照身份和估计参数做 SHA-256,追加未来数据不会改变历史快照。
市场适配层现以 `AssetReturnSnapshot` 固化 simple-return 输入:上游 ingestion snapshot ID、
数据源、价格字段、复权口径、规范化价格值和缺失掩码共同形成内容寻址 ID;不前向填充
停牌/缺失价格。该 ID 同时传入协方差快照和研究运行工件,避免同一研究链出现两套数据
身份。
可选 shrinkage adapter 的评估结论是“保留边界,暂不实现”:当前运行依赖没有声明
scikit-learn,本切片也不修改版本或锁文件。未来只有在依赖治理接受后,才以延迟导入
直接调用 scikit-learn 的 `LedoitWolf` / `OAS`,并让估计器名称、库版本与参数进入
snapshot identity;不复制成熟统计公式,也不让环境中偶然存在的包改变 baseline 行为。
## hikyuu 的定位
[hikyuu](https://github.com/fasiondog/hikyuu) 的 SG / MM / CN / PG 部件化思想、
A 股交易约束和系统组合方式仍有借鉴价值;但其完整 C++/Python 运行时、对象模型和
数据体系不适合作为本项目核心依赖。当前原则是按真实研究链路吸收边界设计,不复制
其框架层级,也不为了“架构完整”预先建设尚无端到端需求的抽象。
+145
View File
@@ -0,0 +1,145 @@
# Retrospective computation contracts v2 (unreleased)
This pure, storage-neutral compatibility path consumes the separate data-contract
major 2.0.0. It does not migrate, reinterpret or relax the accepted v1 contracts.
No financial formula, execution simulation, dependency lock, production database,
publisher or live/paper-order interface changes here. Package version is unchanged;
the new contract major is not a package release or deployment.
## Explicit public boundaries
| Module | Public types/builders | Changed wire identity |
| --- | --- | --- |
| `retrospective_data_contracts` | `RetrospectiveSnapshotEnvelope`, `RetrospectiveFoundationEnvelope` | `rhdsv2`, `rhdfv2`; consume RP-owned 2.0.0 data semantics |
| `retrospective_factor_contracts` | `RetrospectiveFactorSetRef`, typed input/view/causation bindings, `ResolvedRetrospectiveView` | `rhfactorsetv2` |
| `retrospective_backtest_contracts` | `RetrospectiveBacktestRunRef` | `rhbacktestrunv2` |
| `retrospective_artifact_contracts` | `RetrospectiveBacktestEvidenceManifest`, `RetrospectivePerformanceEvidence` and their builders | `rhbacktestevidencev2`, `rhperformancev2` |
| `retrospective_portfolio_risk_contracts` | `RetrospectivePortfolioTarget`, `RetrospectivePortfolioDecision`, `RetrospectiveRiskAssessment`; receipt-digest, decision and assessment builders | `rhportfoliotargetv2`, `rhportfoliodecisionv2`, `rhriskassessmentv2` |
These are separate types and domain-separated content identities. There is no
automatic v1-to-v2 cast. Unknown schema versions and fields are rejected. The
performance wire keeps its named schema `researchhub.performance-evidence.v2`;
the other new computation contracts use `schema_version: 2.0.0`.
FactorDefinition, factor-output byte references, output quality/coverage,
ConstraintSetV1, FreshnessPolicy, ComputationReceipt, CovarianceSnapshot, financial
algorithms, performance metric/methodology IDs and the nine ResearchRunArtifact
tables keep their existing semantics. The table schema remains **1.1.0**. Reusing
these neutral primitives does not make a new-major upstream reference v1-compatible.
## Two clocks, not backdated evidence
Every new result fixes `usage=retrospective_research` and
`historical_availability=not_established`. A business date describes the historical
period being researched. Observation, publication, evaluation, artifact availability,
target creation and computation describe actual events, and must not be backdated.
Public v2 instants require UTC `Z` with at most six fractional digits.
`observation_cutoff` and chunk `observed_by` are upper-bound observations. They are
not the earliest public knowledge time or a PIT cutoff. Unknown earliest knowledge
stays unknown; a supplied knowledge-evidence digest is not authenticated by parsing.
Foundation observation sequences describe retained revisions, not complete original
history. Selected view routes, calendars, corporate-action coverage and lineage
must close exactly within the supplied Foundation.
Required actual order for factor/backtest evidence is:
1. Foundation publication <= factor evaluation <= factor computation <= factor availability.
2. Factor availability <= backtest evaluation <= artifact start <= artifact finish
<= backtest computation <= artifact availability.
3. Artifact availability <= target creation <= portfolio computation <= risk computation.
RetrospectivePortfolioTarget has a historical `effective_at` and a distinct actual
`created_at`. PortfolioDecision carries both plus actual `computed_at`. Covariance
window end <= covariance as-of date <= the historical effective date; covariance
maximum age is measured against that historical date. Manifest maximum age is
measured against **actual** portfolio and risk computation separately. Passing one
age check cannot substitute for the other. Generic v1 receipt timestamps retain
their original normalization; binding compares parsed actual instants.
## Materialized bytes and reference-only reads
Snapshot decoding checks structure, all six blocking-quality declarations,
qualification/time ordering, observation receipts and identities.
`verify_materialized_records` additionally checks supplied chunks, per-chunk and
aggregate content, counts, dimensions, effective ranges and macro effective instants.
Provider/physical paths are forbidden in public metadata and materialized records.
Factor creation requires actual snapshot chunks, selected view schema/content bytes,
and factor-output schema/content bytes. Definition inputs, view availability,
Foundation ancestry and computed digests must close. Reference-only deserialization
is allowed for display/inspection, but input/output validation flags are derived from
supplied bytes, are not serialized claims, and must be re-established for new
computation. Backtest creation requires a factor whose payloads were revalidated.
Reference decoding cannot turn an unverified factor into an admitted compute input.
Backtest manifest decoding rebuilds evidence from the supplied typed run and all
nine actual artifact tables. It checks table/run/config/strategy bindings and time
ordering. Portfolio composition revalidates those retained tables again, rather
than trusting a serialized manifest or mutable Python context. A table digest proves
content binding, not that those tables were produced by the claimed computation.
All content-addressed IDs exclude their own ID field and bind the remainder of the
closed payload. Data/factor/backtest/manifest JSON retains the strict data profile
(no JSON floating-point numbers; financial record decimals are strings). Performance
and S4 preserve the existing finite numeric JSON profile: finite floats, safe ints,
exact booleans, sorted keys, compact separators, UTF-8. Duplicate keys, NaN,
Infinity, noncanonical JSON and extra fields are rejected. Wire revalidation uses
type-sensitive comparisons, including `true` versus `1`. Serializers return
detached copies; internal public maps are immutable.
## Replay, receipts and risk
Backtest v2 replay specification binds immutable input identities, selected calendar
and actions, strategy/execution/cost versions and digests, configuration, code,
environment lock and random seed. It excludes **both actual evaluation and actual
computation time**. These actual times remain in each run's identity. A replay must
retain the same replay specification, append its unique full ancestry, increment
attempt by one, and have parent computation < new actual evaluation <= computation.
This explicit new-major rule allows a later genuine replay without pretending its
evaluation happened at the parent's clock time.
Portfolio computation-input v2 binds the full run and manifest document digests,
new target (including both clocks), objective/model versions and digests, declared
expected returns/covariance/scenario inputs, freshness policy and prior weights.
The receipt separately binds that input, constraints and recomputed outputs/residuals.
Targets and prior holdings must use selected logical instrument IDs, not ad-hoc
symbol matches. Failed/fallback receipts and any actual constraint residual are
rejected, even if a solver declares convergence within a permissive tolerance.
Risk uses the existing labelled Euler decomposition exactly once. Its result binds
the supplied matrix content plus covariance method, bounded estimation window,
observation count, lookback, missing policy, annualization, source dataset/input,
model/budgets/groups and actual computation time. It checks exact labels, finite
symmetry, covariance-source binding and both freshness clocks. Non-PSD,
non-positive portfolio variance or non-closed contributions produce an unavailable,
unqualified result. A budget breach is a ready but unqualified calculation result.
`qualified=true` means only that these calculation checks passed. Every result
remains `decision_eligible=false`, `execution_validation=not_validated`; no portfolio
approval, maker-checker, publication, paper or live permission is granted here.
## Trust, ownership and test evidence
Pure builders accept declarations. Hashes, typed objects, model names, successful
constraint checks and synthetic fixtures do **not** authenticate data or compute
producers. Trusted owner-version bindings and receipt/qualification/view/clock
admission ports remain mandatory. RP owns governance and presentation; Research
Results owns publication. QE supplies validated calculation facts only.
The two data fixtures are public RP candidate vectors from
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (PR #100). EDB producer candidate
`88433dfc9d865ef782465498cdf9454c73920abd` (PR #13) is not a runtime dependency or
accepted owner binding. Acceptance/review gates remain separate from local tests.
`tests/fixtures/retrospective-computation-v2.golden.json` freezes newly constructed
synthetic factor/run/manifest/performance/portfolio/risk payloads and their inputs.
Its artifact matrices are separate synthetic envelope-test inputs: the one-day
public data fixture is **not** claimed to have produced the four-day artifact.
The vector is not an end-to-end data/computation provenance proof or real-data run.
Its deterministic IDs are contract-regression evidence, not admitted source facts.
Focused tests cover v2 goldens, mutation and strict JSON, bytes versus references,
two-clock freshness, replay ancestry, exact table bindings, constraints, receipts,
covariance provenance and numerical findings. Existing v1 tests must also pass.
Rollback is disabling the explicit v2 entry path while retaining v1 and original
immutable results; never retag old results or silently downgrade failed v2 admission.
@@ -0,0 +1,42 @@
# Ledger-backed attribution handoff
## Goal
在 `ExecutionSimulationResult` 日频 Ledger 之上增加轻量、可审计的成交后分析层:
- 逐日隔夜 / 日内资产收益贡献;
- 佣金、印花税、滑点成本独立贡献;
- 贡献闭合到成本后日收益并显式暴露 residual;
- 严格日期对齐的 TE / IR / alpha / beta;
- 标签安全且可分组的 Euler component risk。
- 从 Ledger 股数和收盘估值投影的实际资产 / 现金权重。
## Branch stack
- 当前:`codex/ledger-attribution-20260821`
- 基线:`codex/post-execution-ledger-20260821`
- 再下层:`codex/core-contracts-20260821`(PR #2,尚待用户确认合并)
本分支不得直接合并到 `main`。应按上述顺序逐层审阅;未经用户明确确认,不得合并
L2 PR。
## Open-source decision
调研结论记录在 `docs/OPEN_SOURCE_REFERENCES.md`。Qlib、Zipline、empyrical、
Riskfolio-Lib 和 PyPortfolioOpt 只作为时间语义、Ledger、相对指标与 Euler 风险贡献
的设计参考;本阶段没有新增运行时依赖。
## Verification
- `pytest -q --cov=src --cov-report=term-missing`: 514 passed,9 个既有 SciPy warning,91% coverage;
- `mypy --strict src/`: 15 source files passed;
- 变更范围 `ruff check`: passed;
- 全仓 Ruff:仅 13 个既有 `tests/governance/*` PT009;
- workspace verify/status:passed,预期提示 quant_engine 非 main;
- global Gitea workflow check:passed,23 个无关仓库 warning。
## Next action
先按堆叠顺序审阅 PR。基础 Ledger 分支完成后,再将本分支 rebase 到其最终提交,
运行唯一一次 `ship --ready`;随后将稳定输出适配到 `research_results` 与
`research_platform`,不要在核心层直接写数据库。
@@ -0,0 +1,33 @@
# Post-execution daily Ledger handoff
## 状态
- 分支:`codex/post-execution-ledger-20260821`
- 基线:`codex/core-contracts-20260821`(PR #2,尚未获用户确认合并)
- 本分支不得直接合并到 `main`;先等待 PR #2 合并,再整理基线并创建独立 PR。
- 无账户、券商、数据库或实盘副作用。
## 已完成
- 新增稀疏调仓、完整交易日估值的 `simulate_daily_ledger_with_audit()`。
- 显式分离 execution price 与 valuation price,支持下一日 open 成交、当日 close 估值。
- 成交记录补齐 `side / quantity / price`,并提供 `trades_frame`。
- 提供平台中立的 `ledger_frame`,不携带 `run_id`,不写数据库。
- 新增 `run_factor_backtest_research()`:PIT 因子、下一交易日执行、日频 NAV、首日成本收益和标准绩效。
- 研究区间从首条有效信号日开始,排除因子预热行情对绩效的稀释。
## 验证
- `pytest -q --cov=src --cov-report=term-missing`:500 passed,total coverage 91%。
- `mypy --strict src/`:14 source files passed。
- 本阶段文件 scoped Ruff:passed。
- 全仓 Ruff:仅既有 governance tests 的 13 个 PT009 基线问题。
- workspace verify/status:通过;仅提示功能分支不是引导基线 `main`。
- 全局 Gitea workflow check:通过,23 个既有警告。
## 继续步骤
1. 获得用户对 PR #2 的明确合并确认并按 L2 流程合并。
2. 将本分支整理到更新后的 `main`,重新运行相同全量验证。
3. 为 Ledger 阶段创建独立 PR,执行唯一一次最终 `ship --ready`,等待用户确认合并。
4. 后续在 `research_results` 增加业务投影适配器,再由 `research_platform` 持久化和展示;核心层继续保持无数据库写入。
@@ -0,0 +1,57 @@
# Research artifact contract handoff
## Goal
把完整可信研究链固化成存储中立、版本化、确定性的 `ResearchRunArtifact`,供
`research_results` 持久化和 `research_platform` 查询:
- run identity / schema version / config hash / code revision / data snapshot;
- signal scores / decision weights / signal-to-execution mapping;
- NAV / returns / benchmark / costs;
- trades / realized positions / cash;
- asset and daily return attribution;
- performance including Sortino / TE / IR / alpha / beta;
- reproducible covariance snapshots and annualized Euler component-risk facts;
- canonical JSON / SHA-256 manifest。
## Branch stack
- 当前:`codex/research-artifact-contract-20260821`
- 基线:`codex/ledger-attribution-20260821`(Draft PR #4)
- 下层:Draft PR #3 → Ready PR #2 → `main`
不得绕过堆叠顺序直接合并到 `main`。
## Verification
- `pytest -q`: 540 passed,9 个既有 SciPy warning;
- data-adapter focused coverage 77%(包含未连接真实 ClickHouse 的 I/O 便捷函数);
- `mypy --strict src/`: 16 source files passed;
- changed-scope Ruff: passed;
- no runtime dependency added;
- no database, network, broker or filesystem write side effect in artifact builder。
- 三仓隔离 ClickHouse 黄金链路通过:市场价格 → return snapshot → covariance → artifact →
publisher → reader;使用随机 localhost 端口、tmpfs 和自动容器清理。
## Current risk contract
- artifact schema:`1.1.0`;
- `CovarianceSnapshot` 对输入矩阵深拷贝并显式记录截至日、频率和年化期数;
- `risk_snapshots` 按研究交易日映射,可只生成需要的风险观察日;
- 使用成交后实际持仓,不包含现金风险资产;协方差资产标签必须与研究资产全集一致;
- `covariance_as_of_date` 不得晚于 `trade_date`;无正组合方差时拒绝产物。
- `estimate_covariance_snapshot` 从显式数据快照的日收益生成无前视、complete-case、
SHA-256 可复现的 per-period sample covariance;不包含 I/O 或未来行。
- `prepare_asset_return_snapshot` 从规范化长表行情生成不前向填充的 simple daily returns;
显式 ingestion snapshot ID、源/字段/复权口径、价格值和缺失掩码共同形成
`asset-returns-v1:<sha256>`,并把同一 ID 传给 covariance 与 run artifact。
- artifact builder fail closed:每个 `CovarianceSnapshot.data_snapshot_id` 必须与 run 级
`data_snapshot_id` 完全一致,禁止把其他行情快照的风险分解静默发布到当前研究运行。
- shrinkage 适配器本轮不实现:scikit-learn 尚非声明依赖,未来只允许薄适配
`LedoitWolf` / `OAS`,不复制公式、不依赖环境偶然安装状态。
## Next action
保持 Draft PR #5,不绕过堆叠顺序合并;下游 `research_results` / `research_platform`
继续在现有 Draft 分支消费同一数据 lineage。下一阶段优先把 ingestion snapshot ID 从
真实 ELT 元数据接入调用方,再在依赖治理通过后单独交付可选 shrinkage adapter。
+88
View File
@@ -0,0 +1,88 @@
# Quant Engine retrospective v2 compatibility
Scope: implement the user-authorized retrospective v2 compatibility without changing
v1 semantics, financial algorithms, original results, production databases, deployment
or trading. No claim of complete Quant OS delivery or real-data qualification.
Branch: `codex/research-quant-os-retrospective-contract-v2-20260908`.
Declared base: accepted `main@68dd68392a26251391fbdae40c22eee370adb56e`.
One isolated delivery worktree; the old primary checkout is preserved. This is not
reactivation of an old registered stage or creation of a new stage ledger.
## Dependency baseline
Public RP data-contract candidate: PR #100, initial schemas/goldens at
`7da27e5bd33dc6d06f2c7c60f47029111156293a` (review/acceptance pending).
EDB mapping candidate: PR #13, initial implementation `88433df`, local full
validation passed. Neither candidate is silently treated as accepted owner evidence.
The shared public major is 2.0.0; preserve the accepted v1 paths independently.
Order: public data contracts -> EDB mapping/Foundation -> Quant Engine typed
factor/backtest/portfolio/risk -> RP governance -> Research Results -> RP read.
Accepted owner-version bindings and runtime admission must still close every boundary.
## Internal reuse decision
Need: carry observation-aware inputs and retrospective-only claims through computation.
Existing: strict canonical JSON, immutable envelopes, factor definitions, input/output
closure, numerical algorithms, governed backtest and portfolio/risk contracts.
External candidates: not needed; this is project-owned semantics, not a missing library.
Approach: reuse those primitives and algorithms; introduce explicit new-major wrappers
only where upstream identity, time or usage semantics change.
Risk: reusing the v1 decoder or coercing observed-by into knowledge/PIT would make a
false historical claim. Unknown versions and unsupported usages must fail closed.
## Implemented, not yet accepted or released
Five separate v2 modules now implement immutable DatasetSnapshot/Foundation decoding
and materialized-content verification, FactorSet with explicit v2 nested bindings,
BacktestRunRef and replay ancestry, nine-table BacktestEvidenceManifest,
PerformanceEvidence, PortfolioTarget/Decision and RiskAssessment. The metadata
registers the new major alongside every existing v1 entry. See
`docs/RETROSPECTIVE_COMPUTATION_V2.md` for normative clocks, JSON profiles, input
closure, replay and owner-port boundaries.
Factor definitions, generic output/receipt/constraint/covariance primitives and
financial implementations are reused without semantic edits. Table schema remains
1.1.0; v1 business source, v1 goldens, `pyproject.toml`, `uv.lock` and `ci-profile.yml`
are unchanged. Package version remains unreleased. Only module metadata, its exact
inventory test and README gain v2 alongside the new files.
The frozen synthetic computation vector includes fresh factor/backtest/manifest/
performance/target/portfolio/risk documents and synthetic artifact tables. It is
explicitly **envelope-only**, not an end-to-end claim that the one-day data fixture
produced the four-day synthetic financial artifact. No old real run was rerun,
retagged or backdated.
## Local verification (2026-09-08)
- Full repository unit suite: **1039 passed**, 1166 warnings, 31.50 seconds.
- S4 focused new + unchanged v1 contracts: **120 passed**; new S4 332 statements,
20 branches, 100% measured coverage. Coverage is not source authentication or
proof of complete business semantics.
- All five new source modules passed mypy; all six new test modules, five new
sources and the updated metadata test passed Ruff.
- The combined synthetic vector and metadata smoke checks: **3 passed**.
- Actual negative tests reproduced and fixed missing covariance-estimation context
in result identity, untyped malformed-JSON errors, and risk-time stale-manifest
reuse. Other modules' earlier RED/GREEN evidence remains part of the same turn.
The full suite was run directly against the frozen local environment. This is not
the same claim as remote CI or central ship acceptance; the unchanged declared CI
profile is `lite` with the module-metadata smoke command. Central validation and
Draft PR creation follow the implementation commit. No Ready, merge, accepted
upstream binding or independent-review pass is claimed here.
Actual computation/admission times are distinct from simulated business dates. New
formal outputs cannot inherit the old run's producer identity or be backdated to it.
Real receipt/qualification/view/clock ports remain mandatory; typed objects and hashes
are not source authentication. The optional independent reviewer delegation is still
awaiting the already-requested user choice.
Next: preserve the candidate for review, then carry explicit v2 facts through
RP governance -> Research Results publication -> RP read compatibility. Bind final
accepted upstream versions only when actual acceptance evidence exists. The entire
Quant OS goal is not complete at this intermediate owner unit.
Rollback: disable the explicit v2 path and retain v1 plus immutable artifacts; never
retag v2 into v1 or silently use synthetic evidence for real admission.
+6 -3
View File
@@ -7,7 +7,7 @@ name = "quant_engine"
version = "0.1.0"
description = "量化研究引擎 —— alpha 因子库 + 执行仿真 + 技术指标 + 数据适配 + 回测工具(v1.2.0 从 research_results 抽出)"
readme = "README.md"
requires-python = ">=3.11"
requires-python = ">=3.13,<3.14"
license = { text = "MIT" }
authors = [
{ name = "researchhub team" },
@@ -39,7 +39,7 @@ where = ["src"]
[tool.ruff]
line-length = 100
target-version = "py311"
target-version = "py313"
[tool.ruff.lint]
select = ["E", "F", "W", "I", "N", "UP", "B", "A", "C4", "PT", "RUF"]
@@ -54,10 +54,13 @@ ignore = [
]
[tool.mypy]
python_version = "3.11"
python_version = "3.13"
strict = true
ignore_missing_imports = true
[tool.uv]
index-url = "https://mirrors.cloud.tencent.com/pypi/simple"
[tool.pytest.ini_options]
testpaths = ["tests"]
addopts = "-v --tb=short"
+927 -1
View File
@@ -15,7 +15,9 @@ v1.2.0 Phase 0:5 个基础算子 + 5 个 alpha 公式(alpha001–alpha005)
from __future__ import annotations
from typing import Any
from collections.abc import Callable, Mapping
from types import MappingProxyType
from typing import Any, cast
import numpy as np
import pandas as pd
@@ -209,6 +211,367 @@ def indneutralize(series: pd.Series, groups: pd.Series) -> pd.Series:
return series - series.groupby(groups).transform("mean")
# ── Phase 1 operator contract ──────────────────────────
# This is deliberately a small, stable surface for downstream research
# orchestration. The full alpha158 formula catalogue can continue to grow,
# while callers use one validated dispatch entry point for the first ten
# deterministic building blocks.
ALPHA158_PHASE1_MAX_WINDOW = 252
ALPHA158_PHASE1_OPERATOR_SPECS: dict[str, dict[str, Any]] = {
"rank": {
"name": "rank",
"formula": "rank(series)",
"inputs": ["series"],
"windowed": False,
},
"delta": {
"name": "delta",
"formula": "delta(series, window)",
"inputs": ["series"],
"windowed": True,
},
"ts_mean": {
"name": "ts_mean",
"formula": "ts_mean(series, window)",
"inputs": ["series"],
"windowed": True,
},
"ts_std": {
"name": "ts_std",
"formula": "ts_std(series, window)",
"inputs": ["series"],
"windowed": True,
},
"ts_rank": {
"name": "ts_rank",
"formula": "ts_rank(series, window)",
"inputs": ["series"],
"windowed": True,
},
"correlation": {
"name": "correlation",
"formula": "correlation(series, secondary, window)",
"inputs": ["series", "secondary"],
"windowed": True,
},
"ts_min": {
"name": "ts_min",
"formula": "ts_min(series, window)",
"inputs": ["series"],
"windowed": True,
},
"ts_max": {
"name": "ts_max",
"formula": "ts_max(series, window)",
"inputs": ["series"],
"windowed": True,
},
"ts_sum": {
"name": "ts_sum",
"formula": "ts_sum(series, window)",
"inputs": ["series"],
"windowed": True,
},
"decay_linear": {
"name": "decay_linear",
"formula": "decay_linear(series, window)",
"inputs": ["series"],
"windowed": True,
},
}
_PHASE1_OPERATOR_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"rank": rank,
"delta": delta,
"ts_mean": ts_mean,
"ts_std": ts_std,
"ts_rank": ts_rank,
"correlation": correlation,
"ts_min": ts_min,
"ts_max": ts_max,
"ts_sum": ts_sum,
"decay_linear": decay_linear,
}
def list_phase1_operators() -> tuple[str, ...]:
"""Return the deterministic Phase 1 operator names in stable order."""
return tuple(ALPHA158_PHASE1_OPERATOR_SPECS)
def evaluate_phase1_operator(
name: str,
series: pd.Series,
secondary: pd.Series | None = None,
*,
window: int | None = None,
) -> pd.Series:
"""Evaluate one of the ten Phase 1 operators with a validated contract.
``window`` is required for time-series operators and forbidden for the
cross-sectional ``rank`` operator. Binary ``correlation`` also requires
a same-index secondary series so that callers cannot silently introduce
alignment-dependent results.
"""
if name not in ALPHA158_PHASE1_OPERATOR_SPECS:
raise KeyError(f"operator {name!r} not registered")
is_windowed = bool(ALPHA158_PHASE1_OPERATOR_SPECS[name]["windowed"])
if is_windowed:
if isinstance(window, bool) or not isinstance(window, int) or window <= 0:
raise ValueError(f"window must be a positive integer for {name}")
if window > ALPHA158_PHASE1_MAX_WINDOW:
raise ValueError(
f"window exceeds maximum supported value {ALPHA158_PHASE1_MAX_WINDOW} for {name}"
)
if not is_windowed and window is not None:
raise ValueError(f"window is not supported for {name}")
if name == "correlation":
if secondary is None:
raise ValueError("secondary is required for correlation")
if not series.index.equals(secondary.index):
raise ValueError("secondary index must align with series")
return correlation(series, secondary, window) # type: ignore[arg-type]
if secondary is not None:
raise ValueError(f"secondary is not supported for {name}")
operator = _PHASE1_OPERATOR_FUNCTIONS[name]
if name == "rank":
return operator(series)
return operator(series, window)
# ── Phase 2 cumulative operator contract ──────────────────────────────
# Phase 2 is cumulative: downstream callers can upgrade to one dispatch
# surface covering every existing alpha158 building block, while Phase 1
# names, metadata, ordering, and evaluation remain unchanged.
ALPHA158_PHASE2_MAX_WINDOW = ALPHA158_PHASE1_MAX_WINDOW
ALPHA158_PHASE2_OPERATOR_SPECS: dict[str, dict[str, Any]] = {
name: {
**spec,
"parameters": ["window"] if bool(spec["windowed"]) else [],
}
for name, spec in ALPHA158_PHASE1_OPERATOR_SPECS.items()
}
ALPHA158_PHASE2_OPERATOR_SPECS.update(
{
"ts_argmin": {
"name": "ts_argmin",
"formula": "ts_argmin(series, window)",
"inputs": ["series"],
"parameters": ["window"],
"windowed": True,
},
"ts_argmax": {
"name": "ts_argmax",
"formula": "ts_argmax(series, window)",
"inputs": ["series"],
"parameters": ["window"],
"windowed": True,
},
"product": {
"name": "product",
"formula": "product(series, window)",
"inputs": ["series"],
"parameters": ["window"],
"windowed": True,
},
"returns": {
"name": "returns",
"formula": "returns(series)",
"inputs": ["series"],
"parameters": [],
"windowed": False,
},
"scale": {
"name": "scale",
"formula": "scale(series)",
"inputs": ["series"],
"parameters": [],
"windowed": False,
},
"signed_power": {
"name": "signed_power",
"formula": "signed_power(series, exponent)",
"inputs": ["series"],
"parameters": ["exponent"],
"windowed": False,
},
"stddev": {
"name": "stddev",
"formula": "stddev(series, window)",
"inputs": ["series"],
"parameters": ["window"],
"windowed": True,
},
"covariance": {
"name": "covariance",
"formula": "covariance(series, secondary, window)",
"inputs": ["series", "secondary"],
"parameters": ["window"],
"windowed": True,
},
"log": {
"name": "log",
"formula": "log(series)",
"inputs": ["series"],
"parameters": [],
"windowed": False,
},
"abs_series": {
"name": "abs_series",
"formula": "abs_series(series)",
"inputs": ["series"],
"parameters": [],
"windowed": False,
},
"sign": {
"name": "sign",
"formula": "sign(series)",
"inputs": ["series"],
"parameters": [],
"windowed": False,
},
"max_pair": {
"name": "max_pair",
"formula": "max_pair(series, secondary)",
"inputs": ["series", "secondary"],
"parameters": [],
"windowed": False,
},
"min_pair": {
"name": "min_pair",
"formula": "min_pair(series, secondary)",
"inputs": ["series", "secondary"],
"parameters": [],
"windowed": False,
},
"indneutralize": {
"name": "indneutralize",
"formula": "indneutralize(series, groups)",
"inputs": ["series", "groups"],
"parameters": [],
"windowed": False,
},
}
)
_PHASE2_OPERATOR_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
**_PHASE1_OPERATOR_FUNCTIONS,
"ts_argmin": ts_argmin,
"ts_argmax": ts_argmax,
"product": product,
"returns": returns,
"scale": scale,
"signed_power": signed_power,
"stddev": stddev,
"covariance": covariance,
"log": log,
"abs_series": abs_series,
"sign": sign,
"max_pair": max_pair,
"min_pair": min_pair,
"indneutralize": indneutralize,
}
_PHASE2_WINDOWED_OPERATORS = frozenset(
name for name, spec in ALPHA158_PHASE2_OPERATOR_SPECS.items() if bool(spec["windowed"])
)
_PHASE2_BINARY_OPERATORS = frozenset({"correlation", "covariance", "max_pair", "min_pair"})
def list_phase2_operators() -> tuple[str, ...]:
"""Return all Phase 2 operator names in stable cumulative order."""
return tuple(ALPHA158_PHASE2_OPERATOR_SPECS)
def _validate_phase2_window(name: str, window: int | None) -> int:
if isinstance(window, bool) or not isinstance(window, int) or window <= 0:
raise ValueError(f"window must be a positive integer for {name}")
if window > ALPHA158_PHASE2_MAX_WINDOW:
raise ValueError(
f"window exceeds maximum supported value {ALPHA158_PHASE2_MAX_WINDOW} for {name}"
)
return window
def evaluate_phase2_operator(
name: str,
series: pd.Series,
secondary: pd.Series | None = None,
*,
window: int | None = None,
exponent: float | None = None,
groups: pd.Series | None = None,
) -> pd.Series:
"""Evaluate any existing alpha158 building block through a strict contract.
Phase 2 rejects implicit alignment, missing required arguments, unused
arguments, unbounded windows, and non-finite exponents before dispatch.
"""
if name not in ALPHA158_PHASE2_OPERATOR_SPECS:
raise KeyError(f"operator {name!r} not registered")
if not isinstance(series, pd.Series):
raise TypeError("series must be a pandas Series")
validated_window: int | None = None
if name in _PHASE2_WINDOWED_OPERATORS:
validated_window = _validate_phase2_window(name, window)
elif window is not None:
raise ValueError(f"window is not supported for {name}")
if name in _PHASE2_BINARY_OPERATORS:
if secondary is None:
raise ValueError(f"secondary is required for {name}")
if not isinstance(secondary, pd.Series):
raise TypeError("secondary must be a pandas Series")
if not series.index.equals(secondary.index):
raise ValueError("secondary index must align with series")
elif secondary is not None:
raise ValueError(f"secondary is not supported for {name}")
validated_exponent: float | None = None
if name == "signed_power":
if (
isinstance(exponent, bool)
or not isinstance(exponent, (int, float))
or not np.isfinite(exponent)
):
raise ValueError("exponent must be a finite number for signed_power")
validated_exponent = float(exponent)
elif exponent is not None:
raise ValueError(f"exponent is not supported for {name}")
if name == "indneutralize":
if groups is None:
raise ValueError("groups is required for indneutralize")
if not isinstance(groups, pd.Series):
raise TypeError("groups must be a pandas Series")
if not series.index.equals(groups.index):
raise ValueError("groups index must align with series")
elif groups is not None:
raise ValueError(f"groups is not supported for {name}")
operator = _PHASE2_OPERATOR_FUNCTIONS[name]
if name == "signed_power":
return operator(series, validated_exponent)
if name == "indneutralize":
return operator(series, groups)
if name in {"correlation", "covariance"}:
return operator(series, secondary, validated_window)
if name in {"max_pair", "min_pair"}:
return operator(series, secondary)
if validated_window is not None:
return operator(series, validated_window)
return operator(series)
# ── 组合算子(alpha158 公式样本) ─────────────────────────
@@ -2730,6 +3093,541 @@ def parse_alpha_formula(formula_str: str) -> dict[str, Any]:
return parsed
# ── Phase 3 formula contract: frozen alpha001-alpha050 surface ──────────────
# Formula functions remain the implementation source of truth. This contract
# freezes their callable surface separately from formula dependencies so that
# historical compatibility-only arguments remain explicit without rewriting
# formulas or changing direct-call APIs.
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION = "1.0.0"
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 = (
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
)
_PHASE3_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"alpha_001": alpha_001,
"alpha_002": alpha_002,
"alpha_003": alpha_003,
"alpha_004": alpha_004,
"alpha_005": alpha_005,
"alpha_006": alpha_006,
"alpha_007": alpha_007,
"alpha_008": alpha_008,
"alpha_009": alpha_009,
"alpha_010": alpha_010,
"alpha_011": alpha_011,
"alpha_012": alpha_012,
"alpha_013": alpha_013,
"alpha_014": alpha_014,
"alpha_015": alpha_015,
"alpha_016": alpha_016,
"alpha_017": alpha_017,
"alpha_018": alpha_018,
"alpha_019": alpha_019,
"alpha_020": alpha_020,
"alpha_021": alpha_021,
"alpha_022": alpha_022,
"alpha_023": alpha_023,
"alpha_024": alpha_024,
"alpha_025": alpha_025,
"alpha_026": alpha_026,
"alpha_027": alpha_027,
"alpha_028": alpha_028,
"alpha_029": alpha_029,
"alpha_030": alpha_030,
"alpha_031": alpha_031,
"alpha_032": alpha_032,
"alpha_033": alpha_033,
"alpha_034": alpha_034,
"alpha_035": alpha_035,
"alpha_036": alpha_036,
"alpha_037": alpha_037,
"alpha_038": alpha_038,
"alpha_039": alpha_039,
"alpha_040": alpha_040,
"alpha_041": alpha_041,
"alpha_042": alpha_042,
"alpha_043": alpha_043,
"alpha_044": alpha_044,
"alpha_045": alpha_045,
"alpha_046": alpha_046,
"alpha_047": alpha_047,
"alpha_048": alpha_048,
"alpha_049": alpha_049,
"alpha_050": alpha_050,
}
_PHASE3_FORMULA_INPUT_OVERRIDES: dict[str, list[str]] = {
"alpha_011": ["close", "high", "low"],
"alpha_035": ["volume"],
"alpha_036": ["close"],
"alpha_040": ["high", "low"],
"alpha_042": ["close"],
"alpha_043": ["volume"],
}
_PHASE3_INPUT_CATEGORIES = {
1: "single",
2: "pair",
3: "triple",
4: "quadruple",
}
def _phase3_call_inputs(function: Callable[..., pd.Series]) -> list[str]:
import inspect
parameters = list(inspect.signature(function).parameters.values())
if any(
parameter.kind is not inspect.Parameter.POSITIONAL_OR_KEYWORD
or parameter.default is not inspect.Parameter.empty
for parameter in parameters
):
raise RuntimeError(f"unsupported formula signature for {function.__name__}")
return ["open" if parameter.name == "open_" else parameter.name for parameter in parameters]
def _phase3_string_list(meta: dict[str, Any], field: str, alpha_id: str) -> list[str]:
value = meta[field]
if not isinstance(value, list) or not all(isinstance(item, str) for item in value):
raise RuntimeError(f"{field} must be a list of strings for {alpha_id}")
return list(value)
def _build_phase3_formula_specs() -> dict[str, dict[str, Any]]:
specs: dict[str, dict[str, Any]] = {}
for alpha_id, function in _PHASE3_FORMULA_FUNCTIONS.items():
meta = ALPHA158_REGISTRY[alpha_id]
call_inputs = _phase3_call_inputs(function)
formula_inputs = _PHASE3_FORMULA_INPUT_OVERRIDES.get(alpha_id, call_inputs)
input_category = _PHASE3_INPUT_CATEGORIES.get(len(call_inputs))
if input_category is None:
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
specs[alpha_id] = {
"name": alpha_id,
"contract_version": ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
"formula": meta["formula"],
"category": meta["category"],
"complexity": meta["complexity"],
"parameters": _phase3_string_list(meta, "params", alpha_id),
"description": meta["description"],
"references": _phase3_string_list(meta, "references", alpha_id),
"call_inputs": list(call_inputs),
"formula_inputs": list(formula_inputs),
"input_category": input_category,
}
return specs
def _freeze_phase3_formula_specs(
specs: dict[str, dict[str, Any]],
) -> Mapping[str, Mapping[str, Any]]:
frozen_specs: dict[str, Mapping[str, Any]] = {}
for alpha_id, spec in specs.items():
frozen_specs[alpha_id] = MappingProxyType(
{field: tuple(value) if isinstance(value, list) else value for field, value in spec.items()}
)
return MappingProxyType(frozen_specs)
ALPHA158_PHASE3_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
_freeze_phase3_formula_specs(_build_phase3_formula_specs())
)
def list_phase3_formulas() -> tuple[str, ...]:
"""Return the frozen alpha001-alpha050 formula IDs in stable order."""
return tuple(ALPHA158_PHASE3_FORMULA_SPECS)
def evaluate_phase3_formula(name: str, **inputs: pd.Series) -> pd.Series:
"""Evaluate a Phase 3 formula with an exact, alignment-safe input contract."""
if name not in ALPHA158_PHASE3_FORMULA_SPECS:
raise KeyError(f"formula {name!r} not registered")
spec = ALPHA158_PHASE3_FORMULA_SPECS[name]
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
missing_inputs = [field for field in required_inputs if field not in inputs]
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
if missing_inputs or unexpected_inputs:
details: list[str] = []
if missing_inputs:
details.append(f"missing inputs {missing_inputs}")
if unexpected_inputs:
details.append(f"unexpected inputs {unexpected_inputs}")
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
for field in required_inputs:
if not isinstance(inputs[field], pd.Series):
raise TypeError(f"{field} must be a pandas Series")
primary_field = required_inputs[0]
primary = inputs[primary_field]
for field in required_inputs[1:]:
if not primary.index.equals(inputs[field].index):
raise ValueError(f"{field} index must align with {primary_field}")
function = _PHASE3_FORMULA_FUNCTIONS[name]
return function(*(inputs[field] for field in required_inputs))
# ── Phase 4 formula contract: frozen alpha051-alpha100 surface ──────────────
# Phase 4 extends the versioned formula contract without mutating the Phase 3
# catalogue, digest, dispatch surface, or the existing formula functions.
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION = "1.0.0"
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 = (
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
)
_PHASE4_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"alpha_051": alpha_051,
"alpha_052": alpha_052,
"alpha_053": alpha_053,
"alpha_054": alpha_054,
"alpha_055": alpha_055,
"alpha_056": alpha_056,
"alpha_057": alpha_057,
"alpha_058": alpha_058,
"alpha_059": alpha_059,
"alpha_060": alpha_060,
"alpha_061": alpha_061,
"alpha_062": alpha_062,
"alpha_063": alpha_063,
"alpha_064": alpha_064,
"alpha_065": alpha_065,
"alpha_066": alpha_066,
"alpha_067": alpha_067,
"alpha_068": alpha_068,
"alpha_069": alpha_069,
"alpha_070": alpha_070,
"alpha_071": alpha_071,
"alpha_072": alpha_072,
"alpha_073": alpha_073,
"alpha_074": alpha_074,
"alpha_075": alpha_075,
"alpha_076": alpha_076,
"alpha_077": alpha_077,
"alpha_078": alpha_078,
"alpha_079": alpha_079,
"alpha_080": alpha_080,
"alpha_081": alpha_081,
"alpha_082": alpha_082,
"alpha_083": alpha_083,
"alpha_084": alpha_084,
"alpha_085": alpha_085,
"alpha_086": alpha_086,
"alpha_087": alpha_087,
"alpha_088": alpha_088,
"alpha_089": alpha_089,
"alpha_090": alpha_090,
"alpha_091": alpha_091,
"alpha_092": alpha_092,
"alpha_093": alpha_093,
"alpha_094": alpha_094,
"alpha_095": alpha_095,
"alpha_096": alpha_096,
"alpha_097": alpha_097,
"alpha_098": alpha_098,
"alpha_099": alpha_099,
"alpha_100": alpha_100,
}
_PHASE4_INPUT_CATEGORIES = {
1: "single",
2: "pair",
3: "triple",
4: "quadruple",
5: "quintuple",
}
def _build_phase4_formula_specs() -> dict[str, dict[str, Any]]:
specs: dict[str, dict[str, Any]] = {}
for alpha_id, function in _PHASE4_FORMULA_FUNCTIONS.items():
meta = ALPHA158_REGISTRY[alpha_id]
call_inputs = _phase3_call_inputs(function)
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
if input_category is None:
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
specs[alpha_id] = {
"name": alpha_id,
"contract_version": ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
"formula": meta["formula"],
"category": meta["category"],
"complexity": meta["complexity"],
"parameters": _phase3_string_list(meta, "params", alpha_id),
"description": meta["description"],
"references": _phase3_string_list(meta, "references", alpha_id),
"call_inputs": list(call_inputs),
"formula_inputs": formula_inputs,
"input_category": input_category,
}
return specs
ALPHA158_PHASE4_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
_freeze_phase3_formula_specs(_build_phase4_formula_specs())
)
def list_phase4_formulas() -> tuple[str, ...]:
"""Return the frozen alpha051-alpha100 formula IDs in stable order."""
return tuple(ALPHA158_PHASE4_FORMULA_SPECS)
def evaluate_phase4_formula(name: str, **inputs: pd.Series) -> pd.Series:
"""Evaluate a Phase 4 formula with an exact, alignment-safe input contract."""
if name not in ALPHA158_PHASE4_FORMULA_SPECS:
raise KeyError(f"formula {name!r} not registered")
spec = ALPHA158_PHASE4_FORMULA_SPECS[name]
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
missing_inputs = [field for field in required_inputs if field not in inputs]
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
if missing_inputs or unexpected_inputs:
details: list[str] = []
if missing_inputs:
details.append(f"missing inputs {missing_inputs}")
if unexpected_inputs:
details.append(f"unexpected inputs {unexpected_inputs}")
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
for field in required_inputs:
if not isinstance(inputs[field], pd.Series):
raise TypeError(f"{field} must be a pandas Series")
primary_field = required_inputs[0]
primary = inputs[primary_field]
for field in required_inputs[1:]:
if not primary.index.equals(inputs[field].index):
raise ValueError(f"{field} index must align with {primary_field}")
function = _PHASE4_FORMULA_FUNCTIONS[name]
return function(*(inputs[field] for field in required_inputs))
# ── Phase 5 formula contract: frozen alpha101-alpha150 surface ──────────────
# Phase 5 extends the versioned formula contract without mutating any earlier
# catalogue, digest, dispatch surface, or existing formula implementation.
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION = "1.0.0"
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 = (
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
)
_PHASE5_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"alpha_101": alpha_101,
"alpha_102": alpha_102,
"alpha_103": alpha_103,
"alpha_104": alpha_104,
"alpha_105": alpha_105,
"alpha_106": alpha_106,
"alpha_107": alpha_107,
"alpha_108": alpha_108,
"alpha_109": alpha_109,
"alpha_110": alpha_110,
"alpha_111": alpha_111,
"alpha_112": alpha_112,
"alpha_113": alpha_113,
"alpha_114": alpha_114,
"alpha_115": alpha_115,
"alpha_116": alpha_116,
"alpha_117": alpha_117,
"alpha_118": alpha_118,
"alpha_119": alpha_119,
"alpha_120": alpha_120,
"alpha_121": alpha_121,
"alpha_122": alpha_122,
"alpha_123": alpha_123,
"alpha_124": alpha_124,
"alpha_125": alpha_125,
"alpha_126": alpha_126,
"alpha_127": alpha_127,
"alpha_128": alpha_128,
"alpha_129": alpha_129,
"alpha_130": alpha_130,
"alpha_131": alpha_131,
"alpha_132": alpha_132,
"alpha_133": alpha_133,
"alpha_134": alpha_134,
"alpha_135": alpha_135,
"alpha_136": alpha_136,
"alpha_137": alpha_137,
"alpha_138": alpha_138,
"alpha_139": alpha_139,
"alpha_140": alpha_140,
"alpha_141": alpha_141,
"alpha_142": alpha_142,
"alpha_143": alpha_143,
"alpha_144": alpha_144,
"alpha_145": alpha_145,
"alpha_146": alpha_146,
"alpha_147": alpha_147,
"alpha_148": alpha_148,
"alpha_149": alpha_149,
"alpha_150": alpha_150,
}
def _build_phase5_formula_specs() -> dict[str, dict[str, Any]]:
specs: dict[str, dict[str, Any]] = {}
for alpha_id, function in _PHASE5_FORMULA_FUNCTIONS.items():
meta = ALPHA158_REGISTRY[alpha_id]
call_inputs = _phase3_call_inputs(function)
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
if input_category is None:
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
specs[alpha_id] = {
"name": alpha_id,
"contract_version": ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
"formula": meta["formula"],
"category": meta["category"],
"complexity": meta["complexity"],
"parameters": _phase3_string_list(meta, "params", alpha_id),
"description": meta["description"],
"references": _phase3_string_list(meta, "references", alpha_id),
"call_inputs": list(call_inputs),
"formula_inputs": formula_inputs,
"input_category": input_category,
}
return specs
ALPHA158_PHASE5_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
_freeze_phase3_formula_specs(_build_phase5_formula_specs())
)
def list_phase5_formulas() -> tuple[str, ...]:
"""Return the frozen alpha101-alpha150 formula IDs in stable order."""
return tuple(ALPHA158_PHASE5_FORMULA_SPECS)
def evaluate_phase5_formula(name: str, **inputs: pd.Series) -> pd.Series:
"""Evaluate a Phase 5 formula with an exact, alignment-safe input contract."""
if name not in ALPHA158_PHASE5_FORMULA_SPECS:
raise KeyError(f"formula {name!r} not registered")
spec = ALPHA158_PHASE5_FORMULA_SPECS[name]
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
missing_inputs = [field for field in required_inputs if field not in inputs]
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
if missing_inputs or unexpected_inputs:
details: list[str] = []
if missing_inputs:
details.append(f"missing inputs {missing_inputs}")
if unexpected_inputs:
details.append(f"unexpected inputs {unexpected_inputs}")
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
for field in required_inputs:
if not isinstance(inputs[field], pd.Series):
raise TypeError(f"{field} must be a pandas Series")
primary_field = required_inputs[0]
primary = inputs[primary_field]
for field in required_inputs[1:]:
if len(inputs[field]) != len(primary):
raise ValueError(f"{field} length must match {primary_field}")
if not primary.index.equals(inputs[field].index):
raise ValueError(f"{field} index must align with {primary_field}")
function = _PHASE5_FORMULA_FUNCTIONS[name]
return function(*(inputs[field] for field in required_inputs))
# ── Phase 6 formula contract: frozen alpha151-alpha158 surface ──────────────
# Phase 6 completes the versioned formula contract without mutating any
# earlier catalogue, digest, dispatch surface, or formula implementation.
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION = "1.0.0"
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 = (
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
)
_PHASE6_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"alpha_151": alpha_151,
"alpha_152": alpha_152,
"alpha_153": alpha_153,
"alpha_154": alpha_154,
"alpha_155": alpha_155,
"alpha_156": alpha_156,
"alpha_157": alpha_157,
"alpha_158": alpha_158,
}
def _build_phase6_formula_specs() -> dict[str, dict[str, Any]]:
specs: dict[str, dict[str, Any]] = {}
for alpha_id, function in _PHASE6_FORMULA_FUNCTIONS.items():
meta = ALPHA158_REGISTRY[alpha_id]
call_inputs = _phase3_call_inputs(function)
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
if input_category is None:
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
specs[alpha_id] = {
"name": alpha_id,
"contract_version": ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
"formula": meta["formula"],
"category": meta["category"],
"complexity": meta["complexity"],
"parameters": _phase3_string_list(meta, "params", alpha_id),
"description": meta["description"],
"references": _phase3_string_list(meta, "references", alpha_id),
"call_inputs": list(call_inputs),
"formula_inputs": formula_inputs,
"input_category": input_category,
}
return specs
ALPHA158_PHASE6_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
_freeze_phase3_formula_specs(_build_phase6_formula_specs())
)
def list_phase6_formulas() -> tuple[str, ...]:
"""Return the frozen alpha151-alpha158 formula IDs in stable order."""
return tuple(ALPHA158_PHASE6_FORMULA_SPECS)
def evaluate_phase6_formula(name: str, **inputs: pd.Series) -> pd.Series:
"""Evaluate a Phase 6 formula with an exact, alignment-safe input contract."""
if name not in ALPHA158_PHASE6_FORMULA_SPECS:
raise KeyError(f"formula {name!r} not registered")
spec = ALPHA158_PHASE6_FORMULA_SPECS[name]
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
missing_inputs = [field for field in required_inputs if field not in inputs]
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
if missing_inputs or unexpected_inputs:
details: list[str] = []
if missing_inputs:
details.append(f"missing inputs {missing_inputs}")
if unexpected_inputs:
details.append(f"unexpected inputs {unexpected_inputs}")
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
for field in required_inputs:
if not isinstance(inputs[field], pd.Series):
raise TypeError(f"{field} must be a pandas Series")
primary_field = required_inputs[0]
primary = inputs[primary_field]
for field in required_inputs[1:]:
if len(inputs[field]) != len(primary):
raise ValueError(f"{field} length must match {primary_field}")
if not primary.index.equals(inputs[field].index):
raise ValueError(f"{field} index must align with {primary_field}")
function = _PHASE6_FORMULA_FUNCTIONS[name]
return function(*(inputs[field] for field in required_inputs))
__all__ = [
"rank",
"delta",
@@ -2755,6 +3653,34 @@ __all__ = [
"max_pair",
"min_pair",
"indneutralize",
"ALPHA158_PHASE1_MAX_WINDOW",
"ALPHA158_PHASE1_OPERATOR_SPECS",
"list_phase1_operators",
"evaluate_phase1_operator",
"ALPHA158_PHASE2_MAX_WINDOW",
"ALPHA158_PHASE2_OPERATOR_SPECS",
"list_phase2_operators",
"evaluate_phase2_operator",
"ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE3_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE3_FORMULA_SPECS",
"list_phase3_formulas",
"evaluate_phase3_formula",
"ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE4_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE4_FORMULA_SPECS",
"list_phase4_formulas",
"evaluate_phase4_formula",
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE5_FORMULA_SPECS",
"list_phase5_formulas",
"evaluate_phase5_formula",
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE6_FORMULA_SPECS",
"list_phase6_formulas",
"evaluate_phase6_formula",
"alpha_001",
"alpha_002",
"alpha_003",
File diff suppressed because it is too large Load Diff
+141
View File
@@ -0,0 +1,141 @@
"""Post-execution daily return attribution derived from the portfolio ledger.
The ledger is the source of truth: previous-close holdings explain overnight
PnL, current-close holdings explain intraday PnL, and actual execution costs
remain a separate contribution. Target weights and factor scores are not
accepted here because they are intentions rather than realized positions.
"""
from __future__ import annotations
import math
from dataclasses import dataclass
import pandas as pd
from quant_engine.execution import ExecutionSimulationResult
__all__ = ["DailyReturnAttribution", "compute_daily_return_attribution"]
@dataclass(frozen=True, slots=True, eq=False)
class DailyReturnAttribution:
"""Auditable decomposition of each net portfolio return."""
overnight: pd.DataFrame
intraday: pd.DataFrame
transaction_cost: pd.Series
residual: pd.Series
total_return: pd.Series
@property
def asset_contributions(self) -> pd.DataFrame:
"""Return the combined overnight and intraday contribution by asset."""
return self.overnight + self.intraday
@property
def explained_return(self) -> pd.Series:
"""Return asset contributions plus execution costs, before residual."""
explained = self.asset_contributions.sum(axis=1) + self.transaction_cost
return explained.rename("explained_return")
def _validate_prices(
execution: ExecutionSimulationResult,
execution_prices: pd.DataFrame,
valuation_prices: pd.DataFrame,
) -> pd.DatetimeIndex:
if not isinstance(execution_prices, pd.DataFrame):
raise TypeError("execution_prices must be a pandas DataFrame")
if not isinstance(valuation_prices, pd.DataFrame):
raise TypeError("valuation_prices must be a pandas DataFrame")
if not isinstance(execution_prices.index, pd.DatetimeIndex):
raise TypeError("execution_prices must use a DatetimeIndex")
if not execution_prices.index.equals(valuation_prices.index):
raise ValueError("execution and valuation prices must use matching trading calendars")
if not execution_prices.columns.equals(valuation_prices.columns):
raise ValueError("execution and valuation prices must use matching asset labels")
ledger_index = pd.DatetimeIndex(pd.Timestamp(position.date) for position in execution.positions)
if not ledger_index.equals(execution_prices.index):
raise ValueError("ledger and price histories must use matching trading calendars")
if len(execution.positions) != len(execution.daily_executions):
raise ValueError("ledger positions and executions must have matching lengths")
return execution_prices.index.copy()
def _price_for_held_asset(
prices: pd.DataFrame,
date: pd.Timestamp,
asset: str,
stage: str,
) -> float:
if asset not in prices.columns:
raise ValueError(f"missing {stage} price for held asset {asset} on {date}")
price = float(prices.at[date, asset])
if not math.isfinite(price) or price <= 0:
raise ValueError(f"invalid {stage} price for held asset {asset} on {date}")
return price
def compute_daily_return_attribution(
execution: ExecutionSimulationResult,
execution_prices: pd.DataFrame,
valuation_prices: pd.DataFrame,
) -> DailyReturnAttribution:
"""Decompose net daily returns using realized pre/post-execution holdings.
For each session, previous-close shares earn the move from the previous
close to the current execution price; current-close shares earn the move
from execution price to current close. Actual commissions, stamp tax and
slippage are divided by the same previous NAV denominator. ``residual``
exposes any failure of those components to close to the ledger return.
"""
index = _validate_prices(execution, execution_prices, valuation_prices)
columns = execution_prices.columns.copy()
overnight = pd.DataFrame(0.0, index=index.copy(), columns=columns)
intraday = pd.DataFrame(0.0, index=index.copy(), columns=columns)
cost = pd.Series(0.0, index=index.copy(), name="transaction_cost")
previous_holdings: dict[str, float] = {}
previous_nav = execution.initial_cash
for row_number, (date, position, daily) in enumerate(
zip(index, execution.positions, execution.daily_executions, strict=True)
):
if previous_nav <= 0 or not math.isfinite(previous_nav):
raise ValueError(f"previous portfolio value must be positive and finite on {date}")
for asset, shares in previous_holdings.items():
execution_price = _price_for_held_asset(
execution_prices, date, asset, "execution"
)
previous_close = _price_for_held_asset(
valuation_prices, index[row_number - 1], asset, "previous valuation"
)
overnight.at[date, asset] = shares * (execution_price - previous_close) / previous_nav
for asset, shares in position.holdings.items():
execution_price = _price_for_held_asset(
execution_prices, date, asset, "execution"
)
close_price = _price_for_held_asset(valuation_prices, date, asset, "valuation")
intraday.at[date, asset] = shares * (close_price - execution_price) / previous_nav
cost.at[date] = -sum(item.total_cost for item in daily.executions) / previous_nav
previous_holdings = position.holdings
previous_nav = position.portfolio_value
total_return = pd.Series(
execution.daily_returns.to_numpy(copy=True),
index=index.copy(),
name="total_return",
)
explained = (overnight + intraday).sum(axis=1) + cost
residual = (total_return - explained).rename("residual")
return DailyReturnAttribution(
overnight=overnight,
intraday=intraday,
transaction_cost=cost,
residual=residual,
total_return=total_return,
)
+60 -12
View File
@@ -13,29 +13,28 @@
```python
from quant_engine.backtest import (
compute_nav_from_weights, # 调仓表 → 净值
rebalance_table, # 周期性再平衡
compare_to_benchmark, # 策略 vs 基准
rebalance_periodic, # 周期性再平衡
run_weight_backtest, # 权重 → 统一结果对象
weights_to_long_short, # 多空组合
)
# 1. 调仓表 → 净值
nav = compute_nav_from_weights(
# 调仓表 → 净值、收益、绩效与基准报告
rebalance_table = rebalance_periodic(target_weights, rebalance_dates, returns.index)
result = run_weight_backtest(
weights=rebalance_table, # 每周/每月调仓
stock_returns=returns, # 个股日收益
initial_capital=1.0,
benchmark_nav=benchmark_nav,
)
# 2. 跟基准比
result = compare_to_benchmark(nav, benchmark_nav)
print(result.summary())
print(result.stats())
print(result.benchmark_report())
```
"""
from __future__ import annotations
from pathlib import Path
from collections.abc import Mapping, Sequence
from dataclasses import dataclass
import numpy as np
import pandas as pd
@@ -46,6 +45,26 @@ from quant_engine.metrics import summary as metrics_summary
logger = get_logger(__name__)
@dataclass(frozen=True, slots=True, eq=False)
class BacktestResult:
"""一次权重回测的稳定结果快照。"""
nav: pd.Series
returns: pd.Series
weights: pd.DataFrame
benchmark_nav: pd.Series | None = None
def stats(self, rf: float = 0.0) -> Mapping[str, float]:
"""返回标准绩效指标。"""
return metrics_summary(self.returns, rf)
def benchmark_report(self, rf: float = 0.0) -> pd.DataFrame:
"""返回策略与基准的对比报告。"""
if self.benchmark_nav is None:
raise ValueError("benchmark_nav is required for benchmark comparison")
return compare_to_benchmark(self.nav, self.benchmark_nav, rf)
# ── 调仓表 → 净值 ──────────────────────────────────────
@@ -57,8 +76,9 @@ def compute_nav_from_weights(
) -> pd.Series:
"""从调仓表(日期 × 股票权重)+ 个股日收益 → 净值曲线。
假设:在调仓日之间权重不变(**前向填充**)。
调仓日的权重 = `weights.loc[rebalance_date]`。
假设:输入是该收益测量区间开始前已经生效的持仓权重,并在调仓日之间
保持不变(**前向填充**)。本函数不会把信号日自动解释为执行日;因子分数
应先经交易日历调度和实际执行时点处理,避免把同一时点未知的收益计入。
Args:
weights: 调仓日 × 股票代码 的权重 DataFrame(**0~1**,行和 ≤ 1)
@@ -113,6 +133,29 @@ def compute_returns_from_nav(nav: pd.Series) -> pd.Series:
return nav.pct_change().fillna(0.0)
def run_weight_backtest(
weights: pd.DataFrame,
stock_returns: pd.DataFrame,
initial_capital: float = 1.0,
tc_rate: float = 0.0,
benchmark_nav: pd.Series | None = None,
) -> BacktestResult:
"""执行权重回测并返回隔离于调用方输入的结果快照。"""
weights_snapshot = weights.copy(deep=True)
nav = compute_nav_from_weights(
weights=weights_snapshot,
stock_returns=stock_returns,
initial_capital=initial_capital,
tc_rate=tc_rate,
)
return BacktestResult(
nav=nav,
returns=compute_returns_from_nav(nav),
weights=weights_snapshot,
benchmark_nav=None if benchmark_nav is None else benchmark_nav.copy(deep=True),
)
# ── 调仓工具 ──────────────────────────────────────
@@ -131,6 +174,9 @@ def rebalance_periodic(
Returns:
调仓表 DataFrame(all_dates × 股票代码)
"""
if all_dates.empty:
return pd.DataFrame(index=all_dates, columns=target_weights.index, dtype=float)
table = pd.DataFrame(0.0, index=all_dates, columns=target_weights.index)
for date in rebalance_dates:
if date not in all_dates:
@@ -201,6 +247,8 @@ def compare_to_benchmark(
"""
# 对齐 index
common = strategy_nav.index.intersection(benchmark_nav.index)
if common.empty:
raise ValueError("strategy and benchmark must have overlapping dates")
s = strategy_nav.loc[common]
b = benchmark_nav.loc[common]
+210 -7
View File
@@ -6,22 +6,27 @@
- execution.py 需要**宽表**(date × stock_code)prices / volumes
- Tushare 字段命名:`ts_code / vol(手) / amount(千元) / pct_chg`,且**无 vwap 字段**
本模块提供 6 个纯函数,让新模块直接吃 qtdb_pro 真实数据:
本模块提供可组合的数据适配函数,让新模块直接吃 qtdb_pro 真实数据:
1. `long_to_wide()` — 长表 → 宽表(date × stock_code)
2. `wide_to_long()` — 宽表 → 长表
3. `rename_tushare_columns()` — 列名映射(ts_code→stock_code, vol→volume 等)
4. `add_vwap_proxy()` — vwap 代理(Tushare 无 vwap 字段)
5. `apply_adj_factor()` — 复权(hq_daily × hq_adj_factor 前复权)
6. `prepare_stock_series()` — 单股提取(alpha_factors 输入)
7. `prepare_execution_inputs()` — execution 输入(prices + volumes 宽表)
8. `load_qtdb_daily()` — 便捷加载(qtdb_pro.hq_daily + 可选复权)
7. `prepare_asset_return_snapshot()` — 带稳定 lineage 的资产日收益
8. `prepare_execution_inputs()` — execution 输入(prices + volumes 宽表)
9. `load_qtdb_daily()` — 便捷加载(qtdb_pro.hq_daily + 可选复权)
全部纯 pandas/numpy,零新依赖,mypy strict 兼容。
"""
from __future__ import annotations
import hashlib
import json
from collections.abc import Mapping, Sequence
from dataclasses import dataclass
from datetime import date
from typing import Any
import numpy as np
@@ -32,12 +37,14 @@ from quant_engine.logging import get_logger
logger = get_logger(__name__)
__all__ = [
"AssetReturnSnapshot",
"long_to_wide",
"wide_to_long",
"rename_tushare_columns",
"add_vwap_proxy",
"apply_adj_factor",
"prepare_stock_series",
"prepare_asset_return_snapshot",
"prepare_execution_inputs",
"load_qtdb_daily",
]
@@ -57,6 +64,107 @@ TUSHARE_RENAME: dict[str, str] = {
}
@dataclass(frozen=True, slots=True, init=False, eq=False)
class AssetReturnSnapshot:
"""Immutable-by-interface daily return matrix with reproducible lineage."""
data_snapshot_id: str
source: str
source_snapshot_id: str
price_field: str
adjustment: str
return_method: str
start_date: date
end_date: date
sessions: int
assets: tuple[str, ...]
_returns: pd.DataFrame
def __init__(
self,
*,
data_snapshot_id: str,
source: str,
source_snapshot_id: str,
price_field: str,
adjustment: str,
return_method: str,
start_date: date,
end_date: date,
assets: tuple[str, ...],
returns: pd.DataFrame,
) -> None:
for value, name in (
(data_snapshot_id, "data_snapshot_id"),
(source, "source"),
(source_snapshot_id, "source_snapshot_id"),
(price_field, "price_field"),
(adjustment, "adjustment"),
(return_method, "return_method"),
):
if not isinstance(value, str) or not value.strip():
raise ValueError(f"{name} must be non-empty")
if returns.empty or not isinstance(returns.index, pd.DatetimeIndex):
raise ValueError("returns must contain a DatetimeIndex and at least one session")
if tuple(returns.columns) != assets:
raise ValueError("assets must match returns columns")
if start_date > end_date:
raise ValueError("start_date must not be after end_date")
object.__setattr__(self, "data_snapshot_id", data_snapshot_id.strip())
object.__setattr__(self, "source", source.strip())
object.__setattr__(self, "source_snapshot_id", source_snapshot_id.strip())
object.__setattr__(self, "price_field", price_field.strip())
object.__setattr__(self, "adjustment", adjustment.strip())
object.__setattr__(self, "return_method", return_method.strip())
object.__setattr__(self, "start_date", start_date)
object.__setattr__(self, "end_date", end_date)
object.__setattr__(self, "sessions", len(returns))
object.__setattr__(self, "assets", assets)
object.__setattr__(self, "_returns", returns.copy(deep=True))
@property
def returns(self) -> pd.DataFrame:
"""Return an isolated copy so callers cannot mutate the snapshot."""
return self._returns.copy(deep=True)
def _non_empty(value: str, name: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ValueError(f"{name} must be non-empty")
return value.strip()
def _asset_return_snapshot_id(
prices: pd.DataFrame,
*,
source: str,
source_snapshot_id: str,
price_field: str,
adjustment: str,
) -> str:
values = prices.to_numpy(dtype=float, copy=True)
missing = np.isnan(values)
normalized = np.where(missing, 0.0, values).astype("<f8", copy=False)
metadata = {
"adjustment": adjustment,
"assets": [str(asset) for asset in prices.columns],
"price_field": price_field,
"return_method": "simple",
"schema": "asset-returns-v1",
"sessions": [timestamp.date().isoformat() for timestamp in prices.index],
"shape": list(values.shape),
"source": source,
"source_snapshot_id": source_snapshot_id,
}
digest = hashlib.sha256(
json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode("utf-8")
)
digest.update(missing.astype(np.uint8, copy=False).tobytes(order="C"))
digest.update(normalized.tobytes(order="C"))
return f"asset-returns-v1:{digest.hexdigest()}"
def long_to_wide(
df: pd.DataFrame,
value_col: str = "close",
@@ -283,10 +391,104 @@ def prepare_stock_series(
return series_map
def prepare_asset_return_snapshot(
df: pd.DataFrame,
*,
source: str,
source_snapshot_id: str,
price_col: str = "close",
adjustment: str = "none",
stock_col: str = "stock_code",
date_col: str = "trade_date",
) -> AssetReturnSnapshot:
"""Build deterministic simple daily returns from a long market-price table.
``source_snapshot_id`` must identify the upstream ingestion snapshot. The
resulting ID additionally fingerprints canonical price values and their
missing mask, so changed contents cannot retain the same downstream identity.
Missing prices are never forward-filled.
"""
normalized_source = _non_empty(source, "source")
normalized_source_snapshot_id = _non_empty(
source_snapshot_id,
"source_snapshot_id",
)
normalized_price_col = _non_empty(price_col, "price_col")
normalized_adjustment = _non_empty(adjustment, "adjustment")
if not isinstance(df, pd.DataFrame):
raise TypeError("df must be a pandas DataFrame")
if df.empty:
raise ValueError("df must contain market prices")
required = {date_col, stock_col, normalized_price_col}
missing_columns = sorted(required.difference(df.columns))
if missing_columns:
raise ValueError(f"prepare_asset_return_snapshot: missing columns={missing_columns}")
market = df[[date_col, stock_col, normalized_price_col]].copy()
if any(not isinstance(asset, str) or not asset.strip() for asset in market[stock_col]):
raise ValueError("asset labels must be non-empty strings")
market[stock_col] = market[stock_col].str.strip()
try:
normalized_dates = pd.to_datetime(market[date_col], errors="raise")
except (TypeError, ValueError) as error:
raise ValueError("trade dates must be valid dates") from error
if normalized_dates.isna().any():
raise ValueError("trade dates must be valid dates")
market[date_col] = normalized_dates.dt.normalize()
if market.duplicated(subset=[date_col, stock_col]).any():
raise ValueError("duplicate asset/session prices are not allowed")
try:
market[normalized_price_col] = pd.to_numeric(
market[normalized_price_col],
errors="raise",
)
except (TypeError, ValueError) as error:
raise ValueError("prices must be numeric") from error
observed_prices = market[normalized_price_col].dropna().to_numpy(dtype=float)
if observed_prices.size == 0 or not np.isfinite(observed_prices).all():
raise ValueError("prices must contain positive finite observations")
if (observed_prices <= 0.0).any():
raise ValueError("prices must contain positive finite observations")
prices = market.pivot(
index=date_col,
columns=stock_col,
values=normalized_price_col,
).sort_index()
prices = prices.reindex(sorted(str(asset) for asset in prices.columns), axis="columns")
prices = prices.astype(float)
if len(prices) < 2:
raise ValueError("market prices must contain at least two sessions")
returns = prices.pct_change(fill_method=None)
assets = tuple(str(asset) for asset in prices.columns)
snapshot_id = _asset_return_snapshot_id(
prices,
source=normalized_source,
source_snapshot_id=normalized_source_snapshot_id,
price_field=normalized_price_col,
adjustment=normalized_adjustment,
)
return AssetReturnSnapshot(
data_snapshot_id=snapshot_id,
source=normalized_source,
source_snapshot_id=normalized_source_snapshot_id,
price_field=normalized_price_col,
adjustment=normalized_adjustment,
return_method="simple",
start_date=prices.index[0].date(),
end_date=prices.index[-1].date(),
assets=assets,
returns=returns,
)
def prepare_execution_inputs(
df: pd.DataFrame,
stock_col: str = "stock_code",
date_col: str = "trade_date",
*,
price_col: str = "close",
) -> tuple[pd.DataFrame, pd.DataFrame]:
"""长表行情 → execution 输入(prices + volumes 宽表)。
@@ -294,10 +496,11 @@ def prepare_execution_inputs(
df: 长表行情(含 close / volume 列,Tushare rename 后)
stock_col: 股票代码列名
date_col: 日期列名
price_col: 执行价字段,默认 close;防前视研究可显式选择下一交易日 open
Returns:
(prices_wide, volumes_wide):
- prices_wide: date × stock_code,值=close
- prices_wide: date × stock_code,值=price_col
- volumes_wide: date × stock_code,值=volume(若无 volume 列则全 1.0)
Examples:
@@ -313,9 +516,9 @@ def prepare_execution_inputs(
"""
if df.empty:
return pd.DataFrame(), pd.DataFrame()
if "close" not in df.columns:
raise ValueError(f"prepare_execution_inputs: 缺 close 列,实际列={list(df.columns)}")
prices = long_to_wide(df, value_col="close", date_col=date_col, stock_col=stock_col)
if price_col not in df.columns:
raise ValueError(f"prepare_execution_inputs: 缺 {price_col} 列,实际列={list(df.columns)}")
prices = long_to_wide(df, value_col=price_col, date_col=date_col, stock_col=stock_col)
if "volume" in df.columns:
volumes = long_to_wide(df, value_col="volume", date_col=date_col, stock_col=stock_col)
else:
+495 -147
View File
@@ -10,7 +10,9 @@
借鉴 hikyuu SG/MM/CN/PG 部件化思想(不引入 hikyuu 框架):
- ExecutionConfig:佣金 + 印花税 + 滑点 + 最小交易额 + 止损/止盈阈值
- simulate_execution():从目标权重 → 实际成交金额(应用成本/滑点)
- simulate_multi_day():多日组合仿真(NAV 序列 + 调仓记录)
- simulate_daily_ledger_with_audit():稀疏调仓 + 完整交易日收盘估值 Ledger
- simulate_multi_day_with_audit():目标权重差额调仓(成交/拒绝/持仓/NAV)
- simulate_multi_day():兼容的多日日末持仓快照入口
- check_stop_loss_take_profit():止损/止盈触发判定
- run_end_to_end_poc():signal → 调仓 → 执行 → NAV 端到端 POC
@@ -19,8 +21,9 @@
from __future__ import annotations
import math
from collections.abc import Mapping
from dataclasses import dataclass
from dataclasses import dataclass, replace
from typing import Any
import pandas as pd
@@ -101,6 +104,9 @@ class ExecutionResult:
net_cash_flow: float # 净现金流(买入为负,卖出为正)
partial_fill_pct: float = 1.0 # 实际成交占目标的比例(1.0 = 全部成交)
blocked_reason: str = "" # 阻塞原因(如涨跌停停牌)
side: str = "" # buy / sell;未成交记录也保留目标方向
quantity: float = 0.0 # 实际成交股数
price: float = 0.0 # 未含滑点的参考执行价
def _apply_costs(
@@ -344,19 +350,458 @@ class DailyExecution:
"""单日执行记录。"""
date: str
executions: list[ExecutionResult]
executions: tuple[ExecutionResult, ...]
nav_before: float
nav_after: float
rebalance_triggered: bool
def simulate_multi_day(
@dataclass(frozen=True)
class ExecutionSimulationResult:
"""单次多日仿真的持仓与执行审计结果。"""
initial_cash: float
positions: tuple[DailyPosition, ...]
daily_executions: tuple[DailyExecution, ...]
@property
def nav_series(self) -> pd.Series:
"""返回按日期索引的日末 NAV 副本。"""
return pd.Series(
[position.portfolio_value for position in self.positions],
index=[position.date for position in self.positions],
dtype=float,
)
@property
def normalized_nav_series(self) -> pd.Series:
"""返回以初始资金为 1 的净值曲线副本。"""
nav = self.nav_series
if self.initial_cash == 0:
return pd.Series(0.0, index=nav.index, dtype=float)
return nav / self.initial_cash
@property
def daily_returns(self) -> pd.Series:
"""返回逐日收益;首日相对初始资金计算,保留首日交易成本。"""
nav = self.nav_series
if nav.empty:
return nav
returns = nav.pct_change()
returns.iloc[0] = (
nav.iloc[0] / self.initial_cash - 1.0 if self.initial_cash != 0 else 0.0
)
return returns.fillna(0.0)
@property
def trades_frame(self) -> pd.DataFrame:
"""返回可投影到平台成交明细的实际成交表,不包含纯拒绝记录。"""
columns = [
"trade_date",
"ts_code",
"side",
"qty",
"price",
"amount",
"fee",
"slippage",
]
rows = [
{
"trade_date": daily.date,
"ts_code": execution.stock_code,
"side": execution.side,
"qty": execution.quantity,
"price": execution.price,
"amount": execution.executed_value,
"fee": execution.commission + execution.stamp_tax,
"slippage": execution.slippage_cost,
}
for daily in self.daily_executions
for execution in daily.executions
if execution.quantity > 0
]
return pd.DataFrame(rows, columns=columns)
@property
def ledger_frame(self) -> pd.DataFrame:
"""返回稳定的日频 Ledger 投影,不附加运行元数据或写数据库。"""
columns = [
"trade_date",
"portfolio_value",
"nav",
"pnl",
"pnl_pct",
"position_value",
"cash",
"turnover",
]
previous_value = self.initial_cash
rows: list[dict[str, float | str]] = []
daily_returns = self.daily_returns
for index, (position, daily) in enumerate(
zip(self.positions, self.daily_executions, strict=True)
):
daily_turnover = sum(
execution.executed_value
for execution in daily.executions
if execution.quantity > 0
)
turnover_rate = daily_turnover / daily.nav_before if daily.nav_before > 0 else 0.0
rows.append(
{
"trade_date": position.date,
"portfolio_value": position.portfolio_value,
"nav": (
position.portfolio_value / self.initial_cash
if self.initial_cash != 0
else 0.0
),
"pnl": position.portfolio_value - previous_value,
"pnl_pct": float(daily_returns.iloc[index]),
"position_value": position.portfolio_value - position.cash,
"cash": position.cash,
"turnover": turnover_rate,
}
)
previous_value = position.portfolio_value
return pd.DataFrame(rows, columns=columns)
@property
def total_costs(self) -> float:
"""汇总实际成交产生的成本。"""
return sum(
execution.total_cost
for daily in self.daily_executions
for execution in daily.executions
)
@property
def total_turnover(self) -> float:
"""汇总实际成交金额。"""
return sum(
execution.executed_value
for daily in self.daily_executions
for execution in daily.executions
)
@property
def total_rebalances(self) -> int:
"""返回至少有一笔实际成交的调仓日数量。"""
return sum(daily.rebalance_triggered for daily in self.daily_executions)
@property
def final_portfolio_value(self) -> float:
"""返回最后一个日末 NAV;空输入时返回初始资金。"""
if not self.positions:
return self.initial_cash
return self.positions[-1].portfolio_value
@property
def return_pct(self) -> float:
"""返回相对初始资金的百分比收益。"""
if self.initial_cash == 0:
return 0.0
return (self.final_portfolio_value / self.initial_cash - 1.0) * 100.0
def _blocked_execution(stock_code: str, target_value: float, reason: str) -> ExecutionResult:
"""构造未成交但可审计的执行记录。"""
return ExecutionResult(
stock_code=stock_code,
target_value=target_value,
executed_value=0.0,
commission=0.0,
stamp_tax=0.0,
slippage_cost=0.0,
total_cost=0.0,
net_cash_flow=0.0,
partial_fill_pct=0.0,
blocked_reason=reason,
side="buy" if target_value > 0 else "sell" if target_value < 0 else "",
)
def _validate_target_weights(date: str, targets: Mapping[str, float]) -> dict[str, float]:
"""校验并复制单日长仓目标权重。"""
normalized: dict[str, float] = {}
for stock_code, raw_weight in targets.items():
try:
weight = float(raw_weight)
except (TypeError, ValueError) as error:
raise ValueError(f"target weights on {date!r} must be numeric") from error
if not math.isfinite(weight) or weight < 0:
raise ValueError(f"target weights on {date!r} must be finite and non-negative")
normalized[stock_code] = weight
if sum(normalized.values()) > 1.0 + 1e-12:
raise ValueError(f"target weights on {date!r} must sum to at most 1.0")
return normalized
def _partially_fill_buy(
desired: ExecutionResult,
fill_pct: float,
config: ExecutionConfig,
) -> ExecutionResult:
"""按同一比例缩放买入,保留原始目标金额供审计。"""
actual_target_value = desired.target_value * fill_pct
executed_value, commission, stamp_tax, slippage_cost = _apply_costs(
actual_target_value,
True,
config,
)
total_cost = commission + stamp_tax + slippage_cost
return ExecutionResult(
stock_code=desired.stock_code,
target_value=desired.target_value,
executed_value=executed_value,
commission=commission,
stamp_tax=stamp_tax,
slippage_cost=slippage_cost,
total_cost=total_cost,
net_cash_flow=-(executed_value + commission + stamp_tax),
partial_fill_pct=fill_pct,
blocked_reason="insufficient_cash_partial_fill",
)
def _rebalance_at_prices(
date: str,
targets: Mapping[str, float],
prices: Mapping[str, float],
cash: float,
holdings: dict[str, float],
config: ExecutionConfig,
) -> tuple[float, tuple[ExecutionResult, ...], float, float]:
"""在单一执行时点按目标权重差额调仓,并原地更新 holdings。"""
normalized_targets = _validate_target_weights(date, targets)
for held_code in holdings:
held_price = prices.get(held_code)
if held_price is None or not math.isfinite(held_price) or held_price <= 0:
raise ValueError(f"missing price for held asset {held_code} on {date!r}")
nav_before = cash + sum(
shares * prices.get(stock_code, 0.0)
for stock_code, shares in holdings.items()
)
effective_targets = dict.fromkeys(holdings, 0.0)
effective_targets.update(normalized_targets)
buy_weights: dict[str, float] = {}
sell_weights: dict[str, float] = {}
rejected: list[ExecutionResult] = []
for stock_code, target_weight in effective_targets.items():
price = prices.get(stock_code)
target_value = float(target_weight) * nav_before
if price is None or not math.isfinite(price) or price <= 0:
if target_value != 0 or holdings.get(stock_code, 0.0) != 0:
rejected.append(_blocked_execution(stock_code, target_value, "missing_price"))
continue
current_value = holdings.get(stock_code, 0.0) * price
trade_value = target_value - current_value
if abs(trade_value) < config.min_trade_amount or math.isclose(
trade_value, 0.0, abs_tol=1e-12
):
continue
if nav_before == 0:
rejected.append(_blocked_execution(stock_code, trade_value, "zero_nav"))
continue
destination = buy_weights if trade_value > 0 else sell_weights
destination[stock_code] = trade_value / nav_before
sell_executions = simulate_execution(sell_weights, nav_before, config)
filled: list[ExecutionResult] = []
for raw_execution in sell_executions:
price = prices[raw_execution.stock_code]
quantity = abs(raw_execution.target_value) / price
execution = replace(
raw_execution,
side="sell",
quantity=quantity,
price=price,
)
held = holdings.get(execution.stock_code, 0.0)
holdings[execution.stock_code] = max(0.0, held - quantity)
if holdings[execution.stock_code] < 1e-6:
del holdings[execution.stock_code]
cash += execution.net_cash_flow
filled.append(execution)
desired_buys = simulate_execution(buy_weights, nav_before, config)
required_cash = sum(-execution.net_cash_flow for execution in desired_buys)
buy_fill_pct = min(1.0, max(cash, 0.0) / required_cash) if required_cash > 0 else 1.0
for desired in desired_buys:
if buy_fill_pct == 0:
rejected.append(
_blocked_execution(desired.stock_code, desired.target_value, "insufficient_cash")
)
continue
raw_execution = (
desired
if buy_fill_pct == 1.0
else _partially_fill_buy(desired, buy_fill_pct, config)
)
price = prices[raw_execution.stock_code]
quantity = abs(raw_execution.target_value) * raw_execution.partial_fill_pct / price
execution = replace(
raw_execution,
side="buy",
quantity=quantity,
price=price,
)
holdings[execution.stock_code] = holdings.get(execution.stock_code, 0.0) + quantity
cash += execution.net_cash_flow
if math.isclose(cash, 0.0, abs_tol=1e-9):
cash = 0.0
filled.append(execution)
executions = (*filled, *rejected)
nav_after = cash + sum(
shares * prices.get(stock_code, 0.0)
for stock_code, shares in holdings.items()
)
return cash, executions, nav_before, nav_after
def _validate_sparse_daily_histories(
target_weights_history: list[tuple[str, dict[str, float]]],
execution_price_history: list[tuple[str, dict[str, float]]],
valuation_price_history: list[tuple[str, dict[str, float]]],
) -> tuple[
dict[str, dict[str, float]],
dict[str, dict[str, float]],
list[tuple[str, dict[str, float]]],
]:
"""校验稀疏调仓与完整估值日历,并隔离调用方可变输入。"""
target_dates = [date for date, _ in target_weights_history]
execution_dates = [date for date, _ in execution_price_history]
valuation_dates = [date for date, _ in valuation_price_history]
if len(set(target_dates)) != len(target_dates):
raise ValueError("target_weights_history must contain unique dates")
if len(set(execution_dates)) != len(execution_dates):
raise ValueError("execution_price_history must contain unique dates")
if len(set(valuation_dates)) != len(valuation_dates):
raise ValueError("valuation_price_history must contain unique dates")
if execution_dates != target_dates:
raise ValueError("execution price dates must exactly match target weight dates")
valuation_positions = {date: index for index, date in enumerate(valuation_dates)}
missing_dates = [date for date in target_dates if date not in valuation_positions]
if missing_dates:
raise ValueError(f"target dates must belong to valuation calendar: {missing_dates}")
positions = [valuation_positions[date] for date in target_dates]
if positions != sorted(positions):
raise ValueError("target weights must follow valuation calendar order")
targets = {date: dict(values) for date, values in target_weights_history}
execution_prices = {date: dict(values) for date, values in execution_price_history}
valuation_prices = [(date, dict(values)) for date, values in valuation_price_history]
return targets, execution_prices, valuation_prices
def _simulate_daily_ledger(
target_weights_history: list[tuple[str, dict[str, float]]],
execution_price_history: list[tuple[str, dict[str, float]]],
valuation_price_history: list[tuple[str, dict[str, float]]],
initial_cash: float,
config: ExecutionConfig,
) -> ExecutionSimulationResult:
targets_by_date, execution_prices_by_date, valuation_history = (
_validate_sparse_daily_histories(
target_weights_history,
execution_price_history,
valuation_price_history,
)
)
cash = initial_cash
holdings: dict[str, float] = {}
positions: list[DailyPosition] = []
daily_executions: list[DailyExecution] = []
for date, valuation_prices in valuation_history:
targets = targets_by_date.get(date)
if targets is None:
executions: tuple[ExecutionResult, ...] = ()
nav_before = 0.0
nav_after = 0.0
rebalance_triggered = False
else:
cash, executions, nav_before, nav_after = _rebalance_at_prices(
date,
targets,
execution_prices_by_date[date],
cash,
holdings,
config,
)
rebalance_triggered = any(execution.quantity > 0 for execution in executions)
for held_code in holdings:
valuation_price = valuation_prices.get(held_code)
if (
valuation_price is None
or not math.isfinite(valuation_price)
or valuation_price <= 0
):
raise ValueError(
f"missing valuation price for held asset {held_code} on {date!r}"
)
portfolio_value = cash + sum(
shares * valuation_prices[stock_code]
for stock_code, shares in holdings.items()
)
if targets is None:
nav_before = portfolio_value
nav_after = portfolio_value
positions.append(DailyPosition(date, cash, dict(holdings), portfolio_value))
daily_executions.append(
DailyExecution(
date=date,
executions=executions,
nav_before=nav_before,
nav_after=nav_after,
rebalance_triggered=rebalance_triggered,
)
)
return ExecutionSimulationResult(
initial_cash=initial_cash,
positions=tuple(positions),
daily_executions=tuple(daily_executions),
)
def simulate_daily_ledger_with_audit(
target_weights_history: list[tuple[str, dict[str, float]]],
execution_price_history: list[tuple[str, dict[str, float]]],
valuation_price_history: list[tuple[str, dict[str, float]]],
initial_cash: float,
config: ExecutionConfig | None = None,
) -> ExecutionSimulationResult:
"""以稀疏调仓和完整日历运行成交后持仓 Ledger。
执行价只用于调仓日现金与股数变化,估值价用于每个交易日日末 NAV;二者
显式分离,从而支持“下一日 open 成交、同日 close 估值”的无前视研究。
"""
if not math.isfinite(initial_cash) or initial_cash <= 0:
raise ValueError(f"initial_cash must be positive and finite, got {initial_cash}")
return _simulate_daily_ledger(
target_weights_history,
execution_price_history,
valuation_price_history,
initial_cash,
ExecutionConfig() if config is None else config,
)
def simulate_multi_day_with_audit(
target_weights_history: list[tuple[str, dict[str, float]]],
price_history: list[tuple[str, dict[str, float]]],
initial_cash: float,
config: ExecutionConfig | None = None,
) -> list[DailyPosition]:
"""多日组合仿真(NAV 序列)。
) -> ExecutionSimulationResult:
"""按目标权重差额推进组合,并返回唯一事实来源的审计结果。
Args:
target_weights_history: [(date, {stock_code: target_weight})]
@@ -365,97 +810,45 @@ def simulate_multi_day(
config: 执行配置
Returns:
DailyPosition 列表(每日 NAV 快照)。
日末持仓快照与逐日成交记录组成的结构化审计结果。
Note:
- 调仓频率 = target_weights_history 的频率(每日 / 每周 / 每月都行)
- 每日先按当日 close 估值,再按当日 target 调仓(下一交易日生效)
- 此处简化:调仓使用当日 close 价格
- 每日先按当日 close 估值,再交易“目标市值 - 当前市值”的差额
- 此处简化为当日 close 成交;调用方必须传入已正确滞后的目标权重
"""
if config is None:
config = ExecutionConfig()
if not math.isfinite(initial_cash) or initial_cash < 0:
raise ValueError(f"initial_cash must be finite and non-negative, got {initial_cash}")
if len(target_weights_history) != len(price_history):
raise ValueError("target_weights_history and price_history must have same length")
if not target_weights_history:
return []
cash = initial_cash
holdings: dict[str, float] = {}
positions: list[DailyPosition] = []
for (date, targets), (_, prices) in zip(target_weights_history, price_history, strict=True):
# 1) 先按当日收盘价估值
portfolio_value = cash + sum(
shares * prices.get(code, 0.0) for code, shares in holdings.items()
)
positions.append(
DailyPosition(
date=date,
cash=cash,
holdings=dict(holdings),
portfolio_value=portfolio_value,
for (date, _), (price_date, _) in zip(target_weights_history, price_history, strict=True):
if date != price_date:
raise ValueError(
f"target and price dates must match, got {date!r} and {price_date!r}"
)
)
# 2) 计算 effective_targets(包含需要平仓的零权重)
effective_targets: dict[str, float] = dict(targets)
for held_code in holdings:
if held_code not in effective_targets:
effective_targets[held_code] = 0.0
# 3) 调仓(只对非零目标调用 simulate_execution)
non_zero_targets = {k: v for k, v in effective_targets.items() if v != 0}
results = simulate_execution(non_zero_targets, portfolio_value, config)
# 4) 处理零目标(平仓):构造 ExecutionResult,shares = held(全部卖出)
for stock_code, weight in effective_targets.items():
if weight == 0 and stock_code in holdings and holdings[stock_code] > 0:
price = prices.get(stock_code, 0.0)
if price > 0:
held = holdings[stock_code]
# 全部卖出:target_shares = held
# executed_value = held * price(考虑滑点)
slippage_factor = 1.0 - config.slippage_bps / 10000.0
target_value = -held * price
executed_value = target_value * slippage_factor
commission = abs(executed_value) * config.commission_bps / 10000.0
stamp_tax = abs(executed_value) * config.stamp_tax_bps / 10000.0
slippage_cost = abs(executed_value - target_value)
# 标记净卖出 shares = held
results.append(
ExecutionResult(
stock_code=stock_code,
target_value=target_value,
executed_value=executed_value,
commission=commission,
stamp_tax=stamp_tax,
slippage_cost=slippage_cost,
total_cost=commission + stamp_tax + slippage_cost,
net_cash_flow=executed_value - commission - stamp_tax,
)
)
# 5) 应用执行结果到持仓
for r in results:
cost = r.executed_value + r.commission + r.stamp_tax
proceeds = r.executed_value - r.commission - r.stamp_tax
price = prices.get(r.stock_code, 0.0)
if r.target_value > 0:
# 买入:shares = 正数 executed_value / price,cash 减少 cost
shares = r.executed_value / price if price > 0 else 0.0
holdings[r.stock_code] = holdings.get(r.stock_code, 0.0) + shares
cash -= cost
else:
# 卖出:cash 增加 proceeds 的绝对值(proceeds 本是负的)
held = holdings.get(r.stock_code, 0.0)
if held > 0:
# 如果是 zero-target 触发的全卖(target_value 与持仓市值近似),全部卖出
if abs(r.target_value) >= held * price * 0.95:
sell_shares = held
else:
target_shares = abs(r.executed_value) / price if price > 0 else held
sell_shares = min(held, target_shares)
holdings[r.stock_code] = held - sell_shares
if holdings[r.stock_code] < 1e-6:
del holdings[r.stock_code]
# proceeds 是负的(target_value 负),cash += proceeds 实际是减去
# 但卖出是现金流入,所以应该 cash += abs(proceeds)
cash += abs(proceeds)
return positions
return _simulate_daily_ledger(
target_weights_history,
price_history,
price_history,
initial_cash,
ExecutionConfig() if config is None else config,
)
def simulate_multi_day(
target_weights_history: list[tuple[str, dict[str, float]]],
price_history: list[tuple[str, dict[str, float]]],
initial_cash: float,
config: ExecutionConfig | None = None,
) -> list[DailyPosition]:
"""兼容入口:返回多日仿真的日末持仓快照。"""
result = simulate_multi_day_with_audit(
target_weights_history,
price_history,
initial_cash,
config,
)
return list(result.positions)
def run_end_to_end_poc(
@@ -486,64 +879,16 @@ def run_end_to_end_poc(
config = ExecutionConfig()
if len(signals) != len(prices):
raise ValueError("signals and prices must have same length")
positions = simulate_multi_day(signals, prices, initial_cash, config)
nav_series = pd.Series(
[p.portfolio_value for p in positions], index=[p.date for p in positions]
)
# 计算 total_costs / total_turnover(重放所有执行)
total_cost_acc = 0.0
total_turnover_acc = 0.0
rebalance_count = 0
cash = initial_cash
holdings: dict[str, float] = {}
for (date, targets), (_, price_map) in zip(signals, prices, strict=True):
portfolio_value = cash + sum(
shares * price_map.get(code, 0.0) for code, shares in holdings.items()
)
if targets:
rebalance_count += 1
# 自动平仓:持仓但不在 target 中的股票
effective_targets: dict[str, float] = dict(targets)
for held_code in holdings:
if held_code not in effective_targets:
effective_targets[held_code] = 0.0
results = simulate_execution(effective_targets, portfolio_value, config)
total_cost_acc += total_costs(results)
total_turnover_acc += total_turnover(results)
for r in results:
cost = r.executed_value + r.commission + r.stamp_tax
proceeds = r.executed_value - r.commission - r.stamp_tax
if r.target_value > 0:
shares = (
r.executed_value / price_map[r.stock_code]
if price_map[r.stock_code] > 0
else 0.0
)
holdings[r.stock_code] = holdings.get(r.stock_code, 0.0) + shares
cash -= cost
else:
held = holdings.get(r.stock_code, 0.0)
if held > 0:
sell_shares = min(
held,
abs(r.executed_value / price_map[r.stock_code])
if price_map[r.stock_code] > 0
else held,
)
holdings[r.stock_code] = held - sell_shares
if holdings[r.stock_code] < 1e-6:
del holdings[r.stock_code]
cash += proceeds
audit = simulate_multi_day_with_audit(signals, prices, initial_cash, config)
return {
"positions": positions,
"nav_series": nav_series,
"total_costs": total_cost_acc,
"total_turnover": total_turnover_acc,
"total_rebalances": rebalance_count,
"final_portfolio_value": nav_series.iloc[-1] if len(nav_series) > 0 else initial_cash,
"return_pct": ((nav_series.iloc[-1] / initial_cash) - 1) * 100
if len(nav_series) > 0
else 0.0,
"positions": list(audit.positions),
"daily_executions": list(audit.daily_executions),
"nav_series": audit.nav_series,
"total_costs": audit.total_costs,
"total_turnover": audit.total_turnover,
"total_rebalances": audit.total_rebalances,
"final_portfolio_value": audit.final_portfolio_value,
"return_pct": audit.return_pct,
}
@@ -667,7 +1012,10 @@ __all__ = [
"apply_bid_ask_spread",
"DailyPosition",
"DailyExecution",
"ExecutionSimulationResult",
"simulate_daily_ledger_with_audit",
"simulate_multi_day",
"simulate_multi_day_with_audit",
"run_end_to_end_poc",
"DailyPnL",
"simulate_with_daily_data",
File diff suppressed because it is too large Load Diff
+6 -2
View File
@@ -312,8 +312,8 @@ def ols_regress(
ss_tot = float(((y_arr - y_arr.mean()) ** 2).sum())
r_sq = 1.0 - ss_res / ss_tot if ss_tot > 0 else np.nan
sigma2 = ss_res / max(n - k, 1)
# 协方差矩阵 = sigma2 * (X'X)^-1
xtx_inv = np.linalg.inv(x_arr.T @ x_arr) if sigma2 > 0 else np.full((k, k), np.nan)
# 广义协方差矩阵 = sigma2 * (X'X)^+,伪逆兼容共线因子。
xtx_inv = np.linalg.pinv(x_arr.T @ x_arr) if sigma2 > 0 else np.full((k, k), np.nan)
se = np.sqrt(np.diag(xtx_inv) * sigma2)
t_vals = coef / se if sigma2 > 0 else np.full_like(coef, np.nan)
if add_constant:
@@ -513,6 +513,8 @@ def apply_factor_direction(
Returns:
方向调整后的因子(同向 = 越大越好)
"""
if direction not in {"auto", "forward", "reverse"}:
raise ValueError(f"direction={direction!r} not supported (auto / forward / reverse)")
if factor.empty:
return factor.copy()
if direction == "auto":
@@ -541,6 +543,8 @@ def cross_sectional_rank_with_direction(
Returns:
pd.Series(百分位排名 [0, 1],越大越优)
"""
if direction not in {"auto", "forward", "reverse"}:
raise ValueError(f"direction={direction!r} not supported (auto / forward / reverse)")
if df.empty or factor_col not in df.columns:
return pd.Series(dtype=float)
factor = df[factor_col]
File diff suppressed because it is too large Load Diff
+99 -1
View File
@@ -65,13 +65,28 @@ def sharpe_ratio(r: pd.Series, rf: float = 0.0) -> float:
return (annualized_return(r) - rf) / vol
def sortino_ratio(r: pd.Series, rf: float = 0.0) -> float:
"""Sortino = (年化收益 - rf) / 年化下行偏差。"""
r = _clean(r)
if len(r) < 2:
return 0.0
downside = np.minimum(r.to_numpy(dtype=float), 0.0)
downside_deviation = float(
np.sqrt(np.mean(np.square(downside))) * np.sqrt(TRADING_DAYS_PER_YEAR)
)
if downside_deviation == 0:
return 0.0
return (annualized_return(r) - rf) / downside_deviation
def max_drawdown(r: pd.Series) -> float:
"""最大回撤(负数)。例如 -0.2 表示最大亏 20%。"""
r = _clean(r)
if len(r) < 2:
return 0.0
nav = (1 + r).cumprod()
peak = nav.cummax()
# 初始资金净值为 1;否则首个观测日的亏损会被误当成新的历史高点。
peak = nav.cummax().clip(lower=1.0)
drawdown = (nav - peak) / peak
return float(drawdown.min())
@@ -119,6 +134,7 @@ def summary(r: pd.Series, rf: float = 0.0) -> Mapping[str, float]:
"ann_return": ann_ret,
"ann_volatility": ann_vol,
"sharpe": sharpe_ratio(r, rf),
"sortino": sortino_ratio(r, rf),
"max_drawdown": mdd,
"calmar": calmar_ratio(r),
"win_rate": win_rate(r),
@@ -129,6 +145,62 @@ def summary(r: pd.Series, rf: float = 0.0) -> Mapping[str, float]:
}
def benchmark_summary(
portfolio_returns: pd.Series,
benchmark_returns: pd.Series,
*,
risk_free_daily: float = 0.0,
annualization: int = TRADING_DAYS_PER_YEAR,
) -> Mapping[str, float]:
"""计算成本后组合相对基准的严格对齐绩效。
与通用 ``summary`` 不同,本函数拒绝静默清洗或日期 inner join。alpha
使用日频回归截距的几何年化;基准方差不足时 alpha/beta 为 NaN,明确
表示回归不可估计。
"""
portfolio, benchmark = _validate_benchmark_inputs(
portfolio_returns,
benchmark_returns,
)
if isinstance(annualization, bool) or not isinstance(annualization, int):
raise TypeError("annualization must be an integer")
if annualization <= 0:
raise ValueError("annualization must be positive")
if not np.isfinite(risk_free_daily):
raise ValueError("risk_free_daily must be finite")
active = portfolio - benchmark
active_std = float(active.std())
tracking_error = active_std * float(np.sqrt(annualization))
information_ratio = (
float(active.mean()) / active_std * float(np.sqrt(annualization))
if active_std >= 1e-30
else float("nan")
)
adjusted_portfolio = portfolio - risk_free_daily
adjusted_benchmark = benchmark - risk_free_daily
benchmark_variance = float(adjusted_benchmark.var())
if benchmark_variance < 1e-30:
beta = float("nan")
alpha = float("nan")
else:
beta = float(adjusted_portfolio.cov(adjusted_benchmark) / benchmark_variance)
alpha_daily = float((adjusted_portfolio - beta * adjusted_benchmark).mean())
alpha = (
float((1.0 + alpha_daily) ** annualization - 1.0)
if alpha_daily > -1.0
else float("nan")
)
return {
"n_observations": len(portfolio),
"tracking_error": tracking_error,
"information_ratio": information_ratio,
"alpha": alpha,
"beta": beta,
}
# ── 内部 ──────────────────────────────────────
@@ -137,3 +209,29 @@ def _clean(r: pd.Series) -> pd.Series:
if not isinstance(r, pd.Series):
raise TypeError(f"expected pd.Series, got {type(r).__name__}")
return r.replace([np.inf, -np.inf], np.nan).dropna()
def _validate_benchmark_inputs(
portfolio_returns: pd.Series,
benchmark_returns: pd.Series,
) -> tuple[pd.Series, pd.Series]:
if not isinstance(portfolio_returns, pd.Series):
raise TypeError("portfolio_returns must be a pandas Series")
if not isinstance(benchmark_returns, pd.Series):
raise TypeError("benchmark_returns must be a pandas Series")
if not portfolio_returns.index.equals(benchmark_returns.index):
raise ValueError("portfolio and benchmark returns must use matching indexes")
if not portfolio_returns.index.is_unique:
raise ValueError("portfolio and benchmark indexes must be unique")
if len(portfolio_returns) < 2:
raise ValueError("benchmark metrics require at least two observations")
portfolio = portfolio_returns.astype(float, copy=True)
benchmark = benchmark_returns.astype(float, copy=True)
if not np.isfinite(portfolio.to_numpy()).all() or not np.isfinite(
benchmark.to_numpy()
).all():
raise ValueError("portfolio and benchmark returns must be finite")
if (portfolio < -1.0).any() or (benchmark < -1.0).any():
raise ValueError("simple returns cannot be less than -1")
return portfolio, benchmark
+104
View File
@@ -0,0 +1,104 @@
"""因子分数到目标权重的轻量组合构建闭环。"""
from __future__ import annotations
import numpy as np
import pandas as pd
from pandas.api.types import is_numeric_dtype
__all__ = [
"select_top_k",
"equal_weight",
"scores_to_target_weights",
"scores_to_weight_table",
]
def _validate_top_k(top_k: int) -> None:
if isinstance(top_k, bool) or not isinstance(top_k, int) or top_k <= 0:
raise ValueError("top_k must be positive")
def _validate_gross_exposure(gross_exposure: float) -> None:
if not np.isfinite(gross_exposure) or gross_exposure < 0:
raise ValueError("gross_exposure must be finite and non-negative")
def _validate_score_series(scores: pd.Series) -> None:
if not isinstance(scores, pd.Series):
raise TypeError(f"scores must be a pandas Series, got {type(scores).__name__}")
if not scores.index.is_unique:
raise ValueError("scores must contain unique asset labels")
if not is_numeric_dtype(scores.dtype):
raise TypeError("scores must contain numeric values")
def select_top_k(scores: pd.Series, top_k: int, *, largest: bool = True) -> pd.Index:
"""稳定选择最高或最低的 K 个有效因子分数。"""
_validate_top_k(top_k)
_validate_score_series(scores)
valid_scores = scores.dropna()
ordered = valid_scores.sort_values(ascending=not largest, kind="mergesort")
return ordered.iloc[:top_k].index.copy()
def equal_weight(assets: pd.Index, *, gross_exposure: float = 1.0) -> pd.Series:
"""在已选资产间等权分配指定总敞口。"""
_validate_gross_exposure(gross_exposure)
if not assets.is_unique:
raise ValueError("assets must contain unique asset labels")
if assets.empty:
return pd.Series(index=assets.copy(), dtype=float, name="weight")
weight = gross_exposure / len(assets)
return pd.Series(weight, index=assets.copy(), dtype=float, name="weight")
def scores_to_target_weights(
scores: pd.Series,
top_k: int,
*,
gross_exposure: float = 1.0,
largest: bool = True,
) -> pd.Series:
"""把单期因子分数转换为完整股票池目标权重。"""
_validate_score_series(scores)
selected = select_top_k(scores, top_k, largest=largest)
selected_weights = equal_weight(selected, gross_exposure=gross_exposure)
result = pd.Series(0.0, index=scores.index.copy(), dtype=float, name="weight")
result.loc[selected_weights.index] = selected_weights
return result
def scores_to_weight_table(
scores: pd.DataFrame,
top_k: int,
*,
gross_exposure: float = 1.0,
largest: bool = True,
) -> pd.DataFrame:
"""逐调仓日独立构建目标权重表,避免使用未来分数。"""
if not isinstance(scores, pd.DataFrame):
raise TypeError(f"scores must be a pandas DataFrame, got {type(scores).__name__}")
_validate_top_k(top_k)
_validate_gross_exposure(gross_exposure)
if not scores.index.is_unique:
raise ValueError("scores must contain unique rebalance dates")
if not scores.index.is_monotonic_increasing:
raise ValueError("scores rebalance dates must be in chronological order")
if not scores.columns.is_unique:
raise ValueError("scores must contain unique asset labels")
if not all(is_numeric_dtype(dtype) for dtype in scores.dtypes):
raise TypeError("scores must contain numeric values")
if scores.empty:
return pd.DataFrame(index=scores.index.copy(), columns=scores.columns.copy(), dtype=float)
rows = [
scores_to_target_weights(
row,
top_k,
gross_exposure=gross_exposure,
largest=largest,
).to_numpy()
for _, row in scores.iterrows()
]
return pd.DataFrame(rows, index=scores.index.copy(), columns=scores.columns.copy(), dtype=float)
File diff suppressed because it is too large Load Diff
+401
View File
@@ -0,0 +1,401 @@
"""可信研究链路:因子分数经交易日历滞后后进入执行与日频 Ledger。
本模块只编排现有组合构建与执行组件,不连接账户、券商或实盘订单。
时间契约借鉴 Qlib 的 prediction/trade time 分离与 Backtrader 的 next-bar
执行语义:signal_date 上形成的目标权重,默认最早在下一交易时点执行。
完整回测链路进一步分离 execution price 与日末 valuation price,非调仓日也
持续盯市,并从真实成交后持仓派生日收益和绩效。
"""
from __future__ import annotations
from collections.abc import Mapping
from dataclasses import dataclass
import numpy as np
import pandas as pd
from pandas.api.types import is_numeric_dtype
from quant_engine.attribution import DailyReturnAttribution, compute_daily_return_attribution
from quant_engine.execution import (
ExecutionConfig,
ExecutionSimulationResult,
simulate_daily_ledger_with_audit,
simulate_multi_day_with_audit,
)
from quant_engine.metrics import benchmark_summary, summary as metrics_summary
from quant_engine.portfolio_construction import scores_to_weight_table
__all__ = [
"TargetWeightSchedule",
"FactorExecutionResult",
"FactorBacktestResult",
"schedule_target_weights",
"run_factor_execution_research",
"run_factor_backtest_research",
]
@dataclass(frozen=True, slots=True, eq=False)
class TargetWeightSchedule:
"""保留决策时间和执行时间的目标权重调度快照。"""
decision_weights: pd.DataFrame
signal_to_execution: pd.Series
execution_weights: pd.DataFrame
lag_sessions: int
@dataclass(frozen=True, slots=True, eq=False)
class FactorExecutionResult:
"""因子到执行审计的一次可复现研究结果。"""
factor_scores: pd.DataFrame
execution_prices: pd.DataFrame
schedule: TargetWeightSchedule
execution_price_field: str
execution: ExecutionSimulationResult
@dataclass(frozen=True, slots=True, eq=False)
class FactorBacktestResult:
"""因子、成交后日频 Ledger 与绩效的一次可复现快照。"""
factor_scores: pd.DataFrame
execution_prices: pd.DataFrame
valuation_prices: pd.DataFrame
schedule: TargetWeightSchedule
execution_price_field: str
valuation_price_field: str
execution: ExecutionSimulationResult
@property
def nav(self) -> pd.Series:
"""返回以初始资金归一化为 1 的日频 NAV。"""
return pd.Series(
self.execution.normalized_nav_series.to_numpy(copy=True),
index=self.valuation_prices.index.copy(),
name="nav",
)
@property
def returns(self) -> pd.Series:
"""返回包含首日成本影响的日频收益。"""
return pd.Series(
self.execution.daily_returns.to_numpy(copy=True),
index=self.valuation_prices.index.copy(),
name="returns",
)
@property
def position_weights(self) -> pd.DataFrame:
"""按日末实际股数、收盘估值和账本 NAV 投影资产权重。"""
weights = pd.DataFrame(
0.0,
index=self.valuation_prices.index.copy(),
columns=self.valuation_prices.columns.copy(),
)
for date, position in zip(
self.valuation_prices.index,
self.execution.positions,
strict=True,
):
if position.portfolio_value <= 0:
raise ValueError(f"portfolio value must be positive on {date}")
for asset, shares in position.holdings.items():
weights.at[date, asset] = (
shares * float(self.valuation_prices.at[date, asset])
/ position.portfolio_value
)
return weights
@property
def cash_weights(self) -> pd.Series:
"""返回与实际资产权重使用同一日末 NAV 分母的现金权重。"""
values = []
for date, position in zip(
self.valuation_prices.index,
self.execution.positions,
strict=True,
):
if position.portfolio_value <= 0:
raise ValueError(f"portfolio value must be positive on {date}")
values.append(position.cash / position.portfolio_value)
return pd.Series(
values,
index=self.valuation_prices.index.copy(),
dtype=float,
name="cash_weight",
)
def stats(self, rf: float = 0.0) -> Mapping[str, float]:
"""复用标准绩效口径计算指标。"""
return metrics_summary(self.returns, rf)
def return_attribution(self) -> DailyReturnAttribution:
"""从实际成交后持仓与账本生成逐日净收益归因。"""
return compute_daily_return_attribution(
self.execution,
self.execution_prices,
self.valuation_prices,
)
def benchmark_stats(self, benchmark_returns: pd.Series) -> Mapping[str, float]:
"""计算成本后日收益相对同日基准的 TE、IR、alpha 与 beta。"""
return benchmark_summary(self.returns, benchmark_returns)
def _validate_datetime_index(index: pd.Index, name: str) -> pd.DatetimeIndex:
if not isinstance(index, pd.DatetimeIndex):
raise TypeError(f"{name} must use a DatetimeIndex")
if not index.is_unique:
raise ValueError(f"{name} must contain unique sessions")
if not index.is_monotonic_increasing:
raise ValueError(f"{name} must be in chronological order")
return index
def _validate_decision_weights(decision_weights: pd.DataFrame) -> None:
if not isinstance(decision_weights, pd.DataFrame):
raise TypeError(
f"decision_weights must be a pandas DataFrame, got {type(decision_weights).__name__}"
)
_validate_datetime_index(decision_weights.index, "decision_weights index")
if not decision_weights.columns.is_unique:
raise ValueError("decision_weights must contain unique asset labels")
if not all(is_numeric_dtype(dtype) for dtype in decision_weights.dtypes):
raise TypeError("decision_weights must contain numeric values")
values = decision_weights.to_numpy(dtype=float)
if not np.isfinite(values).all() or (values < 0).any():
raise ValueError("decision_weights must be finite and non-negative")
if (decision_weights.sum(axis=1) > 1.0 + 1e-12).any():
raise ValueError("decision_weights rows must sum to at most 1.0")
def _validate_execution_prices(execution_prices: pd.DataFrame) -> pd.DatetimeIndex:
if not isinstance(execution_prices, pd.DataFrame):
raise TypeError(
f"execution_prices must be a pandas DataFrame, got {type(execution_prices).__name__}"
)
calendar = _validate_datetime_index(execution_prices.index, "execution_prices index")
if not execution_prices.columns.is_unique:
raise ValueError("execution_prices must contain unique asset labels")
if not all(is_numeric_dtype(dtype) for dtype in execution_prices.dtypes):
raise TypeError("execution_prices must contain numeric values")
return calendar
def schedule_target_weights(
decision_weights: pd.DataFrame,
trading_calendar: pd.DatetimeIndex,
*,
lag_sessions: int = 1,
) -> TargetWeightSchedule:
"""将信号日目标权重映射到后续真实交易日,不做整数行盲移位。
所有信号日必须属于 ``trading_calendar``,且日历必须包含每个信号对应的
未来执行日;无法执行的末尾信号会显式失败,避免被静默丢弃。
"""
_validate_decision_weights(decision_weights)
calendar = _validate_datetime_index(trading_calendar, "trading_calendar")
if isinstance(lag_sessions, bool) or not isinstance(lag_sessions, int) or lag_sessions <= 0:
raise ValueError("lag_sessions must be a positive integer")
decision_snapshot = decision_weights.copy(deep=True)
if decision_snapshot.empty:
execution_weights = decision_snapshot.copy(deep=True)
execution_weights.index = pd.DatetimeIndex([], name="execution_date")
mapping = pd.Series(
calendar[:0],
index=decision_snapshot.index.copy(),
name="execution_date",
)
return TargetWeightSchedule(
decision_weights=decision_snapshot,
signal_to_execution=mapping,
execution_weights=execution_weights,
lag_sessions=lag_sessions,
)
signal_positions = calendar.get_indexer(decision_snapshot.index)
if (signal_positions < 0).any():
missing = decision_snapshot.index[signal_positions < 0]
raise ValueError(
"signal dates must be trading sessions; missing="
+ ", ".join(str(date) for date in missing)
)
execution_positions = signal_positions + lag_sessions
if (execution_positions >= len(calendar)).any():
unavailable = decision_snapshot.index[execution_positions >= len(calendar)]
raise ValueError(
"trading_calendar lacks a future execution session for signal dates: "
+ ", ".join(str(date) for date in unavailable)
)
execution_dates = calendar.take(execution_positions)
signal_to_execution = pd.Series(
execution_dates,
index=decision_snapshot.index.copy(),
name="execution_date",
)
execution_weights = decision_snapshot.copy(deep=True)
execution_weights.index = pd.DatetimeIndex(execution_dates, name="execution_date")
return TargetWeightSchedule(
decision_weights=decision_snapshot,
signal_to_execution=signal_to_execution,
execution_weights=execution_weights,
lag_sessions=lag_sessions,
)
def run_factor_execution_research(
factor_scores: pd.DataFrame,
execution_prices: pd.DataFrame,
*,
top_k: int,
execution_price_field: str,
lag_sessions: int = 1,
gross_exposure: float = 1.0,
largest: bool = True,
initial_cash: float = 1_000_000.0,
config: ExecutionConfig | None = None,
) -> FactorExecutionResult:
"""运行因子分数 → 目标权重 → 下一交易时点 → 执行审计链路。
``execution_prices`` 必须代表实际拟执行时点的价格矩阵,例如日频研究中
signal 日收盘生成分数后使用下一交易日 ``open``。价格字段名称被保存在
结果元数据中,但函数不会猜测或重写价格语义。
"""
price_field = execution_price_field.strip()
if not price_field:
raise ValueError("execution_price_field must be non-empty")
calendar = _validate_execution_prices(execution_prices)
factor_snapshot = factor_scores.copy(deep=True)
decision_weights = scores_to_weight_table(
factor_snapshot,
top_k,
gross_exposure=gross_exposure,
largest=largest,
)
schedule = schedule_target_weights(
decision_weights,
calendar,
lag_sessions=lag_sessions,
)
price_snapshot = execution_prices.copy(deep=True)
target_history: list[tuple[str, dict[str, float]]] = []
price_history: list[tuple[str, dict[str, float]]] = []
for execution_date, weights in schedule.execution_weights.iterrows():
date_label = str(pd.Timestamp(execution_date))
target_history.append(
(date_label, {asset: float(weight) for asset, weight in weights.items()})
)
prices = price_snapshot.loc[execution_date]
price_history.append(
(date_label, {asset: float(price) for asset, price in prices.items()})
)
execution = simulate_multi_day_with_audit(
target_history,
price_history,
initial_cash,
config,
)
return FactorExecutionResult(
factor_scores=factor_snapshot,
execution_prices=price_snapshot,
schedule=schedule,
execution_price_field=price_field,
execution=execution,
)
def run_factor_backtest_research(
factor_scores: pd.DataFrame,
execution_prices: pd.DataFrame,
valuation_prices: pd.DataFrame,
*,
top_k: int,
execution_price_field: str,
valuation_price_field: str,
lag_sessions: int = 1,
gross_exposure: float = 1.0,
largest: bool = True,
initial_cash: float = 1_000_000.0,
config: ExecutionConfig | None = None,
) -> FactorBacktestResult:
"""运行 PIT 因子到成交后日频 Ledger、收益与绩效的可信研究链路。"""
execution_field = execution_price_field.strip()
valuation_field = valuation_price_field.strip()
if not execution_field:
raise ValueError("execution_price_field must be non-empty")
if not valuation_field:
raise ValueError("valuation_price_field must be non-empty")
execution_calendar = _validate_execution_prices(execution_prices)
valuation_calendar = _validate_execution_prices(valuation_prices)
if not execution_calendar.equals(valuation_calendar):
raise ValueError("execution and valuation prices must use matching trading calendars")
if not execution_prices.columns.equals(valuation_prices.columns):
raise ValueError("execution and valuation prices must use matching asset labels")
factor_snapshot = factor_scores.copy(deep=True)
execution_snapshot = execution_prices.copy(deep=True)
valuation_snapshot = valuation_prices.copy(deep=True)
decision_weights = scores_to_weight_table(
factor_snapshot,
top_k,
gross_exposure=gross_exposure,
largest=largest,
)
schedule = schedule_target_weights(
decision_weights,
execution_calendar,
lag_sessions=lag_sessions,
)
if decision_weights.empty:
execution_window = execution_snapshot.iloc[:0].copy()
valuation_window = valuation_snapshot.iloc[:0].copy()
else:
research_start = decision_weights.index[0]
execution_window = execution_snapshot.loc[research_start:].copy()
valuation_window = valuation_snapshot.loc[research_start:].copy()
target_history: list[tuple[str, dict[str, float]]] = []
execution_history: list[tuple[str, dict[str, float]]] = []
for execution_date, weights in schedule.execution_weights.iterrows():
date_label = str(pd.Timestamp(execution_date))
target_history.append(
(date_label, {asset: float(weight) for asset, weight in weights.items()})
)
prices = execution_window.loc[execution_date]
execution_history.append(
(date_label, {asset: float(price) for asset, price in prices.items()})
)
valuation_history = [
(
str(pd.Timestamp(valuation_date)),
{asset: float(price) for asset, price in prices.items()},
)
for valuation_date, prices in valuation_window.iterrows()
]
execution = simulate_daily_ledger_with_audit(
target_history,
execution_history,
valuation_history,
initial_cash,
config,
)
return FactorBacktestResult(
factor_scores=factor_snapshot,
execution_prices=execution_window,
valuation_prices=valuation_window,
schedule=schedule,
execution_price_field=execution_field,
valuation_price_field=valuation_field,
execution=execution,
)
@@ -0,0 +1,489 @@
"""Retrospective-only evidence wrappers over the unchanged research fact tables."""
from __future__ import annotations
import json
from collections.abc import Mapping
from dataclasses import dataclass, field
from types import MappingProxyType
from typing import Any, Self, cast
import pandas as pd
from quant_engine.artifact import (
BacktestEvidenceEntry,
EvidenceQualification,
ResearchRunArtifact,
PERFORMANCE_METRIC_SCHEMA_ID,
PERFORMANCE_METHODOLOGY_ID,
PerformanceMethodology,
PerformanceMetric,
_PERFORMANCE_SOURCE_COLUMNS,
_absolute_performance_metrics,
_benchmark_context,
_count_performance_metrics,
_performance_canonical_bytes,
_performance_compare,
_performance_date,
_performance_digest,
_performance_methodology,
_performance_text,
_performance_validate_tree,
_relative_performance_metrics,
_artifact_frames,
_evidence_entries,
_evidence_frame_records,
_manifest_instant,
_run_row,
_table_evidence,
_validate_table_run_ids,
)
from quant_engine.factor_contracts import (
ContractErrorCode,
FactorContractError,
_assert_canonical_profile,
_content_address,
_digest_bytes,
_duplicate_key_pairs,
_freeze_json,
_parse_json_object,
_parse_utc,
_thaw_json,
canonical_json,
canonical_json_bytes,
)
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
from quant_engine.retrospective_data_contracts import _check, _public, _shape
def _validated_run(run: Any) -> RetrospectiveBacktestRunRef:
_check(
type(run) is RetrospectiveBacktestRunRef,
"$.run_ref",
"explicit v2 run reference required",
ContractErrorCode.TYPE_ERROR,
)
factor = run._factor_set
return RetrospectiveBacktestRunRef.from_dict(
run.to_dict(),
dataset_snapshot=factor._dataset_snapshot,
foundation=factor._foundation,
factor_set=factor,
parent=run._parent,
)
def _validated_frames(
artifact: ResearchRunArtifact, run: RetrospectiveBacktestRunRef
) -> dict[str, pd.DataFrame]:
frames = _artifact_frames(artifact)
_validate_table_run_ids(frames, run.run_id)
row = _run_row(frames)
expected = {
"run_id": run.run_id,
"data_snapshot_id": run.dataset_snapshot_id,
"strategy_id": run.strategy_id,
"strategy_version": run.strategy_version,
"code_revision": run.code_revision,
"config_hash": run.configuration_digest.removeprefix("sha256:"),
"schema_version": artifact.schema_version,
}
_check(
set(expected) | {"started_at", "finished_at"} <= set(row.index),
"$.artifact.tables.run",
"run schema fields missing",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
for key, value in expected.items():
_check(
type(row[key]) is str and row[key] == value,
f"$.artifact.tables.run.{key}",
"artifact does not bind exact v2 run",
ContractErrorCode.IDENTITY_MISMATCH,
)
# The unchanged artifact 1.1 timestamp profile admits offsets; public v2
# envelope times remain strict UTC. No knowledge-time inference is performed.
_, started = _manifest_instant(row["started_at"], "$.artifact.tables.run.started_at")
_, finished = _manifest_instant(row["finished_at"], "$.artifact.tables.run.finished_at")
_check(
_parse_utc(run.evaluation_at, "$.run_ref.evaluation_at")
<= started
<= finished
<= _parse_utc(run.computed_at, "$.run_ref.computed_at"),
"$.artifact.tables.run",
"actual evaluation <= start <= finish <= computed required",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
for name, frame in frames.items():
_public(_evidence_frame_records(frame, name), f"$.artifact.tables.{name}")
return frames
@dataclass(frozen=True, slots=True, init=False)
class RetrospectiveBacktestEvidenceManifest:
"""Exact artifact closure, not authenticity, historical or execution authority."""
contract_name: str
schema_version: str
manifest_id: str
run_id: str
profile: str
artifact_schema_version: str
artifact_available_at: str
qualification: EvidenceQualification
evidence_digest: str
evidence: tuple[BacktestEvidenceEntry, ...]
backtest_run_ref: RetrospectiveBacktestRunRef
evidence_scope: str
usage: str
historical_availability: str
observation_cutoff: str
decision_eligible: bool
execution_validation: str
_payload: Mapping[str, Any] = field(repr=False, compare=False)
_artifact: ResearchRunArtifact = field(repr=False, compare=False)
def to_dict(self) -> dict[str, Any]:
return cast(dict[str, Any], _thaw_json(self._payload))
def to_json(self) -> str:
return canonical_json(self.to_dict())
@classmethod
def from_dict(
cls,
value: Any,
*,
artifact: ResearchRunArtifact,
backtest_run_ref: RetrospectiveBacktestRunRef,
) -> Self:
_assert_canonical_profile(value)
row = _shape(
value,
"$",
"contract_name schema_version manifest_id run_id profile artifact_schema_version artifact_available_at qualification "
"run_reference evidence_digest evidence evidence_scope usage historical_availability observation_cutoff decision_eligible execution_validation",
)
_check(
type(row["qualification"]) is str
and row["qualification"] in {"exploratory", "contract_qualified"},
"$.qualification",
"explicit non-legacy contract qualification required",
ContractErrorCode.QUALIFICATION_REJECTED,
)
rebuilt = build_retrospective_backtest_evidence_manifest(
backtest_run_ref,
artifact,
artifact_available_at=row["artifact_available_at"],
qualification=EvidenceQualification(row["qualification"]),
)
_check(
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
"$",
"manifest differs from actual run/table closure",
ContractErrorCode.IDENTITY_MISMATCH,
)
return cast(Self, rebuilt)
@classmethod
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
def build_retrospective_backtest_evidence_manifest(
backtest_run_ref: RetrospectiveBacktestRunRef,
artifact: ResearchRunArtifact,
*,
artifact_available_at: str,
qualification: EvidenceQualification = EvidenceQualification.CONTRACT_QUALIFIED,
expected_table_digests: Mapping[str, str] | None = None,
) -> RetrospectiveBacktestEvidenceManifest:
"""Close new in-memory artifact bytes; never promote an old exploratory run."""
run = _validated_run(backtest_run_ref)
_check(
type(qualification) is EvidenceQualification
and qualification is not EvidenceQualification.LEGACY_EXPLORATORY,
"$.qualification",
"legacy evidence cannot enter the v2 path",
ContractErrorCode.QUALIFICATION_REJECTED,
)
available = _parse_utc(artifact_available_at, "$.artifact_available_at")
_check(
_parse_utc(run.computed_at, "$.run_ref.computed_at") <= available,
"$.artifact_available_at",
"artifact precedes actual computation",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
frames = _validated_frames(artifact, run)
summaries = _table_evidence(frames, expected_table_digests)
reference: dict[str, object] = {"kind": "backtest_run_ref", "value": run.to_dict()}
# Table categories and canonical content hashing have not changed semantics.
evidence = _evidence_entries(summaries, reference, legacy=False)
evidence_digest = _digest_bytes(canonical_json_bytes([item.to_dict() for item in evidence]))
payload = {
"contract_name": "researchhub.backtest-evidence-manifest",
"schema_version": "2.0.0",
"run_id": run.run_id,
"profile": "offline_research_retrospective_v2",
"artifact_schema_version": artifact.schema_version,
"artifact_available_at": artifact_available_at,
"qualification": qualification.value,
"run_reference": reference,
"evidence_digest": evidence_digest,
"evidence": [item.to_dict() for item in evidence],
"evidence_scope": run.evidence_scope,
"usage": run.usage,
"historical_availability": run.historical_availability,
"observation_cutoff": run.observation_cutoff,
"decision_eligible": False,
"execution_validation": "not_validated",
}
payload["manifest_id"] = _content_address(
payload, "manifest_id", "rhbacktestevidencev2:sha256:"
)
instance = object.__new__(RetrospectiveBacktestEvidenceManifest)
values = {
**payload,
"qualification": qualification,
"evidence": evidence,
"backtest_run_ref": run,
"_payload": _freeze_json(payload),
"_artifact": artifact,
}
del values["run_reference"]
for name, value in values.items():
object.__setattr__(instance, name, value)
return instance
def _freeze_numeric_evidence(value: Any) -> Any:
"""Freeze the existing finite-number metric profile, not the data JSON profile."""
if type(value) is dict:
return MappingProxyType(
{key: _freeze_numeric_evidence(item) for key, item in value.items()}
)
if type(value) is list:
return tuple(_freeze_numeric_evidence(item) for item in value)
return value
@dataclass(frozen=True, slots=True, init=False)
class RetrospectivePerformanceEvidence:
"""New upstream/time identity; unchanged finite-number metric/methodology v1."""
methodology: PerformanceMethodology
metrics: tuple[PerformanceMetric, ...]
_payload: Mapping[str, Any] = field(repr=False)
@property
def performance_evidence_id(self) -> str:
return cast(str, self._payload["performance_evidence_id"])
@property
def document_sha256(self) -> str:
return cast(str, self._payload["document_sha256"])
@property
def run_id(self) -> str:
return cast(str, self._payload["run_id"])
def to_dict(self) -> dict[str, Any]:
return cast(dict[str, Any], _thaw_json(self._payload))
def canonical_bytes(self) -> bytes:
return _performance_canonical_bytes(self.to_dict())
def to_json(self) -> str:
return self.canonical_bytes().decode("utf-8")
@classmethod
def from_dict(
cls,
value: Any,
*,
artifact: ResearchRunArtifact,
run_ref: RetrospectiveBacktestRunRef,
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
) -> Self:
_performance_validate_tree(value, "$")
rebuilt = build_retrospective_performance_evidence(artifact, run_ref, evidence_manifest)
_performance_compare(value, rebuilt.to_dict(), "$")
return cast(Self, rebuilt)
@classmethod
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
_check(
type(value) in {str, bytes},
"$",
"canonical JSON text/bytes required",
ContractErrorCode.TYPE_ERROR,
)
raw = value.encode("utf-8") if isinstance(value, str) else value
try:
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
except (UnicodeDecodeError, json.JSONDecodeError) as error:
raise FactorContractError(
ContractErrorCode.INVALID_FORMAT, "$", "invalid performance evidence JSON"
) from error
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
_check(
_performance_canonical_bytes(document) == raw,
"$",
"canonical finite-number JSON required",
ContractErrorCode.INVALID_FORMAT,
)
return cls.from_dict(document, **kwargs)
def build_retrospective_performance_evidence(
artifact: ResearchRunArtifact,
run_ref: RetrospectiveBacktestRunRef,
evidence_manifest: RetrospectiveBacktestEvidenceManifest,
) -> RetrospectivePerformanceEvidence:
"""Bind current tables and existing methodology; no performance recalculation."""
run = _validated_run(run_ref)
_check(
type(evidence_manifest) is RetrospectiveBacktestEvidenceManifest,
"$.evidence_manifest",
"explicit v2 manifest required",
ContractErrorCode.TYPE_ERROR,
)
manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
evidence_manifest.to_dict(), artifact=artifact, backtest_run_ref=run
)
frames = _validated_frames(artifact, run)
performance = frames["performance"]
_check(
len(performance) == 1
and tuple(str(column) for column in performance.columns) == _PERFORMANCE_SOURCE_COLUMNS,
"$.artifact.tables.performance",
"one row in the unchanged closed performance schema required",
ContractErrorCode.ARTIFACT_MISMATCH,
)
performance_row = performance.iloc[0]
run_row = _run_row(frames)
frequency = _performance_text(run_row["frequency"], "$.artifact.tables.run.frequency")
_check(
frequency == "1d",
"$.artifact.tables.run.frequency",
"only existing daily methodology is supported",
)
calendar = _performance_text(run_row["calendar"], "$.artifact.tables.run.calendar")
timezone = _performance_text(run_row["timezone"], "$.artifact.tables.run.timezone")
start_date = _performance_date(run_row["start_date"], "$.artifact.tables.run.start_date")
end_date = _performance_date(run_row["end_date"], "$.artifact.tables.run.end_date")
nav = frames["nav"]
_check(
not nav.empty,
"$.artifact.tables.nav",
"NAV observation window required",
ContractErrorCode.ARTIFACT_MISMATCH,
)
_check(
_performance_date(nav.iloc[0]["trade_date"], "$.artifact.tables.nav.start") == start_date
and _performance_date(nav.iloc[-1]["trade_date"], "$.artifact.tables.nav.end") == end_date,
"$.artifact.tables.nav",
"observation window differs from artifact dates",
ContractErrorCode.ARTIFACT_MISMATCH,
)
benchmark_digest, active_std, benchmark_variance, alpha_domain_unestimable = _benchmark_context(
frames, run_row, performance_row
)
metrics = (
*_absolute_performance_metrics(performance_row),
*_relative_performance_metrics(
performance_row,
benchmark_present=benchmark_digest is not None,
active_std=active_std,
benchmark_variance=benchmark_variance,
alpha_domain_unestimable=alpha_domain_unestimable,
),
*_count_performance_metrics(performance_row),
)
normalized_row: dict[str, object] = {metric.source_column: metric.value for metric in metrics}
normalized_row["run_id"] = run.run_id
row_digest = _performance_digest(
{"columns": list(_PERFORMANCE_SOURCE_COLUMNS), "row": normalized_row}
)
alignment = cast(str, run_row["benchmark_alignment_policy"])
methodology = _performance_methodology(
frequency=frequency, alignment=alignment, code_revision=run.code_revision
)
performance_table = next(
table
for entry in manifest.evidence
for table in entry.tables
if table.logical_name == "performance"
)
run_document = run.to_dict()
payload: dict[str, Any] = {
"schema_version": "researchhub.performance-evidence.v2",
"authority": "quant_engine",
"scope": "offline_retrospective_research_only",
"run_id": run.run_id,
"usage": run.usage,
"historical_availability": run.historical_availability,
"evidence_scope": run.evidence_scope,
"observation_cutoff": run.observation_cutoff,
"decision_eligible": False,
"execution_validation": "not_validated",
"backtest_run_ref_id": run.run_id,
"backtest_run_ref_document_sha256": _digest_bytes(canonical_json_bytes(run_document)),
"backtest_evidence_manifest_id": manifest.manifest_id,
"backtest_evidence_manifest_document_sha256": _digest_bytes(
canonical_json_bytes(manifest.to_dict())
),
"backtest_evidence_manifest_evidence_digest": manifest.evidence_digest,
"backtest_evidence_qualification": manifest.qualification.value,
"research_artifact_schema_version": artifact.schema_version,
"research_artifact_content_digest": "sha256:" + artifact.content_sha256,
"artifact_available_at": manifest.artifact_available_at,
"computed_at": run.computed_at,
"performance_table_logical_name": performance_table.logical_name,
"performance_table_row_count": performance_table.row_count,
"performance_table_schema_digest": performance_table.schema_digest,
"performance_table_content_digest": performance_table.content_digest,
"performance_row_digest": row_digest,
"benchmark_series_digest": benchmark_digest,
"methodology_id": PERFORMANCE_METHODOLOGY_ID,
"metric_schema_id": PERFORMANCE_METRIC_SCHEMA_ID,
**{
key: run_document[key]
for key in (
"dataset_snapshot_id",
"dataset_content_digest",
"dataset_manifest_digest",
"foundation_id",
"foundation_digest",
"factor_set_id",
"factor_set_digest",
"factor_output_content_digest",
"strategy_id",
"strategy_version",
"strategy_digest",
"execution_model_version",
"execution_model_digest",
"cost_model_version",
"cost_model_digest",
"code_revision",
"environment_lock_digest",
"configuration_digest",
)
},
"frequency": frequency,
"calendar": calendar,
"timezone": timezone,
"benchmark_id": run_row["benchmark_id"],
"benchmark_alignment_policy": alignment,
"start_date": start_date,
"end_date": end_date,
"methodology": methodology.to_dict(),
"metrics": [metric.to_dict() for metric in metrics],
}
payload["performance_evidence_id"] = "rhperformancev2:" + _performance_digest(payload)
payload["document_sha256"] = _performance_digest(payload)
instance = object.__new__(RetrospectivePerformanceEvidence)
object.__setattr__(instance, "_payload", _freeze_numeric_evidence(payload))
object.__setattr__(instance, "methodology", methodology)
object.__setattr__(instance, "metrics", metrics)
return instance
@@ -0,0 +1,422 @@
"""Explicit retrospective v2 run identities and offline artifact evidence."""
from __future__ import annotations
from collections.abc import Mapping, Sequence
from dataclasses import dataclass, field
from typing import Any, Self, cast
from quant_engine.factor_contracts import (
ContractErrorCode,
PayloadValidation,
_assert_canonical_profile,
_content_address,
_digest,
_digest_bytes,
_freeze_json,
_git_revision,
_logical_id,
_parse_json_object,
_parse_utc,
_safe_integer,
_semver,
_thaw_json,
canonical_json,
canonical_json_bytes,
)
from quant_engine.retrospective_data_contracts import (
RetrospectiveFoundationEnvelope,
RetrospectiveSnapshotEnvelope,
_IDS,
_check,
_public,
_shape,
_strings,
)
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef, _context
_CONFIG_FIELDS = (
"universe_digest",
"strategy_id",
"strategy_version",
"strategy_digest",
"execution_model_version",
"execution_model_digest",
"cost_model_version",
"cost_model_digest",
"random_seed",
"code_revision",
"environment_lock_digest",
"configuration_digest",
)
_RUN_FIELDS = (
"contract_name schema_version run_id dataset_snapshot_id dataset_content_digest dataset_manifest_digest "
"foundation_id foundation_digest factor_set_id factor_set_digest factor_output_content_digest "
"observation_cutoff evidence_scope usage historical_availability decision_eligible execution_validation "
"universe_digest trading_calendar_revision_ids trading_calendar_digest corporate_action_revision_ids corporate_action_digest "
"strategy_id strategy_version strategy_digest execution_model_version execution_model_digest cost_model_version cost_model_digest "
"random_seed code_revision environment_lock_digest configuration_digest evaluation_at computed_at replay_spec_digest "
"replay_parent_run_id replay_reason replay_attempt replay_ancestor_run_ids"
)
@dataclass(frozen=True, slots=True, init=False)
class RetrospectiveBacktestRunRef:
"""New-major deterministic-input identity with separate actual attempt times."""
contract_name: str
schema_version: str
run_id: str
dataset_snapshot_id: str
dataset_content_digest: str
dataset_manifest_digest: str
foundation_id: str
foundation_digest: str
factor_set_id: str
factor_set_digest: str
factor_output_content_digest: str
observation_cutoff: str
evidence_scope: str
usage: str
historical_availability: str
decision_eligible: bool
execution_validation: str
universe_digest: str
trading_calendar_revision_ids: tuple[str, ...]
trading_calendar_digest: str
corporate_action_revision_ids: tuple[str, ...]
corporate_action_digest: str
strategy_id: str
strategy_version: str
strategy_digest: str
execution_model_version: str
execution_model_digest: str
cost_model_version: str
cost_model_digest: str
random_seed: int
code_revision: str
environment_lock_digest: str
configuration_digest: str
evaluation_at: str
computed_at: str
replay_spec_digest: str
replay_parent_run_id: str | None
replay_reason: str | None
replay_attempt: int
replay_ancestor_run_ids: tuple[str, ...]
input_payload_validation: PayloadValidation = field(compare=False)
_payload: Mapping[str, Any] = field(repr=False, compare=False)
_factor_set: RetrospectiveFactorSetRef = field(repr=False, compare=False)
_parent: RetrospectiveBacktestRunRef | None = field(repr=False, compare=False)
@classmethod
def create(
cls,
*,
dataset_snapshot: RetrospectiveSnapshotEnvelope,
foundation: RetrospectiveFoundationEnvelope,
factor_set: RetrospectiveFactorSetRef,
universe_digest: str,
trading_calendar_revision_ids: Sequence[str],
corporate_action_revision_ids: Sequence[str],
strategy_id: str,
strategy_version: str,
strategy_digest: str,
execution_model_version: str,
execution_model_digest: str,
cost_model_version: str,
cost_model_digest: str,
random_seed: int,
code_revision: str,
environment_lock_digest: str,
configuration_digest: str,
evaluation_at: str,
computed_at: str,
parent: RetrospectiveBacktestRunRef | None = None,
replay_reason: str | None = None,
replay_attempt: int = 0,
) -> Self:
_check(
type(factor_set) is RetrospectiveFactorSetRef,
"$.factor_set",
"explicit v2 factor result required",
ContractErrorCode.TYPE_ERROR,
)
factor_set.require_payloads_revalidated()
return cls._build(
dataset_snapshot=dataset_snapshot,
foundation=foundation,
factor_set=factor_set,
trading_calendar_revision_ids=trading_calendar_revision_ids,
corporate_action_revision_ids=corporate_action_revision_ids,
configuration={
"universe_digest": universe_digest,
"strategy_id": strategy_id,
"strategy_version": strategy_version,
"strategy_digest": strategy_digest,
"execution_model_version": execution_model_version,
"execution_model_digest": execution_model_digest,
"cost_model_version": cost_model_version,
"cost_model_digest": cost_model_digest,
"random_seed": random_seed,
"code_revision": code_revision,
"environment_lock_digest": environment_lock_digest,
"configuration_digest": configuration_digest,
},
evaluation_at=evaluation_at,
computed_at=computed_at,
parent=parent,
replay_reason=replay_reason,
replay_attempt=replay_attempt,
)
@classmethod
def _build(
cls,
*,
dataset_snapshot: Any,
foundation: Any,
factor_set: Any,
trading_calendar_revision_ids: Any,
corporate_action_revision_ids: Any,
configuration: dict[str, Any],
evaluation_at: Any,
computed_at: Any,
parent: RetrospectiveBacktestRunRef | None,
replay_reason: Any,
replay_attempt: Any,
) -> Self:
_check(
type(factor_set) is RetrospectiveFactorSetRef,
"$.factor_set",
"explicit v2 factor result required",
ContractErrorCode.TYPE_ERROR,
)
definitions, snapshot, foundation = _context(
factor_set._definitions, dataset_snapshot, foundation
)
# Reconstruct the serialized factor boundary against the exact supplied inputs.
checked_factor = RetrospectiveFactorSetRef.from_dict(
factor_set.to_dict(),
definitions=definitions,
dataset_snapshot=snapshot,
foundation=foundation,
parent=factor_set._parent,
)
closures: dict[str, tuple[str, ...]] = {}
for field_name, supplied, kind in (
("trading_calendar_revision_ids", trading_calendar_revision_ids, "calendar_revision"),
("corporate_action_revision_ids", corporate_action_revision_ids, "action_revision"),
):
_check(
type(supplied) in {tuple, list},
f"$.{field_name}",
"list/tuple required",
ContractErrorCode.TYPE_ERROR,
)
supplied_ids = tuple(
sorted(
_strings(
list(supplied),
f"$.{field_name}",
_IDS[kind],
1 if kind == "calendar_revision" else 0,
)
)
)
expected_ids = tuple(
sorted(
{
identity
for view_id in checked_factor.selected_view_ref_ids
for identity in getattr(foundation.views[view_id], field_name)
}
)
)
_check(
supplied_ids == expected_ids,
f"$.{field_name}",
"exact selected observation ancestry required",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
closures[field_name] = supplied_ids
_shape(configuration, "$.configuration", " ".join(_CONFIG_FIELDS))
for name, value in configuration.items():
if name.endswith("_digest"):
_digest(value, f"$.{name}")
elif name.endswith("_version"):
_semver(value, f"$.{name}")
elif name == "random_seed":
_safe_integer(value, f"$.{name}", minimum=0)
elif name == "code_revision":
_git_revision(value, f"$.{name}")
else:
_logical_id(value, f"$.{name}")
_public(configuration, "$.configuration")
evaluation = _parse_utc(evaluation_at, "$.evaluation_at")
computed = _parse_utc(computed_at, "$.computed_at")
_check(
_parse_utc(checked_factor.artifact_available_at, "$.factor_set.artifact_available_at")
<= evaluation
<= computed,
"$.computed_at",
"factor availability <= actual evaluation <= computation required",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
content = snapshot.to_dict()["descriptor"]["content"]
spec: dict[str, Any] = {
"dataset_snapshot_id": snapshot.snapshot_id,
"dataset_content_digest": content["content_digest"],
"dataset_manifest_digest": content["manifest_digest"],
"foundation_id": foundation.foundation_id,
"foundation_digest": foundation.foundation_id.removeprefix("rhdfv2:"),
"factor_set_id": checked_factor.factor_set_id,
"factor_set_digest": checked_factor.factor_set_id.removeprefix("rhfactorsetv2:"),
"factor_output_content_digest": checked_factor.output_content_digest,
"observation_cutoff": foundation.observation_cutoff,
"evidence_scope": checked_factor.evidence_scope,
"usage": "retrospective_research",
"historical_availability": "not_established",
"decision_eligible": False,
"execution_validation": "not_validated",
**configuration,
}
for field_name, identities in closures.items():
spec[field_name] = list(identities)
digest_field = (
"trading_calendar_digest"
if field_name == "trading_calendar_revision_ids"
else "corporate_action_digest"
)
spec[digest_field] = _digest_bytes(canonical_json_bytes(list(identities)))
# v2 replay specification excludes BOTH actual attempt times. They remain in
# run_id, so a replay never backdates evaluation to manufacture equality.
replay_spec_digest = _digest_bytes(canonical_json_bytes(spec))
replay_count = _safe_integer(replay_attempt, "$.replay_attempt", minimum=0)
if parent is None:
_check(
replay_reason is None and replay_count == 0,
"$.replay_attempt",
"root must use zero attempt and no reason",
ContractErrorCode.LINEAGE_VIOLATION,
)
parent_id = None
ancestors: tuple[str, ...] = ()
else:
_check(
type(parent) is RetrospectiveBacktestRunRef,
"$.parent",
"exact v2 run parent required",
ContractErrorCode.TYPE_ERROR,
)
_logical_id(replay_reason, "$.replay_reason")
_check(
replay_count == parent.replay_attempt + 1
and replay_spec_digest == parent.replay_spec_digest,
"$.replay_spec_digest",
"replay requires unchanged inputs and the next attempt",
ContractErrorCode.LINEAGE_VIOLATION,
)
_check(
_parse_utc(parent.computed_at, "$.parent.computed_at") < evaluation <= computed,
"$.evaluation_at",
"new actual attempt must follow parent computation",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
parent_id = parent.run_id
ancestors = (*parent.replay_ancestor_run_ids, parent_id)
_check(
len(ancestors) == len(set(ancestors)),
"$.replay_ancestor_run_ids",
"replay cycle",
ContractErrorCode.LINEAGE_VIOLATION,
)
payload = {
"contract_name": "researchhub.backtest-run-ref",
"schema_version": "2.0.0",
**spec,
"evaluation_at": evaluation_at,
"computed_at": computed_at,
"replay_spec_digest": replay_spec_digest,
"replay_parent_run_id": parent_id,
"replay_reason": replay_reason,
"replay_attempt": replay_count,
"replay_ancestor_run_ids": list(ancestors),
}
payload["run_id"] = _content_address(payload, "run_id", "rhbacktestrunv2:sha256:")
_check(
payload["run_id"] not in ancestors,
"$.run_id",
"self-parent cycle",
ContractErrorCode.LINEAGE_VIOLATION,
)
verified = (
factor_set.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
and factor_set.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
)
instance = object.__new__(cls)
for name, value in {
**payload,
**closures,
"replay_ancestor_run_ids": ancestors,
"input_payload_validation": PayloadValidation.PAYLOAD_REVALIDATED
if verified
else PayloadValidation.REFERENCE_ONLY,
"_payload": _freeze_json(payload),
"_factor_set": factor_set,
"_parent": parent,
}.items():
object.__setattr__(instance, name, value)
return instance
def to_dict(self) -> dict[str, Any]:
return cast(dict[str, Any], _thaw_json(self._payload))
def to_json(self) -> str:
return canonical_json(self.to_dict())
def require_inputs_revalidated(self) -> None:
_check(
self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
"$.input_payload_validation",
"reference-only factors cannot admit a new computation",
ContractErrorCode.ARTIFACT_MISMATCH,
)
@classmethod
def from_dict(
cls,
value: Any,
*,
dataset_snapshot: RetrospectiveSnapshotEnvelope,
foundation: RetrospectiveFoundationEnvelope,
factor_set: RetrospectiveFactorSetRef,
parent: RetrospectiveBacktestRunRef | None = None,
) -> Self:
_assert_canonical_profile(value)
_public(value)
row = _shape(value, "$", _RUN_FIELDS)
rebuilt = cls._build(
dataset_snapshot=dataset_snapshot,
foundation=foundation,
factor_set=factor_set,
trading_calendar_revision_ids=row["trading_calendar_revision_ids"],
corporate_action_revision_ids=row["corporate_action_revision_ids"],
configuration={key: row[key] for key in _CONFIG_FIELDS},
evaluation_at=row["evaluation_at"],
computed_at=row["computed_at"],
parent=parent,
replay_reason=row["replay_reason"],
replay_attempt=row["replay_attempt"],
)
_check(
canonical_json_bytes(row) == canonical_json_bytes(rebuilt.to_dict()),
"$",
"serialized run differs from exact v2 input/configuration/lineage closure",
ContractErrorCode.IDENTITY_MISMATCH,
)
return rebuilt
@classmethod
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,721 @@
"""Observation-aware factor results; no historical, governance or execution grant."""
from __future__ import annotations
import json
import re
from collections.abc import Mapping, Sequence
from dataclasses import dataclass, field
from typing import Any, Self, cast
from quant_engine.factor_contracts import (
ActorIdentity,
ContractErrorCode,
FactorDefinition,
OutputArtifactRef,
OutputCoverage,
OutputQuality,
PayloadValidation,
ProducerIdentity,
_DEFINITION_ID,
_FIELD_NAME,
_array,
_assert_canonical_profile,
_canonical_evidence_bytes,
_content_address,
_digest,
_digest_bytes,
_freeze_json,
_git_revision,
_logical_id,
_parse_json_object,
_parse_utc,
_string,
_thaw_json,
canonical_json,
canonical_json_bytes,
validate_factor_catalog,
)
from quant_engine.retrospective_data_contracts import (
RetrospectiveFoundationEnvelope,
RetrospectiveSnapshotEnvelope,
_IDS,
_check,
_choice,
_public,
_restrictions,
_shape,
_strings,
)
_FACTOR_SET_ID = re.compile(r"^rhfactorsetv2:sha256:[0-9a-f]{64}$")
def _instant_text(value: Any, path: str) -> str:
_parse_utc(value, path)
return cast(str, value)
@dataclass(frozen=True, slots=True)
class RetrospectiveInputBinding:
definition_id: str
input_name: str
view_ref_id: str
schema_digest: str
def __post_init__(self) -> None:
_string(self.definition_id, "$.input_bindings[].definition_id", _DEFINITION_ID)
_string(self.input_name, "$.input_bindings[].input_name", _FIELD_NAME)
_string(self.view_ref_id, "$.input_bindings[].view_ref_id", _IDS["view_ref"])
_digest(self.schema_digest, "$.input_bindings[].schema_digest")
def to_dict(self) -> dict[str, Any]:
return {
"definition_id": self.definition_id,
"input_name": self.input_name,
"view_ref_id": self.view_ref_id,
"schema_digest": self.schema_digest,
}
@classmethod
def from_dict(cls, value: Any, path: str = "$.input_bindings[]") -> Self:
row = _shape(value, path, "definition_id input_name view_ref_id schema_digest")
return cls(**row)
@dataclass(frozen=True, slots=True)
class RetrospectiveViewAvailability:
view_ref_id: str
available_at: str
evidence_digest: str
def __post_init__(self) -> None:
_string(self.view_ref_id, "$.view_availability[].view_ref_id", _IDS["view_ref"])
_instant_text(self.available_at, "$.view_availability[].available_at")
_digest(self.evidence_digest, "$.view_availability[].evidence_digest")
def to_dict(self) -> dict[str, Any]:
return {
"view_ref_id": self.view_ref_id,
"available_at": self.available_at,
"evidence_digest": self.evidence_digest,
}
@classmethod
def from_dict(cls, value: Any, path: str = "$.view_availability[]") -> Self:
return cls(**_shape(value, path, "view_ref_id available_at evidence_digest"))
@dataclass(frozen=True, slots=True)
class RetrospectiveCausation:
kind: str
id: str
def __post_init__(self) -> None:
kind = _choice(self.kind, "$.causation.kind", {"foundation", "factor_set"})
_string(
self.id,
"$.causation.id",
_IDS["foundation"] if kind == "foundation" else _FACTOR_SET_ID,
)
def to_dict(self) -> dict[str, Any]:
return {"kind": self.kind, "id": self.id}
@classmethod
def from_dict(cls, value: Any) -> Self:
return cls(**_shape(value, "$.causation", "kind id"))
@dataclass(frozen=True, slots=True)
class ResolvedRetrospectiveView:
"""In-memory logical bytes; no locator, source authentication or transformation claim."""
view_ref_id: str
schema_bytes: bytes
content_bytes: bytes
def __post_init__(self) -> None:
_string(self.view_ref_id, "$.resolved_views[].view_ref_id", _IDS["view_ref"])
for key, value in (
("schema_bytes", self.schema_bytes),
("content_bytes", self.content_bytes),
):
_canonical_evidence_bytes(value, f"$.resolved_views[].{key}")
_public(json.loads(value), f"$.resolved_views[].{key}")
def _typed(values: Any, expected: type[Any], path: str) -> tuple[Any, ...]:
_check(
type(values) in {tuple, list},
path,
"typed list/tuple required",
ContractErrorCode.TYPE_ERROR,
)
_check(
all(type(value) is expected for value in values),
path,
f"{expected.__name__} required",
ContractErrorCode.TYPE_ERROR,
)
return tuple(values)
def _context(
definitions: Sequence[FactorDefinition],
snapshot: Any,
foundation: Any,
) -> tuple[
tuple[FactorDefinition, ...], RetrospectiveSnapshotEnvelope, RetrospectiveFoundationEnvelope
]:
_check(
type(snapshot) is RetrospectiveSnapshotEnvelope,
"$.dataset_snapshot",
"explicit v2 snapshot required",
ContractErrorCode.TYPE_ERROR,
)
_check(
type(foundation) is RetrospectiveFoundationEnvelope,
"$.foundation",
"explicit v2 foundation required",
ContractErrorCode.TYPE_ERROR,
)
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot.to_dict())
snapshot.require_qualified()
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation.to_dict(), snapshot=snapshot)
supplied = _typed(definitions, FactorDefinition, "$.definitions")
# Definitions stay v1, but are parsed again so mutable/caller summaries are not authority.
normalized = validate_factor_catalog(
tuple(FactorDefinition.from_dict(item.to_dict()) for item in supplied)
)
return normalized, snapshot, foundation
def _upstream(
snapshot: RetrospectiveSnapshotEnvelope, foundation: RetrospectiveFoundationEnvelope
) -> dict[str, Any]:
descriptor = snapshot.to_dict()["descriptor"]
return {
"dataset_snapshot_id": snapshot.snapshot_id,
"foundation_id": foundation.foundation_id,
"evidence_scope": snapshot.evidence_scope,
"content_digest": descriptor["content"]["content_digest"],
"manifest_digest": descriptor["content"]["manifest_digest"],
"observation_manifest_digest": _digest_bytes(
canonical_json_bytes(descriptor["observation_manifest"])
),
"time_semantics": descriptor["time_semantics"],
"quality": descriptor["quality"],
"qualification": descriptor["qualification"],
"foundation_readiness": foundation.to_dict()["readiness"],
}
def _input_payloads(
snapshot: RetrospectiveSnapshotEnvelope,
foundation: RetrospectiveFoundationEnvelope,
selected: tuple[str, ...],
dataset_chunks: Any,
resolved_views: Sequence[ResolvedRetrospectiveView] | None,
) -> PayloadValidation:
_check(
(dataset_chunks is None) == (resolved_views is None),
"$.input_payloads",
"snapshot chunks and resolved views must be supplied together",
ContractErrorCode.ARTIFACT_MISMATCH,
)
if dataset_chunks is None:
return PayloadValidation.REFERENCE_ONLY
snapshot.verify_materialized_records(dataset_chunks)
views = _typed(resolved_views, ResolvedRetrospectiveView, "$.resolved_views")
view_ids = [view.view_ref_id for view in views]
_check(
len(view_ids) == len(selected) and set(view_ids) == set(selected),
"$.resolved_views",
"resolved view closure mismatch",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
for item in views:
declared = foundation.views[item.view_ref_id]
# Recheck canonical bytes even for caller-constructed typed payloads.
schema = _canonical_evidence_bytes(item.schema_bytes, "$.resolved_views[].schema_bytes")
content = _canonical_evidence_bytes(item.content_bytes, "$.resolved_views[].content_bytes")
_public(json.loads(schema))
_public(json.loads(content))
_check(
_digest_bytes(schema) == declared.schema_digest
and _digest_bytes(content) == declared.content_digest,
"$.resolved_views",
"view bytes do not match Foundation",
ContractErrorCode.ARTIFACT_MISMATCH,
)
return PayloadValidation.PAYLOAD_REVALIDATED
@dataclass(frozen=True, slots=True, init=False)
class RetrospectiveFactorSetRef:
contract_name: str
schema_version: str
factor_set_id: str
definition_ids: tuple[str, ...]
dataset_snapshot_id: str
foundation_id: str
observation_cutoff: str
selected_view_ref_ids: tuple[str, ...]
input_bindings: tuple[RetrospectiveInputBinding, ...]
view_availability: tuple[RetrospectiveViewAvailability, ...]
upstream_evidence: Mapping[str, Any]
output_quality: OutputQuality
output_coverage: OutputCoverage
output_schema_digest: str
output_content_digest: str
output_artifact_ref: OutputArtifactRef
availability_mode: str
usage: str
historical_availability: str
evaluation_at: str
computed_at: str
artifact_available_at: str
producer: ProducerIdentity
code_revision: str
actor: ActorIdentity
correlation_id: str
causation: RetrospectiveCausation
evidence_scope: str
decision_eligible: bool
payload_validation: PayloadValidation = field(compare=False)
input_payload_validation: PayloadValidation = field(compare=False)
_payload: Mapping[str, Any] = field(repr=False, compare=False)
_definitions: tuple[FactorDefinition, ...] = field(repr=False, compare=False)
_dataset_snapshot: RetrospectiveSnapshotEnvelope = field(repr=False, compare=False)
_foundation: RetrospectiveFoundationEnvelope = field(repr=False, compare=False)
_parent: RetrospectiveFactorSetRef | None = field(repr=False, compare=False)
@classmethod
def create(
cls,
*,
definitions: Sequence[FactorDefinition],
dataset_snapshot: RetrospectiveSnapshotEnvelope,
foundation: RetrospectiveFoundationEnvelope,
selected_view_ref_ids: Sequence[str],
input_bindings: Sequence[RetrospectiveInputBinding],
view_availability: Sequence[RetrospectiveViewAvailability],
dataset_chunks: Any,
resolved_views: Sequence[ResolvedRetrospectiveView],
output_quality: OutputQuality,
output_coverage: OutputCoverage,
output_schema_bytes: bytes,
output_content_bytes: bytes,
output_artifact_ref: OutputArtifactRef,
evaluation_at: str,
computed_at: str,
artifact_available_at: str,
producer: ProducerIdentity,
code_revision: str,
actor: ActorIdentity,
correlation_id: str,
causation: RetrospectiveCausation,
evidence_scope: str,
decision_eligible: bool,
parent: RetrospectiveFactorSetRef | None = None,
) -> Self:
definitions, dataset_snapshot, foundation = _context(
definitions, dataset_snapshot, foundation
)
_check(
type(selected_view_ref_ids) in {list, tuple},
"$.selected_view_ref_ids",
"list/tuple required",
ContractErrorCode.TYPE_ERROR,
)
selected = sorted(
_strings(list(selected_view_ref_ids), "$.selected_view_ref_ids", _IDS["view_ref"], 1)
)
bindings = sorted(
_typed(input_bindings, RetrospectiveInputBinding, "$.input_bindings"),
key=lambda item: (item.definition_id, item.input_name),
)
availability = sorted(
_typed(view_availability, RetrospectiveViewAvailability, "$.view_availability"),
key=lambda item: item.view_ref_id,
)
for value, expected, path in (
(output_quality, OutputQuality, "$.output_quality"),
(output_coverage, OutputCoverage, "$.output_coverage"),
(output_artifact_ref, OutputArtifactRef, "$.output_artifact_ref"),
(producer, ProducerIdentity, "$.producer"),
(actor, ActorIdentity, "$.actor"),
(causation, RetrospectiveCausation, "$.causation"),
):
_check(
type(value) is expected,
path,
f"{expected.__name__} required",
ContractErrorCode.TYPE_ERROR,
)
schema_digest = _digest_bytes(
_canonical_evidence_bytes(output_schema_bytes, "$.output_schema_bytes")
)
content_digest = _digest_bytes(
_canonical_evidence_bytes(output_content_bytes, "$.output_content_bytes")
)
document = {
"contract_name": "researchhub.factor-set-ref",
"schema_version": "2.0.0",
"definition_ids": [definition.definition_id for definition in definitions],
"dataset_snapshot_id": dataset_snapshot.snapshot_id,
"foundation_id": foundation.foundation_id,
"observation_cutoff": foundation.observation_cutoff,
"selected_view_ref_ids": selected,
"input_bindings": [item.to_dict() for item in bindings],
"view_availability": [item.to_dict() for item in availability],
"upstream_evidence": _upstream(dataset_snapshot, foundation),
"output_quality": output_quality.to_dict(),
"output_coverage": output_coverage.to_dict(),
"output_schema_digest": schema_digest,
"output_content_digest": content_digest,
"output_artifact_ref": output_artifact_ref.to_dict(),
"availability_mode": "retrospective_replay",
"usage": "retrospective_research",
"historical_availability": "not_established",
"evaluation_at": evaluation_at,
"computed_at": computed_at,
"artifact_available_at": artifact_available_at,
"producer": producer.to_dict(),
"code_revision": code_revision,
"actor": actor.to_dict(),
"correlation_id": correlation_id,
"causation": causation.to_dict(),
"evidence_scope": evidence_scope,
"decision_eligible": decision_eligible,
}
document["factor_set_id"] = _content_address(
document, "factor_set_id", "rhfactorsetv2:sha256:"
)
result = cls.from_dict(
document,
definitions=definitions,
dataset_snapshot=dataset_snapshot,
foundation=foundation,
parent=parent,
output_schema_bytes=output_schema_bytes,
output_content_bytes=output_content_bytes,
dataset_chunks=dataset_chunks,
resolved_views=resolved_views,
)
result.require_payloads_revalidated()
return result
@classmethod
def from_dict(
cls,
value: Any,
*,
definitions: Sequence[FactorDefinition],
dataset_snapshot: RetrospectiveSnapshotEnvelope,
foundation: RetrospectiveFoundationEnvelope,
parent: RetrospectiveFactorSetRef | None = None,
output_schema_bytes: bytes | None = None,
output_content_bytes: bytes | None = None,
dataset_chunks: Any = None,
resolved_views: Sequence[ResolvedRetrospectiveView] | None = None,
) -> Self:
definitions, dataset_snapshot, foundation = _context(
definitions, dataset_snapshot, foundation
)
_assert_canonical_profile(value)
_public(value)
row = _shape(
value,
"$",
"contract_name schema_version factor_set_id definition_ids dataset_snapshot_id foundation_id observation_cutoff "
"selected_view_ref_ids input_bindings view_availability upstream_evidence output_quality output_coverage "
"output_schema_digest output_content_digest output_artifact_ref availability_mode usage historical_availability "
"evaluation_at computed_at artifact_available_at producer code_revision actor correlation_id causation evidence_scope decision_eligible",
)
_choice(row["contract_name"], "$.contract_name", {"researchhub.factor-set-ref"})
_choice(row["schema_version"], "$.schema_version", {"2.0.0"})
_choice(row["availability_mode"], "$.availability_mode", {"retrospective_replay"})
_restrictions(row, "$")
_check(
type(row["decision_eligible"]) is bool and not row["decision_eligible"],
"$.decision_eligible",
"computation is never decision eligible",
ContractErrorCode.READINESS_ESCALATION,
)
_check(
row["dataset_snapshot_id"] == dataset_snapshot.snapshot_id
and row["foundation_id"] == foundation.foundation_id
and row["observation_cutoff"] == foundation.observation_cutoff,
"$.foundation_id",
"exact snapshot/foundation/cutoff required",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
definition_ids = _strings(row["definition_ids"], "$.definition_ids", _DEFINITION_ID, 1)
_check(
definition_ids == tuple(item.definition_id for item in definitions),
"$.definition_ids",
"normalized exact definitions required",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
selected = _strings(
row["selected_view_ref_ids"], "$.selected_view_ref_ids", _IDS["view_ref"], 1
)
_check(
tuple(sorted(selected)) == selected and set(selected) <= foundation.views.keys(),
"$.selected_view_ref_ids",
"unknown/unnormalized selected views",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
bindings = tuple(
RetrospectiveInputBinding.from_dict(item)
for item in _array(row["input_bindings"], "$.input_bindings", minimum=1, unique=True)
)
keys = [(item.definition_id, item.input_name) for item in bindings]
expected = {
(item.definition_id, input_spec.input_name): input_spec
for item in definitions
for input_spec in item.inputs
}
_check(
len(keys) == len(expected) and set(keys) == expected.keys() and keys == sorted(keys),
"$.input_bindings",
"exact normalized factor input closure required",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
for binding in bindings:
_check(
binding.view_ref_id in selected,
"$.input_bindings",
"unselected view",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
_check(
binding.schema_digest
== expected[(binding.definition_id, binding.input_name)].schema_digest
== foundation.views[binding.view_ref_id].schema_digest,
"$.input_bindings",
"schema mismatch",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
_check(
{item.view_ref_id for item in bindings} == set(selected),
"$.selected_view_ref_ids",
"unused selected view",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
availability = tuple(
RetrospectiveViewAvailability.from_dict(item)
for item in _array(
row["view_availability"], "$.view_availability", minimum=1, unique=True
)
)
_check(
tuple(item.view_ref_id for item in availability) == selected,
"$.view_availability",
"exact normalized selected view availability required",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
for item in availability:
_check(
item.available_at == foundation.views[item.view_ref_id].available_at,
"$.view_availability",
"availability must equal its Foundation fact",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
upstream = _upstream(dataset_snapshot, foundation)
_check(
canonical_json_bytes(row["upstream_evidence"]) == canonical_json_bytes(upstream),
"$.upstream_evidence",
"upstream evidence differs from complete input envelopes",
ContractErrorCode.IDENTITY_MISMATCH,
)
_check(
row["evidence_scope"] == dataset_snapshot.evidence_scope == foundation.evidence_scope,
"$.evidence_scope",
"scope must equal both inputs",
ContractErrorCode.READINESS_ESCALATION,
)
if row["evidence_scope"] == "real_data":
_check(
foundation.real_data_validation_status == "validated",
"$.evidence_scope",
"real-data Foundation validation required",
ContractErrorCode.READINESS_ESCALATION,
)
quality = OutputQuality.from_dict(row["output_quality"])
coverage = OutputCoverage.from_dict(row["output_coverage"])
_check(
quality.status == "passed" and all(item.status == "passed" for item in quality.checks),
"$.output_quality",
"all output checks must pass",
)
_check(
coverage.status == "complete" and coverage.observed_count == coverage.expected_count,
"$.output_coverage",
"complete output coverage required",
)
artifact = OutputArtifactRef.from_dict(row["output_artifact_ref"])
schema_digest = _digest(row["output_schema_digest"], "$.output_schema_digest")
content_digest = _digest(row["output_content_digest"], "$.output_content_digest")
_check(
artifact.schema_digest == schema_digest and artifact.content_digest == content_digest,
"$.output_artifact_ref",
"output artifact mismatch",
ContractErrorCode.ARTIFACT_MISMATCH,
)
_check(
(output_schema_bytes is None) == (output_content_bytes is None),
"$.output_artifact_ref",
"both output payloads required together",
ContractErrorCode.ARTIFACT_MISMATCH,
)
validation = PayloadValidation.REFERENCE_ONLY
if output_schema_bytes is not None and output_content_bytes is not None:
for data, expected_digest, path in (
(output_schema_bytes, schema_digest, "$.output_schema_bytes"),
(output_content_bytes, content_digest, "$.output_content_bytes"),
):
canonical = _canonical_evidence_bytes(data, path)
_public(json.loads(canonical), path)
_check(
_digest_bytes(canonical) == expected_digest,
path,
"output bytes mismatch",
ContractErrorCode.ARTIFACT_MISMATCH,
)
validation = PayloadValidation.PAYLOAD_REVALIDATED
input_validation = _input_payloads(
dataset_snapshot, foundation, selected, dataset_chunks, resolved_views
)
evaluation = _parse_utc(row["evaluation_at"], "$.evaluation_at")
computed = _parse_utc(row["computed_at"], "$.computed_at")
available = _parse_utc(row["artifact_available_at"], "$.artifact_available_at")
_check(
foundation.published_at <= evaluation <= computed <= available,
"$.computed_at",
"input publication <= actual evaluation <= computation <= artifact required",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
for definition in definitions:
_check(
_parse_utc(definition.valid_from, "$.definitions[].valid_from")
<= evaluation
< _parse_utc(definition.valid_until, "$.definitions[].valid_until"),
"$.definitions",
"factor definition is not valid at actual evaluation",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
producer = ProducerIdentity.from_dict(row["producer"])
_check(
producer.id == "quant_engine",
"$.producer.id",
"computation owner must be quant_engine",
ContractErrorCode.LINEAGE_VIOLATION,
)
_git_revision(row["code_revision"], "$.code_revision")
actor = ActorIdentity.from_dict(row["actor"])
correlation = _logical_id(row["correlation_id"], "$.correlation_id")
cause = RetrospectiveCausation.from_dict(row["causation"])
for name, parsed in (
("output_quality", quality),
("output_coverage", coverage),
("output_artifact_ref", artifact),
("producer", producer),
("actor", actor),
("causation", cause),
):
_check(
canonical_json_bytes(row[name]) == canonical_json_bytes(parsed.to_dict()),
f"$.{name}",
"nested contract is not normalized",
ContractErrorCode.INVALID_FORMAT,
)
if cause.kind == "foundation":
_check(
cause.id == foundation.foundation_id and parent is None,
"$.causation",
"exact Foundation cause required",
ContractErrorCode.LINEAGE_VIOLATION,
)
else:
_check(
type(parent) is RetrospectiveFactorSetRef,
"$.causation",
"exact v2 parent object required",
ContractErrorCode.LINEAGE_VIOLATION,
)
assert parent is not None
_check(
cause.id == parent.factor_set_id
and correlation == parent.correlation_id
and row["evidence_scope"] == parent.evidence_scope,
"$.causation",
"parent identity/correlation/scope mismatch",
ContractErrorCode.LINEAGE_VIOLATION,
)
_check(
_parse_utc(parent.artifact_available_at, "$.parent.artifact_available_at")
<= evaluation,
"$.causation",
"parent artifact postdates child evaluation",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
factor_set_id = _string(row["factor_set_id"], "$.factor_set_id", _FACTOR_SET_ID)
_check(
factor_set_id == _content_address(row, "factor_set_id", "rhfactorsetv2:sha256:"),
"$.factor_set_id",
"factor result identity mismatch",
ContractErrorCode.IDENTITY_MISMATCH,
)
_check(
cause.id != factor_set_id,
"$.causation",
"self parent is forbidden",
ContractErrorCode.LINEAGE_VIOLATION,
)
instance = object.__new__(cls)
values = {
**row,
"definition_ids": definition_ids,
"selected_view_ref_ids": selected,
"input_bindings": bindings,
"view_availability": availability,
"upstream_evidence": _freeze_json(upstream),
"output_quality": quality,
"output_coverage": coverage,
"output_artifact_ref": artifact,
"producer": producer,
"actor": actor,
"causation": cause,
"payload_validation": validation,
"input_payload_validation": input_validation,
"_payload": _freeze_json(row),
"_definitions": definitions,
"_dataset_snapshot": dataset_snapshot,
"_foundation": foundation,
"_parent": parent,
}
for name, item in values.items():
object.__setattr__(instance, name, item)
return instance
def require_payloads_revalidated(self) -> None:
_check(
self.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
and self.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED,
"$.payload_validation",
"reference-only data is not computation admission",
ContractErrorCode.ARTIFACT_MISMATCH,
)
def to_dict(self) -> dict[str, Any]:
return cast(dict[str, Any], _thaw_json(self._payload))
def to_json(self) -> str:
return canonical_json(self.to_dict())
@classmethod
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
return cls.from_dict(_parse_json_object(value, "$"), **kwargs)
@@ -0,0 +1,995 @@
"""Retrospective-only portfolio/risk evidence with separate business/actual clocks."""
from __future__ import annotations
import json
import math
import re
from collections.abc import Mapping
from dataclasses import dataclass, field
from types import MappingProxyType
from typing import Any, Self, TypedDict, cast
import pandas as pd
from quant_engine.artifact import (
EvidenceQualification,
_performance_compare,
_performance_validate_tree,
)
from quant_engine.factor_contracts import (
ContractErrorCode,
FactorContractError,
_duplicate_key_pairs,
_parse_utc,
_string,
_thaw_json,
)
from quant_engine.portfolio_risk_contracts import (
ComputationReceipt,
ConstraintSetV1,
FreshnessPolicy,
ReceiptStatus,
PortfolioRiskContractError,
PortfolioRiskContractErrorCode,
RiskAssessmentStatus,
RiskFindingCode,
_CLOSURE_ATOL,
_CLOSURE_RTOL,
_finite_number,
_series_mapping,
_validate_covariance_structure,
_canonical_json,
_constraint_metrics,
_constraint_residuals,
_digest,
_document_sha256,
_immutable_float_mapping,
_mapping_dict,
_payload_digest,
_semver,
_text,
)
from quant_engine.retrospective_artifact_contracts import (
RetrospectiveBacktestEvidenceManifest,
_freeze_numeric_evidence,
_validated_run,
)
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
from quant_engine.retrospective_data_contracts import _IDS, _check, _public, _shape
from quant_engine.risk import CovarianceSnapshot, labeled_component_risk
_RUN_ID = re.compile(r"^rhbacktestrunv2:sha256:[0-9a-f]{64}$")
def _json_object(value: str | bytes) -> dict[str, Any]:
_check(
type(value) in {str, bytes},
"$",
"canonical JSON text/bytes required",
ContractErrorCode.TYPE_ERROR,
)
try:
raw = value.encode("utf-8") if isinstance(value, str) else value
document = json.loads(raw, object_pairs_hook=_duplicate_key_pairs)
except (json.JSONDecodeError, UnicodeError) as error:
raise FactorContractError(
ContractErrorCode.INVALID_FORMAT, "$", "valid UTF-8 JSON required"
) from error
_check(type(document) is dict, "$", "object required", ContractErrorCode.TYPE_ERROR)
_performance_validate_tree(document, "$")
_check(
_canonical_json(document).encode() == raw,
"$",
"canonical numeric JSON required",
ContractErrorCode.INVALID_FORMAT,
)
return cast(dict[str, Any], document)
@dataclass(frozen=True, slots=True, init=False)
class RetrospectivePortfolioTarget:
contract_name: str
schema_version: str
target_id: str
backtest_run_id: str
dataset_snapshot_id: str
weights: Mapping[str, float]
effective_at: str
created_at: str
usage: str
historical_availability: str
_payload: Mapping[str, Any] = field(repr=False, compare=False)
@classmethod
def create(
cls,
*,
backtest_run_id: str,
dataset_snapshot_id: str,
weights: Mapping[str, float],
effective_at: str,
created_at: str,
) -> Self:
_string(backtest_run_id, "$.backtest_run_id", _RUN_ID)
_string(dataset_snapshot_id, "$.dataset_snapshot_id", _IDS["snapshot"])
normalized = _immutable_float_mapping(weights, "$.weights")
_check(bool(normalized), "$.weights", "non-empty target asset set required")
for instrument in normalized:
_string(instrument, "$.weights.keys", _IDS["instrument"])
effective = _parse_utc(effective_at, "$.effective_at")
created = _parse_utc(created_at, "$.created_at")
_check(
effective <= created,
"$.effective_at",
"historical effective time exceeds actual creation",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
payload = {
"contract_name": "researchhub.portfolio-target",
"schema_version": "2.0.0",
"backtest_run_id": backtest_run_id,
"dataset_snapshot_id": dataset_snapshot_id,
"weights": _mapping_dict(normalized),
"effective_at": effective_at,
"created_at": created_at,
"usage": "retrospective_research",
"historical_availability": "not_established",
}
_public(payload)
payload["target_id"] = "rhportfoliotargetv2:" + _payload_digest(payload)
instance = object.__new__(cls)
for key, value in {
**payload,
"weights": normalized,
"_payload": _freeze_numeric_evidence(payload),
}.items():
object.__setattr__(instance, key, value)
return instance
def to_dict(self) -> dict[str, Any]:
return cast(dict[str, Any], _thaw_json(self._payload))
def to_json(self) -> str:
return _canonical_json(self.to_dict())
@classmethod
def from_dict(cls, value: Any) -> Self:
_performance_validate_tree(value, "$")
row = _shape(
value,
"$",
"contract_name schema_version target_id backtest_run_id dataset_snapshot_id weights effective_at created_at usage historical_availability",
)
rebuilt = cls.create(
**{
key: row[key]
for key in (
"backtest_run_id",
"dataset_snapshot_id",
"weights",
"effective_at",
"created_at",
)
}
)
_performance_compare(row, rebuilt.to_dict(), "$")
return rebuilt
@classmethod
def from_json(cls, value: str | bytes) -> Self:
return cls.from_dict(_json_object(value))
@dataclass(frozen=True, slots=True)
class _PortfolioInputs:
run: RetrospectiveBacktestRunRef
manifest: RetrospectiveBacktestEvidenceManifest
target: RetrospectivePortfolioTarget
constraints: ConstraintSetV1
freshness: FreshnessPolicy
weights: Mapping[str, float]
prior: Mapping[str, float] | None
metrics: dict[str, float | int | None]
residuals: dict[str, float]
input_payload: dict[str, object]
def _material(
*,
backtest_run_ref: RetrospectiveBacktestRunRef,
manifest: RetrospectiveBacktestEvidenceManifest,
target: RetrospectivePortfolioTarget,
objective_name: str,
objective_version: str,
objective_digest: str,
model_name: str,
model_version: str,
model_digest: str,
expected_return_digest: str,
covariance_digest: str,
scenario_digest: str,
constraints: ConstraintSetV1,
freshness_policy: FreshnessPolicy,
prior_weights: Mapping[str, float] | None = None,
) -> _PortfolioInputs:
run = _validated_run(backtest_run_ref)
for item, expected, path in (
(manifest, RetrospectiveBacktestEvidenceManifest, "$.manifest"),
(target, RetrospectivePortfolioTarget, "$.target"),
(constraints, ConstraintSetV1, "$.constraints"),
(freshness_policy, FreshnessPolicy, "$.freshness_policy"),
):
_check(
type(item) is expected,
path,
f"explicit {expected.__name__} required",
ContractErrorCode.TYPE_ERROR,
)
checked_manifest = RetrospectiveBacktestEvidenceManifest.from_dict(
manifest.to_dict(), artifact=manifest._artifact, backtest_run_ref=run
)
checked_target = RetrospectivePortfolioTarget.from_dict(target.to_dict())
_check(
checked_manifest.qualification is EvidenceQualification.CONTRACT_QUALIFIED,
"$.manifest.qualification",
"contract-qualified retrospective S3 required",
ContractErrorCode.QUALIFICATION_REJECTED,
)
_check(
checked_target.backtest_run_id == run.run_id
and checked_target.dataset_snapshot_id == run.dataset_snapshot_id,
"$.target",
"target and S3 identities differ",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
foundation = run._factor_set._foundation
selected_routes = {
identity
for view_id in run._factor_set.selected_view_ref_ids
for identity in foundation.views[view_id].instrument_route_revision_ids
}
selected_instruments = {
row["instrument_id"]
for row in foundation.to_dict()["instrument_routes"]
if row["route_revision_id"] in selected_routes
}
_check(
set(checked_target.weights) <= selected_instruments,
"$.target.weights",
"target assets must be selected logical instruments",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
constraints = ConstraintSetV1.from_dict(constraints.to_dict())
freshness_policy = FreshnessPolicy.from_dict(freshness_policy.to_dict())
prior = (
None
if prior_weights is None
else _immutable_float_mapping(prior_weights, "$.prior_weights")
)
if prior is not None:
_check(
set(prior) <= selected_instruments,
"$.prior_weights",
"prior assets outside selected instruments",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
weights = checked_target.weights
metrics = _constraint_metrics(weights, prior)
residuals = _constraint_residuals(constraints, weights, metrics)
payload: dict[str, object] = {
"contract_name": "researchhub.portfolio-computation-input",
"schema_version": "2.0.0",
"usage": run.usage,
"historical_availability": run.historical_availability,
"evidence_scope": run.evidence_scope,
"run_ref_document_sha256": _document_sha256(run.to_json()),
"manifest_document_sha256": _document_sha256(checked_manifest.to_json()),
"portfolio_target": checked_target.to_dict(),
"objective": {
"name": _text(objective_name, "$.objective_name"),
"version": _semver(objective_version, "$.objective_version"),
"digest": _digest(objective_digest, "$.objective_digest"),
},
"model": {
"name": _text(model_name, "$.model_name"),
"version": _semver(model_version, "$.model_version"),
"digest": _digest(model_digest, "$.model_digest"),
},
"expected_return_digest": _digest(expected_return_digest, "$.expected_return_digest"),
"covariance_digest": _digest(covariance_digest, "$.covariance_digest"),
"scenario_digest": _digest(scenario_digest, "$.scenario_digest"),
"freshness_policy_digest": _payload_digest(freshness_policy.to_dict()),
"prior_weights": None if prior is None else _mapping_dict(prior),
}
_public(payload)
return _PortfolioInputs(
run,
checked_manifest,
checked_target,
constraints,
freshness_policy,
weights,
prior,
metrics,
residuals,
payload,
)
def _receipt_digests(inputs: _PortfolioInputs) -> dict[str, str | float]:
return {
"input_digest": _payload_digest(inputs.input_payload),
"constraint_digest": _payload_digest(inputs.constraints.to_dict()),
"output_digest": _payload_digest(
{
"weights": _mapping_dict(inputs.weights),
"metrics": inputs.metrics,
"constraint_residuals": inputs.residuals,
}
),
"max_constraint_residual": max(inputs.residuals.values(), default=0.0),
}
def compute_retrospective_portfolio_receipt_digests(**kwargs: Any) -> Mapping[str, str | float]:
"""Recompute receipt claims; the returned digests are not producer authentication."""
return MappingProxyType(_receipt_digests(_material(**kwargs)))
@dataclass(frozen=True, slots=True, init=False)
class RetrospectivePortfolioDecision:
contract_name: str
schema_version: str
decision_id: str
run_id: str
manifest_id: str
evidence_digest: str
dataset_snapshot_id: str
run_ref_document_sha256: str
manifest_document_sha256: str
source_universe_digest: str
portfolio_asset_set_digest: str
target_id: str
target_weights: Mapping[str, float]
prior_weights: Mapping[str, float] | None
objective_name: str
objective_version: str
objective_digest: str
model_name: str
model_version: str
model_digest: str
expected_return_digest: str
covariance_digest: str
scenario_digest: str
constraints: ConstraintSetV1
freshness_policy: FreshnessPolicy
receipt: ComputationReceipt
gross_exposure: float
net_exposure: float
turnover_l1: float | None
position_count: int
constraint_residuals: Mapping[str, float]
output_digest: str
effective_at: str
created_at: str
computed_at: str
observation_cutoff: str
evidence_scope: str
usage: str
historical_availability: str
decision_eligible: bool
execution_validation: str
_payload: Mapping[str, Any] = field(repr=False, compare=False)
_target: RetrospectivePortfolioTarget = field(repr=False, compare=False)
_run: RetrospectiveBacktestRunRef = field(repr=False, compare=False)
_manifest: RetrospectiveBacktestEvidenceManifest = field(repr=False, compare=False)
def to_dict(self) -> dict[str, Any]:
return cast(dict[str, Any], _thaw_json(self._payload))
def to_json(self) -> str:
return _canonical_json(self.to_dict())
@classmethod
def from_dict(
cls,
value: Any,
*,
backtest_run_ref: RetrospectiveBacktestRunRef,
manifest: RetrospectiveBacktestEvidenceManifest,
target: RetrospectivePortfolioTarget,
) -> Self:
_performance_validate_tree(value, "$")
# Rebuild from independent typed inputs, not from a self-approved target in the wire.
row = _shape(
value,
"$",
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
)
arguments = {
key: row[key]
for key in (
"objective_name",
"objective_version",
"objective_digest",
"model_name",
"model_version",
"model_digest",
"expected_return_digest",
"covariance_digest",
"scenario_digest",
"computed_at",
"prior_weights",
)
}
rebuilt = build_retrospective_portfolio_decision(
**arguments,
backtest_run_ref=backtest_run_ref,
manifest=manifest,
target=target,
constraints=ConstraintSetV1.from_dict(row["constraints"]),
freshness_policy=FreshnessPolicy.from_dict(row["freshness_policy"]),
receipt=ComputationReceipt.from_dict(row["receipt"]),
)
_performance_compare(row, rebuilt.to_dict(), "$")
return cast(Self, rebuilt)
@classmethod
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
return cls.from_dict(_json_object(value), **kwargs)
def build_retrospective_portfolio_decision(
*,
backtest_run_ref: RetrospectiveBacktestRunRef,
manifest: RetrospectiveBacktestEvidenceManifest,
target: RetrospectivePortfolioTarget,
objective_name: str,
objective_version: str,
objective_digest: str,
model_name: str,
model_version: str,
model_digest: str,
expected_return_digest: str,
covariance_digest: str,
scenario_digest: str,
constraints: ConstraintSetV1,
freshness_policy: FreshnessPolicy,
receipt: ComputationReceipt,
computed_at: str,
prior_weights: Mapping[str, float] | None = None,
) -> RetrospectivePortfolioDecision:
"""Verify the existing constraints and receipt, with two explicitly different clocks."""
_check(
type(receipt) is ComputationReceipt,
"$.receipt",
"typed computation receipt required",
ContractErrorCode.TYPE_ERROR,
)
receipt = ComputationReceipt.from_dict(receipt.to_dict())
inputs = _material(
backtest_run_ref=backtest_run_ref,
manifest=manifest,
target=target,
objective_name=objective_name,
objective_version=objective_version,
objective_digest=objective_digest,
model_name=model_name,
model_version=model_version,
model_digest=model_digest,
expected_return_digest=expected_return_digest,
covariance_digest=covariance_digest,
scenario_digest=scenario_digest,
constraints=constraints,
freshness_policy=freshness_policy,
prior_weights=prior_weights,
)
run, manifest, target = inputs.run, inputs.manifest, inputs.target
computed = _parse_utc(computed_at, "$.computed_at")
created = _parse_utc(target.created_at, "$.target.created_at")
available = _parse_utc(manifest.artifact_available_at, "$.manifest.artifact_available_at")
_check(
available <= created <= computed,
"$.target.created_at",
"artifact availability <= actual target creation <= computation required",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
_check(
_parse_utc(receipt.computed_at, "$.receipt.computed_at") == computed,
"$.receipt.computed_at",
"receipt actual time differs from computation",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
_check(
(computed - available).total_seconds() <= inputs.freshness.max_manifest_age_seconds,
"$.manifest.artifact_available_at",
"manifest is stale at actual computation",
)
_check(
receipt.status not in {ReceiptStatus.FAILED, ReceiptStatus.FALLBACK},
"$.receipt.status",
"failed/fallback computation cannot form a result",
ContractErrorCode.QUALIFICATION_REJECTED,
)
digests = _receipt_digests(inputs)
for key, expected in digests.items():
_check(
getattr(receipt, key) == expected,
f"$.receipt.{key}",
"receipt differs from independently recomputed evidence",
ContractErrorCode.ARTIFACT_MISMATCH,
)
_check(
digests["max_constraint_residual"] == 0.0,
"$.constraints",
"target violates supported constraints",
)
payload = {
"contract_name": "researchhub.portfolio-decision",
"schema_version": "2.0.0",
"run_id": run.run_id,
"manifest_id": manifest.manifest_id,
"evidence_digest": manifest.evidence_digest,
"dataset_snapshot_id": run.dataset_snapshot_id,
"run_ref_document_sha256": _document_sha256(run.to_json()),
"manifest_document_sha256": _document_sha256(manifest.to_json()),
"source_universe_digest": run.universe_digest,
"portfolio_asset_set_digest": _payload_digest(sorted(inputs.weights)),
"target_id": target.target_id,
"target_weights": _mapping_dict(inputs.weights),
"prior_weights": None if inputs.prior is None else _mapping_dict(inputs.prior),
"objective_name": objective_name,
"objective_version": objective_version,
"objective_digest": objective_digest,
"model_name": model_name,
"model_version": model_version,
"model_digest": model_digest,
"expected_return_digest": expected_return_digest,
"covariance_digest": covariance_digest,
"scenario_digest": scenario_digest,
"constraints": inputs.constraints.to_dict(),
"freshness_policy": inputs.freshness.to_dict(),
"receipt": receipt.to_dict(),
**inputs.metrics,
"constraint_residuals": inputs.residuals,
"output_digest": digests["output_digest"],
"effective_at": target.effective_at,
"created_at": target.created_at,
"computed_at": computed_at,
"observation_cutoff": run.observation_cutoff,
"evidence_scope": run.evidence_scope,
"usage": run.usage,
"historical_availability": run.historical_availability,
"decision_eligible": False,
"execution_validation": "not_validated",
}
_public(payload)
payload["decision_id"] = "rhportfoliodecisionv2:" + _payload_digest(payload)
instance = object.__new__(RetrospectivePortfolioDecision)
values = {
**payload,
"target_weights": inputs.weights,
"prior_weights": inputs.prior,
"constraints": inputs.constraints,
"freshness_policy": inputs.freshness,
"receipt": receipt,
"constraint_residuals": MappingProxyType(inputs.residuals),
"_payload": _freeze_numeric_evidence(payload),
"_target": target,
"_run": run,
"_manifest": manifest,
}
for key, value in values.items():
object.__setattr__(instance, key, value)
return instance
@dataclass(frozen=True, slots=True, init=False)
class RetrospectiveRiskAssessment:
contract_name: str
schema_version: str
assessment_id: str
decision_id: str
run_id: str
manifest_id: str
dataset_snapshot_id: str
covariance_data_snapshot_id: str
covariance_snapshot_id: str
covariance_as_of_date: str
covariance_method: str
covariance_window_start_date: str
covariance_window_end_date: str
covariance_observations: int | None
covariance_lookback_sessions: int | None
covariance_missing_policy: str
covariance_input_digest: str
covariance_matrix_digest: str
return_frequency: str
periods_per_year: int
risk_model_name: str
risk_model_version: str
risk_model_digest: str
freshness_policy_digest: str
scenario_digest: str
portfolio_volatility_limit: float | None
risk_budget: Mapping[str, float]
groups: Mapping[str, str] | None
marginal_risk: Mapping[str, float]
component_risk: Mapping[str, float]
percentage_risk: Mapping[str, float]
portfolio_volatility: float | None
group_exposure: Mapping[str, float]
findings: tuple[RiskFindingCode, ...]
status: RiskAssessmentStatus
qualified: bool
effective_at: str
computed_at: str
observation_cutoff: str
evidence_scope: str
usage: str
historical_availability: str
decision_eligible: bool
execution_validation: str
_payload: Mapping[str, Any] = field(repr=False, compare=False)
def to_dict(self) -> dict[str, Any]:
return cast(dict[str, Any], _thaw_json(self._payload))
def to_json(self) -> str:
return _canonical_json(self.to_dict())
@classmethod
def from_dict(
cls,
value: Any,
*,
portfolio_decision: RetrospectivePortfolioDecision,
backtest_run_ref: RetrospectiveBacktestRunRef,
manifest: RetrospectiveBacktestEvidenceManifest,
covariance: CovarianceSnapshot,
) -> Self:
_performance_validate_tree(value, "$")
row = _shape(
value,
"$",
" ".join(name for name in cls.__dataclass_fields__ if not name.startswith("_")),
)
rebuilt = assess_retrospective_portfolio_risk(
portfolio_decision=portfolio_decision,
backtest_run_ref=backtest_run_ref,
manifest=manifest,
covariance=covariance,
**{
key: row[key]
for key in (
"risk_model_name",
"risk_model_version",
"risk_model_digest",
"risk_budget",
"portfolio_volatility_limit",
"groups",
"computed_at",
)
},
)
_performance_compare(row, rebuilt.to_dict(), "$")
return cast(Self, rebuilt)
@classmethod
def from_json(cls, value: str | bytes, **kwargs: Any) -> Self:
return cls.from_dict(_json_object(value), **kwargs)
class _RiskContext(TypedDict):
decision: RetrospectivePortfolioDecision
covariance: CovarianceSnapshot
matrix_digest: str
risk_model_name: str
risk_model_version: str
risk_model_digest: str
portfolio_volatility_limit: float | None
risk_budget: Mapping[str, float]
groups: Mapping[str, str] | None
computed_at: str
def _risk_result(
*,
decision: RetrospectivePortfolioDecision,
covariance: CovarianceSnapshot,
matrix_digest: str,
risk_model_name: str,
risk_model_version: str,
risk_model_digest: str,
portfolio_volatility_limit: float | None,
risk_budget: Mapping[str, float],
groups: Mapping[str, str] | None,
marginal: Mapping[str, float],
component: Mapping[str, float],
percentage: Mapping[str, float],
volatility: float | None,
grouped: Mapping[str, float],
findings: tuple[RiskFindingCode, ...],
status: RiskAssessmentStatus,
qualified: bool,
computed_at: str,
) -> RetrospectiveRiskAssessment:
assert covariance.window_start_date is not None
assert covariance.window_end_date is not None
payload = {
"contract_name": "researchhub.risk-assessment",
"schema_version": "2.0.0",
"decision_id": decision.decision_id,
"run_id": decision.run_id,
"manifest_id": decision.manifest_id,
"dataset_snapshot_id": decision.dataset_snapshot_id,
"covariance_data_snapshot_id": covariance.data_snapshot_id,
"covariance_snapshot_id": covariance.snapshot_id,
"covariance_as_of_date": covariance.as_of_date.isoformat(),
"covariance_method": covariance.method,
"covariance_window_start_date": covariance.window_start_date.isoformat(),
"covariance_window_end_date": covariance.window_end_date.isoformat(),
"covariance_observations": covariance.observations,
"covariance_lookback_sessions": covariance.lookback_sessions,
"covariance_missing_policy": covariance.missing_policy,
"covariance_input_digest": "sha256:" + covariance.input_sha256,
"covariance_matrix_digest": matrix_digest,
"return_frequency": covariance.return_frequency,
"periods_per_year": covariance.periods_per_year,
"risk_model_name": risk_model_name,
"risk_model_version": risk_model_version,
"risk_model_digest": risk_model_digest,
"freshness_policy_digest": _payload_digest(decision.freshness_policy.to_dict()),
"scenario_digest": decision.scenario_digest,
"portfolio_volatility_limit": portfolio_volatility_limit,
"risk_budget": _mapping_dict(risk_budget),
"groups": None if groups is None else dict(groups),
"marginal_risk": _mapping_dict(marginal),
"component_risk": _mapping_dict(component),
"percentage_risk": _mapping_dict(percentage),
"portfolio_volatility": volatility,
"group_exposure": _mapping_dict(grouped),
"findings": [finding.value for finding in findings],
"status": status.value,
"qualified": qualified,
"effective_at": decision.effective_at,
"computed_at": computed_at,
"observation_cutoff": decision.observation_cutoff,
"evidence_scope": decision.evidence_scope,
"usage": decision.usage,
"historical_availability": decision.historical_availability,
"decision_eligible": False,
"execution_validation": "not_validated",
}
_public(payload)
payload["assessment_id"] = "rhriskassessmentv2:" + _payload_digest(payload)
instance = object.__new__(RetrospectiveRiskAssessment)
values = {
**payload,
"risk_budget": risk_budget,
"groups": groups,
"marginal_risk": marginal,
"component_risk": component,
"percentage_risk": percentage,
"group_exposure": grouped,
"findings": findings,
"status": status,
"_payload": _freeze_numeric_evidence(payload),
}
for key, value in values.items():
object.__setattr__(instance, key, value)
return instance
def assess_retrospective_portfolio_risk(
*,
portfolio_decision: RetrospectivePortfolioDecision,
backtest_run_ref: RetrospectiveBacktestRunRef,
manifest: RetrospectiveBacktestEvidenceManifest,
covariance: CovarianceSnapshot,
risk_model_name: str,
risk_model_version: str,
risk_model_digest: str,
computed_at: str,
risk_budget: Mapping[str, float] | None = None,
portfolio_volatility_limit: float | None = None,
groups: Mapping[str, str] | None = None,
) -> RetrospectiveRiskAssessment:
"""Use the existing Euler decomposition once; distinguish the two freshness clocks."""
_check(
type(portfolio_decision) is RetrospectivePortfolioDecision,
"$.portfolio_decision",
"explicit v2 portfolio result required",
ContractErrorCode.TYPE_ERROR,
)
_check(
type(covariance) is CovarianceSnapshot,
"$.covariance",
"typed covariance required",
ContractErrorCode.TYPE_ERROR,
)
decision = RetrospectivePortfolioDecision.from_dict(
portfolio_decision.to_dict(),
backtest_run_ref=backtest_run_ref,
manifest=manifest,
target=portfolio_decision._target,
)
actual_computed = _parse_utc(computed_at, "$.computed_at")
_check(
_parse_utc(decision.computed_at, "$.portfolio_decision.computed_at") <= actual_computed,
"$.computed_at",
"risk computation precedes portfolio computation",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
manifest_age = (
actual_computed
- _parse_utc(decision._manifest.artifact_available_at, "$.manifest.artifact_available_at")
).total_seconds()
_check(
0 <= manifest_age <= decision.freshness_policy.max_manifest_age_seconds,
"$.manifest.artifact_available_at",
"manifest is stale at actual risk computation",
)
_check(
covariance.data_snapshot_id == decision.dataset_snapshot_id,
"$.covariance.data_snapshot_id",
"covariance and decision data differ",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
business_date = _parse_utc(decision.effective_at, "$.portfolio_decision.effective_at").date()
_check(
covariance.window_start_date is not None and covariance.window_end_date is not None,
"$.covariance",
"bounded covariance window required",
)
assert covariance.window_start_date is not None
assert covariance.window_end_date is not None
_check(
covariance.window_start_date
<= covariance.window_end_date
<= covariance.as_of_date
<= business_date,
"$.covariance.as_of_date",
"covariance business dates exceed the historical target date",
ContractErrorCode.TIME_ORDER_VIOLATION,
)
_check(
(business_date - covariance.as_of_date).days
<= decision.freshness_policy.max_covariance_age_days,
"$.covariance.as_of_date",
"covariance is stale at historical target date",
)
_check(
_digest("sha256:" + covariance.input_sha256, "$.covariance.input_sha256")
== decision.covariance_digest,
"$.covariance.input_sha256",
"covariance input differs from portfolio receipt",
ContractErrorCode.IDENTITY_MISMATCH,
)
name = _text(risk_model_name, "$.risk_model_name")
version = _semver(risk_model_version, "$.risk_model_version")
model_digest = _digest(risk_model_digest, "$.risk_model_digest")
limit = (
None
if portfolio_volatility_limit is None
else _finite_number(
portfolio_volatility_limit, "$.portfolio_volatility_limit", non_negative=True
)
)
budget: Mapping[str, float] = (
MappingProxyType({})
if risk_budget is None
else _immutable_float_mapping(risk_budget, "$.risk_budget")
)
_check(
all(value >= 0 for value in budget.values())
and set(budget) <= decision.target_weights.keys(),
"$.risk_budget",
"risk budgets must be non-negative and use target labels",
)
normalized_groups = None
if groups is not None:
_check(
isinstance(groups, Mapping),
"$.groups",
"mapping required",
ContractErrorCode.TYPE_ERROR,
)
group_values = {
_text(key, "$.groups.keys"): _text(value, "$.groups.values")
for key, value in groups.items()
}
_check(
set(group_values) == decision.target_weights.keys(),
"$.groups",
"groups must label every target exactly once",
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
)
normalized_groups = MappingProxyType(dict(sorted(group_values.items())))
aligned = _validate_covariance_structure(decision.target_weights, covariance)
matrix_digest = _payload_digest(
{
"assets": sorted(decision.target_weights),
"matrix": aligned.to_numpy(dtype=float).tolist(),
}
)
arguments: _RiskContext = {
"decision": decision,
"covariance": covariance,
"matrix_digest": matrix_digest,
"risk_model_name": name,
"risk_model_version": version,
"risk_model_digest": model_digest,
"portfolio_volatility_limit": limit,
"risk_budget": budget,
"groups": normalized_groups,
"computed_at": computed_at,
}
empty: Mapping[str, float] = MappingProxyType({})
def unavailable(finding: RiskFindingCode) -> RetrospectiveRiskAssessment:
return _risk_result(
**arguments,
marginal=empty,
component=empty,
percentage=empty,
volatility=None,
grouped=empty,
findings=(finding,),
status=RiskAssessmentStatus.UNAVAILABLE,
qualified=False,
)
weights = pd.Series(_mapping_dict(decision.target_weights), dtype=float, name="weight")
try:
decomposition = labeled_component_risk(weights, aligned * covariance.periods_per_year)
except ValueError as error:
finding = {
"covariance must be positive semidefinite": RiskFindingCode.COVARIANCE_NOT_PSD,
"weights and covariance must produce positive portfolio variance": RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE,
}.get(str(error))
if finding is None:
raise PortfolioRiskContractError(
PortfolioRiskContractErrorCode.COMPUTATION_FAILURE,
"$.covariance",
"risk computation failed",
) from error
return unavailable(finding)
marginal = _series_mapping(decomposition.marginal)
component = _series_mapping(decomposition.component)
percentage = _series_mapping(decomposition.percentage)
volatility = _finite_number(
decomposition.portfolio_volatility, "$.risk_output.portfolio_volatility", non_negative=True
)
if not (
set(marginal) == set(component) == set(percentage) == decision.target_weights.keys()
and math.isclose(
sum(component.values()), volatility, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
)
and math.isclose(
sum(percentage.values()), 1.0, rel_tol=_CLOSURE_RTOL, abs_tol=_CLOSURE_ATOL
)
):
return unavailable(RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED)
grouped = (
empty
if normalized_groups is None
else _series_mapping(
decomposition.grouped_component(pd.Series(dict(normalized_groups), dtype="object"))
)
)
breached = (limit is not None and volatility > limit + _CLOSURE_ATOL) or any(
percentage[label] > maximum + _CLOSURE_ATOL for label, maximum in budget.items()
)
return _risk_result(
**arguments,
marginal=marginal,
component=component,
percentage=percentage,
volatility=volatility,
grouped=grouped,
findings=(RiskFindingCode.RISK_BUDGET_BREACH,) if breached else (),
status=RiskAssessmentStatus.READY,
qualified=not breached,
)
+372 -10
View File
@@ -5,10 +5,310 @@
from __future__ import annotations
import numpy as np
from numpy.typing import NDArray
import hashlib
import json
from dataclasses import dataclass
from datetime import date
from typing import Any
import numpy as np
import pandas as pd
from numpy.typing import NDArray
__all__ = [
"ComponentRiskResult",
"CovarianceSnapshot",
"component_var",
"estimate_covariance_snapshot",
"labeled_component_risk",
"marginal_risk_contribution",
"risk_contribution",
]
@dataclass(frozen=True, slots=True, init=False, eq=False)
class CovarianceSnapshot:
"""Immutable-by-interface covariance input with explicit time semantics."""
snapshot_id: str
as_of_date: date
_covariance: pd.DataFrame
return_frequency: str
periods_per_year: int
method: str
window_start_date: date | None
window_end_date: date | None
observations: int | None
lookback_sessions: int | None
missing_policy: str
data_snapshot_id: str
input_sha256: str
def __init__(
self,
*,
snapshot_id: str,
as_of_date: str | date | pd.Timestamp,
covariance: pd.DataFrame,
return_frequency: str,
periods_per_year: int,
method: str = "provided",
window_start_date: str | date | pd.Timestamp | None = None,
window_end_date: str | date | pd.Timestamp | None = None,
observations: int | None = None,
lookback_sessions: int | None = None,
missing_policy: str = "provided",
data_snapshot_id: str = "",
input_sha256: str = "",
) -> None:
if not isinstance(snapshot_id, str) or not snapshot_id.strip():
raise ValueError("snapshot_id must be non-empty")
if not isinstance(return_frequency, str) or not return_frequency.strip():
raise ValueError("return_frequency must be non-empty")
if isinstance(periods_per_year, bool) or not isinstance(periods_per_year, int):
raise TypeError("periods_per_year must be an integer")
if periods_per_year <= 0:
raise ValueError("periods_per_year must be positive")
if not isinstance(covariance, pd.DataFrame):
raise TypeError("covariance must be a pandas DataFrame")
if covariance.empty:
raise ValueError("covariance must contain at least one asset")
if not isinstance(method, str) or not method.strip():
raise ValueError("method must be non-empty")
if not isinstance(missing_policy, str) or not missing_policy.strip():
raise ValueError("missing_policy must be non-empty")
for value, name in (
(observations, "observations"),
(lookback_sessions, "lookback_sessions"),
):
if value is not None and (
isinstance(value, bool) or not isinstance(value, int) or value <= 0
):
raise ValueError(f"{name} must be a positive integer when provided")
if input_sha256 and (
len(input_sha256) != 64
or any(character not in "0123456789abcdef" for character in input_sha256)
):
raise ValueError("input_sha256 must be a lowercase SHA-256 digest")
normalized_as_of = _normalized_date(as_of_date, "as_of_date")
normalized_window_start = (
None
if window_start_date is None
else _normalized_date(window_start_date, "window_start_date")
)
normalized_window_end = (
None
if window_end_date is None
else _normalized_date(window_end_date, "window_end_date")
)
if (normalized_window_start is None) != (normalized_window_end is None):
raise ValueError("window_start_date and window_end_date must be provided together")
if (
normalized_window_start is not None
and normalized_window_end is not None
and normalized_window_start > normalized_window_end
):
raise ValueError("window_start_date must not be after window_end_date")
if normalized_window_end is not None and normalized_window_end > normalized_as_of:
raise ValueError("window_end_date must not be after as_of_date")
object.__setattr__(self, "snapshot_id", snapshot_id.strip())
object.__setattr__(self, "as_of_date", normalized_as_of)
object.__setattr__(self, "_covariance", covariance.copy(deep=True))
object.__setattr__(self, "return_frequency", return_frequency.strip())
object.__setattr__(self, "periods_per_year", periods_per_year)
object.__setattr__(self, "method", method.strip())
object.__setattr__(self, "window_start_date", normalized_window_start)
object.__setattr__(self, "window_end_date", normalized_window_end)
object.__setattr__(self, "observations", observations)
object.__setattr__(self, "lookback_sessions", lookback_sessions)
object.__setattr__(self, "missing_policy", missing_policy.strip())
object.__setattr__(self, "data_snapshot_id", data_snapshot_id.strip())
object.__setattr__(self, "input_sha256", input_sha256)
@property
def covariance(self) -> pd.DataFrame:
"""Return an isolated copy so callers cannot mutate the snapshot."""
return self._covariance.copy(deep=True)
def _normalized_date(value: object, name: str) -> date:
try:
timestamp = pd.Timestamp(value)
except (TypeError, ValueError) as error:
raise ValueError(f"{name} must be a valid date") from error
if pd.isna(timestamp):
raise ValueError(f"{name} must be a valid date")
return date(int(timestamp.year), int(timestamp.month), int(timestamp.day))
def _positive_integer(value: int, name: str, *, minimum: int = 1) -> int:
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
raise ValueError(f"{name} must be an integer of at least {minimum}")
return value
def _input_fingerprint(window: pd.DataFrame, session_dates: list[date]) -> str:
values = window.to_numpy(dtype=float, copy=True)
missing = np.isnan(values)
normalized = np.where(missing, 0.0, values).astype("<f8", copy=False)
metadata = {
"assets": [str(asset) for asset in window.columns],
"sessions": [session.isoformat() for session in session_dates],
"shape": list(values.shape),
}
digest = hashlib.sha256(
json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode("utf-8")
)
digest.update(missing.astype(np.uint8, copy=False).tobytes(order="C"))
digest.update(normalized.tobytes(order="C"))
return digest.hexdigest()
def estimate_covariance_snapshot(
asset_returns: pd.DataFrame,
*,
as_of_date: str | date | pd.Timestamp,
lookback_sessions: int,
min_observations: int,
data_snapshot_id: str,
return_frequency: str = "1d",
periods_per_year: int = 252,
) -> CovarianceSnapshot:
"""Estimate a deterministic per-period sample covariance without look-ahead.
The selected lookback window is truncated at ``as_of_date`` before any
calculation. Rows containing a missing asset return are removed as complete
cases, preventing pairwise sample sets from producing an ambiguous matrix.
"""
if not isinstance(asset_returns, pd.DataFrame):
raise TypeError("asset_returns must be a pandas DataFrame")
if asset_returns.empty or asset_returns.shape[1] == 0:
raise ValueError("asset_returns must contain observations and assets")
if not isinstance(asset_returns.index, pd.DatetimeIndex):
raise TypeError("asset_returns index must be a DatetimeIndex")
if not asset_returns.index.is_unique or not asset_returns.index.is_monotonic_increasing:
raise ValueError("asset_returns index must be unique and strictly increasing")
if not asset_returns.columns.is_unique:
raise ValueError("asset_returns must contain unique asset labels")
if any(not isinstance(asset, str) or not asset.strip() for asset in asset_returns.columns):
raise ValueError("asset_returns asset labels must be non-empty strings")
lookback = _positive_integer(lookback_sessions, "lookback_sessions")
minimum = _positive_integer(min_observations, "min_observations", minimum=2)
if minimum > lookback:
raise ValueError("min_observations must not exceed lookback_sessions")
normalized_data_snapshot_id = data_snapshot_id.strip()
if not normalized_data_snapshot_id:
raise ValueError("data_snapshot_id must be non-empty")
normalized_as_of = _normalized_date(as_of_date, "as_of_date")
returns = asset_returns.astype(float, copy=True)
values = returns.to_numpy()
if np.isinf(values).any():
raise ValueError("asset_returns must not contain infinite values")
session_dates = [
_normalized_date(index_value, "asset_returns index") for index_value in returns.index
]
if len(set(session_dates)) != len(session_dates):
raise ValueError("asset_returns must contain at most one observation per session date")
historical_mask = [session <= normalized_as_of for session in session_dates]
window = returns.loc[historical_mask].tail(lookback)
if window.empty:
raise ValueError("asset_returns contain no observations on or before as_of_date")
window_dates = [
_normalized_date(index_value, "asset_returns index") for index_value in window.index
]
complete = window.dropna(axis=0, how="any")
if len(complete) < minimum:
raise ValueError(
f"complete observations must be at least {minimum}; received {len(complete)}"
)
covariance = complete.cov(ddof=1)
covariance_values = covariance.to_numpy()
if not np.isfinite(covariance_values).all():
raise ValueError("sample covariance must be finite")
input_sha256 = _input_fingerprint(window, window_dates)
identity = {
"as_of_date": normalized_as_of.isoformat(),
"assets": list(returns.columns),
"data_snapshot_id": normalized_data_snapshot_id,
"estimator": "sample-cov-v1",
"input_sha256": input_sha256,
"lookback_sessions": lookback,
"min_observations": minimum,
"missing_policy": "complete_case",
"observations": len(complete),
"periods_per_year": periods_per_year,
"return_frequency": return_frequency,
"window_end_date": window_dates[-1].isoformat(),
"window_start_date": window_dates[0].isoformat(),
}
identity_bytes = json.dumps(
identity,
sort_keys=True,
separators=(",", ":"),
).encode("utf-8")
digest = hashlib.sha256(identity_bytes)
digest.update(covariance_values.astype("<f8", copy=False).tobytes(order="C"))
snapshot_id = f"sample-cov-v1:{digest.hexdigest()}"
return CovarianceSnapshot(
snapshot_id=snapshot_id,
as_of_date=normalized_as_of,
covariance=covariance,
return_frequency=return_frequency,
periods_per_year=periods_per_year,
method="sample",
window_start_date=window_dates[0],
window_end_date=window_dates[-1],
observations=len(complete),
lookback_sessions=lookback,
missing_policy="complete_case",
data_snapshot_id=normalized_data_snapshot_id,
input_sha256=input_sha256,
)
@dataclass(frozen=True, slots=True, eq=False)
class ComponentRiskResult:
"""Label-preserving Euler decomposition of portfolio volatility."""
portfolio_volatility: float
marginal: pd.Series
component: pd.Series
percentage: pd.Series
def grouped_component(self, groups: pd.Series) -> pd.Series:
"""Aggregate asset component risk by an explicitly aligned label series."""
if not isinstance(groups, pd.Series):
raise TypeError("groups must be a pandas Series")
if not groups.index.is_unique:
raise ValueError("groups must contain unique asset labels")
if not self.component.index.difference(groups.index).empty or not groups.index.difference(
self.component.index
).empty:
raise ValueError("groups and component risk must use the same asset labels")
aligned = groups.reindex(self.component.index)
if aligned.isna().any():
raise ValueError("groups must contain a non-missing label for every asset")
grouped = self.component.groupby(aligned, sort=True).sum()
grouped.name = "component_risk"
return grouped
def _validate_inputs(weights: NDArray[Any], cov: NDArray[Any]) -> tuple[NDArray[Any], NDArray[Any]]:
"""Normalize a portfolio vector and its covariance matrix."""
w = np.asarray(weights, dtype=float).ravel()
covariance = np.asarray(cov, dtype=float)
k = w.size
if k == 0:
raise ValueError("weights must contain at least one asset")
if covariance.shape != (k, k):
raise ValueError(f"cov shape {covariance.shape} does not match weights length {k}")
return w, covariance
def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
"""风险贡献率 (RC_i): w_i * (Σw)_i / w'Σw。
@@ -25,11 +325,8 @@ def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
Returns:
RC: 风险贡献向量 (k,), Σ=1
"""
w = np.asarray(weights, dtype=float).ravel()
cov = np.asarray(cov, dtype=float)
w, cov = _validate_inputs(weights, cov)
k = w.size
if cov.shape != (k, k):
raise ValueError(f"cov 形状 {cov.shape} 与 weights 长度 {k} 不匹配")
port_var = float(w @ cov @ w)
if port_var <= 0:
@@ -41,13 +338,78 @@ def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
def marginal_risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
"""边际风险贡献 (MRC_i): (Σw)_i。"""
w = np.asarray(weights, dtype=float).ravel()
cov = np.asarray(cov, dtype=float)
w, cov = _validate_inputs(weights, cov)
return cov @ w # type: ignore[no-any-return]
def component_var(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
"""成分方差: w_i · (Σw)_i; 与 RC 的关系 RC_i = CV_i / w'Σw。"""
w = np.asarray(weights, dtype=float).ravel()
cov = np.asarray(cov, dtype=float)
w, cov = _validate_inputs(weights, cov)
return w * (cov @ w) # type: ignore[no-any-return]
def labeled_component_risk(
weights: pd.Series,
covariance: pd.DataFrame,
) -> ComponentRiskResult:
"""Return a label-safe Euler decomposition that sums to portfolio volatility.
The covariance matrix may use a different asset order, but its row and
column label sets must exactly match ``weights``. Invalid or indefinite
covariance input is rejected instead of silently producing misleading risk
percentages.
"""
if not isinstance(weights, pd.Series):
raise TypeError("weights must be a pandas Series")
if not isinstance(covariance, pd.DataFrame):
raise TypeError("covariance must be a pandas DataFrame")
if weights.empty:
raise ValueError("weights must contain at least one asset")
if not weights.index.is_unique:
raise ValueError("weights must contain unique asset labels")
if not covariance.index.is_unique or not covariance.columns.is_unique:
raise ValueError("covariance must contain unique asset labels")
if not weights.index.difference(covariance.index).empty or not covariance.index.difference(
weights.index
).empty:
raise ValueError("weights and covariance must use the same asset labels")
if not weights.index.difference(covariance.columns).empty or not covariance.columns.difference(
weights.index
).empty:
raise ValueError("weights and covariance must use the same asset labels")
aligned_weights = weights.astype(float, copy=True)
aligned_covariance = covariance.reindex(
index=weights.index,
columns=weights.index,
).astype(float, copy=True)
weight_values = aligned_weights.to_numpy()
covariance_values = aligned_covariance.to_numpy()
if not np.isfinite(weight_values).all():
raise ValueError("weights must be finite")
if not np.isfinite(covariance_values).all():
raise ValueError("covariance must be finite")
if not np.allclose(covariance_values, covariance_values.T, rtol=1e-10, atol=1e-12):
raise ValueError("covariance must be symmetric")
eigenvalues = np.linalg.eigvalsh(covariance_values)
scale = max(1.0, float(np.max(np.abs(eigenvalues))))
if float(eigenvalues.min()) < -1e-10 * scale:
raise ValueError("covariance must be positive semidefinite")
portfolio_variance = float(weight_values @ covariance_values @ weight_values)
if portfolio_variance <= 0 or not np.isfinite(portfolio_variance):
raise ValueError("weights and covariance must produce positive portfolio variance")
portfolio_volatility = float(np.sqrt(portfolio_variance))
marginal_values = covariance_values @ weight_values / portfolio_volatility
component_values = weight_values * marginal_values
percentage_values = component_values / portfolio_volatility
return ComponentRiskResult(
portfolio_volatility=portfolio_volatility,
marginal=pd.Series(marginal_values, index=weights.index.copy(), name="marginal_risk"),
component=pd.Series(component_values, index=weights.index.copy(), name="component_risk"),
percentage=pd.Series(
percentage_values,
index=weights.index.copy(),
name="risk_contribution",
),
)
+17
View File
@@ -0,0 +1,17 @@
{
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"replay_spec_digest": "sha256:20f07fcb526bc38b4ba3d71d6c3b00a8c63ad96cab0fecd869326503ab42d98b",
"manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117",
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
"table_content_digests": {
"run": "sha256:7d947ec93f714641669cbb14bd69dd8cf30387aabb7f3a086918fc2e878ab05c",
"signals": "sha256:72dc15064cc45d7c51d2dd4b8c3a6d8c7d4155232d5d70d3c9e7697fab70ce50",
"trades": "sha256:2b8b9321e7993941ac486cf50cab5b4c6425b2571a4701f0eb02273ec9a96c53",
"positions": "sha256:b456a48fab51742b05084ca6dcaf01c03dfe5b39ad71215ada070e9b1f59f7ea",
"nav": "sha256:25649efce860b76410f87dbd36c81085887620086dc0b48b07099918f65d9c78",
"performance": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
"attribution": "sha256:ad0f12f668d0a2d9ae5b3636d29989ab4bafcfb95f5de77abeb52a6d9e95d366",
"attribution_daily": "sha256:d4459ad650f88871d7b1e40392033037b5818ef92fdf1ee021868929fc1455a5",
"risk": "sha256:4816dd4812b5ff2e97e74bf34ca2221bfde675b387a96e277683557f0ad7d975"
}
}
+206
View File
@@ -0,0 +1,206 @@
{
"dataset_snapshot": {
"contract_name": "researchhub.dataset-snapshot",
"schema_version": "1.0.0",
"snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"descriptor": {
"dataset": {
"dataset_id": "rhdataset:market:0123456789abcdef0123456789abcdef",
"dataset_kind": "market",
"record_schema_version": "1.0.0",
"dimensions": ["instrument_id", "effective_time"]
},
"published_at": "2026-01-02T07:05:00Z",
"time_semantics": {
"effective_time": {
"start_inclusive": "2026-01-02T07:00:00Z",
"end_inclusive": "2026-01-02T07:00:00Z"
},
"knowledge_time": {
"start_inclusive": "2026-01-02T07:01:00Z",
"end_inclusive": "2026-01-02T07:01:00Z"
},
"pit_cutoff": "2026-01-02T07:01:00Z"
},
"content": {
"digest_algorithm": "sha256",
"canonicalization": "RFC8785",
"record_order": "canonical-record-byte-order",
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
"logical_manifest": {
"record_count": 2,
"chunks": [
{
"chunk_index": 0,
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
"record_count": 2
}
]
},
"manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
"record_count": 2
},
"lineage": {
"publisher": {"id": "researchhub.data", "version": "1.0.0"},
"transformation": {
"id": "rhtransform:00112233445566778899aabbccddeeff",
"version": "1.0.0"
},
"upstream_snapshot_ids": [],
"upstream_content_digests": []
},
"quality": {
"status": "passed",
"checks": [
{
"check_id": "completeness",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:876fc2fcc6414ddc3f824a47f475d34c82a53d2bda5dc72a234a3f3164e8e2ec"
},
{
"check_id": "pit_time_integrity",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:90a6cc46b9f2ab317a1c6d14dc173784e19b5621cd7fe956e8f41338cbdc5944"
}
]
},
"qualification": {
"status": "qualified",
"policy_id": "researchhub.dataset-snapshot.pit",
"policy_version": "1.0.0",
"evaluated_at": "2026-01-02T07:04:00Z",
"evidence_digest": "sha256:e192462f9022f2b477f73cdbe9e6c9f891ebcfdc2b4ed4f8ddd7b1ff107ee6a6"
}
}
},
"data_foundation": {
"contract_name": "researchhub.data-foundation",
"schema_version": "1.0.0",
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"pit_cutoff": "2026-01-03T00:00:00Z",
"instrument_routes": [
{
"route_revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
"revision_number": 1,
"symbol": "600000",
"mic": "XSHG",
"currency": "CNY",
"asset_class": "equity",
"instrument_type": "stock",
"calendar_id": "rhcalendar:11112222333344445555666677778888",
"effective_from": "2020-01-01T00:00:00Z",
"knowledge_time": "2026-01-01T07:00:00Z",
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
}
],
"trading_calendar_revisions": [
{
"calendar_revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
"calendar_id": "rhcalendar:11112222333344445555666677778888",
"session_date": "2026-01-02",
"revision_number": 1,
"status": "open",
"sessions": [
{"opens_at": "2026-01-02T01:30:00Z", "closes_at": "2026-01-02T07:00:00Z"}
],
"knowledge_time": "2026-01-01T08:00:00Z",
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
}
],
"corporate_action_revisions": [
{
"action_revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
"action_id": "rhaction:99998888777766665555444433332222",
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
"revision_number": 1,
"action_type": "cash_dividend",
"status": "confirmed",
"effective_time": "2026-01-02T00:00:00Z",
"knowledge_time": "2026-01-01T09:00:00Z",
"terms_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
}
],
"standardized_views": [
{
"view_ref_id": "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c",
"view_id": "rhview:abcdef0123456789abcdef0123456789",
"view_version": "1.0.0",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"pit_cutoff": "2026-01-03T00:00:00Z",
"schema_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
"content_digest": "sha256:123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef0",
"transformation_digest": "sha256:23456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef01",
"instrument_route_revision_ids": [
"rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86"
],
"trading_calendar_revision_ids": [
"rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
],
"corporate_action_revision_ids": [
"rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
]
}
],
"revision_lineage": [
{
"revision_kind": "instrument_route",
"revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
"revision_number": 1,
"knowledge_time": "2026-01-01T07:00:00Z",
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
},
{
"revision_kind": "trading_calendar",
"revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
"revision_number": 1,
"knowledge_time": "2026-01-01T08:00:00Z",
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
},
{
"revision_kind": "corporate_action",
"revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
"revision_number": 1,
"knowledge_time": "2026-01-01T09:00:00Z",
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
}
],
"readiness": {
"evidence_scope": "synthetic_fixture",
"contract_validation": {
"status": "validated",
"evidence_digests": [
"sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee"
]
},
"real_data_validation": {"status": "not_validated", "evidence_digests": []},
"production_validation": {"status": "not_validated", "evidence_digests": []},
"live_validation": {"status": "not_validated", "evidence_digests": []}
}
},
"output_schema": {
"columns": ["evaluation_at", "factor_id", "instrument_id", "value"],
"schema_version": "1.0.0"
},
"output_content": {
"rows": [
{
"evaluation_at": "2026-01-03T11:00:00Z",
"factor_id": "alpha_005",
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
"value": "0.125"
}
]
},
"expected": {
"definition_id": "rhfactorv1:sha256:978fb8000d318373844a5e044ca14bf377e01ebe8d85964b826ecd2af9085ce9",
"input_schema_digest": "sha256:4501aeab99b4bcc25a1b8813ebe197fb498053fd710d73746bf20fc8eeb4bfa7",
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"output_artifact_id": "rhfactoroutputv1:sha256:a4803b5ff66d12d3a0e7e8e5b8cca953cbc137e2bf41514b7c4e5f05da5ee68b",
"legacy_binding_id": "rhlegacyfactorv1:sha256:541bc5a9469f9c8e4c2d696a9972fc5f2e6e2bef218b86f728823994b915dede"
}
}
+871
View File
@@ -0,0 +1,871 @@
{
"cases": {
"absent": {
"artifact_available_at": "2026-01-08T02:05:00Z",
"authority": "quant_engine",
"backtest_evidence_manifest_document_sha256": "sha256:b7e9139e021de3122bee9376a8c8387fca3db75fe2a698adccf33471caf971d2",
"backtest_evidence_manifest_evidence_digest": "sha256:fae26a93d754e98f437bf9e4b635fc1cdc4f85d004a4bde396823b52e2115de5",
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:3f67fcad23c75684ffeb325d8405b2f81139a732e237d30fbb30d23f91e32726",
"backtest_evidence_qualification": "contract_qualified",
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"benchmark_alignment_policy": "none",
"benchmark_id": "",
"benchmark_series_digest": null,
"calendar": "CN-A",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
"cost_model_version": "1.0.0",
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"document_sha256": "sha256:729852af307fd4a94ac566ae45df2e57b3d45c4fbdf45820351b5ef19cdf0ad3",
"end_date": "2026-01-08",
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
"execution_model_version": "1.0.0",
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"frequency": "1d",
"methodology": {
"alpha": "daily_ols_intercept_geometric_annualization",
"annual_risk_free": 0.0,
"annualized_return": "geometric_compound",
"annualized_volatility": "sample_std_sqrt_periods",
"benchmark_alignment": "none",
"benchmark_risk_free_daily": 0.0,
"beta": "sample_covariance_over_sample_variance",
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"implementation_module": "quant_engine.metrics",
"implementation_version": "researchhub.quant-performance-methodology.v1",
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"periods_per_year": 252,
"return_type": "simple",
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
"source_frequency": "1d",
"total_return": "final_nav_minus_one",
"tracking_error": "sample_std_active_return_sqrt_periods",
"win_rate": "positive_daily_return_count_over_observation_count"
},
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"metrics": [
{
"availability": "available",
"key": "total_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "total_ret",
"unit": "ratio",
"value": 0.575
},
{
"availability": "available",
"key": "annualized_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_ret",
"unit": "ratio_per_year",
"value": 2683336646708.1
},
{
"availability": "available",
"key": "annualized_volatility",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_volatility",
"unit": "ratio_per_year",
"value": 1.38901943830891
},
{
"availability": "available",
"key": "sharpe_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sharpe",
"unit": "ratio",
"value": 1931820803008.3313
},
{
"availability": "available",
"key": "sortino_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sortino",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "maximum_drawdown",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "max_dd",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "calmar_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "calmar",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "win_rate",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "win_rate",
"unit": "ratio",
"value": 0.75
},
{
"availability": "benchmark_absent",
"key": "tracking_error",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "tracking_error",
"unit": "ratio_per_year",
"value": null
},
{
"availability": "benchmark_absent",
"key": "information_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "ir",
"unit": "ratio",
"value": null
},
{
"availability": "benchmark_absent",
"key": "alpha",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "alpha",
"unit": "ratio_per_year",
"value": null
},
{
"availability": "benchmark_absent",
"key": "beta",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "beta",
"unit": "ratio",
"value": null
},
{
"availability": "available",
"key": "trade_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_trades",
"unit": "count",
"value": 3
},
{
"availability": "available",
"key": "day_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_days",
"unit": "count",
"value": 4
}
],
"performance_evidence_id": "rhperformanceevidencev1:sha256:d735abe76c28db429e107cb5ba6a1929c2c0f2084a6c4ac41b38ffbc02b2d2bd",
"performance_row_digest": "sha256:4e6329b0db4731513fe793491a163dd358e3a6f7baa84d3c94a4d4d2355c7a7d",
"performance_table_content_digest": "sha256:074b8511ff9a2154235e4dde8bac300658cb69f900e49f332d61560b82be9af5",
"performance_table_logical_name": "performance",
"performance_table_row_count": 1,
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
"research_artifact_content_digest": "sha256:ad85ee1317bd8ce5fbef8fddc268b91be8678701ba5b8fbf6d61bf670f918bf9",
"research_artifact_schema_version": "1.1.0",
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"schema_version": "researchhub.performance-evidence.v1",
"scope": "offline_research_only",
"start_date": "2026-01-05",
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
"strategy_id": "alpha-top1",
"strategy_version": "1.0.0",
"timezone": "Asia/Shanghai"
},
"estimable": {
"artifact_available_at": "2026-01-08T02:05:00Z",
"authority": "quant_engine",
"backtest_evidence_manifest_document_sha256": "sha256:e2e0e63fb4c733d2dba9a290511b6c8b0132bfe877dfe89ee7e614b74b87bc51",
"backtest_evidence_manifest_evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117",
"backtest_evidence_qualification": "contract_qualified",
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"benchmark_alignment_policy": "exact_session_index",
"benchmark_id": "000300.SH",
"benchmark_series_digest": "sha256:1c564e4d12af0d59b1fd707d7816da2895238969d9207470984db63cfeac423a",
"calendar": "CN-A",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
"cost_model_version": "1.0.0",
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"document_sha256": "sha256:a6a1a3313f511505c0cf57fddeebdf20a3b0f83a2958f31332d5597d172165e7",
"end_date": "2026-01-08",
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
"execution_model_version": "1.0.0",
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"frequency": "1d",
"methodology": {
"alpha": "daily_ols_intercept_geometric_annualization",
"annual_risk_free": 0.0,
"annualized_return": "geometric_compound",
"annualized_volatility": "sample_std_sqrt_periods",
"benchmark_alignment": "exact_session_index",
"benchmark_risk_free_daily": 0.0,
"beta": "sample_covariance_over_sample_variance",
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"implementation_module": "quant_engine.metrics",
"implementation_version": "researchhub.quant-performance-methodology.v1",
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"periods_per_year": 252,
"return_type": "simple",
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
"source_frequency": "1d",
"total_return": "final_nav_minus_one",
"tracking_error": "sample_std_active_return_sqrt_periods",
"win_rate": "positive_daily_return_count_over_observation_count"
},
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"metrics": [
{
"availability": "available",
"key": "total_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "total_ret",
"unit": "ratio",
"value": 0.575
},
{
"availability": "available",
"key": "annualized_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_ret",
"unit": "ratio_per_year",
"value": 2683336646708.1
},
{
"availability": "available",
"key": "annualized_volatility",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_volatility",
"unit": "ratio_per_year",
"value": 1.38901943830891
},
{
"availability": "available",
"key": "sharpe_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sharpe",
"unit": "ratio",
"value": 1931820803008.3313
},
{
"availability": "available",
"key": "sortino_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sortino",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "maximum_drawdown",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "max_dd",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "calmar_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "calmar",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "win_rate",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "win_rate",
"unit": "ratio",
"value": 0.75
},
{
"availability": "available",
"key": "tracking_error",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "tracking_error",
"unit": "ratio_per_year",
"value": 1.3032171729991897
},
{
"availability": "available",
"key": "information_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "ir",
"unit": "ratio",
"value": 22.801264912443322
},
{
"availability": "available",
"key": "alpha",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "alpha",
"unit": "ratio_per_year",
"value": 123663320625.66454
},
{
"availability": "available",
"key": "beta",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "beta",
"unit": "ratio",
"value": 3.2500000000000013
},
{
"availability": "available",
"key": "trade_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_trades",
"unit": "count",
"value": 3
},
{
"availability": "available",
"key": "day_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_days",
"unit": "count",
"value": 4
}
],
"performance_evidence_id": "rhperformanceevidencev1:sha256:f1f72c34d7427bd0d2e273a66b9677d112aa51c2417ea1cda8245e51e334e4f6",
"performance_row_digest": "sha256:bad2b506679a6ca13d80ab00859a442ba692f3872d42b0b45bbc3b1d6ffe72a4",
"performance_table_content_digest": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
"performance_table_logical_name": "performance",
"performance_table_row_count": 1,
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
"research_artifact_content_digest": "sha256:6ec566d73da3966d3d5f946313e2b0181993374c9f4ceb4b56e165cc6793d382",
"research_artifact_schema_version": "1.1.0",
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"schema_version": "researchhub.performance-evidence.v1",
"scope": "offline_research_only",
"start_date": "2026-01-05",
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
"strategy_id": "alpha-top1",
"strategy_version": "1.0.0",
"timezone": "Asia/Shanghai"
},
"zero_active_variance": {
"artifact_available_at": "2026-01-08T02:05:00Z",
"authority": "quant_engine",
"backtest_evidence_manifest_document_sha256": "sha256:8a13212575154d97f85011ed54ed3bb588f73a691926b3ca03b360d563e9fc29",
"backtest_evidence_manifest_evidence_digest": "sha256:8c0c2052ab93aa920254bc7f4d89435d839b7150ecaaaa798703e89ddf876ec1",
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:699ad35ee30a294082c607c40617097aa90d5d58343e7d8c8185d8a53f52596c",
"backtest_evidence_qualification": "contract_qualified",
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"benchmark_alignment_policy": "exact_session_index",
"benchmark_id": "000300.SH",
"benchmark_series_digest": "sha256:7668b77ac689cd542005fae42cb2abd48fdf772d2f982086dca409606653e6ac",
"calendar": "CN-A",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
"cost_model_version": "1.0.0",
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"document_sha256": "sha256:44a04be0105f29e1cbf8d2430b6b0eb084853d428bc2ccd5834dfdbcf9482bd7",
"end_date": "2026-01-08",
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
"execution_model_version": "1.0.0",
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"frequency": "1d",
"methodology": {
"alpha": "daily_ols_intercept_geometric_annualization",
"annual_risk_free": 0.0,
"annualized_return": "geometric_compound",
"annualized_volatility": "sample_std_sqrt_periods",
"benchmark_alignment": "exact_session_index",
"benchmark_risk_free_daily": 0.0,
"beta": "sample_covariance_over_sample_variance",
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"implementation_module": "quant_engine.metrics",
"implementation_version": "researchhub.quant-performance-methodology.v1",
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"periods_per_year": 252,
"return_type": "simple",
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
"source_frequency": "1d",
"total_return": "final_nav_minus_one",
"tracking_error": "sample_std_active_return_sqrt_periods",
"win_rate": "positive_daily_return_count_over_observation_count"
},
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"metrics": [
{
"availability": "available",
"key": "total_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "total_ret",
"unit": "ratio",
"value": 0.575
},
{
"availability": "available",
"key": "annualized_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_ret",
"unit": "ratio_per_year",
"value": 2683336646708.1
},
{
"availability": "available",
"key": "annualized_volatility",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_volatility",
"unit": "ratio_per_year",
"value": 1.38901943830891
},
{
"availability": "available",
"key": "sharpe_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sharpe",
"unit": "ratio",
"value": 1931820803008.3313
},
{
"availability": "available",
"key": "sortino_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sortino",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "maximum_drawdown",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "max_dd",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "calmar_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "calmar",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "win_rate",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "win_rate",
"unit": "ratio",
"value": 0.75
},
{
"availability": "available",
"key": "tracking_error",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "tracking_error",
"unit": "ratio_per_year",
"value": 0.0
},
{
"availability": "not_estimable_active_variance",
"key": "information_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "ir",
"unit": "ratio",
"value": null
},
{
"availability": "available",
"key": "alpha",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "alpha",
"unit": "ratio_per_year",
"value": 0.0
},
{
"availability": "available",
"key": "beta",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "beta",
"unit": "ratio",
"value": 1.0000000000000002
},
{
"availability": "available",
"key": "trade_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_trades",
"unit": "count",
"value": 3
},
{
"availability": "available",
"key": "day_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_days",
"unit": "count",
"value": 4
}
],
"performance_evidence_id": "rhperformanceevidencev1:sha256:e6730d25d90c85a6370adb72bc45c1bf505235f77b72521dd8197898ca369fe5",
"performance_row_digest": "sha256:6528c47fa9416e350aa5010477dbd8c83a35cecf87aa5bc413cec95a9765114d",
"performance_table_content_digest": "sha256:7d57a966431c68093a9f6193ae62f79a63d33832f11dd68524ed9bcd2cb6257c",
"performance_table_logical_name": "performance",
"performance_table_row_count": 1,
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
"research_artifact_content_digest": "sha256:b17ab9e158a7ceeb5107f6fe8a61f332009c314186f3436d4750aa44d6ca3856",
"research_artifact_schema_version": "1.1.0",
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"schema_version": "researchhub.performance-evidence.v1",
"scope": "offline_research_only",
"start_date": "2026-01-05",
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
"strategy_id": "alpha-top1",
"strategy_version": "1.0.0",
"timezone": "Asia/Shanghai"
},
"zero_benchmark_variance": {
"artifact_available_at": "2026-01-08T02:05:00Z",
"authority": "quant_engine",
"backtest_evidence_manifest_document_sha256": "sha256:2389b804f5d8d20a37b31f596b52892484777c5da799dd68865a082ee90e6e5b",
"backtest_evidence_manifest_evidence_digest": "sha256:54bb315bc1940ef6df79999cf03ebb8d10defaf526bf7802d63da33f612520ec",
"backtest_evidence_manifest_id": "rhbacktestevidencev1:sha256:ad9559e1f8feca28bb310765216a975935f05f51057139c2d02fd2fe7dfe501e",
"backtest_evidence_qualification": "contract_qualified",
"backtest_run_ref_document_sha256": "sha256:6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
"backtest_run_ref_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"benchmark_alignment_policy": "exact_session_index",
"benchmark_id": "000300.SH",
"benchmark_series_digest": "sha256:d0f1e954d796e95254bcccc893d6e3a8f9d541c3028b88b2c0b24c1658bb2408",
"calendar": "CN-A",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"configuration_digest": "sha256:d28b49ea1bebc78d0023667f4fb9990b1fb45807176c2193b458525def056f4d",
"cost_model_digest": "sha256:8888888888888888888888888888888888888888888888888888888888888888",
"cost_model_version": "1.0.0",
"dataset_content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
"dataset_manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"document_sha256": "sha256:077f3ec84802aff2f046365a3a5f61f475b1ad63ec05e96742844bc531bcb551",
"end_date": "2026-01-08",
"environment_lock_digest": "sha256:9999999999999999999999999999999999999999999999999999999999999999",
"execution_model_digest": "sha256:7777777777777777777777777777777777777777777777777777777777777777",
"execution_model_version": "1.0.0",
"factor_output_content_digest": "sha256:d78751460dc27fc796163c462d946924ce2bcb45dcfceb5e4b77f76093befc9e",
"factor_set_digest": "sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
"foundation_digest": "sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
"frequency": "1d",
"methodology": {
"alpha": "daily_ols_intercept_geometric_annualization",
"annual_risk_free": 0.0,
"annualized_return": "geometric_compound",
"annualized_volatility": "sample_std_sqrt_periods",
"benchmark_alignment": "exact_session_index",
"benchmark_risk_free_daily": 0.0,
"beta": "sample_covariance_over_sample_variance",
"calmar_ratio": "unadjusted_annualized_return_over_absolute_maximum_drawdown",
"code_revision": "dddddddddddddddddddddddddddddddddddddddd",
"implementation_module": "quant_engine.metrics",
"implementation_version": "researchhub.quant-performance-methodology.v1",
"information_ratio": "mean_active_over_sample_std_active_sqrt_periods",
"maximum_drawdown": "non_positive_peak_to_trough_ratio_with_initial_nav_one",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"periods_per_year": 252,
"return_type": "simple",
"sharpe_ratio": "annualized_return_minus_annual_risk_free_over_annualized_volatility",
"sortino_ratio": "annualized_return_minus_annual_risk_free_over_root_mean_square_negative_returns_sqrt_periods",
"source_frequency": "1d",
"total_return": "final_nav_minus_one",
"tracking_error": "sample_std_active_return_sqrt_periods",
"win_rate": "positive_daily_return_count_over_observation_count"
},
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"metrics": [
{
"availability": "available",
"key": "total_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "total_ret",
"unit": "ratio",
"value": 0.575
},
{
"availability": "available",
"key": "annualized_return",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_ret",
"unit": "ratio_per_year",
"value": 2683336646708.1
},
{
"availability": "available",
"key": "annualized_volatility",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "ann_volatility",
"unit": "ratio_per_year",
"value": 1.38901943830891
},
{
"availability": "available",
"key": "sharpe_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sharpe",
"unit": "ratio",
"value": 1931820803008.3313
},
{
"availability": "available",
"key": "sortino_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "sortino",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "maximum_drawdown",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "max_dd",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "calmar_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "calmar",
"unit": "ratio",
"value": 0.0
},
{
"availability": "available",
"key": "win_rate",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "win_rate",
"unit": "ratio",
"value": 0.75
},
{
"availability": "available",
"key": "tracking_error",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "tracking_error",
"unit": "ratio_per_year",
"value": 1.38901943830891
},
{
"availability": "available",
"key": "information_ratio",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "ir",
"unit": "ratio",
"value": 22.299903907544408
},
{
"availability": "not_estimable_benchmark_variance",
"key": "alpha",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "alpha",
"unit": "ratio_per_year",
"value": null
},
{
"availability": "not_estimable_benchmark_variance",
"key": "beta",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": true,
"source_column": "beta",
"unit": "ratio",
"value": null
},
{
"availability": "available",
"key": "trade_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_trades",
"unit": "count",
"value": 3
},
{
"availability": "available",
"key": "day_count",
"methodology_id": "researchhub.quant-performance-methodology.v1",
"metric_schema_id": "researchhub.quant-performance-metrics.v1",
"nullable": false,
"source_column": "n_days",
"unit": "count",
"value": 4
}
],
"performance_evidence_id": "rhperformanceevidencev1:sha256:86133a6b3c3091477cdc4618ebc294e671c9ba0a3ef5015a068f92bb575729b9",
"performance_row_digest": "sha256:90c9aa2f4168c15ff9ac5d4da2a35c3e915bb4e04736e5a440ea80c787a99553",
"performance_table_content_digest": "sha256:1570b64d83ad2a849607e6f5203f5ec2f0291a8d6cbb00b8edfe0ae030b7967d",
"performance_table_logical_name": "performance",
"performance_table_row_count": 1,
"performance_table_schema_digest": "sha256:16cef93a679761103ae405e622b7929abbfe07115bf276be4f164da5a16128d0",
"research_artifact_content_digest": "sha256:fed7a28888a6e98ccce78e7d84f22d4e3636e928485861af7a97f90eddc91f94",
"research_artifact_schema_version": "1.1.0",
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"schema_version": "researchhub.performance-evidence.v1",
"scope": "offline_research_only",
"start_date": "2026-01-05",
"strategy_digest": "sha256:6666666666666666666666666666666666666666666666666666666666666666",
"strategy_id": "alpha-top1",
"strategy_version": "1.0.0",
"timezone": "Asia/Shanghai"
}
},
"schema_version": 1,
"source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c",
"source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603"
}
+129
View File
@@ -0,0 +1,129 @@
{
"portfolio_decision": {
"computed_at": "2026-01-08T03:01:00Z",
"constraint_residuals": {
"gross_exposure_max": 0.0,
"net_exposure_max": 0.0,
"net_exposure_min": 0.0,
"position_count_max": 0.0,
"single_asset_max": 0.0,
"single_asset_min": 0.0,
"turnover_max": 0.0
},
"constraints": {
"gross_exposure_max": 1.0,
"net_exposure_max": 1.0,
"net_exposure_min": 1.0,
"position_count_max": 2,
"schema_version": "1.0.0",
"single_asset_max": 0.7,
"single_asset_min": 0.2,
"turnover_max": 0.2
},
"contract_name": "researchhub.portfolio-decision",
"covariance_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
"effective_at": "2026-01-08T03:00:00Z",
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
"expected_return_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd",
"freshness_policy": {
"max_covariance_age_days": 0,
"max_manifest_age_seconds": 3600,
"schema_version": "1.0.0"
},
"gross_exposure": 1.0,
"manifest_document_sha256": "fbf54218f770528978f9ccd35577e1ab00877143ea397a576e071e75f0afbab0",
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
"model_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
"model_name": "deterministic_weights",
"model_version": "1.0.0",
"net_exposure": 1.0,
"objective_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
"objective_name": "long_only_allocation",
"objective_version": "1.0.0",
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
"portfolio_asset_set_digest": "sha256:b64e3448a83a5b86466465080361c1a7e1157a27ddccd4b68069cb18caffb74a",
"position_count": 2,
"prior_weights": {
"A": 0.5,
"B": 0.5
},
"receipt": {
"algorithm": "bounded_allocation",
"algorithm_version": "1.0.0",
"computed_at": "2026-01-08T03:01:00Z",
"constraint_digest": "sha256:34df0e5c00f503748ff936f9cd415a8947169181f8ca0917bce97dd606e08e94",
"implementation_digest": "sha256:ffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffff",
"input_digest": "sha256:ebd8ff957115e1adfd84eafa5cea49470356194d747b64ee5897858f9dd067b7",
"iterations": null,
"max_constraint_residual": 0.0,
"objective_value": null,
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
"parameter_digest": "sha256:0000000000000000000000000000000000000000000000000000000000000000",
"schema_version": "1.0.0",
"solver_config_digest": null,
"solver_name": null,
"solver_required": false,
"solver_version": null,
"status": "completed",
"tolerance": 1e-12
},
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"run_ref_document_sha256": "6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
"schema_version": "1.0.0",
"source_universe_digest": "sha256:5555555555555555555555555555555555555555555555555555555555555555",
"target_id": "portfolio-target:synthetic-v1",
"target_weights": {
"A": 0.6,
"B": 0.4
},
"turnover_l1": 0.19999999999999996
},
"risk_assessment": {
"assessment_id": "rhriskassessmentv1:sha256:dbc38825cffcf6d95bd0216d22dbba4e0d4a1359ca5924a3ee529b99a7d78b6d",
"component_risk": {
"A": 1.4549226783578566,
"B": 1.4549226783578568
},
"contract_name": "researchhub.risk-assessment",
"covariance_as_of_date": "2026-01-08",
"covariance_data_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"covariance_input_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
"covariance_snapshot_id": "covariance:synthetic-v1",
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
"findings": [],
"freshness_policy_digest": "sha256:833f58f4d1056f0450f7369fd4edd70534cc74e9e55cd60f05b2b3d7a2979763",
"group_exposure": {
"equity": 1.4549226783578566,
"fixed_income": 1.4549226783578568
},
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
"marginal_risk": {
"A": 2.424871130596428,
"B": 3.637306695894642
},
"percentage_risk": {
"A": 0.49999999999999983,
"B": 0.49999999999999994
},
"periods_per_year": 252,
"portfolio_volatility": 2.909845356715714,
"portfolio_volatility_limit": 10.0,
"qualified": true,
"return_frequency": "1d",
"risk_budget": {
"A": 0.8,
"B": 0.8
},
"risk_model_digest": "sha256:2222222222222222222222222222222222222222222222222222222222222222",
"risk_model_name": "euler_volatility",
"risk_model_version": "1.0.0",
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
"schema_version": "1.0.0",
"status": "ready"
}
}
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,175 @@
{
"contract_name": "researchhub.data-foundation",
"schema_version": "2.0.0",
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
"observation_cutoff": "2026-09-08T01:01:00Z",
"usage": "retrospective_research",
"historical_availability": "not_established",
"published_at": "2026-09-08T01:05:00Z",
"instrument_routes": [
{
"observation_sequence": 1,
"observed_by": "2026-09-08T01:00:00Z",
"earliest_external_knowledge": {
"status": "unknown"
},
"history_completeness": "not_established",
"instrument_id": "rhinstrument:11111111111111111111111111111111",
"symbol": "SIM0",
"mic": "XSHG",
"currency": "CNY",
"asset_class": "equity",
"instrument_type": "stock",
"calendar_id": "rhcalendar:33333333333333333333333333333333",
"effective_from": "2018-01-01T00:00:00Z",
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb",
"route_revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9"
},
{
"observation_sequence": 1,
"observed_by": "2026-09-08T01:00:00Z",
"earliest_external_knowledge": {
"status": "unknown"
},
"history_completeness": "not_established",
"instrument_id": "rhinstrument:22222222222222222222222222222222",
"symbol": "SIM1",
"mic": "XSHG",
"currency": "CNY",
"asset_class": "equity",
"instrument_type": "stock",
"calendar_id": "rhcalendar:33333333333333333333333333333333",
"effective_from": "2018-01-01T00:00:00Z",
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799",
"route_revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
}
],
"trading_calendar_revisions": [
{
"observation_sequence": 1,
"observed_by": "2026-09-08T01:00:00Z",
"earliest_external_knowledge": {
"status": "unknown"
},
"history_completeness": "not_established",
"calendar_id": "rhcalendar:33333333333333333333333333333333",
"session_date": "2018-01-02",
"status": "open",
"sessions": [
{
"opens_at": "2018-01-02T01:30:00Z",
"closes_at": "2018-01-02T07:00:00Z"
}
],
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a",
"calendar_revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
}
],
"corporate_action_revisions": [],
"standardized_views": [
{
"view_id": "rhview:66666666666666666666666666666666",
"view_version": "2.0.0",
"dataset_snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905",
"observation_cutoff": "2026-09-08T01:01:00Z",
"schema_digest": "sha256:bb660b16a5dfccb771e6a263b56aedfebc18a2ea9e13d6b934a6873810f4bd9d",
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
"transformation_digest": "sha256:b2376dbe4424b38e5b27874c998f082aecc086179f611896ebaa0fdeb5362c68",
"instrument_route_revision_ids": [
"rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
"rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92"
],
"trading_calendar_revision_ids": [
"rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25"
],
"corporate_action_revision_ids": [],
"usage": "retrospective_research",
"historical_availability": "not_established",
"available_at": "2026-09-08T01:04:00Z",
"view_ref_id": "rhviewrefv2:sha256:3ce598c8db700ee409e82671b3c8b7689a281ba211ee691082b3bc4f6168e309"
}
],
"observation_lineage": [
{
"revision_kind": "instrument_route",
"revision_id": "rhroutev2:sha256:026558e7edbd96e437a9a6526601fdb746f42c598f7e76502906239365b9c9d9",
"observation_sequence": 1,
"observed_by": "2026-09-08T01:00:00Z",
"earliest_external_knowledge": {
"status": "unknown"
},
"history_completeness": "not_established",
"evidence_digest": "sha256:ca276a7c5cc35f580fad6a98e20dc36fde35762e9579456fd427d6457ee1b2eb"
},
{
"revision_kind": "instrument_route",
"revision_id": "rhroutev2:sha256:fab06ff6642fd9ed1206b75e4c0341195d29e308b9842eeba44df5c67ee5bd92",
"observation_sequence": 1,
"observed_by": "2026-09-08T01:00:00Z",
"earliest_external_knowledge": {
"status": "unknown"
},
"history_completeness": "not_established",
"evidence_digest": "sha256:ee1434f987443cfc3d097b54be8cb6b39d924ec6f161da86e15042b48d75b799"
},
{
"revision_kind": "trading_calendar",
"revision_id": "rhcalv2:sha256:aba57b52bf328c6455e1a9f51037953646e811aaf0e466032092b310db030e25",
"observation_sequence": 1,
"observed_by": "2026-09-08T01:00:00Z",
"earliest_external_knowledge": {
"status": "unknown"
},
"history_completeness": "not_established",
"evidence_digest": "sha256:10f007ef0c12b2f444230dffdeb5bf185325419bfb0fe2df338730998de7e28a"
}
],
"corporate_action_coverage": [
{
"instrument_id": "rhinstrument:11111111111111111111111111111111",
"effective_time": {
"start_inclusive": "2018-01-02T07:00:00Z",
"end_inclusive": "2018-01-02T07:00:00Z"
},
"observed_by": "2026-09-08T01:00:00Z",
"status": "validated",
"evidence_digests": [
"sha256:233cc2968985e953e59e408b6619a837a792ae17f8afba32261f861de8fe9c7c"
]
},
{
"instrument_id": "rhinstrument:22222222222222222222222222222222",
"effective_time": {
"start_inclusive": "2018-01-02T07:00:00Z",
"end_inclusive": "2018-01-02T07:00:00Z"
},
"observed_by": "2026-09-08T01:00:00Z",
"status": "validated",
"evidence_digests": [
"sha256:21f2594fc42e46168a65d7cac7e9b52aeabf24ecc5fb409530c07ad56ed3986c"
]
}
],
"readiness": {
"evidence_scope": "synthetic_fixture",
"contract_validation": {
"status": "validated",
"evidence_digests": [
"sha256:62f2abbbb79f4ff9164e10cc4b21df37f0964224a4c546f978b97794241ff5b3"
]
},
"real_data_validation": {
"status": "not_validated",
"evidence_digests": []
},
"production_validation": {
"status": "not_validated",
"evidence_digests": []
},
"live_validation": {
"status": "not_validated",
"evidence_digests": []
}
},
"foundation_id": "rhdfv2:sha256:656d8fa2a9e81c87dd8c748bca092a96fb45d3ad9798b5cb2f68c7d4158c8260"
}
@@ -0,0 +1,120 @@
{
"contract_name": "researchhub.dataset-snapshot",
"schema_version": "2.0.0",
"evidence_scope": "synthetic_fixture",
"descriptor": {
"dataset": {
"dataset_id": "rhdataset:market:44444444444444444444444444444444",
"dataset_kind": "market",
"record_schema_version": "2.0.0",
"dimensions": [
"instrument_id",
"effective_time"
]
},
"published_at": "2026-09-08T01:03:00Z",
"time_semantics": {
"effective_time": {
"start_inclusive": "2018-01-02T07:00:00Z",
"end_inclusive": "2018-01-02T07:00:00Z"
},
"observation_cutoff": "2026-09-08T01:01:00Z",
"earliest_external_knowledge": {
"status": "unknown"
},
"historical_availability": "not_established"
},
"content": {
"digest_algorithm": "sha256",
"canonicalization": "RFC8785",
"record_order": "canonical-record-byte-order",
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
"logical_manifest": {
"record_count": 2,
"chunks": [
{
"chunk_index": 0,
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
"record_count": 2
}
]
},
"manifest_digest": "sha256:1c847123277f7973f117b21e597eefdadc93a401e2bec99c94dbb1fb2e3d31c7",
"record_count": 2
},
"observation_manifest": {
"batches": [
{
"chunk_index": 0,
"content_digest": "sha256:b1c0e47b94ffd57e1d34394138d966803c049afc75e1dc0a1a80fea82de5c54a",
"record_count": 2,
"observation_kind": "observed_by",
"observed_by": "2026-09-08T01:00:00Z",
"evidence_digest": "sha256:6f35f19a2ede0610d461b458a0f2cbc96f40cee6e90fa6592fd66daf617d3ec0"
}
]
},
"lineage": {
"publisher": {
"id": "researchhub.data",
"version": "2.0.0"
},
"transformation": {
"id": "rhtransform:55555555555555555555555555555555",
"version": "2.0.0"
},
"upstream_snapshot_ids": [],
"upstream_content_digests": []
},
"quality": {
"status": "passed",
"checks": [
{
"check_id": "completeness",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:519b01c60ad85c2b03d6c4f29027fe9b3672a8f6ae3ca8a2d831b281a61c5095"
},
{
"check_id": "duplicate_identity",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:7c585c7471ab45886fce037061b5ab0d5f981aede5d5190021963676b0ec0fcc"
},
{
"check_id": "observation_coverage",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:dde5693c3ad79a01a162f9fc5a40c69d3c284965e283c91c5cd456904e0ef737"
},
{
"check_id": "historical_claim_policy",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:f4cf8da48f92e03d75bb28b2569061e649f382ba360d879fef8e41e35169c228"
},
{
"check_id": "range_validity",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:6af08355c5e29d846e1c48783dccde791e8a3f48f8416149413217b7c2cbb52a"
},
{
"check_id": "schema_conformance",
"status": "passed",
"severity": "blocking",
"evidence_digest": "sha256:fefdc5b81eba128af92f15a60feb7bb6f40b2e7277bf2beb48c1bb9399ee564b"
}
]
},
"qualification": {
"status": "qualified",
"usage": "retrospective_research",
"policy_id": "researchhub.dataset-snapshot.retrospective",
"policy_version": "2.0.0",
"evaluated_at": "2026-09-08T01:02:00Z",
"evidence_digest": "sha256:70c3ddcadf095877f17a1365c8d75a2e5a7078a27722fb8b698e61d350c6bb2d"
}
},
"snapshot_id": "rhdsv2:sha256:fff0c2f4721407dd8264dacaaec389047fe5533258e940611bd61fb9c8a90905"
}
+11 -3
View File
@@ -9,14 +9,22 @@ ROOT = Path(__file__).resolve().parents[2]
class CiContractTests(unittest.TestCase):
def test_ci_is_one_dependency_free_lite_gate(self) -> None:
def test_ci_is_one_locked_shared_runtime_lite_gate(self) -> None:
workflow = (ROOT / ".gitea/workflows/ci.yml").read_text(encoding="utf-8")
jobs = workflow.split("jobs:", 1)[1]
self.assertEqual(re.findall(r"(?m)^ ([a-z][a-z0-9_-]*):\s*$", jobs), ["lite"])
self.assertIn("actions/checkout@524e936cd9e579adf00e308bfdf971aebc7de09e", workflow)
self.assertIn("persist-credentials: false", workflow)
self.assertIn("python3 tests/governance/test_module_spec.py", workflow)
for forbidden in ("setup-python", "pip ", "curl ", "wget ", "docker pull"):
self.assertIn("UV_PYTHON_DOWNLOADS: never", workflow)
self.assertIn('test "$(python3 --version)" = "Python 3.13.15"', workflow)
self.assertIn(
'test "$(uv --version | cut -d\' \' -f1-2)" = "uv 0.12.3"',
workflow,
)
self.assertIn("uv sync --locked --extra dev", workflow)
self.assertIn("uv run --locked --no-sync python", workflow)
self.assertIn("tests/governance/test_ci_contract.py", workflow)
for forbidden in ("setup-python", "setup-uv", "pip ", "curl ", "wget ", "docker pull"):
self.assertNotIn(forbidden, workflow)
+91 -20
View File
@@ -1,32 +1,103 @@
from __future__ import annotations
import json
import unittest
from pathlib import Path
ROOT = Path(__file__).resolve().parents[2]
class ModuleSpecTests(unittest.TestCase):
def test_module_spec_declares_pure_research_engine_boundary(self) -> None:
spec = json.loads((ROOT / "MODULE_SPEC.yaml").read_text(encoding="utf-8"))
self.assertEqual(spec["module_id"], "quant_engine")
self.assertEqual(spec["authority"]["subject"], spec["module_id"])
self.assertEqual(spec["repository"]["type"], "research_engine")
self.assertEqual(spec["bounded_context"]["domain"], "quantitative-research-engine")
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
for term in ("investment advice", "live order", "credentials", "source facts"):
self.assertIn(term, prohibited)
self.assertEqual(spec["contracts"], {"provides": [], "consumes": []})
self.assertEqual(spec["dependencies"], [])
self.assertTrue(
all(
command["required"] and not command["network"]
for command in spec["verification"]["commands"]
)
)
def test_module_spec_declares_pure_research_engine_boundary() -> None:
spec = json.loads((ROOT / "MODULE_SPEC.yaml").read_text(encoding="utf-8"))
assert spec["module_id"] == "quant_engine"
assert spec["authority"]["subject"] == spec["module_id"]
assert spec["repository"]["type"] == "research_engine"
assert spec["bounded_context"]["domain"] == "quantitative-research-engine"
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
for term in ("investment advice", "live order", "credentials", "source facts"):
assert term in prohibited
assert spec["authority"]["revision"] == 6
assert {
(item["contract_id"], item["version"])
for item in spec["contracts"]["provides"]
} == {
("researchhub.factor-definition", "1.0.0"),
("researchhub.factor-set-ref", "1.0.0"),
("researchhub.backtest-run-ref", "1.0.0"),
("researchhub.backtest-evidence-manifest", "1.0.0"),
("researchhub.performance-evidence", "1.0.0"),
("researchhub.portfolio-decision", "1.0.0"),
("researchhub.risk-assessment", "1.0.0"),
("researchhub.factor-set-ref", "2.0.0"),
("researchhub.backtest-run-ref", "2.0.0"),
("researchhub.backtest-evidence-manifest", "2.0.0"),
("researchhub.performance-evidence", "2.0.0"),
("researchhub.portfolio-target", "2.0.0"),
("researchhub.portfolio-decision", "2.0.0"),
("researchhub.risk-assessment", "2.0.0"),
}
expected_paths = {
"researchhub.factor-definition": "src/quant_engine/factor_contracts.py",
"researchhub.factor-set-ref": "src/quant_engine/factor_contracts.py",
"researchhub.backtest-run-ref": "src/quant_engine/governed_pipeline.py",
"researchhub.backtest-evidence-manifest": "src/quant_engine/artifact.py",
"researchhub.performance-evidence": "src/quant_engine/artifact.py",
"researchhub.portfolio-decision": "src/quant_engine/portfolio_risk_contracts.py",
"researchhub.risk-assessment": "src/quant_engine/portfolio_risk_contracts.py",
}
assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"])
assert {
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
if item["version"] == "1.0.0"
} == expected_paths
assert {
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
if item["version"] == "2.0.0"
} == {
"researchhub.factor-set-ref": "src/quant_engine/retrospective_factor_contracts.py",
"researchhub.backtest-run-ref": "src/quant_engine/retrospective_backtest_contracts.py",
"researchhub.backtest-evidence-manifest": "src/quant_engine/retrospective_artifact_contracts.py",
"researchhub.performance-evidence": "src/quant_engine/retrospective_artifact_contracts.py",
"researchhub.portfolio-target": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
"researchhub.portfolio-decision": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
"researchhub.risk-assessment": "src/quant_engine/retrospective_portfolio_risk_contracts.py",
}
assert {
(item["contract_id"], item["version"])
for item in spec["contracts"]["consumes"]
} == {
("researchhub.dataset-snapshot", "1.0.0"),
("researchhub.data-foundation", "1.0.0"),
("researchhub.dataset-snapshot", "2.0.0"),
("researchhub.data-foundation", "2.0.0"),
}
assert all(
item["authority"] == "researchhub.data"
for item in spec["contracts"]["consumes"]
)
assert spec["dependencies"] == []
capabilities = {item["id"]: item for item in spec["capabilities"]}
evidence_contract = capabilities["backtest-evidence-contracts"]
assert evidence_contract["status"] == "operational"
evidence_summary = evidence_contract["summary"].lower()
for term in ("performance-methodology", "without recomputation", "decision authority"):
assert term in evidence_summary
portfolio_contract = capabilities["portfolio-risk-computation-contracts"]
assert portfolio_contract["status"] == "operational"
summary = portfolio_contract["summary"].lower()
for term in ("receipt", "portfolio decisions", "risk assessments", "without"):
assert term in summary
retrospective = capabilities["retrospective-computation-contracts"]
assert retrospective["status"] == "operational"
for term in ("two clocks", "retrospective", "no historical-availability", "no execution"):
assert term in retrospective["summary"].lower()
for term in ("approval", "maker-checker", "publication", "paper", "live"):
assert term in prohibited
assert all(
command["required"] and not command["network"]
for command in spec["verification"]["commands"]
)
if __name__ == "__main__":
unittest.main()
test_module_spec_declares_pure_research_engine_boundary()
+976
View File
@@ -6,8 +6,30 @@ import numpy as np
import pandas as pd
import pytest
import quant_engine.alpha_factors as alpha_factors_module
from quant_engine.factor_contracts import (
FactorContractError,
FactorInput,
ProducerIdentity,
factor_definition_from_alpha158,
factor_input_schema_digest,
)
from quant_engine.alpha_factors import (
ALPHA158_REGISTRY,
ALPHA158_PHASE1_OPERATOR_SPECS,
ALPHA158_PHASE2_OPERATOR_SPECS,
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE3_FORMULA_SPECS,
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE4_FORMULA_SPECS,
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE5_FORMULA_SPECS,
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE6_FORMULA_SPECS,
alpha_001,
alpha_002,
alpha_003,
@@ -166,6 +188,18 @@ from quant_engine.alpha_factors import (
alpha_156,
alpha_157,
alpha_158,
evaluate_phase1_operator,
evaluate_phase2_operator,
evaluate_phase3_formula,
evaluate_phase4_formula,
evaluate_phase5_formula,
evaluate_phase6_formula,
list_phase1_operators,
list_phase2_operators,
list_phase3_formulas,
list_phase4_formulas,
list_phase5_formulas,
list_phase6_formulas,
correlation,
covariance,
decay_linear,
@@ -407,6 +441,50 @@ def test_alpha_registry_required_fields():
assert required <= set(meta.keys()), f"{alpha_id} missing fields"
def test_alpha_registry_adapts_to_definition_without_copying_formula_or_inputs():
factor_input = FactorInput(
"market",
"sha256:" + "1" * 64,
tuple(ALPHA158_REGISTRY["alpha_005"]["inputs"]),
)
definition = factor_definition_from_alpha158(
"alpha_005",
version="1.0.0",
parameters={},
inputs=(factor_input,),
implementation_digest="sha256:" + "2" * 64,
input_schema_digest=factor_input_schema_digest((factor_input,)),
valid_from="2026-01-01T00:00:00Z",
valid_until="2027-01-01T00:00:00Z",
warmup_sessions=10,
lag_sessions=1,
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
)
assert definition.formula == ALPHA158_REGISTRY["alpha_005"]["formula"]
assert definition.inputs[0].required_columns == tuple(
ALPHA158_REGISTRY["alpha_005"]["inputs"]
)
incomplete = FactorInput("market", "sha256:" + "1" * 64, ("close",))
with pytest.raises(FactorContractError, match="exactly correspond"):
factor_definition_from_alpha158(
"alpha_005",
version="1.0.0",
parameters={},
inputs=(incomplete,),
implementation_digest="sha256:" + "2" * 64,
input_schema_digest=factor_input_schema_digest((incomplete,)),
valid_from="2026-01-01T00:00:00Z",
valid_until="2027-01-01T00:00:00Z",
warmup_sessions=10,
lag_sessions=1,
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
)
def test_get_alpha_meta_success():
"""已知 alpha_id 返回完整 meta。"""
meta = get_alpha_meta("alpha_001")
@@ -1232,3 +1310,901 @@ def test_parse_alpha_formula_round_trip_jsonb():
serialized = json.dumps(parsed)
assert isinstance(serialized, str)
assert "ts_rank" in serialized
# ── v1.2.0 Phase 1: deterministic operator dispatch contract ──────────────
def test_phase1_operator_catalog_is_explicit_and_serializable():
"""Phase 1 exposes a stable, JSON-friendly catalog for downstream callers."""
import json
expected = {
"rank",
"delta",
"ts_mean",
"ts_std",
"ts_rank",
"correlation",
"ts_min",
"ts_max",
"ts_sum",
"decay_linear",
}
assert set(list_phase1_operators()) == expected
assert set(ALPHA158_PHASE1_OPERATOR_SPECS) == expected
json.dumps(ALPHA158_PHASE1_OPERATOR_SPECS)
for name, spec in ALPHA158_PHASE1_OPERATOR_SPECS.items():
assert spec["name"] == name
assert isinstance(spec["inputs"], list)
assert isinstance(spec["formula"], str)
def test_phase1_unary_operators_preserve_index_and_are_deterministic():
values = pd.Series([1.0, 2.0, 3.0, 4.0], index=["a", "b", "c", "d"])
first = evaluate_phase1_operator("rank", values)
second = evaluate_phase1_operator("rank", values)
pd.testing.assert_series_equal(first, second)
assert first.index.equals(values.index)
assert first.iloc[-1] == pytest.approx(1.0)
@pytest.mark.parametrize(
("name", "window"),
[
("delta", 2),
("ts_mean", 2),
("ts_std", 2),
("ts_rank", 2),
("ts_min", 2),
("ts_max", 2),
("ts_sum", 2),
("decay_linear", 2),
],
)
def test_phase1_windowed_operators_require_explicit_window(name: str, window: int):
values = pd.Series([1.0, 2.0, 3.0, 4.0])
result = evaluate_phase1_operator(name, values, window=window)
assert result.index.equals(values.index)
with pytest.raises(ValueError, match="window"):
evaluate_phase1_operator(name, values)
with pytest.raises(ValueError, match="positive integer"):
evaluate_phase1_operator(name, values, window=1.5) # type: ignore[arg-type]
def test_phase1_binary_correlation_requires_aligned_secondary_input():
values = pd.Series([1.0, 2.0, 3.0, 4.0])
other = pd.Series([4.0, 3.0, 2.0, 1.0])
result = evaluate_phase1_operator("correlation", values, other, window=2)
assert result.iloc[-1] == pytest.approx(-1.0)
with pytest.raises(ValueError, match="secondary"):
evaluate_phase1_operator("correlation", values, window=2)
def test_phase1_dispatch_rejects_unknown_or_unused_arguments():
values = pd.Series([1.0, 2.0, 3.0])
with pytest.raises(KeyError, match="not registered"):
evaluate_phase1_operator("unknown", values)
with pytest.raises(ValueError, match="window"):
evaluate_phase1_operator("rank", values, window=2)
with pytest.raises(ValueError, match="secondary"):
evaluate_phase1_operator("rank", values, values)
def test_phase1_dispatch_rejects_window_above_supported_limit():
values = pd.Series([1.0, 2.0, 3.0])
with pytest.raises(ValueError, match="maximum"):
evaluate_phase1_operator("ts_mean", values, window=2**63)
# ── v1.2.0 Phase 2: cumulative deterministic operator contract ─────────────
def test_phase2_operator_catalog_is_cumulative_stable_and_serializable():
"""Phase 2 exposes all existing building blocks without changing Phase 1."""
import json
phase1 = list_phase1_operators()
expected_phase2 = (
*phase1,
"ts_argmin",
"ts_argmax",
"product",
"returns",
"scale",
"signed_power",
"stddev",
"covariance",
"log",
"abs_series",
"sign",
"max_pair",
"min_pair",
"indneutralize",
)
assert list_phase2_operators() == expected_phase2
assert tuple(ALPHA158_PHASE2_OPERATOR_SPECS) == expected_phase2
assert tuple(ALPHA158_PHASE1_OPERATOR_SPECS) == phase1
json.dumps(ALPHA158_PHASE2_OPERATOR_SPECS)
for name, spec in ALPHA158_PHASE2_OPERATOR_SPECS.items():
assert spec["name"] == name
assert isinstance(spec["inputs"], list)
assert isinstance(spec["parameters"], list)
assert isinstance(spec["formula"], str)
@pytest.mark.parametrize("name", ["ts_argmin", "ts_argmax", "product", "stddev"])
def test_phase2_windowed_unary_dispatch_is_deterministic(name: str):
values = pd.Series([3.0, 1.0, 4.0, 2.0], index=["a", "b", "c", "d"])
first = evaluate_phase2_operator(name, values, window=3)
second = evaluate_phase2_operator(name, values, window=3)
pd.testing.assert_series_equal(first, second)
assert first.index.equals(values.index)
with pytest.raises(ValueError, match="window"):
evaluate_phase2_operator(name, values)
@pytest.mark.parametrize("name", ["returns", "scale", "log", "abs_series", "sign"])
def test_phase2_unary_dispatch_rejects_unused_arguments(name: str):
values = pd.Series([1.0, 2.0, 4.0], index=["a", "b", "c"])
result = evaluate_phase2_operator(name, values)
assert result.index.equals(values.index)
with pytest.raises(ValueError, match="window"):
evaluate_phase2_operator(name, values, window=2)
with pytest.raises(ValueError, match="secondary"):
evaluate_phase2_operator(name, values, secondary=values)
@pytest.mark.parametrize(
("name", "window"),
[("correlation", 2), ("covariance", 2), ("max_pair", None), ("min_pair", None)],
)
def test_phase2_binary_dispatch_requires_aligned_secondary(name: str, window: int | None):
values = pd.Series([1.0, 2.0, 3.0], index=["a", "b", "c"])
secondary = pd.Series([3.0, 2.0, 1.0], index=values.index)
result = evaluate_phase2_operator(name, values, secondary=secondary, window=window)
assert result.index.equals(values.index)
with pytest.raises(ValueError, match="secondary is required"):
evaluate_phase2_operator(name, values, window=window)
with pytest.raises(ValueError, match="secondary index"):
evaluate_phase2_operator(
name,
values,
secondary=secondary.rename(index={"c": "z"}),
window=window,
)
def test_phase2_signed_power_requires_finite_numeric_exponent():
values = pd.Series([-4.0, 0.0, 9.0])
result = evaluate_phase2_operator("signed_power", values, exponent=0.5)
pd.testing.assert_series_equal(result, pd.Series([-2.0, 0.0, 3.0]))
for exponent in (None, True, float("inf"), float("nan"), "2"):
with pytest.raises(ValueError, match="exponent"):
evaluate_phase2_operator( # type: ignore[arg-type]
"signed_power",
values,
exponent=exponent,
)
def test_phase2_indneutralize_requires_aligned_groups():
values = pd.Series([1.0, 3.0, 10.0, 14.0], index=["a", "b", "c", "d"])
groups = pd.Series(["x", "x", "y", "y"], index=values.index)
result = evaluate_phase2_operator("indneutralize", values, groups=groups)
pd.testing.assert_series_equal(result, pd.Series([-1.0, 1.0, -2.0, 2.0], index=values.index))
with pytest.raises(ValueError, match="groups is required"):
evaluate_phase2_operator("indneutralize", values)
with pytest.raises(ValueError, match="groups index"):
evaluate_phase2_operator(
"indneutralize",
values,
groups=groups.rename(index={"d": "z"}),
)
def test_phase2_dispatch_validates_primary_series_and_unused_parameters():
values = pd.Series([1.0, 2.0, 3.0])
with pytest.raises(TypeError, match="series must be a pandas Series"):
evaluate_phase2_operator("rank", [1.0, 2.0, 3.0]) # type: ignore[arg-type]
with pytest.raises(KeyError, match="not registered"):
evaluate_phase2_operator("unknown", values)
with pytest.raises(ValueError, match="exponent"):
evaluate_phase2_operator("rank", values, exponent=2.0)
with pytest.raises(ValueError, match="groups"):
evaluate_phase2_operator("rank", values, groups=pd.Series(["x", "x", "x"]))
with pytest.raises(ValueError, match="maximum"):
evaluate_phase2_operator("product", values, window=253)
# ── Alpha158 Phase 3: versioned alpha001-alpha050 formula contract ──────────
def _phase3_market_inputs() -> dict[str, pd.Series]:
positions = np.arange(80, dtype=float)
index = pd.RangeIndex(len(positions), name="row")
open_ = pd.Series(100.0 + positions * 0.2 + np.sin(positions / 4.0), index=index)
close = pd.Series(100.5 + positions * 0.18 + np.cos(positions / 5.0), index=index)
high = pd.Series(np.maximum(open_, close) + 1.0, index=index)
low = pd.Series(np.minimum(open_, close) - 1.0, index=index)
volume = pd.Series(1_000.0 + positions**1.3 + 20.0 * np.sin(positions / 3.0), index=index)
vwap = (open_ + close + high + low) / 4.0
return {
"open": open_,
"close": close,
"high": high,
"low": low,
"volume": volume,
"vwap": vwap,
}
def test_phase3_formula_catalog_is_versioned_exact_and_content_addressed():
import hashlib
import json
from collections import Counter
expected_ids = tuple(f"alpha_{number:03d}" for number in range(1, 51))
expected_fields = {
"name",
"contract_version",
"formula",
"category",
"complexity",
"parameters",
"description",
"references",
"call_inputs",
"formula_inputs",
"input_category",
}
assert ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION == "1.0.0"
assert list_phase3_formulas() == expected_ids
assert tuple(ALPHA158_PHASE3_FORMULA_SPECS) == expected_ids
assert Counter(
spec["input_category"] for spec in ALPHA158_PHASE3_FORMULA_SPECS.values()
) == {"single": 19, "pair": 26, "triple": 4, "quadruple": 1}
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
assert set(spec) == expected_fields
assert spec["name"] == alpha_id
assert spec["contract_version"] == ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
serializable_specs = {
alpha_id: {
field: list(value) if isinstance(value, tuple) else value
for field, value in spec.items()
}
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items()
}
encoded = json.dumps(
serializable_specs,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
).encode()
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE3_FORMULA_CATALOG_SHA256
assert ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 == (
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
)
def test_phase3_formula_catalog_is_recursively_immutable():
import operator
with pytest.raises(TypeError):
operator.setitem(ALPHA158_PHASE3_FORMULA_SPECS, "alpha_001", {})
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"],
"formula",
"changed",
)
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"]["call_inputs"],
0,
"volume",
)
def test_phase3_catalog_freezes_callable_signatures_without_rewriting_formulas():
import inspect
legacy_formula_input_differences = {
"alpha_011": ("close", "high", "low"),
"alpha_035": ("volume",),
"alpha_036": ("close",),
"alpha_040": ("high", "low"),
"alpha_042": ("close",),
"alpha_043": ("volume",),
}
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
function = getattr(alpha_factors_module, alpha_id)
signature_inputs = tuple(
"open" if name == "open_" else name
for name in inspect.signature(function).parameters
)
assert spec["call_inputs"] == signature_inputs
assert spec["formula_inputs"] == legacy_formula_input_differences.get(
alpha_id,
signature_inputs,
)
assert ALPHA158_PHASE3_FORMULA_SPECS["alpha_035"]["call_inputs"] == (
"close",
"volume",
)
assert ALPHA158_REGISTRY["alpha_035"]["inputs"] == ["volume"]
def test_phase3_dispatch_matches_all_existing_alpha001_alpha050_functions():
inputs = _phase3_market_inputs()
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
call_inputs = spec["call_inputs"]
function = getattr(alpha_factors_module, alpha_id)
expected = function(*(inputs[name] for name in call_inputs))
actual = evaluate_phase3_formula(
alpha_id,
**{name: inputs[name] for name in reversed(call_inputs)},
)
pd.testing.assert_series_equal(actual, expected)
def test_phase3_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
inputs = _phase3_market_inputs()
with pytest.raises(KeyError, match="not registered"):
evaluate_phase3_formula("alpha_051", close=inputs["close"])
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
evaluate_phase3_formula("alpha_005", close=inputs["close"])
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
evaluate_phase3_formula(
"alpha_005",
close=inputs["close"],
volume=inputs["volume"],
vwap=inputs["vwap"],
)
with pytest.raises(TypeError, match="close must be a pandas Series"):
evaluate_phase3_formula( # type: ignore[arg-type]
"alpha_005",
close=[1.0, 2.0],
volume=inputs["volume"],
)
def test_phase3_dispatch_rejects_implicit_series_alignment():
inputs = _phase3_market_inputs()
misaligned_volume = inputs["volume"].rename(index={79: 80})
with pytest.raises(ValueError, match="volume index must align with close"):
evaluate_phase3_formula(
"alpha_005",
close=inputs["close"],
volume=misaligned_volume,
)
# ── Alpha158 Phase 4: versioned alpha051-alpha100 formula contract ──────────
def test_phase4_formula_catalog_is_versioned_exact_and_content_addressed():
import hashlib
import json
from collections import Counter
expected_ids = tuple(f"alpha_{number:03d}" for number in range(51, 101))
expected_fields = {
"name",
"contract_version",
"formula",
"category",
"complexity",
"parameters",
"description",
"references",
"call_inputs",
"formula_inputs",
"input_category",
}
assert ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION == "1.0.0"
assert list_phase4_formulas() == expected_ids
assert tuple(ALPHA158_PHASE4_FORMULA_SPECS) == expected_ids
assert Counter(
spec["input_category"] for spec in ALPHA158_PHASE4_FORMULA_SPECS.values()
) == {"single": 1, "pair": 27, "triple": 11, "quadruple": 10, "quintuple": 1}
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
assert set(spec) == expected_fields
assert spec["name"] == alpha_id
assert spec["contract_version"] == ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
serializable_specs = {
alpha_id: {
field: list(value) if isinstance(value, tuple) else value
for field, value in spec.items()
}
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items()
}
encoded = json.dumps(
serializable_specs,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
).encode()
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE4_FORMULA_CATALOG_SHA256
assert ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 == (
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
)
def test_phase4_formula_catalog_is_recursively_immutable():
import operator
with pytest.raises(TypeError):
operator.setitem(ALPHA158_PHASE4_FORMULA_SPECS, "alpha_051", {})
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"],
"formula",
"changed",
)
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"]["call_inputs"],
0,
"volume",
)
def test_phase4_catalog_freezes_callable_and_formula_inputs():
import inspect
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
function = getattr(alpha_factors_module, alpha_id)
signature_inputs = tuple(
"open" if name == "open_" else name
for name in inspect.signature(function).parameters
)
assert spec["call_inputs"] == signature_inputs
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
assert ALPHA158_PHASE4_FORMULA_SPECS["alpha_055"]["call_inputs"] == (
"open",
"high",
"low",
"volume",
"close",
)
def test_phase4_dispatch_matches_all_existing_alpha051_alpha100_functions():
inputs = _phase3_market_inputs()
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
call_inputs = spec["call_inputs"]
function = getattr(alpha_factors_module, alpha_id)
expected = function(*(inputs[name] for name in call_inputs))
actual = evaluate_phase4_formula(
alpha_id,
**{name: inputs[name] for name in reversed(call_inputs)},
)
pd.testing.assert_series_equal(actual, expected)
def test_phase4_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
inputs = _phase3_market_inputs()
with pytest.raises(KeyError, match="not registered"):
evaluate_phase4_formula("alpha_050", close=inputs["close"])
with pytest.raises(ValueError, match=r"missing inputs.*low"):
evaluate_phase4_formula("alpha_051", high=inputs["high"])
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
evaluate_phase4_formula(
"alpha_051",
high=inputs["high"],
low=inputs["low"],
vwap=inputs["vwap"],
)
with pytest.raises(TypeError, match="high must be a pandas Series"):
evaluate_phase4_formula( # type: ignore[arg-type]
"alpha_051",
high=[1.0, 2.0],
low=inputs["low"],
)
def test_phase4_dispatch_rejects_implicit_series_alignment():
inputs = _phase3_market_inputs()
misaligned_low = inputs["low"].rename(index={79: 80})
with pytest.raises(ValueError, match="low index must align with high"):
evaluate_phase4_formula(
"alpha_051",
high=inputs["high"],
low=misaligned_low,
)
# ── Alpha158 Phase 5: versioned alpha101-alpha150 formula contract ──────────
def test_phase5_formula_catalog_is_versioned_exact_and_content_addressed():
import hashlib
import json
from collections import Counter
expected_ids = tuple(f"alpha_{number:03d}" for number in range(101, 151))
expected_fields = {
"name",
"contract_version",
"formula",
"category",
"complexity",
"parameters",
"description",
"references",
"call_inputs",
"formula_inputs",
"input_category",
}
assert ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION == "1.0.0"
assert list_phase5_formulas() == expected_ids
assert tuple(ALPHA158_PHASE5_FORMULA_SPECS) == expected_ids
assert Counter(
spec["input_category"] for spec in ALPHA158_PHASE5_FORMULA_SPECS.values()
) == {"pair": 33, "triple": 14, "quadruple": 3}
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
assert set(spec) == expected_fields
assert spec["name"] == alpha_id
assert spec["contract_version"] == ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
serializable_specs = {
alpha_id: {
field: list(value) if isinstance(value, tuple) else value
for field, value in spec.items()
}
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items()
}
encoded = json.dumps(
serializable_specs,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
).encode()
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE5_FORMULA_CATALOG_SHA256
assert ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 == (
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
)
def test_phase5_formula_catalog_is_recursively_immutable():
import operator
with pytest.raises(TypeError):
operator.setitem(ALPHA158_PHASE5_FORMULA_SPECS, "alpha_101", {})
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"],
"formula",
"changed",
)
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"],
0,
"volume",
)
def test_phase5_catalog_freezes_callable_and_formula_inputs():
import inspect
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
function = getattr(alpha_factors_module, alpha_id)
signature_inputs = tuple(
"open" if name == "open_" else name
for name in inspect.signature(function).parameters
)
assert spec["call_inputs"] == signature_inputs
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
assert ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"] == (
"close",
"high",
"low",
)
def test_phase5_dispatch_matches_all_existing_alpha101_alpha150_functions():
inputs = _phase3_market_inputs()
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
call_inputs = spec["call_inputs"]
function = getattr(alpha_factors_module, alpha_id)
expected = function(*(inputs[name] for name in call_inputs))
actual = evaluate_phase5_formula(
alpha_id,
**{name: inputs[name] for name in reversed(call_inputs)},
)
pd.testing.assert_series_equal(actual, expected)
def test_phase5_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
inputs = _phase3_market_inputs()
with pytest.raises(KeyError, match="not registered"):
evaluate_phase5_formula("alpha_100", close=inputs["close"])
with pytest.raises(KeyError, match="not registered"):
evaluate_phase5_formula("alpha_151", close=inputs["close"])
with pytest.raises(ValueError, match=r"missing inputs.*low"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=inputs["high"],
)
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=inputs["high"],
low=inputs["low"],
vwap=inputs["vwap"],
)
with pytest.raises(TypeError, match="high must be a pandas Series"):
evaluate_phase5_formula( # type: ignore[arg-type]
"alpha_101",
close=inputs["close"],
high=[1.0, 2.0],
low=inputs["low"],
)
def test_phase5_dispatch_rejects_length_and_index_alignment_errors():
inputs = _phase3_market_inputs()
shorter_low = inputs["low"].iloc[:-1]
misaligned_high = inputs["high"].rename(index={79: 80})
with pytest.raises(ValueError, match="low length must match close"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=inputs["high"],
low=shorter_low,
)
with pytest.raises(ValueError, match="high index must align with close"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=misaligned_high,
low=inputs["low"],
)
def test_phase5_contract_is_publicly_exported():
assert {
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE5_FORMULA_SPECS",
"list_phase5_formulas",
"evaluate_phase5_formula",
} <= set(alpha_factors_module.__all__)
# ── Alpha158 Phase 6: versioned alpha151-alpha158 formula contract ──────────
def test_phase6_formula_catalog_is_versioned_exact_and_content_addressed():
import hashlib
import json
from collections import Counter
expected_ids = tuple(f"alpha_{number:03d}" for number in range(151, 159))
expected_fields = {
"name",
"contract_version",
"formula",
"category",
"complexity",
"parameters",
"description",
"references",
"call_inputs",
"formula_inputs",
"input_category",
}
assert ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION == "1.0.0"
assert list_phase6_formulas() == expected_ids
assert tuple(ALPHA158_PHASE6_FORMULA_SPECS) == expected_ids
assert Counter(
spec["input_category"] for spec in ALPHA158_PHASE6_FORMULA_SPECS.values()
) == {"pair": 6, "triple": 2}
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
assert set(spec) == expected_fields
assert spec["name"] == alpha_id
assert spec["contract_version"] == ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
serializable_specs = {
alpha_id: {
field: list(value) if isinstance(value, tuple) else value
for field, value in spec.items()
}
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items()
}
encoded = json.dumps(
serializable_specs,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
).encode()
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE6_FORMULA_CATALOG_SHA256
assert ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 == (
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
)
def test_phase6_formula_catalog_is_recursively_immutable():
import operator
with pytest.raises(TypeError):
operator.setitem(ALPHA158_PHASE6_FORMULA_SPECS, "alpha_151", {})
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"],
"formula",
"changed",
)
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"]["call_inputs"],
0,
"volume",
)
def test_phase6_catalog_freezes_callable_and_formula_inputs():
import inspect
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
function = getattr(alpha_factors_module, alpha_id)
signature_inputs = tuple(
"open" if name == "open_" else name
for name in inspect.signature(function).parameters
)
assert spec["call_inputs"] == signature_inputs
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
assert ALPHA158_PHASE6_FORMULA_SPECS["alpha_158"]["call_inputs"] == (
"high",
"low",
"volume",
)
def test_phase6_dispatch_matches_all_existing_alpha151_alpha158_functions():
inputs = _phase3_market_inputs()
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
call_inputs = spec["call_inputs"]
function = getattr(alpha_factors_module, alpha_id)
expected = function(*(inputs[name] for name in call_inputs))
actual = evaluate_phase6_formula(
alpha_id,
**{name: inputs[name] for name in reversed(call_inputs)},
)
pd.testing.assert_series_equal(actual, expected)
def test_phase6_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
inputs = _phase3_market_inputs()
with pytest.raises(KeyError, match="not registered"):
evaluate_phase6_formula("alpha_150", close=inputs["close"])
with pytest.raises(KeyError, match="not registered"):
evaluate_phase6_formula("alpha_159", close=inputs["close"])
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
evaluate_phase6_formula("alpha_151", close=inputs["close"])
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
evaluate_phase6_formula(
"alpha_151",
close=inputs["close"],
volume=inputs["volume"],
vwap=inputs["vwap"],
)
with pytest.raises(TypeError, match="volume must be a pandas Series"):
evaluate_phase6_formula( # type: ignore[arg-type]
"alpha_151",
close=inputs["close"],
volume=[1.0, 2.0],
)
def test_phase6_dispatch_rejects_length_and_index_alignment_errors():
inputs = _phase3_market_inputs()
shorter_volume = inputs["volume"].iloc[:-1]
misaligned_low = inputs["low"].rename(index={79: 80})
with pytest.raises(ValueError, match="volume length must match close"):
evaluate_phase6_formula(
"alpha_151",
close=inputs["close"],
volume=shorter_volume,
)
with pytest.raises(ValueError, match="low index must align with high"):
evaluate_phase6_formula(
"alpha_158",
high=inputs["high"],
low=misaligned_low,
volume=inputs["volume"],
)
def test_phase6_preserves_complete_alpha001_alpha158_formula_body_fingerprint():
import ast
import hashlib
import inspect
import json
import textwrap
fingerprints = {}
for number in range(1, 159):
alpha_id = f"alpha_{number:03d}"
source = textwrap.dedent(inspect.getsource(getattr(alpha_factors_module, alpha_id)))
node = ast.parse(source).body[0]
assert isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef))
body = ast.dump(
ast.Module(body=node.body, type_ignores=[]),
include_attributes=False,
)
fingerprints[alpha_id] = hashlib.sha256(body.encode()).hexdigest()
encoded = json.dumps(
fingerprints,
sort_keys=True,
separators=(",", ":"),
).encode()
assert hashlib.sha256(encoded).hexdigest() == (
"f3ae807983cf8ca083d0b924c3807ffd84a62a5bd354f9efb1a1a16ca40c7da6"
)
def test_phase6_contract_is_publicly_exported():
assert {
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE6_FORMULA_SPECS",
"list_phase6_formulas",
"evaluate_phase6_formula",
} <= set(alpha_factors_module.__all__)
+309
View File
@@ -0,0 +1,309 @@
"""Stable research-run artifact contracts for downstream persistence."""
from __future__ import annotations
import json
from datetime import date
import pandas as pd
import pytest
from quant_engine.artifact import (
RESEARCH_ARTIFACT_SCHEMA_VERSION,
ResearchRunArtifact,
build_research_run_artifact,
)
from quant_engine.execution import ExecutionConfig
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
from quant_engine.risk import CovarianceSnapshot
def _backtest_result() -> FactorBacktestResult:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
scores = pd.DataFrame(
{"A": [2.0, 0.0], "B": [1.0, 3.0]},
index=dates[:2],
)
opens = pd.DataFrame(
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
index=dates,
)
closes = pd.DataFrame(
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
index=dates,
)
return run_factor_backtest_research(
scores,
opens,
closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
),
)
def _build(
result: FactorBacktestResult,
*,
parameters: dict[str, object] | None = None,
risk_snapshots: dict[date, CovarianceSnapshot] | None = None,
) -> ResearchRunArtifact:
benchmark = pd.Series(
[0.0, 0.01, -0.01, 0.02],
index=result.returns.index,
name="benchmark_return",
)
return build_research_run_artifact(
result,
run_id="run-20260105-a",
strategy_id="alpha-top1",
strategy_name="Alpha Top 1",
strategy_version="1.0.0",
engine_version="1.2.0",
code_revision="3b1ad07",
data_snapshot_id="qtdb-pro-20260108-v1",
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-01-08T10:00:00+08:00",
finished_at="2026-01-08T10:01:00+08:00",
parameters=parameters or {"top_k": 1, "lag_sessions": 1},
benchmark_id="000300.SH",
benchmark_returns=benchmark,
risk_snapshots=risk_snapshots,
)
def test_research_artifact_projects_versioned_queryable_fact_tables() -> None:
result = _backtest_result()
artifact = _build(result)
assert artifact.schema_version == RESEARCH_ARTIFACT_SCHEMA_VERSION
assert artifact.run.loc[0, "run_id"] == "run-20260105-a"
assert artifact.run.loc[0, "benchmark_alignment_policy"] == "exact_session_index"
assert artifact.nav["run_id"].unique().tolist() == ["run-20260105-a"]
assert artifact.nav["pnl_pct"].tolist() == pytest.approx(result.returns.tolist())
assert artifact.nav["benchmark_return"].tolist() == pytest.approx(
[0.0, 0.01, -0.01, 0.02]
)
assert artifact.signals.columns.tolist() == [
"run_id",
"signal_date",
"execution_date",
"asset_id",
"factor_score",
"target_weight",
]
first_signal = artifact.signals[
artifact.signals["signal_date"] == result.factor_scores.index[0].date()
]
assert first_signal.set_index("asset_id").loc["A", "factor_score"] == 2.0
assert first_signal.set_index("asset_id").loc["A", "target_weight"] == 1.0
assert first_signal["execution_date"].unique().tolist() == [
result.schedule.signal_to_execution.iloc[0].date()
]
assert set(artifact.trades["side"]) == {"buy", "sell"}
assert artifact.trades["trade_id"].is_unique
assert artifact.trades["trade_id"].str.startswith("run-20260105-a:").all()
assert artifact.trades["signal_id"].str.startswith("run-20260105-a:signal:").all()
assert {"security", "cash"}.issubset(set(artifact.positions["asset_type"]))
assert artifact.positions.groupby("trade_date")["weight"].sum().tolist() == pytest.approx(
[1.0, 1.0, 1.0, 1.0]
)
assert set(artifact.attribution.columns) == {
"run_id",
"trade_date",
"asset_id",
"overnight",
"intraday",
"asset_total",
}
assert artifact.attribution_daily["residual"].abs().max() < 1e-12
assert artifact.risk.empty
assert artifact.risk.columns.tolist() == [
"run_id",
"trade_date",
"asset_id",
"weight",
"marginal_risk",
"component_risk",
"risk_contribution",
"covariance_snapshot_id",
"covariance_as_of_date",
"risk_measure",
"return_frequency",
"periods_per_year",
]
assert artifact.performance.loc[0, "n_trades"] == len(artifact.trades)
assert artifact.performance.loc[0, "ir"] == pytest.approx(
result.benchmark_stats(pd.Series([0.0, 0.01, -0.01, 0.02], index=result.returns.index))[
"information_ratio"
]
)
assert "sortino" in artifact.performance.columns
def test_research_artifact_projects_annualized_risk_from_actual_positions() -> None:
result = _backtest_result()
trade_date = result.position_weights.index[-1].date()
covariance = pd.DataFrame(
[[0.0001, 0.00002], [0.00002, 0.0004]],
index=["A", "B"],
columns=["A", "B"],
)
snapshot = CovarianceSnapshot(
snapshot_id="cov-20260107-v1",
as_of_date="2026-01-07",
covariance=covariance,
return_frequency="1d",
periods_per_year=252,
data_snapshot_id="qtdb-pro-20260108-v1",
)
artifact = _build(result, risk_snapshots={trade_date: snapshot})
risk = artifact.risk.set_index("asset_id")
expected_weights = result.position_weights.loc[pd.Timestamp(trade_date)]
assert artifact.schema_version == "1.1.0"
assert risk.index.tolist() == ["A", "B"]
assert risk["weight"].tolist() == pytest.approx(expected_weights.tolist())
assert risk["covariance_snapshot_id"].unique().tolist() == ["cov-20260107-v1"]
assert risk["covariance_as_of_date"].unique().tolist() == [date(2026, 1, 7)]
assert risk["risk_measure"].unique().tolist() == ["annualized_volatility"]
assert risk["return_frequency"].unique().tolist() == ["1d"]
assert risk["periods_per_year"].unique().tolist() == [252]
assert risk["component_risk"].sum() == pytest.approx((0.0004 * 252) ** 0.5)
assert risk["risk_contribution"].sum() == pytest.approx(1.0)
def test_research_artifact_rejects_risk_from_a_different_data_snapshot() -> None:
result = _backtest_result()
trade_date = result.position_weights.index[-1].date()
covariance = pd.DataFrame(
[[0.0001, 0.0], [0.0, 0.0004]],
index=["A", "B"],
columns=["A", "B"],
)
with pytest.raises(ValueError, match="data lineage differs"):
_build(
result,
risk_snapshots={
trade_date: CovarianceSnapshot(
snapshot_id="foreign-covariance",
as_of_date="2026-01-07",
covariance=covariance,
return_frequency="1d",
periods_per_year=252,
data_snapshot_id="different-market-snapshot",
)
},
)
def test_research_artifact_rejects_future_or_misaligned_risk_snapshots() -> None:
result = _backtest_result()
trade_date = result.position_weights.index[-1].date()
covariance = pd.DataFrame(
[[0.0001, 0.0], [0.0, 0.0004]],
index=["A", "B"],
columns=["A", "B"],
)
with pytest.raises(ValueError, match="must not be after trade date"):
_build(
result,
risk_snapshots={
trade_date: CovarianceSnapshot(
snapshot_id="future-covariance",
as_of_date="2026-01-09",
covariance=covariance,
return_frequency="1d",
periods_per_year=252,
data_snapshot_id="qtdb-pro-20260108-v1",
)
},
)
with pytest.raises(ValueError, match="same asset labels"):
_build(
result,
risk_snapshots={
trade_date: CovarianceSnapshot(
snapshot_id="incomplete-universe",
as_of_date="2026-01-07",
covariance=covariance.loc[["B"], ["B"]],
return_frequency="1d",
periods_per_year=252,
data_snapshot_id="qtdb-pro-20260108-v1",
)
},
)
def test_research_artifact_serialization_and_hashes_are_deterministic() -> None:
result = _backtest_result()
first = _build(result, parameters={"top_k": 1, "lag_sessions": 1})
second = _build(result, parameters={"lag_sessions": 1, "top_k": 1})
assert first.run.loc[0, "config_hash"] == second.run.loc[0, "config_hash"]
assert first.content_sha256 == second.content_sha256
assert first.manifest() == second.manifest()
decoded = json.loads(first.canonical_json())
assert decoded["schema_version"] == RESEARCH_ARTIFACT_SCHEMA_VERSION
assert decoded["tables"]["nav"][0]["trade_date"] == "2026-01-05"
leaked_copy = first.nav
leaked_copy.loc[0, "nav"] = -999.0
assert first.nav.loc[0, "nav"] != -999.0
assert first.content_sha256 == second.content_sha256
def test_research_artifact_requires_complete_reproducibility_identity() -> None:
result = _backtest_result()
with pytest.raises(ValueError, match="code_revision"):
build_research_run_artifact(
result,
run_id="run-1",
strategy_id="alpha-top1",
strategy_name="Alpha Top 1",
strategy_version="1.0.0",
engine_version="1.2.0",
code_revision="",
data_snapshot_id="snapshot-1",
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-01-08T10:00:00+08:00",
finished_at="2026-01-08T10:01:00+08:00",
parameters={},
)
def test_research_artifact_requires_benchmark_identity_and_returns_together() -> None:
result = _backtest_result()
with pytest.raises(ValueError, match="benchmark_id and benchmark_returns"):
build_research_run_artifact(
result,
run_id="run-1",
strategy_id="alpha-top1",
strategy_name="Alpha Top 1",
strategy_version="1.0.0",
engine_version="1.2.0",
code_revision="3b1ad07",
data_snapshot_id="snapshot-1",
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-01-08T10:00:00+08:00",
finished_at="2026-01-08T10:01:00+08:00",
parameters={},
benchmark_id="000300.SH",
)
+118
View File
@@ -0,0 +1,118 @@
"""Post-execution return attribution contracts."""
from __future__ import annotations
import pandas as pd
import pytest
from quant_engine.attribution import DailyReturnAttribution
from quant_engine.execution import ExecutionConfig
from quant_engine.research_pipeline import run_factor_backtest_research
def _zero_cost_config() -> ExecutionConfig:
return ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
def test_daily_attribution_closes_across_rebalance_and_holding_days() -> None:
"""开盘换仓时,隔夜和日内贡献必须来自实际换仓前后持仓。"""
dates = pd.date_range("2026-01-05", periods=4, freq="B")
scores = pd.DataFrame(
{"A": [2.0, 0.0], "B": [1.0, 3.0]},
index=dates[:2],
)
opens = pd.DataFrame(
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
index=dates,
)
closes = pd.DataFrame(
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
index=dates,
)
result = run_factor_backtest_research(
scores,
opens,
closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=_zero_cost_config(),
)
attribution = result.return_attribution()
assert isinstance(attribution, DailyReturnAttribution)
assert attribution.overnight.loc[dates[2], "A"] == pytest.approx(0.25)
assert attribution.intraday.loc[dates[2], "B"] == pytest.approx(-0.125)
assert attribution.asset_contributions.loc[dates[3], "B"] == pytest.approx(1 / 6)
pd.testing.assert_series_equal(
attribution.total_return,
result.returns.rename("total_return"),
)
pd.testing.assert_series_equal(
attribution.explained_return + attribution.residual,
attribution.total_return,
check_names=False,
)
assert attribution.residual.abs().max() < 1e-12
def test_daily_attribution_reports_execution_cost_separately() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
scores = pd.DataFrame({"A": [1.0]}, index=dates[:1])
prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates)
config = ExecutionConfig(
commission_bps=10,
stamp_tax_bps=0,
slippage_bps=10,
min_trade_amount=0,
)
result = run_factor_backtest_research(
scores,
prices,
prices,
top_k=1,
gross_exposure=0.5,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=config,
)
attribution = result.return_attribution()
execution = result.execution.daily_executions[1].executions[0]
assert attribution.asset_contributions.loc[dates[1], "A"] == 0.0
assert attribution.transaction_cost.loc[dates[1]] == pytest.approx(
-execution.total_cost / 1_000.0
)
assert attribution.total_return.loc[dates[1]] == pytest.approx(
attribution.transaction_cost.loc[dates[1]]
)
assert attribution.residual.loc[dates[1]] == pytest.approx(0.0, abs=1e-12)
def test_return_attribution_is_empty_for_empty_research_result() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
scores = pd.DataFrame(columns=["A"], index=pd.DatetimeIndex([]), dtype=float)
prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates)
result = run_factor_backtest_research(
scores,
prices,
prices,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
)
attribution = result.return_attribution()
assert attribution.overnight.empty
assert attribution.intraday.empty
assert attribution.total_return.empty
+209
View File
@@ -0,0 +1,209 @@
"""Backtest contract tests for weights, NAV, rebalancing, and benchmarks."""
from __future__ import annotations
import pandas as pd
import pytest
from quant_engine.backtest import (
BacktestResult,
compare_to_benchmark,
compute_nav_from_weights,
compute_returns_from_nav,
rebalance_periodic,
run_weight_backtest,
weights_to_long_short,
)
def test_compute_nav_from_weights_forward_fills_rebalance_weights() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
weights = pd.DataFrame({"A": [0.5], "B": [0.5]}, index=dates[:1])
returns = pd.DataFrame({"A": [0.10, 0.00, -0.10], "B": [0.00, 0.10, 0.00]}, index=dates)
nav = compute_nav_from_weights(weights, returns, initial_capital=100.0)
expected = pd.Series([105.0, 110.25, 104.7375], index=dates)
pd.testing.assert_series_equal(nav, expected)
def test_compute_nav_stays_in_cash_before_first_rebalance() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
weights = pd.DataFrame({"A": [1.0]}, index=dates[1:2])
returns = pd.DataFrame({"A": [0.50, 0.10, 0.10]}, index=dates)
nav = compute_nav_from_weights(weights, returns)
pd.testing.assert_series_equal(nav, pd.Series([1.0, 1.1, 1.21], index=dates))
def test_compute_nav_ignores_weight_columns_without_returns() -> None:
dates = pd.date_range("2026-01-05", periods=2, freq="B")
weights = pd.DataFrame({"A": [0.5], "MISSING": [0.5]}, index=dates[:1])
returns = pd.DataFrame({"A": [0.10, 0.10]}, index=dates)
nav = compute_nav_from_weights(weights, returns)
pd.testing.assert_series_equal(nav, pd.Series([1.05, 1.1025], index=dates))
def test_compute_nav_charges_configured_turnover_cost() -> None:
dates = pd.date_range("2026-01-05", periods=2, freq="B")
weights = pd.DataFrame({"A": [1.0]}, index=dates[:1])
returns = pd.DataFrame({"A": [0.0, 0.0]}, index=dates)
nav = compute_nav_from_weights(weights, returns, tc_rate=0.01)
pd.testing.assert_series_equal(nav, pd.Series([0.995, 0.995], index=dates))
def test_compute_returns_from_nav_preserves_index_and_sets_initial_zero() -> None:
nav = pd.Series([100.0, 110.0, 99.0], index=pd.date_range("2026-01-05", periods=3))
result = compute_returns_from_nav(nav)
pd.testing.assert_series_equal(result, pd.Series([0.0, 0.1, -0.1], index=nav.index))
def test_rebalance_periodic_maps_weekend_to_previous_trading_day() -> None:
dates = pd.date_range("2026-01-05", periods=5, freq="B")
target = pd.Series({"A": 0.6, "B": 0.4})
result = rebalance_periodic(target, [pd.Timestamp("2026-01-10")], dates)
assert result.loc[pd.Timestamp("2026-01-08")].sum() == 0.0
pd.testing.assert_series_equal(
result.loc[pd.Timestamp("2026-01-09")], target, check_names=False
)
def test_rebalance_periodic_accepts_empty_trading_calendar() -> None:
target = pd.Series({"A": 1.0})
result = rebalance_periodic(
target,
[pd.Timestamp("2026-01-05")],
pd.DatetimeIndex([]),
)
assert result.empty
assert result.columns.tolist() == ["A"]
def test_weights_to_long_short_allocates_each_leg() -> None:
result = weights_to_long_short(["A", "B"], ["C"], long_weight=0.6, short_weight=0.4)
assert result["A"] == pytest.approx(0.3)
assert result["B"] == pytest.approx(0.3)
assert result["C"] == pytest.approx(-0.4)
assert result.sum() == pytest.approx(0.2)
def test_weights_to_long_short_keeps_explicit_universe() -> None:
result = weights_to_long_short(["A"], [], all_tickers=["A", "B"])
pd.testing.assert_series_equal(result, pd.Series({"A": 0.5, "B": 0.0}))
def test_compare_to_benchmark_returns_report_table() -> None:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
strategy = pd.Series([1.0, 1.1, 1.0, 1.2], index=dates)
benchmark = pd.Series([1.0, 1.0, 1.05, 1.1], index=dates)
result = compare_to_benchmark(strategy, benchmark)
assert result.columns.tolist() == ["策略", "基准"]
assert result.loc["n_days", "策略"] == 4
assert result.loc["累计收益", "策略"] == pytest.approx(0.2)
assert result.loc["累计收益", "基准"] == pytest.approx(0.1)
def test_compare_to_benchmark_rejects_non_overlapping_dates() -> None:
strategy = pd.Series([1.0], index=[pd.Timestamp("2026-01-05")])
benchmark = pd.Series([1.0], index=[pd.Timestamp("2026-02-05")])
with pytest.raises(ValueError, match="overlapping dates"):
compare_to_benchmark(strategy, benchmark)
# ── 统一回测结果门面 ──────────────────────────────────────
def test_run_weight_backtest_returns_nav_returns_and_input_snapshot() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
weights = pd.DataFrame({"A": [1.0]}, index=dates[:1])
stock_returns = pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates)
result = run_weight_backtest(weights, stock_returns, initial_capital=100.0)
assert isinstance(result, BacktestResult)
pd.testing.assert_series_equal(
result.nav,
pd.Series([110.0, 99.0, 118.8], index=dates),
)
pd.testing.assert_series_equal(
result.returns,
pd.Series([0.0, -0.1, 0.2], index=dates),
)
pd.testing.assert_frame_equal(result.weights, weights)
def test_backtest_result_stats_reuses_standard_metrics_contract() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
result = run_weight_backtest(
pd.DataFrame({"A": [1.0]}, index=dates[:1]),
pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates),
)
stats = result.stats(rf=0.02)
assert stats["n_days"] == 3
assert stats["ann_return"] == pytest.approx(
(1.0 * 0.9 * 1.2) ** (252 / 3) - 1.0
)
assert "sharpe" in stats
assert stats["drawback"] == stats["max_drawdown"]
def test_backtest_result_builds_benchmark_report() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
benchmark = pd.Series([1.0, 1.05, 1.10], index=dates, name="benchmark")
result = run_weight_backtest(
pd.DataFrame({"A": [1.0]}, index=dates[:1]),
pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates),
benchmark_nav=benchmark,
)
report = result.benchmark_report()
assert report.columns.tolist() == ["策略", "基准"]
assert report.loc["累计收益", "基准"] == pytest.approx(0.10)
def test_backtest_result_requires_benchmark_for_comparison() -> None:
dates = pd.date_range("2026-01-05", periods=2, freq="B")
result = run_weight_backtest(
pd.DataFrame({"A": [1.0]}, index=dates[:1]),
pd.DataFrame({"A": [0.0, 0.0]}, index=dates),
)
with pytest.raises(ValueError, match="benchmark_nav"):
result.benchmark_report()
def test_backtest_result_isolated_from_mutated_caller_inputs() -> None:
dates = pd.date_range("2026-01-05", periods=2, freq="B")
weights = pd.DataFrame({"A": [1.0]}, index=dates[:1])
benchmark = pd.Series([1.0, 1.1], index=dates)
result = run_weight_backtest(
weights,
pd.DataFrame({"A": [0.0, 0.0]}, index=dates),
benchmark_nav=benchmark,
)
weights.iloc[0, 0] = 0.0
benchmark.iloc[1] = 99.0
assert result.weights.iloc[0, 0] == 1.0
assert result.benchmark_nav is not None
assert result.benchmark_nav.iloc[1] == 1.1
+773
View File
@@ -0,0 +1,773 @@
"""Backtest run-reference and closed-evidence contract conformance."""
from __future__ import annotations
import copy
import hashlib
import json
from dataclasses import replace
from datetime import UTC, datetime
from pathlib import Path
from typing import Any
import pandas as pd
import pytest
from quant_engine.artifact import (
BacktestEvidenceManifest,
EvidenceQualification,
RESEARCH_ARTIFACT_SCHEMA_VERSION,
ResearchRunArtifact,
build_backtest_evidence_manifest,
build_legacy_backtest_evidence_manifest,
build_research_run_artifact,
)
from quant_engine.execution import ExecutionConfig
from quant_engine.factor_contracts import (
ActorIdentity,
AvailabilityMode,
Causation,
DataFoundationEnvelope,
DatasetSnapshotEnvelope,
FactorInput,
FactorSetRef,
InputBinding,
OutputArtifactRef,
OutputCoverage,
OutputQuality,
OutputQualityCheck,
ProducerIdentity,
ViewAvailability,
canonical_json_bytes,
factor_definition_from_alpha158,
factor_input_schema_digest,
)
from quant_engine.governed_pipeline import (
BacktestContractError,
BacktestContractErrorCode,
BacktestRun,
BacktestRunRef,
)
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
ROOT = Path(__file__).resolve().parents[1]
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
BACKTEST_FIXTURE = ROOT / "tests" / "fixtures" / "backtest-evidence-v1.golden.json"
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
def _sha256(value: bytes) -> str:
return f"sha256:{hashlib.sha256(value).hexdigest()}"
def _accepted_authorities(
*,
factor_evaluation_at: str = "2026-01-03T11:00:00Z",
factor_computed_at: str = "2026-01-03T10:15:00Z",
factor_artifact_available_at: str = "2026-01-03T10:20:00Z",
factor_availability_mode: AvailabilityMode = AvailabilityMode.AS_AVAILABLE,
) -> tuple[
DatasetSnapshotEnvelope,
DataFoundationEnvelope,
FactorSetRef,
]:
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
definition = factor_definition_from_alpha158(
"alpha_005",
version="1.0.0",
parameters={},
inputs=(factor_input,),
implementation_digest="sha256:" + "1" * 64,
input_schema_digest=factor_input_schema_digest((factor_input,)),
valid_from="2026-01-01T00:00:00.000000Z",
valid_until="2027-01-01T00:00:00Z",
warmup_sessions=10,
lag_sessions=1,
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
)
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
output_content_bytes = canonical_json_bytes(fixture["output_content"])
artifact_ref = OutputArtifactRef.create(
schema_digest=_sha256(output_schema_bytes),
content_digest=_sha256(output_content_bytes),
)
factor_set = FactorSetRef.create(
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
selected_view_ref_ids=(VIEW_REF_ID,),
input_bindings=(
InputBinding(definition.definition_id, "market", VIEW_REF_ID, VIEW_SCHEMA_DIGEST),
),
view_availability=(
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
),
output_quality=OutputQuality(
"passed",
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
),
output_coverage=OutputCoverage(
"complete", 1, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64
),
output_schema_bytes=output_schema_bytes,
output_content_bytes=output_content_bytes,
output_artifact_ref=artifact_ref,
availability_mode=factor_availability_mode,
evaluation_at=factor_evaluation_at,
computed_at=factor_computed_at,
artifact_available_at=factor_artifact_available_at,
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
actor=ActorIdentity("service", "factor_worker_v1"),
correlation_id="research_run_001",
causation=Causation("foundation", foundation.foundation_id),
evidence_scope="synthetic_fixture",
decision_eligible=False,
)
return snapshot, foundation, factor_set
def _config_digest(parameters: dict[str, object] | None = None) -> str:
encoded = json.dumps(
PARAMETERS if parameters is None else parameters,
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
).encode("utf-8")
return _sha256(encoded)
def _run_ref(**overrides: Any) -> BacktestRunRef:
snapshot, foundation, factor_set = _accepted_authorities()
arguments: dict[str, Any] = {
"dataset_snapshot": snapshot,
"foundation": foundation,
"factor_set": factor_set,
"universe_digest": "sha256:" + "5" * 64,
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
"strategy_id": "alpha-top1",
"strategy_version": "1.0.0",
"strategy_digest": "sha256:" + "6" * 64,
"execution_model_version": "1.0.0",
"execution_model_digest": "sha256:" + "7" * 64,
"cost_model_version": "1.0.0",
"cost_model_digest": "sha256:" + "8" * 64,
"random_seed": 7,
"code_revision": "d" * 40,
"environment_lock_digest": "sha256:" + "9" * 64,
"configuration_digest": _config_digest(),
"evaluation_at": "2026-01-08T01:00:00Z",
"computed_at": "2026-01-08T02:00:00Z",
}
arguments.update(overrides)
return BacktestRunRef.create(**arguments)
def _backtest_result() -> FactorBacktestResult:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
opens = pd.DataFrame(
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
index=dates,
)
closes = pd.DataFrame(
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
index=dates,
)
return run_factor_backtest_research(
scores,
opens,
closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
),
)
def _artifact(run_ref: BacktestRunRef, *, run_id: str | None = None) -> ResearchRunArtifact:
result = _backtest_result()
benchmark = pd.Series(
[0.0, 0.01, -0.01, 0.02],
index=result.returns.index,
name="benchmark_return",
)
return build_research_run_artifact(
result,
run_id=run_ref.run_id if run_id is None else run_id,
strategy_id=run_ref.strategy_id,
strategy_name="Alpha Top 1",
strategy_version=run_ref.strategy_version,
engine_version="1.2.0",
code_revision=run_ref.code_revision,
data_snapshot_id=run_ref.dataset_snapshot_id,
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-01-08T10:00:00+08:00",
finished_at="2026-01-08T10:01:00+08:00",
parameters=PARAMETERS,
benchmark_id="000300.SH",
benchmark_returns=benchmark,
)
def _assert_error(
error: pytest.ExceptionInfo[BacktestContractError],
code: BacktestContractErrorCode,
path: str,
) -> None:
assert error.value.code is code
assert error.value.path == path
def test_backtest_run_ref_is_deterministic_and_binds_only_opaque_authorities() -> None:
first = _run_ref()
second = _run_ref()
assert first == second
assert first.run_id.startswith("rhbacktestrunv1:sha256:")
assert first.replay_spec_digest.startswith("sha256:")
assert first.dataset_snapshot_id.startswith("rhdsv1:sha256:")
assert first.foundation_id.startswith("rhdfv1:sha256:")
assert first.factor_set_id.startswith("rhfactorsetv1:sha256:")
assert first.trading_calendar_revision_ids == (CALENDAR_REVISION_ID,)
assert first.corporate_action_revision_ids == (ACTION_REVISION_ID,)
assert first.replay_parent_run_id is None
assert first.replay_attempt == 0
assert first.replay_ancestor_run_ids == ()
snapshot, foundation, factor_set = _accepted_authorities()
assert BacktestRunRef.from_dict(
first.to_dict(),
dataset_snapshot=snapshot,
foundation=foundation,
factor_set=factor_set,
) == first
forbidden = ("latest", "locator", "uri", "credential", "provider", "broker")
assert not any(token in first.to_json().lower() for token in forbidden)
@pytest.mark.parametrize(
("field", "value"),
[
("universe_digest", "sha256:" + "a" * 64),
("strategy_digest", "sha256:" + "b" * 64),
("execution_model_digest", "sha256:" + "c" * 64),
("cost_model_digest", "sha256:" + "e" * 64),
("random_seed", 8),
("code_revision", "e" * 40),
("environment_lock_digest", "sha256:" + "f" * 64),
("configuration_digest", "sha256:" + "0" * 64),
("evaluation_at", "2026-01-08T01:00:01Z"),
("computed_at", "2026-01-08T02:00:01Z"),
],
)
def test_every_governed_run_input_mutation_changes_run_identity(
field: str,
value: object,
) -> None:
assert _run_ref(**{field: value}).run_id != _run_ref().run_id
def test_backtest_run_ref_rejects_unclosed_upstream_and_unsafe_scalars() -> None:
with pytest.raises(BacktestContractError) as wrong_calendar:
_run_ref(trading_calendar_revision_ids=())
_assert_error(
wrong_calendar,
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.trading_calendar_revision_ids",
)
with pytest.raises(BacktestContractError) as wrong_action:
_run_ref(corporate_action_revision_ids=())
_assert_error(
wrong_action,
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.corporate_action_revision_ids",
)
with pytest.raises(BacktestContractError) as bool_seed:
_run_ref(random_seed=True)
_assert_error(bool_seed, BacktestContractErrorCode.TYPE_ERROR, "$.random_seed")
with pytest.raises(BacktestContractError) as bad_revision:
_run_ref(code_revision="abc")
_assert_error(bad_revision, BacktestContractErrorCode.INVALID_FORMAT, "$.code_revision")
with pytest.raises(BacktestContractError) as bad_digest:
_run_ref(universe_digest="5" * 64)
_assert_error(bad_digest, BacktestContractErrorCode.INVALID_FORMAT, "$.universe_digest")
with pytest.raises(BacktestContractError) as lookahead:
_run_ref(computed_at="2026-01-08T00:59:59Z")
_assert_error(lookahead, BacktestContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
with pytest.raises(BacktestContractError) as factor_type:
_run_ref(factor_set="rhfactorsetv1:sha256:" + "0" * 64)
_assert_error(factor_type, BacktestContractErrorCode.TYPE_ERROR, "$.factor_set")
@pytest.mark.parametrize(
("factor_times", "expected_path"),
[
(
{
"factor_evaluation_at": "2026-01-08T01:00:01Z",
"factor_computed_at": "2026-01-08T00:59:59Z",
"factor_artifact_available_at": "2026-01-08T01:00:00Z",
},
"$.evaluation_at",
),
(
{
"factor_evaluation_at": "2026-01-03T11:00:00Z",
"factor_computed_at": "2026-01-08T01:00:00Z",
"factor_artifact_available_at": "2026-01-08T01:00:01Z",
"factor_availability_mode": AvailabilityMode.RETROSPECTIVE_REPLAY,
},
"$.evaluation_at",
),
],
)
def test_run_ref_evaluation_closes_factor_pit(
factor_times: dict[str, Any],
expected_path: str,
) -> None:
snapshot, foundation, factor_set = _accepted_authorities(**factor_times)
with pytest.raises(BacktestContractError) as lookahead:
_run_ref(
dataset_snapshot=snapshot,
foundation=foundation,
factor_set=factor_set,
)
_assert_error(
lookahead,
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
expected_path,
)
def test_run_ref_rejects_aliases_locators_unsafe_integers_and_invalid_text() -> None:
with pytest.raises(BacktestContractError) as mutable_alias:
_run_ref(strategy_id="latest")
_assert_error(
mutable_alias,
BacktestContractErrorCode.INVALID_FORMAT,
"$.strategy_id",
)
with pytest.raises(BacktestContractError) as physical_uri:
_run_ref(execution_model_version="s3://model-bucket/current")
_assert_error(
physical_uri,
BacktestContractErrorCode.INVALID_FORMAT,
"$.execution_model_version",
)
with pytest.raises(BacktestContractError) as unsafe_seed:
_run_ref(random_seed=2**53)
_assert_error(
unsafe_seed,
BacktestContractErrorCode.INVALID_VALUE,
"$.random_seed",
)
with pytest.raises(BacktestContractError) as invalid_unicode:
_run_ref(strategy_id="\ud800")
_assert_error(
invalid_unicode,
BacktestContractErrorCode.INVALID_FORMAT,
"$.strategy_id",
)
run_ref = _run_ref()
mixed_keys = run_ref.to_dict()
mixed_keys[1] = "not-a-contract-key" # type: ignore[index]
snapshot, foundation, factor_set = _accepted_authorities()
with pytest.raises(BacktestContractError) as invalid_key:
BacktestRunRef.from_dict(
mixed_keys,
dataset_snapshot=snapshot,
foundation=foundation,
factor_set=factor_set,
)
_assert_error(invalid_key, BacktestContractErrorCode.TYPE_ERROR, "$")
@pytest.mark.parametrize(
"physical_id",
["db.table", "source_alpha", "wind.model", "qtdb_view", "bloomberg-signal"],
)
def test_run_ref_rejects_physical_terms_in_logical_ids(physical_id: str) -> None:
with pytest.raises(BacktestContractError) as physical:
_run_ref(strategy_id=physical_id)
_assert_error(
physical,
BacktestContractErrorCode.INVALID_FORMAT,
"$.strategy_id",
)
@pytest.mark.parametrize("version", ["1.0.0-.", "1.0.0-foo..bar", "1.0.0-01"])
def test_run_ref_requires_strict_semver_prerelease_identifiers(version: str) -> None:
with pytest.raises(BacktestContractError) as invalid:
_run_ref(strategy_version=version)
_assert_error(
invalid,
BacktestContractErrorCode.INVALID_FORMAT,
"$.strategy_version",
)
assert _run_ref(strategy_version="1.0.0-alpha.1").strategy_version == "1.0.0-alpha.1"
def test_replay_lineage_is_acyclic_and_cannot_claim_changed_inputs() -> None:
parent = _run_ref()
replay = _run_ref(
computed_at="2026-01-08T03:00:00Z",
parent=parent,
replay_reason="deterministic_reproduction",
replay_attempt=1,
)
assert replay.run_id != parent.run_id
assert replay.replay_spec_digest == parent.replay_spec_digest
assert replay.replay_parent_run_id == parent.run_id
assert replay.replay_ancestor_run_ids == (parent.run_id,)
with pytest.raises(BacktestContractError) as changed_input:
_run_ref(
universe_digest="sha256:" + "a" * 64,
computed_at="2026-01-08T03:00:00Z",
parent=parent,
replay_reason="changed_universe",
replay_attempt=1,
)
_assert_error(
changed_input,
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_spec_digest",
)
with pytest.raises(BacktestContractError) as skipped_attempt:
_run_ref(
computed_at="2026-01-08T03:00:00Z",
parent=parent,
replay_reason="skipped_attempt",
replay_attempt=2,
)
_assert_error(
skipped_attempt,
BacktestContractErrorCode.LINEAGE_VIOLATION,
"$.replay_attempt",
)
def test_offline_research_manifest_closes_exact_existing_evidence_mapping() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
first = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
)
second = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
)
assert first == second
assert first.manifest_id.startswith("rhbacktestevidencev1:sha256:")
assert first.run_id == run_ref.run_id
assert first.profile == "offline_research_v1"
assert first.qualification is EvidenceQualification.CONTRACT_QUALIFIED
mapping = {
item.category: tuple(table.logical_name for table in item.tables)
for item in first.evidence
}
assert mapping == {
"run": ("run",),
"signal": ("signals",),
"fill": ("trades",),
"position_nav": ("positions", "nav"),
"performance": ("performance",),
"attribution": ("attribution", "attribution_daily"),
"risk_snapshot": ("risk",),
"replay": (),
}
assert "order" not in mapping
assert "rejection" not in mapping
risk = next(item for item in first.evidence if item.category == "risk_snapshot")
assert risk.tables[0].row_count == 0
assert risk.tables[0].schema_digest.startswith("sha256:")
changed_performance = artifact.performance
changed_performance.loc[0, "n_days"] += 1
changed_artifact = replace(artifact, _performance=changed_performance)
changed = build_backtest_evidence_manifest(
run_ref,
changed_artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
assert changed.manifest_id != first.manifest_id
assert run_ref.run_id == first.run_id == changed.run_id
def test_manifest_rejects_missing_mismatched_or_duplicate_evidence() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
with pytest.raises(BacktestContractError) as wrong_run:
build_backtest_evidence_manifest(
run_ref,
_artifact(run_ref, run_id="different-run"),
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
wrong_run,
BacktestContractErrorCode.IDENTITY_MISMATCH,
"$.artifact.tables.run.run_id",
)
missing_signals = replace(artifact, _signals=None) # type: ignore[arg-type]
with pytest.raises(BacktestContractError) as missing_table:
build_backtest_evidence_manifest(
run_ref,
missing_signals,
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
missing_table,
BacktestContractErrorCode.TYPE_ERROR,
"$.artifact.tables.signals",
)
with pytest.raises(BacktestContractError) as digest_mismatch:
build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
expected_table_digests={"performance": "sha256:" + "0" * 64},
)
_assert_error(
digest_mismatch,
BacktestContractErrorCode.EVIDENCE_MISMATCH,
"$.artifact.tables.performance.content_digest",
)
manifest = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
duplicate = manifest.to_dict()
duplicate["evidence"].append(copy.deepcopy(duplicate["evidence"][0]))
with pytest.raises(BacktestContractError) as duplicate_category:
BacktestEvidenceManifest.from_dict(
duplicate,
backtest_run_ref=run_ref,
artifact=artifact,
)
_assert_error(
duplicate_category,
BacktestContractErrorCode.INVALID_VALUE,
"$.evidence[8].category",
)
def test_manifest_binds_supported_schema_and_has_collision_free_cell_encoding() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
with pytest.raises(BacktestContractError) as unsupported_schema:
build_backtest_evidence_manifest(
run_ref,
replace(artifact, schema_version="999.0.0"),
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
unsupported_schema,
BacktestContractErrorCode.INVALID_VALUE,
"$.artifact.schema_version",
)
identities: set[str] = set()
for value in (float("nan"), float("inf"), float("-inf")):
performance = artifact.performance
performance.loc[0, "alpha"] = value
manifest = build_backtest_evidence_manifest(
run_ref,
replace(artifact, _performance=performance),
artifact_available_at="2026-01-08T02:05:00Z",
)
assert manifest.artifact_schema_version == RESEARCH_ARTIFACT_SCHEMA_VERSION
identities.add(manifest.manifest_id)
assert len(identities) == 3
content_digests: set[str] = set()
for value in (float("nan"), {"non_finite_float": "nan"}):
performance = artifact.performance.astype(object)
performance.at[0, "alpha"] = value
manifest = build_backtest_evidence_manifest(
run_ref,
replace(artifact, _performance=performance),
artifact_available_at="2026-01-08T02:05:00Z",
)
performance_entry = next(
entry for entry in manifest.evidence if entry.category == "performance"
)
content_digests.add(performance_entry.tables[0].content_digest)
assert len(content_digests) == 2
unsupported = artifact.performance.astype(object)
unsupported.loc[0, "alpha"] = object()
with pytest.raises(BacktestContractError) as unsupported_cell:
build_backtest_evidence_manifest(
run_ref,
replace(artifact, _performance=unsupported),
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
unsupported_cell,
BacktestContractErrorCode.TYPE_ERROR,
"$.artifact.tables.performance.rows[0].alpha",
)
invalid_nested_key = artifact.performance.astype(object)
invalid_nested_key.at[0, "alpha"] = {"\ud800": "value"}
with pytest.raises(BacktestContractError) as invalid_utf8:
build_backtest_evidence_manifest(
run_ref,
replace(artifact, _performance=invalid_nested_key),
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
invalid_utf8,
BacktestContractErrorCode.INVALID_FORMAT,
"$.artifact.tables.performance.rows[0].alpha.keys",
)
unsafe_integer = artifact.performance.astype(object)
unsafe_integer.loc[0, "alpha"] = 10**5000
with pytest.raises(BacktestContractError) as unsafe_cell:
build_backtest_evidence_manifest(
run_ref,
replace(artifact, _performance=unsafe_integer),
artifact_available_at="2026-01-08T02:05:00Z",
)
_assert_error(
unsafe_cell,
BacktestContractErrorCode.INVALID_VALUE,
"$.artifact.tables.performance.rows[0].alpha",
)
def test_artifact_canonical_content_has_typed_collision_free_cell_encoding() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
content_hashes: set[str] = set()
for value in (
float("nan"),
float("inf"),
float("-inf"),
{"non_finite_float": "nan"},
):
performance = artifact.performance.astype(object)
performance.at[0, "alpha"] = value
mutated = replace(artifact, _performance=performance)
content_hashes.add(mutated.content_sha256)
assert "non_finite_float" in mutated.canonical_json()
assert len(content_hashes) == 4
unsupported = artifact.performance.astype(object)
unsupported.loc[0, "alpha"] = object()
with pytest.raises(BacktestContractError) as unsupported_cell:
replace(artifact, _performance=unsupported).canonical_json()
_assert_error(
unsupported_cell,
BacktestContractErrorCode.TYPE_ERROR,
"$.tables.performance.rows[0].alpha",
)
def test_legacy_bridge_is_explicit_and_cannot_be_contract_qualified() -> None:
run_ref = _run_ref()
legacy_run = BacktestRun(
run_id="legacy-run-001",
dataset_snapshot_id=run_ref.dataset_snapshot_id,
factor_version_id="alpha_005@1.0.0",
strategy_version_id="alpha-top1@1.0.0",
code_revision=run_ref.code_revision,
config_hash=_config_digest().removeprefix("sha256:"),
created_at=datetime(2026, 1, 8, 2, 0, tzinfo=UTC),
)
artifact = _artifact(run_ref, run_id=legacy_run.run_id)
manifest = build_legacy_backtest_evidence_manifest(
legacy_run,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
assert manifest.qualification is EvidenceQualification.LEGACY_EXPLORATORY
assert manifest.run_id == legacy_run.run_id
assert manifest.backtest_run_ref is None
assert manifest.to_dict()["run_reference"]["kind"] == "legacy_backtest_run"
assert BacktestEvidenceManifest.from_dict(
manifest.to_dict(),
artifact=artifact,
) == manifest
with pytest.raises(BacktestContractError) as implicit_promotion:
build_backtest_evidence_manifest( # type: ignore[arg-type]
legacy_run,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
)
_assert_error(
implicit_promotion,
BacktestContractErrorCode.TYPE_ERROR,
"$.backtest_run_ref",
)
def test_golden_contract_and_architecture_boundary() -> None:
run_ref = _run_ref()
artifact = _artifact(run_ref)
manifest = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
golden = json.loads(BACKTEST_FIXTURE.read_text(encoding="utf-8"))
table_digests = {
table.logical_name: table.content_digest
for item in manifest.evidence
for table in item.tables
}
assert golden == {
"run_id": run_ref.run_id,
"replay_spec_digest": run_ref.replay_spec_digest,
"manifest_id": manifest.manifest_id,
"evidence_digest": manifest.evidence_digest,
"table_content_digests": table_digests,
}
governed_source = (ROOT / "src" / "quant_engine" / "governed_pipeline.py").read_text(
encoding="utf-8"
)
artifact_source = (ROOT / "src" / "quant_engine" / "artifact.py").read_text(
encoding="utf-8"
)
assert "from quant_engine.artifact" not in governed_source
assert "BacktestRunRef" in governed_source
assert "BacktestEvidenceManifest" not in governed_source
assert "BacktestEvidenceManifest" in artifact_source
assert not (ROOT / "src" / "quant_engine" / "backtest_contracts.py").exists()
+184
View File
@@ -7,10 +7,12 @@ import pandas as pd
import pytest
from quant_engine.data_adapter import (
AssetReturnSnapshot,
add_vwap_proxy,
apply_adj_factor,
load_qtdb_daily,
long_to_wide,
prepare_asset_return_snapshot,
prepare_execution_inputs,
prepare_stock_series,
rename_tushare_columns,
@@ -266,6 +268,17 @@ def test_prepare_execution_inputs_basic(tushare_long: pd.DataFrame) -> None:
assert volumes.iloc[0, 0] == pytest.approx(1000.0)
def test_prepare_execution_inputs_can_select_next_session_open_price(
tushare_long: pd.DataFrame,
) -> None:
"""显式 price_col=open 时应生成开盘执行价矩阵。"""
renamed = rename_tushare_columns(tushare_long)
prices, _volumes = prepare_execution_inputs(renamed, price_col="open")
assert prices.iloc[0, 0] == pytest.approx(10.0)
def test_prepare_execution_inputs_no_volume() -> None:
"""无 volume 列 → volumes 全 1.0。"""
df = pd.DataFrame(
@@ -286,6 +299,177 @@ def test_prepare_execution_inputs_missing_close_raises() -> None:
prepare_execution_inputs(df)
def test_prepare_execution_inputs_missing_selected_price_raises() -> None:
df = pd.DataFrame({"stock_code": ["A"], "trade_date": ["2024-01-01"], "close": [10.0]})
with pytest.raises(ValueError, match="缺 open"):
prepare_execution_inputs(df, price_col="open")
# ── prepare_asset_return_snapshot ────────────────────────────
def _daily_prices() -> pd.DataFrame:
return pd.DataFrame(
{
"stock_code": ["B", "A", "B", "A", "B", "A"],
"trade_date": [
"2024-01-02",
"2024-01-01",
"2024-01-01",
"2024-01-03",
"2024-01-03",
"2024-01-02",
],
"close": [18.0, 10.0, 20.0, 12.1, 19.8, 11.0],
}
)
def test_prepare_asset_return_snapshot_is_stable_and_immutable_by_interface() -> None:
snapshot = prepare_asset_return_snapshot(
_daily_prices(),
source="qtdb_pro.hq_daily",
source_snapshot_id="hq-daily:2024-01-03:v1",
adjustment="qfq",
)
assert isinstance(snapshot, AssetReturnSnapshot)
assert snapshot.data_snapshot_id.startswith("asset-returns-v1:")
assert snapshot.source == "qtdb_pro.hq_daily"
assert snapshot.source_snapshot_id == "hq-daily:2024-01-03:v1"
assert snapshot.price_field == "close"
assert snapshot.adjustment == "qfq"
assert snapshot.return_method == "simple"
assert snapshot.start_date.isoformat() == "2024-01-01"
assert snapshot.end_date.isoformat() == "2024-01-03"
assert snapshot.sessions == 3
assert snapshot.assets == ("A", "B")
expected = pd.DataFrame(
{
"A": [np.nan, 0.1, 0.1],
"B": [np.nan, -0.1, 0.1],
},
index=pd.to_datetime(["2024-01-01", "2024-01-02", "2024-01-03"]),
)
expected.index.name = "trade_date"
expected.columns.name = "stock_code"
pd.testing.assert_frame_equal(snapshot.returns, expected)
exposed = snapshot.returns
exposed.iloc[1, 0] = 999.0
assert snapshot.returns.iloc[1, 0] == pytest.approx(0.1)
def test_asset_return_snapshot_identity_is_order_independent_and_content_addressed() -> None:
kwargs = {
"source": "qtdb_pro.hq_daily",
"source_snapshot_id": "hq-daily:2024-01-03:v1",
"adjustment": "none",
}
baseline = prepare_asset_return_snapshot(_daily_prices(), **kwargs)
shuffled = prepare_asset_return_snapshot(
_daily_prices().sample(frac=1.0, random_state=7),
**kwargs,
)
changed_prices = _daily_prices().copy()
changed_prices.loc[changed_prices["close"] == 12.1, "close"] = 12.2
changed_content = prepare_asset_return_snapshot(changed_prices, **kwargs)
changed_source = prepare_asset_return_snapshot(
_daily_prices(),
source="qtdb_pro.hq_daily",
source_snapshot_id="hq-daily:2024-01-03:v2",
adjustment="none",
)
assert shuffled.data_snapshot_id == baseline.data_snapshot_id
assert changed_content.data_snapshot_id != baseline.data_snapshot_id
assert changed_source.data_snapshot_id != baseline.data_snapshot_id
def test_prepare_asset_return_snapshot_does_not_fill_missing_prices() -> None:
prices = _daily_prices()
prices.loc[
(prices["stock_code"] == "A") & (prices["trade_date"] == "2024-01-02"),
"close",
] = np.nan
snapshot = prepare_asset_return_snapshot(
prices,
source="qtdb_pro.hq_daily",
source_snapshot_id="hq-daily:missing-middle",
)
assert pd.isna(snapshot.returns.loc[pd.Timestamp("2024-01-02"), "A"])
assert pd.isna(snapshot.returns.loc[pd.Timestamp("2024-01-03"), "A"])
def test_prepare_asset_return_snapshot_rejects_duplicate_sessions() -> None:
duplicate = pd.concat([_daily_prices(), _daily_prices().iloc[[0]]], ignore_index=True)
with pytest.raises(ValueError, match="duplicate"):
prepare_asset_return_snapshot(
duplicate,
source="qtdb_pro.hq_daily",
source_snapshot_id="hq-daily:duplicate",
)
@pytest.mark.parametrize("invalid_price", [0.0, -1.0, np.inf])
def test_prepare_asset_return_snapshot_rejects_invalid_prices(invalid_price: float) -> None:
prices = _daily_prices()
prices.loc[0, "close"] = invalid_price
with pytest.raises(ValueError, match="positive finite"):
prepare_asset_return_snapshot(
prices,
source="qtdb_pro.hq_daily",
source_snapshot_id="hq-daily:invalid-price",
)
@pytest.mark.parametrize(
("source", "source_snapshot_id", "adjustment"),
[
("", "source-1", "none"),
("qtdb_pro.hq_daily", "", "none"),
("qtdb_pro.hq_daily", "source-1", ""),
],
)
def test_prepare_asset_return_snapshot_requires_explicit_identity_semantics(
source: str,
source_snapshot_id: str,
adjustment: str,
) -> None:
with pytest.raises(ValueError, match="must be non-empty"):
prepare_asset_return_snapshot(
_daily_prices(),
source=source,
source_snapshot_id=source_snapshot_id,
adjustment=adjustment,
)
def test_asset_return_snapshot_feeds_reproducible_covariance_lineage() -> None:
from quant_engine.risk import estimate_covariance_snapshot
market_snapshot = prepare_asset_return_snapshot(
_daily_prices(),
source="qtdb_pro.hq_daily",
source_snapshot_id="hq-daily:2024-01-03:v1",
)
covariance_snapshot = estimate_covariance_snapshot(
market_snapshot.returns,
as_of_date=market_snapshot.end_date,
lookback_sessions=3,
min_observations=2,
data_snapshot_id=market_snapshot.data_snapshot_id,
)
assert covariance_snapshot.data_snapshot_id == market_snapshot.data_snapshot_id
assert covariance_snapshot.snapshot_id.startswith("sample-cov-v1:")
# ── 端到端:长表 → 适配 → alpha158 + execution ──────────────
+307 -14
View File
@@ -11,6 +11,7 @@ import pytest
from quant_engine.execution import (
ExecutionConfig,
ExecutionResult,
ExecutionSimulationResult,
apply_bid_ask_spread,
apply_volume_constraint,
check_price_limit,
@@ -19,7 +20,9 @@ from quant_engine.execution import (
compute_realized_pnl,
run_end_to_end_poc,
simulate_execution,
simulate_daily_ledger_with_audit,
simulate_multi_day,
simulate_multi_day_with_audit,
simulate_with_daily_data,
total_costs,
total_turnover,
@@ -327,24 +330,26 @@ def test_simulate_multi_day_length_mismatch_raises():
def test_simulate_multi_day_first_day_value_equals_initial():
"""第一天 portfolio_value = initial_cash(无持仓)。"""
"""零成本下第一天日末 NAV 等于初始资金。"""
signals = [("d1", {"A": 1.0})]
prices = [("d1", {"A": 10.0})]
positions = simulate_multi_day(signals, prices, 1_000_000.0)
# 第一天 NAV = 1_000_000(无持仓),第二天才是调仓后
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
positions = simulate_multi_day(signals, prices, 1_000_000.0, config)
assert positions[0].portfolio_value == 1_000_000.0
assert positions[0].holdings == {"A": 100_000.0}
def test_simulate_multi_day_holdings_evolution():
"""调仓后 holdings 演化。
注意:positions[i] 是第 i 天 rebalance 之前的快照。
所以要看 d2 rebalance 后的 holdings,需要看 positions[2](d3 的快照)。
"""
"""日末快照应反映当天调仓后的 holdings。"""
signals = [
("d1", {"A": 0.5, "B": 0.5}),
("d2", {"A": 1.0, "B": 0.0}), # 全仓 A
("d3", {"A": 1.0, "B": 0.0}), # 第三天的快照才能看到 d2 rebalance 后的 holdings
("d3", {"A": 1.0, "B": 0.0}),
]
prices = [
("d1", {"A": 10.0, "B": 20.0}),
@@ -352,9 +357,288 @@ def test_simulate_multi_day_holdings_evolution():
("d3", {"A": 12.0, "B": 22.0}),
]
positions = simulate_multi_day(signals, prices, 1_000_000.0)
# d3 的 PRE-trade snapshot 应该只有 A(B 在 d2 被平仓)
assert "B" not in positions[2].holdings
assert "A" in positions[2].holdings
assert "B" not in positions[1].holdings
assert "A" in positions[1].holdings
def test_simulate_multi_day_with_audit_rebalances_target_weights_by_delta():
"""相同目标权重不应在每个交易日重复买入。"""
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
targets = [(date, {"A": 1.0}) for date in ("d1", "d2", "d3")]
prices = [(date, {"A": 10.0}) for date in ("d1", "d2", "d3")]
result = simulate_multi_day_with_audit(targets, prices, 1_000.0, config)
assert isinstance(result, ExecutionSimulationResult)
assert [len(day.executions) for day in result.daily_executions] == [1, 0, 0]
assert result.total_turnover == pytest.approx(1_000.0)
assert [position.cash for position in result.positions] == pytest.approx([0.0, 0.0, 0.0])
assert [position.holdings["A"] for position in result.positions] == pytest.approx(
[100.0, 100.0, 100.0]
)
assert [position.portfolio_value for position in result.positions] == pytest.approx(
[1_000.0, 1_000.0, 1_000.0]
)
def test_simulate_multi_day_with_audit_records_costs_without_replay():
"""成交成本与日末 NAV 应来自同一次状态推进。"""
targets = [("d1", {"A": 1.0}), ("d2", {"A": 1.0})]
prices = [("d1", {"A": 10.0}), ("d2", {"A": 10.0})]
result = simulate_multi_day_with_audit(targets, prices, 1_000.0)
first_day = result.daily_executions[0]
assert first_day.nav_before == pytest.approx(1_000.0)
assert first_day.nav_after == pytest.approx(result.positions[0].portfolio_value)
assert result.total_costs == pytest.approx(sum(r.total_cost for r in first_day.executions))
assert result.final_portfolio_value == pytest.approx(1_000.0 - result.total_costs)
assert result.daily_executions[1].executions == ()
def test_simulate_multi_day_with_audit_never_spends_more_cash_than_available():
"""满仓目标应按可用现金部分成交,不能用负现金隐式加杠杆。"""
result = simulate_multi_day_with_audit(
[("d1", {"A": 1.0})],
[("d1", {"A": 10.0})],
1_000.0,
)
execution = result.daily_executions[0].executions[0]
assert result.positions[0].cash >= -1e-9
assert 0 < execution.partial_fill_pct < 1
assert execution.blocked_reason == "insufficient_cash_partial_fill"
assert result.final_portfolio_value == pytest.approx(1_000.0 - result.total_costs)
@pytest.mark.parametrize(
"targets",
[
{"A": -0.1},
{"A": 0.6, "B": 0.5},
{"A": float("nan")},
],
)
def test_simulate_multi_day_with_audit_rejects_invalid_long_only_weights(targets):
"""多日 A 股目标必须是有限、非负且合计不超过 100% 的权重。"""
with pytest.raises(ValueError, match="target weights"):
simulate_multi_day_with_audit(
[("d1", targets)],
[("d1", {"A": 10.0, "B": 10.0})],
1_000.0,
)
def test_simulate_multi_day_with_audit_requires_price_for_existing_holding():
"""已有持仓缺价时无法可信估值,必须失败而不是把市值记为零。"""
with pytest.raises(ValueError, match="missing price for held asset A"):
simulate_multi_day_with_audit(
[("d1", {"A": 1.0}), ("d2", {"A": 1.0})],
[("d1", {"A": 10.0}), ("d2", {})],
1_000.0,
)
def test_simulate_multi_day_with_audit_records_unpriced_target_rejection():
"""缺失价格的目标不能吞掉现金,且必须留下拒绝原因。"""
result = simulate_multi_day_with_audit(
[("d1", {"A": 1.0})],
[("d1", {"B": 10.0})],
1_000.0,
)
rejection = result.daily_executions[0].executions[0]
assert rejection.stock_code == "A"
assert rejection.executed_value == 0.0
assert rejection.partial_fill_pct == 0.0
assert rejection.blocked_reason == "missing_price"
assert result.positions[0].cash == 1_000.0
assert result.positions[0].holdings == {}
def test_simulate_multi_day_with_audit_requires_matching_dates():
"""权重与价格日期错位必须显式失败,不能按位置静默配对。"""
with pytest.raises(ValueError, match="dates must match"):
simulate_multi_day_with_audit(
[("d1", {"A": 1.0})],
[("d2", {"A": 10.0})],
1_000.0,
)
# ── 逐交易日 Ledger:成交时点与估值时点分离 ─────────────────
def test_daily_ledger_marks_every_session_after_sparse_open_execution() -> None:
"""下一日开盘成交后,应按每日收盘价持续盯市,而非只记录调仓日。"""
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
result = simulate_daily_ledger_with_audit(
target_weights_history=[("d1", {"A": 1.0})],
execution_price_history=[("d1", {"A": 10.0})],
valuation_price_history=[
("d0", {"A": 9.0}),
("d1", {"A": 11.0}),
("d2", {"A": 12.0}),
],
initial_cash=1_000.0,
config=config,
)
assert [position.date for position in result.positions] == ["d0", "d1", "d2"]
assert [position.portfolio_value for position in result.positions] == pytest.approx(
[1_000.0, 1_100.0, 1_200.0]
)
assert [len(day.executions) for day in result.daily_executions] == [0, 1, 0]
fill = result.daily_executions[1].executions[0]
assert fill.side == "buy"
assert fill.quantity == pytest.approx(100.0)
assert fill.price == pytest.approx(10.0)
pd.testing.assert_series_equal(
result.normalized_nav_series,
pd.Series([1.0, 1.1, 1.2], index=["d0", "d1", "d2"], dtype=float),
)
pd.testing.assert_series_equal(
result.daily_returns,
pd.Series([0.0, 0.1, 1.2 / 1.1 - 1.0], index=["d0", "d1", "d2"]),
)
def test_daily_ledger_first_session_cost_reduces_first_return() -> None:
"""首个估值日发生交易时,费用必须进入相对初始资金的首日收益。"""
result = simulate_daily_ledger_with_audit(
target_weights_history=[("d0", {"A": 1.0})],
execution_price_history=[("d0", {"A": 10.0})],
valuation_price_history=[("d0", {"A": 10.0})],
initial_cash=1_000.0,
)
assert result.total_costs > 0
assert result.daily_returns.iloc[0] == pytest.approx(
result.final_portfolio_value / result.initial_cash - 1.0
)
assert result.daily_returns.iloc[0] < 0
def test_daily_ledger_nav_is_rebuildable_and_trades_are_projectable() -> None:
"""Ledger 必须同时支持现金守恒校验和平台成交表投影。"""
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
result = simulate_daily_ledger_with_audit(
target_weights_history=[
("d1", {"A": 1.0, "B": 0.0}),
("d2", {"A": 0.0, "B": 1.0}),
],
execution_price_history=[
("d1", {"A": 10.0, "B": 20.0}),
("d2", {"A": 11.0, "B": 22.0}),
],
valuation_price_history=[
("d0", {"A": 9.0, "B": 19.0}),
("d1", {"A": 10.5, "B": 21.0}),
("d2", {"A": 12.0, "B": 24.0}),
],
initial_cash=1_000.0,
config=config,
)
close_prices = {
"d0": {"A": 9.0, "B": 19.0},
"d1": {"A": 10.5, "B": 21.0},
"d2": {"A": 12.0, "B": 24.0},
}
for position in result.positions:
rebuilt = position.cash + sum(
shares * close_prices[position.date][asset]
for asset, shares in position.holdings.items()
)
assert position.portfolio_value == pytest.approx(rebuilt)
trades = result.trades_frame
assert trades.columns.tolist() == [
"trade_date",
"ts_code",
"side",
"qty",
"price",
"amount",
"fee",
"slippage",
]
assert trades["side"].tolist() == ["buy", "sell", "buy"]
assert (trades["qty"] > 0).all()
def test_daily_ledger_frame_matches_platform_projection_contract() -> None:
"""核心层输出稳定日频投影,但不携带 run_id 或执行数据库写入。"""
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
result = simulate_daily_ledger_with_audit(
target_weights_history=[("d1", {"A": 1.0})],
execution_price_history=[("d1", {"A": 10.0})],
valuation_price_history=[
("d0", {"A": 9.0}),
("d1", {"A": 11.0}),
("d2", {"A": 12.0}),
],
initial_cash=1_000.0,
config=config,
)
ledger = result.ledger_frame
assert ledger.columns.tolist() == [
"trade_date",
"portfolio_value",
"nav",
"pnl",
"pnl_pct",
"position_value",
"cash",
"turnover",
]
assert ledger["trade_date"].tolist() == ["d0", "d1", "d2"]
assert ledger["nav"].tolist() == pytest.approx([1.0, 1.1, 1.2])
assert ledger["pnl"].tolist() == pytest.approx([0.0, 100.0, 100.0])
assert ledger["pnl_pct"].tolist() == pytest.approx([0.0, 0.1, 1.2 / 1.1 - 1.0])
assert ledger["position_value"].tolist() == pytest.approx([0.0, 1_100.0, 1_200.0])
assert ledger["cash"].tolist() == pytest.approx([1_000.0, 0.0, 0.0])
assert ledger["turnover"].tolist() == pytest.approx([0.0, 1.0, 0.0])
def test_daily_ledger_rejects_missing_close_for_held_asset() -> None:
"""已有持仓缺少收盘估值价时必须 fail closed。"""
with pytest.raises(ValueError, match="missing valuation price for held asset A"):
simulate_daily_ledger_with_audit(
target_weights_history=[("d0", {"A": 1.0})],
execution_price_history=[("d0", {"A": 10.0})],
valuation_price_history=[("d0", {"A": 10.0}), ("d1", {})],
initial_cash=1_000.0,
)
def test_daily_ledger_requires_positive_initial_cash() -> None:
"""可信收益曲线需要正初始资金作为归一化基准。"""
with pytest.raises(ValueError, match="initial_cash must be positive"):
simulate_daily_ledger_with_audit([], [], [], initial_cash=0.0)
# ── v1.2.0 Phase 1:端到端 POC(run_end_to_end_poc) ─────
@@ -449,6 +733,15 @@ def test_run_end_to_end_poc_costs_recorded():
result = run_end_to_end_poc(signals, prices, 1_000_000.0)
assert result["total_costs"] > 0
assert result["total_turnover"] > 0
executions = [
execution
for daily in result["daily_executions"]
for execution in daily.executions
]
assert result["total_costs"] == pytest.approx(sum(item.total_cost for item in executions))
assert result["total_turnover"] == pytest.approx(
sum(item.executed_value for item in executions)
)
# ── v1.2.0 Phase 2: T+1 / 涨跌停 / 部分成交 / 买卖价差 ─────
@@ -741,8 +1034,8 @@ def test_compute_realized_pnl_sell_realizes():
target_weights_history=targets,
)
pnl_list = compute_realized_pnl(positions)
# 第三天(卖出兑现)应有 realized 正利润(cash 从 -800 → 2M = +2M)
assert pnl_list[2].realized_pnl > 0
# 第二天日末快照已包含当日卖出,现金流入应在当天反映。
assert pnl_list[1].realized_pnl > 0
# ── O3: end-to-end 端到端测试(集成多个函数) ──────────────
+938
View File
@@ -0,0 +1,938 @@
"""Versioned factor-definition and factor-set contract conformance."""
from __future__ import annotations
import copy
import hashlib
import json
from dataclasses import FrozenInstanceError
from pathlib import Path
from typing import Any, Callable
import pytest
from quant_engine.factor_contracts import (
ActorIdentity,
AvailabilityMode,
ContractErrorCode,
Causation,
DataFoundationEnvelope,
DatasetSnapshotEnvelope,
FactorContractError,
FactorDefinition,
FactorInput,
FactorSetRef,
HistoricalAvailability,
InputBinding,
LegacyFactorBinding,
OutputArtifactRef,
OutputCoverage,
OutputQuality,
OutputQualityCheck,
PayloadValidation,
ProducerIdentity,
TypedParameter,
ViewAvailability,
canonical_json_bytes,
factor_definition_from_alpha158,
factor_input_schema_digest,
validate_factor_catalog,
)
from quant_engine.governed_pipeline import (
FactorVersion,
bind_legacy_factor,
project_legacy_factor,
)
FIXTURE_PATH = Path(__file__).parent / "fixtures" / "factor-contracts-v1.golden.json"
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
def _golden() -> dict[str, Any]:
loaded = json.loads(FIXTURE_PATH.read_text(encoding="utf-8"))
assert isinstance(loaded, dict)
return loaded
def _sha256(value: bytes) -> str:
return f"sha256:{hashlib.sha256(value).hexdigest()}"
def _reidentify(item: dict[str, Any], field: str, prefix: str) -> None:
payload = {key: value for key, value in item.items() if key != field}
item[field] = f"{prefix}{hashlib.sha256(canonical_json_bytes(payload)).hexdigest()}"
def _snapshot_and_foundation(
fixture: dict[str, Any] | None = None,
) -> tuple[DatasetSnapshotEnvelope, DataFoundationEnvelope]:
source = _golden() if fixture is None else fixture
return (
DatasetSnapshotEnvelope.from_dict(source["dataset_snapshot"]),
DataFoundationEnvelope.from_dict(source["data_foundation"]),
)
def _definition(
*,
inputs: tuple[FactorInput, ...] | None = None,
**overrides: Any,
) -> FactorDefinition:
factor_inputs = inputs or (
FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume")),
)
arguments: dict[str, Any] = {
"factor_id": "alpha_005",
"version": "1.0.0",
"formula": "correlation(close, volume, 10)",
"parameters": {},
"implementation_digest": "sha256:" + "1" * 64,
"input_schema_digest": factor_input_schema_digest(factor_inputs),
"inputs": factor_inputs,
"valid_from": "2026-01-01T00:00:00.000000Z",
"valid_until": "2027-01-01T00:00:00Z",
"warmup_sessions": 10,
"lag_sessions": 1,
"producer": ProducerIdentity("quant_engine", "1.0.0"),
"code_revision": "c" * 40,
}
arguments.update(overrides)
return FactorDefinition.create(**arguments)
def _golden_definition() -> FactorDefinition:
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
return factor_definition_from_alpha158(
"alpha_005",
version="1.0.0",
parameters={},
inputs=(factor_input,),
implementation_digest="sha256:" + "1" * 64,
input_schema_digest=factor_input_schema_digest((factor_input,)),
valid_from="2026-01-01T00:00:00.000000Z",
valid_until="2027-01-01T00:00:00Z",
warmup_sessions=10,
lag_sessions=1,
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
)
def _factor_set_arguments(
*,
fixture: dict[str, Any] | None = None,
snapshot: DatasetSnapshotEnvelope | None = None,
foundation: DataFoundationEnvelope | None = None,
definition: FactorDefinition | None = None,
) -> dict[str, Any]:
source = _golden() if fixture is None else fixture
if snapshot is None or foundation is None:
parsed_snapshot, parsed_foundation = _snapshot_and_foundation(source)
snapshot = snapshot or parsed_snapshot
foundation = foundation or parsed_foundation
selected_definition = definition or _golden_definition()
output_schema_bytes = canonical_json_bytes(source["output_schema"])
output_content_bytes = canonical_json_bytes(source["output_content"])
artifact = OutputArtifactRef.create(
schema_digest=_sha256(output_schema_bytes),
content_digest=_sha256(output_content_bytes),
)
return {
"definitions": (selected_definition,),
"dataset_snapshot": snapshot,
"foundation": foundation,
"selected_view_ref_ids": (VIEW_REF_ID,),
"input_bindings": (
InputBinding(
selected_definition.definition_id,
"market",
VIEW_REF_ID,
VIEW_SCHEMA_DIGEST,
),
),
"view_availability": (
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
),
"output_quality": OutputQuality(
"passed",
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
),
"output_coverage": OutputCoverage(
"complete",
1,
1,
"row",
"alpha_005.cn_a",
"sha256:" + "4" * 64,
),
"output_schema_bytes": output_schema_bytes,
"output_content_bytes": output_content_bytes,
"output_artifact_ref": artifact,
"availability_mode": AvailabilityMode.AS_AVAILABLE,
"evaluation_at": "2026-01-03T11:00:00Z",
"computed_at": "2026-01-03T10:15:00Z",
"artifact_available_at": "2026-01-03T10:20:00Z",
"producer": ProducerIdentity("quant_engine", "1.0.0"),
"code_revision": "c" * 40,
"actor": ActorIdentity("service", "factor_worker_v1"),
"correlation_id": "research_run_001",
"causation": Causation("foundation", foundation.foundation_id),
"evidence_scope": "synthetic_fixture",
"decision_eligible": False,
}
def _factor_set(**overrides: Any) -> FactorSetRef:
arguments = _factor_set_arguments()
arguments.update(overrides)
return FactorSetRef.create(**arguments)
def _assert_error(
error: pytest.ExceptionInfo[FactorContractError],
code: ContractErrorCode,
path: str,
) -> None:
assert error.value.code is code
assert error.value.path == path
def _mutate_artifact_schema_binding(value: dict[str, Any]) -> None:
artifact = value["output_artifact_ref"]
artifact["schema_digest"] = "sha256:" + "0" * 64
_reidentify(artifact, "artifact_id", "rhfactoroutputv1:sha256:")
def test_golden_contracts_are_content_addressed_round_trippable_and_deeply_immutable() -> None:
fixture = _golden()
original_snapshot = copy.deepcopy(fixture["dataset_snapshot"])
original_foundation = copy.deepcopy(fixture["data_foundation"])
snapshot, foundation = _snapshot_and_foundation(fixture)
definition = _golden_definition()
factor_set = FactorSetRef.create(**_factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition))
binding = LegacyFactorBinding.create(
definition=definition,
legacy_factor_id="factor:demo-momentum",
legacy_version="1.0.0",
legacy_definition_sha256="b" * 64,
legacy_dataset_schema_version="1.0.0",
canonical_input_schema_digest=definition.input_schema_digest,
correspondence_evidence_digest="sha256:" + "5" * 64,
)
assert snapshot.pit_cutoff == "2026-01-02T07:01:00Z"
assert foundation.pit_cutoff == factor_set.pit_cutoff == "2026-01-03T00:00:00Z"
assert snapshot.pit_cutoff != foundation.pit_cutoff
assert definition.definition_id == fixture["expected"]["definition_id"]
assert definition.input_schema_digest == fixture["expected"]["input_schema_digest"]
assert factor_set.factor_set_id == fixture["expected"]["factor_set_id"]
assert factor_set.output_artifact_ref.artifact_id == fixture["expected"]["output_artifact_id"]
assert binding.binding_id == fixture["expected"]["legacy_binding_id"]
assert not definition.to_json().endswith("\n")
assert not factor_set.to_json().endswith("\n")
assert FactorDefinition.from_json(definition.to_json()) == definition
reparsed = FactorSetRef.from_json(
factor_set.to_json(),
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
output_schema_bytes=canonical_json_bytes(fixture["output_schema"]),
output_content_bytes=canonical_json_bytes(fixture["output_content"]),
)
reference_only = FactorSetRef.from_json(
factor_set.to_json(),
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
)
assert reparsed.factor_set_id == factor_set.factor_set_id
assert reparsed.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
assert reference_only.payload_validation is PayloadValidation.REFERENCE_ONLY
fixture["dataset_snapshot"]["descriptor"]["dataset"]["dimensions"].append("forbidden")
fixture["data_foundation"]["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
assert snapshot.to_dict() == original_snapshot
assert foundation.to_dict() == original_foundation
returned = snapshot.to_dict()
returned["descriptor"]["dataset"]["dimensions"].append("also_forbidden")
assert snapshot.to_dict() == original_snapshot
with pytest.raises(FrozenInstanceError):
snapshot.snapshot_id = "rhdsv1:sha256:" + "0" * 64 # type: ignore[misc]
@pytest.mark.parametrize("variant", ["whitespace", "key_order"])
def test_contract_decoders_reject_non_canonical_json(variant: str) -> None:
fixture = _golden()
snapshot, foundation = _snapshot_and_foundation(fixture)
definition = _golden_definition()
factor_set = FactorSetRef.create(
**_factor_set_arguments(
fixture=fixture,
snapshot=snapshot,
foundation=foundation,
definition=definition,
)
)
binding = LegacyFactorBinding.create(
definition=definition,
legacy_factor_id="factor:demo-momentum",
legacy_version="1.0.0",
legacy_definition_sha256="b" * 64,
legacy_dataset_schema_version="1.0.0",
canonical_input_schema_digest=definition.input_schema_digest,
correspondence_evidence_digest="sha256:" + "5" * 64,
)
def non_canonical(value: str) -> str:
if variant == "whitespace":
return value + "\n"
loaded = json.loads(value)
reversed_items = dict(reversed(tuple(loaded.items())))
return json.dumps(reversed_items, ensure_ascii=False, separators=(",", ":"))
decoders = (
lambda value: FactorDefinition.from_json(value),
lambda value: FactorSetRef.from_json(
value,
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
),
lambda value: LegacyFactorBinding.from_json(value, definition=definition),
)
for decoder, encoded in zip(
decoders,
(definition.to_json(), factor_set.to_json(), binding.to_json()),
strict=True,
):
with pytest.raises(FactorContractError) as exc_info:
decoder(non_canonical(encoded))
assert exc_info.value.code is ContractErrorCode.INVALID_FORMAT
assert exc_info.value.path == "$"
def test_factor_definition_identity_is_order_independent_where_semantics_are_unordered() -> None:
first_input = FactorInput("prices", "sha256:" + "6" * 64, ("close",))
second_input = FactorInput("volumes", "sha256:" + "7" * 64, ("volume",))
inputs = (first_input, second_input)
parameters_a = {
"window": TypedParameter("integer", 10),
"weights": TypedParameter("json", {"fast": [1, 2], "slow": [3, 4]}),
}
parameters_b = {
"weights": TypedParameter("json", {"slow": [3, 4], "fast": [1, 2]}),
"window": TypedParameter("integer", 10),
}
first = _definition(
inputs=inputs,
parameters=parameters_a,
input_schema_digest=factor_input_schema_digest(inputs),
)
second = _definition(
inputs=tuple(reversed(inputs)),
parameters=parameters_b,
input_schema_digest=factor_input_schema_digest(tuple(reversed(inputs))),
)
assert first.definition_id == second.definition_id
assert first.to_json() == second.to_json()
semantic_changes = (
_definition(factor_id="alpha_006"),
_definition(version="1.0.1"),
_definition(formula="correlation(close, volume, 11)"),
_definition(parameters={"window": TypedParameter("integer", 10)}),
_definition(implementation_digest="sha256:" + "9" * 64),
_definition(valid_until="2027-01-02T00:00:00Z"),
_definition(warmup_sessions=11),
_definition(lag_sessions=2),
_definition(producer=ProducerIdentity("quant_engine", "1.0.1")),
_definition(code_revision="d" * 40),
)
assert all(changed.definition_id != _golden_definition().definition_id for changed in semantic_changes)
assert len({changed.definition_id for changed in semantic_changes}) == len(semantic_changes)
def test_parameter_types_decimal_profile_and_detached_nested_values_are_strict() -> None:
nested = {"ordered": [1, {"flag": True}]}
parameter = TypedParameter("json", nested)
nested["ordered"].append(2)
definition = _definition(parameters={"payload": parameter})
assert definition.to_dict()["parameters"]["payload"]["value"] == {
"ordered": [1, {"flag": True}]
}
integer_definition = _definition(parameters={"value": TypedParameter("integer", 1)})
string_definition = _definition(parameters={"value": TypedParameter("string", "1")})
assert integer_definition.definition_id != string_definition.definition_id
for parameter_type, value, code in (
("decimal", "1.0", ContractErrorCode.INVALID_FORMAT),
("decimal", "1e3", ContractErrorCode.INVALID_FORMAT),
("decimal", "-0", ContractErrorCode.INVALID_FORMAT),
("integer", True, ContractErrorCode.TYPE_ERROR),
("json", 1.5, ContractErrorCode.TYPE_ERROR),
("json", {"é": "bad-key"}, ContractErrorCode.INVALID_FORMAT),
("json", 9_007_199_254_740_992, ContractErrorCode.INVALID_VALUE),
):
with pytest.raises(FactorContractError) as error:
TypedParameter(parameter_type, value)
assert error.value.code is code
assert TypedParameter("decimal", "10.25").to_dict()["value"] == "10.25"
def test_catalog_rejects_duplicate_and_overlapping_logical_validity_but_allows_adjacency() -> None:
base = _golden_definition()
adjacent = _definition(valid_from="2027-01-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
assert len(validate_factor_catalog((adjacent, base))) == 2
with pytest.raises(FactorContractError) as duplicate:
validate_factor_catalog((base, base))
_assert_error(duplicate, ContractErrorCode.INVALID_VALUE, "$.definitions")
overlapping = _definition(valid_from="2026-06-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
with pytest.raises(FactorContractError) as overlap:
validate_factor_catalog((base, overlapping))
_assert_error(overlap, ContractErrorCode.TIME_ORDER_VIOLATION, "$.definitions")
def test_upstream_contracts_reject_unknown_fields_identity_forgery_and_unqualified_input() -> None:
unknown = _golden()["dataset_snapshot"]
unknown["provider"] = "forbidden"
with pytest.raises(FactorContractError) as unknown_error:
DatasetSnapshotEnvelope.from_dict(unknown)
_assert_error(unknown_error, ContractErrorCode.UNKNOWN_FIELD, "$.provider")
forged = _golden()["data_foundation"]
forged["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
with pytest.raises(FactorContractError) as forged_error:
DataFoundationEnvelope.from_dict(forged)
assert forged_error.value.code is ContractErrorCode.IDENTITY_MISMATCH
assert forged_error.value.path.endswith("view_ref_id")
rejected_source = _golden()
rejected_source["dataset_snapshot"]["descriptor"]["qualification"]["status"] = "rejected"
_reidentify(rejected_source["dataset_snapshot"], "snapshot_id", "rhdsv1:sha256:")
rejected_snapshot = DatasetSnapshotEnvelope.from_dict(rejected_source["dataset_snapshot"])
_, foundation = _snapshot_and_foundation()
with pytest.raises(FactorContractError) as rejected_error:
FactorSetRef.create(
**_factor_set_arguments(snapshot=rejected_snapshot, foundation=foundation)
)
_assert_error(
rejected_error,
ContractErrorCode.QUALIFICATION_REJECTED,
"$.dataset_snapshot.descriptor.qualification",
)
def test_foundation_rejects_future_knowledge_and_per_view_calendar_borrowing() -> None:
future = _golden()["data_foundation"]
action = future["corporate_action_revisions"][0]
old_action_id = action["action_revision_id"]
action["knowledge_time"] = "2026-01-03T00:00:01Z"
_reidentify(action, "action_revision_id", "rhcav1:sha256:")
future["standardized_views"][0]["corporate_action_revision_ids"] = [action["action_revision_id"]]
lineage = next(item for item in future["revision_lineage"] if item["revision_id"] == old_action_id)
lineage["revision_id"] = action["action_revision_id"]
lineage["knowledge_time"] = action["knowledge_time"]
_reidentify(future["standardized_views"][0], "view_ref_id", "rhviewrefv1:sha256:")
_reidentify(future, "foundation_id", "rhdfv1:sha256:")
with pytest.raises(FactorContractError) as future_error:
DataFoundationEnvelope.from_dict(future)
_assert_error(
future_error,
ContractErrorCode.TIME_ORDER_VIOLATION,
"$.revision_lineage.knowledge_time",
)
uncovered = _golden()["data_foundation"]
original_route_id = uncovered["instrument_routes"][0]["route_revision_id"]
second_calendar = copy.deepcopy(uncovered["trading_calendar_revisions"][0])
second_calendar["calendar_id"] = "rhcalendar:99990000111122223333444455556666"
_reidentify(second_calendar, "calendar_revision_id", "rhcalv1:sha256:")
uncovered["trading_calendar_revisions"].append(second_calendar)
route = uncovered["instrument_routes"][0]
route["calendar_id"] = second_calendar["calendar_id"]
_reidentify(route, "route_revision_id", "rhroutev1:sha256:")
route_lineage = next(item for item in uncovered["revision_lineage"] if item["revision_id"] == original_route_id)
route_lineage["revision_id"] = route["route_revision_id"]
uncovered["revision_lineage"].append(
{
"revision_kind": "trading_calendar",
"revision_id": second_calendar["calendar_revision_id"],
"revision_number": 1,
"knowledge_time": second_calendar["knowledge_time"],
"evidence_digest": second_calendar["evidence_digest"],
}
)
view = uncovered["standardized_views"][0]
view["instrument_route_revision_ids"] = [route["route_revision_id"]]
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
_reidentify(uncovered, "foundation_id", "rhdfv1:sha256:")
with pytest.raises(FactorContractError) as calendar_error:
DataFoundationEnvelope.from_dict(uncovered)
assert calendar_error.value.code is ContractErrorCode.INPUT_CLOSURE_VIOLATION
assert "selected route calendar" in calendar_error.value.detail
def _replay_fixture() -> dict[str, Any]:
fixture = _golden()
snapshot = fixture["dataset_snapshot"]
snapshot["descriptor"]["published_at"] = "2026-01-04T00:00:00Z"
_reidentify(snapshot, "snapshot_id", "rhdsv1:sha256:")
foundation = fixture["data_foundation"]
foundation["dataset_snapshot_id"] = snapshot["snapshot_id"]
for view in foundation["standardized_views"]:
view["dataset_snapshot_id"] = snapshot["snapshot_id"]
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
return fixture
def test_as_available_and_retrospective_replay_keep_distinct_time_claims() -> None:
as_available = _factor_set()
assert as_available.historical_availability is HistoricalAvailability.DECLARED_AS_AVAILABLE
replay_source = _replay_fixture()
snapshot, foundation = _snapshot_and_foundation(replay_source)
replay_view_id = next(iter(foundation.views))
arguments = _factor_set_arguments(
fixture=replay_source,
snapshot=snapshot,
foundation=foundation,
)
arguments.update(
selected_view_ref_ids=(replay_view_id,),
input_bindings=(
InputBinding(
arguments["definitions"][0].definition_id,
"market",
replay_view_id,
VIEW_SCHEMA_DIGEST,
),
),
view_availability=(
ViewAvailability(replay_view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),
),
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
computed_at="2026-01-04T00:20:00Z",
artifact_available_at="2026-01-04T00:25:00Z",
causation=Causation("foundation", foundation.foundation_id),
)
replay = FactorSetRef.create(**arguments)
assert replay.evaluation_at == "2026-01-03T11:00:00Z"
assert replay.computed_at == "2026-01-04T00:20:00Z"
assert replay.historical_availability is HistoricalAvailability.NOT_ESTABLISHED
replay_source_args = _factor_set_arguments(
fixture=replay_source,
snapshot=snapshot,
foundation=foundation,
)
replay_source_args.update(
selected_view_ref_ids=(replay_view_id,),
input_bindings=(
InputBinding(
replay_source_args["definitions"][0].definition_id,
"market",
replay_view_id,
VIEW_SCHEMA_DIGEST,
),
),
view_availability=(
ViewAvailability(replay_view_id, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
),
causation=Causation("foundation", foundation.foundation_id),
)
with pytest.raises(FactorContractError) as late_publication:
FactorSetRef.create(**replay_source_args)
_assert_error(
late_publication,
ContractErrorCode.TIME_ORDER_VIOLATION,
"$.dataset_snapshot.descriptor.published_at",
)
@pytest.mark.parametrize(
("overrides", "path"),
[
({"view_availability": (ViewAvailability(VIEW_REF_ID, "2026-01-03T00:00:01Z", "sha256:" + "2" * 64),)}, "$.view_availability"),
({"computed_at": "2026-01-02T23:40:00Z"}, "$.computed_at"),
({"artifact_available_at": "2026-01-03T10:14:00Z"}, "$.artifact_available_at"),
({"artifact_available_at": "2026-01-03T11:00:01Z"}, "$.artifact_available_at"),
({"evaluation_at": "2026-01-03T11:00:00"}, "$.evaluation_at"),
],
)
def test_as_available_time_failures_are_typed(overrides: dict[str, Any], path: str) -> None:
with pytest.raises(FactorContractError) as error:
_factor_set(**overrides)
assert error.value.code in {
ContractErrorCode.INVALID_FORMAT,
ContractErrorCode.TIME_ORDER_VIOLATION,
}
assert error.value.path == path
def test_replay_rejects_backdating_and_historical_availability_promotion() -> None:
source = _replay_fixture()
snapshot, foundation = _snapshot_and_foundation(source)
view_id = next(iter(foundation.views))
arguments = _factor_set_arguments(fixture=source, snapshot=snapshot, foundation=foundation)
definition = arguments["definitions"][0]
arguments.update(
selected_view_ref_ids=(view_id,),
input_bindings=(InputBinding(definition.definition_id, "market", view_id, VIEW_SCHEMA_DIGEST),),
view_availability=(ViewAvailability(view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),),
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
computed_at="2026-01-04T00:20:00Z",
artifact_available_at="2026-01-04T00:25:00Z",
causation=Causation("foundation", foundation.foundation_id),
)
replay = FactorSetRef.create(**arguments)
promoted = replay.to_dict()
promoted["historical_availability"] = "declared_as_available"
_reidentify(promoted, "factor_set_id", "rhfactorsetv1:sha256:")
with pytest.raises(FactorContractError) as promotion_error:
FactorSetRef.from_dict(
promoted,
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
)
_assert_error(
promotion_error,
ContractErrorCode.READINESS_ESCALATION,
"$.historical_availability",
)
arguments["computed_at"] = "2026-01-03T11:30:00Z"
with pytest.raises(FactorContractError) as backdated_error:
FactorSetRef.create(**arguments)
_assert_error(backdated_error, ContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
def _multi_view_fixture() -> tuple[dict[str, Any], str]:
fixture = _golden()
foundation = fixture["data_foundation"]
second = copy.deepcopy(foundation["standardized_views"][0])
second["view_id"] = "rhview:11111111222222223333333344444444"
second["schema_digest"] = "sha256:" + "6" * 64
second["content_digest"] = "sha256:" + "7" * 64
second["transformation_digest"] = "sha256:" + "8" * 64
_reidentify(second, "view_ref_id", "rhviewrefv1:sha256:")
foundation["standardized_views"].append(second)
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
return fixture, second["view_ref_id"]
def test_multi_input_mapping_requires_exact_consumption_closure_and_is_order_independent() -> None:
fixture, second_view_id = _multi_view_fixture()
snapshot, foundation = _snapshot_and_foundation(fixture)
inputs = (
FactorInput("prices", VIEW_SCHEMA_DIGEST, ("close",)),
FactorInput("volumes", "sha256:" + "6" * 64, ("volume",)),
)
definition = _definition(
inputs=inputs,
formula="correlation(close, volume, 10)",
input_schema_digest=factor_input_schema_digest(inputs),
)
first_binding = InputBinding(definition.definition_id, "prices", VIEW_REF_ID, VIEW_SCHEMA_DIGEST)
second_binding = InputBinding(definition.definition_id, "volumes", second_view_id, "sha256:" + "6" * 64)
first_availability = ViewAvailability(VIEW_REF_ID, "2026-01-02T23:40:00Z", "sha256:" + "2" * 64)
second_availability = ViewAvailability(second_view_id, "2026-01-02T23:50:00Z", "sha256:" + "6" * 64)
base = _factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition)
base.update(
selected_view_ref_ids=(VIEW_REF_ID, second_view_id),
input_bindings=(first_binding, second_binding),
view_availability=(first_availability, second_availability),
causation=Causation("foundation", foundation.foundation_id),
)
first = FactorSetRef.create(**base)
reordered = dict(base)
reordered.update(
selected_view_ref_ids=(second_view_id, VIEW_REF_ID),
input_bindings=(second_binding, first_binding),
view_availability=(second_availability, first_availability),
)
assert FactorSetRef.create(**reordered).factor_set_id == first.factor_set_id
for invalid_bindings, invalid_views in (
((first_binding,), (VIEW_REF_ID, second_view_id)),
((first_binding, second_binding), (VIEW_REF_ID,)),
((first_binding, second_binding), (VIEW_REF_ID, second_view_id, VIEW_REF_ID)),
):
invalid = dict(base)
invalid.update(input_bindings=invalid_bindings, selected_view_ref_ids=invalid_views)
with pytest.raises(FactorContractError) as error:
FactorSetRef.create(**invalid)
assert error.value.code in {
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
ContractErrorCode.INVALID_VALUE,
}
@pytest.mark.parametrize(
("mutate", "code", "path"),
[
(lambda value: value["producer"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.producer.id"),
(lambda value: value["producer"].pop("version"), ContractErrorCode.MISSING_FIELD, "$.producer.version"),
(lambda value: value["producer"].update(id="other_engine"), ContractErrorCode.LINEAGE_VIOLATION, "$.producer.id"),
(lambda value: value["producer"].update(version="latest"), ContractErrorCode.INVALID_FORMAT, "$.producer.version"),
(lambda value: value.update(code_revision="bad"), ContractErrorCode.INVALID_FORMAT, "$.code_revision"),
(lambda value: value["actor"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.actor.kind"),
(lambda value: value["actor"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.actor.id"),
(lambda value: value["actor"].update(kind="robot"), ContractErrorCode.INVALID_VALUE, "$.actor.kind"),
(lambda value: value["actor"].update(id="latest"), ContractErrorCode.INVALID_VALUE, "$.actor.id"),
(lambda value: value.pop("correlation_id"), ContractErrorCode.MISSING_FIELD, "$.correlation_id"),
(lambda value: value.update(correlation_id="latest"), ContractErrorCode.INVALID_VALUE, "$.correlation_id"),
(lambda value: value["causation"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.causation.kind"),
(lambda value: value["causation"].update(kind="run"), ContractErrorCode.INVALID_VALUE, "$.causation.kind"),
(lambda value: value["causation"].update(id="rhdfv1:sha256:" + "0" * 64), ContractErrorCode.LINEAGE_VIOLATION, "$.causation.id"),
(lambda value: value.pop("output_artifact_ref"), ContractErrorCode.MISSING_FIELD, "$.output_artifact_ref"),
(lambda value: value["output_artifact_ref"].update(artifact_id="rhfactoroutputv1:sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.output_artifact_ref.artifact_id"),
(_mutate_artifact_schema_binding, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref"),
(lambda value: value.update(availability_mode="implicit_fallback"), ContractErrorCode.INVALID_VALUE, "$.availability_mode"),
(lambda value: value.pop("computed_at"), ContractErrorCode.MISSING_FIELD, "$.computed_at"),
(lambda value: value.update(decision_eligible=True), ContractErrorCode.READINESS_ESCALATION, "$.decision_eligible"),
(lambda value: value.update(evidence_scope="real_data"), ContractErrorCode.READINESS_ESCALATION, "$.evidence_scope"),
(lambda value: value["upstream_evidence"].update(qualification_evidence_digest="sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.upstream_evidence"),
],
)
def test_lineage_artifact_and_readiness_fields_have_independent_typed_negatives(
mutate: Callable[[dict[str, Any]], Any],
code: ContractErrorCode,
path: str,
) -> None:
factor_set = _factor_set()
value = factor_set.to_dict()
mutate(value)
if "factor_set_id" in value:
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
snapshot, foundation = _snapshot_and_foundation()
with pytest.raises(FactorContractError) as error:
FactorSetRef.from_dict(
value,
definitions=(_golden_definition(),),
dataset_snapshot=snapshot,
foundation=foundation,
)
_assert_error(error, code, path)
def test_output_schema_content_bytes_cannot_be_swapped_or_forged() -> None:
factor_set = _factor_set()
fixture = _golden()
snapshot, foundation = _snapshot_and_foundation()
schema_bytes = canonical_json_bytes(fixture["output_schema"])
content_bytes = canonical_json_bytes(fixture["output_content"])
with pytest.raises(FactorContractError) as swapped:
FactorSetRef.from_dict(
factor_set.to_dict(),
definitions=(_golden_definition(),),
dataset_snapshot=snapshot,
foundation=foundation,
output_schema_bytes=content_bytes,
output_content_bytes=schema_bytes,
)
_assert_error(swapped, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref")
with pytest.raises(FactorContractError) as noncanonical:
FactorSetRef.create(
**{
**_factor_set_arguments(),
"output_schema_bytes": json.dumps(fixture["output_schema"], indent=2).encode(),
}
)
_assert_error(noncanonical, ContractErrorCode.INVALID_FORMAT, "$.output_schema_bytes")
def test_unsuccessful_output_quality_or_coverage_cannot_form_a_factor_set() -> None:
with pytest.raises(FactorContractError) as failed_quality:
_factor_set(
output_quality=OutputQuality(
"failed",
(OutputQualityCheck("finite_values", "failed", "sha256:" + "3" * 64),),
)
)
_assert_error(failed_quality, ContractErrorCode.INVALID_VALUE, "$.output_quality")
for coverage in (
OutputCoverage("incomplete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
OutputCoverage("complete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
):
with pytest.raises(FactorContractError) as incomplete:
_factor_set(output_coverage=coverage)
_assert_error(incomplete, ContractErrorCode.INVALID_VALUE, "$.output_coverage")
def test_external_snapshot_definition_and_view_references_cannot_be_substituted() -> None:
factor_set = _factor_set()
snapshot, foundation = _snapshot_and_foundation()
value = factor_set.to_dict()
value["dataset_snapshot_id"] = "rhdsv1:sha256:" + "0" * 64
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
with pytest.raises(FactorContractError) as snapshot_error:
FactorSetRef.from_dict(
value,
definitions=(_golden_definition(),),
dataset_snapshot=snapshot,
foundation=foundation,
)
_assert_error(
snapshot_error,
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.dataset_snapshot_id",
)
value = factor_set.to_dict()
value["definition_ids"] = ["rhfactorv1:sha256:" + "0" * 64]
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
with pytest.raises(FactorContractError) as definition_error:
FactorSetRef.from_dict(
value,
definitions=(_golden_definition(),),
dataset_snapshot=snapshot,
foundation=foundation,
)
_assert_error(
definition_error,
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.definition_ids",
)
arguments = _factor_set_arguments()
arguments["selected_view_ref_ids"] = ("rhviewrefv1:sha256:" + "0" * 64,)
with pytest.raises(FactorContractError) as view_error:
FactorSetRef.create(**arguments)
_assert_error(
view_error,
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
"$.selected_view_ref_ids",
)
@pytest.mark.parametrize(
"invalid_definition_id",
[
{"unexpected": "object"},
["array"],
42,
True,
None,
],
)
def test_factor_set_ref_definition_ids_reject_non_string_types(
invalid_definition_id: Any,
) -> None:
factor_set = _factor_set()
definition = _golden_definition()
value = factor_set.to_dict()
value["definition_ids"] = [definition.definition_id, invalid_definition_id]
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
snapshot, foundation = _snapshot_and_foundation()
with pytest.raises(FactorContractError) as error:
FactorSetRef.from_json(
canonical_json_bytes(value),
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
)
_assert_error(error, ContractErrorCode.TYPE_ERROR, "$.definition_ids[1]")
def test_factor_set_ref_definition_ids_still_reject_duplicate_strings() -> None:
factor_set = _factor_set()
definition = _golden_definition()
value = factor_set.to_dict()
value["definition_ids"] = [definition.definition_id, definition.definition_id]
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
snapshot, foundation = _snapshot_and_foundation()
with pytest.raises(FactorContractError) as error:
FactorSetRef.from_json(
canonical_json_bytes(value),
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
)
_assert_error(error, ContractErrorCode.INVALID_VALUE, "$.definition_ids")
def test_factor_set_parent_requires_exact_identity_and_correlation() -> None:
parent = _factor_set()
child_arguments = _factor_set_arguments()
child_arguments.update(
output_content_bytes=canonical_json_bytes({"rows": [{"value": "0.250"}]}),
causation=Causation("factor_set", parent.factor_set_id),
parent=parent,
)
child_arguments["output_artifact_ref"] = OutputArtifactRef.create(
schema_digest=_sha256(child_arguments["output_schema_bytes"]),
content_digest=_sha256(child_arguments["output_content_bytes"]),
)
child = FactorSetRef.create(**child_arguments)
assert child.causation.id == parent.factor_set_id
missing_parent = child.to_dict()
snapshot, foundation = _snapshot_and_foundation()
with pytest.raises(FactorContractError) as missing_error:
FactorSetRef.from_dict(
missing_parent,
definitions=(_golden_definition(),),
dataset_snapshot=snapshot,
foundation=foundation,
)
_assert_error(missing_error, ContractErrorCode.LINEAGE_VIOLATION, "$.causation")
wrong_correlation = dict(child_arguments)
wrong_correlation["correlation_id"] = "different_run"
with pytest.raises(FactorContractError) as correlation_error:
FactorSetRef.create(**wrong_correlation)
_assert_error(correlation_error, ContractErrorCode.LINEAGE_VIOLATION, "$.correlation_id")
def test_legacy_bridge_is_explicit_lossy_and_preserves_all_four_historical_fields() -> None:
definition = _golden_definition()
legacy = FactorVersion(
factor_id="factor:demo-momentum",
version="1.0.0",
definition_sha256="b" * 64,
dataset_schema_version="1.0.0",
)
binding = LegacyFactorBinding.create(
definition=definition,
legacy_factor_id=legacy.factor_id,
legacy_version=legacy.version,
legacy_definition_sha256=legacy.definition_sha256,
legacy_dataset_schema_version=legacy.dataset_schema_version,
canonical_input_schema_digest=definition.input_schema_digest,
correspondence_evidence_digest="sha256:" + "5" * 64,
)
assert bind_legacy_factor(legacy, definition, binding) is definition
assert project_legacy_factor(definition, binding) == legacy
assert legacy.version_id == "factor:demo-momentum@1.0.0"
assert legacy.definition_sha256 != definition.definition_id.rsplit(":", maxsplit=1)[-1]
assert LegacyFactorBinding.from_json(binding.to_json(), definition=definition) == binding
mismatched = FactorVersion(
factor_id="factor:different",
version=legacy.version,
definition_sha256=legacy.definition_sha256,
dataset_schema_version=legacy.dataset_schema_version,
)
with pytest.raises(FactorContractError) as mismatch_error:
bind_legacy_factor(mismatched, definition, binding)
_assert_error(mismatch_error, ContractErrorCode.LEGACY_BINDING_MISMATCH, "$.binding")
def test_bare_legacy_factor_or_id_cannot_enter_factor_set_contract() -> None:
legacy = FactorVersion("factor:demo-momentum", "1.0.0", "b" * 64, "1.0.0")
arguments = _factor_set_arguments()
arguments["definitions"] = (legacy,)
with pytest.raises(FactorContractError) as legacy_error:
FactorSetRef.create(**arguments)
_assert_error(legacy_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
arguments["definitions"] = (legacy.version_id,)
with pytest.raises(FactorContractError) as id_error:
FactorSetRef.create(**arguments)
_assert_error(id_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
+173
View File
@@ -0,0 +1,173 @@
"""Contracts for reusable factor diagnostics and transformations."""
from __future__ import annotations
import numpy as np
import pandas as pd
import pytest
from quant_engine.factor_library import (
annualized_sharpe,
apply_factor_direction,
cross_sectional_momentum,
cross_sectional_pct_rank,
cross_sectional_rank_with_direction,
ic_summary,
jb_test,
kurtosis,
ols_regress,
rolling_annual_vol,
rolling_zscore,
skewness,
spearman_ic,
time_series_momentum,
turnover,
winsorize,
)
def test_turnover_supports_one_way_and_round_trip_conventions() -> None:
weights = pd.DataFrame({"A": [1.0, 0.0], "B": [0.0, 1.0]})
pd.testing.assert_series_equal(turnover(weights), pd.Series([1.0], index=[1]))
pd.testing.assert_series_equal(
turnover(weights, divide_by_two=False), pd.Series([2.0], index=[1])
)
assert turnover(weights.iloc[:1]).empty
def test_ic_functions_measure_monotonic_relationship() -> None:
factor = pd.Series([1.0, 2.0, 3.0, 4.0])
forward = pd.Series([10.0, 20.0, 30.0, 40.0])
assert spearman_ic(factor, forward) == pytest.approx(1.0)
result = ic_summary(factor, forward, periods=(1,), method="pearson")
assert result.loc[1, "ic_mean"] == pytest.approx(1.0)
assert result.loc[1, "n"] == 4
def test_ic_summary_rejects_unknown_method() -> None:
with pytest.raises(ValueError, match="not supported"):
ic_summary(pd.Series([1, 2, 3]), pd.Series([1, 2, 3]), method="kendall")
def test_winsorize_clips_tails_and_preserves_nan() -> None:
values = pd.Series([0.0, 1.0, 2.0, 100.0, np.nan])
result = winsorize(values, lower=0.25, upper=0.75)
assert result.iloc[0] == pytest.approx(0.75)
assert result.iloc[3] == pytest.approx(26.5)
assert pd.isna(result.iloc[4])
def test_distribution_diagnostics_handle_short_samples() -> None:
assert np.isnan(skewness(pd.Series([1.0, 2.0])))
assert np.isnan(kurtosis(pd.Series([1.0, 2.0, 3.0])))
jb, p_value = jb_test(pd.Series(range(7), dtype=float))
assert np.isnan(jb)
assert np.isnan(p_value)
def test_distribution_diagnostics_return_finite_values() -> None:
values = pd.Series([-2.0, -1.0, -0.5, 0.0, 0.25, 0.75, 1.0, 3.0])
assert np.isfinite(skewness(values))
assert np.isfinite(kurtosis(values))
jb, p_value = jb_test(values)
assert jb >= 0
assert 0 <= p_value <= 1
def test_ols_recovers_linear_coefficients_and_residual_index() -> None:
index = pd.date_range("2026-01-01", periods=8)
factor = pd.Series(np.arange(8, dtype=float), index=index, name="factor")
target = 1.5 + 2.0 * factor
result = ols_regress(target, factor)
assert result.alpha == pytest.approx(1.5)
assert result.beta["factor"] == pytest.approx(2.0)
assert result.r_squared == pytest.approx(1.0)
assert result.n == 8
assert result.resid.index.equals(index)
def test_ols_handles_collinear_factors_without_crashing() -> None:
x = pd.DataFrame({"a": np.arange(8, dtype=float), "b": np.arange(8, dtype=float)})
y = pd.Series(1.0 + x["a"])
result = ols_regress(y, x)
assert result.n == 8
assert np.isfinite(result.beta).all()
np.testing.assert_allclose(result.resid, 0.0, atol=1e-12)
def test_ols_short_sample_returns_empty_estimate() -> None:
result = ols_regress(pd.Series([1.0, 2.0]), pd.Series([1.0, 2.0], name="x"))
assert np.isnan(result.alpha)
assert result.beta.empty
assert result.n == 2
def test_momentum_and_rolling_transforms_match_manual_values() -> None:
prices = pd.DataFrame({"A": [100.0, 110.0, 121.0, 133.1]})
momentum = cross_sectional_momentum(prices, lookback=2, skip=0)
assert momentum.iloc[2, 0] == pytest.approx(0.21)
returns = pd.Series([0.1, 0.1, -0.5, -0.5])
pd.testing.assert_series_equal(
time_series_momentum(returns, lookback=2),
pd.Series([0, 1, -1, -1]),
)
values = pd.Series([1.0, 2.0, 3.0])
zscore = rolling_zscore(values, window=3)
assert zscore.iloc[-1] == pytest.approx(1.0)
annual_vol = rolling_annual_vol(returns, window=2, min_periods=2, trading_days=4)
assert annual_vol.iloc[1] == pytest.approx(0.0)
def test_rank_helpers_support_global_and_grouped_ranking() -> None:
frame = pd.DataFrame(
{"factor": [3.0, 1.0, 2.0, 4.0], "industry": ["x", "x", "y", "y"]}
)
global_rank = cross_sectional_pct_rank(frame, "factor", ascending=True)
grouped_rank = cross_sectional_pct_rank(
frame, "factor", group_col="industry", ascending=True
)
assert global_rank.tolist() == [0.75, 0.25, 0.5, 1.0]
assert grouped_rank.tolist() == [1.0, 0.5, 0.5, 1.0]
assert cross_sectional_pct_rank(frame, "missing").empty
def test_factor_direction_and_directional_rank() -> None:
pe = pd.Series([10.0, 20.0], name="pe_ttm")
pd.testing.assert_series_equal(apply_factor_direction(pe), -pe)
frame = pd.DataFrame({"pe_ttm": [10.0, 20.0], "roe": [0.1, 0.2]})
assert cross_sectional_rank_with_direction(frame, "pe_ttm").tolist() == [1.0, 0.5]
assert cross_sectional_rank_with_direction(frame, "roe").tolist() == [0.5, 1.0]
@pytest.mark.parametrize("direction", ["sideways", "", "REVERSE"])
def test_factor_direction_rejects_unknown_values(direction: str) -> None:
factor = pd.Series([1.0, 2.0], name="roe")
with pytest.raises(ValueError, match="direction"):
apply_factor_direction(factor, direction=direction)
with pytest.raises(ValueError, match="direction"):
cross_sectional_rank_with_direction(
pd.DataFrame({"roe": factor}), "roe", direction=direction
)
def test_annualized_sharpe_handles_empty_and_nonzero_returns() -> None:
assert annualized_sharpe(pd.Series(dtype=float)) == 0.0
returns = pd.Series([0.01, -0.01, 0.02, 0.0])
expected = returns.mean() * 252 / (returns.std() * np.sqrt(252))
assert annualized_sharpe(returns) == pytest.approx(expected)
+338
View File
@@ -0,0 +1,338 @@
"""Governed Personal Quant OS vertical-slice contracts."""
from __future__ import annotations
from datetime import UTC, datetime
import pandas as pd
import pytest
from quant_engine.execution import ExecutionConfig
from quant_engine.governed_pipeline import (
DatasetSnapshot,
FactorVersion,
PaperOrderIntent,
RiskDecisionStatus,
RiskPolicy,
StrategyStage,
StrategyVersion,
create_paper_order_intent,
run_governed_factor_slice,
)
def _calendar() -> pd.DatetimeIndex:
return pd.date_range("2026-01-05", periods=4, freq="B")
def _scores() -> pd.DataFrame:
dates = _calendar()
return pd.DataFrame(
{"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]},
index=dates[:2],
)
def _prices() -> tuple[pd.DataFrame, pd.DataFrame]:
dates = _calendar()
opens = pd.DataFrame(
{"A": [10.0, 10.0, 10.2, 10.4], "B": [20.0, 20.0, 20.5, 21.0], "C": [30.0, 30.0, 30.0, 30.0]},
index=dates,
)
closes = opens * 1.01
return opens, closes
def _snapshot() -> DatasetSnapshot:
return DatasetSnapshot(
snapshot_id="dataset:cn-a-daily-20260108-v1",
schema_version="1.0.0",
content_sha256="a" * 64,
effective_at=datetime(2026, 1, 8, 7, tzinfo=UTC),
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
ingested_at=datetime(2026, 1, 8, 8, 5, tzinfo=UTC),
)
def _factor() -> FactorVersion:
return FactorVersion(
factor_id="factor:demo-momentum",
version="1.0.0",
definition_sha256="b" * 64,
dataset_schema_version="1.0.0",
)
def _strategy() -> StrategyVersion:
return StrategyVersion(
strategy_id="strategy:demo-top2",
version="1.0.0",
factor_version_id="factor:demo-momentum@1.0.0",
stage=StrategyStage.APPROVED,
)
def _execution_config() -> ExecutionConfig:
return ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
def test_governed_slice_is_reproducible_and_creates_only_paper_intent() -> None:
opens, closes = _prices()
created_at = datetime(2026, 1, 9, 1, tzinfo=UTC)
policy = RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
)
result = run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=_strategy(),
risk_policy=policy,
code_revision="c" * 40,
created_at=created_at,
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
assert result.backtest_run.dataset_snapshot_id == _snapshot().snapshot_id
assert result.backtest_run.factor_version_id == _factor().version_id
assert result.backtest_run.strategy_version_id == _strategy().version_id
assert result.backtest_run.code_revision == "c" * 40
assert len(result.backtest_run.config_hash) == 64
assert result.portfolio_target.backtest_run_id == result.backtest_run.run_id
assert result.risk_decision.status is RiskDecisionStatus.APPROVED
assert result.risk_decision.portfolio_target_id == result.portfolio_target.target_id
assert result.order_intent is not None
assert result.order_intent.environment == "paper"
assert result.order_intent.risk_decision_id == result.risk_decision.decision_id
assert result.order_intent.portfolio_target_id == result.portfolio_target.target_id
assert result.factor_version.version_id == "factor:demo-momentum@1.0.0"
assert result.factor_version.definition_sha256 == "b" * 64
assert result.strategy_version.version_id == "strategy:demo-top2@1.0.0"
assert result.backtest_run.run_id == (
"backtest-run:e74403571f6a73c98b380220748957a422518c0bed883fabc4b93ebe13f05a37"
)
assert result.backtest_run.config_hash == (
"40a3c804a2dc940161a626d1a5d25817c13463e685005e37fc48c41d1e20b87b"
)
assert result.portfolio_target.target_id == (
"portfolio-target:ab2d398489aa9a292ee155a1098e9340beac0a924874cdaeb5b7b4d379ac9ce8"
)
assert result.risk_decision.decision_id == (
"risk-decision:95924926bd327e44beeb15a63f47d14c5e77b97533fc3227fba2eda68e9b423d"
)
assert result.order_intent.intent_id == (
"order-intent:73c349086c2c05b68424ace9286896eb21507b9080da5ff9b96b58c88d8f6ac1"
)
repeated = run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=_strategy(),
risk_policy=policy,
code_revision="c" * 40,
created_at=created_at,
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
assert repeated.backtest_run.run_id == result.backtest_run.run_id
assert repeated.portfolio_target.target_id == result.portfolio_target.target_id
assert repeated.risk_decision.decision_id == result.risk_decision.decision_id
assert repeated.order_intent == result.order_intent
def test_risk_rejection_blocks_order_intent() -> None:
opens, closes = _prices()
result = run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=_strategy(),
risk_policy=RiskPolicy(
policy_id="risk:no-concentration@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.4,
max_positions=10,
),
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
assert result.risk_decision.status is RiskDecisionStatus.REJECTED
assert any("single-asset weight" in reason for reason in result.risk_decision.reasons)
assert result.order_intent is None
with pytest.raises(ValueError, match="approved risk decision"):
create_paper_order_intent(result.portfolio_target, result.risk_decision)
with pytest.raises(ValueError, match="approved risk decision"):
PaperOrderIntent(result.portfolio_target, result.risk_decision)
def test_dataset_snapshot_requires_point_in_time_ordering_and_aware_times() -> None:
with pytest.raises(ValueError, match="timezone-aware"):
DatasetSnapshot(
snapshot_id="dataset:invalid",
schema_version="1.0.0",
content_sha256="a" * 64,
effective_at=datetime(2026, 1, 8, 7),
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
ingested_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
)
with pytest.raises(ValueError, match="effective_at <= available_at <= ingested_at"):
DatasetSnapshot(
snapshot_id="dataset:invalid",
schema_version="1.0.0",
content_sha256="a" * 64,
effective_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
ingested_at=datetime(2026, 1, 8, 10, tzinfo=UTC),
)
def test_strategy_factor_lineage_must_match() -> None:
opens, closes = _prices()
mismatched = StrategyVersion(
strategy_id="strategy:demo-top2",
version="1.0.0",
factor_version_id="factor:other@1.0.0",
stage=StrategyStage.APPROVED,
)
with pytest.raises(ValueError, match="factor lineage"):
run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=mismatched,
risk_policy=RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
),
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
def test_governed_slice_requires_matching_schema_and_snapshot_available_by_run_time() -> None:
opens, closes = _prices()
common = {
"factor_scores": _scores(),
"execution_prices": opens,
"valuation_prices": closes,
"strategy_version": _strategy(),
"risk_policy": RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
),
"code_revision": "c" * 40,
"top_k": 2,
"execution_price_field": "open",
"valuation_price_field": "close",
"execution_config": _execution_config(),
}
with pytest.raises(ValueError, match="dataset schema"):
run_governed_factor_slice(
dataset_snapshot=_snapshot(),
factor_version=FactorVersion(
factor_id="factor:demo-momentum",
version="1.0.0",
definition_sha256="b" * 64,
dataset_schema_version="2.0.0",
),
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
**common,
)
with pytest.raises(ValueError, match="available before the research run"):
run_governed_factor_slice(
dataset_snapshot=_snapshot(),
factor_version=_factor(),
created_at=datetime(2026, 1, 8, 7, 30, tzinfo=UTC),
**common,
)
future_scores = _scores()
future_scores.index = pd.date_range("2026-01-12", periods=2, freq="B")
with pytest.raises(ValueError, match="future decision dates"):
run_governed_factor_slice(
dataset_snapshot=_snapshot(),
factor_version=_factor(),
factor_scores=future_scores,
execution_prices=opens,
valuation_prices=closes,
strategy_version=common["strategy_version"],
risk_policy=common["risk_policy"],
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
def test_paper_intent_requires_approved_strategy_stage() -> None:
opens, closes = _prices()
validated = StrategyVersion(
strategy_id="strategy:demo-top2",
version="1.0.0",
factor_version_id=_factor().version_id,
stage=StrategyStage.VALIDATED,
)
with pytest.raises(ValueError, match="Approved or Paper"):
run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=validated,
risk_policy=RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
),
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
+163
View File
@@ -0,0 +1,163 @@
"""Mathematical contracts for the standard performance metrics."""
from __future__ import annotations
import numpy as np
import pandas as pd
import pytest
from quant_engine.metrics import (
TRADING_DAYS_PER_YEAR,
annualized_return,
annualized_volatility,
benchmark_summary,
calmar_ratio,
max_drawdown,
sharpe_ratio,
sortino_ratio,
summary,
win_rate,
)
def test_annualized_return_uses_compounded_simple_returns() -> None:
returns = pd.Series([0.10, -0.10])
expected = 0.99 ** (TRADING_DAYS_PER_YEAR / 2) - 1.0
assert annualized_return(returns) == pytest.approx(expected)
def test_annualized_volatility_uses_sample_standard_deviation() -> None:
returns = pd.Series([0.01, 0.03, 0.02])
assert annualized_volatility(returns) == pytest.approx(
returns.std() * np.sqrt(TRADING_DAYS_PER_YEAR)
)
def test_sharpe_ratio_subtracts_annual_risk_free_rate() -> None:
returns = pd.Series([0.01, -0.005, 0.02, 0.0])
result = sharpe_ratio(returns, rf=0.02)
assert result == pytest.approx(
(annualized_return(returns) - 0.02) / annualized_volatility(returns)
)
def test_zero_volatility_metrics_return_zero() -> None:
returns = pd.Series([0.0, 0.0, 0.0])
assert sharpe_ratio(returns) == 0.0
assert sortino_ratio(returns) == 0.0
assert calmar_ratio(returns) == 0.0
def test_sortino_ratio_uses_all_sessions_for_downside_deviation() -> None:
returns = pd.Series([0.02, -0.01, 0.0, -0.03])
downside = np.minimum(returns.to_numpy(), 0.0)
downside_deviation = np.sqrt(np.mean(np.square(downside))) * np.sqrt(
TRADING_DAYS_PER_YEAR
)
assert sortino_ratio(returns) == pytest.approx(
annualized_return(returns) / downside_deviation
)
def test_max_drawdown_includes_loss_from_initial_capital() -> None:
returns = pd.Series([-0.20, 0.0])
assert max_drawdown(returns) == pytest.approx(-0.20)
def test_max_drawdown_tracks_peak_to_trough_loss() -> None:
returns = pd.Series([0.10, -0.20, 0.05])
assert max_drawdown(returns) == pytest.approx(-0.20)
def test_metrics_clean_nan_and_infinite_values() -> None:
returns = pd.Series([0.10, np.nan, np.inf, -0.05, -np.inf])
assert win_rate(returns) == 0.5
assert summary(returns)["n_days"] == 2
def test_summary_aliases_match_canonical_fields() -> None:
result = summary(pd.Series([0.01, -0.02, 0.03]))
assert result["annual_yield"] == result["ann_return"]
assert result["annual_sd"] == result["ann_volatility"]
assert result["drawback"] == result["max_drawdown"]
@pytest.mark.parametrize(
"metric",
[
annualized_return,
annualized_volatility,
sharpe_ratio,
sortino_ratio,
max_drawdown,
calmar_ratio,
win_rate,
],
)
def test_metrics_reject_non_series_input(metric) -> None:
with pytest.raises(TypeError, match=r"expected pd\.Series"):
metric([0.01, 0.02])
def test_short_and_empty_series_return_zero() -> None:
assert annualized_return(pd.Series(dtype=float)) == 0.0
assert annualized_volatility(pd.Series([0.01])) == 0.0
assert max_drawdown(pd.Series([0.01])) == 0.0
assert win_rate(pd.Series(dtype=float)) == 0.0
def test_benchmark_summary_uses_aligned_active_returns_and_regression() -> None:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
benchmark = pd.Series([-0.01, 0.0, 0.01, 0.02], index=dates)
portfolio = 0.001 + 1.5 * benchmark
active = portfolio - benchmark
result = benchmark_summary(portfolio, benchmark)
assert result["n_observations"] == 4
assert result["tracking_error"] == pytest.approx(
active.std() * np.sqrt(TRADING_DAYS_PER_YEAR)
)
assert result["information_ratio"] == pytest.approx(
active.mean() / active.std() * np.sqrt(TRADING_DAYS_PER_YEAR)
)
assert result["beta"] == pytest.approx(1.5)
assert result["alpha"] == pytest.approx(1.001**TRADING_DAYS_PER_YEAR - 1.0)
def test_benchmark_summary_rejects_silent_calendar_alignment() -> None:
portfolio = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-05", periods=2))
benchmark = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-06", periods=2))
with pytest.raises(ValueError, match="matching indexes"):
benchmark_summary(portfolio, benchmark)
def test_benchmark_summary_rejects_missing_observations() -> None:
dates = pd.date_range("2026-01-05", periods=2)
portfolio = pd.Series([0.01, np.nan], index=dates)
benchmark = pd.Series([0.0, 0.01], index=dates)
with pytest.raises(ValueError, match="finite"):
benchmark_summary(portfolio, benchmark)
def test_benchmark_summary_marks_constant_benchmark_regression_unestimable() -> None:
dates = pd.date_range("2026-01-05", periods=3)
portfolio = pd.Series([0.01, -0.01, 0.02], index=dates)
benchmark = pd.Series([0.0, 0.0, 0.0], index=dates)
result = benchmark_summary(portfolio, benchmark)
assert np.isnan(result["alpha"])
assert np.isnan(result["beta"])
+727
View File
@@ -0,0 +1,727 @@
"""Closed performance-evidence contract conformance tests."""
from __future__ import annotations
import copy
import hashlib
import json
from dataclasses import replace
from pathlib import Path
from typing import Any
import numpy as np
import pandas as pd
import pytest
from quant_engine.artifact import (
PERFORMANCE_EVIDENCE_SCHEMA_VERSION,
PERFORMANCE_METRIC_SCHEMA_ID,
PERFORMANCE_METHODOLOGY_ID,
BacktestEvidenceManifest,
EvidenceQualification,
PerformanceEvidenceError,
PerformanceEvidenceErrorCode,
PerformanceEvidenceV1,
PerformanceMetricAvailability,
ResearchRunArtifact,
build_backtest_evidence_manifest,
build_performance_evidence,
build_research_run_artifact,
)
from quant_engine.execution import ExecutionConfig
from quant_engine.factor_contracts import (
ActorIdentity,
AvailabilityMode,
Causation,
DataFoundationEnvelope,
DatasetSnapshotEnvelope,
FactorInput,
FactorSetRef,
InputBinding,
OutputArtifactRef,
OutputCoverage,
OutputQuality,
OutputQualityCheck,
ProducerIdentity,
ViewAvailability,
canonical_json_bytes,
factor_definition_from_alpha158,
factor_input_schema_digest,
)
from quant_engine.governed_pipeline import BacktestRunRef
from quant_engine.metrics import TRADING_DAYS_PER_YEAR, benchmark_summary, summary
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
ROOT = Path(__file__).resolve().parents[1]
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
PERFORMANCE_FIXTURE = (
ROOT / "tests" / "fixtures" / "performance-evidence-v1.golden.json"
)
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
def _sha256(value: bytes) -> str:
return f"sha256:{hashlib.sha256(value).hexdigest()}"
def _accepted_authorities() -> tuple[
DatasetSnapshotEnvelope,
DataFoundationEnvelope,
FactorSetRef,
]:
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
definition = factor_definition_from_alpha158(
"alpha_005",
version="1.0.0",
parameters={},
inputs=(factor_input,),
implementation_digest="sha256:" + "1" * 64,
input_schema_digest=factor_input_schema_digest((factor_input,)),
valid_from="2026-01-01T00:00:00.000000Z",
valid_until="2027-01-01T00:00:00Z",
warmup_sessions=10,
lag_sessions=1,
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
)
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
output_content_bytes = canonical_json_bytes(fixture["output_content"])
artifact_ref = OutputArtifactRef.create(
schema_digest=_sha256(output_schema_bytes),
content_digest=_sha256(output_content_bytes),
)
factor_set = FactorSetRef.create(
definitions=(definition,),
dataset_snapshot=snapshot,
foundation=foundation,
selected_view_ref_ids=(VIEW_REF_ID,),
input_bindings=(
InputBinding(
definition.definition_id,
"market",
VIEW_REF_ID,
VIEW_SCHEMA_DIGEST,
),
),
view_availability=(
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
),
output_quality=OutputQuality(
"passed",
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
),
output_coverage=OutputCoverage(
"complete",
1,
1,
"row",
"alpha_005.cn_a",
"sha256:" + "4" * 64,
),
output_schema_bytes=output_schema_bytes,
output_content_bytes=output_content_bytes,
output_artifact_ref=artifact_ref,
availability_mode=AvailabilityMode.AS_AVAILABLE,
evaluation_at="2026-01-03T11:00:00Z",
computed_at="2026-01-03T10:15:00Z",
artifact_available_at="2026-01-03T10:20:00Z",
producer=ProducerIdentity("quant_engine", "1.0.0"),
code_revision="c" * 40,
actor=ActorIdentity("service", "factor_worker_v1"),
correlation_id="research_run_001",
causation=Causation("foundation", foundation.foundation_id),
evidence_scope="synthetic_fixture",
decision_eligible=False,
)
return snapshot, foundation, factor_set
def _configuration_digest() -> str:
return _sha256(
json.dumps(
PARAMETERS,
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
).encode("utf-8")
)
def _run_ref(**overrides: Any) -> BacktestRunRef:
snapshot, foundation, factor_set = _accepted_authorities()
arguments: dict[str, Any] = {
"dataset_snapshot": snapshot,
"foundation": foundation,
"factor_set": factor_set,
"universe_digest": "sha256:" + "5" * 64,
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
"strategy_id": "alpha-top1",
"strategy_version": "1.0.0",
"strategy_digest": "sha256:" + "6" * 64,
"execution_model_version": "1.0.0",
"execution_model_digest": "sha256:" + "7" * 64,
"cost_model_version": "1.0.0",
"cost_model_digest": "sha256:" + "8" * 64,
"random_seed": 7,
"code_revision": "d" * 40,
"environment_lock_digest": "sha256:" + "9" * 64,
"configuration_digest": _configuration_digest(),
"evaluation_at": "2026-01-08T01:00:00Z",
"computed_at": "2026-01-08T02:00:00Z",
}
arguments.update(overrides)
return BacktestRunRef.create(**arguments)
def _backtest_result() -> FactorBacktestResult:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
opens = pd.DataFrame(
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
index=dates,
)
closes = pd.DataFrame(
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
index=dates,
)
return run_factor_backtest_research(
scores,
opens,
closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
),
)
def _artifact(
run_ref: BacktestRunRef,
benchmark_kind: str,
) -> tuple[ResearchRunArtifact, FactorBacktestResult]:
result = _backtest_result()
benchmark_id: str | None
benchmark_returns: pd.Series | None
if benchmark_kind == "absent":
benchmark_id = None
benchmark_returns = None
elif benchmark_kind == "estimable":
benchmark_id = "000300.SH"
benchmark_returns = pd.Series(
[0.0, 0.01, -0.01, 0.02],
index=result.returns.index,
name="benchmark_return",
)
elif benchmark_kind == "zero_active_variance":
benchmark_id = "000300.SH"
benchmark_returns = result.returns.rename("benchmark_return")
elif benchmark_kind == "zero_benchmark_variance":
benchmark_id = "000300.SH"
benchmark_returns = pd.Series(
np.zeros(len(result.returns)),
index=result.returns.index,
name="benchmark_return",
)
else:
raise AssertionError(f"unknown benchmark_kind: {benchmark_kind}")
artifact = build_research_run_artifact(
result,
run_id=run_ref.run_id,
strategy_id=run_ref.strategy_id,
strategy_name="Alpha Top 1",
strategy_version=run_ref.strategy_version,
engine_version="1.2.0",
code_revision=run_ref.code_revision,
data_snapshot_id=run_ref.dataset_snapshot_id,
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-01-08T10:00:00+08:00",
finished_at="2026-01-08T10:01:00+08:00",
parameters=PARAMETERS,
benchmark_id=benchmark_id,
benchmark_returns=benchmark_returns,
)
return artifact, result
def _case(
benchmark_kind: str,
) -> tuple[PerformanceEvidenceV1, ResearchRunArtifact, BacktestRunRef, BacktestEvidenceManifest]:
run_ref = _run_ref()
artifact, _ = _artifact(run_ref, benchmark_kind)
manifest = build_backtest_evidence_manifest(
run_ref,
artifact,
artifact_available_at="2026-01-08T02:05:00Z",
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
)
return (
build_performance_evidence(artifact, run_ref, manifest),
artifact,
run_ref,
manifest,
)
def _metric_map(evidence: PerformanceEvidenceV1) -> dict[str, Any]:
return {metric.key: metric for metric in evidence.metrics}
def _mutate_frozen(value: Any, field: str, replacement: object) -> Any:
changed = copy.copy(value)
object.__setattr__(changed, field, replacement)
return changed
def _assert_error(
error: pytest.ExceptionInfo[PerformanceEvidenceError],
code: PerformanceEvidenceErrorCode,
path: str,
) -> None:
assert error.value.code is code
assert error.value.path == path
def test_present_evidence_is_deterministic_content_addressed_and_three_party_closed() -> None:
first, artifact, run_ref, manifest = _case("estimable")
second = build_performance_evidence(artifact, run_ref, manifest)
assert first == second
assert first.schema_version == PERFORMANCE_EVIDENCE_SCHEMA_VERSION
assert first.performance_evidence_id.startswith("rhperformanceevidencev1:sha256:")
assert first.document_sha256.startswith("sha256:")
assert first.authority == "quant_engine"
assert first.scope == "offline_research_only"
assert first.run_id == first.backtest_run_ref_id == run_ref.run_id == manifest.run_id
assert first.backtest_evidence_manifest_id == manifest.manifest_id
assert first.backtest_evidence_manifest_evidence_digest == manifest.evidence_digest
assert first.backtest_evidence_qualification == "contract_qualified"
assert first.research_artifact_content_digest == f"sha256:{artifact.content_sha256}"
assert first.performance_table_logical_name == "performance"
assert first.performance_table_row_count == 1
assert first.performance_row_digest.startswith("sha256:")
assert first.benchmark_series_digest is not None
assert first.canonical_bytes() == first.to_json().encode("utf-8")
assert not first.canonical_bytes().endswith(b"\n")
document_payload = first.to_dict()
document_payload.pop("document_sha256")
expected_document = json.dumps(
document_payload,
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
).encode("utf-8")
assert _sha256(expected_document) == first.document_sha256
assert PerformanceEvidenceV1.from_dict(
first.to_dict(),
artifact=artifact,
run_ref=run_ref,
evidence_manifest=manifest,
) == first
def test_methodology_and_metrics_bind_the_actual_artifact_builder_path() -> None:
evidence, artifact, _, _ = _case("estimable")
result = _backtest_result()
expected_absolute = summary(result.returns, rf=0.0)
benchmark = artifact.nav.set_index("trade_date")["benchmark_return"]
benchmark.index = result.returns.index
expected_relative = benchmark_summary(
result.returns,
benchmark,
risk_free_daily=0.0,
annualization=TRADING_DAYS_PER_YEAR,
)
metrics = _metric_map(evidence)
assert evidence.methodology.methodology_id == PERFORMANCE_METHODOLOGY_ID
assert evidence.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID
assert evidence.methodology.return_type == "simple"
assert evidence.methodology.source_frequency == "1d"
assert evidence.methodology.periods_per_year == TRADING_DAYS_PER_YEAR == 252
assert evidence.methodology.annual_risk_free == 0.0
assert evidence.methodology.benchmark_risk_free_daily == 0.0
assert evidence.methodology.benchmark_alignment == "exact_session_index"
assert metrics["annualized_return"].value == pytest.approx(
expected_absolute["ann_return"]
)
assert metrics["sharpe_ratio"].value == pytest.approx(expected_absolute["sharpe"])
assert metrics["tracking_error"].value == pytest.approx(
expected_relative["tracking_error"]
)
assert metrics["alpha"].value == pytest.approx(expected_relative["alpha"])
assert all(metric.methodology_id == PERFORMANCE_METHODOLOGY_ID for metric in metrics.values())
assert all(metric.metric_schema_id == PERFORMANCE_METRIC_SCHEMA_ID for metric in metrics.values())
def test_relative_metric_availability_is_closed_for_present_absent_and_unestimable() -> None:
present, *_ = _case("estimable")
absent, *_ = _case("absent")
zero_active, *_ = _case("zero_active_variance")
zero_benchmark, *_ = _case("zero_benchmark_variance")
present_metrics = _metric_map(present)
assert all(
present_metrics[key].availability is PerformanceMetricAvailability.AVAILABLE
for key in ("tracking_error", "information_ratio", "alpha", "beta")
)
absent_metrics = _metric_map(absent)
assert absent.benchmark_series_digest is None
assert absent.benchmark_id == ""
assert absent.benchmark_alignment_policy == "none"
assert all(
absent_metrics[key].value is None
and absent_metrics[key].availability
is PerformanceMetricAvailability.BENCHMARK_ABSENT
for key in ("tracking_error", "information_ratio", "alpha", "beta")
)
zero_active_metrics = _metric_map(zero_active)
assert zero_active_metrics["tracking_error"].value == pytest.approx(0.0)
assert (
zero_active_metrics["information_ratio"].availability
is PerformanceMetricAvailability.NOT_ESTIMABLE_ACTIVE_VARIANCE
)
assert zero_active_metrics["information_ratio"].value is None
zero_benchmark_metrics = _metric_map(zero_benchmark)
assert np.isfinite(zero_benchmark_metrics["tracking_error"].value)
for key in ("alpha", "beta"):
assert zero_benchmark_metrics[key].value is None
assert (
zero_benchmark_metrics[key].availability
is PerformanceMetricAvailability.NOT_ESTIMABLE_BENCHMARK_VARIANCE
)
def test_golden_covers_present_absent_and_both_unestimable_states() -> None:
expected = {
"schema_version": 1,
"source_commit": "a724e1e57a99d1304a932d01ee836bac56c5c15c",
"source_tree": "4774e88442d25bf79a54eab3d7106ff4d0ba9603",
"cases": {
name: _case(name)[0].to_dict()
for name in (
"estimable",
"zero_active_variance",
"zero_benchmark_variance",
"absent",
)
},
}
assert json.loads(PERFORMANCE_FIXTURE.read_text(encoding="utf-8")) == expected
@pytest.mark.parametrize(
("owner", "field", "replacement", "code", "path"),
[
(
"run_ref",
"run_id",
"rhbacktestrunv1:sha256:" + "0" * 64,
PerformanceEvidenceErrorCode.IDENTITY_MISMATCH,
"$.backtest_run_ref.run_id",
),
(
"manifest",
"manifest_id",
"rhbacktestevidencev1:sha256:" + "0" * 64,
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
"$.backtest_evidence_manifest.manifest_id",
),
(
"manifest",
"qualification",
EvidenceQualification.EXPLORATORY,
PerformanceEvidenceErrorCode.AUTHORITY_REJECTED,
"$.backtest_evidence_manifest.qualification",
),
],
)
def test_owner_identity_and_authority_mismatches_fail_closed(
owner: str,
field: str,
replacement: object,
code: PerformanceEvidenceErrorCode,
path: str,
) -> None:
_, artifact, run_ref, manifest = _case("estimable")
changed_run_ref = _mutate_frozen(run_ref, field, replacement) if owner == "run_ref" else run_ref
changed_manifest = (
_mutate_frozen(manifest, field, replacement) if owner == "manifest" else manifest
)
with pytest.raises(PerformanceEvidenceError) as rejected:
build_performance_evidence(artifact, changed_run_ref, changed_manifest)
_assert_error(rejected, code, path)
def test_performance_table_row_and_benchmark_digest_mismatches_fail_closed() -> None:
evidence, artifact, run_ref, manifest = _case("estimable")
performance = artifact.performance
performance.loc[0, "n_days"] += 1
changed_artifact = replace(artifact, _performance=performance)
with pytest.raises(PerformanceEvidenceError) as table_mismatch:
build_performance_evidence(changed_artifact, run_ref, manifest)
_assert_error(
table_mismatch,
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
"$.artifact.tables.performance.content_digest",
)
payload = evidence.to_dict()
payload["benchmark_series_digest"] = "sha256:" + "0" * 64
with pytest.raises(PerformanceEvidenceError) as benchmark_mismatch:
PerformanceEvidenceV1.from_dict(
payload,
artifact=artifact,
run_ref=run_ref,
evidence_manifest=manifest,
)
_assert_error(
benchmark_mismatch,
PerformanceEvidenceErrorCode.BENCHMARK_INVALID,
"$.benchmark_series_digest",
)
def test_manifest_closure_covers_non_performance_artifact_tables() -> None:
_, artifact, run_ref, manifest = _case("estimable")
nav = artifact.nav
nav.loc[0, "nav"] += 0.01
changed_artifact = replace(artifact, _nav=nav)
with pytest.raises(PerformanceEvidenceError) as rejected:
build_performance_evidence(changed_artifact, run_ref, manifest)
_assert_error(
rejected,
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
"$.artifact.tables.nav.content_digest",
)
def test_row_and_benchmark_mutations_change_their_digests_and_document_identity() -> None:
original, artifact, run_ref, _ = _case("estimable")
performance = artifact.performance
performance.loc[0, "sharpe"] += 0.01
changed_performance_artifact = replace(artifact, _performance=performance)
changed_performance_manifest = build_backtest_evidence_manifest(
run_ref,
changed_performance_artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
changed_performance = build_performance_evidence(
changed_performance_artifact,
run_ref,
changed_performance_manifest,
)
assert changed_performance.performance_row_digest != original.performance_row_digest
assert changed_performance.performance_evidence_id != original.performance_evidence_id
nav = artifact.nav
nav.loc[0, "benchmark_nav"] += 0.01
changed_benchmark_artifact = replace(artifact, _nav=nav)
changed_benchmark_manifest = build_backtest_evidence_manifest(
run_ref,
changed_benchmark_artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
changed_benchmark = build_performance_evidence(
changed_benchmark_artifact,
run_ref,
changed_benchmark_manifest,
)
assert changed_benchmark.benchmark_series_digest != original.benchmark_series_digest
assert changed_benchmark.performance_row_digest == original.performance_row_digest
assert changed_benchmark.performance_evidence_id != original.performance_evidence_id
def test_relative_metric_null_reasons_cannot_be_invented() -> None:
_, artifact, run_ref, _ = _case("estimable")
performance = artifact.performance
performance.loc[0, "alpha"] = float("nan")
changed_artifact = replace(artifact, _performance=performance)
changed_manifest = build_backtest_evidence_manifest(
run_ref,
changed_artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
with pytest.raises(PerformanceEvidenceError) as false_alpha_domain:
build_performance_evidence(changed_artifact, run_ref, changed_manifest)
_assert_error(
false_alpha_domain,
PerformanceEvidenceErrorCode.METRIC_INVALID,
"$.metrics.alpha.value",
)
_, absent_artifact, absent_run_ref, _ = _case("absent")
absent_performance = absent_artifact.performance
absent_performance.loc[0, "tracking_error"] = 0.0
changed_absent = replace(absent_artifact, _performance=absent_performance)
changed_absent_manifest = build_backtest_evidence_manifest(
absent_run_ref,
changed_absent,
artifact_available_at="2026-01-08T02:05:00Z",
)
with pytest.raises(PerformanceEvidenceError) as false_absence:
build_performance_evidence(
changed_absent,
absent_run_ref,
changed_absent_manifest,
)
_assert_error(
false_absence,
PerformanceEvidenceErrorCode.BENCHMARK_INVALID,
"$.metrics.tracking_error.availability",
)
def test_artifact_builder_enforces_strict_benchmark_session_alignment() -> None:
run_ref = _run_ref()
result = _backtest_result()
misaligned = pd.Series(
[0.0, 0.01, -0.01, 0.02],
index=result.returns.index.shift(1, freq="B"),
)
with pytest.raises(ValueError, match="matching indexes"):
build_research_run_artifact(
result,
run_id=run_ref.run_id,
strategy_id=run_ref.strategy_id,
strategy_name="Alpha Top 1",
strategy_version=run_ref.strategy_version,
engine_version="1.2.0",
code_revision=run_ref.code_revision,
data_snapshot_id=run_ref.dataset_snapshot_id,
calendar="CN-A",
timezone="Asia/Shanghai",
started_at="2026-01-08T10:00:00+08:00",
finished_at="2026-01-08T10:01:00+08:00",
parameters=PARAMETERS,
benchmark_id="000300.SH",
benchmark_returns=misaligned,
)
@pytest.mark.parametrize(
("column", "value", "path"),
[
("total_ret", -1.01, "$.metrics.total_return.value"),
("ann_ret", -1.01, "$.metrics.annualized_return.value"),
("ann_volatility", -0.01, "$.metrics.annualized_volatility.value"),
("max_dd", 0.01, "$.metrics.maximum_drawdown.value"),
("win_rate", 1.01, "$.metrics.win_rate.value"),
("tracking_error", -0.01, "$.metrics.tracking_error.value"),
("n_trades", True, "$.metrics.trade_count.value"),
],
)
def test_metric_domains_reject_invalid_source_values(
column: str,
value: object,
path: str,
) -> None:
_, artifact, run_ref, _ = _case("estimable")
performance = artifact.performance.astype(object)
performance.at[0, column] = value
changed_artifact = replace(artifact, _performance=performance)
changed_manifest = build_backtest_evidence_manifest(
run_ref,
changed_artifact,
artifact_available_at="2026-01-08T02:05:00Z",
)
with pytest.raises(PerformanceEvidenceError) as rejected:
build_performance_evidence(changed_artifact, run_ref, changed_manifest)
_assert_error(rejected, PerformanceEvidenceErrorCode.METRIC_INVALID, path)
def test_closed_parser_rejects_unknown_non_ascii_non_finite_bool_and_unsafe_integer() -> None:
evidence, artifact, run_ref, manifest = _case("estimable")
mutations: list[tuple[dict[str, Any], PerformanceEvidenceErrorCode, str]] = []
unknown = evidence.to_dict()
unknown["unexpected"] = "value"
mutations.append((unknown, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.unexpected"))
non_ascii = evidence.to_dict()
non_ascii["métric"] = "value"
mutations.append((non_ascii, PerformanceEvidenceErrorCode.TYPE_ERROR, "$.métric"))
non_finite = evidence.to_dict()
non_finite["metrics"][0]["value"] = float("inf")
mutations.append(
(non_finite, PerformanceEvidenceErrorCode.METRIC_INVALID, "$.metrics[0].value")
)
bool_number = evidence.to_dict()
bool_number["methodology"]["periods_per_year"] = True
mutations.append(
(
bool_number,
PerformanceEvidenceErrorCode.METHODOLOGY_MISMATCH,
"$.methodology.periods_per_year",
)
)
unsafe = evidence.to_dict()
unsafe["performance_table_row_count"] = 2**53
mutations.append(
(
unsafe,
PerformanceEvidenceErrorCode.EVIDENCE_MISMATCH,
"$.performance_table_row_count",
)
)
for payload, code, path in mutations:
with pytest.raises(PerformanceEvidenceError) as rejected:
PerformanceEvidenceV1.from_dict(
payload,
artifact=artifact,
run_ref=run_ref,
evidence_manifest=manifest,
)
_assert_error(rejected, code, path)
def test_public_mapping_has_no_raw_inputs_storage_or_runtime_authority() -> None:
evidence, *_ = _case("estimable")
payload = evidence.to_dict()
serialized = evidence.to_json().lower()
forbidden_keys = {
"parameters",
"params_json",
"returns",
"nav",
"benchmark_series",
"table_bytes",
"locator",
"uri",
"credential",
"decision_eligible",
"publication_eligible",
"paper_trading",
"live_trading",
"investment_advice",
}
def keys(value: object) -> set[str]:
if isinstance(value, dict):
return set(value) | {key for item in value.values() for key in keys(item)}
if isinstance(value, list):
return {key for item in value for key in keys(item)}
return set()
assert not (keys(payload) & forbidden_keys)
for token in ("postgres://", "mysql://", "s3://", "credential", "broker"):
assert token not in serialized
+157
View File
@@ -0,0 +1,157 @@
"""Factor-score portfolio construction and backtest integration contracts."""
from __future__ import annotations
import numpy as np
import pandas as pd
import pytest
from quant_engine.backtest import run_weight_backtest
from quant_engine.portfolio_construction import (
equal_weight,
scores_to_target_weights,
scores_to_weight_table,
select_top_k,
)
def test_select_top_k_ignores_nan_and_breaks_ties_by_input_order() -> None:
scores = pd.Series([1.0, 1.0, np.nan, 0.5], index=["B", "A", "C", "D"])
selected = select_top_k(scores, top_k=2)
assert selected.tolist() == ["B", "A"]
def test_select_top_k_can_select_lowest_scores() -> None:
scores = pd.Series([3.0, 1.0, 2.0], index=["A", "B", "C"])
selected = select_top_k(scores, top_k=2, largest=False)
assert selected.tolist() == ["B", "C"]
def test_equal_weight_allocates_requested_gross_exposure() -> None:
result = equal_weight(pd.Index(["A", "B", "C"]), gross_exposure=0.9)
pd.testing.assert_series_equal(
result,
pd.Series([0.3, 0.3, 0.3], index=["A", "B", "C"], name="weight"),
)
def test_equal_weight_returns_empty_float_series_for_no_assets() -> None:
result = equal_weight(pd.Index([], dtype=object))
assert result.empty
assert result.dtype == float
assert result.name == "weight"
def test_scores_to_target_weights_keeps_full_universe_with_zero_for_unselected() -> None:
scores = pd.Series([0.2, 0.8, 0.5], index=["A", "B", "C"])
result = scores_to_target_weights(scores, top_k=2)
pd.testing.assert_series_equal(
result,
pd.Series([0.0, 0.5, 0.5], index=scores.index, name="weight"),
)
def test_scores_to_target_weights_divides_exposure_over_available_scores() -> None:
scores = pd.Series([1.0, np.nan, 0.5], index=["A", "B", "C"])
result = scores_to_target_weights(scores, top_k=5, gross_exposure=0.8)
pd.testing.assert_series_equal(
result,
pd.Series([0.4, 0.0, 0.4], index=scores.index, name="weight"),
)
def test_scores_to_weight_table_constructs_each_rebalance_independently() -> None:
dates = pd.to_datetime(["2026-01-05", "2026-01-07"])
scores = pd.DataFrame(
{"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]},
index=dates,
)
result = scores_to_weight_table(scores, top_k=2)
expected = pd.DataFrame(
{"A": [0.5, 0.0], "B": [0.5, 0.5], "C": [0.0, 0.5]},
index=dates,
)
pd.testing.assert_frame_equal(result, expected)
changed_future = scores.copy()
changed_future.iloc[1] = [100.0, -100.0, 0.0]
changed_result = scores_to_weight_table(changed_future, top_k=2)
pd.testing.assert_series_equal(result.iloc[0], changed_result.iloc[0])
def test_effective_holding_weights_flow_into_weight_backtest() -> None:
dates = pd.date_range("2026-01-05", periods=3, freq="B")
effective_weights = pd.DataFrame(
{"A": [1.0, 0.0], "B": [0.0, 1.0]},
index=dates[[0, 2]],
)
stock_returns = pd.DataFrame(
{"A": [0.10, 0.0, 0.0], "B": [0.0, 0.0, 0.20]},
index=dates,
)
result = run_weight_backtest(effective_weights, stock_returns)
pd.testing.assert_series_equal(result.nav, pd.Series([1.1, 1.1, 1.32], index=dates))
pd.testing.assert_frame_equal(result.weights, effective_weights)
@pytest.mark.parametrize("top_k", [0, -1])
def test_portfolio_construction_rejects_non_positive_top_k(top_k: int) -> None:
scores = pd.Series([1.0], index=["A"])
with pytest.raises(ValueError, match="top_k must be positive"):
select_top_k(scores, top_k=top_k)
@pytest.mark.parametrize("gross_exposure", [-0.1, np.inf, np.nan])
def test_equal_weight_rejects_invalid_gross_exposure(gross_exposure: float) -> None:
with pytest.raises(ValueError, match="gross_exposure"):
equal_weight(pd.Index(["A"]), gross_exposure=gross_exposure)
def test_portfolio_construction_rejects_duplicate_assets() -> None:
duplicate_scores = pd.Series([1.0, 2.0], index=["A", "A"])
with pytest.raises(ValueError, match="unique asset labels"):
scores_to_target_weights(duplicate_scores, top_k=1)
def test_weight_table_rejects_duplicate_rebalance_dates() -> None:
duplicate_date = pd.Timestamp("2026-01-05")
scores = pd.DataFrame(
{"A": [1.0, 2.0]},
index=[duplicate_date, duplicate_date],
)
with pytest.raises(ValueError, match="unique rebalance dates"):
scores_to_weight_table(scores, top_k=1)
def test_weight_table_rejects_unsorted_rebalance_dates() -> None:
scores = pd.DataFrame(
{"A": [1.0, 2.0]},
index=pd.to_datetime(["2026-01-07", "2026-01-05"]),
)
with pytest.raises(ValueError, match="chronological order"):
scores_to_weight_table(scores, top_k=1)
def test_weight_table_rejects_non_numeric_scores() -> None:
scores = pd.DataFrame({"A": ["high"], "B": ["low"]})
with pytest.raises(TypeError, match="numeric"):
scores_to_weight_table(scores, top_k=1)
File diff suppressed because it is too large Load Diff
+336
View File
@@ -0,0 +1,336 @@
"""No-lookahead factor-score to execution-audit integration contracts."""
from __future__ import annotations
import pandas as pd
import pytest
from quant_engine.execution import ExecutionConfig
from quant_engine.research_pipeline import (
FactorBacktestResult,
FactorExecutionResult,
TargetWeightSchedule,
run_factor_backtest_research,
run_factor_execution_research,
schedule_target_weights,
)
def _calendar() -> pd.DatetimeIndex:
return pd.date_range("2026-01-05", periods=4, freq="B")
def _factor_scores() -> pd.DataFrame:
dates = _calendar()
return pd.DataFrame(
{"A": [2.0, 0.0], "B": [1.0, 3.0]},
index=dates[:2],
)
def _next_session_open_prices() -> pd.DataFrame:
dates = _calendar()
return pd.DataFrame(
{"A": [1.0, 10.0, 10.0, 10.0], "B": [1.0, 10.0, 20.0, 20.0]},
index=dates,
)
def test_schedule_target_weights_maps_signal_to_next_trading_session() -> None:
dates = _calendar()
decision_weights = pd.DataFrame(
{"A": [1.0, 0.0], "B": [0.0, 1.0]},
index=dates[:2],
)
schedule = schedule_target_weights(decision_weights, dates, lag_sessions=1)
assert isinstance(schedule, TargetWeightSchedule)
assert schedule.lag_sessions == 1
pd.testing.assert_series_equal(
schedule.signal_to_execution,
pd.Series(dates[1:3], index=dates[:2], name="execution_date"),
)
expected = decision_weights.copy()
expected.index = dates[1:3]
expected.index.name = "execution_date"
pd.testing.assert_frame_equal(schedule.execution_weights, expected)
assert (schedule.execution_weights.index > schedule.signal_to_execution.index).all()
def test_factor_execution_research_uses_next_session_prices() -> None:
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
result = run_factor_execution_research(
_factor_scores(),
_next_session_open_prices(),
top_k=1,
execution_price_field="open",
initial_cash=1_000.0,
config=config,
)
assert isinstance(result, FactorExecutionResult)
assert result.execution_price_field == "open"
assert result.execution.daily_executions[0].date == str(_calendar()[1])
assert result.execution.positions[0].holdings == {"A": 100.0}
assert result.execution.positions[1].holdings == {"B": 50.0}
assert result.execution.final_portfolio_value == pytest.approx(1_000.0)
def test_factor_execution_result_snapshots_research_inputs() -> None:
scores = _factor_scores()
prices = _next_session_open_prices()
result = run_factor_execution_research(
scores,
prices,
top_k=1,
execution_price_field="open",
)
scores.iloc[0, 0] = -999.0
prices.iloc[1, 0] = 999.0
assert result.factor_scores.iloc[0, 0] == 2.0
assert result.execution_prices.loc[_calendar()[1], "A"] == 10.0
assert result.execution.positions[0].holdings["A"] < 200_000.0
@pytest.mark.parametrize("lag_sessions", [0, -1, True])
def test_schedule_target_weights_requires_positive_integer_lag(lag_sessions: int) -> None:
with pytest.raises(ValueError, match="lag_sessions"):
schedule_target_weights(
pd.DataFrame({"A": [1.0]}, index=_calendar()[:1]),
_calendar(),
lag_sessions=lag_sessions,
)
def test_schedule_target_weights_rejects_signal_outside_trading_calendar() -> None:
weekend = pd.Timestamp("2026-01-10")
with pytest.raises(ValueError, match="signal dates must be trading sessions"):
schedule_target_weights(
pd.DataFrame({"A": [1.0]}, index=[weekend]),
_calendar(),
)
def test_schedule_target_weights_rejects_missing_future_execution_session() -> None:
dates = _calendar()
with pytest.raises(ValueError, match="future execution session"):
schedule_target_weights(
pd.DataFrame({"A": [1.0]}, index=dates[-1:]),
dates,
)
def test_factor_execution_research_requires_explicit_price_field() -> None:
with pytest.raises(ValueError, match="execution_price_field"):
run_factor_execution_research(
_factor_scores(),
_next_session_open_prices(),
top_k=1,
execution_price_field="",
)
def test_factor_execution_research_accepts_empty_scores() -> None:
scores = pd.DataFrame(columns=["A", "B"], index=pd.DatetimeIndex([]), dtype=float)
result = run_factor_execution_research(
scores,
_next_session_open_prices(),
top_k=1,
execution_price_field="open",
)
assert result.schedule.execution_weights.empty
assert result.execution.positions == ()
def test_factor_backtest_research_runs_signal_to_daily_performance_without_lookahead() -> None:
"""信号日保持现金,下一日开盘成交后才参与当日收盘收益。"""
dates = _calendar()
scores = pd.DataFrame({"A": [2.0], "B": [1.0]}, index=dates[:1])
opens = pd.DataFrame(
{"A": [1.0, 10.0, 10.0, 10.0], "B": [1.0, 20.0, 20.0, 20.0]},
index=dates,
)
closes = pd.DataFrame(
{"A": [500.0, 11.0, 12.0, 12.0], "B": [500.0, 20.0, 20.0, 20.0]},
index=dates,
)
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
result = run_factor_backtest_research(
scores,
execution_prices=opens,
valuation_prices=closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=config,
)
assert isinstance(result, FactorBacktestResult)
assert result.execution_price_field == "open"
assert result.valuation_price_field == "close"
pd.testing.assert_series_equal(
result.nav,
pd.Series([1.0, 1.1, 1.2, 1.2], index=dates, name="nav"),
)
pd.testing.assert_series_equal(
result.returns,
pd.Series([0.0, 0.1, 1.2 / 1.1 - 1.0, 0.0], index=dates, name="returns"),
)
assert result.stats()["n_days"] == 4
assert result.execution.daily_executions[0].executions == ()
assert result.execution.daily_executions[1].executions[0].price == 10.0
def test_factor_backtest_result_snapshots_both_price_semantics() -> None:
scores = pd.DataFrame({"A": [1.0]}, index=_calendar()[:1])
opens = pd.DataFrame({"A": [10.0, 10.0, 10.0, 10.0]}, index=_calendar())
closes = pd.DataFrame({"A": [10.0, 11.0, 12.0, 13.0]}, index=_calendar())
result = run_factor_backtest_research(
scores,
execution_prices=opens,
valuation_prices=closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
)
opens.iloc[1, 0] = 999.0
closes.iloc[1, 0] = 999.0
assert result.execution_prices.iloc[1, 0] == 10.0
assert result.valuation_prices.iloc[1, 0] == 11.0
def test_factor_backtest_research_requires_matching_daily_calendars() -> None:
scores = pd.DataFrame({"A": [1.0]}, index=_calendar()[:1])
opens = pd.DataFrame({"A": [10.0, 10.0, 10.0, 10.0]}, index=_calendar())
closes = pd.DataFrame({"A": [10.0, 11.0, 12.0]}, index=_calendar()[:3])
with pytest.raises(ValueError, match="matching trading calendars"):
run_factor_backtest_research(
scores,
execution_prices=opens,
valuation_prices=closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
)
def test_factor_backtest_starts_at_first_signal_instead_of_price_warmup() -> None:
"""因子预热行情不能作为空仓日混入研究绩效区间。"""
dates = pd.date_range("2026-01-05", periods=5, freq="B")
scores = pd.DataFrame({"A": [1.0]}, index=dates[2:3])
opens = pd.DataFrame({"A": [1.0, 1.0, 1.0, 10.0, 10.0]}, index=dates)
closes = pd.DataFrame({"A": [100.0, 200.0, 300.0, 11.0, 12.0]}, index=dates)
config = ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
result = run_factor_backtest_research(
scores,
execution_prices=opens,
valuation_prices=closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=config,
)
assert result.nav.index.equals(dates[2:])
pd.testing.assert_series_equal(
result.nav,
pd.Series([1.0, 1.1, 1.2], index=dates[2:], name="nav"),
)
assert result.stats()["n_days"] == 3
def test_factor_backtest_exposes_net_benchmark_metrics() -> None:
dates = _calendar()
scores = pd.DataFrame({"A": [1.0]}, index=dates[:1])
prices = pd.DataFrame({"A": [10.0, 10.0, 11.0, 11.0]}, index=dates)
result = run_factor_backtest_research(
scores,
prices,
prices,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
config=ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
),
)
benchmark = pd.Series([0.0, 0.01, -0.01, 0.0], index=dates)
relative = result.benchmark_stats(benchmark)
assert relative["n_observations"] == len(result.returns)
assert relative["tracking_error"] > 0
def test_factor_backtest_projects_actual_close_weights_from_ledger() -> None:
dates = _calendar()
scores = pd.DataFrame({"A": [1.0], "B": [0.0]}, index=dates[:1])
opens = pd.DataFrame(
{"A": [10.0, 10.0, 10.0, 10.0], "B": [20.0, 20.0, 20.0, 20.0]},
index=dates,
)
closes = pd.DataFrame(
{"A": [10.0, 11.0, 12.0, 12.0], "B": [20.0, 20.0, 20.0, 20.0]},
index=dates,
)
result = run_factor_backtest_research(
scores,
opens,
closes,
top_k=1,
gross_exposure=0.5,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1_000.0,
config=ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
),
)
weights = result.position_weights
cash = result.cash_weights
assert weights.index.equals(result.nav.index)
assert weights.columns.tolist() == ["A", "B"]
assert weights.loc[dates[0]].sum() == 0.0
assert cash.loc[dates[0]] == 1.0
assert weights.loc[dates[1], "A"] == pytest.approx(550.0 / 1_050.0)
pd.testing.assert_series_equal(
weights.sum(axis=1) + cash,
pd.Series(1.0, index=dates),
check_names=False,
)
@@ -0,0 +1,311 @@
"""Fresh synthetic artifact assembly; no reuse or relabelling of real outputs."""
from __future__ import annotations
import hashlib
import json
from dataclasses import replace
from typing import Any
import pandas as pd
import pytest
from quant_engine.artifact import (
EvidenceQualification,
PerformanceEvidenceError,
ResearchRunArtifact,
build_research_run_artifact,
build_backtest_evidence_manifest,
build_performance_evidence,
)
from quant_engine.execution import ExecutionConfig
from quant_engine.factor_contracts import FactorContractError
from quant_engine.governed_pipeline import BacktestContractError
from quant_engine.research_pipeline import run_factor_backtest_research
from quant_engine.retrospective_artifact_contracts import (
build_retrospective_backtest_evidence_manifest,
build_retrospective_performance_evidence,
RetrospectiveBacktestEvidenceManifest,
RetrospectivePerformanceEvidence,
)
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
from test_retrospective_backtest_contracts import run_arguments
from test_retrospective_data_contracts import identify, replace_at
CONTRACT_ERRORS = (FactorContractError, BacktestContractError, PerformanceEvidenceError)
def synthetic_artifact(run: RetrospectiveBacktestRunRef) -> ResearchRunArtifact:
# Artifact-envelope tests, not an end-to-end proof of factor/source authenticity.
# The existing financial methods receive new, in-memory synthetic matrices.
dates = pd.date_range("2018-01-02", periods=4, freq="B")
scores = pd.DataFrame({"SIM0": [2.0, 0.0], "SIM1": [1.0, 3.0]}, index=dates[:2])
opens = pd.DataFrame(
{"SIM0": [10.0, 10.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 20.0, 21.0]}, index=dates
)
closes = pd.DataFrame(
{"SIM0": [10.0, 12.0, 15.0, 15.0], "SIM1": [20.0, 20.0, 18.0, 21.0]}, index=dates
)
result = run_factor_backtest_research(
scores,
opens,
closes,
top_k=1,
execution_price_field="open",
valuation_price_field="close",
initial_cash=1000.0,
config=ExecutionConfig(
commission_bps=0, stamp_tax_bps=0, slippage_bps=0, min_trade_amount=0
),
)
benchmark = pd.Series(
[0.0, 0.01, -0.01, 0.02], index=result.returns.index, name="benchmark_return"
)
return build_research_run_artifact(
result,
run_id=run.run_id,
strategy_id=run.strategy_id,
strategy_name="Synthetic Top 1",
strategy_version=run.strategy_version,
engine_version="0.1.0",
code_revision=run.code_revision,
data_snapshot_id=run.dataset_snapshot_id,
calendar="CN-A",
timezone="Asia/Shanghai",
started_at=run.evaluation_at,
finished_at=run.computed_at,
parameters={"lag_sessions": 1, "top_k": 1},
benchmark_id="synthetic.benchmark",
benchmark_returns=benchmark,
)
def test_manifest_closes_all_nine_existing_tables_without_changing_their_schema() -> None:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
manifest = build_retrospective_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
)
wire = manifest.to_dict()
assert wire["schema_version"] == "2.0.0"
assert wire["artifact_schema_version"] == "1.1.0"
assert wire["run_id"] == run.run_id
assert wire["usage"] == "retrospective_research"
assert wire["historical_availability"] == "not_established"
assert wire["execution_validation"] == "not_validated"
assert wire["decision_eligible"] is False
assert manifest.manifest_id.startswith("rhbacktestevidencev2:sha256:")
assert len({table.logical_name for item in manifest.evidence for table in item.tables}) == 9
def test_performance_v2_keeps_existing_metric_methods_and_binds_all_upstreams() -> None:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
manifest = build_retrospective_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
)
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
wire = evidence.to_dict()
assert wire["schema_version"] == "researchhub.performance-evidence.v2"
assert wire["methodology_id"] == "researchhub.quant-performance-methodology.v1"
assert wire["metric_schema_id"] == "researchhub.quant-performance-metrics.v1"
assert wire["research_artifact_schema_version"] == "1.1.0"
assert wire["backtest_run_ref_id"] == run.run_id
assert wire["backtest_evidence_manifest_id"] == manifest.manifest_id
assert wire["historical_availability"] == "not_established"
assert wire["usage"] == "retrospective_research"
assert wire["start_date"] == "2018-01-02"
assert wire["end_date"] == "2018-01-05"
assert wire["artifact_available_at"] == "2026-09-08T01:11:00Z"
assert evidence.run_id == run.run_id
assert evidence.performance_evidence_id.startswith("rhperformancev2:sha256:")
for metric in evidence.metrics:
if metric.value is not None:
assert metric.value == artifact.performance.iloc[0][metric.source_column]
assert (
RetrospectivePerformanceEvidence.from_json(
evidence.to_json(), artifact=artifact, run_ref=run, evidence_manifest=manifest
)
== evidence
)
assert (
RetrospectiveBacktestEvidenceManifest.from_json(
manifest.to_json(), artifact=artifact, backtest_run_ref=run
)
== manifest
)
@pytest.mark.parametrize(
("path", "value"),
[
("schema_version", "1.0.0"),
("run_id", "rhbacktestrunv2:sha256:" + "0" * 64),
("profile", "offline_research_v1"),
("historical_availability", "established"),
("decision_eligible", True),
("execution_validation", "validated"),
("evidence_scope", "real_data"),
("artifact_available_at", "2026-09-08T01:09:00Z"),
("artifact_schema_version", "2.0.0"),
("qualification", "legacy_exploratory"),
("evidence_digest", "sha256:" + "0" * 64),
("evidence.0.tables.0.row_count", True),
("evidence.0.tables.0.content_digest", "sha256:" + "0" * 64),
("run_reference.value.factor_set_id", "rhfactorsetv2:sha256:" + "0" * 64),
],
)
def test_manifest_rejects_reidentified_claims_without_actual_table_closure(
path: str, value: Any
) -> None:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
row = build_retrospective_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
).to_dict()
replace_at(row, path, value)
identify(row, "manifest_id", "rhbacktestevidencev2:")
with pytest.raises(CONTRACT_ERRORS):
RetrospectiveBacktestEvidenceManifest.from_dict(
row, artifact=artifact, backtest_run_ref=run
)
@pytest.mark.parametrize(
("table", "column", "value"),
[
("run", "data_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
("run", "config_hash", "0" * 64),
("run", "code_revision", "0" * 40),
("run", "started_at", "2018-01-02T07:00:00Z"),
("run", "finished_at", "2026-09-08T01:12:00Z"),
("signals", "asset_id", "/private/data.csv"),
("nav", "run_id", "old.run"),
("performance", "run_id", "old.run"),
],
)
def test_manifest_rejects_artifact_identity_time_or_private_data_mismatch(
table: str, column: str, value: Any
) -> None:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
frame = getattr(artifact, table)
frame.loc[frame.index[0], column] = value
forged = replace(artifact, **{"_" + table: frame})
with pytest.raises(CONTRACT_ERRORS):
build_retrospective_backtest_evidence_manifest(
run, forged, artifact_available_at="2026-09-08T01:13:00Z"
)
def test_each_table_is_reconciled_and_legacy_apis_cannot_accept_v2() -> None:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
manifest = build_retrospective_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
)
for item in manifest.evidence:
for table in item.tables:
with pytest.raises(CONTRACT_ERRORS):
build_retrospective_backtest_evidence_manifest(
run,
artifact,
artifact_available_at="2026-09-08T01:11:00Z",
expected_table_digests={table.logical_name: "sha256:" + "0" * 64},
)
with pytest.raises(CONTRACT_ERRORS):
build_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
)
with pytest.raises(CONTRACT_ERRORS):
build_performance_evidence(artifact, run, manifest)
with pytest.raises(CONTRACT_ERRORS):
build_retrospective_backtest_evidence_manifest(
run,
artifact,
artifact_available_at="2026-09-08T01:11:00Z",
qualification=EvidenceQualification.LEGACY_EXPLORATORY,
)
def seal_performance(row: dict[str, Any]) -> None:
def sha(document: Any) -> str:
return (
"sha256:"
+ hashlib.sha256(
json.dumps(
document,
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
).encode()
).hexdigest()
)
row.pop("document_sha256", None)
row.pop("performance_evidence_id", None)
row["performance_evidence_id"] = "rhperformancev2:" + sha(row)
row["document_sha256"] = sha(row)
@pytest.mark.parametrize(
("path", "value"),
[
("schema_version", "researchhub.performance-evidence.v1"),
("scope", "live"),
("historical_availability", "established"),
("decision_eligible", True),
("evidence_scope", "real_data"),
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
("backtest_run_ref_document_sha256", "sha256:" + "0" * 64),
("backtest_evidence_manifest_id", "rhbacktestevidencev2:sha256:" + "0" * 64),
("methodology.periods_per_year", 365),
("metric_schema_id", "new.metric"),
("metrics.0.value", 0.0),
("metrics.0.nullable", True),
("start_date", "2017-01-01"),
("artifact_available_at", "2018-01-02T07:00:00Z"),
],
)
def test_performance_never_accepts_reidentified_changed_facts(path: str, value: Any) -> None:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
manifest = build_retrospective_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
)
row = build_retrospective_performance_evidence(artifact, run, manifest).to_dict()
replace_at(row, path, value)
seal_performance(row)
with pytest.raises(CONTRACT_ERRORS):
RetrospectivePerformanceEvidence.from_dict(
row, artifact=artifact, run_ref=run, evidence_manifest=manifest
)
def test_performance_has_immutable_finite_canonical_payload_and_current_tables() -> None:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
manifest = build_retrospective_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
)
evidence = build_retrospective_performance_evidence(artifact, run, manifest)
assert evidence.document_sha256.startswith("sha256:")
exported = evidence.to_dict()
exported["metrics"][0]["value"] = 9.0
assert evidence.to_dict()["metrics"][0]["value"] != 9.0
for data in (
evidence.to_json() + "\n",
'{"schema_version":"x",' + evidence.to_json()[1:],
"null",
"{bad",
):
with pytest.raises(CONTRACT_ERRORS):
RetrospectivePerformanceEvidence.from_json(
data, artifact=artifact, run_ref=run, evidence_manifest=manifest
)
frame = artifact.performance
frame.loc[0, "total_ret"] = 0.0
forged = replace(artifact, _performance=frame)
with pytest.raises(CONTRACT_ERRORS):
build_retrospective_performance_evidence(forged, run, manifest)
@@ -0,0 +1,157 @@
"""Offline synthetic v2 backtest evidence and replay boundaries."""
from __future__ import annotations
from typing import Any
import pytest
from quant_engine.factor_contracts import FactorContractError, PayloadValidation
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
from quant_engine.retrospective_factor_contracts import RetrospectiveFactorSetRef
from test_retrospective_data_contracts import digest, identify, replace_at
from test_retrospective_factor_contracts import factor_arguments, decoding_arguments
def run_arguments() -> dict[str, Any]:
arguments = factor_arguments()
factor = RetrospectiveFactorSetRef.create(**arguments)
view = next(iter(arguments["foundation"].views.values()))
return {
"dataset_snapshot": arguments["dataset_snapshot"],
"foundation": arguments["foundation"],
"factor_set": factor,
"universe_digest": digest({"synthetic_universe": 2}),
"trading_calendar_revision_ids": view.trading_calendar_revision_ids,
"corporate_action_revision_ids": view.corporate_action_revision_ids,
"strategy_id": "synthetic.top1",
"strategy_version": "1.0.0",
"strategy_digest": digest({"synthetic_strategy": "top1"}),
"execution_model_version": "1.0.0",
"execution_model_digest": digest({"synthetic_execution": 1}),
"cost_model_version": "1.0.0",
"cost_model_digest": digest({"synthetic_cost": 1}),
"random_seed": 7,
"code_revision": "d" * 40,
"environment_lock_digest": digest({"synthetic_lock": 1}),
"configuration_digest": digest({"lag_sessions": 1, "top_k": 1}),
"evaluation_at": "2026-09-08T01:09:00Z",
"computed_at": "2026-09-08T01:10:00Z",
}
def run_context(arguments: dict[str, Any]) -> dict[str, Any]:
return {key: arguments[key] for key in ("dataset_snapshot", "foundation", "factor_set")}
def test_run_identity_closes_observation_inputs_and_preserves_configurations() -> None:
arguments = run_arguments()
run = RetrospectiveBacktestRunRef.create(**arguments)
document = run.to_dict()
assert document["schema_version"] == "2.0.0"
assert run.run_id.startswith("rhbacktestrunv2:sha256:")
assert run.dataset_snapshot_id == arguments["dataset_snapshot"].snapshot_id
assert run.foundation_id == arguments["foundation"].foundation_id
assert run.factor_set_id == arguments["factor_set"].factor_set_id
assert document["usage"] == "retrospective_research"
assert document["historical_availability"] == "not_established"
assert document["decision_eligible"] is False
assert document["execution_validation"] == "not_validated"
assert document["observation_cutoff"] == arguments["foundation"].observation_cutoff
assert document["replay_attempt"] == 0
assert RetrospectiveBacktestRunRef.from_json(run.to_json(), **run_context(arguments)) == run
@pytest.mark.parametrize(
("path", "value"),
[
("schema_version", "1.0.0"),
("schema_version", "2.1.0"),
("historical_availability", "established"),
("usage", "as_available"),
("execution_validation", "validated"),
("decision_eligible", True),
("decision_eligible", 0),
("dataset_content_digest", "sha256:" + "0" * 64),
("foundation_digest", "sha256:" + "0" * 64),
("factor_set_digest", "sha256:" + "0" * 64),
("factor_output_content_digest", "sha256:" + "0" * 64),
("observation_cutoff", "2018-01-02T07:00:00Z"),
("evidence_scope", "real_data"),
("trading_calendar_revision_ids", []),
("corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
("evaluation_at", "2026-09-08T01:07:00Z"),
("computed_at", "2026-09-08T01:08:00Z"),
("computed_at", "2026-09-08T01:10:00.0000001Z"),
("random_seed", True),
("strategy_version", "latest"),
("configuration_digest", "../private/a"),
("code_revision", "unknown"),
("replay_attempt", 1),
("replay_reason", "retry"),
("replay_spec_digest", "sha256:" + "0" * 64),
("replay_ancestor_run_ids", ["rhbacktestrunv2:sha256:" + "0" * 64]),
],
)
def test_reidentified_run_must_match_exact_context(path: str, value: Any) -> None:
arguments = run_arguments()
row = RetrospectiveBacktestRunRef.create(**arguments).to_dict()
replace_at(row, path, value)
identify(row, "run_id", "rhbacktestrunv2:")
with pytest.raises(FactorContractError):
RetrospectiveBacktestRunRef.from_dict(row, **run_context(arguments))
def test_replay_keeps_input_spec_but_requires_new_actual_attempt_times() -> None:
arguments = run_arguments()
root = RetrospectiveBacktestRunRef.create(**arguments)
replay_args = {
**arguments,
"parent": root,
"replay_reason": "synthetic.retry",
"replay_attempt": 1,
"evaluation_at": "2026-09-08T01:12:00Z",
"computed_at": "2026-09-08T01:13:00Z",
}
replay = RetrospectiveBacktestRunRef.create(**replay_args)
assert replay.replay_spec_digest == root.replay_spec_digest
assert replay.run_id != root.run_id
assert replay.replay_ancestor_run_ids == (root.run_id,)
assert replay.evaluation_at != root.evaluation_at
assert (
RetrospectiveBacktestRunRef.from_json(
replay.to_json(), **run_context(arguments), parent=root
)
== replay
)
for changes in (
{"random_seed": 9},
{"configuration_digest": digest({"different_configuration": 1})},
{"evaluation_at": root.evaluation_at},
{"replay_attempt": 2},
{"replay_reason": None},
{"parent": None},
):
with pytest.raises(FactorContractError):
RetrospectiveBacktestRunRef.create(**{**replay_args, **changes})
def test_reference_only_factors_can_be_read_but_not_used_to_create_new_runs() -> None:
arguments = run_arguments()
run = RetrospectiveBacktestRunRef.create(**arguments)
factor = arguments["factor_set"]
reference = RetrospectiveFactorSetRef.from_dict(
factor.to_dict(),
definitions=factor._definitions,
dataset_snapshot=arguments["dataset_snapshot"],
foundation=arguments["foundation"],
)
with pytest.raises(FactorContractError):
RetrospectiveBacktestRunRef.create(**{**arguments, "factor_set": reference})
restored = RetrospectiveBacktestRunRef.from_dict(
run.to_dict(), **{**run_context(arguments), "factor_set": reference}
)
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
with pytest.raises(FactorContractError):
restored.require_inputs_revalidated()
run.require_inputs_revalidated()
@@ -0,0 +1,88 @@
"""Frozen synthetic interoperability vector, not end-to-end source provenance."""
from __future__ import annotations
import ast
import json
from pathlib import Path
from typing import Any
from quant_engine.artifact import _evidence_frame_records
from quant_engine.retrospective_artifact_contracts import build_retrospective_performance_evidence
from quant_engine.retrospective_portfolio_risk_contracts import (
assess_retrospective_portfolio_risk,
)
from test_retrospective_factor_contracts import factor_arguments
from test_retrospective_portfolio_risk_contracts import portfolio_arguments, risk_arguments
ROOT = Path(__file__).resolve().parents[1]
VECTOR = ROOT / "tests" / "fixtures" / "retrospective-computation-v2.golden.json"
def build_vector() -> dict[str, Any]:
portfolio = portfolio_arguments()
risk = risk_arguments(portfolio)
run = portfolio["backtest_run_ref"]
manifest = portfolio["manifest"]
artifact = manifest._artifact
factor = factor_arguments()
return {
"fixture_kind": "synthetic_retrospective_contract_vector",
"artifact_data_provenance": "envelope_test_only_not_end_to_end",
"source_authenticity": "not_established",
"factor_definitions": [item.to_dict() for item in factor["definitions"]],
"dataset_chunks": factor["dataset_chunks"],
"resolved_view_schema": {"synthetic_schema": "neutral_close_v2"},
"factor_output_schema": json.loads(factor["output_schema_bytes"]),
"factor_output_records": json.loads(factor["output_content_bytes"]),
"factor_set": run._factor_set.to_dict(),
"backtest_run_ref": run.to_dict(),
"artifact_tables": {
name: _evidence_frame_records(frame, name)
for name, frame in artifact.table_frames().items()
},
"backtest_evidence_manifest": manifest.to_dict(),
"performance_evidence": build_retrospective_performance_evidence(
artifact, run, manifest
).to_dict(),
"portfolio_target": portfolio["target"].to_dict(),
"portfolio_decision": risk["portfolio_decision"].to_dict(),
"risk_assessment": assess_retrospective_portfolio_risk(**risk).to_dict(),
"covariance_matrix": risk["covariance"].covariance.to_dict(),
}
def test_synthetic_vector_matches_all_current_owner_serializers() -> None:
expected = VECTOR.read_text(encoding="utf-8")
actual = (
json.dumps(build_vector(), ensure_ascii=False, sort_keys=True, indent=2, allow_nan=False)
+ "\n"
)
assert actual == expected
def test_v2_modules_do_not_import_data_owners_publishers_or_execution_authority() -> None:
modules = sorted((ROOT / "src" / "quant_engine").glob("retrospective_*_contracts.py"))
assert len(modules) == 5
for path in modules:
tree = ast.parse(path.read_text(encoding="utf-8"))
imports = {
alias.name
for node in ast.walk(tree)
if isinstance(node, ast.Import)
for alias in node.names
} | {node.module or "" for node in ast.walk(tree) if isinstance(node, ast.ImportFrom)}
assert not any(
name.startswith(("research_results", "research_platform", "edb_data_core"))
for name in imports
)
called = {
node.func.id
for node in ast.walk(tree)
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
}
assert not called & {
"create_paper_order_intent",
"run_governed_factor_slice",
"evaluate_portfolio_risk",
}
+636
View File
@@ -0,0 +1,636 @@
"""QE-owned decoding tests against RP's synthetic public v2 golden vectors."""
from __future__ import annotations
import hashlib
import json
from copy import deepcopy
from dataclasses import FrozenInstanceError
from pathlib import Path
from typing import Any
import pytest
from quant_engine.factor_contracts import (
DataFoundationEnvelope,
DatasetSnapshotEnvelope,
FactorContractError,
canonical_json_bytes,
)
from quant_engine.retrospective_data_contracts import (
RetrospectiveFoundationEnvelope,
RetrospectiveSnapshotEnvelope,
)
FIXTURES = Path(__file__).parent / "fixtures"
def golden(kind: str) -> dict[str, Any]:
return json.loads((FIXTURES / f"retrospective-{kind}-v2.golden.json").read_text())
def digest(value: Any) -> str:
return "sha256:" + hashlib.sha256(canonical_json_bytes(value)).hexdigest()
def identify(value: dict[str, Any], field: str, prefix: str) -> None:
value[field] = prefix + digest({key: item for key, item in value.items() if key != field})
def records() -> list[dict[str, Any]]:
return [
{
"effective_time": "2018-01-02T07:00:00Z",
"instrument_id": "rhinstrument:" + "1" * 32,
"metric": "close",
"value": "101.25",
},
{
"effective_time": "2018-01-02T07:00:00Z",
"instrument_id": "rhinstrument:" + "2" * 32,
"metric": "close",
"value": "87.50",
},
]
def bind_records(source: dict[str, Any], chunks: list[list[dict[str, Any]]]) -> None:
def records_digest(rows: list[dict[str, Any]]) -> str:
data = b"[" + b",".join(sorted(canonical_json_bytes(row) for row in rows)) + b"]"
return "sha256:" + hashlib.sha256(data).hexdigest()
manifest = {
"record_count": sum(len(rows) for rows in chunks),
"chunks": [
{
"chunk_index": index,
"content_digest": records_digest(rows),
"record_count": len(rows),
}
for index, rows in enumerate(chunks)
],
}
source["descriptor"]["content"].update(
{
"record_count": manifest["record_count"],
"logical_manifest": manifest,
"manifest_digest": digest(manifest),
"content_digest": records_digest([row for chunk in chunks for row in chunk]),
}
)
source["descriptor"]["observation_manifest"]["batches"] = [
{
**chunk,
"observation_kind": "observed_by",
"observed_by": "2026-09-08T01:00:00Z",
"evidence_digest": digest({"synthetic_receipt": index}),
}
for index, chunk in enumerate(manifest["chunks"])
]
identify(source, "snapshot_id", "rhdsv2:")
def replace_at(source: dict[str, Any], path: str, value: Any) -> None:
target: Any = source
keys = path.split(".")
for key in keys[:-1]:
target = target[int(key)] if isinstance(target, list) else target[key]
target[int(keys[-1]) if isinstance(target, list) else keys[-1]] = value
COLLECTIONS = (
(
"instrument_routes",
"route_revision_id",
"rhroutev2:",
"instrument_route",
"instrument_route_revision_ids",
),
(
"trading_calendar_revisions",
"calendar_revision_id",
"rhcalv2:",
"trading_calendar",
"trading_calendar_revision_ids",
),
(
"corporate_action_revisions",
"action_revision_id",
"rhcav2:",
"corporate_action",
"corporate_action_revision_ids",
),
)
def seal_foundation(source: dict[str, Any], *, rebuild_lineage: bool = True) -> None:
lineage = []
for name, key, prefix, kind, view_key in COLLECTIONS:
replacements = {}
for row in sorted(source[name], key=lambda row: row["observation_sequence"]):
old = row[key]
if "supersedes_observation_id" in row:
row["supersedes_observation_id"] = replacements.get(
row["supersedes_observation_id"], row["supersedes_observation_id"]
)
identify(row, key, prefix)
replacements[old] = row[key]
lineage.append(
{
"revision_kind": kind,
"revision_id": row[key],
**{
field: row[field]
for field in (
"observation_sequence",
"observed_by",
"earliest_external_knowledge",
"history_completeness",
"evidence_digest",
"supersedes_observation_id",
)
if field in row
},
}
)
for view in source["standardized_views"]:
view[view_key] = [replacements.get(item, item) for item in view[view_key]]
if rebuild_lineage:
source["observation_lineage"] = lineage
for view in source["standardized_views"]:
identify(view, "view_ref_id", "rhviewrefv2:")
identify(source, "foundation_id", "rhdfv2:")
def parse_foundation(source: dict[str, Any]) -> RetrospectiveFoundationEnvelope:
return RetrospectiveFoundationEnvelope.from_dict(
source, snapshot=RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
)
def test_public_snapshot_golden_is_an_explicit_observation_contract() -> None:
source = golden("dataset-snapshot")
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
assert snapshot.to_dict() == source
assert snapshot.snapshot_id == source["snapshot_id"]
assert snapshot.observation_cutoff == "2026-09-08T01:01:00Z"
assert snapshot.evidence_scope == "synthetic_fixture"
assert snapshot.earliest_external_knowledge == {"status": "unknown"}
assert not hasattr(snapshot, "pit_cutoff")
assert not hasattr(snapshot, "knowledge_time")
snapshot.require_qualified()
assert RetrospectiveSnapshotEnvelope.from_json(snapshot.to_json()) == snapshot
def test_public_foundation_golden_binds_exact_typed_snapshot() -> None:
source = golden("data-foundation")
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
foundation = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
assert foundation.to_dict() == source
assert foundation.dataset_snapshot_id == snapshot.snapshot_id
assert foundation.observation_cutoff == snapshot.observation_cutoff
assert foundation.evidence_scope == snapshot.evidence_scope
assert foundation.real_data_validation_status == "not_validated"
assert not hasattr(foundation, "pit_cutoff")
assert (
RetrospectiveFoundationEnvelope.from_json(foundation.to_json(), snapshot=snapshot)
== foundation
)
def test_empty_logical_dimension_is_rejected_after_rebinding_all_bytes() -> None:
source = golden("dataset-snapshot")
rows = records()
rows[0]["instrument_id"] = ""
bind_records(source, [rows])
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
with pytest.raises(FactorContractError, match="dimension"):
snapshot.verify_materialized_records([rows])
def test_boolean_lineage_sequence_does_not_equal_integer_one() -> None:
source = golden("data-foundation")
source["observation_lineage"][0]["observation_sequence"] = True
identify(source, "foundation_id", "rhdfv2:")
with pytest.raises(FactorContractError):
parse_foundation(source)
@pytest.mark.parametrize(
("path", "value"),
[
("schema_version", "1.0.0"),
("schema_version", "2.1.0"),
("descriptor.time_semantics.pit_cutoff", "2018-01-02T07:00:00Z"),
("descriptor.time_semantics.knowledge_time", {"start_inclusive": "2018-01-02T07:00:00Z"}),
("descriptor.time_semantics.historical_availability", "established"),
("descriptor.time_semantics.observation_cutoff", "2018-01-02T07:00:00Z"),
("descriptor.time_semantics.observation_cutoff", "2026-09-08T01:01:00.0000001Z"),
("descriptor.time_semantics.observation_cutoff", "2026-09-08T09:01:00+08:00"),
(
"descriptor.time_semantics.earliest_external_knowledge",
{"status": "unknown", "evidence_digest": "sha256:" + "0" * 64},
),
("descriptor.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
("descriptor.published_at", "2026-02-30T00:00:00Z"),
("descriptor.published_at", "2026-09-08T01:00:00Z"),
("descriptor.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
("descriptor.qualification.usage", "as_available"),
("descriptor.qualification.policy_version", "1.0.0"),
("descriptor.quality.checks.0.severity", "advisory"),
("descriptor.quality.checks.0.status", "failed"),
("descriptor.quality.checks.0.check_id", "schema_conformance"),
("descriptor.quality.status", "failed"),
("descriptor.content.record_count", True),
("descriptor.content.record_count", 9007199254740992),
("descriptor.content.record_count", 2.0),
("descriptor.content.content_digest", "bad"),
("descriptor.content.manifest_digest", "sha256:" + "0" * 64),
("descriptor.observation_manifest.batches", []),
("descriptor.observation_manifest.batches.0.record_count", 1),
("descriptor.observation_manifest.batches.0.chunk_index", True),
("descriptor.observation_manifest.batches.0.observed_by", "2026-09-08T01:02:00Z"),
("descriptor.observation_manifest.batches.0.observation_kind", "first_published_at"),
("descriptor.lineage.transformation.id", "rhtransform:private"),
("descriptor.dataset.dimensions", ["instrument_id", "knowledge_time"]),
("descriptor.dataset.dataset_id", "rhdataset:macroeconomic:" + "4" * 32),
],
)
def test_snapshot_rejects_reidentified_invalid_declarations(path: str, value: Any) -> None:
source = golden("dataset-snapshot")
replace_at(source, path, value)
# Noncanonical numbers are rejected before identity formation.
if type(value) is not float and value != 9007199254740992:
identify(source, "snapshot_id", "rhdsv2:")
with pytest.raises(FactorContractError):
RetrospectiveSnapshotEnvelope.from_dict(source)
def test_snapshot_materialized_records_bind_the_public_golden_and_chunks() -> None:
source = golden("dataset-snapshot")
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
snapshot.verify_materialized_records([records()])
snapshot.verify_materialized_records([list(reversed(records()))])
chunks = [[records()[0]], [records()[1]]]
bind_records(source, chunks)
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records(chunks)
with pytest.raises(FactorContractError):
snapshot.verify_materialized_records(chunks)
rows = records()
rows[0]["value"] = "0"
with pytest.raises(FactorContractError):
snapshot.verify_materialized_records([rows])
@pytest.mark.parametrize(
"mutation", ["duplicate", "legacy", "range", "location", "missing_effective"]
)
def test_materialized_bad_records_rejected_even_with_matching_digests(mutation: str) -> None:
source = golden("dataset-snapshot")
rows = records()
if mutation == "duplicate":
rows.append(deepcopy(rows[0]))
elif mutation == "legacy":
rows[0]["knowledge_time"] = "2026-09-08T01:00:00Z"
elif mutation == "range":
rows[0]["effective_time"] = "2018-01-01T07:00:00Z"
elif mutation == "location":
rows[0]["value"] = "/private/records.csv"
else:
del rows[0]["effective_time"]
bind_records(source, [rows])
with pytest.raises(FactorContractError):
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
def test_unknown_or_evidenced_knowledge_never_changes_usage() -> None:
source = golden("dataset-snapshot")
source["descriptor"]["time_semantics"]["earliest_external_knowledge"] = {
"status": "evidenced",
"range": {
"start_inclusive": "2018-01-02T07:00:00Z",
"end_inclusive": "2018-01-02T07:00:00Z",
},
"evidence_digest": digest({"synthetic_earliest": True}),
}
identify(source, "snapshot_id", "rhdsv2:")
parsed = RetrospectiveSnapshotEnvelope.from_dict(source)
assert (
parsed.to_dict()["descriptor"]["time_semantics"]["historical_availability"]
== "not_established"
)
source["descriptor"]["time_semantics"]["earliest_external_knowledge"]["range"][
"end_inclusive"
] = "2026-09-08T01:01:00Z"
identify(source, "snapshot_id", "rhdsv2:")
with pytest.raises(FactorContractError):
RetrospectiveSnapshotEnvelope.from_dict(source)
def test_rejected_snapshot_remains_readable_but_cannot_support_foundation() -> None:
source = golden("dataset-snapshot")
source["descriptor"]["qualification"]["status"] = "rejected"
identify(source, "snapshot_id", "rhdsv2:")
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
with pytest.raises(FactorContractError):
snapshot.require_qualified()
foundation = golden("data-foundation")
foundation["dataset_snapshot_id"] = snapshot.snapshot_id
for view in foundation["standardized_views"]:
view["dataset_snapshot_id"] = snapshot.snapshot_id
seal_foundation(foundation)
with pytest.raises(FactorContractError):
RetrospectiveFoundationEnvelope.from_dict(foundation, snapshot=snapshot)
@pytest.mark.parametrize(
("path", "value"),
[
("schema_version", "1.0.0"),
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
("observation_cutoff", "2026-09-08T01:00:00Z"),
("published_at", "2026-09-08T01:03:00Z"),
("usage", "paper_trading"),
("historical_availability", "established"),
("instrument_routes.0.observation_sequence", 2),
("instrument_routes.0.supersedes_observation_id", "rhroutev2:sha256:" + "0" * 64),
("instrument_routes.0.observed_by", "2026-09-08T01:02:00Z"),
("instrument_routes.0.history_completeness", "complete"),
(
"instrument_routes.0.earliest_external_knowledge",
{"status": "unknown", "earliest_at": "2018-01-01T00:00:00Z"},
),
("instrument_routes.0.instrument_type", "index"),
("instrument_routes.0.symbol", "WIND.TEST"),
("instrument_routes.0.symbol", "A" * 33),
("instrument_routes.0.calendar_id", "rhcalendar:" + "0" * 32),
("trading_calendar_revisions.0.status", "closed"),
("trading_calendar_revisions.0.sessions", []),
("trading_calendar_revisions.0.sessions.0.closes_at", "2018-01-02T00:00:00Z"),
("trading_calendar_revisions.0.session_date", "2018-02-30"),
("standardized_views.0.instrument_route_revision_ids", []),
("standardized_views.0.trading_calendar_revision_ids", []),
("standardized_views.0.corporate_action_revision_ids", ["rhcav2:sha256:" + "0" * 64]),
("standardized_views.0.observation_cutoff", "2018-01-02T07:00:00Z"),
("standardized_views.0.available_at", "2026-09-08T01:02:00Z"),
("standardized_views.0.available_at", "2026-09-08T01:06:00Z"),
("standardized_views.0.usage", "as_available"),
("corporate_action_coverage", []),
("corporate_action_coverage.0.instrument_id", "rhinstrument:" + "0" * 32),
("corporate_action_coverage.0.effective_time.end_inclusive", "2018-01-01T07:00:00Z"),
("corporate_action_coverage.0.effective_time.start_inclusive", "2018-01-02T07:00:01Z"),
("corporate_action_coverage.0.observed_by", "2026-09-08T01:02:00Z"),
("corporate_action_coverage.0.evidence_digests", []),
("readiness.evidence_scope", "real_data"),
("readiness.contract_validation.evidence_digests", []),
(
"readiness.real_data_validation",
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
),
(
"readiness.production_validation",
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
),
(
"readiness.live_validation",
{"status": "validated", "evidence_digests": ["sha256:" + "0" * 64]},
),
],
)
def test_foundation_rejects_semantic_forgery_after_reidentification(path: str, value: Any) -> None:
source = golden("data-foundation")
replace_at(source, path, value)
seal_foundation(source)
with pytest.raises(FactorContractError):
parse_foundation(source)
def test_v1_and_v2_never_coerce_each_other() -> None:
with pytest.raises(FactorContractError):
DatasetSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
with pytest.raises(FactorContractError):
DataFoundationEnvelope.from_dict(golden("data-foundation"))
old = json.loads((FIXTURES / "factor-contracts-v1.golden.json").read_text())
with pytest.raises(FactorContractError):
RetrospectiveSnapshotEnvelope.from_dict(old["dataset_snapshot"])
with pytest.raises(FactorContractError):
parse_foundation(old["data_foundation"])
with pytest.raises(FactorContractError):
RetrospectiveFoundationEnvelope.from_dict(
golden("data-foundation"),
snapshot=DatasetSnapshotEnvelope.from_dict(old["dataset_snapshot"]),
)
def test_deep_immutability_and_strict_canonical_json() -> None:
source = golden("dataset-snapshot")
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
source["descriptor"]["quality"]["status"] = "failed"
snapshot.to_dict()["descriptor"]["qualification"]["status"] = "rejected"
snapshot.require_qualified()
with pytest.raises(FrozenInstanceError):
snapshot._payload = {}
with pytest.raises(TypeError):
snapshot.earliest_external_knowledge["status"] = "evidenced"
foundation = parse_foundation(golden("data-foundation"))
with pytest.raises(TypeError):
foundation.views["new"] = next(iter(foundation.views.values()))
for decoder, document in (
(RetrospectiveSnapshotEnvelope.from_json, golden("dataset-snapshot")),
(
lambda value: RetrospectiveFoundationEnvelope.from_json(value, snapshot=snapshot),
golden("data-foundation"),
),
):
wire = canonical_json_bytes(document)
with pytest.raises(FactorContractError):
decoder(wire + b"\n")
with pytest.raises(FactorContractError):
decoder(b'{"schema_version":"2.0.0",' + wire[1:])
def test_macro_period_is_not_inferred_as_an_effective_instant() -> None:
source = golden("dataset-snapshot")
source["descriptor"]["dataset"].update(
dataset_id="rhdataset:macroeconomic:" + "4" * 32,
dataset_kind="macroeconomic",
dimensions=["series_id", "observation_period"],
)
rows = [
{
"series_id": "cpi",
"observation_period": "2018-01",
"effective_time": "2018-01-02T07:00:00Z",
"value": "2.1",
}
]
bind_records(source, [rows])
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
del rows[0]["effective_time"]
bind_records(source, [rows])
with pytest.raises(FactorContractError):
RetrospectiveSnapshotEnvelope.from_dict(source).verify_materialized_records([rows])
def with_successor() -> dict[str, Any]:
source = golden("data-foundation")
previous = source["instrument_routes"][0]
successor = deepcopy(previous)
successor.update(
observation_sequence=2,
observed_by="2026-09-08T01:00:30Z",
symbol="SIM0B",
supersedes_observation_id=previous["route_revision_id"],
)
identify(successor, "route_revision_id", "rhroutev2:")
source["instrument_routes"].append(successor)
source["standardized_views"][0]["instrument_route_revision_ids"].append(
successor["route_revision_id"]
)
seal_foundation(source)
return source
def test_retained_successor_requires_parent_time_and_view_ancestry() -> None:
source = with_successor()
parsed = parse_foundation(source)
assert parsed.foundation_id == source["foundation_id"]
assert parsed.published_at.isoformat() == "2026-09-08T01:05:00+00:00"
assert parsed.contract_evidence_digests
assert len(next(iter(parsed.views.values())).instrument_route_revision_ids) == 3
for mutation in (
"missing_parent",
"equal_time",
"omitted_ancestor",
"duplicate_sequence",
"wrong_lineage",
):
forged = deepcopy(source)
if mutation == "missing_parent":
forged["instrument_routes"][-1]["supersedes_observation_id"] = (
"rhroutev2:sha256:" + "0" * 64
)
elif mutation == "equal_time":
forged["instrument_routes"][-1]["observed_by"] = forged["instrument_routes"][0][
"observed_by"
]
elif mutation == "omitted_ancestor":
forged["standardized_views"][0]["instrument_route_revision_ids"].remove(
forged["instrument_routes"][0]["route_revision_id"]
)
elif mutation == "duplicate_sequence":
forged["instrument_routes"][-1]["observation_sequence"] = 1
else:
forged["observation_lineage"][0]["evidence_digest"] = "sha256:" + "0" * 64
seal_foundation(forged, rebuild_lineage=mutation != "wrong_lineage")
with pytest.raises(FactorContractError):
parse_foundation(forged)
def test_index_and_closed_calendar_are_explicit_non_execution_data() -> None:
source = golden("data-foundation")
source["instrument_routes"][0].update(asset_class="index", instrument_type="index")
source["trading_calendar_revisions"][0].update(status="closed", sessions=[])
seal_foundation(source)
parsed = parse_foundation(source)
assert parsed.to_dict()["instrument_routes"][0]["instrument_type"] == "index"
assert parsed.to_dict()["readiness"]["live_validation"]["status"] == "not_validated"
def test_real_scope_is_still_a_declaration_with_separate_evidence_and_coverage() -> None:
snapshot_source = golden("dataset-snapshot")
snapshot_source["evidence_scope"] = "real_data"
identify(snapshot_source, "snapshot_id", "rhdsv2:")
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_source)
source = golden("data-foundation")
source["dataset_snapshot_id"] = snapshot.snapshot_id
for view in source["standardized_views"]:
view["dataset_snapshot_id"] = snapshot.snapshot_id
source["readiness"]["evidence_scope"] = "real_data"
source["readiness"]["real_data_validation"] = {
"status": "validated",
"evidence_digests": [digest({"synthetic_real_claim_test": 1})],
}
seal_foundation(source)
parsed = RetrospectiveFoundationEnvelope.from_dict(source, snapshot=snapshot)
assert parsed.real_data_validation_status == "validated" # NOT actual real-data evidence.
for mutation in ("coverage", "reuse"):
forged = deepcopy(source)
if mutation == "coverage":
forged["corporate_action_coverage"][0].update(
status="not_validated", evidence_digests=[]
)
else:
forged["readiness"]["real_data_validation"] = deepcopy(
forged["readiness"]["contract_validation"]
)
seal_foundation(forged)
with pytest.raises(FactorContractError):
RetrospectiveFoundationEnvelope.from_dict(forged, snapshot=snapshot)
synthetic = golden("data-foundation")
synthetic["corporate_action_coverage"][0].update(status="not_validated", evidence_digests=[])
seal_foundation(synthetic)
assert parse_foundation(synthetic).real_data_validation_status == "not_validated"
def test_action_must_belong_to_view_selected_instrument() -> None:
source = golden("data-foundation")
route = source["instrument_routes"][0]
action = {
"action_id": "rhaction:" + "7" * 32,
"instrument_id": route["instrument_id"],
"observation_sequence": 1,
"observed_by": route["observed_by"],
"earliest_external_knowledge": {
"status": "evidenced",
"earliest_at": "2018-01-01T00:00:00Z",
"evidence_digest": digest({"synthetic_action_earliest": 1}),
},
"history_completeness": "not_established",
"evidence_digest": digest({"synthetic_action": 1}),
"action_type": "cash_dividend",
"status": "confirmed",
"effective_time": "2018-01-02T07:00:00Z",
"terms_digest": digest({"synthetic_terms": 1}),
}
identify(action, "action_revision_id", "rhcav2:")
source["corporate_action_revisions"] = [action]
source["standardized_views"][0]["corporate_action_revision_ids"] = [
action["action_revision_id"]
]
seal_foundation(source)
assert len(parse_foundation(source).to_dict()["corporate_action_revisions"]) == 1
source["standardized_views"][0]["instrument_route_revision_ids"].remove(
route["route_revision_id"]
)
seal_foundation(source)
with pytest.raises(FactorContractError):
parse_foundation(source)
def test_views_cannot_borrow_calendar_selection_from_each_other() -> None:
source = golden("data-foundation")
calendar = deepcopy(source["trading_calendar_revisions"][0])
calendar["calendar_id"] = "rhcalendar:" + "9" * 32
identify(calendar, "calendar_revision_id", "rhcalv2:")
source["trading_calendar_revisions"].append(calendar)
view = deepcopy(source["standardized_views"][0])
view["view_id"] = "rhview:" + "8" * 32
view["trading_calendar_revision_ids"] = [calendar["calendar_revision_id"]]
source["standardized_views"].append(view)
seal_foundation(source)
with pytest.raises(FactorContractError):
parse_foundation(source)
@pytest.mark.parametrize(
"location", ["/private/a", "./a", "../a", "C:\\data\\a", "\\\\host\\a", "s3://private/a"]
)
def test_materialized_values_cannot_carry_physical_locations(location: str) -> None:
source = golden("dataset-snapshot")
rows = records()
rows[0]["value"] = location
bind_records(source, [rows])
snapshot = RetrospectiveSnapshotEnvelope.from_dict(source)
with pytest.raises(FactorContractError):
snapshot.verify_materialized_records([rows])
@@ -0,0 +1,367 @@
"""Synthetic v2 computation boundaries; never source authentication."""
from __future__ import annotations
from copy import deepcopy
from typing import Any
import pytest
from quant_engine.factor_contracts import (
ActorIdentity,
FactorContractError,
FactorDefinition,
FactorInput,
FactorSetRef,
OutputArtifactRef,
OutputCoverage,
OutputQuality,
OutputQualityCheck,
PayloadValidation,
ProducerIdentity,
canonical_json_bytes,
factor_input_schema_digest,
)
from quant_engine.retrospective_data_contracts import (
RetrospectiveFoundationEnvelope,
RetrospectiveSnapshotEnvelope,
)
from quant_engine.retrospective_factor_contracts import (
ResolvedRetrospectiveView,
RetrospectiveCausation,
RetrospectiveFactorSetRef,
RetrospectiveInputBinding,
RetrospectiveViewAvailability,
)
from test_retrospective_data_contracts import (
digest,
golden,
identify,
records,
replace_at,
seal_foundation,
)
def factor_arguments() -> dict[str, Any]:
snapshot = RetrospectiveSnapshotEnvelope.from_dict(golden("dataset-snapshot"))
foundation = RetrospectiveFoundationEnvelope.from_dict(
golden("data-foundation"), snapshot=snapshot
)
view = next(iter(foundation.views.values()))
factor_inputs = (FactorInput("market", view.schema_digest, ("value",)),)
definition = FactorDefinition.create(
factor_id="neutral_close",
version="1.0.0",
formula="value",
parameters={},
implementation_digest=digest({"synthetic_formula": "identity"}),
input_schema_digest=factor_input_schema_digest(factor_inputs),
inputs=factor_inputs,
valid_from="2026-01-01T00:00:00Z",
valid_until="2027-01-01T00:00:00Z",
warmup_sessions=0,
lag_sessions=1,
producer=ProducerIdentity("quant_engine", "0.1.0"),
code_revision="c" * 40,
)
schema = {"fields": ["instrument_id", "value"]}
output = [{"instrument_id": row["instrument_id"], "value": row["value"]} for row in records()]
return {
"definitions": (definition,),
"dataset_snapshot": snapshot,
"foundation": foundation,
"selected_view_ref_ids": (view.view_ref_id,),
"input_bindings": (
RetrospectiveInputBinding(
definition.definition_id, "market", view.view_ref_id, view.schema_digest
),
),
"view_availability": (
RetrospectiveViewAvailability(
view.view_ref_id, view.available_at, digest({"synthetic_view_receipt": 1})
),
),
"dataset_chunks": [records()],
"resolved_views": (
ResolvedRetrospectiveView(
view.view_ref_id,
canonical_json_bytes({"synthetic_schema": "neutral_close_v2"}),
canonical_json_bytes(sorted(records(), key=canonical_json_bytes)),
),
),
"output_quality": OutputQuality(
"passed",
(OutputQualityCheck("finite_values", "passed", digest({"synthetic_quality": 1})),),
),
"output_coverage": OutputCoverage(
"complete", 2, 2, "records", "synthetic_market", digest({"synthetic_coverage": 1})
),
"output_schema_bytes": canonical_json_bytes(schema),
"output_content_bytes": canonical_json_bytes(output),
"output_artifact_ref": OutputArtifactRef.create(
schema_digest=digest(schema), content_digest=digest(output)
),
"evaluation_at": "2026-09-08T01:06:00Z",
"computed_at": "2026-09-08T01:07:00Z",
"artifact_available_at": "2026-09-08T01:08:00Z",
"producer": ProducerIdentity("quant_engine", "0.1.0"),
"code_revision": "d" * 40,
"actor": ActorIdentity("service", "synthetic.research"),
"correlation_id": "synthetic.retrospective",
"causation": RetrospectiveCausation("foundation", foundation.foundation_id),
"evidence_scope": "synthetic_fixture",
"decision_eligible": False,
}
def decoding_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
return {key: arguments[key] for key in ("definitions", "dataset_snapshot", "foundation")}
def test_factor_result_closes_v2_inputs_and_preserves_v1_definition() -> None:
arguments = factor_arguments()
result = RetrospectiveFactorSetRef.create(**arguments)
wire = result.to_dict()
assert result.schema_version == "2.0.0"
assert result.factor_set_id.startswith("rhfactorsetv2:sha256:")
assert result.definition_ids == (arguments["definitions"][0].definition_id,)
assert result.definition_ids[0].startswith("rhfactorv1:")
assert wire["usage"] == "retrospective_research"
assert wire["availability_mode"] == "retrospective_replay"
assert wire["historical_availability"] == "not_established"
assert wire["observation_cutoff"] == arguments["foundation"].observation_cutoff
assert wire["decision_eligible"] is False
assert "pit_cutoff" not in wire
assert result.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
assert result.input_payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
restored = RetrospectiveFactorSetRef.from_json(
result.to_json(), **decoding_arguments(arguments)
)
assert restored.to_dict() == wire
assert restored.payload_validation is PayloadValidation.REFERENCE_ONLY
assert restored.input_payload_validation is PayloadValidation.REFERENCE_ONLY
@pytest.mark.parametrize(
("path", "value"),
[
("schema_version", "1.0.0"),
("schema_version", "2.1.0"),
("contract_name", "researchhub.dataset-snapshot"),
("dataset_snapshot_id", "rhdsv2:sha256:" + "0" * 64),
("foundation_id", "rhdfv2:sha256:" + "0" * 64),
("observation_cutoff", "2018-01-02T07:00:00Z"),
("pit_cutoff", "2018-01-02T07:00:00Z"),
("selected_view_ref_ids", []),
("selected_view_ref_ids", ["rhviewrefv2:sha256:" + "0" * 64]),
("definition_ids", ["rhfactorv1:sha256:" + "0" * 64]),
("input_bindings", []),
("input_bindings.0.input_name", "volume"),
("input_bindings.0.schema_digest", "sha256:" + "0" * 64),
("input_bindings.0.view_ref_id", "rhviewrefv2:sha256:" + "0" * 64),
("view_availability", []),
("view_availability.0.available_at", "2026-09-08T01:03:30Z"),
("view_availability.0.available_at", "2026-09-08T01:04:00.0000001Z"),
("upstream_evidence.quality.checks.0.status", "failed"),
("upstream_evidence.qualification.evaluated_at", "2018-01-02T07:00:00Z"),
("upstream_evidence.time_semantics.earliest_external_knowledge", {"status": "evidenced"}),
("evidence_scope", "real_data"),
("output_quality.status", "failed"),
("output_quality.checks.0.status", "failed"),
("output_coverage.status", "incomplete"),
("output_coverage.observed_count", 1),
("output_schema_digest", "sha256:" + "0" * 64),
("output_artifact_ref.artifact_id", "rhfactoroutputv1:sha256:" + "0" * 64),
("availability_mode", "as_available"),
("usage", "paper_trading"),
("historical_availability", "declared_as_available"),
("decision_eligible", True),
("decision_eligible", 0),
("evaluation_at", "2018-01-02T07:00:00Z"),
("evaluation_at", "2026-09-08T01:04:00Z"),
("computed_at", "2026-09-08T01:05:00Z"),
("artifact_available_at", "2026-09-08T01:06:00Z"),
("producer.id", "research_platform"),
("code_revision", "unknown"),
("actor.id", "https://private/a"),
("causation.id", "rhdfv2:sha256:" + "0" * 64),
],
)
def test_factor_rejects_reidentified_semantic_forgery(path: str, value: Any) -> None:
arguments = factor_arguments()
row = RetrospectiveFactorSetRef.create(**arguments).to_dict()
replace_at(row, path, value)
identify(row, "factor_set_id", "rhfactorsetv2:")
with pytest.raises(FactorContractError):
RetrospectiveFactorSetRef.from_dict(row, **decoding_arguments(arguments))
def test_payload_validation_is_never_inherited_from_serialization() -> None:
arguments = factor_arguments()
result = RetrospectiveFactorSetRef.create(**arguments)
kwargs = decoding_arguments(arguments)
reference = RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs)
with pytest.raises(FactorContractError):
reference.require_payloads_revalidated()
checked = RetrospectiveFactorSetRef.from_dict(
result.to_dict(),
**kwargs,
**{
key: arguments[key]
for key in (
"output_schema_bytes",
"output_content_bytes",
"dataset_chunks",
"resolved_views",
)
},
)
checked.require_payloads_revalidated()
assert checked == result
for extra in (
{"output_schema_bytes": arguments["output_schema_bytes"]},
{"dataset_chunks": arguments["dataset_chunks"]},
{"resolved_views": arguments["resolved_views"]},
{"output_schema_bytes": arguments["output_schema_bytes"], "output_content_bytes": b"[]"},
{"dataset_chunks": arguments["dataset_chunks"], "resolved_views": []},
):
with pytest.raises(FactorContractError):
RetrospectiveFactorSetRef.from_dict(result.to_dict(), **kwargs, **extra)
def test_create_verifies_actual_snapshot_and_each_resolved_view() -> None:
for mutation in (
"content",
"schema",
"snapshot",
"duplicate_view",
"noncanonical",
"unknown_view",
):
arguments = factor_arguments()
view = arguments["resolved_views"][0]
if mutation == "content":
arguments["resolved_views"] = (
ResolvedRetrospectiveView(view.view_ref_id, view.schema_bytes, b"[]"),
)
elif mutation == "schema":
arguments["resolved_views"] = (
ResolvedRetrospectiveView(view.view_ref_id, b"{}", view.content_bytes),
)
elif mutation == "snapshot":
arguments["dataset_chunks"][0][0]["value"] = "0"
elif mutation == "duplicate_view":
arguments["resolved_views"] = (view, view)
elif mutation == "unknown_view":
arguments["resolved_views"] = (
ResolvedRetrospectiveView(
"rhviewrefv2:sha256:" + "0" * 64, view.schema_bytes, view.content_bytes
),
)
else:
arguments["output_content_bytes"] += b"\n"
with pytest.raises(FactorContractError):
RetrospectiveFactorSetRef.create(**arguments)
def test_parent_requires_exact_correlation_scope_and_actual_availability() -> None:
arguments = factor_arguments()
parent = RetrospectiveFactorSetRef.create(**arguments)
child_args = {
**arguments,
"parent": parent,
"causation": RetrospectiveCausation("factor_set", parent.factor_set_id),
"evaluation_at": "2026-09-08T01:09:00Z",
"computed_at": "2026-09-08T01:10:00Z",
"artifact_available_at": "2026-09-08T01:11:00Z",
}
child = RetrospectiveFactorSetRef.create(**child_args)
assert child.factor_set_id != parent.factor_set_id
assert (
RetrospectiveFactorSetRef.from_json(
child.to_json(), **decoding_arguments(arguments), parent=parent
)
== child
)
for changes in (
{"parent": None},
{"correlation_id": "different.correlation"},
{"causation": RetrospectiveCausation("factor_set", "rhfactorsetv2:sha256:" + "0" * 64)},
{"evaluation_at": "2026-09-08T01:07:59Z"},
{"causation": arguments["causation"]},
):
with pytest.raises(FactorContractError):
RetrospectiveFactorSetRef.create(**{**child_args, **changes})
def test_definition_validity_is_checked_at_actual_evaluation() -> None:
arguments = factor_arguments()
arguments.update(
evaluation_at="2027-01-01T00:00:00Z",
computed_at="2027-01-01T00:01:00Z",
artifact_available_at="2027-01-01T00:02:00Z",
)
with pytest.raises(FactorContractError):
RetrospectiveFactorSetRef.create(**arguments)
def test_factor_contract_is_immutable_and_v1_does_not_accept_it() -> None:
arguments = factor_arguments()
result = RetrospectiveFactorSetRef.create(**arguments)
exported = result.to_dict()
exported["upstream_evidence"]["quality"]["status"] = "failed"
assert result.upstream_evidence["quality"]["status"] == "passed"
with pytest.raises(TypeError):
result.upstream_evidence["quality"]["status"] = "failed"
with pytest.raises(FactorContractError):
FactorSetRef.from_dict(result.to_dict(), **decoding_arguments(arguments))
with pytest.raises(FactorContractError):
RetrospectiveFactorSetRef.from_json(
result.to_json() + "\n", **decoding_arguments(arguments)
)
with pytest.raises(FactorContractError):
RetrospectiveInputBinding(
arguments["definitions"][0].definition_id,
"market",
"rhviewrefv1:sha256:" + "0" * 64,
"sha256:" + "0" * 64,
)
def test_missing_synthetic_or_real_readiness_cannot_be_relabelled() -> None:
arguments = factor_arguments()
snapshot_row = arguments["dataset_snapshot"].to_dict()
snapshot_row["evidence_scope"] = "real_data"
identify(snapshot_row, "snapshot_id", "rhdsv2:")
snapshot = RetrospectiveSnapshotEnvelope.from_dict(snapshot_row)
foundation_row = arguments["foundation"].to_dict()
foundation_row["dataset_snapshot_id"] = snapshot.snapshot_id
foundation_row["readiness"]["evidence_scope"] = "real_data"
for view in foundation_row["standardized_views"]:
view["dataset_snapshot_id"] = snapshot.snapshot_id
seal_foundation(foundation_row)
foundation = RetrospectiveFoundationEnvelope.from_dict(foundation_row, snapshot=snapshot)
view = next(iter(foundation.views.values()))
arguments.update(
dataset_snapshot=snapshot,
foundation=foundation,
evidence_scope="real_data",
selected_view_ref_ids=(view.view_ref_id,),
input_bindings=(
RetrospectiveInputBinding(
arguments["definitions"][0].definition_id,
"market",
view.view_ref_id,
view.schema_digest,
),
),
view_availability=(
RetrospectiveViewAvailability(
view.view_ref_id, view.available_at, digest({"synthetic_view": 1})
),
),
causation=RetrospectiveCausation("foundation", foundation.foundation_id),
)
with pytest.raises(FactorContractError, match="real-data"):
RetrospectiveFactorSetRef.create(**arguments)
@@ -0,0 +1,655 @@
"""New synthetic S4 evidence; historical valuation is not actual availability."""
from __future__ import annotations
import hashlib
import json
from dataclasses import FrozenInstanceError, replace
from typing import Any
import pandas as pd
import pytest
from quant_engine.portfolio_risk_contracts import (
ComputationReceipt,
ConstraintSetV1,
FreshnessPolicy,
PortfolioRiskContractError,
RiskAssessmentStatus,
RiskFindingCode,
)
from quant_engine.artifact import EvidenceQualification, PerformanceEvidenceError
from quant_engine.factor_contracts import FactorContractError
from quant_engine.governed_pipeline import BacktestContractError
from quant_engine.risk import ComponentRiskResult, CovarianceSnapshot, labeled_component_risk
import quant_engine.retrospective_portfolio_risk_contracts as contracts
from quant_engine.retrospective_artifact_contracts import (
build_retrospective_backtest_evidence_manifest,
)
from quant_engine.retrospective_backtest_contracts import RetrospectiveBacktestRunRef
from quant_engine.retrospective_portfolio_risk_contracts import (
RetrospectivePortfolioDecision,
RetrospectivePortfolioTarget,
RetrospectiveRiskAssessment,
build_retrospective_portfolio_decision,
compute_retrospective_portfolio_receipt_digests,
assess_retrospective_portfolio_risk,
)
from test_retrospective_artifact_contracts import synthetic_artifact
from test_retrospective_backtest_contracts import run_arguments
from test_retrospective_data_contracts import digest, replace_at
ASSETS = ("rhinstrument:" + "1" * 32, "rhinstrument:" + "2" * 32)
CONTRACT_ERRORS = (
FactorContractError,
PortfolioRiskContractError,
BacktestContractError,
PerformanceEvidenceError,
)
def portfolio_arguments() -> dict[str, Any]:
run = RetrospectiveBacktestRunRef.create(**run_arguments())
artifact = synthetic_artifact(run)
manifest = build_retrospective_backtest_evidence_manifest(
run, artifact, artifact_available_at="2026-09-08T01:11:00Z"
)
target = RetrospectivePortfolioTarget.create(
backtest_run_id=run.run_id,
dataset_snapshot_id=run.dataset_snapshot_id,
weights={ASSETS[0]: 0.6, ASSETS[1]: 0.4},
effective_at="2018-01-05T07:00:00Z",
created_at="2026-09-08T01:12:00Z",
)
return {
"backtest_run_ref": run,
"manifest": manifest,
"target": target,
"objective_name": "synthetic_allocation",
"objective_version": "1.0.0",
"objective_digest": digest({"synthetic_objective": 1}),
"model_name": "bounded_weights",
"model_version": "1.0.0",
"model_digest": digest({"synthetic_model": 1}),
"expected_return_digest": digest({"synthetic_returns": 1}),
"covariance_digest": "sha256:" + "a" * 64,
"scenario_digest": digest({"synthetic_scenario": 1}),
"constraints": ConstraintSetV1(
gross_exposure_max=1.0,
net_exposure_min=1.0,
net_exposure_max=1.0,
single_asset_min=0.2,
single_asset_max=0.7,
position_count_max=2,
turnover_max=0.2,
),
"freshness_policy": FreshnessPolicy(
max_manifest_age_seconds=3600, max_covariance_age_days=0
),
"prior_weights": {ASSETS[0]: 0.5, ASSETS[1]: 0.5},
"computed_at": "2026-09-08T01:13:00Z",
}
def portfolio_receipt(arguments: dict[str, Any], **changes: Any) -> ComputationReceipt:
values = compute_retrospective_portfolio_receipt_digests(
**{key: value for key, value in arguments.items() if key not in {"computed_at", "receipt"}}
)
return ComputationReceipt(
**{
"algorithm": "bounded_weights",
"algorithm_version": "1.0.0",
"implementation_digest": digest({"synthetic_implementation": 1}),
"parameter_digest": digest({"synthetic_parameters": 1}),
"input_digest": values["input_digest"],
"constraint_digest": values["constraint_digest"],
"output_digest": values["output_digest"],
"status": "completed",
"solver_required": False,
"solver_name": None,
"solver_version": None,
"solver_config_digest": None,
"iterations": None,
"objective_value": None,
"max_constraint_residual": values["max_constraint_residual"],
"tolerance": 1e-12,
"computed_at": arguments["computed_at"],
**changes,
}
)
def test_target_separates_historical_effective_time_from_actual_creation() -> None:
arguments = portfolio_arguments()
target = arguments["target"]
assert target.effective_at == "2018-01-05T07:00:00Z"
assert target.created_at == "2026-09-08T01:12:00Z"
assert target.target_id.startswith("rhportfoliotargetv2:sha256:")
assert target.to_dict()["usage"] == "retrospective_research"
def test_portfolio_decision_preserves_constraints_and_actual_receipt_time() -> None:
arguments = portfolio_arguments()
decision = build_retrospective_portfolio_decision(
**arguments, receipt=portfolio_receipt(arguments)
)
assert decision.decision_id.startswith("rhportfoliodecisionv2:sha256:")
assert decision.effective_at == "2018-01-05T07:00:00Z"
assert decision.created_at == "2026-09-08T01:12:00Z"
assert decision.computed_at == "2026-09-08T01:13:00Z"
assert decision.gross_exposure == 1.0
assert decision.position_count == 2
assert decision.to_dict()["decision_eligible"] is False
def covariance(arguments: dict[str, Any], **changes: Any) -> CovarianceSnapshot:
return CovarianceSnapshot(
**{
"snapshot_id": "covariance:synthetic-retrospective",
"as_of_date": "2018-01-05",
"covariance": pd.DataFrame([[0.04, 0.01], [0.01, 0.09]], index=ASSETS, columns=ASSETS),
"return_frequency": "1d",
"periods_per_year": 252,
"method": "provided",
"window_start_date": "2018-01-02",
"window_end_date": "2018-01-05",
"observations": 4,
"lookback_sessions": 4,
"missing_policy": "complete_case",
"data_snapshot_id": arguments["backtest_run_ref"].dataset_snapshot_id,
"input_sha256": "a" * 64,
**changes,
}
)
def risk_arguments(arguments: dict[str, Any]) -> dict[str, Any]:
decision = build_retrospective_portfolio_decision(
**arguments, receipt=portfolio_receipt(arguments)
)
return {
"portfolio_decision": decision,
"backtest_run_ref": arguments["backtest_run_ref"],
"manifest": arguments["manifest"],
"covariance": covariance(arguments),
"risk_model_name": "euler_volatility",
"risk_model_version": "1.0.0",
"risk_model_digest": digest({"synthetic_risk_model": 1}),
"risk_budget": {ASSETS[0]: 0.8, ASSETS[1]: 0.8},
"portfolio_volatility_limit": 10.0,
"groups": {ASSETS[0]: "equity", ASSETS[1]: "fixed_income"},
"computed_at": "2026-09-08T01:14:00Z",
}
def test_risk_uses_historical_business_age_and_actual_computation_time() -> None:
arguments = risk_arguments(portfolio_arguments())
result = assess_retrospective_portfolio_risk(**arguments)
assert result.assessment_id.startswith("rhriskassessmentv2:sha256:")
assert result.qualified is True
assert result.effective_at == "2018-01-05T07:00:00Z"
assert result.computed_at == "2026-09-08T01:14:00Z"
assert result.to_dict()["decision_eligible"] is False
assert result.to_dict()["execution_validation"] == "not_validated"
assert sum(result.percentage_risk.values()) == pytest.approx(1.0)
assert sum(result.component_risk.values()) == pytest.approx(result.portfolio_volatility)
def test_new_risk_computation_cannot_reuse_stale_actual_manifest_time() -> None:
arguments = risk_arguments(portfolio_arguments())
arguments["computed_at"] = "2026-09-08T02:11:01Z"
with pytest.raises(FactorContractError, match="stale"):
assess_retrospective_portfolio_risk(**arguments)
def target_with(arguments: dict[str, Any], **changes: Any) -> RetrospectivePortfolioTarget:
row = arguments["target"].to_dict()
return RetrospectivePortfolioTarget.create(
**{
key: value
for key, value in {**row, **changes}.items()
if key
in {"backtest_run_id", "dataset_snapshot_id", "weights", "effective_at", "created_at"}
}
)
def decision_context(arguments: dict[str, Any]) -> dict[str, Any]:
return {key: arguments[key] for key in ("backtest_run_ref", "manifest", "target")}
def assessment_context(arguments: dict[str, Any]) -> dict[str, Any]:
return {
key: arguments[key]
for key in ("portfolio_decision", "backtest_run_ref", "manifest", "covariance")
}
def reidentify(row: dict[str, Any], field: str, prefix: str) -> None:
row.pop(field, None)
encoded = json.dumps(
row, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False
)
row[field] = prefix + "sha256:" + hashlib.sha256(encoded.encode()).hexdigest()
@pytest.mark.parametrize(
"parser",
[RetrospectivePortfolioTarget, RetrospectivePortfolioDecision, RetrospectiveRiskAssessment],
)
def test_json_syntax_failures_use_typed_contract_errors(parser: Any) -> None:
with pytest.raises(FactorContractError):
parser.from_json(b"{")
@pytest.mark.parametrize(
"change",
[
{"method": "alternate_estimator"},
{"window_start_date": "2018-01-03"},
{"window_end_date": "2018-01-04"},
{"observations": 3},
{"lookback_sessions": 5},
{"missing_policy": "alternate_missing_policy"},
],
)
def test_covariance_estimation_context_is_bound_into_the_result_identity(
change: dict[str, Any],
) -> None:
base = portfolio_arguments()
arguments = risk_arguments(base)
original = assess_retrospective_portfolio_risk(**arguments)
arguments["covariance"] = covariance(base, **change)
changed = assess_retrospective_portfolio_risk(**arguments)
assert changed.assessment_id != original.assessment_id
def test_canonical_roundtrips_and_immutable_results() -> None:
base = portfolio_arguments()
target = base["target"]
assert RetrospectivePortfolioTarget.from_json(target.to_json().encode()) == target
decision = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base))
assert (
RetrospectivePortfolioDecision.from_json(decision.to_json(), **decision_context(base))
== decision
)
arguments = risk_arguments(base)
result = assess_retrospective_portfolio_risk(**arguments)
assert (
RetrospectiveRiskAssessment.from_json(
result.to_json().encode(), **assessment_context(arguments)
)
== result
)
with pytest.raises(TypeError):
target.weights[ASSETS[0]] = 0.1
with pytest.raises(FrozenInstanceError):
target.created_at = "2018-01-05T07:00:00Z"
with pytest.raises(TypeError):
decision.target_weights[ASSETS[0]] = 0.1
with pytest.raises(TypeError):
result.component_risk[ASSETS[0]] = 0.1
detached = result.to_dict()
detached["component_risk"][ASSETS[0]] = 0.1
assert detached != result.to_dict()
@pytest.mark.parametrize(
"change",
[
{"weights": {}},
{"weights": {"SIM0": 1.0}},
{"weights": {ASSETS[0]: float("nan")}},
{"weights": {ASSETS[0]: True}},
{"backtest_run_id": "rhbacktestrunv1:sha256:" + "0" * 64},
{"dataset_snapshot_id": "rhds:sha256:" + "0" * 64},
{"effective_at": "2026-09-09T01:00:00Z"},
{"created_at": "2026-09-08T01:12:00.1234567Z"},
{"effective_at": "2018-01-05T15:00:00+08:00"},
],
)
def test_target_rejects_legacy_ambiguous_and_nonfinite_inputs(change: dict[str, Any]) -> None:
with pytest.raises(CONTRACT_ERRORS):
target_with(portfolio_arguments(), **change)
@pytest.mark.parametrize(
"path,value",
[
("usage", "live"),
("historical_availability", "established"),
("schema_version", "1.0.0"),
("extra", True),
("target_id", "rhportfoliotargetv2:sha256:" + "0" * 64),
],
)
def test_target_rejects_wire_mutations(path: str, value: Any) -> None:
row = portfolio_arguments()["target"].to_dict()
row[path] = value
with pytest.raises(CONTRACT_ERRORS):
RetrospectivePortfolioTarget.from_dict(row)
@pytest.mark.parametrize("field", ["input_digest", "constraint_digest", "output_digest"])
def test_receipt_digests_are_recomputed(field: str) -> None:
arguments = portfolio_arguments()
receipt = portfolio_receipt(arguments, **{field: "sha256:" + "0" * 64})
with pytest.raises(FactorContractError, match="independently recomputed"):
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
@pytest.mark.parametrize("status", ["failed", "fallback"])
def test_failed_or_fallback_solver_cannot_form_a_decision(status: str) -> None:
arguments = portfolio_arguments()
receipt = portfolio_receipt(
arguments,
status=status,
solver_required=True,
solver_name="synthetic_solver",
solver_version="1.0.0",
solver_config_digest=digest({"synthetic_solver": 1}),
iterations=1,
objective_value=0.0,
)
with pytest.raises(FactorContractError, match="failed/fallback"):
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
@pytest.mark.parametrize(
"change",
[
{"backtest_run_id": "rhbacktestrunv2:sha256:" + "0" * 64},
{"dataset_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
{"weights": {"rhinstrument:" + "f" * 32: 1.0}},
{"created_at": "2026-09-08T01:10:00Z"},
{"created_at": "2026-09-08T01:14:00Z"},
],
)
def test_decision_closes_target_identity_assets_and_actual_time(change: dict[str, Any]) -> None:
arguments = portfolio_arguments()
receipt = portfolio_receipt(arguments)
arguments["target"] = target_with(arguments, **change)
with pytest.raises(CONTRACT_ERRORS):
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
def test_actual_manifest_freshness_boundary_and_receipt_time() -> None:
arguments = portfolio_arguments()
arguments["computed_at"] = "2026-09-08T02:11:00Z"
assert (
build_retrospective_portfolio_decision(
**arguments, receipt=portfolio_receipt(arguments)
).computed_at
== arguments["computed_at"]
)
arguments["computed_at"] = "2026-09-08T02:11:00.000001Z"
with pytest.raises(FactorContractError, match="stale"):
build_retrospective_portfolio_decision(**arguments, receipt=portfolio_receipt(arguments))
arguments["computed_at"] = "2026-09-08T01:13:00Z"
with pytest.raises(FactorContractError, match="receipt actual time"):
build_retrospective_portfolio_decision(
**arguments, receipt=portfolio_receipt(arguments, computed_at="2026-09-08T01:13:01Z")
)
def test_manifest_tables_and_qualification_are_revalidated_at_s4_boundary() -> None:
arguments = portfolio_arguments()
receipt = portfolio_receipt(arguments)
manifest = arguments["manifest"]
artifact = manifest._artifact
arguments["manifest"] = build_retrospective_backtest_evidence_manifest(
arguments["backtest_run_ref"],
artifact,
artifact_available_at=manifest.artifact_available_at,
qualification=EvidenceQualification.EXPLORATORY,
)
with pytest.raises(FactorContractError, match="contract-qualified"):
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
arguments["manifest"] = manifest
# Public access is an isolated copy. Simulate corruption of the retained bytes,
# beyond that normal interface, to exercise the consumer's independent recheck.
artifact._performance.loc[0, "n_days"] += 1
with pytest.raises(CONTRACT_ERRORS):
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
def test_constraint_residuals_and_prior_assets_cannot_be_bypassed() -> None:
arguments = portfolio_arguments()
arguments["constraints"] = ConstraintSetV1(gross_exposure_max=0.9)
# A solver may report convergence within its tolerance; actual contract constraints still bind.
receipt = portfolio_receipt(
arguments,
status="converged",
solver_required=True,
solver_name="synthetic_solver",
solver_version="1.0.0",
solver_config_digest=digest({"synthetic_solver": 1}),
iterations=1,
objective_value=0.0,
tolerance=0.2,
)
with pytest.raises(FactorContractError, match="violates supported constraints"):
build_retrospective_portfolio_decision(**arguments, receipt=receipt)
arguments = portfolio_arguments()
arguments["prior_weights"] = {"rhinstrument:" + "f" * 32: 0.5}
with pytest.raises(FactorContractError, match="prior assets"):
compute_retrospective_portfolio_receipt_digests(
**{key: value for key, value in arguments.items() if key != "computed_at"}
)
arguments["prior_weights"] = None
with pytest.raises(PortfolioRiskContractError, match="prior"):
portfolio_receipt(arguments)
def test_optional_prior_budget_limit_and_groups_have_explicit_empty_semantics() -> None:
base = portfolio_arguments()
base["constraints"] = ConstraintSetV1(gross_exposure_max=1.0)
base["prior_weights"] = None
arguments = risk_arguments(base)
arguments.update(risk_budget=None, portfolio_volatility_limit=None, groups=None)
result = assess_retrospective_portfolio_risk(**arguments)
assert result.qualified is True
assert result.risk_budget == {}
assert result.group_exposure == {}
assert result.groups is None
@pytest.mark.parametrize(
"path,value",
[
("decision_eligible", True),
("execution_validation", "validated"),
("historical_availability", "established"),
("gross_exposure", True),
("position_count", 2.0),
("target_weights." + ASSETS[0], 0.5),
("schema_version", "1.0.0"),
("observation_cutoff", "2018-01-05T07:00:00Z"),
("extra", True),
],
)
def test_decision_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
base = portfolio_arguments()
row = build_retrospective_portfolio_decision(**base, receipt=portfolio_receipt(base)).to_dict()
replace_at(row, path, value)
reidentify(row, "decision_id", "rhportfoliodecisionv2:")
with pytest.raises(CONTRACT_ERRORS):
RetrospectivePortfolioDecision.from_dict(row, **decision_context(base))
@pytest.mark.parametrize(
"change",
[
{"as_of_date": "2018-01-06"},
{"as_of_date": "2018-01-04", "window_end_date": "2018-01-04"},
{"window_start_date": None, "window_end_date": None},
{"data_snapshot_id": "rhdsv2:sha256:" + "0" * 64},
{"input_sha256": "b" * 64},
],
)
def test_covariance_business_time_bounds_and_source_binding(change: dict[str, Any]) -> None:
base = portfolio_arguments()
arguments = risk_arguments(base)
arguments["covariance"] = covariance(base, **change)
with pytest.raises(FactorContractError):
assess_retrospective_portfolio_risk(**arguments)
@pytest.mark.parametrize(
"matrix,index,columns",
[
([[float("nan"), 0.0], [0.0, 0.1]], ASSETS, ASSETS),
([[0.1, 0.1], [0.0, 0.1]], ASSETS, ASSETS),
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], ASSETS[0]), ASSETS),
([[0.1, 0.0], [0.0, 0.1]], (ASSETS[0], "unknown"), ASSETS),
([[0.1, 0.0], [0.0, 0.1]], ASSETS, (ASSETS[0], "unknown")),
],
)
def test_covariance_structure_is_checked_before_computation(
matrix: Any, index: Any, columns: Any
) -> None:
base = portfolio_arguments()
arguments = risk_arguments(base)
arguments["covariance"] = covariance(
base, covariance=pd.DataFrame(matrix, index=index, columns=columns)
)
with pytest.raises(PortfolioRiskContractError):
assess_retrospective_portfolio_risk(**arguments)
@pytest.mark.parametrize(
"change",
[
{"risk_budget": {ASSETS[0]: -0.1}},
{"risk_budget": {"unknown": 0.1}},
{"portfolio_volatility_limit": -0.1},
{"groups": {ASSETS[0]: "equity"}},
{"groups": []},
{"risk_model_version": "latest"},
{"risk_model_name": "/private/model"},
{"computed_at": "2026-09-08T01:12:59Z"},
{"portfolio_decision": object()},
{"covariance": object()},
],
)
def test_risk_rejects_invalid_models_budgets_clocks_and_untyped_inputs(
change: dict[str, Any],
) -> None:
arguments = risk_arguments(portfolio_arguments())
arguments.update(change)
with pytest.raises(CONTRACT_ERRORS):
assess_retrospective_portfolio_risk(**arguments)
@pytest.mark.parametrize(
"matrix,finding",
[
([[1.0, 2.0], [2.0, 1.0]], RiskFindingCode.COVARIANCE_NOT_PSD),
([[0.0, 0.0], [0.0, 0.0]], RiskFindingCode.PORTFOLIO_VARIANCE_NON_POSITIVE),
],
)
def test_numerical_unavailability_is_not_qualification(
matrix: Any, finding: RiskFindingCode
) -> None:
base = portfolio_arguments()
arguments = risk_arguments(base)
arguments["covariance"] = covariance(
base, covariance=pd.DataFrame(matrix, index=ASSETS, columns=ASSETS)
)
result = assess_retrospective_portfolio_risk(**arguments)
assert result.status is RiskAssessmentStatus.UNAVAILABLE
assert result.qualified is False
assert result.findings == (finding,)
assert result.portfolio_volatility is None
def test_risk_uses_the_existing_numeric_implementation_exactly_once(
monkeypatch: pytest.MonkeyPatch,
) -> None:
arguments = risk_arguments(portfolio_arguments())
calls = []
def recorded(weights: Any, matrix: Any) -> ComponentRiskResult:
calls.append((weights, matrix))
return labeled_component_risk(weights, matrix)
monkeypatch.setattr(contracts, "labeled_component_risk", recorded)
result = assess_retrospective_portfolio_risk(**arguments)
assert len(calls) == 1
expected = labeled_component_risk(*calls[0])
assert result.component_risk == expected.component.to_dict()
assert result.portfolio_volatility == expected.portfolio_volatility
def test_unknown_numeric_failures_are_sanitized(monkeypatch: pytest.MonkeyPatch) -> None:
def failed(*args: Any) -> ComponentRiskResult:
raise ValueError("synthetic internal detail")
monkeypatch.setattr(contracts, "labeled_component_risk", failed)
with pytest.raises(PortfolioRiskContractError, match="risk computation failed") as error:
assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
assert "internal detail" not in str(error.value)
def test_nonclosed_decomposition_is_unavailable(monkeypatch: pytest.MonkeyPatch) -> None:
def nonclosed(weights: Any, matrix: Any) -> ComponentRiskResult:
output = labeled_component_risk(weights, matrix)
return replace(output, component=output.component * 0.5)
monkeypatch.setattr(contracts, "labeled_component_risk", nonclosed)
result = assess_retrospective_portfolio_risk(**risk_arguments(portfolio_arguments()))
assert result.status is RiskAssessmentStatus.UNAVAILABLE
assert result.findings == (RiskFindingCode.RISK_CONTRIBUTION_NOT_CLOSED,)
@pytest.mark.parametrize(
"change", [{"portfolio_volatility_limit": 0.0}, {"risk_budget": {ASSETS[0]: 0.0}}]
)
def test_budget_breach_keeps_ready_but_unqualified_evidence(change: dict[str, Any]) -> None:
arguments = risk_arguments(portfolio_arguments())
arguments.update(change)
result = assess_retrospective_portfolio_risk(**arguments)
assert result.status is RiskAssessmentStatus.READY
assert result.qualified is False
assert result.findings == (RiskFindingCode.RISK_BUDGET_BREACH,)
assert result.decision_eligible is False
@pytest.mark.parametrize(
"path,value",
[
("decision_eligible", True),
("execution_validation", "validated"),
("historical_availability", "established"),
("qualified", 1),
("portfolio_volatility", 1.0),
("component_risk." + ASSETS[0], 1.0),
("schema_version", "1.0.0"),
("covariance_matrix_digest", "sha256:" + "0" * 64),
("extra", True),
],
)
def test_risk_rejects_reidentified_forged_wire(path: str, value: Any) -> None:
arguments = risk_arguments(portfolio_arguments())
row = assess_retrospective_portfolio_risk(**arguments).to_dict()
replace_at(row, path, value)
reidentify(row, "assessment_id", "rhriskassessmentv2:")
with pytest.raises(CONTRACT_ERRORS):
RetrospectiveRiskAssessment.from_dict(row, **assessment_context(arguments))
@pytest.mark.parametrize(
"raw",
[
b'{"x":1,"x":2}',
b'{ "x":1}',
b"[]",
b'{"x":NaN}',
b'{"x":Infinity}',
b'{"x":9007199254740992}',
1,
],
)
def test_json_profiles_reject_ambiguous_nonfinite_and_noncanonical_input(raw: Any) -> None:
with pytest.raises(CONTRACT_ERRORS):
RetrospectivePortfolioTarget.from_json(raw)
+270
View File
@@ -0,0 +1,270 @@
"""Risk contribution contracts and validation tests."""
from __future__ import annotations
import numpy as np
import pandas as pd
import pytest
from quant_engine.risk import (
ComponentRiskResult,
CovarianceSnapshot,
component_var,
estimate_covariance_snapshot,
labeled_component_risk,
marginal_risk_contribution,
risk_contribution,
)
def test_estimate_covariance_snapshot_is_complete_case_and_reproducible() -> None:
dates = pd.date_range("2026-01-05", periods=6, freq="B")
returns = pd.DataFrame(
{
"A": [0.01, 0.02, 0.03, 0.04, 0.05, 99.0],
"B": [0.02, 0.01, np.nan, 0.03, 0.04, -99.0],
},
index=dates,
)
as_of = dates[4]
snapshot = estimate_covariance_snapshot(
returns,
as_of_date=as_of,
lookback_sessions=4,
min_observations=3,
data_snapshot_id="market-returns-20260109-v1",
return_frequency="1d",
periods_per_year=252,
)
expected_window = returns.loc[:as_of].tail(4)
expected = expected_window.dropna(how="any").cov()
pd.testing.assert_frame_equal(snapshot.covariance, expected)
assert snapshot.snapshot_id.startswith("sample-cov-v1:")
assert snapshot.as_of_date == as_of.date()
assert snapshot.method == "sample"
assert snapshot.window_start_date == expected_window.index[0].date()
assert snapshot.window_end_date == as_of.date()
assert snapshot.observations == 3
assert snapshot.lookback_sessions == 4
assert snapshot.missing_policy == "complete_case"
assert snapshot.data_snapshot_id == "market-returns-20260109-v1"
assert len(snapshot.input_sha256) == 64
future_changed = returns.copy()
future_changed.loc[dates[-1], :] = [1_000_000.0, -1_000_000.0]
repeated = estimate_covariance_snapshot(
future_changed,
as_of_date=as_of,
lookback_sessions=4,
min_observations=3,
data_snapshot_id="market-returns-20260109-v1",
return_frequency="1d",
periods_per_year=252,
)
assert repeated.snapshot_id == snapshot.snapshot_id
pd.testing.assert_frame_equal(repeated.covariance, snapshot.covariance)
def test_covariance_snapshot_identity_captures_data_and_estimator_contract() -> None:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
returns = pd.DataFrame(
{"A": [0.01, 0.02, -0.01, 0.03], "B": [0.02, -0.01, 0.01, 0.04]},
index=dates,
)
base = estimate_covariance_snapshot(
returns,
as_of_date=dates[-1],
lookback_sessions=4,
min_observations=3,
data_snapshot_id="snapshot-a",
)
different_source = estimate_covariance_snapshot(
returns,
as_of_date=dates[-1],
lookback_sessions=4,
min_observations=3,
data_snapshot_id="snapshot-b",
)
assert base.snapshot_id != different_source.snapshot_id
assert base.covariance.equals(different_source.covariance)
def test_estimate_covariance_snapshot_rejects_ambiguous_or_insufficient_history() -> None:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
returns = pd.DataFrame(
{"A": [0.01, np.nan, 0.03, 0.04], "B": [0.02, 0.01, np.nan, 0.03]},
index=dates,
)
with pytest.raises(ValueError, match="complete observations"):
estimate_covariance_snapshot(
returns,
as_of_date=dates[-1],
lookback_sessions=4,
min_observations=3,
data_snapshot_id="snapshot-a",
)
with pytest.raises(ValueError, match="strictly increasing"):
estimate_covariance_snapshot(
returns.iloc[::-1],
as_of_date=dates[-1],
lookback_sessions=4,
min_observations=2,
data_snapshot_id="snapshot-a",
)
def test_covariance_snapshot_is_validated_and_immutable_by_interface() -> None:
covariance = pd.DataFrame(
[[0.04, 0.01], [0.01, 0.09]],
index=["A", "B"],
columns=["A", "B"],
)
snapshot = CovarianceSnapshot(
snapshot_id="cov-20260107-v1",
as_of_date="2026-01-07",
covariance=covariance,
return_frequency="1d",
periods_per_year=252,
)
covariance.loc["A", "A"] = 999.0
leaked_copy = snapshot.covariance
leaked_copy.loc["B", "B"] = 999.0
assert snapshot.as_of_date == pd.Timestamp("2026-01-07").date()
assert snapshot.covariance.loc["A", "A"] == pytest.approx(0.04)
assert snapshot.covariance.loc["B", "B"] == pytest.approx(0.09)
@pytest.mark.parametrize(
("kwargs", "message"),
[
({"snapshot_id": ""}, "snapshot_id"),
({"return_frequency": ""}, "return_frequency"),
({"periods_per_year": 0}, "periods_per_year"),
],
)
def test_covariance_snapshot_rejects_incomplete_identity(
kwargs: dict[str, object],
message: str,
) -> None:
values: dict[str, object] = {
"snapshot_id": "cov-20260107-v1",
"as_of_date": "2026-01-07",
"covariance": pd.DataFrame([[0.04]], index=["A"], columns=["A"]),
"return_frequency": "1d",
"periods_per_year": 252,
}
values.update(kwargs)
with pytest.raises((TypeError, ValueError), match=message):
CovarianceSnapshot(**values)
def test_risk_contribution_sums_to_one_for_positive_portfolio_variance() -> None:
weights = np.array([0.5, 0.5])
covariance = np.diag([1.0, 4.0])
result = risk_contribution(weights, covariance)
np.testing.assert_allclose(result, [0.2, 0.8])
assert result.sum() == pytest.approx(1.0)
def test_zero_variance_portfolio_falls_back_to_equal_contribution() -> None:
result = risk_contribution(np.array([0.2, 0.3, 0.5]), np.zeros((3, 3)))
np.testing.assert_allclose(result, np.full(3, 1 / 3))
def test_marginal_and_component_risk_follow_matrix_identities() -> None:
weights = np.array([0.25, 0.75])
covariance = np.array([[0.04, 0.01], [0.01, 0.09]])
marginal = marginal_risk_contribution(weights, covariance)
component = component_var(weights, covariance)
np.testing.assert_allclose(marginal, covariance @ weights)
np.testing.assert_allclose(component, weights * marginal)
assert component.sum() == pytest.approx(weights @ covariance @ weights)
@pytest.mark.parametrize(
"function",
[risk_contribution, marginal_risk_contribution, component_var],
)
def test_risk_functions_reject_covariance_shape_mismatch(function) -> None:
with pytest.raises(ValueError, match="does not match weights length"):
function(np.array([0.5, 0.5]), np.eye(3))
@pytest.mark.parametrize(
"function",
[risk_contribution, marginal_risk_contribution, component_var],
)
def test_risk_functions_reject_empty_portfolio(function) -> None:
with pytest.raises(ValueError, match="at least one asset"):
function(np.array([]), np.empty((0, 0)))
def test_labeled_component_risk_aligns_covariance_and_closes_to_volatility() -> None:
weights = pd.Series({"A": 0.25, "B": 0.75}, name="weight")
covariance = pd.DataFrame(
[[0.09, 0.01], [0.01, 0.04]],
index=["B", "A"],
columns=["B", "A"],
)
result = labeled_component_risk(weights, covariance)
aligned = covariance.reindex(index=weights.index, columns=weights.index)
expected_volatility = float(np.sqrt(weights @ aligned @ weights))
assert isinstance(result, ComponentRiskResult)
assert result.component.index.tolist() == ["A", "B"]
assert result.portfolio_volatility == pytest.approx(expected_volatility)
assert result.component.sum() == pytest.approx(expected_volatility)
assert result.percentage.sum() == pytest.approx(1.0)
def test_component_risk_groups_actual_asset_contributions_by_label() -> None:
weights = pd.Series({"A": 0.2, "B": 0.3, "C": 0.5})
covariance = pd.DataFrame(np.diag([0.04, 0.09, 0.16]), index=weights.index, columns=weights.index)
groups = pd.Series({"C": "growth", "A": "value", "B": "value"})
result = labeled_component_risk(weights, covariance)
grouped = result.grouped_component(groups)
assert grouped.index.tolist() == ["growth", "value"]
assert grouped.loc["value"] == pytest.approx(
result.component.loc["A"] + result.component.loc["B"]
)
assert grouped.sum() == pytest.approx(result.portfolio_volatility)
def test_labeled_component_risk_rejects_asset_label_mismatch() -> None:
weights = pd.Series({"A": 0.5, "B": 0.5})
covariance = pd.DataFrame(np.eye(2), index=["A", "C"], columns=["A", "C"])
with pytest.raises(ValueError, match="same asset labels"):
labeled_component_risk(weights, covariance)
def test_labeled_component_risk_rejects_invalid_covariance() -> None:
weights = pd.Series({"A": 0.5, "B": 0.5})
asymmetric = pd.DataFrame([[1.0, 0.2], [0.1, 1.0]], index=weights.index, columns=weights.index)
with pytest.raises(ValueError, match="symmetric"):
labeled_component_risk(weights, asymmetric)
def test_labeled_component_risk_rejects_zero_variance_portfolio() -> None:
weights = pd.Series({"A": 0.5, "B": 0.5})
covariance = pd.DataFrame(np.zeros((2, 2)), index=weights.index, columns=weights.index)
with pytest.raises(ValueError, match="positive portfolio variance"):
labeled_component_risk(weights, covariance)
Generated
+408
View File
@@ -0,0 +1,408 @@
version = 1
revision = 3
requires-python = "==3.13.*"
resolution-markers = [
"sys_platform == 'win32'",
"sys_platform == 'emscripten'",
"sys_platform != 'emscripten' and sys_platform != 'win32'",
]
[[package]]
name = "ast-serialize"
version = "0.8.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/e1/a9/11851c3e02a3fea2ddc9932d1fdc7d2edaeecc0d2e11bc5f2a7fde2b0934/ast_serialize-0.8.0.tar.gz", hash = "sha256:6c37c43e4004dfb42d321ddedc569dc17ff4259296f3af577c9ea46a809bc010" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/4c/11/911210c3c78923273a9211a2b6cfc4c8aa723b30dab3e1c8d19afb983b40/ast_serialize-0.8.0-cp315-abi3.abi3t-macosx_10_12_x86_64.whl", hash = "sha256:86b8a1e6d90467345356098b040150e82fbc26d24a7a202224b13dc1f6264ca0" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/77/89/6282881c8587606638db153cbe21e1e0c4d1f3970dee1aa0610a1c62a026/ast_serialize-0.8.0-cp315-abi3.abi3t-macosx_11_0_arm64.whl", hash = "sha256:39e92ff8e8cb45947fe9007174b2950e1fb098e6abd00266a13cd3bcf6675068" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/97/78/a9f846a03a340ff3728c915f23338ca742742f3292700559cdb3ad999b1e/ast_serialize-0.8.0-cp315-abi3.abi3t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:c85d8d18db5b2dfcb3b7e38a4d600ca35504c0ed8a6f75cd1c811e4ffe248a15" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/c0/15/aba6ef8a988a6eceb6f0359589aac509e29ae2dba67fd9bfd5af0c3f13e7/ast_serialize-0.8.0-cp315-abi3.abi3t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:9830ff7e764f74d9eefb01170c61a9f0fd2c027dac5fcb72e064decd57d56371" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/94/29/3f63d696ea7c5b8abadcecc3505be51bd900daaccc522ed8322fa5b05a93/ast_serialize-0.8.0-cp315-abi3.abi3t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:6479d9722a4cd21b578f5478074c41e6169f04811996ec881655560f703a5bba" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e2/5d/0aac338604ff59df5774d4304307898982252f325ff7cafe31d52fedcb65/ast_serialize-0.8.0-cp315-abi3.abi3t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:a63bed264e818cd83eec11feed0f50aa162542b91132ef58afebc857182763a5" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/23/ca/9f1ef795bb724719532bd86dbec11e5b66857d3fbe9b6772baec0191a6ed/ast_serialize-0.8.0-cp315-abi3.abi3t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9d187197d234aa45d6cfa2b096be5f666e8cc2e7eb3722d0ab8926293cf5720c" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/dc/25/5e061372d2ed953b9ba3b9c4f73de3b8e9234cda3f6c088db4686801d0e1/ast_serialize-0.8.0-cp315-abi3.abi3t-manylinux_2_31_riscv64.whl", hash = "sha256:2d39a56282cfcc0d8eeea37267c754be59c98d48505c23b1dae5c6011f3813dd" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/a8/c1/ae7da218053120635a4ca802366c69f707203641af95372eeb83f70dfd52/ast_serialize-0.8.0-cp315-abi3.abi3t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:f7cc5f10386994c0f4844f1e6d6a97127e9b478660eb6dec2b257644f0acab64" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/2e/89/271d1f49c5269fcddcc789ea3f25be401f6723fc1138aeda539f4d05516d/ast_serialize-0.8.0-cp315-abi3.abi3t-musllinux_1_2_aarch64.whl", hash = "sha256:6102f2f985c2e542be85cd857678ec9356fefa792b93cadfadd31139f5696f27" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/55/be/4e7d77fcf571ac7cb5cf7115a20c36642bd7d29473b45dfaaefeb9618f90/ast_serialize-0.8.0-cp315-abi3.abi3t-musllinux_1_2_armv7l.whl", hash = "sha256:3a8660fe66667b76a6e9dccd1d33e66b229fde3b308db991c041609226c005b6" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8b/ae/ed1de2db7e019d4236fbc164ffa5ef9a6022a300a342bbf142d21b7c141e/ast_serialize-0.8.0-cp315-abi3.abi3t-musllinux_1_2_i686.whl", hash = "sha256:e7266307e5fba39836edb79def8608887af48820508bff3c5f2941e1e04d1534" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/92/89/5fea507fae5c5f18b7dc7f95e5c00956574b8c717b8fd2049c504fab0b18/ast_serialize-0.8.0-cp315-abi3.abi3t-musllinux_1_2_ppc64le.whl", hash = "sha256:4ca7e6fd1ad845d1cc649dc2ecd499db2f8f46af5bf8da7b70dd858774cc038b" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/42/71/478d69df21b64e064554a68134c94be304270316ca676a94e63c389a636a/ast_serialize-0.8.0-cp315-abi3.abi3t-musllinux_1_2_riscv64.whl", hash = "sha256:2880350b13d3eae69a0d70bc1fb6c9bfaca4dbd0e20ba8cd1aa483080b56ff06" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/5e/2d/8962dc8d5b3a9dc27b36f9db199afa25264c741505469d9ec10ffbfd2ba7/ast_serialize-0.8.0-cp315-abi3.abi3t-musllinux_1_2_x86_64.whl", hash = "sha256:ab0f9a59f7d63d0d441b56b9a818b273705264352d5115cfee12e940e816d958" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/4f/22/14d2ad4fd1d1bcd0dc687ca268e0630069f45162496260c0efb70ee0ea72/ast_serialize-0.8.0-cp315-abi3.abi3t-win32.whl", hash = "sha256:0485a25ef519c62e749ee3c1ad8070e591b380d67226349eb5a70b228dc1ac4a" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/18/1d/84a327c0202a41aa5fdba3ade33904d6d8f3b9e6806fa83568d835395850/ast_serialize-0.8.0-cp315-abi3.abi3t-win_amd64.whl", hash = "sha256:bd84d60bca7079e741be4ac5dbe237751a59d7f6f9f0126b11880d63822cbe16" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8c/92/74556dec52fde85a2ad84ed159991b916241043788609c15d8b77e14570b/ast_serialize-0.8.0-cp315-abi3.abi3t-win_arm64.whl", hash = "sha256:057769b5921336eb2d9124f2a731b42ed05ffdac559b840dbdf6f3937cf153dc" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/d9/e3/6142e920fec6ef7bccabd8c24ed8ed99f8bdc6cb8b065e1df7c6a3b2d667/ast_serialize-0.8.0-cp39-abi3-macosx_10_12_x86_64.whl", hash = "sha256:e1bd223df0f6c96b396975fa604cb33bce53d9b4a0185490be4c4a289f7c9c87" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/a6/e9/6e8be8df02b35d85e2b8809f7f1cfa290bdf5882b55127a539d049482db0/ast_serialize-0.8.0-cp39-abi3-macosx_11_0_arm64.whl", hash = "sha256:ddd3b61f45c132da66c5476b281891e08c1fd87fbdabe8a6973e1622efc85f06" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8c/80/7e0fd2e2e2aba257820db4a8657c4c356844d36b914b20a4af294bcfb902/ast_serialize-0.8.0-cp39-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:1f9caa63fad8241257ae401b5ff0a64026c6adb36b8e86cbe8782d9ea505daf6" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b0/6a/3bae0af06f9b1bae3001c44d64215f5b567877e7aae9ffd45db11c3a7647/ast_serialize-0.8.0-cp39-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:3926fa117b5e65019853a2969966d11c7175af377a3425991f3fe73784412405" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/6f/c4/ce2d41a1bc22508e82618901f7e10f2a5e2f9556553fea90624daf9875e2/ast_serialize-0.8.0-cp39-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:485f1113af805e9e170b95ef993ca3fbd4f89c04bab25c58b4fc632d854801ab" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/1a/90/f5058f209756dd70e958b7538aaa82d25d24944baf9ec8ae6f27b06fcacc/ast_serialize-0.8.0-cp39-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:3ccebbed24f1281062d5852353c72c47502955926cfcb8345ffb3a44d87ff3d3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/bf/32/7f77ea87fa0836daab706ed5cb7f903bb25fa26a77439011aee626af11d8/ast_serialize-0.8.0-cp39-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:252f883290d1cdb728eb7fe1d9a7221b88af5a329aae0bc91ddee4dafb820331" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/eb/5a/75b82ad2725b5e8e8c742732f9e76c6738a292d0709e1f60d10a973730b4/ast_serialize-0.8.0-cp39-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:96abc072ad29db8d02194afd47d68987322622787daceae82398d7b69f3ba2e6" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/4e/54/8c20ed4eea805516a3fd23dd4a721ce28c64f50f0e4b359969f60a8c97a6/ast_serialize-0.8.0-cp39-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:9118ad3e369727060b2696fc4078f250ecffca4248ba87f537f55cea9f9dce06" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/cb/5b/9f14430f12fe830b656fb38f8e2e05ee13b02a88967660bef46af0ab22a8/ast_serialize-0.8.0-cp39-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:f359df4bd921918af8bebd142a376c77511d7151cc8ba852760b587b5a4a54f3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/2d/3d/084882eca93c842bd4262591a071ec7f825340644035e51501208cc5a8d4/ast_serialize-0.8.0-cp39-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:e94f9121d13fa36cbf21314783c77d05ae3a0868decd18cf5233fdcc6de49ac8" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/ce/73/ea84852096c2036c61cc0b2f97b90242207419f534dc671060ee1c8e05cb/ast_serialize-0.8.0-cp39-abi3-musllinux_1_2_i686.whl", hash = "sha256:54f95b486018d262bcb387a9afd96f0da74508b442762b80c769454a6fbb3ee3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/cb/88/287b9a5300c1f2f651d259f670931b63110adc265b7613c885b44c5bc53d/ast_serialize-0.8.0-cp39-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:4c38b915511e32bc718c49dbce98ff9af36bac0ad6a604f58000cd5e3aecdba7" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/ee/f3/1bc3a79afcf0c2a8d2c37182d0d659d1545a9d7f7f6dc9cf3e63d6c17135/ast_serialize-0.8.0-cp39-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:9a2ef9cf12f2de4f1028c42c1dd7d775255e0fb3e5bb48896c97e35ef52366fe" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/5c/cd/440c798957e14e31776bfeb024d8fafe0bb1d5b89c51c2f067e69938f7b0/ast_serialize-0.8.0-cp39-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:6f18048fe9f6dd266bd577cdec48bdcecb74faaa01fe941324435483b013ed2a" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/4f/4a/587eb36dcc240a54c8660f599464516b469ecad96f0dbdb6bccbedb50745/ast_serialize-0.8.0-cp39-abi3-win32.whl", hash = "sha256:31883542dd6c94d178f5db3d32fbd69c5eb88b3a7c018e7ac8cc0c45195ddbed" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/5f/a4/3e887bbd92164e183cb6e412c6a3e9198ddd446d7fe405958293ef5ef49c/ast_serialize-0.8.0-cp39-abi3-win_amd64.whl", hash = "sha256:861794565b06337005c1447ef23103a3d5a627d08bdc827870d00d0b28ef5f51" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/25/6c/b400476d3ceba681ab929787edc9554f6d88fcc69435eb681b00fc0457a5/ast_serialize-0.8.0-cp39-abi3-win_arm64.whl", hash = "sha256:b2a5978662fd4db463dfb4b974d2b10ac6430b98f5333aabc7051909df3561d0" },
]
[[package]]
name = "colorama"
version = "0.4.6"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/d8/53/6f443c9a4a8358a93a6792e2acffb9d9d5cb0a5cfd8802644b7b1c9a02e4/colorama-0.4.6.tar.gz", hash = "sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6" },
]
[[package]]
name = "coverage"
version = "7.15.4"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/be/c3/4f2195f512fb172aa425a8803a874b2baa9ba7f80ff7b6080998761fc701/coverage-7.15.4.tar.gz", hash = "sha256:0548198fff07ccf4faf469520bce1c2eceb1ce3e62891921138dec10907f9d00" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f1/84/651a9310859673aaa3b3203f1aa1641ca60fcf2494683e1c9474c7172780/coverage-7.15.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:c705b28feb2775dc82a25f1d473a370bc37ff93f5177f4e29ce2425f560f6921" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/82/f9/4dcf700137e8af550670f4d74d1b63828ce93e1e2b05e5f10710eb2ea987/coverage-7.15.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:3ff205ab5e3ecc670f6a4dd19d9cbf12ede53dd41cfc1e15716ec961ea6d314e" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/07/4a/612ff1e780b3fbfd637486f542f84adc5503873d8b5d279dec1ffeef9414/coverage-7.15.4-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:5172326e861a38b48b48befca15e0f477a26b283337a33a739c8fed229934e36" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b0/04/d1cff1c2ead4708a6a79c01d3736b6a25bd38a36678398f72a8dd33dfad9/coverage-7.15.4-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:12b59c90084e3234fb11184886bf4a40f4f16a8c8f867be2e087b81f8e8868d4" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b9/80/d34e13fb4b293cbdb9665838cf5522077b8ad14ef947550631a4bced36a5/coverage-7.15.4-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:349062d66f00b40fa2c1c222438bad25fabf755631b5d82937fe985c8008615c" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/0f/e7/2c5fe7636fdb0732fe0f09f308a5b066864078b7fc61f6678e8478554f2e/coverage-7.15.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:4256ced708e598e05209bc1a8ab4074e04a51dba4c62fb45926a229af675ace7" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/92/28/9689f0858dfff59c2ea688938ab9fa2925631235df67126a42b6c5c70ae1/coverage-7.15.4-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:d80f974b20782d9612c8b4c9beeca867074c7cf4079d1419843fa25a26428b25" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f9/e2/785077c230c157243eb5aa9a26c3be260ecd02001bead54a3cada3df8e03/coverage-7.15.4-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:2e179f19bfe1d31f8eeeaa12990194d761c4f62f0759661000bca6cd8729f40b" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/d4/90/e20371b17b40f912f21305c2db2f30efa3de306f7320fc916804872c85a4/coverage-7.15.4-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:8bc16bb47b7679670eceff71d78bfb7d6e5b143f6c2cd117487ec7c75e0d4b78" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/05/49/25371987ee459a5f67c0427fb75c74f9358e65f2c71fe75bf41c1b6c5fcb/coverage-7.15.4-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:1cd685005cd2c4200adfc14cf39a603b9320efab3f18a8f7f156d20c9cc3345f" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/30/6e/32e67467f6154bf4f1c4f63b05acc5097cba4237d45bbeeea446b52e8ac1/coverage-7.15.4-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:337399ad2c93b3acd2a937627dae8b3e86b66707cd3d3e856347999aadf1ef8d" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/03/c1/8b24192e89286399765155251f99ee9f070a9d637109018ac23d99b99f6f/coverage-7.15.4-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:96e257121228ec5cd2bb919276e94ac11074471bc37d68dbae0e8308cce15fff" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/16/6f/8b41ebdf67c87854e17c035336a90f1cfbad0c14c2a584301be6ff148718/coverage-7.15.4-cp313-cp313-win32.whl", hash = "sha256:c65a9e0dfc6143491879da4e13b5e30f8be192055de508d737fb14601edbd22c" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e0/e2/2946c7f0b42b152ecb21ff1bdad72e3d301e790c0c487e4a86e8c9f69347/coverage-7.15.4-cp313-cp313-win_amd64.whl", hash = "sha256:2ff8f5e9b8f7a94f0c11c45631eee103dbcb7d63274edd12c56efe1be690b3b4" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/9e/83/3f4a69957f48ae7a0aba76c34743f88963d607b19e03f3f8e66f91cae0f9/coverage-7.15.4-cp313-cp313-win_arm64.whl", hash = "sha256:6e0a8a5083b096487d6cfced94cdd514d8f5db6f113610fb36c0620edb1028cf" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b4/d9/e70c286c979378f061d8266e279b686ab0b0b688e1fe0af864684f23a77d/coverage-7.15.4-py3-none-any.whl", hash = "sha256:964730a1e9de9c0cf11be6a1a3c79ce419c34882842abd256086ba4698705e84" },
]
[[package]]
name = "iniconfig"
version = "2.3.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/72/34/14ca021ce8e5dfedc35312d08ba8bf51fdd999c576889fc2c24cb97f4f10/iniconfig-2.3.0.tar.gz", hash = "sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/cb/b1/3846dd7f199d53cb17f49cba7e651e9ce294d8497c8c150530ed11865bb8/iniconfig-2.3.0-py3-none-any.whl", hash = "sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12" },
]
[[package]]
name = "librt"
version = "0.15.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/36/9b/356320fbae2ac8467e21c5e73e1389c80468e4998c62cc7d3536cc51b614/librt-0.15.0.tar.gz", hash = "sha256:4e66cbe84437497d951b799d3e1551291b6fb3d643820a7014b3655d57a59162" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e7/42/467b53a601b406ccd7b97c1fd54b59cb34f9185ad5ce7e9d5c3c4e8961c8/librt-0.15.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:db13ca398005abcbe538deda87b686d9bd08b7001cf40c4c06b444960ae10a26" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/3e/e6/36c2299b7a94b84fdd01220d8a777a71be5be0925bb0dbdf71c0a06a34d9/librt-0.15.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:aa1f1995789dca3698bc550aaceb09a51bd5df0a057ff84ff15296cd1975b801" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/c9/b6/ed5071f9325845e670bd36012757419767fbf56af77ed483077b9e4db541/librt-0.15.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:55456ea87d8df21808446d03817be2f65e20391c1c615d9187440dff28cd08dc" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/7f/81/6450c67c3615d87704bcbc21323fafc69c799b06a044c447529f725d4b01/librt-0.15.0-cp313-cp313-manylinux2014_i686.manylinux_2_17_i686.manylinux_2_28_i686.whl", hash = "sha256:5a86a5a08c2235316bdb359d5dbb6ce0abfca7fac06363103e2c5af571d92f95" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e1/d6/5f52b722bc75076954b3bfd49be15ea362df4d580c6fb315d0f617100d30/librt-0.15.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:e56b6a368529bed262da40ce13f8fef590db0479819cca84f16a1f01ac356d0b" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8d/e2/c08fd1d36ce63ea5a12b85c5d37f4550b5f86a692167e41e5a74222607ae/librt-0.15.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:234d8d394721fa0d786af15ebf1f3fb7f3ed82fd1cd0cde45c2f247b5d4281d2" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/3f/d8/d9482fcbeb177b9eb87bb3899eeb3b42be690313c652f9e146b1d0681fb2/librt-0.15.0-cp313-cp313-manylinux_2_34_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:d8363d7accb0286ac3a0e633f396e93800dafb8150494505daf9515bbda591f3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/10/cc/075171517b41f861753034fbb151b42cfc83bcc853849f24f5e66fd60ccf/librt-0.15.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:0f0ee3644d951f31055ad07d77d92520e84505dd7a432cc4cd501dd70ee06785" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b0/03/42c2330f37eeb475b6affeedd06518f60035f323af3a839335e3fc9fef2d/librt-0.15.0-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:2cfd1a81a648806e6a7717be4cc4d1bb392fa229752bf8444ba365e381e984d6" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/57/1e/1ad4c5638f7e64d8560328bd25c54b409a661bdb6ff254b38ff90744288d/librt-0.15.0-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:a6cd22c9da0d866558e46a041f1cc0c2bbb26b61b137b2347fa834c332e1d101" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/49/41/39fa7d15db1204cd1cbe6514680fbdc243adf754a0885061308f43afc013/librt-0.15.0-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:6d5225ef8801e4ea5e482fa9b5dfb891dd9ef6f6d870f1f25d449ca2c70ac218" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/1e/88/c6dcf0dd8e26dc0c9a499a2abab8646c86dcaf9ecea9524cb46d3686331a/librt-0.15.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:6d28a05796b99f749bf8794f17ba9ba1612d0076b802e9cfc62c554634e9ce3b" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/1b/9b/ab54c71a7918a7c34fa5327fb61390a77446a07a146fbfb1165250a61035/librt-0.15.0-cp313-cp313-pyemscripten_2025_0_wasm32.whl", hash = "sha256:2067ff438048cead9d223ca5675bae2a25e520a7c3e6c1498bf9c6892d22caab" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8d/b2/4f9a243bb892395f3becb80789ade13771701091f9f07ab8230247953ba8/librt-0.15.0-cp313-cp313-win32.whl", hash = "sha256:1cd3b721f24c206398b9e26da3c3a9c011e6e89d06f318ba8ebefc30f1003890" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/bf/af/64aff4885a40b93132382f2c314647d722574605416504379184ef3045ea/librt-0.15.0-cp313-cp313-win_amd64.whl", hash = "sha256:f395a4a9a03ac062dbe9a9f82e0c720502e590a38feee6a757bc82e9c63afbd8" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/27/83/335bccf6c7cb9028cb0b54aead27d9ece3f01f83bc6baa2abace5da655c1/librt-0.15.0-cp313-cp313-win_arm64.whl", hash = "sha256:0a15cb554761247d84a3ec0cbdf4078d70725384f0e4662c0fa3b26266eb60ad" },
]
[[package]]
name = "loguru"
version = "0.7.3"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "colorama", marker = "sys_platform == 'win32'" },
{ name = "win32-setctime", marker = "sys_platform == 'win32'" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/3a/05/a1dae3dffd1116099471c643b8924f5aa6524411dc6c63fdae648c4f1aca/loguru-0.7.3.tar.gz", hash = "sha256:19480589e77d47b8d85b2c827ad95d49bf31b0dcde16593892eb51dd18706eb6" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/0c/29/0348de65b8cc732daa3e33e67806420b2ae89bdce2b04af740289c5c6c8c/loguru-0.7.3-py3-none-any.whl", hash = "sha256:31a33c10c8e1e10422bfd431aeb5d351c7cf7fa671e3c4df004162264b28220c" },
]
[[package]]
name = "mypy"
version = "2.3.1"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "ast-serialize" },
{ name = "librt", marker = "platform_python_implementation != 'PyPy'" },
{ name = "mypy-extensions" },
{ name = "pathspec" },
{ name = "typing-extensions" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/82/6a/878cc1097d4035f82bd516658d0c528d2a9955bc7b363afcbd0b07fea11b/mypy-2.3.1.tar.gz", hash = "sha256:47c1b1207258513a9d93495f69c8be9de73916186f0e52703e8c461b7a623419" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/de/cf/862010ee800ca9c2bd0c4c0dacf0f092e5411824a09b8f97ad4be8fe250e/mypy-2.3.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:114dff494000f18bd10d5d95d84b8567b26da60279ecbe838131841df20e635d" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/75/5a/3f3a2107b41e3e92e617e25daaee121413b91e9784bea733131ed4fecc5d/mypy-2.3.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c8637731bb5eee3671eb2c3200827aa3564ed8a9309ecee4d1afe77e6d031bdb" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8b/41/04dc4fe7e63d7820fa4eff272e95157d30cbea921388f3ab3fe77794cd0b/mypy-2.3.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:1c80fbc405ed8020f5ff3802dc18cf060197bcdd3fbdd6a26ef2fd34dfdd5226" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/96/fc/c3053b26b9054949285aa868cb6af8c10e7591541cacd79c5dcc06a1fcf9/mypy-2.3.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:84081f538ce27375045c02e3d7f81bd11d853400621ae245d87ce7b6c420ec74" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/70/4e/d77daab008bbc4e5001374d7928f4a260d28f0e6747af444fc4763f7a310/mypy-2.3.1-cp313-cp313-win_amd64.whl", hash = "sha256:e9144ac16fde007096f9563eb2041b4433c2d705c4218edeb79e7e9d01035ee6" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f0/f8/7eb68c136e4abd30569fe31ef2bfcb7eceae9952cab80017c04cd09f5d0c/mypy-2.3.1-cp313-cp313-win_arm64.whl", hash = "sha256:77ad9529e67dca28e511f5cd5671436584ce91f6d3bac159a353158187b986ac" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8e/41/9675c7a1e78edecfba0b79e587a52594c56e189368261dc7b3a7fffb9527/mypy-2.3.1-py3-none-any.whl", hash = "sha256:6ed5c7e3419083268e5c9258bd1c1ef91af44a9e89374dbcaf37b775716e72eb" },
]
[[package]]
name = "mypy-extensions"
version = "1.1.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/a2/6e/371856a3fb9d31ca8dac321cda606860fa4548858c0cc45d9d1d4ca2628b/mypy_extensions-1.1.0.tar.gz", hash = "sha256:52e68efc3284861e772bbcd66823fde5ae21fd2fdb51c62a211403730b916558" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/79/7b/2c79738432f5c924bef5071f933bcc9efd0473bac3b4aa584a6f7c1c8df8/mypy_extensions-1.1.0-py3-none-any.whl", hash = "sha256:1be4cccdb0f2482337c4743e60421de3a356cd97508abadd57d47403e94f5505" },
]
[[package]]
name = "numpy"
version = "2.5.2"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/9a/80/db0b4559e57ec36362bedbb05530a87fafbcb6067708c946967a41d449e7/numpy-2.5.2.tar.gz", hash = "sha256:d482d171c406ae88c5b19cad3b6a1c4c5209f886ab74bc44c2c865c23f52d860" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f5/d2/6b24738a0ef4557d189b150046cd07823c50e4273e8aebd651222e24306f/numpy-2.5.2-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:8e4cb9a754c8a0c62eaa88273a5fba3391f4a610d1dee893c0755da31c083f15" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/65/60/f2d208d366f263f39c6e69ed309290717aab41078b6d04c9be2a84fa2a07/numpy-2.5.2-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:52c808f96484f5571a5cc863775ce50247c17dfb3b0361f8ed6b4b0456f80080" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/3c/79/81e0bf24f4d020a2b1d5cd297a9f60c3f24eeb116f9bba5870443f7b6a4a/numpy-2.5.2-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:29d81e97f668489cba8ebfd796b9bdd453525d35dd9e162e2daec94bf3fc7740" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/ba/cc/e3141cf06d1a8a2c7e107543fe1269c1d1af760d4d683c0794a4ee1127c2/numpy-2.5.2-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:afb3f0632d6b2e3ba04dbce8d1e48d321b369138b73830b5ca371a0e8d479d56" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/29/f1/2a64a307d92c5d98f5255a4014eb43bb6103ee477087b61ecae44a3aa9b9/numpy-2.5.2-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0aadf13b60048d501e05fa699efaf7734e2494f3498a4c2a5521d822640324f3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/7b/44/59a1eb68e773c4098d107ef34a0dbdeca501d72ffcfbff9a7707343921ce/numpy-2.5.2-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:29b86ff8a6cc556b47ec6b64b194815cc80e6bf5eedcc6cddfd65318cb0b4eee" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8a/4c/3e54d4ddbc359a1295f8b633e8106bcd4d7d4a206e82df051bdfb3058755/numpy-2.5.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:6950c4b7dd562453090548ba7f5da7e59f57f85663f15d5dcc60e249192f7e59" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f2/9f/02e371638ebf19b66d46231e4be52999e87f32d1961b113bc45656608b22/numpy-2.5.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:b9727f472d2f3888053b8a75ab0cb94745a9de224bb5846dbadc0092101bc71d" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/eb/ae/ad6645abc7a3510fe48e8ea1ab4598166f500057ef4ebf38bfad4f1577de/numpy-2.5.2-cp313-cp313-win32.whl", hash = "sha256:4f9744f9fbdcea0bc552e8f19e1f141f811a3f9bc2be2cc6e86d982cab23e3f4" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/15/20/f3489f86d81ea460b2bcdceaed094142ca6579f6be0ec527b781d39afe68/numpy-2.5.2-cp313-cp313-win_amd64.whl", hash = "sha256:85aaccb24182c25df891ad0ec333585967e115269d5f1b17f2c9ae005bc96657" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/d5/21/35b31dde1b283b79de828b80f876afd8c94e28fe1e9c375f89e261cc4c0d/numpy-2.5.2-cp313-cp313-win_arm64.whl", hash = "sha256:bd68ece1553d2023c09a4226d9e41c586ad2d20594d1a456186c33513d2cb3f2" },
]
[[package]]
name = "packaging"
version = "26.3"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", hash = "sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", hash = "sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c" },
]
[[package]]
name = "pandas"
version = "3.0.5"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "numpy" },
{ name = "python-dateutil" },
{ name = "tzdata", marker = "sys_platform == 'emscripten' or sys_platform == 'win32'" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/be/4f/5f3422a2afec5ffc46308b79e53291365a93748b498ac2e58bead0197916/pandas-3.0.5.tar.gz", hash = "sha256:dca3734d6ab7c906e6730f0788b0a1dbb9f2467731f9711f77995c8e9d62d712" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/bf/09/7b95c4a0025227d6f118c4039b423412ac6a982db02864166185d812fbc7/pandas-3.0.5-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:c1c05a767fe8e5b4fe9e1c29806829c582052eaedb9120a3da83ba3f69e24a5b" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/8d/0c/dc78fd8c4da477b4b5e8ad37295af352190d21ef63a9ee1bc071753074cc/pandas-3.0.5-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:b86765f268b56f7e665b93bce9d5df69dee7f99e595cf8fb839483ab315942a3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/3e/71/3592c055cf44df9808550f9368ceda80ff2b224d355ef73fe251dcda1802/pandas-3.0.5-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c597ecf5616b5c420372c1d4d4c00dbbfba7398bea857dcc984347e1ea48417b" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e3/70/4363150359f95b4cb4bcbb34ca23572bb5495749a621a8f3d5a1ddfd293c/pandas-3.0.5-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4b11c36e218331d0387cbe3a0a5f75162357a1d92d57b2b08a336ff94b19b2be" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f7/d0/317e7a0c67c0e69fa905a0161409397a7dc2d46ff611f6ca4803352c042b/pandas-3.0.5-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:cf52e1f61d229496da17dc7ab54acdee627357e7008fd4fecba3d0ba2937fa58" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f1/8d/36dade89b49e4f9d5cbdbe863772581f98c0c6d78fc39ad4c557f6f2e17e/pandas-3.0.5-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:db172144bb56422bd157812f3b021eacc255451470b31e2c633c349490a1cfee" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/9c/ba/18c4ec8a746e177da05a9e7a7963781d8ea195780724f854601b6ebd6b78/pandas-3.0.5-cp313-cp313-win_amd64.whl", hash = "sha256:0d298e951f23016ce4699951d044ae6418dbc91bf68cefca0f77666fcbb4e5c6" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/de/ec/28a57266b753799a87b8bc79e7887ac6fd981b8c6d2978a0b7e7b6bd708c/pandas-3.0.5-cp313-cp313-win_arm64.whl", hash = "sha256:66266d3442a5e8b3c90274c2b8b230bee42dd1c286bc822cc2f9f2c7e12b883e" },
]
[[package]]
name = "pathspec"
version = "1.1.1"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/5a/82/42f767fc1c1143d6fd36efb827202a2d997a375e160a71eb2888a925aac1/pathspec-1.1.1.tar.gz", hash = "sha256:17db5ecd524104a120e173814c90367a96a98d07c45b2e10c2f3919fff91bf5a" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f1/d9/7fb5aa316bc299258e68c73ba3bddbc499654a07f151cba08f6153988714/pathspec-1.1.1-py3-none-any.whl", hash = "sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189" },
]
[[package]]
name = "pluggy"
version = "1.6.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/f9/e2/3e91f31a7d2b083fe6ef3fa267035b518369d9511ffab804f839851d2779/pluggy-1.6.0.tar.gz", hash = "sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746" },
]
[[package]]
name = "pygments"
version = "2.21.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/49/2e/ced460408999b33da6b31b0021b0f37d329e202d4169aeb164493778f25b/pygments-2.21.0.tar.gz", hash = "sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/71/46/17f022dd3e953bf20a04a028a21ec746d942f8d2af30fa0f124fa0e6a684/pygments-2.21.0-py3-none-any.whl", hash = "sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9" },
]
[[package]]
name = "pytest"
version = "9.1.1"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "colorama", marker = "sys_platform == 'win32'" },
{ name = "iniconfig" },
{ name = "packaging" },
{ name = "pluggy" },
{ name = "pygments" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/e4/47/b9efed96c114afcfa3c9d3fe98a76a1d14c74a9e266d397cf6eb64be5e01/pytest-9.1.1.tar.gz", hash = "sha256:1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", hash = "sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c" },
]
[[package]]
name = "pytest-asyncio"
version = "1.4.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "pytest" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/43/7c/d36d04db312ecf4298932ef77e6e4a9e8ad017906e24e34f0b0c361a2473/pytest_asyncio-1.4.0.tar.gz", hash = "sha256:c6c0d2259945122819f171a32ecea2c349ead889ee28176caaf492143424be42" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/03/e2/08a497ef684b88559c9cc5f4ad53a37e7b99e727094a86d6ea32536d5d3c/pytest_asyncio-1.4.0-py3-none-any.whl", hash = "sha256:933ca923a23075a87fb7070c0ec272a6848489824d887c85c812670932835aa1" },
]
[[package]]
name = "pytest-cov"
version = "7.1.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "coverage" },
{ name = "pluggy" },
{ name = "pytest" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/b1/51/a849f96e117386044471c8ec2bd6cfebacda285da9525c9106aeb28da671/pytest_cov-7.1.0.tar.gz", hash = "sha256:30674f2b5f6351aa09702a9c8c364f6a01c27aae0c1366ae8016160d1efc56b2" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/9d/7a/d968e294073affff457b041c2be9868a40c1c71f4a35fcc1e45e5493067b/pytest_cov-7.1.0-py3-none-any.whl", hash = "sha256:a0461110b7865f9a271aa1b51e516c9a95de9d696734a2f71e3e78f46e1d4678" },
]
[[package]]
name = "python-dateutil"
version = "2.9.0.post0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "six" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/66/c0/0c8b6ad9f17a802ee498c46e004a0eb49bc148f2fd230864601a86dcf6db/python-dateutil-2.9.0.post0.tar.gz", hash = "sha256:37dd54208da7e1cd875388217d5e00ebd4179249f90fb72437e91a35459a0ad3" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/ec/57/56b9bcc3c9c6a792fcbaf139543cee77261f3651ca9da0c93f5c1221264b/python_dateutil-2.9.0.post0-py2.py3-none-any.whl", hash = "sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427" },
]
[[package]]
name = "quant-engine"
version = "0.1.0"
source = { editable = "." }
dependencies = [
{ name = "loguru" },
{ name = "numpy" },
{ name = "pandas" },
{ name = "scipy" },
]
[package.optional-dependencies]
dev = [
{ name = "mypy" },
{ name = "pytest" },
{ name = "pytest-asyncio" },
{ name = "pytest-cov" },
{ name = "ruff" },
]
[package.metadata]
requires-dist = [
{ name = "loguru", specifier = ">=0.7" },
{ name = "mypy", marker = "extra == 'dev'", specifier = ">=1.10" },
{ name = "numpy", specifier = ">=1.24" },
{ name = "pandas", specifier = ">=2.0" },
{ name = "pytest", marker = "extra == 'dev'", specifier = ">=8.0" },
{ name = "pytest-asyncio", marker = "extra == 'dev'", specifier = ">=0.23" },
{ name = "pytest-cov", marker = "extra == 'dev'", specifier = ">=4.1" },
{ name = "ruff", marker = "extra == 'dev'", specifier = ">=0.4" },
{ name = "scipy", specifier = ">=1.10" },
]
provides-extras = ["dev"]
[[package]]
name = "ruff"
version = "0.16.4"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", hash = "sha256:13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", hash = "sha256:df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", hash = "sha256:8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", hash = "sha256:312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", hash = "sha256:05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", hash = "sha256:a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21" },
]
[[package]]
name = "scipy"
version = "1.18.1"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
dependencies = [
{ name = "numpy" },
]
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/7e/74/66de6258867beb2ef08f35f9f2ac017a52cacd5081714d239ff1a442d458/scipy-1.18.1.tar.gz", hash = "sha256:52c4b7422442aba924d03ad4019852b08a92e64ea187b933135687bfe2747307" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b6/55/4540ee0f9c42a9ad7109d0d1a8cc70de54c3572b01c6693a2b1c70e90ceb/scipy-1.18.1-cp313-cp313-macosx_10_15_x86_64.whl", hash = "sha256:3ab3523da44749156e1f68b464dc56af11ae4cbc5c739a49d05f32b982eca9f3" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/2a/f5/769f36d14922b8071a43e95d24d18b6bdafad10d7f5cf647867e1ac052bc/scipy-1.18.1-cp313-cp313-macosx_12_0_arm64.whl", hash = "sha256:e6fb6a55cc0ba97b59a1f288fb86dc6fce8bdfc0fffcbfd015e3a954bf2a2d93" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/9a/d7/21d890274f75ea37a8209d5519e72da3da90302e3b9fb8397a0918386a62/scipy-1.18.1-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:ea324d9dd34c38bfb9bec8ca4d1b407db97dbb74029f566b8e322b1b6fe56fe6" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/ec/01/798430ecea2e78ec7c02663d5f71c007bb6abeca931080debd40d7fa55ea/scipy-1.18.1-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:75b00eb8fb802090aa903f4ea1c7f5a584779f967361e68b7e98e531cc2d7174" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e6/5f/4634e9d35c68496e4e34cb6946eafab044458e6cedab42b40b6588e475b6/scipy-1.18.1-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d416b16cccfd70fbf62400e84d0bb2f4e6af519a45557f1692c749b37f14b315" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/41/48/6450ed9243315322bbc19ac57b9b70d66a20bf1d38d124c96bc4bf6af9ea/scipy-1.18.1-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fdaf5ea890a6183d0565f51a61799d67081bd5b1cf03c5f4b3fd3732108625c9" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/00/bd/bf5a4be6a3525676499f6dff307991739ff6fdcad1481b1aeb6745339f58/scipy-1.18.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:c825cef2f49e46753726a7181a8e199804a912b29519ada542c6ebc654951899" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/bd/4e/3c45c33e00a77996c4b1cb707929f833ba7b1d522ee29f882512c330676d/scipy-1.18.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:e3b417bf8c2c7c16e8f58ad91db17783ec911ac16e7b50eb6eab6e809b4f5b07" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/93/0e/e0348fbc0dbab65c114cf78957e7dfeb49f8e8b556b4d930cc12ff195e18/scipy-1.18.1-cp313-cp313-win_amd64.whl", hash = "sha256:559ed65f60c1af5a03f3912605a1b5114f522c7c32fb23c3376ae8f03219fe28" },
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/50/a8/6a77f5f267c555108f0a864b6db714363dab567a8266422a79a385f9232b/scipy-1.18.1-cp313-cp313-win_arm64.whl", hash = "sha256:cd479fc04dd9401e3b4f49e76518768ef99c4f517a98c284eb091fd725719adf" },
]
[[package]]
name = "six"
version = "1.17.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/94/e7/b2c673351809dca68a0e064b6af791aa332cf192da575fd474ed7d6f16a2/six-1.17.0.tar.gz", hash = "sha256:ff70335d468e7eb6ec65b95b99d3a2836546063f63acc5171de367e834932a81" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/b7/ce/149a00dd41f10bc29e5921b496af8b574d8413afcd5e30dfa0ed46c2cc5e/six-1.17.0-py2.py3-none-any.whl", hash = "sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274" },
]
[[package]]
name = "typing-extensions"
version = "4.16.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/f6/cc/6253133b5bb138fc3306cebfbda2c520f545d36b5be2c7255cc528bb45d6/typing_extensions-4.16.0.tar.gz", hash = "sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/49/d3/b8441a820a491ddfc024b0b0cf0393375b75ea13866d9c66727e54c2fc80/typing_extensions-4.16.0-py3-none-any.whl", hash = "sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8" },
]
[[package]]
name = "tzdata"
version = "2026.3"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/92/ff/5a28bdfd8c3ebec42564ac7d0e54ca3db65044a9314a97f9564fa7a1e926/tzdata-2026.3.tar.gz", hash = "sha256:4a1518b8993086a7982523e071643f3c0e5f213e75b21318e78bcabfff9d1415" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e5/6d/b53b99a9f2766d095985947a5782f1702cabb129a34f7a802d7197af832f/tzdata-2026.3-py2.py3-none-any.whl", hash = "sha256:dc096730c87af6cab1b171c9d532be840741ff5d459015e7f6947bd7d7e54931" },
]
[[package]]
name = "win32-setctime"
version = "1.2.0"
source = { registry = "https://mirrors.cloud.tencent.com/pypi/simple" }
sdist = { url = "https://mirrors.cloud.tencent.com/pypi/packages/b3/8f/705086c9d734d3b663af0e9bb3d4de6578d08f46b1b101c2442fd9aecaa2/win32_setctime-1.2.0.tar.gz", hash = "sha256:ae1fdf948f5640aae05c511ade119313fb6a30d7eabe25fef9764dca5873c4c0" }
wheels = [
{ url = "https://mirrors.cloud.tencent.com/pypi/packages/e1/07/c6fe3ad3e685340704d314d765b7912993bcb8dc198f0e7a89382d37974b/win32_setctime-1.2.0-py3-none-any.whl", hash = "sha256:95d644c4e708aba81dc3704a116d8cbc974d70b3bdb8be1d150e36be6e9d1390" },
]