Compare commits
14
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8c5af40dad | ||
|
|
d4ee6f005f | ||
|
|
78d65b4db0 | ||
|
|
598c2b92a2 | ||
|
|
62ed09842d | ||
|
|
e782e223f7 | ||
|
|
015c1a3602 | ||
|
|
2bc8aea435 | ||
|
|
03e38d5123 | ||
|
|
90a43adda2 | ||
|
|
e72fe0a8d1 | ||
|
|
fd3014c286 | ||
|
|
38a984b245 | ||
|
|
8a30bf5ebc |
+18
-2
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"module_id": "quant_engine",
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 1, "effective_from": "2026-08-20T00:00:00+08:00"},
|
||||
"authority": {"scope": "module_metadata", "subject": "quant_engine", "owner": "quant-engine-owner", "source": "MODULE_SPEC.yaml", "revision": 4, "effective_from": "2026-09-01T00:00:00+08:00"},
|
||||
"repository": {"name": "quant_engine", "workspace_id": "researchhub", "type": "research_engine", "maturity": "operational"},
|
||||
"bounded_context": {
|
||||
"domain": "quantitative-research-engine",
|
||||
@@ -11,6 +11,7 @@
|
||||
"Submitting live orders, routing trades, managing brokerage accounts, or claiming transaction execution",
|
||||
"Owning market-data source facts, research-result publication, or platform presentation state",
|
||||
"Loading provider credentials, brokerage credentials, or production secrets",
|
||||
"Granting portfolio approval, maker-checker decisions, publication eligibility, paper execution, or live execution authority",
|
||||
"Changing financial model semantics through module metadata"
|
||||
]
|
||||
},
|
||||
@@ -18,13 +19,28 @@
|
||||
{"id": "factor-and-indicator-calculation", "summary": "Calculate reusable alpha factors and technical indicators from caller-supplied data.", "status": "operational"},
|
||||
{"id": "execution-simulation", "summary": "Simulate costs, slippage, market constraints, fills, NAV, and PnL without live order routing.", "status": "operational"},
|
||||
{"id": "portfolio-backtesting", "summary": "Run weight-based backtests and benchmark comparisons.", "status": "operational"},
|
||||
{"id": "backtest-evidence-contracts", "summary": "Identify governed offline backtest inputs and close existing research artifact evidence without persistence or decision authority.", "status": "operational"},
|
||||
{"id": "portfolio-risk-computation-contracts", "summary": "Verify deterministic portfolio-computation receipts and expose S3-bound portfolio decisions and risk assessments without adding algorithms or execution authority.", "status": "operational"},
|
||||
{"id": "risk-and-performance-analysis", "summary": "Calculate portfolio decomposition, risk contribution, and performance statistics.", "status": "operational"}
|
||||
],
|
||||
"data": {"owns": [
|
||||
{"asset_id": "quantitative-model-implementations", "kind": "model", "classification": "internal"},
|
||||
{"asset_id": "simulation-and-metric-results", "kind": "artifact", "classification": "confidential"}
|
||||
]},
|
||||
"contracts": {"provides": [], "consumes": []},
|
||||
"contracts": {
|
||||
"provides": [
|
||||
{"contract_id": "researchhub.factor-definition", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
|
||||
{"contract_id": "researchhub.factor-set-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/factor_contracts.py"},
|
||||
{"contract_id": "researchhub.backtest-run-ref", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/governed_pipeline.py"},
|
||||
{"contract_id": "researchhub.backtest-evidence-manifest", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/artifact.py"},
|
||||
{"contract_id": "researchhub.portfolio-decision", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"},
|
||||
{"contract_id": "researchhub.risk-assessment", "version": "1.0.0", "authority": "quant_engine", "path": "src/quant_engine/portfolio_risk_contracts.py"}
|
||||
],
|
||||
"consumes": [
|
||||
{"contract_id": "researchhub.dataset-snapshot", "version": "1.0.0", "authority": "researchhub.data", "admission": "qualified_immutable_envelope"},
|
||||
{"contract_id": "researchhub.data-foundation", "version": "1.0.0", "authority": "researchhub.data", "admission": "content_addressed_selected_views"}
|
||||
]
|
||||
},
|
||||
"dependencies": [],
|
||||
"agent_context": {
|
||||
"default_entrypoints": [
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
|
||||
| 仓库 | 角色 |
|
||||
|---|---|
|
||||
| `quant_engine` | **纯回测核心**(alpha + execution + indicators + data_adapter + backtest + metrics) |
|
||||
| `quant_engine` | **纯研究核心**(alpha + execution + ledger + attribution + risk + metrics) |
|
||||
| `research_results` | 业务集成(47 个 proj 调度 + 注册 + 平台对接) |
|
||||
| `tushare2db_pro_aoge` | 数据层(行情 ELT) |
|
||||
| `research_platform` | 展示层(FastAPI + Next.js) |
|
||||
@@ -19,14 +19,21 @@
|
||||
## 模块
|
||||
|
||||
- `alpha_factors` — 158 alpha 公式 + 24 基础算子(移植自 qlib alpha158)
|
||||
- `execution` — 执行仿真(成本/滑点/T+1/涨跌停/部分成交/价差)+ 多日 NAV + PnL 拆解(借鉴 hikyuu 部件化思想)
|
||||
- `factor_contracts` — `FactorDefinition` / `FactorSetRef` v1 纯计算合同、严格 PIT/availability 输入准入与显式 legacy 投影
|
||||
- `execution` — A 股长仓执行仿真(成本/滑点/现金约束)+ 稀疏调仓/完整交易日 Ledger + 可投影成交与 NAV 审计;T+1、涨跌停、成交量与价差提供独立约束函数
|
||||
- `indicators` — 50+ 技术指标(MACD / KDJ / 布林 / ATR / ADX / 等)
|
||||
- `data_adapter` — 桥接 qtdb_pro 长表与新模块(rename / long-wide / 复权 / vwap 代理)
|
||||
- `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark)
|
||||
- `metrics` — 绩效(年化收益 / 波动率 / Sharpe / 最大回撤 / Calmar)
|
||||
- `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表
|
||||
- `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排)
|
||||
- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;同时拥有输入/配置/重放血缘决定的 `BacktestRunRef`
|
||||
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表,以及只映射现有表的 `BacktestEvidenceManifest`
|
||||
- `portfolio_risk_contracts` — S3 证据闭合的 `PortfolioDecision` / `RiskAssessment` v1;独立复核 freshness、约束与 computation receipt,并复用既有标签安全风险分解
|
||||
- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计
|
||||
- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效
|
||||
- `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test)
|
||||
- `portfolio_decomp` — 组合分解(risk_parity / mean_variance / 因子归因)
|
||||
- `risk` — 风险指标(边际 / 风险贡献)
|
||||
- `risk` — ndarray 低层风险公式 + 标签安全、可分组的 Euler 成分风险分解
|
||||
- `perf_stats` — 详细绩效(与 metrics 并存)
|
||||
- `logging` — 统一 logger(标准库 + 可选 loguru)
|
||||
|
||||
@@ -51,6 +58,9 @@ pytest # 单元测试
|
||||
pytest --cov=src # 覆盖率
|
||||
mypy --strict src/ # 类型检查
|
||||
ruff check src/ tests/ # lint
|
||||
|
||||
# 无网络、无数据库、无券商的架构烟测
|
||||
uv run python -m quant_engine.governed_pipeline
|
||||
```
|
||||
|
||||
## 使用
|
||||
@@ -58,8 +68,13 @@ ruff check src/ tests/ # lint
|
||||
```python
|
||||
from quant_engine.alpha_factors import alpha_001, alpha_005, ALPHA158_REGISTRY
|
||||
from quant_engine.execution import (
|
||||
ExecutionConfig, simulate_with_daily_data, compute_realized_pnl,
|
||||
ExecutionConfig, simulate_daily_ledger_with_audit,
|
||||
simulate_multi_day_with_audit, simulate_with_daily_data,
|
||||
)
|
||||
from quant_engine.research_pipeline import (
|
||||
run_factor_backtest_research, run_factor_execution_research,
|
||||
)
|
||||
from quant_engine.backtest import run_weight_backtest
|
||||
from quant_engine.indicators import macd, bollinger, kdj
|
||||
from quant_engine.data_adapter import (
|
||||
long_to_wide, wide_to_long, rename_tushare_columns,
|
||||
@@ -70,10 +85,303 @@ from quant_engine.data_adapter import (
|
||||
|
||||
# 端到端:qtdb_pro 长表 → 适配 → alpha158 → execution
|
||||
df = load_qtdb_daily(["000001.SZ"], "2024-01-01", with_adj=True)
|
||||
prices, volumes = prepare_execution_inputs(df)
|
||||
result = simulate_with_daily_data(prices, initial_cash=1_000_000.0)
|
||||
close_prices, volumes = prepare_execution_inputs(df)
|
||||
open_prices, _ = prepare_execution_inputs(df, price_col="open")
|
||||
result = simulate_with_daily_data(close_prices, initial_cash=1_000_000.0)
|
||||
|
||||
# 已正确滞后的目标权重 → 现金约束执行 → 唯一来源的成交/拒绝/日末持仓/NAV
|
||||
execution = simulate_multi_day_with_audit(
|
||||
target_weights_history=[
|
||||
("2024-01-02", {"000001.SZ": 1.0}),
|
||||
("2024-01-03", {"000001.SZ": 1.0}),
|
||||
],
|
||||
price_history=[
|
||||
("2024-01-02", {"000001.SZ": 10.0}),
|
||||
("2024-01-03", {"000001.SZ": 10.5}),
|
||||
],
|
||||
initial_cash=1_000_000.0,
|
||||
config=ExecutionConfig(),
|
||||
)
|
||||
print(execution.nav_series)
|
||||
print(execution.daily_executions)
|
||||
|
||||
# 多期因子分数(必须是 point-in-time 数据)→ Top-K → 下一交易日 open 执行
|
||||
factor_execution = run_factor_execution_research(
|
||||
factor_scores,
|
||||
top_k=20,
|
||||
execution_prices=open_prices,
|
||||
execution_price_field="open",
|
||||
initial_cash=1_000_000.0,
|
||||
)
|
||||
|
||||
# 推荐研究入口:同一交易日历上显式区分 open 成交和 close 估值。
|
||||
# 因子日保持现金,下一交易日成交后的真实持仓才参与当日收盘收益。
|
||||
factor_backtest = run_factor_backtest_research(
|
||||
factor_scores,
|
||||
top_k=20,
|
||||
execution_prices=open_prices,
|
||||
valuation_prices=close_prices,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000_000.0,
|
||||
config=ExecutionConfig(),
|
||||
)
|
||||
print(factor_backtest.nav)
|
||||
print(factor_backtest.returns)
|
||||
print(factor_backtest.stats())
|
||||
print(factor_backtest.execution.ledger_frame)
|
||||
print(factor_backtest.execution.trades_frame)
|
||||
print(factor_backtest.position_weights) # 实际日末资产权重
|
||||
print(factor_backtest.cash_weights)
|
||||
|
||||
# 所有分析都以实际成交后的 Ledger 为事实源,不直接使用目标权重伪造结果。
|
||||
attribution = factor_backtest.return_attribution()
|
||||
print(attribution.asset_contributions)
|
||||
print(attribution.transaction_cost)
|
||||
print(attribution.residual) # 应接近 0;否则说明贡献未闭合到账本收益
|
||||
|
||||
# benchmark_returns 必须与成本后 factor_backtest.returns 使用完全相同的日期索引。
|
||||
print(factor_backtest.benchmark_stats(benchmark_returns))
|
||||
|
||||
# 下游稳定交付:显式提供代码版本、数据快照和时区,不在核心层写数据库。
|
||||
from quant_engine.artifact import build_research_run_artifact
|
||||
from quant_engine.data_adapter import prepare_asset_return_snapshot
|
||||
from quant_engine.risk import estimate_covariance_snapshot
|
||||
|
||||
risk_date = factor_backtest.position_weights.index[-1].date()
|
||||
market_snapshot = prepare_asset_return_snapshot(
|
||||
qtdb_daily_long,
|
||||
source="qtdb_pro.hq_daily",
|
||||
source_snapshot_id="<upstream-ingestion-snapshot-id>",
|
||||
adjustment="qfq",
|
||||
)
|
||||
risk_snapshot = estimate_covariance_snapshot(
|
||||
market_snapshot.returns,
|
||||
as_of_date=risk_date,
|
||||
lookback_sessions=252,
|
||||
min_observations=120,
|
||||
data_snapshot_id=market_snapshot.data_snapshot_id,
|
||||
)
|
||||
|
||||
artifact = build_research_run_artifact(
|
||||
factor_backtest,
|
||||
run_id="research-run-001",
|
||||
strategy_id="alpha-top20",
|
||||
strategy_name="Alpha Top 20",
|
||||
strategy_version="1.0.0",
|
||||
engine_version="1.2.0",
|
||||
code_revision="<git-sha>",
|
||||
data_snapshot_id=market_snapshot.data_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-08-21T10:00:00+08:00",
|
||||
finished_at="2026-08-21T10:01:00+08:00",
|
||||
parameters={"top_k": 20, "lag_sessions": 1},
|
||||
benchmark_id="000300.SH",
|
||||
benchmark_returns=benchmark_returns,
|
||||
risk_snapshots={risk_date: risk_snapshot},
|
||||
)
|
||||
print(artifact.manifest())
|
||||
|
||||
# run_weight_backtest 是低层算子:只接受收益区间开始前已经生效的持仓权重。
|
||||
# 不要把 signal-date 的 factor_scores/decision_weights 直接传给它。
|
||||
backtest = run_weight_backtest(
|
||||
weights=effective_holding_weights,
|
||||
stock_returns=daily_returns,
|
||||
initial_capital=1_000_000.0,
|
||||
benchmark_nav=benchmark_nav,
|
||||
)
|
||||
print(factor_execution.schedule.signal_to_execution)
|
||||
print(factor_execution.execution.daily_executions)
|
||||
print(backtest.stats())
|
||||
print(backtest.benchmark_report())
|
||||
```
|
||||
|
||||
## 因子/特征合同 v1
|
||||
|
||||
`quant_engine.factor_contracts` 提供 `researchhub.factor-definition` 与
|
||||
`researchhub.factor-set-ref` `1.0.0`。合同使用受限 canonical JSON:只接受 ASCII
|
||||
lower-snake-case object key、UTF-8 string、bool/null 和 safe integer;小数参数必须用显式
|
||||
canonical decimal string。定义、输入映射、上游证据、输出 schema/content 和 lineage 的任一
|
||||
语义变化都会产生新 identity。
|
||||
|
||||
创建 `FactorSetRef` 必须提供完整且可重算 identity 的 `DatasetSnapshotEnvelope` 与
|
||||
`DataFoundationEnvelope`,不能用 ID 字符串或布尔值代替资格证明。每个因子输入都要映射到一个
|
||||
实际选中的 `StandardizedViewRef`,schema 必须同时匹配定义和 view;未消费、缺失、重复或跨
|
||||
snapshot/Foundation/PIT 的 view 都会失败关闭。snapshot PIT 可以早于 Foundation/view PIT,
|
||||
但始终满足 knowledge ≤ snapshot PIT ≤ Foundation/view/FactorSet PIT ≤ evaluation。
|
||||
|
||||
```python
|
||||
from quant_engine.factor_contracts import (
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorDefinition,
|
||||
FactorSetRef,
|
||||
)
|
||||
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(dataset_snapshot_v1)
|
||||
foundation = DataFoundationEnvelope.from_dict(data_foundation_v1)
|
||||
|
||||
# definition 必须是完整的 FactorDefinition;FactorSetRef.create 还要求显式 input bindings、
|
||||
# view availability、output quality/coverage、canonical output bytes 和 immutable artifact ref。
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
**explicit_factor_set_evidence,
|
||||
)
|
||||
```
|
||||
|
||||
`availability_mode="as_available"` 声明 source/view 和计算产物在历史 evaluation 前实际可用;
|
||||
`"retrospective_replay"` 保留历史 evaluation,但要求真实 publication/view creation、compute 和
|
||||
artifact 时间位于之后,并固定 `historical_availability="not_established"`。两种模式都不会授予
|
||||
decision、real-data、production、paper 或 live readiness。
|
||||
|
||||
旧 `governed_pipeline.FactorVersion` 的四字段构造器、`factor_id@version`、run/target/risk/order
|
||||
identity 均保持不变。迁移只能通过 content-addressed `LegacyFactorBinding`,再显式调用
|
||||
`bind_legacy_factor()` 或 `project_legacy_factor()`;后者是有损投影,不表示旧 digest 与新定义
|
||||
digest 等价,也不会把旧 run 静默升级为新合同。
|
||||
|
||||
## 回测引用与证据合同 v1
|
||||
|
||||
`quant_engine.governed_pipeline.BacktestRunRef` 是合格回测运行身份的唯一权威。`run_id` 只由
|
||||
已验收的 Dataset Snapshot / Data Foundation / `FactorSetRef` 身份、universe、日历与公司行动
|
||||
祖先、策略、执行/成本模型、严格整数 seed、完整代码提交、环境锁、配置、时间和重放血缘决定;
|
||||
它不包含任何输出摘要。重放必须绑定直接父运行、连续 attempt 和不变的
|
||||
`replay_spec_digest`,输入漂移或血缘环会失败关闭。
|
||||
|
||||
`quant_engine.artifact.BacktestEvidenceManifest` 只摘要 `ResearchRunArtifact` 已有的九张事实表。
|
||||
固定 `offline_research_v1` 映射为 `run`、`signal`、`fill`、`position_nav`、`performance`、
|
||||
`attribution`、`risk_snapshot` 与 `replay`;每张表都保留列模式摘要、行数和内容摘要,空 risk
|
||||
表也必须有稳定 schema。`manifest_id` 由完整 RunRef 与输出证据决定,因此结果变化不会反向改变
|
||||
`run_id`。当前 artifact 不拥有订单或拒绝事实,所以此画像明确不声明 `order` / `rejection`。
|
||||
|
||||
旧 `BacktestRun` 只能通过 `build_legacy_backtest_evidence_manifest()` 显式映射为
|
||||
`LEGACY_EXPLORATORY`;不能隐式提升为 `CONTRACT_QUALIFIED`。所有资格均只描述离线证据闭合,
|
||||
不表示投资有效、组合获批、Paper、生产或实盘就绪。
|
||||
|
||||
## 组合决策与风险评估合同 v1
|
||||
|
||||
`quant_engine.portfolio_risk_contracts` 是现有计算 owner 外围的薄合同层。创建
|
||||
`PortfolioDecision` 必须同时提供完整 `BacktestRunRef`、嵌入同一 RunRef 的
|
||||
`CONTRACT_QUALIFIED` 非 legacy `BacktestEvidenceManifest`、现有 `PortfolioTarget`、
|
||||
`FreshnessPolicy`、`ConstraintSetV1` 与 `ComputationReceipt`。适配器会从权威输入独立重算
|
||||
receipt 的 input/constraint/output digest、敞口、持仓数和 L1 turnover 残差;receipt 自报
|
||||
成功、fallback 或放宽 tolerance 均不能替代复核。
|
||||
|
||||
```python
|
||||
from quant_engine.portfolio_risk_contracts import (
|
||||
ComputationReceipt,
|
||||
ConstraintSetV1,
|
||||
FreshnessPolicy,
|
||||
assess_portfolio_risk,
|
||||
build_portfolio_decision,
|
||||
compute_portfolio_receipt_digests,
|
||||
)
|
||||
|
||||
freshness = FreshnessPolicy(
|
||||
max_manifest_age_seconds=3600,
|
||||
max_covariance_age_days=5,
|
||||
)
|
||||
constraints = ConstraintSetV1(
|
||||
gross_exposure_max=1.0,
|
||||
single_asset_max=0.10,
|
||||
position_count_max=20,
|
||||
turnover_max=0.30,
|
||||
)
|
||||
|
||||
# 生产者先形成公开 canonical digest;decision 构建时仍会独立重算。
|
||||
expected = compute_portfolio_receipt_digests(
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
target=portfolio_target,
|
||||
objective_name="long_only_allocation",
|
||||
objective_version="1.0.0",
|
||||
objective_digest=objective_digest,
|
||||
model_name="factor_weighting",
|
||||
model_version="1.0.0",
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
|
||||
receipt = ComputationReceipt(
|
||||
algorithm="factor_weighting",
|
||||
algorithm_version="1.0.0",
|
||||
implementation_digest=implementation_digest,
|
||||
parameter_digest=parameter_digest,
|
||||
input_digest=expected["input_digest"],
|
||||
constraint_digest=expected["constraint_digest"],
|
||||
output_digest=expected["output_digest"],
|
||||
status="completed",
|
||||
solver_required=False,
|
||||
solver_name=None,
|
||||
solver_version=None,
|
||||
solver_config_digest=None,
|
||||
iterations=None,
|
||||
objective_value=None,
|
||||
max_constraint_residual=expected["max_constraint_residual"],
|
||||
tolerance=1e-12,
|
||||
computed_at=computed_at,
|
||||
)
|
||||
|
||||
decision = build_portfolio_decision(
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
target=portfolio_target,
|
||||
objective_name="long_only_allocation",
|
||||
objective_version="1.0.0",
|
||||
objective_digest=objective_digest,
|
||||
model_name="factor_weighting",
|
||||
model_version="1.0.0",
|
||||
model_digest=model_digest,
|
||||
expected_return_digest=expected_return_digest,
|
||||
covariance_digest=covariance_digest,
|
||||
scenario_digest=scenario_digest,
|
||||
constraints=constraints,
|
||||
freshness_policy=freshness,
|
||||
receipt=receipt,
|
||||
computed_at=computed_at,
|
||||
prior_weights=prior_weights,
|
||||
)
|
||||
|
||||
assessment = assess_portfolio_risk(
|
||||
portfolio_decision=decision,
|
||||
backtest_run_ref=run_ref,
|
||||
manifest=evidence_manifest,
|
||||
covariance=covariance_snapshot,
|
||||
risk_model_name="euler_volatility",
|
||||
risk_model_version="1.0.0",
|
||||
risk_model_digest=risk_model_digest,
|
||||
)
|
||||
```
|
||||
|
||||
`source_universe_digest` 保留 S3 研究 universe 身份,`portfolio_asset_set_digest` 只描述实际
|
||||
目标资产标签;二者不会互相冒充成员证明。风险评估在任何数值计算前要求 covariance、target、
|
||||
RunRef 的 dataset identity 三方一致,并且只调用一次现有 `labeled_component_risk()`。合同中的
|
||||
`qualified` 仅表示 S4.1 计算证据闭合,不授予 maker-checker、发布、订单、Paper、生产或实盘权限。
|
||||
|
||||
## 治理垂直切片
|
||||
|
||||
`governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。
|
||||
调用方必须显式提供 `DatasetSnapshot`、`FactorVersion`、`StrategyVersion`、代码提交和
|
||||
`RiskPolicy`。模块只会生成 `environment="paper"` 的订单意图,不连接数据库、数据供应商或
|
||||
券商;风险决策为拒绝时,订单意图固定为空,直接调用创建函数也会失败关闭。
|
||||
|
||||
该切片对应 ResearchHub 架构的首个可执行验收链路:
|
||||
|
||||
```text
|
||||
DatasetSnapshot → FactorVersion → StrategyVersion → BacktestRun
|
||||
→ PortfolioTarget → RiskDecision → PaperOrderIntent
|
||||
```
|
||||
|
||||
平台总架构、五仓职责和十二层能力映射仍以 `research_platform/docs/architecture/` 为权威;
|
||||
本仓只拥有纯计算与离线模拟合同。
|
||||
|
||||
## 与 research_results 的关系
|
||||
|
||||
`research_results` 依赖 `quant_engine`(通过 re-export 保持向后兼容):
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# Open-source design references
|
||||
|
||||
本项目采用“借鉴稳定语义、保留轻量实现”的策略。引入新量化能力前先检查成熟
|
||||
开源案例;除非维护成本和许可证收益明确优于本地小型实现,否则不增加框架级依赖。
|
||||
|
||||
## 2026-08-21:成交后归因与相对绩效
|
||||
|
||||
| 项目 | 借鉴内容 | 当前决策 |
|
||||
|---|---|---|
|
||||
| [Qlib](https://github.com/microsoft/qlib) | 信号时间与交易时间分离、成本前后超额收益分开报告 | 借鉴语义;不引入完整框架 |
|
||||
| [Zipline](https://github.com/quantopian/zipline) | Ledger / transaction / portfolio value 状态模型 | 以现有 `ExecutionSimulationResult` 承担事实源 |
|
||||
| [empyrical](https://github.com/quantopian/empyrical) | beta 协方差口径、alpha 几何年化、年化因子 | 移植小型公式;不增加老旧运行时依赖 |
|
||||
| [Riskfolio-Lib](https://github.com/dcajasn/Riskfolio-Lib) | Euler component risk 与分组/因子风险贡献 | 只实现当前需要的 pandas/numpy 标签安全封装 |
|
||||
| [PyPortfolioOpt](https://github.com/PyPortfolio/PyPortfolioOpt) | 协方差估计与优化器解耦 | 留作未来风险模型适配器参考 |
|
||||
|
||||
当前核心不新增依赖。逐日收益归因必须从实际换仓前后持仓、成交记录、执行价和
|
||||
收盘估值推导;因子分数与目标权重只是意图,不能作为成交后归因事实源。
|
||||
|
||||
## 2026-08-21:研究运行工件
|
||||
|
||||
- 借鉴 [Qlib Recorder / RecordTemplate](https://github.com/microsoft/qlib/blob/main/qlib/workflow/record_temp.py)
|
||||
将 signal、portfolio analysis 和 risk analysis 分成稳定事实,但不引入 Qlib 运行时;
|
||||
- 借鉴 [MLflow Tracking](https://mlflow.org/docs/latest/tracking/) 的 run / params /
|
||||
metrics / artifacts 分层,但 MLflow 只保留为未来可选 exporter;
|
||||
- HTML、PNG 和 tearsheet 是可再生展示物,不能替代 NAV、成交、持仓、归因和绩效事实。
|
||||
|
||||
因此 `ResearchRunArtifact` 使用显式 `schema_version`、`config_hash`、代码版本和数据
|
||||
快照身份,并提供确定性 JSON / SHA-256 manifest;核心层仍不写数据库或 artifact store。
|
||||
|
||||
schema `1.1.0` 将 Qlib 的独立 risk-analysis artifact 思路与 Riskfolio-Lib 的 Euler
|
||||
component-risk 语义结合,但只保留本项目需要的轻量合同:协方差快照必须声明
|
||||
`snapshot_id`、`as_of_date`、收益频率和年化期数;风险从成交后的实际日末持仓计算,
|
||||
component risk 闭合到年化组合波动,percentage contribution 闭合到 1。未来日期、资产
|
||||
标签不完整和零方差组合都直接失败,不以默认值伪造结果。
|
||||
|
||||
## 2026-08-21:协方差快照估计
|
||||
|
||||
| 项目 | 借鉴内容 | 当前决策 |
|
||||
|---|---|---|
|
||||
| [PyPortfolioOpt risk models](https://github.com/PyPortfolio/PyPortfolioOpt/blob/main/pypfopt/risk_models.py) | 将收益输入、协方差估计器和组合优化解耦;sample / EWM / shrinkage 使用统一标签输出 | 借鉴可替换估计器边界,不引入完整包 |
|
||||
| [scikit-learn covariance](https://github.com/scikit-learn/scikit-learn/blob/main/sklearn/covariance/_shrunk_covariance.py) | 维护成熟的 Ledoit–Wolf / OAS shrinkage 实现 | 未来作为可选 adapter;不复制统计公式 |
|
||||
| [Qlib structured risk model](https://github.com/microsoft/qlib/blob/main/qlib/model/riskmodel/structured.py) | PCA/FA 结构化协方差和固定随机状态 | 留作因子风险模型阶段,不进入当前 baseline |
|
||||
|
||||
当前 `estimate_covariance_snapshot` 只编排 pandas 的 sample covariance:先按 `as_of_date`
|
||||
截断,再取固定 session 窗口,使用 complete-case 行并拒绝历史不足;禁止 pandas 默认的
|
||||
pairwise 样本集合产生含义不一致的矩阵。snapshot ID 对窗口数据、缺失掩码、上游数据
|
||||
快照身份和估计参数做 SHA-256,追加未来数据不会改变历史快照。
|
||||
|
||||
市场适配层现以 `AssetReturnSnapshot` 固化 simple-return 输入:上游 ingestion snapshot ID、
|
||||
数据源、价格字段、复权口径、规范化价格值和缺失掩码共同形成内容寻址 ID;不前向填充
|
||||
停牌/缺失价格。该 ID 同时传入协方差快照和研究运行工件,避免同一研究链出现两套数据
|
||||
身份。
|
||||
|
||||
可选 shrinkage adapter 的评估结论是“保留边界,暂不实现”:当前运行依赖没有声明
|
||||
scikit-learn,本切片也不修改版本或锁文件。未来只有在依赖治理接受后,才以延迟导入
|
||||
直接调用 scikit-learn 的 `LedoitWolf` / `OAS`,并让估计器名称、库版本与参数进入
|
||||
snapshot identity;不复制成熟统计公式,也不让环境中偶然存在的包改变 baseline 行为。
|
||||
|
||||
## hikyuu 的定位
|
||||
|
||||
[hikyuu](https://github.com/fasiondog/hikyuu) 的 SG / MM / CN / PG 部件化思想、
|
||||
A 股交易约束和系统组合方式仍有借鉴价值;但其完整 C++/Python 运行时、对象模型和
|
||||
数据体系不适合作为本项目核心依赖。当前原则是按真实研究链路吸收边界设计,不复制
|
||||
其框架层级,也不为了“架构完整”预先建设尚无端到端需求的抽象。
|
||||
@@ -0,0 +1,42 @@
|
||||
# Ledger-backed attribution handoff
|
||||
|
||||
## Goal
|
||||
|
||||
在 `ExecutionSimulationResult` 日频 Ledger 之上增加轻量、可审计的成交后分析层:
|
||||
|
||||
- 逐日隔夜 / 日内资产收益贡献;
|
||||
- 佣金、印花税、滑点成本独立贡献;
|
||||
- 贡献闭合到成本后日收益并显式暴露 residual;
|
||||
- 严格日期对齐的 TE / IR / alpha / beta;
|
||||
- 标签安全且可分组的 Euler component risk。
|
||||
- 从 Ledger 股数和收盘估值投影的实际资产 / 现金权重。
|
||||
|
||||
## Branch stack
|
||||
|
||||
- 当前:`codex/ledger-attribution-20260821`
|
||||
- 基线:`codex/post-execution-ledger-20260821`
|
||||
- 再下层:`codex/core-contracts-20260821`(PR #2,尚待用户确认合并)
|
||||
|
||||
本分支不得直接合并到 `main`。应按上述顺序逐层审阅;未经用户明确确认,不得合并
|
||||
L2 PR。
|
||||
|
||||
## Open-source decision
|
||||
|
||||
调研结论记录在 `docs/OPEN_SOURCE_REFERENCES.md`。Qlib、Zipline、empyrical、
|
||||
Riskfolio-Lib 和 PyPortfolioOpt 只作为时间语义、Ledger、相对指标与 Euler 风险贡献
|
||||
的设计参考;本阶段没有新增运行时依赖。
|
||||
|
||||
## Verification
|
||||
|
||||
- `pytest -q --cov=src --cov-report=term-missing`: 514 passed,9 个既有 SciPy warning,91% coverage;
|
||||
- `mypy --strict src/`: 15 source files passed;
|
||||
- 变更范围 `ruff check`: passed;
|
||||
- 全仓 Ruff:仅 13 个既有 `tests/governance/*` PT009;
|
||||
- workspace verify/status:passed,预期提示 quant_engine 非 main;
|
||||
- global Gitea workflow check:passed,23 个无关仓库 warning。
|
||||
|
||||
## Next action
|
||||
|
||||
先按堆叠顺序审阅 PR。基础 Ledger 分支完成后,再将本分支 rebase 到其最终提交,
|
||||
运行唯一一次 `ship --ready`;随后将稳定输出适配到 `research_results` 与
|
||||
`research_platform`,不要在核心层直接写数据库。
|
||||
@@ -0,0 +1,33 @@
|
||||
# Post-execution daily Ledger handoff
|
||||
|
||||
## 状态
|
||||
|
||||
- 分支:`codex/post-execution-ledger-20260821`
|
||||
- 基线:`codex/core-contracts-20260821`(PR #2,尚未获用户确认合并)
|
||||
- 本分支不得直接合并到 `main`;先等待 PR #2 合并,再整理基线并创建独立 PR。
|
||||
- 无账户、券商、数据库或实盘副作用。
|
||||
|
||||
## 已完成
|
||||
|
||||
- 新增稀疏调仓、完整交易日估值的 `simulate_daily_ledger_with_audit()`。
|
||||
- 显式分离 execution price 与 valuation price,支持下一日 open 成交、当日 close 估值。
|
||||
- 成交记录补齐 `side / quantity / price`,并提供 `trades_frame`。
|
||||
- 提供平台中立的 `ledger_frame`,不携带 `run_id`,不写数据库。
|
||||
- 新增 `run_factor_backtest_research()`:PIT 因子、下一交易日执行、日频 NAV、首日成本收益和标准绩效。
|
||||
- 研究区间从首条有效信号日开始,排除因子预热行情对绩效的稀释。
|
||||
|
||||
## 验证
|
||||
|
||||
- `pytest -q --cov=src --cov-report=term-missing`:500 passed,total coverage 91%。
|
||||
- `mypy --strict src/`:14 source files passed。
|
||||
- 本阶段文件 scoped Ruff:passed。
|
||||
- 全仓 Ruff:仅既有 governance tests 的 13 个 PT009 基线问题。
|
||||
- workspace verify/status:通过;仅提示功能分支不是引导基线 `main`。
|
||||
- 全局 Gitea workflow check:通过,23 个既有警告。
|
||||
|
||||
## 继续步骤
|
||||
|
||||
1. 获得用户对 PR #2 的明确合并确认并按 L2 流程合并。
|
||||
2. 将本分支整理到更新后的 `main`,重新运行相同全量验证。
|
||||
3. 为 Ledger 阶段创建独立 PR,执行唯一一次最终 `ship --ready`,等待用户确认合并。
|
||||
4. 后续在 `research_results` 增加业务投影适配器,再由 `research_platform` 持久化和展示;核心层继续保持无数据库写入。
|
||||
@@ -0,0 +1,57 @@
|
||||
# Research artifact contract handoff
|
||||
|
||||
## Goal
|
||||
|
||||
把完整可信研究链固化成存储中立、版本化、确定性的 `ResearchRunArtifact`,供
|
||||
`research_results` 持久化和 `research_platform` 查询:
|
||||
|
||||
- run identity / schema version / config hash / code revision / data snapshot;
|
||||
- signal scores / decision weights / signal-to-execution mapping;
|
||||
- NAV / returns / benchmark / costs;
|
||||
- trades / realized positions / cash;
|
||||
- asset and daily return attribution;
|
||||
- performance including Sortino / TE / IR / alpha / beta;
|
||||
- reproducible covariance snapshots and annualized Euler component-risk facts;
|
||||
- canonical JSON / SHA-256 manifest。
|
||||
|
||||
## Branch stack
|
||||
|
||||
- 当前:`codex/research-artifact-contract-20260821`
|
||||
- 基线:`codex/ledger-attribution-20260821`(Draft PR #4)
|
||||
- 下层:Draft PR #3 → Ready PR #2 → `main`
|
||||
|
||||
不得绕过堆叠顺序直接合并到 `main`。
|
||||
|
||||
## Verification
|
||||
|
||||
- `pytest -q`: 540 passed,9 个既有 SciPy warning;
|
||||
- data-adapter focused coverage 77%(包含未连接真实 ClickHouse 的 I/O 便捷函数);
|
||||
- `mypy --strict src/`: 16 source files passed;
|
||||
- changed-scope Ruff: passed;
|
||||
- no runtime dependency added;
|
||||
- no database, network, broker or filesystem write side effect in artifact builder。
|
||||
- 三仓隔离 ClickHouse 黄金链路通过:市场价格 → return snapshot → covariance → artifact →
|
||||
publisher → reader;使用随机 localhost 端口、tmpfs 和自动容器清理。
|
||||
|
||||
## Current risk contract
|
||||
|
||||
- artifact schema:`1.1.0`;
|
||||
- `CovarianceSnapshot` 对输入矩阵深拷贝并显式记录截至日、频率和年化期数;
|
||||
- `risk_snapshots` 按研究交易日映射,可只生成需要的风险观察日;
|
||||
- 使用成交后实际持仓,不包含现金风险资产;协方差资产标签必须与研究资产全集一致;
|
||||
- `covariance_as_of_date` 不得晚于 `trade_date`;无正组合方差时拒绝产物。
|
||||
- `estimate_covariance_snapshot` 从显式数据快照的日收益生成无前视、complete-case、
|
||||
SHA-256 可复现的 per-period sample covariance;不包含 I/O 或未来行。
|
||||
- `prepare_asset_return_snapshot` 从规范化长表行情生成不前向填充的 simple daily returns;
|
||||
显式 ingestion snapshot ID、源/字段/复权口径、价格值和缺失掩码共同形成
|
||||
`asset-returns-v1:<sha256>`,并把同一 ID 传给 covariance 与 run artifact。
|
||||
- artifact builder fail closed:每个 `CovarianceSnapshot.data_snapshot_id` 必须与 run 级
|
||||
`data_snapshot_id` 完全一致,禁止把其他行情快照的风险分解静默发布到当前研究运行。
|
||||
- shrinkage 适配器本轮不实现:scikit-learn 尚非声明依赖,未来只允许薄适配
|
||||
`LedoitWolf` / `OAS`,不复制公式、不依赖环境偶然安装状态。
|
||||
|
||||
## Next action
|
||||
|
||||
保持 Draft PR #5,不绕过堆叠顺序合并;下游 `research_results` / `research_platform`
|
||||
继续在现有 Draft 分支消费同一数据 lineage。下一阶段优先把 ingestion snapshot ID 从
|
||||
真实 ELT 元数据接入调用方,再在依赖治理通过后单独交付可选 shrinkage adapter。
|
||||
@@ -15,7 +15,9 @@ v1.2.0 Phase 0:5 个基础算子 + 5 个 alpha 公式(alpha001–alpha005)
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
from collections.abc import Callable, Mapping
|
||||
from types import MappingProxyType
|
||||
from typing import Any, cast
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
@@ -209,6 +211,367 @@ def indneutralize(series: pd.Series, groups: pd.Series) -> pd.Series:
|
||||
return series - series.groupby(groups).transform("mean")
|
||||
|
||||
|
||||
# ── Phase 1 operator contract ──────────────────────────
|
||||
|
||||
# This is deliberately a small, stable surface for downstream research
|
||||
# orchestration. The full alpha158 formula catalogue can continue to grow,
|
||||
# while callers use one validated dispatch entry point for the first ten
|
||||
# deterministic building blocks.
|
||||
ALPHA158_PHASE1_MAX_WINDOW = 252
|
||||
|
||||
ALPHA158_PHASE1_OPERATOR_SPECS: dict[str, dict[str, Any]] = {
|
||||
"rank": {
|
||||
"name": "rank",
|
||||
"formula": "rank(series)",
|
||||
"inputs": ["series"],
|
||||
"windowed": False,
|
||||
},
|
||||
"delta": {
|
||||
"name": "delta",
|
||||
"formula": "delta(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_mean": {
|
||||
"name": "ts_mean",
|
||||
"formula": "ts_mean(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_std": {
|
||||
"name": "ts_std",
|
||||
"formula": "ts_std(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_rank": {
|
||||
"name": "ts_rank",
|
||||
"formula": "ts_rank(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"correlation": {
|
||||
"name": "correlation",
|
||||
"formula": "correlation(series, secondary, window)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_min": {
|
||||
"name": "ts_min",
|
||||
"formula": "ts_min(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_max": {
|
||||
"name": "ts_max",
|
||||
"formula": "ts_max(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_sum": {
|
||||
"name": "ts_sum",
|
||||
"formula": "ts_sum(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
"decay_linear": {
|
||||
"name": "decay_linear",
|
||||
"formula": "decay_linear(series, window)",
|
||||
"inputs": ["series"],
|
||||
"windowed": True,
|
||||
},
|
||||
}
|
||||
|
||||
_PHASE1_OPERATOR_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"rank": rank,
|
||||
"delta": delta,
|
||||
"ts_mean": ts_mean,
|
||||
"ts_std": ts_std,
|
||||
"ts_rank": ts_rank,
|
||||
"correlation": correlation,
|
||||
"ts_min": ts_min,
|
||||
"ts_max": ts_max,
|
||||
"ts_sum": ts_sum,
|
||||
"decay_linear": decay_linear,
|
||||
}
|
||||
|
||||
|
||||
def list_phase1_operators() -> tuple[str, ...]:
|
||||
"""Return the deterministic Phase 1 operator names in stable order."""
|
||||
return tuple(ALPHA158_PHASE1_OPERATOR_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase1_operator(
|
||||
name: str,
|
||||
series: pd.Series,
|
||||
secondary: pd.Series | None = None,
|
||||
*,
|
||||
window: int | None = None,
|
||||
) -> pd.Series:
|
||||
"""Evaluate one of the ten Phase 1 operators with a validated contract.
|
||||
|
||||
``window`` is required for time-series operators and forbidden for the
|
||||
cross-sectional ``rank`` operator. Binary ``correlation`` also requires
|
||||
a same-index secondary series so that callers cannot silently introduce
|
||||
alignment-dependent results.
|
||||
"""
|
||||
if name not in ALPHA158_PHASE1_OPERATOR_SPECS:
|
||||
raise KeyError(f"operator {name!r} not registered")
|
||||
|
||||
is_windowed = bool(ALPHA158_PHASE1_OPERATOR_SPECS[name]["windowed"])
|
||||
if is_windowed:
|
||||
if isinstance(window, bool) or not isinstance(window, int) or window <= 0:
|
||||
raise ValueError(f"window must be a positive integer for {name}")
|
||||
if window > ALPHA158_PHASE1_MAX_WINDOW:
|
||||
raise ValueError(
|
||||
f"window exceeds maximum supported value {ALPHA158_PHASE1_MAX_WINDOW} for {name}"
|
||||
)
|
||||
if not is_windowed and window is not None:
|
||||
raise ValueError(f"window is not supported for {name}")
|
||||
|
||||
if name == "correlation":
|
||||
if secondary is None:
|
||||
raise ValueError("secondary is required for correlation")
|
||||
if not series.index.equals(secondary.index):
|
||||
raise ValueError("secondary index must align with series")
|
||||
return correlation(series, secondary, window) # type: ignore[arg-type]
|
||||
|
||||
if secondary is not None:
|
||||
raise ValueError(f"secondary is not supported for {name}")
|
||||
|
||||
operator = _PHASE1_OPERATOR_FUNCTIONS[name]
|
||||
if name == "rank":
|
||||
return operator(series)
|
||||
return operator(series, window)
|
||||
|
||||
|
||||
# ── Phase 2 cumulative operator contract ──────────────────────────────
|
||||
|
||||
# Phase 2 is cumulative: downstream callers can upgrade to one dispatch
|
||||
# surface covering every existing alpha158 building block, while Phase 1
|
||||
# names, metadata, ordering, and evaluation remain unchanged.
|
||||
ALPHA158_PHASE2_MAX_WINDOW = ALPHA158_PHASE1_MAX_WINDOW
|
||||
|
||||
ALPHA158_PHASE2_OPERATOR_SPECS: dict[str, dict[str, Any]] = {
|
||||
name: {
|
||||
**spec,
|
||||
"parameters": ["window"] if bool(spec["windowed"]) else [],
|
||||
}
|
||||
for name, spec in ALPHA158_PHASE1_OPERATOR_SPECS.items()
|
||||
}
|
||||
ALPHA158_PHASE2_OPERATOR_SPECS.update(
|
||||
{
|
||||
"ts_argmin": {
|
||||
"name": "ts_argmin",
|
||||
"formula": "ts_argmin(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"ts_argmax": {
|
||||
"name": "ts_argmax",
|
||||
"formula": "ts_argmax(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"product": {
|
||||
"name": "product",
|
||||
"formula": "product(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"returns": {
|
||||
"name": "returns",
|
||||
"formula": "returns(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"scale": {
|
||||
"name": "scale",
|
||||
"formula": "scale(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"signed_power": {
|
||||
"name": "signed_power",
|
||||
"formula": "signed_power(series, exponent)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["exponent"],
|
||||
"windowed": False,
|
||||
},
|
||||
"stddev": {
|
||||
"name": "stddev",
|
||||
"formula": "stddev(series, window)",
|
||||
"inputs": ["series"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"covariance": {
|
||||
"name": "covariance",
|
||||
"formula": "covariance(series, secondary, window)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"parameters": ["window"],
|
||||
"windowed": True,
|
||||
},
|
||||
"log": {
|
||||
"name": "log",
|
||||
"formula": "log(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"abs_series": {
|
||||
"name": "abs_series",
|
||||
"formula": "abs_series(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"sign": {
|
||||
"name": "sign",
|
||||
"formula": "sign(series)",
|
||||
"inputs": ["series"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"max_pair": {
|
||||
"name": "max_pair",
|
||||
"formula": "max_pair(series, secondary)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"min_pair": {
|
||||
"name": "min_pair",
|
||||
"formula": "min_pair(series, secondary)",
|
||||
"inputs": ["series", "secondary"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
"indneutralize": {
|
||||
"name": "indneutralize",
|
||||
"formula": "indneutralize(series, groups)",
|
||||
"inputs": ["series", "groups"],
|
||||
"parameters": [],
|
||||
"windowed": False,
|
||||
},
|
||||
}
|
||||
)
|
||||
|
||||
_PHASE2_OPERATOR_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
**_PHASE1_OPERATOR_FUNCTIONS,
|
||||
"ts_argmin": ts_argmin,
|
||||
"ts_argmax": ts_argmax,
|
||||
"product": product,
|
||||
"returns": returns,
|
||||
"scale": scale,
|
||||
"signed_power": signed_power,
|
||||
"stddev": stddev,
|
||||
"covariance": covariance,
|
||||
"log": log,
|
||||
"abs_series": abs_series,
|
||||
"sign": sign,
|
||||
"max_pair": max_pair,
|
||||
"min_pair": min_pair,
|
||||
"indneutralize": indneutralize,
|
||||
}
|
||||
|
||||
_PHASE2_WINDOWED_OPERATORS = frozenset(
|
||||
name for name, spec in ALPHA158_PHASE2_OPERATOR_SPECS.items() if bool(spec["windowed"])
|
||||
)
|
||||
_PHASE2_BINARY_OPERATORS = frozenset({"correlation", "covariance", "max_pair", "min_pair"})
|
||||
|
||||
|
||||
def list_phase2_operators() -> tuple[str, ...]:
|
||||
"""Return all Phase 2 operator names in stable cumulative order."""
|
||||
return tuple(ALPHA158_PHASE2_OPERATOR_SPECS)
|
||||
|
||||
|
||||
def _validate_phase2_window(name: str, window: int | None) -> int:
|
||||
if isinstance(window, bool) or not isinstance(window, int) or window <= 0:
|
||||
raise ValueError(f"window must be a positive integer for {name}")
|
||||
if window > ALPHA158_PHASE2_MAX_WINDOW:
|
||||
raise ValueError(
|
||||
f"window exceeds maximum supported value {ALPHA158_PHASE2_MAX_WINDOW} for {name}"
|
||||
)
|
||||
return window
|
||||
|
||||
|
||||
def evaluate_phase2_operator(
|
||||
name: str,
|
||||
series: pd.Series,
|
||||
secondary: pd.Series | None = None,
|
||||
*,
|
||||
window: int | None = None,
|
||||
exponent: float | None = None,
|
||||
groups: pd.Series | None = None,
|
||||
) -> pd.Series:
|
||||
"""Evaluate any existing alpha158 building block through a strict contract.
|
||||
|
||||
Phase 2 rejects implicit alignment, missing required arguments, unused
|
||||
arguments, unbounded windows, and non-finite exponents before dispatch.
|
||||
"""
|
||||
if name not in ALPHA158_PHASE2_OPERATOR_SPECS:
|
||||
raise KeyError(f"operator {name!r} not registered")
|
||||
if not isinstance(series, pd.Series):
|
||||
raise TypeError("series must be a pandas Series")
|
||||
|
||||
validated_window: int | None = None
|
||||
if name in _PHASE2_WINDOWED_OPERATORS:
|
||||
validated_window = _validate_phase2_window(name, window)
|
||||
elif window is not None:
|
||||
raise ValueError(f"window is not supported for {name}")
|
||||
|
||||
if name in _PHASE2_BINARY_OPERATORS:
|
||||
if secondary is None:
|
||||
raise ValueError(f"secondary is required for {name}")
|
||||
if not isinstance(secondary, pd.Series):
|
||||
raise TypeError("secondary must be a pandas Series")
|
||||
if not series.index.equals(secondary.index):
|
||||
raise ValueError("secondary index must align with series")
|
||||
elif secondary is not None:
|
||||
raise ValueError(f"secondary is not supported for {name}")
|
||||
|
||||
validated_exponent: float | None = None
|
||||
if name == "signed_power":
|
||||
if (
|
||||
isinstance(exponent, bool)
|
||||
or not isinstance(exponent, (int, float))
|
||||
or not np.isfinite(exponent)
|
||||
):
|
||||
raise ValueError("exponent must be a finite number for signed_power")
|
||||
validated_exponent = float(exponent)
|
||||
elif exponent is not None:
|
||||
raise ValueError(f"exponent is not supported for {name}")
|
||||
|
||||
if name == "indneutralize":
|
||||
if groups is None:
|
||||
raise ValueError("groups is required for indneutralize")
|
||||
if not isinstance(groups, pd.Series):
|
||||
raise TypeError("groups must be a pandas Series")
|
||||
if not series.index.equals(groups.index):
|
||||
raise ValueError("groups index must align with series")
|
||||
elif groups is not None:
|
||||
raise ValueError(f"groups is not supported for {name}")
|
||||
|
||||
operator = _PHASE2_OPERATOR_FUNCTIONS[name]
|
||||
if name == "signed_power":
|
||||
return operator(series, validated_exponent)
|
||||
if name == "indneutralize":
|
||||
return operator(series, groups)
|
||||
if name in {"correlation", "covariance"}:
|
||||
return operator(series, secondary, validated_window)
|
||||
if name in {"max_pair", "min_pair"}:
|
||||
return operator(series, secondary)
|
||||
if validated_window is not None:
|
||||
return operator(series, validated_window)
|
||||
return operator(series)
|
||||
|
||||
|
||||
# ── 组合算子(alpha158 公式样本) ─────────────────────────
|
||||
|
||||
|
||||
@@ -2730,6 +3093,541 @@ def parse_alpha_formula(formula_str: str) -> dict[str, Any]:
|
||||
return parsed
|
||||
|
||||
|
||||
# ── Phase 3 formula contract: frozen alpha001-alpha050 surface ──────────────
|
||||
|
||||
# Formula functions remain the implementation source of truth. This contract
|
||||
# freezes their callable surface separately from formula dependencies so that
|
||||
# historical compatibility-only arguments remain explicit without rewriting
|
||||
# formulas or changing direct-call APIs.
|
||||
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 = (
|
||||
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
|
||||
)
|
||||
|
||||
_PHASE3_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_001": alpha_001,
|
||||
"alpha_002": alpha_002,
|
||||
"alpha_003": alpha_003,
|
||||
"alpha_004": alpha_004,
|
||||
"alpha_005": alpha_005,
|
||||
"alpha_006": alpha_006,
|
||||
"alpha_007": alpha_007,
|
||||
"alpha_008": alpha_008,
|
||||
"alpha_009": alpha_009,
|
||||
"alpha_010": alpha_010,
|
||||
"alpha_011": alpha_011,
|
||||
"alpha_012": alpha_012,
|
||||
"alpha_013": alpha_013,
|
||||
"alpha_014": alpha_014,
|
||||
"alpha_015": alpha_015,
|
||||
"alpha_016": alpha_016,
|
||||
"alpha_017": alpha_017,
|
||||
"alpha_018": alpha_018,
|
||||
"alpha_019": alpha_019,
|
||||
"alpha_020": alpha_020,
|
||||
"alpha_021": alpha_021,
|
||||
"alpha_022": alpha_022,
|
||||
"alpha_023": alpha_023,
|
||||
"alpha_024": alpha_024,
|
||||
"alpha_025": alpha_025,
|
||||
"alpha_026": alpha_026,
|
||||
"alpha_027": alpha_027,
|
||||
"alpha_028": alpha_028,
|
||||
"alpha_029": alpha_029,
|
||||
"alpha_030": alpha_030,
|
||||
"alpha_031": alpha_031,
|
||||
"alpha_032": alpha_032,
|
||||
"alpha_033": alpha_033,
|
||||
"alpha_034": alpha_034,
|
||||
"alpha_035": alpha_035,
|
||||
"alpha_036": alpha_036,
|
||||
"alpha_037": alpha_037,
|
||||
"alpha_038": alpha_038,
|
||||
"alpha_039": alpha_039,
|
||||
"alpha_040": alpha_040,
|
||||
"alpha_041": alpha_041,
|
||||
"alpha_042": alpha_042,
|
||||
"alpha_043": alpha_043,
|
||||
"alpha_044": alpha_044,
|
||||
"alpha_045": alpha_045,
|
||||
"alpha_046": alpha_046,
|
||||
"alpha_047": alpha_047,
|
||||
"alpha_048": alpha_048,
|
||||
"alpha_049": alpha_049,
|
||||
"alpha_050": alpha_050,
|
||||
}
|
||||
|
||||
_PHASE3_FORMULA_INPUT_OVERRIDES: dict[str, list[str]] = {
|
||||
"alpha_011": ["close", "high", "low"],
|
||||
"alpha_035": ["volume"],
|
||||
"alpha_036": ["close"],
|
||||
"alpha_040": ["high", "low"],
|
||||
"alpha_042": ["close"],
|
||||
"alpha_043": ["volume"],
|
||||
}
|
||||
|
||||
_PHASE3_INPUT_CATEGORIES = {
|
||||
1: "single",
|
||||
2: "pair",
|
||||
3: "triple",
|
||||
4: "quadruple",
|
||||
}
|
||||
|
||||
|
||||
def _phase3_call_inputs(function: Callable[..., pd.Series]) -> list[str]:
|
||||
import inspect
|
||||
|
||||
parameters = list(inspect.signature(function).parameters.values())
|
||||
if any(
|
||||
parameter.kind is not inspect.Parameter.POSITIONAL_OR_KEYWORD
|
||||
or parameter.default is not inspect.Parameter.empty
|
||||
for parameter in parameters
|
||||
):
|
||||
raise RuntimeError(f"unsupported formula signature for {function.__name__}")
|
||||
return ["open" if parameter.name == "open_" else parameter.name for parameter in parameters]
|
||||
|
||||
|
||||
def _phase3_string_list(meta: dict[str, Any], field: str, alpha_id: str) -> list[str]:
|
||||
value = meta[field]
|
||||
if not isinstance(value, list) or not all(isinstance(item, str) for item in value):
|
||||
raise RuntimeError(f"{field} must be a list of strings for {alpha_id}")
|
||||
return list(value)
|
||||
|
||||
|
||||
def _build_phase3_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE3_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _PHASE3_FORMULA_INPUT_OVERRIDES.get(alpha_id, call_inputs)
|
||||
input_category = _PHASE3_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": list(formula_inputs),
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
def _freeze_phase3_formula_specs(
|
||||
specs: dict[str, dict[str, Any]],
|
||||
) -> Mapping[str, Mapping[str, Any]]:
|
||||
frozen_specs: dict[str, Mapping[str, Any]] = {}
|
||||
for alpha_id, spec in specs.items():
|
||||
frozen_specs[alpha_id] = MappingProxyType(
|
||||
{field: tuple(value) if isinstance(value, list) else value for field, value in spec.items()}
|
||||
)
|
||||
return MappingProxyType(frozen_specs)
|
||||
|
||||
|
||||
ALPHA158_PHASE3_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase3_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase3_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha001-alpha050 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE3_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase3_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 3 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE3_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE3_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE3_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 4 formula contract: frozen alpha051-alpha100 surface ──────────────
|
||||
|
||||
# Phase 4 extends the versioned formula contract without mutating the Phase 3
|
||||
# catalogue, digest, dispatch surface, or the existing formula functions.
|
||||
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 = (
|
||||
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
|
||||
)
|
||||
|
||||
_PHASE4_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_051": alpha_051,
|
||||
"alpha_052": alpha_052,
|
||||
"alpha_053": alpha_053,
|
||||
"alpha_054": alpha_054,
|
||||
"alpha_055": alpha_055,
|
||||
"alpha_056": alpha_056,
|
||||
"alpha_057": alpha_057,
|
||||
"alpha_058": alpha_058,
|
||||
"alpha_059": alpha_059,
|
||||
"alpha_060": alpha_060,
|
||||
"alpha_061": alpha_061,
|
||||
"alpha_062": alpha_062,
|
||||
"alpha_063": alpha_063,
|
||||
"alpha_064": alpha_064,
|
||||
"alpha_065": alpha_065,
|
||||
"alpha_066": alpha_066,
|
||||
"alpha_067": alpha_067,
|
||||
"alpha_068": alpha_068,
|
||||
"alpha_069": alpha_069,
|
||||
"alpha_070": alpha_070,
|
||||
"alpha_071": alpha_071,
|
||||
"alpha_072": alpha_072,
|
||||
"alpha_073": alpha_073,
|
||||
"alpha_074": alpha_074,
|
||||
"alpha_075": alpha_075,
|
||||
"alpha_076": alpha_076,
|
||||
"alpha_077": alpha_077,
|
||||
"alpha_078": alpha_078,
|
||||
"alpha_079": alpha_079,
|
||||
"alpha_080": alpha_080,
|
||||
"alpha_081": alpha_081,
|
||||
"alpha_082": alpha_082,
|
||||
"alpha_083": alpha_083,
|
||||
"alpha_084": alpha_084,
|
||||
"alpha_085": alpha_085,
|
||||
"alpha_086": alpha_086,
|
||||
"alpha_087": alpha_087,
|
||||
"alpha_088": alpha_088,
|
||||
"alpha_089": alpha_089,
|
||||
"alpha_090": alpha_090,
|
||||
"alpha_091": alpha_091,
|
||||
"alpha_092": alpha_092,
|
||||
"alpha_093": alpha_093,
|
||||
"alpha_094": alpha_094,
|
||||
"alpha_095": alpha_095,
|
||||
"alpha_096": alpha_096,
|
||||
"alpha_097": alpha_097,
|
||||
"alpha_098": alpha_098,
|
||||
"alpha_099": alpha_099,
|
||||
"alpha_100": alpha_100,
|
||||
}
|
||||
|
||||
_PHASE4_INPUT_CATEGORIES = {
|
||||
1: "single",
|
||||
2: "pair",
|
||||
3: "triple",
|
||||
4: "quadruple",
|
||||
5: "quintuple",
|
||||
}
|
||||
|
||||
|
||||
def _build_phase4_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE4_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE4_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase4_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase4_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha051-alpha100 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE4_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase4_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 4 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE4_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE4_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE4_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 5 formula contract: frozen alpha101-alpha150 surface ──────────────
|
||||
|
||||
# Phase 5 extends the versioned formula contract without mutating any earlier
|
||||
# catalogue, digest, dispatch surface, or existing formula implementation.
|
||||
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 = (
|
||||
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
|
||||
)
|
||||
|
||||
_PHASE5_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_101": alpha_101,
|
||||
"alpha_102": alpha_102,
|
||||
"alpha_103": alpha_103,
|
||||
"alpha_104": alpha_104,
|
||||
"alpha_105": alpha_105,
|
||||
"alpha_106": alpha_106,
|
||||
"alpha_107": alpha_107,
|
||||
"alpha_108": alpha_108,
|
||||
"alpha_109": alpha_109,
|
||||
"alpha_110": alpha_110,
|
||||
"alpha_111": alpha_111,
|
||||
"alpha_112": alpha_112,
|
||||
"alpha_113": alpha_113,
|
||||
"alpha_114": alpha_114,
|
||||
"alpha_115": alpha_115,
|
||||
"alpha_116": alpha_116,
|
||||
"alpha_117": alpha_117,
|
||||
"alpha_118": alpha_118,
|
||||
"alpha_119": alpha_119,
|
||||
"alpha_120": alpha_120,
|
||||
"alpha_121": alpha_121,
|
||||
"alpha_122": alpha_122,
|
||||
"alpha_123": alpha_123,
|
||||
"alpha_124": alpha_124,
|
||||
"alpha_125": alpha_125,
|
||||
"alpha_126": alpha_126,
|
||||
"alpha_127": alpha_127,
|
||||
"alpha_128": alpha_128,
|
||||
"alpha_129": alpha_129,
|
||||
"alpha_130": alpha_130,
|
||||
"alpha_131": alpha_131,
|
||||
"alpha_132": alpha_132,
|
||||
"alpha_133": alpha_133,
|
||||
"alpha_134": alpha_134,
|
||||
"alpha_135": alpha_135,
|
||||
"alpha_136": alpha_136,
|
||||
"alpha_137": alpha_137,
|
||||
"alpha_138": alpha_138,
|
||||
"alpha_139": alpha_139,
|
||||
"alpha_140": alpha_140,
|
||||
"alpha_141": alpha_141,
|
||||
"alpha_142": alpha_142,
|
||||
"alpha_143": alpha_143,
|
||||
"alpha_144": alpha_144,
|
||||
"alpha_145": alpha_145,
|
||||
"alpha_146": alpha_146,
|
||||
"alpha_147": alpha_147,
|
||||
"alpha_148": alpha_148,
|
||||
"alpha_149": alpha_149,
|
||||
"alpha_150": alpha_150,
|
||||
}
|
||||
|
||||
|
||||
def _build_phase5_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE5_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE5_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase5_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase5_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha101-alpha150 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE5_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase5_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 5 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE5_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE5_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if len(inputs[field]) != len(primary):
|
||||
raise ValueError(f"{field} length must match {primary_field}")
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE5_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
# ── Phase 6 formula contract: frozen alpha151-alpha158 surface ──────────────
|
||||
|
||||
# Phase 6 completes the versioned formula contract without mutating any
|
||||
# earlier catalogue, digest, dispatch surface, or formula implementation.
|
||||
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION = "1.0.0"
|
||||
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 = (
|
||||
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
|
||||
)
|
||||
|
||||
_PHASE6_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
|
||||
"alpha_151": alpha_151,
|
||||
"alpha_152": alpha_152,
|
||||
"alpha_153": alpha_153,
|
||||
"alpha_154": alpha_154,
|
||||
"alpha_155": alpha_155,
|
||||
"alpha_156": alpha_156,
|
||||
"alpha_157": alpha_157,
|
||||
"alpha_158": alpha_158,
|
||||
}
|
||||
|
||||
|
||||
def _build_phase6_formula_specs() -> dict[str, dict[str, Any]]:
|
||||
specs: dict[str, dict[str, Any]] = {}
|
||||
for alpha_id, function in _PHASE6_FORMULA_FUNCTIONS.items():
|
||||
meta = ALPHA158_REGISTRY[alpha_id]
|
||||
call_inputs = _phase3_call_inputs(function)
|
||||
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
|
||||
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
|
||||
if input_category is None:
|
||||
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
|
||||
specs[alpha_id] = {
|
||||
"name": alpha_id,
|
||||
"contract_version": ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
|
||||
"formula": meta["formula"],
|
||||
"category": meta["category"],
|
||||
"complexity": meta["complexity"],
|
||||
"parameters": _phase3_string_list(meta, "params", alpha_id),
|
||||
"description": meta["description"],
|
||||
"references": _phase3_string_list(meta, "references", alpha_id),
|
||||
"call_inputs": list(call_inputs),
|
||||
"formula_inputs": formula_inputs,
|
||||
"input_category": input_category,
|
||||
}
|
||||
return specs
|
||||
|
||||
|
||||
ALPHA158_PHASE6_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
|
||||
_freeze_phase3_formula_specs(_build_phase6_formula_specs())
|
||||
)
|
||||
|
||||
|
||||
def list_phase6_formulas() -> tuple[str, ...]:
|
||||
"""Return the frozen alpha151-alpha158 formula IDs in stable order."""
|
||||
return tuple(ALPHA158_PHASE6_FORMULA_SPECS)
|
||||
|
||||
|
||||
def evaluate_phase6_formula(name: str, **inputs: pd.Series) -> pd.Series:
|
||||
"""Evaluate a Phase 6 formula with an exact, alignment-safe input contract."""
|
||||
if name not in ALPHA158_PHASE6_FORMULA_SPECS:
|
||||
raise KeyError(f"formula {name!r} not registered")
|
||||
|
||||
spec = ALPHA158_PHASE6_FORMULA_SPECS[name]
|
||||
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
|
||||
missing_inputs = [field for field in required_inputs if field not in inputs]
|
||||
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
|
||||
if missing_inputs or unexpected_inputs:
|
||||
details: list[str] = []
|
||||
if missing_inputs:
|
||||
details.append(f"missing inputs {missing_inputs}")
|
||||
if unexpected_inputs:
|
||||
details.append(f"unexpected inputs {unexpected_inputs}")
|
||||
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
|
||||
|
||||
for field in required_inputs:
|
||||
if not isinstance(inputs[field], pd.Series):
|
||||
raise TypeError(f"{field} must be a pandas Series")
|
||||
|
||||
primary_field = required_inputs[0]
|
||||
primary = inputs[primary_field]
|
||||
for field in required_inputs[1:]:
|
||||
if len(inputs[field]) != len(primary):
|
||||
raise ValueError(f"{field} length must match {primary_field}")
|
||||
if not primary.index.equals(inputs[field].index):
|
||||
raise ValueError(f"{field} index must align with {primary_field}")
|
||||
|
||||
function = _PHASE6_FORMULA_FUNCTIONS[name]
|
||||
return function(*(inputs[field] for field in required_inputs))
|
||||
|
||||
|
||||
__all__ = [
|
||||
"rank",
|
||||
"delta",
|
||||
@@ -2755,6 +3653,34 @@ __all__ = [
|
||||
"max_pair",
|
||||
"min_pair",
|
||||
"indneutralize",
|
||||
"ALPHA158_PHASE1_MAX_WINDOW",
|
||||
"ALPHA158_PHASE1_OPERATOR_SPECS",
|
||||
"list_phase1_operators",
|
||||
"evaluate_phase1_operator",
|
||||
"ALPHA158_PHASE2_MAX_WINDOW",
|
||||
"ALPHA158_PHASE2_OPERATOR_SPECS",
|
||||
"list_phase2_operators",
|
||||
"evaluate_phase2_operator",
|
||||
"ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE3_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE3_FORMULA_SPECS",
|
||||
"list_phase3_formulas",
|
||||
"evaluate_phase3_formula",
|
||||
"ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE4_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE4_FORMULA_SPECS",
|
||||
"list_phase4_formulas",
|
||||
"evaluate_phase4_formula",
|
||||
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE5_FORMULA_SPECS",
|
||||
"list_phase5_formulas",
|
||||
"evaluate_phase5_formula",
|
||||
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE6_FORMULA_SPECS",
|
||||
"list_phase6_formulas",
|
||||
"evaluate_phase6_formula",
|
||||
"alpha_001",
|
||||
"alpha_002",
|
||||
"alpha_003",
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,141 @@
|
||||
"""Post-execution daily return attribution derived from the portfolio ledger.
|
||||
|
||||
The ledger is the source of truth: previous-close holdings explain overnight
|
||||
PnL, current-close holdings explain intraday PnL, and actual execution costs
|
||||
remain a separate contribution. Target weights and factor scores are not
|
||||
accepted here because they are intentions rather than realized positions.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from quant_engine.execution import ExecutionSimulationResult
|
||||
|
||||
__all__ = ["DailyReturnAttribution", "compute_daily_return_attribution"]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, eq=False)
|
||||
class DailyReturnAttribution:
|
||||
"""Auditable decomposition of each net portfolio return."""
|
||||
|
||||
overnight: pd.DataFrame
|
||||
intraday: pd.DataFrame
|
||||
transaction_cost: pd.Series
|
||||
residual: pd.Series
|
||||
total_return: pd.Series
|
||||
|
||||
@property
|
||||
def asset_contributions(self) -> pd.DataFrame:
|
||||
"""Return the combined overnight and intraday contribution by asset."""
|
||||
return self.overnight + self.intraday
|
||||
|
||||
@property
|
||||
def explained_return(self) -> pd.Series:
|
||||
"""Return asset contributions plus execution costs, before residual."""
|
||||
explained = self.asset_contributions.sum(axis=1) + self.transaction_cost
|
||||
return explained.rename("explained_return")
|
||||
|
||||
|
||||
def _validate_prices(
|
||||
execution: ExecutionSimulationResult,
|
||||
execution_prices: pd.DataFrame,
|
||||
valuation_prices: pd.DataFrame,
|
||||
) -> pd.DatetimeIndex:
|
||||
if not isinstance(execution_prices, pd.DataFrame):
|
||||
raise TypeError("execution_prices must be a pandas DataFrame")
|
||||
if not isinstance(valuation_prices, pd.DataFrame):
|
||||
raise TypeError("valuation_prices must be a pandas DataFrame")
|
||||
if not isinstance(execution_prices.index, pd.DatetimeIndex):
|
||||
raise TypeError("execution_prices must use a DatetimeIndex")
|
||||
if not execution_prices.index.equals(valuation_prices.index):
|
||||
raise ValueError("execution and valuation prices must use matching trading calendars")
|
||||
if not execution_prices.columns.equals(valuation_prices.columns):
|
||||
raise ValueError("execution and valuation prices must use matching asset labels")
|
||||
|
||||
ledger_index = pd.DatetimeIndex(pd.Timestamp(position.date) for position in execution.positions)
|
||||
if not ledger_index.equals(execution_prices.index):
|
||||
raise ValueError("ledger and price histories must use matching trading calendars")
|
||||
if len(execution.positions) != len(execution.daily_executions):
|
||||
raise ValueError("ledger positions and executions must have matching lengths")
|
||||
return execution_prices.index.copy()
|
||||
|
||||
|
||||
def _price_for_held_asset(
|
||||
prices: pd.DataFrame,
|
||||
date: pd.Timestamp,
|
||||
asset: str,
|
||||
stage: str,
|
||||
) -> float:
|
||||
if asset not in prices.columns:
|
||||
raise ValueError(f"missing {stage} price for held asset {asset} on {date}")
|
||||
price = float(prices.at[date, asset])
|
||||
if not math.isfinite(price) or price <= 0:
|
||||
raise ValueError(f"invalid {stage} price for held asset {asset} on {date}")
|
||||
return price
|
||||
|
||||
|
||||
def compute_daily_return_attribution(
|
||||
execution: ExecutionSimulationResult,
|
||||
execution_prices: pd.DataFrame,
|
||||
valuation_prices: pd.DataFrame,
|
||||
) -> DailyReturnAttribution:
|
||||
"""Decompose net daily returns using realized pre/post-execution holdings.
|
||||
|
||||
For each session, previous-close shares earn the move from the previous
|
||||
close to the current execution price; current-close shares earn the move
|
||||
from execution price to current close. Actual commissions, stamp tax and
|
||||
slippage are divided by the same previous NAV denominator. ``residual``
|
||||
exposes any failure of those components to close to the ledger return.
|
||||
"""
|
||||
index = _validate_prices(execution, execution_prices, valuation_prices)
|
||||
columns = execution_prices.columns.copy()
|
||||
overnight = pd.DataFrame(0.0, index=index.copy(), columns=columns)
|
||||
intraday = pd.DataFrame(0.0, index=index.copy(), columns=columns)
|
||||
cost = pd.Series(0.0, index=index.copy(), name="transaction_cost")
|
||||
|
||||
previous_holdings: dict[str, float] = {}
|
||||
previous_nav = execution.initial_cash
|
||||
for row_number, (date, position, daily) in enumerate(
|
||||
zip(index, execution.positions, execution.daily_executions, strict=True)
|
||||
):
|
||||
if previous_nav <= 0 or not math.isfinite(previous_nav):
|
||||
raise ValueError(f"previous portfolio value must be positive and finite on {date}")
|
||||
|
||||
for asset, shares in previous_holdings.items():
|
||||
execution_price = _price_for_held_asset(
|
||||
execution_prices, date, asset, "execution"
|
||||
)
|
||||
previous_close = _price_for_held_asset(
|
||||
valuation_prices, index[row_number - 1], asset, "previous valuation"
|
||||
)
|
||||
overnight.at[date, asset] = shares * (execution_price - previous_close) / previous_nav
|
||||
|
||||
for asset, shares in position.holdings.items():
|
||||
execution_price = _price_for_held_asset(
|
||||
execution_prices, date, asset, "execution"
|
||||
)
|
||||
close_price = _price_for_held_asset(valuation_prices, date, asset, "valuation")
|
||||
intraday.at[date, asset] = shares * (close_price - execution_price) / previous_nav
|
||||
|
||||
cost.at[date] = -sum(item.total_cost for item in daily.executions) / previous_nav
|
||||
previous_holdings = position.holdings
|
||||
previous_nav = position.portfolio_value
|
||||
|
||||
total_return = pd.Series(
|
||||
execution.daily_returns.to_numpy(copy=True),
|
||||
index=index.copy(),
|
||||
name="total_return",
|
||||
)
|
||||
explained = (overnight + intraday).sum(axis=1) + cost
|
||||
residual = (total_return - explained).rename("residual")
|
||||
return DailyReturnAttribution(
|
||||
overnight=overnight,
|
||||
intraday=intraday,
|
||||
transaction_cost=cost,
|
||||
residual=residual,
|
||||
total_return=total_return,
|
||||
)
|
||||
@@ -13,29 +13,28 @@
|
||||
|
||||
```python
|
||||
from quant_engine.backtest import (
|
||||
compute_nav_from_weights, # 调仓表 → 净值
|
||||
rebalance_table, # 周期性再平衡
|
||||
compare_to_benchmark, # 策略 vs 基准
|
||||
rebalance_periodic, # 周期性再平衡
|
||||
run_weight_backtest, # 权重 → 统一结果对象
|
||||
weights_to_long_short, # 多空组合
|
||||
)
|
||||
|
||||
# 1. 调仓表 → 净值
|
||||
nav = compute_nav_from_weights(
|
||||
# 调仓表 → 净值、收益、绩效与基准报告
|
||||
rebalance_table = rebalance_periodic(target_weights, rebalance_dates, returns.index)
|
||||
result = run_weight_backtest(
|
||||
weights=rebalance_table, # 每周/每月调仓
|
||||
stock_returns=returns, # 个股日收益
|
||||
initial_capital=1.0,
|
||||
benchmark_nav=benchmark_nav,
|
||||
)
|
||||
|
||||
# 2. 跟基准比
|
||||
result = compare_to_benchmark(nav, benchmark_nav)
|
||||
print(result.summary())
|
||||
print(result.stats())
|
||||
print(result.benchmark_report())
|
||||
```
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
@@ -46,6 +45,26 @@ from quant_engine.metrics import summary as metrics_summary
|
||||
logger = get_logger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, eq=False)
|
||||
class BacktestResult:
|
||||
"""一次权重回测的稳定结果快照。"""
|
||||
|
||||
nav: pd.Series
|
||||
returns: pd.Series
|
||||
weights: pd.DataFrame
|
||||
benchmark_nav: pd.Series | None = None
|
||||
|
||||
def stats(self, rf: float = 0.0) -> Mapping[str, float]:
|
||||
"""返回标准绩效指标。"""
|
||||
return metrics_summary(self.returns, rf)
|
||||
|
||||
def benchmark_report(self, rf: float = 0.0) -> pd.DataFrame:
|
||||
"""返回策略与基准的对比报告。"""
|
||||
if self.benchmark_nav is None:
|
||||
raise ValueError("benchmark_nav is required for benchmark comparison")
|
||||
return compare_to_benchmark(self.nav, self.benchmark_nav, rf)
|
||||
|
||||
|
||||
# ── 调仓表 → 净值 ──────────────────────────────────────
|
||||
|
||||
|
||||
@@ -57,8 +76,9 @@ def compute_nav_from_weights(
|
||||
) -> pd.Series:
|
||||
"""从调仓表(日期 × 股票权重)+ 个股日收益 → 净值曲线。
|
||||
|
||||
假设:在调仓日之间权重不变(**前向填充**)。
|
||||
调仓日的权重 = `weights.loc[rebalance_date]`。
|
||||
假设:输入是该收益测量区间开始前已经生效的持仓权重,并在调仓日之间
|
||||
保持不变(**前向填充**)。本函数不会把信号日自动解释为执行日;因子分数
|
||||
应先经交易日历调度和实际执行时点处理,避免把同一时点未知的收益计入。
|
||||
|
||||
Args:
|
||||
weights: 调仓日 × 股票代码 的权重 DataFrame(**0~1**,行和 ≤ 1)
|
||||
@@ -113,6 +133,29 @@ def compute_returns_from_nav(nav: pd.Series) -> pd.Series:
|
||||
return nav.pct_change().fillna(0.0)
|
||||
|
||||
|
||||
def run_weight_backtest(
|
||||
weights: pd.DataFrame,
|
||||
stock_returns: pd.DataFrame,
|
||||
initial_capital: float = 1.0,
|
||||
tc_rate: float = 0.0,
|
||||
benchmark_nav: pd.Series | None = None,
|
||||
) -> BacktestResult:
|
||||
"""执行权重回测并返回隔离于调用方输入的结果快照。"""
|
||||
weights_snapshot = weights.copy(deep=True)
|
||||
nav = compute_nav_from_weights(
|
||||
weights=weights_snapshot,
|
||||
stock_returns=stock_returns,
|
||||
initial_capital=initial_capital,
|
||||
tc_rate=tc_rate,
|
||||
)
|
||||
return BacktestResult(
|
||||
nav=nav,
|
||||
returns=compute_returns_from_nav(nav),
|
||||
weights=weights_snapshot,
|
||||
benchmark_nav=None if benchmark_nav is None else benchmark_nav.copy(deep=True),
|
||||
)
|
||||
|
||||
|
||||
# ── 调仓工具 ──────────────────────────────────────
|
||||
|
||||
|
||||
@@ -131,6 +174,9 @@ def rebalance_periodic(
|
||||
Returns:
|
||||
调仓表 DataFrame(all_dates × 股票代码)
|
||||
"""
|
||||
if all_dates.empty:
|
||||
return pd.DataFrame(index=all_dates, columns=target_weights.index, dtype=float)
|
||||
|
||||
table = pd.DataFrame(0.0, index=all_dates, columns=target_weights.index)
|
||||
for date in rebalance_dates:
|
||||
if date not in all_dates:
|
||||
@@ -201,6 +247,8 @@ def compare_to_benchmark(
|
||||
"""
|
||||
# 对齐 index
|
||||
common = strategy_nav.index.intersection(benchmark_nav.index)
|
||||
if common.empty:
|
||||
raise ValueError("strategy and benchmark must have overlapping dates")
|
||||
s = strategy_nav.loc[common]
|
||||
b = benchmark_nav.loc[common]
|
||||
|
||||
|
||||
@@ -6,22 +6,27 @@
|
||||
- execution.py 需要**宽表**(date × stock_code)prices / volumes
|
||||
- Tushare 字段命名:`ts_code / vol(手) / amount(千元) / pct_chg`,且**无 vwap 字段**
|
||||
|
||||
本模块提供 6 个纯函数,让新模块直接吃 qtdb_pro 真实数据:
|
||||
本模块提供可组合的数据适配函数,让新模块直接吃 qtdb_pro 真实数据:
|
||||
1. `long_to_wide()` — 长表 → 宽表(date × stock_code)
|
||||
2. `wide_to_long()` — 宽表 → 长表
|
||||
3. `rename_tushare_columns()` — 列名映射(ts_code→stock_code, vol→volume 等)
|
||||
4. `add_vwap_proxy()` — vwap 代理(Tushare 无 vwap 字段)
|
||||
5. `apply_adj_factor()` — 复权(hq_daily × hq_adj_factor 前复权)
|
||||
6. `prepare_stock_series()` — 单股提取(alpha_factors 输入)
|
||||
7. `prepare_execution_inputs()` — execution 输入(prices + volumes 宽表)
|
||||
8. `load_qtdb_daily()` — 便捷加载(qtdb_pro.hq_daily + 可选复权)
|
||||
7. `prepare_asset_return_snapshot()` — 带稳定 lineage 的资产日收益
|
||||
8. `prepare_execution_inputs()` — execution 输入(prices + volumes 宽表)
|
||||
9. `load_qtdb_daily()` — 便捷加载(qtdb_pro.hq_daily + 可选复权)
|
||||
|
||||
全部纯 pandas/numpy,零新依赖,mypy strict 兼容。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from datetime import date
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
@@ -32,12 +37,14 @@ from quant_engine.logging import get_logger
|
||||
logger = get_logger(__name__)
|
||||
|
||||
__all__ = [
|
||||
"AssetReturnSnapshot",
|
||||
"long_to_wide",
|
||||
"wide_to_long",
|
||||
"rename_tushare_columns",
|
||||
"add_vwap_proxy",
|
||||
"apply_adj_factor",
|
||||
"prepare_stock_series",
|
||||
"prepare_asset_return_snapshot",
|
||||
"prepare_execution_inputs",
|
||||
"load_qtdb_daily",
|
||||
]
|
||||
@@ -57,6 +64,107 @@ TUSHARE_RENAME: dict[str, str] = {
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False, eq=False)
|
||||
class AssetReturnSnapshot:
|
||||
"""Immutable-by-interface daily return matrix with reproducible lineage."""
|
||||
|
||||
data_snapshot_id: str
|
||||
source: str
|
||||
source_snapshot_id: str
|
||||
price_field: str
|
||||
adjustment: str
|
||||
return_method: str
|
||||
start_date: date
|
||||
end_date: date
|
||||
sessions: int
|
||||
assets: tuple[str, ...]
|
||||
_returns: pd.DataFrame
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
data_snapshot_id: str,
|
||||
source: str,
|
||||
source_snapshot_id: str,
|
||||
price_field: str,
|
||||
adjustment: str,
|
||||
return_method: str,
|
||||
start_date: date,
|
||||
end_date: date,
|
||||
assets: tuple[str, ...],
|
||||
returns: pd.DataFrame,
|
||||
) -> None:
|
||||
for value, name in (
|
||||
(data_snapshot_id, "data_snapshot_id"),
|
||||
(source, "source"),
|
||||
(source_snapshot_id, "source_snapshot_id"),
|
||||
(price_field, "price_field"),
|
||||
(adjustment, "adjustment"),
|
||||
(return_method, "return_method"),
|
||||
):
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ValueError(f"{name} must be non-empty")
|
||||
if returns.empty or not isinstance(returns.index, pd.DatetimeIndex):
|
||||
raise ValueError("returns must contain a DatetimeIndex and at least one session")
|
||||
if tuple(returns.columns) != assets:
|
||||
raise ValueError("assets must match returns columns")
|
||||
if start_date > end_date:
|
||||
raise ValueError("start_date must not be after end_date")
|
||||
|
||||
object.__setattr__(self, "data_snapshot_id", data_snapshot_id.strip())
|
||||
object.__setattr__(self, "source", source.strip())
|
||||
object.__setattr__(self, "source_snapshot_id", source_snapshot_id.strip())
|
||||
object.__setattr__(self, "price_field", price_field.strip())
|
||||
object.__setattr__(self, "adjustment", adjustment.strip())
|
||||
object.__setattr__(self, "return_method", return_method.strip())
|
||||
object.__setattr__(self, "start_date", start_date)
|
||||
object.__setattr__(self, "end_date", end_date)
|
||||
object.__setattr__(self, "sessions", len(returns))
|
||||
object.__setattr__(self, "assets", assets)
|
||||
object.__setattr__(self, "_returns", returns.copy(deep=True))
|
||||
|
||||
@property
|
||||
def returns(self) -> pd.DataFrame:
|
||||
"""Return an isolated copy so callers cannot mutate the snapshot."""
|
||||
return self._returns.copy(deep=True)
|
||||
|
||||
|
||||
def _non_empty(value: str, name: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ValueError(f"{name} must be non-empty")
|
||||
return value.strip()
|
||||
|
||||
|
||||
def _asset_return_snapshot_id(
|
||||
prices: pd.DataFrame,
|
||||
*,
|
||||
source: str,
|
||||
source_snapshot_id: str,
|
||||
price_field: str,
|
||||
adjustment: str,
|
||||
) -> str:
|
||||
values = prices.to_numpy(dtype=float, copy=True)
|
||||
missing = np.isnan(values)
|
||||
normalized = np.where(missing, 0.0, values).astype("<f8", copy=False)
|
||||
metadata = {
|
||||
"adjustment": adjustment,
|
||||
"assets": [str(asset) for asset in prices.columns],
|
||||
"price_field": price_field,
|
||||
"return_method": "simple",
|
||||
"schema": "asset-returns-v1",
|
||||
"sessions": [timestamp.date().isoformat() for timestamp in prices.index],
|
||||
"shape": list(values.shape),
|
||||
"source": source,
|
||||
"source_snapshot_id": source_snapshot_id,
|
||||
}
|
||||
digest = hashlib.sha256(
|
||||
json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
||||
)
|
||||
digest.update(missing.astype(np.uint8, copy=False).tobytes(order="C"))
|
||||
digest.update(normalized.tobytes(order="C"))
|
||||
return f"asset-returns-v1:{digest.hexdigest()}"
|
||||
|
||||
|
||||
def long_to_wide(
|
||||
df: pd.DataFrame,
|
||||
value_col: str = "close",
|
||||
@@ -283,10 +391,104 @@ def prepare_stock_series(
|
||||
return series_map
|
||||
|
||||
|
||||
def prepare_asset_return_snapshot(
|
||||
df: pd.DataFrame,
|
||||
*,
|
||||
source: str,
|
||||
source_snapshot_id: str,
|
||||
price_col: str = "close",
|
||||
adjustment: str = "none",
|
||||
stock_col: str = "stock_code",
|
||||
date_col: str = "trade_date",
|
||||
) -> AssetReturnSnapshot:
|
||||
"""Build deterministic simple daily returns from a long market-price table.
|
||||
|
||||
``source_snapshot_id`` must identify the upstream ingestion snapshot. The
|
||||
resulting ID additionally fingerprints canonical price values and their
|
||||
missing mask, so changed contents cannot retain the same downstream identity.
|
||||
Missing prices are never forward-filled.
|
||||
"""
|
||||
normalized_source = _non_empty(source, "source")
|
||||
normalized_source_snapshot_id = _non_empty(
|
||||
source_snapshot_id,
|
||||
"source_snapshot_id",
|
||||
)
|
||||
normalized_price_col = _non_empty(price_col, "price_col")
|
||||
normalized_adjustment = _non_empty(adjustment, "adjustment")
|
||||
if not isinstance(df, pd.DataFrame):
|
||||
raise TypeError("df must be a pandas DataFrame")
|
||||
if df.empty:
|
||||
raise ValueError("df must contain market prices")
|
||||
required = {date_col, stock_col, normalized_price_col}
|
||||
missing_columns = sorted(required.difference(df.columns))
|
||||
if missing_columns:
|
||||
raise ValueError(f"prepare_asset_return_snapshot: missing columns={missing_columns}")
|
||||
|
||||
market = df[[date_col, stock_col, normalized_price_col]].copy()
|
||||
if any(not isinstance(asset, str) or not asset.strip() for asset in market[stock_col]):
|
||||
raise ValueError("asset labels must be non-empty strings")
|
||||
market[stock_col] = market[stock_col].str.strip()
|
||||
try:
|
||||
normalized_dates = pd.to_datetime(market[date_col], errors="raise")
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError("trade dates must be valid dates") from error
|
||||
if normalized_dates.isna().any():
|
||||
raise ValueError("trade dates must be valid dates")
|
||||
market[date_col] = normalized_dates.dt.normalize()
|
||||
if market.duplicated(subset=[date_col, stock_col]).any():
|
||||
raise ValueError("duplicate asset/session prices are not allowed")
|
||||
|
||||
try:
|
||||
market[normalized_price_col] = pd.to_numeric(
|
||||
market[normalized_price_col],
|
||||
errors="raise",
|
||||
)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError("prices must be numeric") from error
|
||||
observed_prices = market[normalized_price_col].dropna().to_numpy(dtype=float)
|
||||
if observed_prices.size == 0 or not np.isfinite(observed_prices).all():
|
||||
raise ValueError("prices must contain positive finite observations")
|
||||
if (observed_prices <= 0.0).any():
|
||||
raise ValueError("prices must contain positive finite observations")
|
||||
|
||||
prices = market.pivot(
|
||||
index=date_col,
|
||||
columns=stock_col,
|
||||
values=normalized_price_col,
|
||||
).sort_index()
|
||||
prices = prices.reindex(sorted(str(asset) for asset in prices.columns), axis="columns")
|
||||
prices = prices.astype(float)
|
||||
if len(prices) < 2:
|
||||
raise ValueError("market prices must contain at least two sessions")
|
||||
returns = prices.pct_change(fill_method=None)
|
||||
assets = tuple(str(asset) for asset in prices.columns)
|
||||
snapshot_id = _asset_return_snapshot_id(
|
||||
prices,
|
||||
source=normalized_source,
|
||||
source_snapshot_id=normalized_source_snapshot_id,
|
||||
price_field=normalized_price_col,
|
||||
adjustment=normalized_adjustment,
|
||||
)
|
||||
return AssetReturnSnapshot(
|
||||
data_snapshot_id=snapshot_id,
|
||||
source=normalized_source,
|
||||
source_snapshot_id=normalized_source_snapshot_id,
|
||||
price_field=normalized_price_col,
|
||||
adjustment=normalized_adjustment,
|
||||
return_method="simple",
|
||||
start_date=prices.index[0].date(),
|
||||
end_date=prices.index[-1].date(),
|
||||
assets=assets,
|
||||
returns=returns,
|
||||
)
|
||||
|
||||
|
||||
def prepare_execution_inputs(
|
||||
df: pd.DataFrame,
|
||||
stock_col: str = "stock_code",
|
||||
date_col: str = "trade_date",
|
||||
*,
|
||||
price_col: str = "close",
|
||||
) -> tuple[pd.DataFrame, pd.DataFrame]:
|
||||
"""长表行情 → execution 输入(prices + volumes 宽表)。
|
||||
|
||||
@@ -294,10 +496,11 @@ def prepare_execution_inputs(
|
||||
df: 长表行情(含 close / volume 列,Tushare rename 后)
|
||||
stock_col: 股票代码列名
|
||||
date_col: 日期列名
|
||||
price_col: 执行价字段,默认 close;防前视研究可显式选择下一交易日 open
|
||||
|
||||
Returns:
|
||||
(prices_wide, volumes_wide):
|
||||
- prices_wide: date × stock_code,值=close
|
||||
- prices_wide: date × stock_code,值=price_col
|
||||
- volumes_wide: date × stock_code,值=volume(若无 volume 列则全 1.0)
|
||||
|
||||
Examples:
|
||||
@@ -313,9 +516,9 @@ def prepare_execution_inputs(
|
||||
"""
|
||||
if df.empty:
|
||||
return pd.DataFrame(), pd.DataFrame()
|
||||
if "close" not in df.columns:
|
||||
raise ValueError(f"prepare_execution_inputs: 缺 close 列,实际列={list(df.columns)}")
|
||||
prices = long_to_wide(df, value_col="close", date_col=date_col, stock_col=stock_col)
|
||||
if price_col not in df.columns:
|
||||
raise ValueError(f"prepare_execution_inputs: 缺 {price_col} 列,实际列={list(df.columns)}")
|
||||
prices = long_to_wide(df, value_col=price_col, date_col=date_col, stock_col=stock_col)
|
||||
if "volume" in df.columns:
|
||||
volumes = long_to_wide(df, value_col="volume", date_col=date_col, stock_col=stock_col)
|
||||
else:
|
||||
|
||||
+495
-147
@@ -10,7 +10,9 @@
|
||||
借鉴 hikyuu SG/MM/CN/PG 部件化思想(不引入 hikyuu 框架):
|
||||
- ExecutionConfig:佣金 + 印花税 + 滑点 + 最小交易额 + 止损/止盈阈值
|
||||
- simulate_execution():从目标权重 → 实际成交金额(应用成本/滑点)
|
||||
- simulate_multi_day():多日组合仿真(NAV 序列 + 调仓记录)
|
||||
- simulate_daily_ledger_with_audit():稀疏调仓 + 完整交易日收盘估值 Ledger
|
||||
- simulate_multi_day_with_audit():目标权重差额调仓(成交/拒绝/持仓/NAV)
|
||||
- simulate_multi_day():兼容的多日日末持仓快照入口
|
||||
- check_stop_loss_take_profit():止损/止盈触发判定
|
||||
- run_end_to_end_poc():signal → 调仓 → 执行 → NAV 端到端 POC
|
||||
|
||||
@@ -19,8 +21,9 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from dataclasses import dataclass, replace
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
@@ -101,6 +104,9 @@ class ExecutionResult:
|
||||
net_cash_flow: float # 净现金流(买入为负,卖出为正)
|
||||
partial_fill_pct: float = 1.0 # 实际成交占目标的比例(1.0 = 全部成交)
|
||||
blocked_reason: str = "" # 阻塞原因(如涨跌停停牌)
|
||||
side: str = "" # buy / sell;未成交记录也保留目标方向
|
||||
quantity: float = 0.0 # 实际成交股数
|
||||
price: float = 0.0 # 未含滑点的参考执行价
|
||||
|
||||
|
||||
def _apply_costs(
|
||||
@@ -344,19 +350,458 @@ class DailyExecution:
|
||||
"""单日执行记录。"""
|
||||
|
||||
date: str
|
||||
executions: list[ExecutionResult]
|
||||
executions: tuple[ExecutionResult, ...]
|
||||
nav_before: float
|
||||
nav_after: float
|
||||
rebalance_triggered: bool
|
||||
|
||||
|
||||
def simulate_multi_day(
|
||||
@dataclass(frozen=True)
|
||||
class ExecutionSimulationResult:
|
||||
"""单次多日仿真的持仓与执行审计结果。"""
|
||||
|
||||
initial_cash: float
|
||||
positions: tuple[DailyPosition, ...]
|
||||
daily_executions: tuple[DailyExecution, ...]
|
||||
|
||||
@property
|
||||
def nav_series(self) -> pd.Series:
|
||||
"""返回按日期索引的日末 NAV 副本。"""
|
||||
return pd.Series(
|
||||
[position.portfolio_value for position in self.positions],
|
||||
index=[position.date for position in self.positions],
|
||||
dtype=float,
|
||||
)
|
||||
|
||||
@property
|
||||
def normalized_nav_series(self) -> pd.Series:
|
||||
"""返回以初始资金为 1 的净值曲线副本。"""
|
||||
nav = self.nav_series
|
||||
if self.initial_cash == 0:
|
||||
return pd.Series(0.0, index=nav.index, dtype=float)
|
||||
return nav / self.initial_cash
|
||||
|
||||
@property
|
||||
def daily_returns(self) -> pd.Series:
|
||||
"""返回逐日收益;首日相对初始资金计算,保留首日交易成本。"""
|
||||
nav = self.nav_series
|
||||
if nav.empty:
|
||||
return nav
|
||||
returns = nav.pct_change()
|
||||
returns.iloc[0] = (
|
||||
nav.iloc[0] / self.initial_cash - 1.0 if self.initial_cash != 0 else 0.0
|
||||
)
|
||||
return returns.fillna(0.0)
|
||||
|
||||
@property
|
||||
def trades_frame(self) -> pd.DataFrame:
|
||||
"""返回可投影到平台成交明细的实际成交表,不包含纯拒绝记录。"""
|
||||
columns = [
|
||||
"trade_date",
|
||||
"ts_code",
|
||||
"side",
|
||||
"qty",
|
||||
"price",
|
||||
"amount",
|
||||
"fee",
|
||||
"slippage",
|
||||
]
|
||||
rows = [
|
||||
{
|
||||
"trade_date": daily.date,
|
||||
"ts_code": execution.stock_code,
|
||||
"side": execution.side,
|
||||
"qty": execution.quantity,
|
||||
"price": execution.price,
|
||||
"amount": execution.executed_value,
|
||||
"fee": execution.commission + execution.stamp_tax,
|
||||
"slippage": execution.slippage_cost,
|
||||
}
|
||||
for daily in self.daily_executions
|
||||
for execution in daily.executions
|
||||
if execution.quantity > 0
|
||||
]
|
||||
return pd.DataFrame(rows, columns=columns)
|
||||
|
||||
@property
|
||||
def ledger_frame(self) -> pd.DataFrame:
|
||||
"""返回稳定的日频 Ledger 投影,不附加运行元数据或写数据库。"""
|
||||
columns = [
|
||||
"trade_date",
|
||||
"portfolio_value",
|
||||
"nav",
|
||||
"pnl",
|
||||
"pnl_pct",
|
||||
"position_value",
|
||||
"cash",
|
||||
"turnover",
|
||||
]
|
||||
previous_value = self.initial_cash
|
||||
rows: list[dict[str, float | str]] = []
|
||||
daily_returns = self.daily_returns
|
||||
for index, (position, daily) in enumerate(
|
||||
zip(self.positions, self.daily_executions, strict=True)
|
||||
):
|
||||
daily_turnover = sum(
|
||||
execution.executed_value
|
||||
for execution in daily.executions
|
||||
if execution.quantity > 0
|
||||
)
|
||||
turnover_rate = daily_turnover / daily.nav_before if daily.nav_before > 0 else 0.0
|
||||
rows.append(
|
||||
{
|
||||
"trade_date": position.date,
|
||||
"portfolio_value": position.portfolio_value,
|
||||
"nav": (
|
||||
position.portfolio_value / self.initial_cash
|
||||
if self.initial_cash != 0
|
||||
else 0.0
|
||||
),
|
||||
"pnl": position.portfolio_value - previous_value,
|
||||
"pnl_pct": float(daily_returns.iloc[index]),
|
||||
"position_value": position.portfolio_value - position.cash,
|
||||
"cash": position.cash,
|
||||
"turnover": turnover_rate,
|
||||
}
|
||||
)
|
||||
previous_value = position.portfolio_value
|
||||
return pd.DataFrame(rows, columns=columns)
|
||||
|
||||
@property
|
||||
def total_costs(self) -> float:
|
||||
"""汇总实际成交产生的成本。"""
|
||||
return sum(
|
||||
execution.total_cost
|
||||
for daily in self.daily_executions
|
||||
for execution in daily.executions
|
||||
)
|
||||
|
||||
@property
|
||||
def total_turnover(self) -> float:
|
||||
"""汇总实际成交金额。"""
|
||||
return sum(
|
||||
execution.executed_value
|
||||
for daily in self.daily_executions
|
||||
for execution in daily.executions
|
||||
)
|
||||
|
||||
@property
|
||||
def total_rebalances(self) -> int:
|
||||
"""返回至少有一笔实际成交的调仓日数量。"""
|
||||
return sum(daily.rebalance_triggered for daily in self.daily_executions)
|
||||
|
||||
@property
|
||||
def final_portfolio_value(self) -> float:
|
||||
"""返回最后一个日末 NAV;空输入时返回初始资金。"""
|
||||
if not self.positions:
|
||||
return self.initial_cash
|
||||
return self.positions[-1].portfolio_value
|
||||
|
||||
@property
|
||||
def return_pct(self) -> float:
|
||||
"""返回相对初始资金的百分比收益。"""
|
||||
if self.initial_cash == 0:
|
||||
return 0.0
|
||||
return (self.final_portfolio_value / self.initial_cash - 1.0) * 100.0
|
||||
|
||||
|
||||
def _blocked_execution(stock_code: str, target_value: float, reason: str) -> ExecutionResult:
|
||||
"""构造未成交但可审计的执行记录。"""
|
||||
return ExecutionResult(
|
||||
stock_code=stock_code,
|
||||
target_value=target_value,
|
||||
executed_value=0.0,
|
||||
commission=0.0,
|
||||
stamp_tax=0.0,
|
||||
slippage_cost=0.0,
|
||||
total_cost=0.0,
|
||||
net_cash_flow=0.0,
|
||||
partial_fill_pct=0.0,
|
||||
blocked_reason=reason,
|
||||
side="buy" if target_value > 0 else "sell" if target_value < 0 else "",
|
||||
)
|
||||
|
||||
|
||||
def _validate_target_weights(date: str, targets: Mapping[str, float]) -> dict[str, float]:
|
||||
"""校验并复制单日长仓目标权重。"""
|
||||
normalized: dict[str, float] = {}
|
||||
for stock_code, raw_weight in targets.items():
|
||||
try:
|
||||
weight = float(raw_weight)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError(f"target weights on {date!r} must be numeric") from error
|
||||
if not math.isfinite(weight) or weight < 0:
|
||||
raise ValueError(f"target weights on {date!r} must be finite and non-negative")
|
||||
normalized[stock_code] = weight
|
||||
if sum(normalized.values()) > 1.0 + 1e-12:
|
||||
raise ValueError(f"target weights on {date!r} must sum to at most 1.0")
|
||||
return normalized
|
||||
|
||||
|
||||
def _partially_fill_buy(
|
||||
desired: ExecutionResult,
|
||||
fill_pct: float,
|
||||
config: ExecutionConfig,
|
||||
) -> ExecutionResult:
|
||||
"""按同一比例缩放买入,保留原始目标金额供审计。"""
|
||||
actual_target_value = desired.target_value * fill_pct
|
||||
executed_value, commission, stamp_tax, slippage_cost = _apply_costs(
|
||||
actual_target_value,
|
||||
True,
|
||||
config,
|
||||
)
|
||||
total_cost = commission + stamp_tax + slippage_cost
|
||||
return ExecutionResult(
|
||||
stock_code=desired.stock_code,
|
||||
target_value=desired.target_value,
|
||||
executed_value=executed_value,
|
||||
commission=commission,
|
||||
stamp_tax=stamp_tax,
|
||||
slippage_cost=slippage_cost,
|
||||
total_cost=total_cost,
|
||||
net_cash_flow=-(executed_value + commission + stamp_tax),
|
||||
partial_fill_pct=fill_pct,
|
||||
blocked_reason="insufficient_cash_partial_fill",
|
||||
)
|
||||
|
||||
|
||||
def _rebalance_at_prices(
|
||||
date: str,
|
||||
targets: Mapping[str, float],
|
||||
prices: Mapping[str, float],
|
||||
cash: float,
|
||||
holdings: dict[str, float],
|
||||
config: ExecutionConfig,
|
||||
) -> tuple[float, tuple[ExecutionResult, ...], float, float]:
|
||||
"""在单一执行时点按目标权重差额调仓,并原地更新 holdings。"""
|
||||
normalized_targets = _validate_target_weights(date, targets)
|
||||
for held_code in holdings:
|
||||
held_price = prices.get(held_code)
|
||||
if held_price is None or not math.isfinite(held_price) or held_price <= 0:
|
||||
raise ValueError(f"missing price for held asset {held_code} on {date!r}")
|
||||
|
||||
nav_before = cash + sum(
|
||||
shares * prices.get(stock_code, 0.0)
|
||||
for stock_code, shares in holdings.items()
|
||||
)
|
||||
effective_targets = dict.fromkeys(holdings, 0.0)
|
||||
effective_targets.update(normalized_targets)
|
||||
buy_weights: dict[str, float] = {}
|
||||
sell_weights: dict[str, float] = {}
|
||||
rejected: list[ExecutionResult] = []
|
||||
|
||||
for stock_code, target_weight in effective_targets.items():
|
||||
price = prices.get(stock_code)
|
||||
target_value = float(target_weight) * nav_before
|
||||
if price is None or not math.isfinite(price) or price <= 0:
|
||||
if target_value != 0 or holdings.get(stock_code, 0.0) != 0:
|
||||
rejected.append(_blocked_execution(stock_code, target_value, "missing_price"))
|
||||
continue
|
||||
|
||||
current_value = holdings.get(stock_code, 0.0) * price
|
||||
trade_value = target_value - current_value
|
||||
if abs(trade_value) < config.min_trade_amount or math.isclose(
|
||||
trade_value, 0.0, abs_tol=1e-12
|
||||
):
|
||||
continue
|
||||
if nav_before == 0:
|
||||
rejected.append(_blocked_execution(stock_code, trade_value, "zero_nav"))
|
||||
continue
|
||||
destination = buy_weights if trade_value > 0 else sell_weights
|
||||
destination[stock_code] = trade_value / nav_before
|
||||
|
||||
sell_executions = simulate_execution(sell_weights, nav_before, config)
|
||||
filled: list[ExecutionResult] = []
|
||||
for raw_execution in sell_executions:
|
||||
price = prices[raw_execution.stock_code]
|
||||
quantity = abs(raw_execution.target_value) / price
|
||||
execution = replace(
|
||||
raw_execution,
|
||||
side="sell",
|
||||
quantity=quantity,
|
||||
price=price,
|
||||
)
|
||||
held = holdings.get(execution.stock_code, 0.0)
|
||||
holdings[execution.stock_code] = max(0.0, held - quantity)
|
||||
if holdings[execution.stock_code] < 1e-6:
|
||||
del holdings[execution.stock_code]
|
||||
cash += execution.net_cash_flow
|
||||
filled.append(execution)
|
||||
|
||||
desired_buys = simulate_execution(buy_weights, nav_before, config)
|
||||
required_cash = sum(-execution.net_cash_flow for execution in desired_buys)
|
||||
buy_fill_pct = min(1.0, max(cash, 0.0) / required_cash) if required_cash > 0 else 1.0
|
||||
for desired in desired_buys:
|
||||
if buy_fill_pct == 0:
|
||||
rejected.append(
|
||||
_blocked_execution(desired.stock_code, desired.target_value, "insufficient_cash")
|
||||
)
|
||||
continue
|
||||
raw_execution = (
|
||||
desired
|
||||
if buy_fill_pct == 1.0
|
||||
else _partially_fill_buy(desired, buy_fill_pct, config)
|
||||
)
|
||||
price = prices[raw_execution.stock_code]
|
||||
quantity = abs(raw_execution.target_value) * raw_execution.partial_fill_pct / price
|
||||
execution = replace(
|
||||
raw_execution,
|
||||
side="buy",
|
||||
quantity=quantity,
|
||||
price=price,
|
||||
)
|
||||
holdings[execution.stock_code] = holdings.get(execution.stock_code, 0.0) + quantity
|
||||
cash += execution.net_cash_flow
|
||||
if math.isclose(cash, 0.0, abs_tol=1e-9):
|
||||
cash = 0.0
|
||||
filled.append(execution)
|
||||
|
||||
executions = (*filled, *rejected)
|
||||
nav_after = cash + sum(
|
||||
shares * prices.get(stock_code, 0.0)
|
||||
for stock_code, shares in holdings.items()
|
||||
)
|
||||
return cash, executions, nav_before, nav_after
|
||||
|
||||
|
||||
def _validate_sparse_daily_histories(
|
||||
target_weights_history: list[tuple[str, dict[str, float]]],
|
||||
execution_price_history: list[tuple[str, dict[str, float]]],
|
||||
valuation_price_history: list[tuple[str, dict[str, float]]],
|
||||
) -> tuple[
|
||||
dict[str, dict[str, float]],
|
||||
dict[str, dict[str, float]],
|
||||
list[tuple[str, dict[str, float]]],
|
||||
]:
|
||||
"""校验稀疏调仓与完整估值日历,并隔离调用方可变输入。"""
|
||||
target_dates = [date for date, _ in target_weights_history]
|
||||
execution_dates = [date for date, _ in execution_price_history]
|
||||
valuation_dates = [date for date, _ in valuation_price_history]
|
||||
if len(set(target_dates)) != len(target_dates):
|
||||
raise ValueError("target_weights_history must contain unique dates")
|
||||
if len(set(execution_dates)) != len(execution_dates):
|
||||
raise ValueError("execution_price_history must contain unique dates")
|
||||
if len(set(valuation_dates)) != len(valuation_dates):
|
||||
raise ValueError("valuation_price_history must contain unique dates")
|
||||
if execution_dates != target_dates:
|
||||
raise ValueError("execution price dates must exactly match target weight dates")
|
||||
|
||||
valuation_positions = {date: index for index, date in enumerate(valuation_dates)}
|
||||
missing_dates = [date for date in target_dates if date not in valuation_positions]
|
||||
if missing_dates:
|
||||
raise ValueError(f"target dates must belong to valuation calendar: {missing_dates}")
|
||||
positions = [valuation_positions[date] for date in target_dates]
|
||||
if positions != sorted(positions):
|
||||
raise ValueError("target weights must follow valuation calendar order")
|
||||
|
||||
targets = {date: dict(values) for date, values in target_weights_history}
|
||||
execution_prices = {date: dict(values) for date, values in execution_price_history}
|
||||
valuation_prices = [(date, dict(values)) for date, values in valuation_price_history]
|
||||
return targets, execution_prices, valuation_prices
|
||||
|
||||
|
||||
def _simulate_daily_ledger(
|
||||
target_weights_history: list[tuple[str, dict[str, float]]],
|
||||
execution_price_history: list[tuple[str, dict[str, float]]],
|
||||
valuation_price_history: list[tuple[str, dict[str, float]]],
|
||||
initial_cash: float,
|
||||
config: ExecutionConfig,
|
||||
) -> ExecutionSimulationResult:
|
||||
targets_by_date, execution_prices_by_date, valuation_history = (
|
||||
_validate_sparse_daily_histories(
|
||||
target_weights_history,
|
||||
execution_price_history,
|
||||
valuation_price_history,
|
||||
)
|
||||
)
|
||||
cash = initial_cash
|
||||
holdings: dict[str, float] = {}
|
||||
positions: list[DailyPosition] = []
|
||||
daily_executions: list[DailyExecution] = []
|
||||
|
||||
for date, valuation_prices in valuation_history:
|
||||
targets = targets_by_date.get(date)
|
||||
if targets is None:
|
||||
executions: tuple[ExecutionResult, ...] = ()
|
||||
nav_before = 0.0
|
||||
nav_after = 0.0
|
||||
rebalance_triggered = False
|
||||
else:
|
||||
cash, executions, nav_before, nav_after = _rebalance_at_prices(
|
||||
date,
|
||||
targets,
|
||||
execution_prices_by_date[date],
|
||||
cash,
|
||||
holdings,
|
||||
config,
|
||||
)
|
||||
rebalance_triggered = any(execution.quantity > 0 for execution in executions)
|
||||
|
||||
for held_code in holdings:
|
||||
valuation_price = valuation_prices.get(held_code)
|
||||
if (
|
||||
valuation_price is None
|
||||
or not math.isfinite(valuation_price)
|
||||
or valuation_price <= 0
|
||||
):
|
||||
raise ValueError(
|
||||
f"missing valuation price for held asset {held_code} on {date!r}"
|
||||
)
|
||||
portfolio_value = cash + sum(
|
||||
shares * valuation_prices[stock_code]
|
||||
for stock_code, shares in holdings.items()
|
||||
)
|
||||
if targets is None:
|
||||
nav_before = portfolio_value
|
||||
nav_after = portfolio_value
|
||||
positions.append(DailyPosition(date, cash, dict(holdings), portfolio_value))
|
||||
daily_executions.append(
|
||||
DailyExecution(
|
||||
date=date,
|
||||
executions=executions,
|
||||
nav_before=nav_before,
|
||||
nav_after=nav_after,
|
||||
rebalance_triggered=rebalance_triggered,
|
||||
)
|
||||
)
|
||||
|
||||
return ExecutionSimulationResult(
|
||||
initial_cash=initial_cash,
|
||||
positions=tuple(positions),
|
||||
daily_executions=tuple(daily_executions),
|
||||
)
|
||||
|
||||
|
||||
def simulate_daily_ledger_with_audit(
|
||||
target_weights_history: list[tuple[str, dict[str, float]]],
|
||||
execution_price_history: list[tuple[str, dict[str, float]]],
|
||||
valuation_price_history: list[tuple[str, dict[str, float]]],
|
||||
initial_cash: float,
|
||||
config: ExecutionConfig | None = None,
|
||||
) -> ExecutionSimulationResult:
|
||||
"""以稀疏调仓和完整日历运行成交后持仓 Ledger。
|
||||
|
||||
执行价只用于调仓日现金与股数变化,估值价用于每个交易日日末 NAV;二者
|
||||
显式分离,从而支持“下一日 open 成交、同日 close 估值”的无前视研究。
|
||||
"""
|
||||
if not math.isfinite(initial_cash) or initial_cash <= 0:
|
||||
raise ValueError(f"initial_cash must be positive and finite, got {initial_cash}")
|
||||
return _simulate_daily_ledger(
|
||||
target_weights_history,
|
||||
execution_price_history,
|
||||
valuation_price_history,
|
||||
initial_cash,
|
||||
ExecutionConfig() if config is None else config,
|
||||
)
|
||||
|
||||
|
||||
def simulate_multi_day_with_audit(
|
||||
target_weights_history: list[tuple[str, dict[str, float]]],
|
||||
price_history: list[tuple[str, dict[str, float]]],
|
||||
initial_cash: float,
|
||||
config: ExecutionConfig | None = None,
|
||||
) -> list[DailyPosition]:
|
||||
"""多日组合仿真(NAV 序列)。
|
||||
) -> ExecutionSimulationResult:
|
||||
"""按目标权重差额推进组合,并返回唯一事实来源的审计结果。
|
||||
|
||||
Args:
|
||||
target_weights_history: [(date, {stock_code: target_weight})]
|
||||
@@ -365,97 +810,45 @@ def simulate_multi_day(
|
||||
config: 执行配置
|
||||
|
||||
Returns:
|
||||
DailyPosition 列表(每日 NAV 快照)。
|
||||
日末持仓快照与逐日成交记录组成的结构化审计结果。
|
||||
|
||||
Note:
|
||||
- 调仓频率 = target_weights_history 的频率(每日 / 每周 / 每月都行)
|
||||
- 每日先按当日 close 估值,再按当日 target 调仓(下一交易日生效)
|
||||
- 此处简化:调仓使用当日 close 价格
|
||||
- 每日先按当日 close 估值,再交易“目标市值 - 当前市值”的差额
|
||||
- 此处简化为当日 close 成交;调用方必须传入已正确滞后的目标权重
|
||||
"""
|
||||
if config is None:
|
||||
config = ExecutionConfig()
|
||||
if not math.isfinite(initial_cash) or initial_cash < 0:
|
||||
raise ValueError(f"initial_cash must be finite and non-negative, got {initial_cash}")
|
||||
if len(target_weights_history) != len(price_history):
|
||||
raise ValueError("target_weights_history and price_history must have same length")
|
||||
if not target_weights_history:
|
||||
return []
|
||||
cash = initial_cash
|
||||
holdings: dict[str, float] = {}
|
||||
positions: list[DailyPosition] = []
|
||||
for (date, targets), (_, prices) in zip(target_weights_history, price_history, strict=True):
|
||||
# 1) 先按当日收盘价估值
|
||||
portfolio_value = cash + sum(
|
||||
shares * prices.get(code, 0.0) for code, shares in holdings.items()
|
||||
)
|
||||
positions.append(
|
||||
DailyPosition(
|
||||
date=date,
|
||||
cash=cash,
|
||||
holdings=dict(holdings),
|
||||
portfolio_value=portfolio_value,
|
||||
for (date, _), (price_date, _) in zip(target_weights_history, price_history, strict=True):
|
||||
if date != price_date:
|
||||
raise ValueError(
|
||||
f"target and price dates must match, got {date!r} and {price_date!r}"
|
||||
)
|
||||
)
|
||||
# 2) 计算 effective_targets(包含需要平仓的零权重)
|
||||
effective_targets: dict[str, float] = dict(targets)
|
||||
for held_code in holdings:
|
||||
if held_code not in effective_targets:
|
||||
effective_targets[held_code] = 0.0
|
||||
# 3) 调仓(只对非零目标调用 simulate_execution)
|
||||
non_zero_targets = {k: v for k, v in effective_targets.items() if v != 0}
|
||||
results = simulate_execution(non_zero_targets, portfolio_value, config)
|
||||
# 4) 处理零目标(平仓):构造 ExecutionResult,shares = held(全部卖出)
|
||||
for stock_code, weight in effective_targets.items():
|
||||
if weight == 0 and stock_code in holdings and holdings[stock_code] > 0:
|
||||
price = prices.get(stock_code, 0.0)
|
||||
if price > 0:
|
||||
held = holdings[stock_code]
|
||||
# 全部卖出:target_shares = held
|
||||
# executed_value = held * price(考虑滑点)
|
||||
slippage_factor = 1.0 - config.slippage_bps / 10000.0
|
||||
target_value = -held * price
|
||||
executed_value = target_value * slippage_factor
|
||||
commission = abs(executed_value) * config.commission_bps / 10000.0
|
||||
stamp_tax = abs(executed_value) * config.stamp_tax_bps / 10000.0
|
||||
slippage_cost = abs(executed_value - target_value)
|
||||
# 标记净卖出 shares = held
|
||||
results.append(
|
||||
ExecutionResult(
|
||||
stock_code=stock_code,
|
||||
target_value=target_value,
|
||||
executed_value=executed_value,
|
||||
commission=commission,
|
||||
stamp_tax=stamp_tax,
|
||||
slippage_cost=slippage_cost,
|
||||
total_cost=commission + stamp_tax + slippage_cost,
|
||||
net_cash_flow=executed_value - commission - stamp_tax,
|
||||
)
|
||||
)
|
||||
# 5) 应用执行结果到持仓
|
||||
for r in results:
|
||||
cost = r.executed_value + r.commission + r.stamp_tax
|
||||
proceeds = r.executed_value - r.commission - r.stamp_tax
|
||||
price = prices.get(r.stock_code, 0.0)
|
||||
if r.target_value > 0:
|
||||
# 买入:shares = 正数 executed_value / price,cash 减少 cost
|
||||
shares = r.executed_value / price if price > 0 else 0.0
|
||||
holdings[r.stock_code] = holdings.get(r.stock_code, 0.0) + shares
|
||||
cash -= cost
|
||||
else:
|
||||
# 卖出:cash 增加 proceeds 的绝对值(proceeds 本是负的)
|
||||
held = holdings.get(r.stock_code, 0.0)
|
||||
if held > 0:
|
||||
# 如果是 zero-target 触发的全卖(target_value 与持仓市值近似),全部卖出
|
||||
if abs(r.target_value) >= held * price * 0.95:
|
||||
sell_shares = held
|
||||
else:
|
||||
target_shares = abs(r.executed_value) / price if price > 0 else held
|
||||
sell_shares = min(held, target_shares)
|
||||
holdings[r.stock_code] = held - sell_shares
|
||||
if holdings[r.stock_code] < 1e-6:
|
||||
del holdings[r.stock_code]
|
||||
# proceeds 是负的(target_value 负),cash += proceeds 实际是减去
|
||||
# 但卖出是现金流入,所以应该 cash += abs(proceeds)
|
||||
cash += abs(proceeds)
|
||||
return positions
|
||||
return _simulate_daily_ledger(
|
||||
target_weights_history,
|
||||
price_history,
|
||||
price_history,
|
||||
initial_cash,
|
||||
ExecutionConfig() if config is None else config,
|
||||
)
|
||||
|
||||
|
||||
def simulate_multi_day(
|
||||
target_weights_history: list[tuple[str, dict[str, float]]],
|
||||
price_history: list[tuple[str, dict[str, float]]],
|
||||
initial_cash: float,
|
||||
config: ExecutionConfig | None = None,
|
||||
) -> list[DailyPosition]:
|
||||
"""兼容入口:返回多日仿真的日末持仓快照。"""
|
||||
result = simulate_multi_day_with_audit(
|
||||
target_weights_history,
|
||||
price_history,
|
||||
initial_cash,
|
||||
config,
|
||||
)
|
||||
return list(result.positions)
|
||||
|
||||
|
||||
def run_end_to_end_poc(
|
||||
@@ -486,64 +879,16 @@ def run_end_to_end_poc(
|
||||
config = ExecutionConfig()
|
||||
if len(signals) != len(prices):
|
||||
raise ValueError("signals and prices must have same length")
|
||||
positions = simulate_multi_day(signals, prices, initial_cash, config)
|
||||
nav_series = pd.Series(
|
||||
[p.portfolio_value for p in positions], index=[p.date for p in positions]
|
||||
)
|
||||
# 计算 total_costs / total_turnover(重放所有执行)
|
||||
total_cost_acc = 0.0
|
||||
total_turnover_acc = 0.0
|
||||
rebalance_count = 0
|
||||
cash = initial_cash
|
||||
holdings: dict[str, float] = {}
|
||||
for (date, targets), (_, price_map) in zip(signals, prices, strict=True):
|
||||
portfolio_value = cash + sum(
|
||||
shares * price_map.get(code, 0.0) for code, shares in holdings.items()
|
||||
)
|
||||
if targets:
|
||||
rebalance_count += 1
|
||||
# 自动平仓:持仓但不在 target 中的股票
|
||||
effective_targets: dict[str, float] = dict(targets)
|
||||
for held_code in holdings:
|
||||
if held_code not in effective_targets:
|
||||
effective_targets[held_code] = 0.0
|
||||
results = simulate_execution(effective_targets, portfolio_value, config)
|
||||
total_cost_acc += total_costs(results)
|
||||
total_turnover_acc += total_turnover(results)
|
||||
for r in results:
|
||||
cost = r.executed_value + r.commission + r.stamp_tax
|
||||
proceeds = r.executed_value - r.commission - r.stamp_tax
|
||||
if r.target_value > 0:
|
||||
shares = (
|
||||
r.executed_value / price_map[r.stock_code]
|
||||
if price_map[r.stock_code] > 0
|
||||
else 0.0
|
||||
)
|
||||
holdings[r.stock_code] = holdings.get(r.stock_code, 0.0) + shares
|
||||
cash -= cost
|
||||
else:
|
||||
held = holdings.get(r.stock_code, 0.0)
|
||||
if held > 0:
|
||||
sell_shares = min(
|
||||
held,
|
||||
abs(r.executed_value / price_map[r.stock_code])
|
||||
if price_map[r.stock_code] > 0
|
||||
else held,
|
||||
)
|
||||
holdings[r.stock_code] = held - sell_shares
|
||||
if holdings[r.stock_code] < 1e-6:
|
||||
del holdings[r.stock_code]
|
||||
cash += proceeds
|
||||
audit = simulate_multi_day_with_audit(signals, prices, initial_cash, config)
|
||||
return {
|
||||
"positions": positions,
|
||||
"nav_series": nav_series,
|
||||
"total_costs": total_cost_acc,
|
||||
"total_turnover": total_turnover_acc,
|
||||
"total_rebalances": rebalance_count,
|
||||
"final_portfolio_value": nav_series.iloc[-1] if len(nav_series) > 0 else initial_cash,
|
||||
"return_pct": ((nav_series.iloc[-1] / initial_cash) - 1) * 100
|
||||
if len(nav_series) > 0
|
||||
else 0.0,
|
||||
"positions": list(audit.positions),
|
||||
"daily_executions": list(audit.daily_executions),
|
||||
"nav_series": audit.nav_series,
|
||||
"total_costs": audit.total_costs,
|
||||
"total_turnover": audit.total_turnover,
|
||||
"total_rebalances": audit.total_rebalances,
|
||||
"final_portfolio_value": audit.final_portfolio_value,
|
||||
"return_pct": audit.return_pct,
|
||||
}
|
||||
|
||||
|
||||
@@ -667,7 +1012,10 @@ __all__ = [
|
||||
"apply_bid_ask_spread",
|
||||
"DailyPosition",
|
||||
"DailyExecution",
|
||||
"ExecutionSimulationResult",
|
||||
"simulate_daily_ledger_with_audit",
|
||||
"simulate_multi_day",
|
||||
"simulate_multi_day_with_audit",
|
||||
"run_end_to_end_poc",
|
||||
"DailyPnL",
|
||||
"simulate_with_daily_data",
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -312,8 +312,8 @@ def ols_regress(
|
||||
ss_tot = float(((y_arr - y_arr.mean()) ** 2).sum())
|
||||
r_sq = 1.0 - ss_res / ss_tot if ss_tot > 0 else np.nan
|
||||
sigma2 = ss_res / max(n - k, 1)
|
||||
# 协方差矩阵 = sigma2 * (X'X)^-1
|
||||
xtx_inv = np.linalg.inv(x_arr.T @ x_arr) if sigma2 > 0 else np.full((k, k), np.nan)
|
||||
# 广义协方差矩阵 = sigma2 * (X'X)^+,伪逆兼容共线因子。
|
||||
xtx_inv = np.linalg.pinv(x_arr.T @ x_arr) if sigma2 > 0 else np.full((k, k), np.nan)
|
||||
se = np.sqrt(np.diag(xtx_inv) * sigma2)
|
||||
t_vals = coef / se if sigma2 > 0 else np.full_like(coef, np.nan)
|
||||
if add_constant:
|
||||
@@ -513,6 +513,8 @@ def apply_factor_direction(
|
||||
Returns:
|
||||
方向调整后的因子(同向 = 越大越好)
|
||||
"""
|
||||
if direction not in {"auto", "forward", "reverse"}:
|
||||
raise ValueError(f"direction={direction!r} not supported (auto / forward / reverse)")
|
||||
if factor.empty:
|
||||
return factor.copy()
|
||||
if direction == "auto":
|
||||
@@ -541,6 +543,8 @@ def cross_sectional_rank_with_direction(
|
||||
Returns:
|
||||
pd.Series(百分位排名 [0, 1],越大越优)
|
||||
"""
|
||||
if direction not in {"auto", "forward", "reverse"}:
|
||||
raise ValueError(f"direction={direction!r} not supported (auto / forward / reverse)")
|
||||
if df.empty or factor_col not in df.columns:
|
||||
return pd.Series(dtype=float)
|
||||
factor = df[factor_col]
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -65,13 +65,28 @@ def sharpe_ratio(r: pd.Series, rf: float = 0.0) -> float:
|
||||
return (annualized_return(r) - rf) / vol
|
||||
|
||||
|
||||
def sortino_ratio(r: pd.Series, rf: float = 0.0) -> float:
|
||||
"""Sortino = (年化收益 - rf) / 年化下行偏差。"""
|
||||
r = _clean(r)
|
||||
if len(r) < 2:
|
||||
return 0.0
|
||||
downside = np.minimum(r.to_numpy(dtype=float), 0.0)
|
||||
downside_deviation = float(
|
||||
np.sqrt(np.mean(np.square(downside))) * np.sqrt(TRADING_DAYS_PER_YEAR)
|
||||
)
|
||||
if downside_deviation == 0:
|
||||
return 0.0
|
||||
return (annualized_return(r) - rf) / downside_deviation
|
||||
|
||||
|
||||
def max_drawdown(r: pd.Series) -> float:
|
||||
"""最大回撤(负数)。例如 -0.2 表示最大亏 20%。"""
|
||||
r = _clean(r)
|
||||
if len(r) < 2:
|
||||
return 0.0
|
||||
nav = (1 + r).cumprod()
|
||||
peak = nav.cummax()
|
||||
# 初始资金净值为 1;否则首个观测日的亏损会被误当成新的历史高点。
|
||||
peak = nav.cummax().clip(lower=1.0)
|
||||
drawdown = (nav - peak) / peak
|
||||
return float(drawdown.min())
|
||||
|
||||
@@ -119,6 +134,7 @@ def summary(r: pd.Series, rf: float = 0.0) -> Mapping[str, float]:
|
||||
"ann_return": ann_ret,
|
||||
"ann_volatility": ann_vol,
|
||||
"sharpe": sharpe_ratio(r, rf),
|
||||
"sortino": sortino_ratio(r, rf),
|
||||
"max_drawdown": mdd,
|
||||
"calmar": calmar_ratio(r),
|
||||
"win_rate": win_rate(r),
|
||||
@@ -129,6 +145,62 @@ def summary(r: pd.Series, rf: float = 0.0) -> Mapping[str, float]:
|
||||
}
|
||||
|
||||
|
||||
def benchmark_summary(
|
||||
portfolio_returns: pd.Series,
|
||||
benchmark_returns: pd.Series,
|
||||
*,
|
||||
risk_free_daily: float = 0.0,
|
||||
annualization: int = TRADING_DAYS_PER_YEAR,
|
||||
) -> Mapping[str, float]:
|
||||
"""计算成本后组合相对基准的严格对齐绩效。
|
||||
|
||||
与通用 ``summary`` 不同,本函数拒绝静默清洗或日期 inner join。alpha
|
||||
使用日频回归截距的几何年化;基准方差不足时 alpha/beta 为 NaN,明确
|
||||
表示回归不可估计。
|
||||
"""
|
||||
portfolio, benchmark = _validate_benchmark_inputs(
|
||||
portfolio_returns,
|
||||
benchmark_returns,
|
||||
)
|
||||
if isinstance(annualization, bool) or not isinstance(annualization, int):
|
||||
raise TypeError("annualization must be an integer")
|
||||
if annualization <= 0:
|
||||
raise ValueError("annualization must be positive")
|
||||
if not np.isfinite(risk_free_daily):
|
||||
raise ValueError("risk_free_daily must be finite")
|
||||
|
||||
active = portfolio - benchmark
|
||||
active_std = float(active.std())
|
||||
tracking_error = active_std * float(np.sqrt(annualization))
|
||||
information_ratio = (
|
||||
float(active.mean()) / active_std * float(np.sqrt(annualization))
|
||||
if active_std >= 1e-30
|
||||
else float("nan")
|
||||
)
|
||||
|
||||
adjusted_portfolio = portfolio - risk_free_daily
|
||||
adjusted_benchmark = benchmark - risk_free_daily
|
||||
benchmark_variance = float(adjusted_benchmark.var())
|
||||
if benchmark_variance < 1e-30:
|
||||
beta = float("nan")
|
||||
alpha = float("nan")
|
||||
else:
|
||||
beta = float(adjusted_portfolio.cov(adjusted_benchmark) / benchmark_variance)
|
||||
alpha_daily = float((adjusted_portfolio - beta * adjusted_benchmark).mean())
|
||||
alpha = (
|
||||
float((1.0 + alpha_daily) ** annualization - 1.0)
|
||||
if alpha_daily > -1.0
|
||||
else float("nan")
|
||||
)
|
||||
return {
|
||||
"n_observations": len(portfolio),
|
||||
"tracking_error": tracking_error,
|
||||
"information_ratio": information_ratio,
|
||||
"alpha": alpha,
|
||||
"beta": beta,
|
||||
}
|
||||
|
||||
|
||||
# ── 内部 ──────────────────────────────────────
|
||||
|
||||
|
||||
@@ -137,3 +209,29 @@ def _clean(r: pd.Series) -> pd.Series:
|
||||
if not isinstance(r, pd.Series):
|
||||
raise TypeError(f"expected pd.Series, got {type(r).__name__}")
|
||||
return r.replace([np.inf, -np.inf], np.nan).dropna()
|
||||
|
||||
|
||||
def _validate_benchmark_inputs(
|
||||
portfolio_returns: pd.Series,
|
||||
benchmark_returns: pd.Series,
|
||||
) -> tuple[pd.Series, pd.Series]:
|
||||
if not isinstance(portfolio_returns, pd.Series):
|
||||
raise TypeError("portfolio_returns must be a pandas Series")
|
||||
if not isinstance(benchmark_returns, pd.Series):
|
||||
raise TypeError("benchmark_returns must be a pandas Series")
|
||||
if not portfolio_returns.index.equals(benchmark_returns.index):
|
||||
raise ValueError("portfolio and benchmark returns must use matching indexes")
|
||||
if not portfolio_returns.index.is_unique:
|
||||
raise ValueError("portfolio and benchmark indexes must be unique")
|
||||
if len(portfolio_returns) < 2:
|
||||
raise ValueError("benchmark metrics require at least two observations")
|
||||
|
||||
portfolio = portfolio_returns.astype(float, copy=True)
|
||||
benchmark = benchmark_returns.astype(float, copy=True)
|
||||
if not np.isfinite(portfolio.to_numpy()).all() or not np.isfinite(
|
||||
benchmark.to_numpy()
|
||||
).all():
|
||||
raise ValueError("portfolio and benchmark returns must be finite")
|
||||
if (portfolio < -1.0).any() or (benchmark < -1.0).any():
|
||||
raise ValueError("simple returns cannot be less than -1")
|
||||
return portfolio, benchmark
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
"""因子分数到目标权重的轻量组合构建闭环。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from pandas.api.types import is_numeric_dtype
|
||||
|
||||
__all__ = [
|
||||
"select_top_k",
|
||||
"equal_weight",
|
||||
"scores_to_target_weights",
|
||||
"scores_to_weight_table",
|
||||
]
|
||||
|
||||
|
||||
def _validate_top_k(top_k: int) -> None:
|
||||
if isinstance(top_k, bool) or not isinstance(top_k, int) or top_k <= 0:
|
||||
raise ValueError("top_k must be positive")
|
||||
|
||||
|
||||
def _validate_gross_exposure(gross_exposure: float) -> None:
|
||||
if not np.isfinite(gross_exposure) or gross_exposure < 0:
|
||||
raise ValueError("gross_exposure must be finite and non-negative")
|
||||
|
||||
|
||||
def _validate_score_series(scores: pd.Series) -> None:
|
||||
if not isinstance(scores, pd.Series):
|
||||
raise TypeError(f"scores must be a pandas Series, got {type(scores).__name__}")
|
||||
if not scores.index.is_unique:
|
||||
raise ValueError("scores must contain unique asset labels")
|
||||
if not is_numeric_dtype(scores.dtype):
|
||||
raise TypeError("scores must contain numeric values")
|
||||
|
||||
|
||||
def select_top_k(scores: pd.Series, top_k: int, *, largest: bool = True) -> pd.Index:
|
||||
"""稳定选择最高或最低的 K 个有效因子分数。"""
|
||||
_validate_top_k(top_k)
|
||||
_validate_score_series(scores)
|
||||
valid_scores = scores.dropna()
|
||||
ordered = valid_scores.sort_values(ascending=not largest, kind="mergesort")
|
||||
return ordered.iloc[:top_k].index.copy()
|
||||
|
||||
|
||||
def equal_weight(assets: pd.Index, *, gross_exposure: float = 1.0) -> pd.Series:
|
||||
"""在已选资产间等权分配指定总敞口。"""
|
||||
_validate_gross_exposure(gross_exposure)
|
||||
if not assets.is_unique:
|
||||
raise ValueError("assets must contain unique asset labels")
|
||||
if assets.empty:
|
||||
return pd.Series(index=assets.copy(), dtype=float, name="weight")
|
||||
weight = gross_exposure / len(assets)
|
||||
return pd.Series(weight, index=assets.copy(), dtype=float, name="weight")
|
||||
|
||||
|
||||
def scores_to_target_weights(
|
||||
scores: pd.Series,
|
||||
top_k: int,
|
||||
*,
|
||||
gross_exposure: float = 1.0,
|
||||
largest: bool = True,
|
||||
) -> pd.Series:
|
||||
"""把单期因子分数转换为完整股票池目标权重。"""
|
||||
_validate_score_series(scores)
|
||||
selected = select_top_k(scores, top_k, largest=largest)
|
||||
selected_weights = equal_weight(selected, gross_exposure=gross_exposure)
|
||||
result = pd.Series(0.0, index=scores.index.copy(), dtype=float, name="weight")
|
||||
result.loc[selected_weights.index] = selected_weights
|
||||
return result
|
||||
|
||||
|
||||
def scores_to_weight_table(
|
||||
scores: pd.DataFrame,
|
||||
top_k: int,
|
||||
*,
|
||||
gross_exposure: float = 1.0,
|
||||
largest: bool = True,
|
||||
) -> pd.DataFrame:
|
||||
"""逐调仓日独立构建目标权重表,避免使用未来分数。"""
|
||||
if not isinstance(scores, pd.DataFrame):
|
||||
raise TypeError(f"scores must be a pandas DataFrame, got {type(scores).__name__}")
|
||||
_validate_top_k(top_k)
|
||||
_validate_gross_exposure(gross_exposure)
|
||||
if not scores.index.is_unique:
|
||||
raise ValueError("scores must contain unique rebalance dates")
|
||||
if not scores.index.is_monotonic_increasing:
|
||||
raise ValueError("scores rebalance dates must be in chronological order")
|
||||
if not scores.columns.is_unique:
|
||||
raise ValueError("scores must contain unique asset labels")
|
||||
if not all(is_numeric_dtype(dtype) for dtype in scores.dtypes):
|
||||
raise TypeError("scores must contain numeric values")
|
||||
if scores.empty:
|
||||
return pd.DataFrame(index=scores.index.copy(), columns=scores.columns.copy(), dtype=float)
|
||||
|
||||
rows = [
|
||||
scores_to_target_weights(
|
||||
row,
|
||||
top_k,
|
||||
gross_exposure=gross_exposure,
|
||||
largest=largest,
|
||||
).to_numpy()
|
||||
for _, row in scores.iterrows()
|
||||
]
|
||||
return pd.DataFrame(rows, index=scores.index.copy(), columns=scores.columns.copy(), dtype=float)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,401 @@
|
||||
"""可信研究链路:因子分数经交易日历滞后后进入执行与日频 Ledger。
|
||||
|
||||
本模块只编排现有组合构建与执行组件,不连接账户、券商或实盘订单。
|
||||
时间契约借鉴 Qlib 的 prediction/trade time 分离与 Backtrader 的 next-bar
|
||||
执行语义:signal_date 上形成的目标权重,默认最早在下一交易时点执行。
|
||||
完整回测链路进一步分离 execution price 与日末 valuation price,非调仓日也
|
||||
持续盯市,并从真实成交后持仓派生日收益和绩效。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from pandas.api.types import is_numeric_dtype
|
||||
|
||||
from quant_engine.attribution import DailyReturnAttribution, compute_daily_return_attribution
|
||||
from quant_engine.execution import (
|
||||
ExecutionConfig,
|
||||
ExecutionSimulationResult,
|
||||
simulate_daily_ledger_with_audit,
|
||||
simulate_multi_day_with_audit,
|
||||
)
|
||||
from quant_engine.metrics import benchmark_summary, summary as metrics_summary
|
||||
from quant_engine.portfolio_construction import scores_to_weight_table
|
||||
|
||||
__all__ = [
|
||||
"TargetWeightSchedule",
|
||||
"FactorExecutionResult",
|
||||
"FactorBacktestResult",
|
||||
"schedule_target_weights",
|
||||
"run_factor_execution_research",
|
||||
"run_factor_backtest_research",
|
||||
]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, eq=False)
|
||||
class TargetWeightSchedule:
|
||||
"""保留决策时间和执行时间的目标权重调度快照。"""
|
||||
|
||||
decision_weights: pd.DataFrame
|
||||
signal_to_execution: pd.Series
|
||||
execution_weights: pd.DataFrame
|
||||
lag_sessions: int
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, eq=False)
|
||||
class FactorExecutionResult:
|
||||
"""因子到执行审计的一次可复现研究结果。"""
|
||||
|
||||
factor_scores: pd.DataFrame
|
||||
execution_prices: pd.DataFrame
|
||||
schedule: TargetWeightSchedule
|
||||
execution_price_field: str
|
||||
execution: ExecutionSimulationResult
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, eq=False)
|
||||
class FactorBacktestResult:
|
||||
"""因子、成交后日频 Ledger 与绩效的一次可复现快照。"""
|
||||
|
||||
factor_scores: pd.DataFrame
|
||||
execution_prices: pd.DataFrame
|
||||
valuation_prices: pd.DataFrame
|
||||
schedule: TargetWeightSchedule
|
||||
execution_price_field: str
|
||||
valuation_price_field: str
|
||||
execution: ExecutionSimulationResult
|
||||
|
||||
@property
|
||||
def nav(self) -> pd.Series:
|
||||
"""返回以初始资金归一化为 1 的日频 NAV。"""
|
||||
return pd.Series(
|
||||
self.execution.normalized_nav_series.to_numpy(copy=True),
|
||||
index=self.valuation_prices.index.copy(),
|
||||
name="nav",
|
||||
)
|
||||
|
||||
@property
|
||||
def returns(self) -> pd.Series:
|
||||
"""返回包含首日成本影响的日频收益。"""
|
||||
return pd.Series(
|
||||
self.execution.daily_returns.to_numpy(copy=True),
|
||||
index=self.valuation_prices.index.copy(),
|
||||
name="returns",
|
||||
)
|
||||
|
||||
@property
|
||||
def position_weights(self) -> pd.DataFrame:
|
||||
"""按日末实际股数、收盘估值和账本 NAV 投影资产权重。"""
|
||||
weights = pd.DataFrame(
|
||||
0.0,
|
||||
index=self.valuation_prices.index.copy(),
|
||||
columns=self.valuation_prices.columns.copy(),
|
||||
)
|
||||
for date, position in zip(
|
||||
self.valuation_prices.index,
|
||||
self.execution.positions,
|
||||
strict=True,
|
||||
):
|
||||
if position.portfolio_value <= 0:
|
||||
raise ValueError(f"portfolio value must be positive on {date}")
|
||||
for asset, shares in position.holdings.items():
|
||||
weights.at[date, asset] = (
|
||||
shares * float(self.valuation_prices.at[date, asset])
|
||||
/ position.portfolio_value
|
||||
)
|
||||
return weights
|
||||
|
||||
@property
|
||||
def cash_weights(self) -> pd.Series:
|
||||
"""返回与实际资产权重使用同一日末 NAV 分母的现金权重。"""
|
||||
values = []
|
||||
for date, position in zip(
|
||||
self.valuation_prices.index,
|
||||
self.execution.positions,
|
||||
strict=True,
|
||||
):
|
||||
if position.portfolio_value <= 0:
|
||||
raise ValueError(f"portfolio value must be positive on {date}")
|
||||
values.append(position.cash / position.portfolio_value)
|
||||
return pd.Series(
|
||||
values,
|
||||
index=self.valuation_prices.index.copy(),
|
||||
dtype=float,
|
||||
name="cash_weight",
|
||||
)
|
||||
|
||||
def stats(self, rf: float = 0.0) -> Mapping[str, float]:
|
||||
"""复用标准绩效口径计算指标。"""
|
||||
return metrics_summary(self.returns, rf)
|
||||
|
||||
def return_attribution(self) -> DailyReturnAttribution:
|
||||
"""从实际成交后持仓与账本生成逐日净收益归因。"""
|
||||
return compute_daily_return_attribution(
|
||||
self.execution,
|
||||
self.execution_prices,
|
||||
self.valuation_prices,
|
||||
)
|
||||
|
||||
def benchmark_stats(self, benchmark_returns: pd.Series) -> Mapping[str, float]:
|
||||
"""计算成本后日收益相对同日基准的 TE、IR、alpha 与 beta。"""
|
||||
return benchmark_summary(self.returns, benchmark_returns)
|
||||
|
||||
|
||||
def _validate_datetime_index(index: pd.Index, name: str) -> pd.DatetimeIndex:
|
||||
if not isinstance(index, pd.DatetimeIndex):
|
||||
raise TypeError(f"{name} must use a DatetimeIndex")
|
||||
if not index.is_unique:
|
||||
raise ValueError(f"{name} must contain unique sessions")
|
||||
if not index.is_monotonic_increasing:
|
||||
raise ValueError(f"{name} must be in chronological order")
|
||||
return index
|
||||
|
||||
|
||||
def _validate_decision_weights(decision_weights: pd.DataFrame) -> None:
|
||||
if not isinstance(decision_weights, pd.DataFrame):
|
||||
raise TypeError(
|
||||
f"decision_weights must be a pandas DataFrame, got {type(decision_weights).__name__}"
|
||||
)
|
||||
_validate_datetime_index(decision_weights.index, "decision_weights index")
|
||||
if not decision_weights.columns.is_unique:
|
||||
raise ValueError("decision_weights must contain unique asset labels")
|
||||
if not all(is_numeric_dtype(dtype) for dtype in decision_weights.dtypes):
|
||||
raise TypeError("decision_weights must contain numeric values")
|
||||
values = decision_weights.to_numpy(dtype=float)
|
||||
if not np.isfinite(values).all() or (values < 0).any():
|
||||
raise ValueError("decision_weights must be finite and non-negative")
|
||||
if (decision_weights.sum(axis=1) > 1.0 + 1e-12).any():
|
||||
raise ValueError("decision_weights rows must sum to at most 1.0")
|
||||
|
||||
|
||||
def _validate_execution_prices(execution_prices: pd.DataFrame) -> pd.DatetimeIndex:
|
||||
if not isinstance(execution_prices, pd.DataFrame):
|
||||
raise TypeError(
|
||||
f"execution_prices must be a pandas DataFrame, got {type(execution_prices).__name__}"
|
||||
)
|
||||
calendar = _validate_datetime_index(execution_prices.index, "execution_prices index")
|
||||
if not execution_prices.columns.is_unique:
|
||||
raise ValueError("execution_prices must contain unique asset labels")
|
||||
if not all(is_numeric_dtype(dtype) for dtype in execution_prices.dtypes):
|
||||
raise TypeError("execution_prices must contain numeric values")
|
||||
return calendar
|
||||
|
||||
|
||||
def schedule_target_weights(
|
||||
decision_weights: pd.DataFrame,
|
||||
trading_calendar: pd.DatetimeIndex,
|
||||
*,
|
||||
lag_sessions: int = 1,
|
||||
) -> TargetWeightSchedule:
|
||||
"""将信号日目标权重映射到后续真实交易日,不做整数行盲移位。
|
||||
|
||||
所有信号日必须属于 ``trading_calendar``,且日历必须包含每个信号对应的
|
||||
未来执行日;无法执行的末尾信号会显式失败,避免被静默丢弃。
|
||||
"""
|
||||
_validate_decision_weights(decision_weights)
|
||||
calendar = _validate_datetime_index(trading_calendar, "trading_calendar")
|
||||
if isinstance(lag_sessions, bool) or not isinstance(lag_sessions, int) or lag_sessions <= 0:
|
||||
raise ValueError("lag_sessions must be a positive integer")
|
||||
|
||||
decision_snapshot = decision_weights.copy(deep=True)
|
||||
if decision_snapshot.empty:
|
||||
execution_weights = decision_snapshot.copy(deep=True)
|
||||
execution_weights.index = pd.DatetimeIndex([], name="execution_date")
|
||||
mapping = pd.Series(
|
||||
calendar[:0],
|
||||
index=decision_snapshot.index.copy(),
|
||||
name="execution_date",
|
||||
)
|
||||
return TargetWeightSchedule(
|
||||
decision_weights=decision_snapshot,
|
||||
signal_to_execution=mapping,
|
||||
execution_weights=execution_weights,
|
||||
lag_sessions=lag_sessions,
|
||||
)
|
||||
|
||||
signal_positions = calendar.get_indexer(decision_snapshot.index)
|
||||
if (signal_positions < 0).any():
|
||||
missing = decision_snapshot.index[signal_positions < 0]
|
||||
raise ValueError(
|
||||
"signal dates must be trading sessions; missing="
|
||||
+ ", ".join(str(date) for date in missing)
|
||||
)
|
||||
|
||||
execution_positions = signal_positions + lag_sessions
|
||||
if (execution_positions >= len(calendar)).any():
|
||||
unavailable = decision_snapshot.index[execution_positions >= len(calendar)]
|
||||
raise ValueError(
|
||||
"trading_calendar lacks a future execution session for signal dates: "
|
||||
+ ", ".join(str(date) for date in unavailable)
|
||||
)
|
||||
|
||||
execution_dates = calendar.take(execution_positions)
|
||||
signal_to_execution = pd.Series(
|
||||
execution_dates,
|
||||
index=decision_snapshot.index.copy(),
|
||||
name="execution_date",
|
||||
)
|
||||
execution_weights = decision_snapshot.copy(deep=True)
|
||||
execution_weights.index = pd.DatetimeIndex(execution_dates, name="execution_date")
|
||||
return TargetWeightSchedule(
|
||||
decision_weights=decision_snapshot,
|
||||
signal_to_execution=signal_to_execution,
|
||||
execution_weights=execution_weights,
|
||||
lag_sessions=lag_sessions,
|
||||
)
|
||||
|
||||
|
||||
def run_factor_execution_research(
|
||||
factor_scores: pd.DataFrame,
|
||||
execution_prices: pd.DataFrame,
|
||||
*,
|
||||
top_k: int,
|
||||
execution_price_field: str,
|
||||
lag_sessions: int = 1,
|
||||
gross_exposure: float = 1.0,
|
||||
largest: bool = True,
|
||||
initial_cash: float = 1_000_000.0,
|
||||
config: ExecutionConfig | None = None,
|
||||
) -> FactorExecutionResult:
|
||||
"""运行因子分数 → 目标权重 → 下一交易时点 → 执行审计链路。
|
||||
|
||||
``execution_prices`` 必须代表实际拟执行时点的价格矩阵,例如日频研究中
|
||||
signal 日收盘生成分数后使用下一交易日 ``open``。价格字段名称被保存在
|
||||
结果元数据中,但函数不会猜测或重写价格语义。
|
||||
"""
|
||||
price_field = execution_price_field.strip()
|
||||
if not price_field:
|
||||
raise ValueError("execution_price_field must be non-empty")
|
||||
calendar = _validate_execution_prices(execution_prices)
|
||||
|
||||
factor_snapshot = factor_scores.copy(deep=True)
|
||||
decision_weights = scores_to_weight_table(
|
||||
factor_snapshot,
|
||||
top_k,
|
||||
gross_exposure=gross_exposure,
|
||||
largest=largest,
|
||||
)
|
||||
schedule = schedule_target_weights(
|
||||
decision_weights,
|
||||
calendar,
|
||||
lag_sessions=lag_sessions,
|
||||
)
|
||||
price_snapshot = execution_prices.copy(deep=True)
|
||||
|
||||
target_history: list[tuple[str, dict[str, float]]] = []
|
||||
price_history: list[tuple[str, dict[str, float]]] = []
|
||||
for execution_date, weights in schedule.execution_weights.iterrows():
|
||||
date_label = str(pd.Timestamp(execution_date))
|
||||
target_history.append(
|
||||
(date_label, {asset: float(weight) for asset, weight in weights.items()})
|
||||
)
|
||||
prices = price_snapshot.loc[execution_date]
|
||||
price_history.append(
|
||||
(date_label, {asset: float(price) for asset, price in prices.items()})
|
||||
)
|
||||
|
||||
execution = simulate_multi_day_with_audit(
|
||||
target_history,
|
||||
price_history,
|
||||
initial_cash,
|
||||
config,
|
||||
)
|
||||
return FactorExecutionResult(
|
||||
factor_scores=factor_snapshot,
|
||||
execution_prices=price_snapshot,
|
||||
schedule=schedule,
|
||||
execution_price_field=price_field,
|
||||
execution=execution,
|
||||
)
|
||||
|
||||
|
||||
def run_factor_backtest_research(
|
||||
factor_scores: pd.DataFrame,
|
||||
execution_prices: pd.DataFrame,
|
||||
valuation_prices: pd.DataFrame,
|
||||
*,
|
||||
top_k: int,
|
||||
execution_price_field: str,
|
||||
valuation_price_field: str,
|
||||
lag_sessions: int = 1,
|
||||
gross_exposure: float = 1.0,
|
||||
largest: bool = True,
|
||||
initial_cash: float = 1_000_000.0,
|
||||
config: ExecutionConfig | None = None,
|
||||
) -> FactorBacktestResult:
|
||||
"""运行 PIT 因子到成交后日频 Ledger、收益与绩效的可信研究链路。"""
|
||||
execution_field = execution_price_field.strip()
|
||||
valuation_field = valuation_price_field.strip()
|
||||
if not execution_field:
|
||||
raise ValueError("execution_price_field must be non-empty")
|
||||
if not valuation_field:
|
||||
raise ValueError("valuation_price_field must be non-empty")
|
||||
|
||||
execution_calendar = _validate_execution_prices(execution_prices)
|
||||
valuation_calendar = _validate_execution_prices(valuation_prices)
|
||||
if not execution_calendar.equals(valuation_calendar):
|
||||
raise ValueError("execution and valuation prices must use matching trading calendars")
|
||||
if not execution_prices.columns.equals(valuation_prices.columns):
|
||||
raise ValueError("execution and valuation prices must use matching asset labels")
|
||||
|
||||
factor_snapshot = factor_scores.copy(deep=True)
|
||||
execution_snapshot = execution_prices.copy(deep=True)
|
||||
valuation_snapshot = valuation_prices.copy(deep=True)
|
||||
decision_weights = scores_to_weight_table(
|
||||
factor_snapshot,
|
||||
top_k,
|
||||
gross_exposure=gross_exposure,
|
||||
largest=largest,
|
||||
)
|
||||
schedule = schedule_target_weights(
|
||||
decision_weights,
|
||||
execution_calendar,
|
||||
lag_sessions=lag_sessions,
|
||||
)
|
||||
if decision_weights.empty:
|
||||
execution_window = execution_snapshot.iloc[:0].copy()
|
||||
valuation_window = valuation_snapshot.iloc[:0].copy()
|
||||
else:
|
||||
research_start = decision_weights.index[0]
|
||||
execution_window = execution_snapshot.loc[research_start:].copy()
|
||||
valuation_window = valuation_snapshot.loc[research_start:].copy()
|
||||
|
||||
target_history: list[tuple[str, dict[str, float]]] = []
|
||||
execution_history: list[tuple[str, dict[str, float]]] = []
|
||||
for execution_date, weights in schedule.execution_weights.iterrows():
|
||||
date_label = str(pd.Timestamp(execution_date))
|
||||
target_history.append(
|
||||
(date_label, {asset: float(weight) for asset, weight in weights.items()})
|
||||
)
|
||||
prices = execution_window.loc[execution_date]
|
||||
execution_history.append(
|
||||
(date_label, {asset: float(price) for asset, price in prices.items()})
|
||||
)
|
||||
|
||||
valuation_history = [
|
||||
(
|
||||
str(pd.Timestamp(valuation_date)),
|
||||
{asset: float(price) for asset, price in prices.items()},
|
||||
)
|
||||
for valuation_date, prices in valuation_window.iterrows()
|
||||
]
|
||||
execution = simulate_daily_ledger_with_audit(
|
||||
target_history,
|
||||
execution_history,
|
||||
valuation_history,
|
||||
initial_cash,
|
||||
config,
|
||||
)
|
||||
return FactorBacktestResult(
|
||||
factor_scores=factor_snapshot,
|
||||
execution_prices=execution_window,
|
||||
valuation_prices=valuation_window,
|
||||
schedule=schedule,
|
||||
execution_price_field=execution_field,
|
||||
valuation_price_field=valuation_field,
|
||||
execution=execution,
|
||||
)
|
||||
+372
-10
@@ -5,10 +5,310 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
from numpy.typing import NDArray
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from datetime import date
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from numpy.typing import NDArray
|
||||
|
||||
__all__ = [
|
||||
"ComponentRiskResult",
|
||||
"CovarianceSnapshot",
|
||||
"component_var",
|
||||
"estimate_covariance_snapshot",
|
||||
"labeled_component_risk",
|
||||
"marginal_risk_contribution",
|
||||
"risk_contribution",
|
||||
]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, init=False, eq=False)
|
||||
class CovarianceSnapshot:
|
||||
"""Immutable-by-interface covariance input with explicit time semantics."""
|
||||
|
||||
snapshot_id: str
|
||||
as_of_date: date
|
||||
_covariance: pd.DataFrame
|
||||
return_frequency: str
|
||||
periods_per_year: int
|
||||
method: str
|
||||
window_start_date: date | None
|
||||
window_end_date: date | None
|
||||
observations: int | None
|
||||
lookback_sessions: int | None
|
||||
missing_policy: str
|
||||
data_snapshot_id: str
|
||||
input_sha256: str
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
snapshot_id: str,
|
||||
as_of_date: str | date | pd.Timestamp,
|
||||
covariance: pd.DataFrame,
|
||||
return_frequency: str,
|
||||
periods_per_year: int,
|
||||
method: str = "provided",
|
||||
window_start_date: str | date | pd.Timestamp | None = None,
|
||||
window_end_date: str | date | pd.Timestamp | None = None,
|
||||
observations: int | None = None,
|
||||
lookback_sessions: int | None = None,
|
||||
missing_policy: str = "provided",
|
||||
data_snapshot_id: str = "",
|
||||
input_sha256: str = "",
|
||||
) -> None:
|
||||
if not isinstance(snapshot_id, str) or not snapshot_id.strip():
|
||||
raise ValueError("snapshot_id must be non-empty")
|
||||
if not isinstance(return_frequency, str) or not return_frequency.strip():
|
||||
raise ValueError("return_frequency must be non-empty")
|
||||
if isinstance(periods_per_year, bool) or not isinstance(periods_per_year, int):
|
||||
raise TypeError("periods_per_year must be an integer")
|
||||
if periods_per_year <= 0:
|
||||
raise ValueError("periods_per_year must be positive")
|
||||
if not isinstance(covariance, pd.DataFrame):
|
||||
raise TypeError("covariance must be a pandas DataFrame")
|
||||
if covariance.empty:
|
||||
raise ValueError("covariance must contain at least one asset")
|
||||
if not isinstance(method, str) or not method.strip():
|
||||
raise ValueError("method must be non-empty")
|
||||
if not isinstance(missing_policy, str) or not missing_policy.strip():
|
||||
raise ValueError("missing_policy must be non-empty")
|
||||
for value, name in (
|
||||
(observations, "observations"),
|
||||
(lookback_sessions, "lookback_sessions"),
|
||||
):
|
||||
if value is not None and (
|
||||
isinstance(value, bool) or not isinstance(value, int) or value <= 0
|
||||
):
|
||||
raise ValueError(f"{name} must be a positive integer when provided")
|
||||
if input_sha256 and (
|
||||
len(input_sha256) != 64
|
||||
or any(character not in "0123456789abcdef" for character in input_sha256)
|
||||
):
|
||||
raise ValueError("input_sha256 must be a lowercase SHA-256 digest")
|
||||
|
||||
normalized_as_of = _normalized_date(as_of_date, "as_of_date")
|
||||
normalized_window_start = (
|
||||
None
|
||||
if window_start_date is None
|
||||
else _normalized_date(window_start_date, "window_start_date")
|
||||
)
|
||||
normalized_window_end = (
|
||||
None
|
||||
if window_end_date is None
|
||||
else _normalized_date(window_end_date, "window_end_date")
|
||||
)
|
||||
if (normalized_window_start is None) != (normalized_window_end is None):
|
||||
raise ValueError("window_start_date and window_end_date must be provided together")
|
||||
if (
|
||||
normalized_window_start is not None
|
||||
and normalized_window_end is not None
|
||||
and normalized_window_start > normalized_window_end
|
||||
):
|
||||
raise ValueError("window_start_date must not be after window_end_date")
|
||||
if normalized_window_end is not None and normalized_window_end > normalized_as_of:
|
||||
raise ValueError("window_end_date must not be after as_of_date")
|
||||
|
||||
object.__setattr__(self, "snapshot_id", snapshot_id.strip())
|
||||
object.__setattr__(self, "as_of_date", normalized_as_of)
|
||||
object.__setattr__(self, "_covariance", covariance.copy(deep=True))
|
||||
object.__setattr__(self, "return_frequency", return_frequency.strip())
|
||||
object.__setattr__(self, "periods_per_year", periods_per_year)
|
||||
object.__setattr__(self, "method", method.strip())
|
||||
object.__setattr__(self, "window_start_date", normalized_window_start)
|
||||
object.__setattr__(self, "window_end_date", normalized_window_end)
|
||||
object.__setattr__(self, "observations", observations)
|
||||
object.__setattr__(self, "lookback_sessions", lookback_sessions)
|
||||
object.__setattr__(self, "missing_policy", missing_policy.strip())
|
||||
object.__setattr__(self, "data_snapshot_id", data_snapshot_id.strip())
|
||||
object.__setattr__(self, "input_sha256", input_sha256)
|
||||
|
||||
@property
|
||||
def covariance(self) -> pd.DataFrame:
|
||||
"""Return an isolated copy so callers cannot mutate the snapshot."""
|
||||
return self._covariance.copy(deep=True)
|
||||
|
||||
|
||||
def _normalized_date(value: object, name: str) -> date:
|
||||
try:
|
||||
timestamp = pd.Timestamp(value)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError(f"{name} must be a valid date") from error
|
||||
if pd.isna(timestamp):
|
||||
raise ValueError(f"{name} must be a valid date")
|
||||
return date(int(timestamp.year), int(timestamp.month), int(timestamp.day))
|
||||
|
||||
|
||||
def _positive_integer(value: int, name: str, *, minimum: int = 1) -> int:
|
||||
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
|
||||
raise ValueError(f"{name} must be an integer of at least {minimum}")
|
||||
return value
|
||||
|
||||
|
||||
def _input_fingerprint(window: pd.DataFrame, session_dates: list[date]) -> str:
|
||||
values = window.to_numpy(dtype=float, copy=True)
|
||||
missing = np.isnan(values)
|
||||
normalized = np.where(missing, 0.0, values).astype("<f8", copy=False)
|
||||
metadata = {
|
||||
"assets": [str(asset) for asset in window.columns],
|
||||
"sessions": [session.isoformat() for session in session_dates],
|
||||
"shape": list(values.shape),
|
||||
}
|
||||
digest = hashlib.sha256(
|
||||
json.dumps(metadata, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
||||
)
|
||||
digest.update(missing.astype(np.uint8, copy=False).tobytes(order="C"))
|
||||
digest.update(normalized.tobytes(order="C"))
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def estimate_covariance_snapshot(
|
||||
asset_returns: pd.DataFrame,
|
||||
*,
|
||||
as_of_date: str | date | pd.Timestamp,
|
||||
lookback_sessions: int,
|
||||
min_observations: int,
|
||||
data_snapshot_id: str,
|
||||
return_frequency: str = "1d",
|
||||
periods_per_year: int = 252,
|
||||
) -> CovarianceSnapshot:
|
||||
"""Estimate a deterministic per-period sample covariance without look-ahead.
|
||||
|
||||
The selected lookback window is truncated at ``as_of_date`` before any
|
||||
calculation. Rows containing a missing asset return are removed as complete
|
||||
cases, preventing pairwise sample sets from producing an ambiguous matrix.
|
||||
"""
|
||||
if not isinstance(asset_returns, pd.DataFrame):
|
||||
raise TypeError("asset_returns must be a pandas DataFrame")
|
||||
if asset_returns.empty or asset_returns.shape[1] == 0:
|
||||
raise ValueError("asset_returns must contain observations and assets")
|
||||
if not isinstance(asset_returns.index, pd.DatetimeIndex):
|
||||
raise TypeError("asset_returns index must be a DatetimeIndex")
|
||||
if not asset_returns.index.is_unique or not asset_returns.index.is_monotonic_increasing:
|
||||
raise ValueError("asset_returns index must be unique and strictly increasing")
|
||||
if not asset_returns.columns.is_unique:
|
||||
raise ValueError("asset_returns must contain unique asset labels")
|
||||
if any(not isinstance(asset, str) or not asset.strip() for asset in asset_returns.columns):
|
||||
raise ValueError("asset_returns asset labels must be non-empty strings")
|
||||
|
||||
lookback = _positive_integer(lookback_sessions, "lookback_sessions")
|
||||
minimum = _positive_integer(min_observations, "min_observations", minimum=2)
|
||||
if minimum > lookback:
|
||||
raise ValueError("min_observations must not exceed lookback_sessions")
|
||||
normalized_data_snapshot_id = data_snapshot_id.strip()
|
||||
if not normalized_data_snapshot_id:
|
||||
raise ValueError("data_snapshot_id must be non-empty")
|
||||
normalized_as_of = _normalized_date(as_of_date, "as_of_date")
|
||||
|
||||
returns = asset_returns.astype(float, copy=True)
|
||||
values = returns.to_numpy()
|
||||
if np.isinf(values).any():
|
||||
raise ValueError("asset_returns must not contain infinite values")
|
||||
session_dates = [
|
||||
_normalized_date(index_value, "asset_returns index") for index_value in returns.index
|
||||
]
|
||||
if len(set(session_dates)) != len(session_dates):
|
||||
raise ValueError("asset_returns must contain at most one observation per session date")
|
||||
historical_mask = [session <= normalized_as_of for session in session_dates]
|
||||
window = returns.loc[historical_mask].tail(lookback)
|
||||
if window.empty:
|
||||
raise ValueError("asset_returns contain no observations on or before as_of_date")
|
||||
window_dates = [
|
||||
_normalized_date(index_value, "asset_returns index") for index_value in window.index
|
||||
]
|
||||
complete = window.dropna(axis=0, how="any")
|
||||
if len(complete) < minimum:
|
||||
raise ValueError(
|
||||
f"complete observations must be at least {minimum}; received {len(complete)}"
|
||||
)
|
||||
|
||||
covariance = complete.cov(ddof=1)
|
||||
covariance_values = covariance.to_numpy()
|
||||
if not np.isfinite(covariance_values).all():
|
||||
raise ValueError("sample covariance must be finite")
|
||||
input_sha256 = _input_fingerprint(window, window_dates)
|
||||
identity = {
|
||||
"as_of_date": normalized_as_of.isoformat(),
|
||||
"assets": list(returns.columns),
|
||||
"data_snapshot_id": normalized_data_snapshot_id,
|
||||
"estimator": "sample-cov-v1",
|
||||
"input_sha256": input_sha256,
|
||||
"lookback_sessions": lookback,
|
||||
"min_observations": minimum,
|
||||
"missing_policy": "complete_case",
|
||||
"observations": len(complete),
|
||||
"periods_per_year": periods_per_year,
|
||||
"return_frequency": return_frequency,
|
||||
"window_end_date": window_dates[-1].isoformat(),
|
||||
"window_start_date": window_dates[0].isoformat(),
|
||||
}
|
||||
identity_bytes = json.dumps(
|
||||
identity,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
).encode("utf-8")
|
||||
digest = hashlib.sha256(identity_bytes)
|
||||
digest.update(covariance_values.astype("<f8", copy=False).tobytes(order="C"))
|
||||
snapshot_id = f"sample-cov-v1:{digest.hexdigest()}"
|
||||
return CovarianceSnapshot(
|
||||
snapshot_id=snapshot_id,
|
||||
as_of_date=normalized_as_of,
|
||||
covariance=covariance,
|
||||
return_frequency=return_frequency,
|
||||
periods_per_year=periods_per_year,
|
||||
method="sample",
|
||||
window_start_date=window_dates[0],
|
||||
window_end_date=window_dates[-1],
|
||||
observations=len(complete),
|
||||
lookback_sessions=lookback,
|
||||
missing_policy="complete_case",
|
||||
data_snapshot_id=normalized_data_snapshot_id,
|
||||
input_sha256=input_sha256,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True, eq=False)
|
||||
class ComponentRiskResult:
|
||||
"""Label-preserving Euler decomposition of portfolio volatility."""
|
||||
|
||||
portfolio_volatility: float
|
||||
marginal: pd.Series
|
||||
component: pd.Series
|
||||
percentage: pd.Series
|
||||
|
||||
def grouped_component(self, groups: pd.Series) -> pd.Series:
|
||||
"""Aggregate asset component risk by an explicitly aligned label series."""
|
||||
if not isinstance(groups, pd.Series):
|
||||
raise TypeError("groups must be a pandas Series")
|
||||
if not groups.index.is_unique:
|
||||
raise ValueError("groups must contain unique asset labels")
|
||||
if not self.component.index.difference(groups.index).empty or not groups.index.difference(
|
||||
self.component.index
|
||||
).empty:
|
||||
raise ValueError("groups and component risk must use the same asset labels")
|
||||
aligned = groups.reindex(self.component.index)
|
||||
if aligned.isna().any():
|
||||
raise ValueError("groups must contain a non-missing label for every asset")
|
||||
grouped = self.component.groupby(aligned, sort=True).sum()
|
||||
grouped.name = "component_risk"
|
||||
return grouped
|
||||
|
||||
|
||||
def _validate_inputs(weights: NDArray[Any], cov: NDArray[Any]) -> tuple[NDArray[Any], NDArray[Any]]:
|
||||
"""Normalize a portfolio vector and its covariance matrix."""
|
||||
w = np.asarray(weights, dtype=float).ravel()
|
||||
covariance = np.asarray(cov, dtype=float)
|
||||
k = w.size
|
||||
if k == 0:
|
||||
raise ValueError("weights must contain at least one asset")
|
||||
if covariance.shape != (k, k):
|
||||
raise ValueError(f"cov shape {covariance.shape} does not match weights length {k}")
|
||||
return w, covariance
|
||||
|
||||
|
||||
def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
|
||||
"""风险贡献率 (RC_i): w_i * (Σw)_i / w'Σw。
|
||||
@@ -25,11 +325,8 @@ def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
|
||||
Returns:
|
||||
RC: 风险贡献向量 (k,), Σ=1
|
||||
"""
|
||||
w = np.asarray(weights, dtype=float).ravel()
|
||||
cov = np.asarray(cov, dtype=float)
|
||||
w, cov = _validate_inputs(weights, cov)
|
||||
k = w.size
|
||||
if cov.shape != (k, k):
|
||||
raise ValueError(f"cov 形状 {cov.shape} 与 weights 长度 {k} 不匹配")
|
||||
|
||||
port_var = float(w @ cov @ w)
|
||||
if port_var <= 0:
|
||||
@@ -41,13 +338,78 @@ def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
|
||||
|
||||
def marginal_risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
|
||||
"""边际风险贡献 (MRC_i): (Σw)_i。"""
|
||||
w = np.asarray(weights, dtype=float).ravel()
|
||||
cov = np.asarray(cov, dtype=float)
|
||||
w, cov = _validate_inputs(weights, cov)
|
||||
return cov @ w # type: ignore[no-any-return]
|
||||
|
||||
|
||||
def component_var(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]:
|
||||
"""成分方差: w_i · (Σw)_i; 与 RC 的关系 RC_i = CV_i / w'Σw。"""
|
||||
w = np.asarray(weights, dtype=float).ravel()
|
||||
cov = np.asarray(cov, dtype=float)
|
||||
w, cov = _validate_inputs(weights, cov)
|
||||
return w * (cov @ w) # type: ignore[no-any-return]
|
||||
|
||||
|
||||
def labeled_component_risk(
|
||||
weights: pd.Series,
|
||||
covariance: pd.DataFrame,
|
||||
) -> ComponentRiskResult:
|
||||
"""Return a label-safe Euler decomposition that sums to portfolio volatility.
|
||||
|
||||
The covariance matrix may use a different asset order, but its row and
|
||||
column label sets must exactly match ``weights``. Invalid or indefinite
|
||||
covariance input is rejected instead of silently producing misleading risk
|
||||
percentages.
|
||||
"""
|
||||
if not isinstance(weights, pd.Series):
|
||||
raise TypeError("weights must be a pandas Series")
|
||||
if not isinstance(covariance, pd.DataFrame):
|
||||
raise TypeError("covariance must be a pandas DataFrame")
|
||||
if weights.empty:
|
||||
raise ValueError("weights must contain at least one asset")
|
||||
if not weights.index.is_unique:
|
||||
raise ValueError("weights must contain unique asset labels")
|
||||
if not covariance.index.is_unique or not covariance.columns.is_unique:
|
||||
raise ValueError("covariance must contain unique asset labels")
|
||||
if not weights.index.difference(covariance.index).empty or not covariance.index.difference(
|
||||
weights.index
|
||||
).empty:
|
||||
raise ValueError("weights and covariance must use the same asset labels")
|
||||
if not weights.index.difference(covariance.columns).empty or not covariance.columns.difference(
|
||||
weights.index
|
||||
).empty:
|
||||
raise ValueError("weights and covariance must use the same asset labels")
|
||||
|
||||
aligned_weights = weights.astype(float, copy=True)
|
||||
aligned_covariance = covariance.reindex(
|
||||
index=weights.index,
|
||||
columns=weights.index,
|
||||
).astype(float, copy=True)
|
||||
weight_values = aligned_weights.to_numpy()
|
||||
covariance_values = aligned_covariance.to_numpy()
|
||||
if not np.isfinite(weight_values).all():
|
||||
raise ValueError("weights must be finite")
|
||||
if not np.isfinite(covariance_values).all():
|
||||
raise ValueError("covariance must be finite")
|
||||
if not np.allclose(covariance_values, covariance_values.T, rtol=1e-10, atol=1e-12):
|
||||
raise ValueError("covariance must be symmetric")
|
||||
eigenvalues = np.linalg.eigvalsh(covariance_values)
|
||||
scale = max(1.0, float(np.max(np.abs(eigenvalues))))
|
||||
if float(eigenvalues.min()) < -1e-10 * scale:
|
||||
raise ValueError("covariance must be positive semidefinite")
|
||||
|
||||
portfolio_variance = float(weight_values @ covariance_values @ weight_values)
|
||||
if portfolio_variance <= 0 or not np.isfinite(portfolio_variance):
|
||||
raise ValueError("weights and covariance must produce positive portfolio variance")
|
||||
portfolio_volatility = float(np.sqrt(portfolio_variance))
|
||||
marginal_values = covariance_values @ weight_values / portfolio_volatility
|
||||
component_values = weight_values * marginal_values
|
||||
percentage_values = component_values / portfolio_volatility
|
||||
return ComponentRiskResult(
|
||||
portfolio_volatility=portfolio_volatility,
|
||||
marginal=pd.Series(marginal_values, index=weights.index.copy(), name="marginal_risk"),
|
||||
component=pd.Series(component_values, index=weights.index.copy(), name="component_risk"),
|
||||
percentage=pd.Series(
|
||||
percentage_values,
|
||||
index=weights.index.copy(),
|
||||
name="risk_contribution",
|
||||
),
|
||||
)
|
||||
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
{
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"replay_spec_digest": "sha256:20f07fcb526bc38b4ba3d71d6c3b00a8c63ad96cab0fecd869326503ab42d98b",
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:f94733e849433f62f3da1e1ec8999891b93d49c7d657832d64e64bef9e211117",
|
||||
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"table_content_digests": {
|
||||
"run": "sha256:7d947ec93f714641669cbb14bd69dd8cf30387aabb7f3a086918fc2e878ab05c",
|
||||
"signals": "sha256:72dc15064cc45d7c51d2dd4b8c3a6d8c7d4155232d5d70d3c9e7697fab70ce50",
|
||||
"trades": "sha256:2b8b9321e7993941ac486cf50cab5b4c6425b2571a4701f0eb02273ec9a96c53",
|
||||
"positions": "sha256:b456a48fab51742b05084ca6dcaf01c03dfe5b39ad71215ada070e9b1f59f7ea",
|
||||
"nav": "sha256:25649efce860b76410f87dbd36c81085887620086dc0b48b07099918f65d9c78",
|
||||
"performance": "sha256:0856439ea7ec84e38887ccfa0067324f2e9cd293543e4ef683b9c2beb9eb34cf",
|
||||
"attribution": "sha256:ad0f12f668d0a2d9ae5b3636d29989ab4bafcfb95f5de77abeb52a6d9e95d366",
|
||||
"attribution_daily": "sha256:d4459ad650f88871d7b1e40392033037b5818ef92fdf1ee021868929fc1455a5",
|
||||
"risk": "sha256:4816dd4812b5ff2e97e74bf34ca2221bfde675b387a96e277683557f0ad7d975"
|
||||
}
|
||||
}
|
||||
+206
@@ -0,0 +1,206 @@
|
||||
{
|
||||
"dataset_snapshot": {
|
||||
"contract_name": "researchhub.dataset-snapshot",
|
||||
"schema_version": "1.0.0",
|
||||
"snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"descriptor": {
|
||||
"dataset": {
|
||||
"dataset_id": "rhdataset:market:0123456789abcdef0123456789abcdef",
|
||||
"dataset_kind": "market",
|
||||
"record_schema_version": "1.0.0",
|
||||
"dimensions": ["instrument_id", "effective_time"]
|
||||
},
|
||||
"published_at": "2026-01-02T07:05:00Z",
|
||||
"time_semantics": {
|
||||
"effective_time": {
|
||||
"start_inclusive": "2026-01-02T07:00:00Z",
|
||||
"end_inclusive": "2026-01-02T07:00:00Z"
|
||||
},
|
||||
"knowledge_time": {
|
||||
"start_inclusive": "2026-01-02T07:01:00Z",
|
||||
"end_inclusive": "2026-01-02T07:01:00Z"
|
||||
},
|
||||
"pit_cutoff": "2026-01-02T07:01:00Z"
|
||||
},
|
||||
"content": {
|
||||
"digest_algorithm": "sha256",
|
||||
"canonicalization": "RFC8785",
|
||||
"record_order": "canonical-record-byte-order",
|
||||
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"logical_manifest": {
|
||||
"record_count": 2,
|
||||
"chunks": [
|
||||
{
|
||||
"chunk_index": 0,
|
||||
"content_digest": "sha256:44ea11ba64dc2e6fd55c6d8e038c5edc84ee38d6fd46e38b15a1d5409662a020",
|
||||
"record_count": 2
|
||||
}
|
||||
]
|
||||
},
|
||||
"manifest_digest": "sha256:d991bb2f8f6b80525f93c51e0b371213a3ed4649dffb073ed4605bfbd32349bd",
|
||||
"record_count": 2
|
||||
},
|
||||
"lineage": {
|
||||
"publisher": {"id": "researchhub.data", "version": "1.0.0"},
|
||||
"transformation": {
|
||||
"id": "rhtransform:00112233445566778899aabbccddeeff",
|
||||
"version": "1.0.0"
|
||||
},
|
||||
"upstream_snapshot_ids": [],
|
||||
"upstream_content_digests": []
|
||||
},
|
||||
"quality": {
|
||||
"status": "passed",
|
||||
"checks": [
|
||||
{
|
||||
"check_id": "completeness",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:876fc2fcc6414ddc3f824a47f475d34c82a53d2bda5dc72a234a3f3164e8e2ec"
|
||||
},
|
||||
{
|
||||
"check_id": "pit_time_integrity",
|
||||
"status": "passed",
|
||||
"severity": "blocking",
|
||||
"evidence_digest": "sha256:90a6cc46b9f2ab317a1c6d14dc173784e19b5621cd7fe956e8f41338cbdc5944"
|
||||
}
|
||||
]
|
||||
},
|
||||
"qualification": {
|
||||
"status": "qualified",
|
||||
"policy_id": "researchhub.dataset-snapshot.pit",
|
||||
"policy_version": "1.0.0",
|
||||
"evaluated_at": "2026-01-02T07:04:00Z",
|
||||
"evidence_digest": "sha256:e192462f9022f2b477f73cdbe9e6c9f891ebcfdc2b4ed4f8ddd7b1ff107ee6a6"
|
||||
}
|
||||
}
|
||||
},
|
||||
"data_foundation": {
|
||||
"contract_name": "researchhub.data-foundation",
|
||||
"schema_version": "1.0.0",
|
||||
"foundation_id": "rhdfv1:sha256:d848237ab753ee9432ae78ec1f93b6ac45c8072d6694023b7f288203daf9d838",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"pit_cutoff": "2026-01-03T00:00:00Z",
|
||||
"instrument_routes": [
|
||||
{
|
||||
"route_revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"revision_number": 1,
|
||||
"symbol": "600000",
|
||||
"mic": "XSHG",
|
||||
"currency": "CNY",
|
||||
"asset_class": "equity",
|
||||
"instrument_type": "stock",
|
||||
"calendar_id": "rhcalendar:11112222333344445555666677778888",
|
||||
"effective_from": "2020-01-01T00:00:00Z",
|
||||
"knowledge_time": "2026-01-01T07:00:00Z",
|
||||
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
}
|
||||
],
|
||||
"trading_calendar_revisions": [
|
||||
{
|
||||
"calendar_revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
|
||||
"calendar_id": "rhcalendar:11112222333344445555666677778888",
|
||||
"session_date": "2026-01-02",
|
||||
"revision_number": 1,
|
||||
"status": "open",
|
||||
"sessions": [
|
||||
{"opens_at": "2026-01-02T01:30:00Z", "closes_at": "2026-01-02T07:00:00Z"}
|
||||
],
|
||||
"knowledge_time": "2026-01-01T08:00:00Z",
|
||||
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
}
|
||||
],
|
||||
"corporate_action_revisions": [
|
||||
{
|
||||
"action_revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
|
||||
"action_id": "rhaction:99998888777766665555444433332222",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"revision_number": 1,
|
||||
"action_type": "cash_dividend",
|
||||
"status": "confirmed",
|
||||
"effective_time": "2026-01-02T00:00:00Z",
|
||||
"knowledge_time": "2026-01-01T09:00:00Z",
|
||||
"terms_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
|
||||
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
|
||||
}
|
||||
],
|
||||
"standardized_views": [
|
||||
{
|
||||
"view_ref_id": "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c",
|
||||
"view_id": "rhview:abcdef0123456789abcdef0123456789",
|
||||
"view_version": "1.0.0",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"pit_cutoff": "2026-01-03T00:00:00Z",
|
||||
"schema_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
|
||||
"content_digest": "sha256:123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef0",
|
||||
"transformation_digest": "sha256:23456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef01",
|
||||
"instrument_route_revision_ids": [
|
||||
"rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86"
|
||||
],
|
||||
"trading_calendar_revision_ids": [
|
||||
"rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
],
|
||||
"corporate_action_revision_ids": [
|
||||
"rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
]
|
||||
}
|
||||
],
|
||||
"revision_lineage": [
|
||||
{
|
||||
"revision_kind": "instrument_route",
|
||||
"revision_id": "rhroutev1:sha256:ca67013250e28ab4cce16570607379a8792400e62415ee6cb71c75508e2f3d86",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T07:00:00Z",
|
||||
"evidence_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||
},
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T08:00:00Z",
|
||||
"evidence_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||
},
|
||||
{
|
||||
"revision_kind": "corporate_action",
|
||||
"revision_id": "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275",
|
||||
"revision_number": 1,
|
||||
"knowledge_time": "2026-01-01T09:00:00Z",
|
||||
"evidence_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd"
|
||||
}
|
||||
],
|
||||
"readiness": {
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"contract_validation": {
|
||||
"status": "validated",
|
||||
"evidence_digests": [
|
||||
"sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee"
|
||||
]
|
||||
},
|
||||
"real_data_validation": {"status": "not_validated", "evidence_digests": []},
|
||||
"production_validation": {"status": "not_validated", "evidence_digests": []},
|
||||
"live_validation": {"status": "not_validated", "evidence_digests": []}
|
||||
}
|
||||
},
|
||||
"output_schema": {
|
||||
"columns": ["evaluation_at", "factor_id", "instrument_id", "value"],
|
||||
"schema_version": "1.0.0"
|
||||
},
|
||||
"output_content": {
|
||||
"rows": [
|
||||
{
|
||||
"evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"factor_id": "alpha_005",
|
||||
"instrument_id": "rhinstrument:0123456789abcdef0123456789abcdef",
|
||||
"value": "0.125"
|
||||
}
|
||||
]
|
||||
},
|
||||
"expected": {
|
||||
"definition_id": "rhfactorv1:sha256:978fb8000d318373844a5e044ca14bf377e01ebe8d85964b826ecd2af9085ce9",
|
||||
"input_schema_digest": "sha256:4501aeab99b4bcc25a1b8813ebe197fb498053fd710d73746bf20fc8eeb4bfa7",
|
||||
"factor_set_id": "rhfactorsetv1:sha256:e9339581cf569e92459f672e8081337712e7bf98e58ad42d60d7ed13f9b5a021",
|
||||
"output_artifact_id": "rhfactoroutputv1:sha256:a4803b5ff66d12d3a0e7e8e5b8cca953cbc137e2bf41514b7c4e5f05da5ee68b",
|
||||
"legacy_binding_id": "rhlegacyfactorv1:sha256:541bc5a9469f9c8e4c2d696a9972fc5f2e6e2bef218b86f728823994b915dede"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,129 @@
|
||||
{
|
||||
"portfolio_decision": {
|
||||
"computed_at": "2026-01-08T03:01:00Z",
|
||||
"constraint_residuals": {
|
||||
"gross_exposure_max": 0.0,
|
||||
"net_exposure_max": 0.0,
|
||||
"net_exposure_min": 0.0,
|
||||
"position_count_max": 0.0,
|
||||
"single_asset_max": 0.0,
|
||||
"single_asset_min": 0.0,
|
||||
"turnover_max": 0.0
|
||||
},
|
||||
"constraints": {
|
||||
"gross_exposure_max": 1.0,
|
||||
"net_exposure_max": 1.0,
|
||||
"net_exposure_min": 1.0,
|
||||
"position_count_max": 2,
|
||||
"schema_version": "1.0.0",
|
||||
"single_asset_max": 0.7,
|
||||
"single_asset_min": 0.2,
|
||||
"turnover_max": 0.2
|
||||
},
|
||||
"contract_name": "researchhub.portfolio-decision",
|
||||
"covariance_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
|
||||
"effective_at": "2026-01-08T03:00:00Z",
|
||||
"evidence_digest": "sha256:f3913894d032c389c64eef59058b3cbc694cb9cc14ce9cee2699f0068200b650",
|
||||
"expected_return_digest": "sha256:dddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddddd",
|
||||
"freshness_policy": {
|
||||
"max_covariance_age_days": 0,
|
||||
"max_manifest_age_seconds": 3600,
|
||||
"schema_version": "1.0.0"
|
||||
},
|
||||
"gross_exposure": 1.0,
|
||||
"manifest_document_sha256": "fbf54218f770528978f9ccd35577e1ab00877143ea397a576e071e75f0afbab0",
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
|
||||
"model_digest": "sha256:cccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccccc",
|
||||
"model_name": "deterministic_weights",
|
||||
"model_version": "1.0.0",
|
||||
"net_exposure": 1.0,
|
||||
"objective_digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
|
||||
"objective_name": "long_only_allocation",
|
||||
"objective_version": "1.0.0",
|
||||
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
|
||||
"portfolio_asset_set_digest": "sha256:b64e3448a83a5b86466465080361c1a7e1157a27ddccd4b68069cb18caffb74a",
|
||||
"position_count": 2,
|
||||
"prior_weights": {
|
||||
"A": 0.5,
|
||||
"B": 0.5
|
||||
},
|
||||
"receipt": {
|
||||
"algorithm": "bounded_allocation",
|
||||
"algorithm_version": "1.0.0",
|
||||
"computed_at": "2026-01-08T03:01:00Z",
|
||||
"constraint_digest": "sha256:34df0e5c00f503748ff936f9cd415a8947169181f8ca0917bce97dd606e08e94",
|
||||
"implementation_digest": "sha256:ffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffff",
|
||||
"input_digest": "sha256:ebd8ff957115e1adfd84eafa5cea49470356194d747b64ee5897858f9dd067b7",
|
||||
"iterations": null,
|
||||
"max_constraint_residual": 0.0,
|
||||
"objective_value": null,
|
||||
"output_digest": "sha256:bd3b964c628c8648322d036e57dd6f444ca287017d1578bab3689d07d32b28ce",
|
||||
"parameter_digest": "sha256:0000000000000000000000000000000000000000000000000000000000000000",
|
||||
"schema_version": "1.0.0",
|
||||
"solver_config_digest": null,
|
||||
"solver_name": null,
|
||||
"solver_required": false,
|
||||
"solver_version": null,
|
||||
"status": "completed",
|
||||
"tolerance": 1e-12
|
||||
},
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"run_ref_document_sha256": "6a798adb3e0568aca84ed3e3a285b92d181ec569462fc003952a8c0d8382ae3a",
|
||||
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
|
||||
"schema_version": "1.0.0",
|
||||
"source_universe_digest": "sha256:5555555555555555555555555555555555555555555555555555555555555555",
|
||||
"target_id": "portfolio-target:synthetic-v1",
|
||||
"target_weights": {
|
||||
"A": 0.6,
|
||||
"B": 0.4
|
||||
},
|
||||
"turnover_l1": 0.19999999999999996
|
||||
},
|
||||
"risk_assessment": {
|
||||
"assessment_id": "rhriskassessmentv1:sha256:dbc38825cffcf6d95bd0216d22dbba4e0d4a1359ca5924a3ee529b99a7d78b6d",
|
||||
"component_risk": {
|
||||
"A": 1.4549226783578566,
|
||||
"B": 1.4549226783578568
|
||||
},
|
||||
"contract_name": "researchhub.risk-assessment",
|
||||
"covariance_as_of_date": "2026-01-08",
|
||||
"covariance_data_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"covariance_input_digest": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
|
||||
"covariance_snapshot_id": "covariance:synthetic-v1",
|
||||
"dataset_snapshot_id": "rhdsv1:sha256:f63a29b4795c63fb7d6b2d3b5544cee9274b633db77c75c63d50a340c0827d57",
|
||||
"decision_id": "rhportfoliodecisionv1:sha256:0e6e5ce2fa08de8006cc392610695a013327fc80645bfa4d7a908ad3f615bebd",
|
||||
"findings": [],
|
||||
"freshness_policy_digest": "sha256:833f58f4d1056f0450f7369fd4edd70534cc74e9e55cd60f05b2b3d7a2979763",
|
||||
"group_exposure": {
|
||||
"equity": 1.4549226783578566,
|
||||
"fixed_income": 1.4549226783578568
|
||||
},
|
||||
"manifest_id": "rhbacktestevidencev1:sha256:681c49cbdfb3e221b273e7bc616602ad80a7ab01206970807ad4294b176ebc75",
|
||||
"marginal_risk": {
|
||||
"A": 2.424871130596428,
|
||||
"B": 3.637306695894642
|
||||
},
|
||||
"percentage_risk": {
|
||||
"A": 0.49999999999999983,
|
||||
"B": 0.49999999999999994
|
||||
},
|
||||
"periods_per_year": 252,
|
||||
"portfolio_volatility": 2.909845356715714,
|
||||
"portfolio_volatility_limit": 10.0,
|
||||
"qualified": true,
|
||||
"return_frequency": "1d",
|
||||
"risk_budget": {
|
||||
"A": 0.8,
|
||||
"B": 0.8
|
||||
},
|
||||
"risk_model_digest": "sha256:2222222222222222222222222222222222222222222222222222222222222222",
|
||||
"risk_model_name": "euler_volatility",
|
||||
"risk_model_version": "1.0.0",
|
||||
"run_id": "rhbacktestrunv1:sha256:5036c771c44a2adade9ea590eee0d8cf824ff8f519fadd8ba424e5d914856386",
|
||||
"scenario_digest": "sha256:eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee",
|
||||
"schema_version": "1.0.0",
|
||||
"status": "ready"
|
||||
}
|
||||
}
|
||||
@@ -1,32 +1,70 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
class ModuleSpecTests(unittest.TestCase):
|
||||
def test_module_spec_declares_pure_research_engine_boundary(self) -> None:
|
||||
spec = json.loads((ROOT / "MODULE_SPEC.yaml").read_text(encoding="utf-8"))
|
||||
self.assertEqual(spec["module_id"], "quant_engine")
|
||||
self.assertEqual(spec["authority"]["subject"], spec["module_id"])
|
||||
self.assertEqual(spec["repository"]["type"], "research_engine")
|
||||
self.assertEqual(spec["bounded_context"]["domain"], "quantitative-research-engine")
|
||||
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
|
||||
for term in ("investment advice", "live order", "credentials", "source facts"):
|
||||
self.assertIn(term, prohibited)
|
||||
self.assertEqual(spec["contracts"], {"provides": [], "consumes": []})
|
||||
self.assertEqual(spec["dependencies"], [])
|
||||
self.assertTrue(
|
||||
all(
|
||||
command["required"] and not command["network"]
|
||||
for command in spec["verification"]["commands"]
|
||||
)
|
||||
)
|
||||
def test_module_spec_declares_pure_research_engine_boundary() -> None:
|
||||
spec = json.loads((ROOT / "MODULE_SPEC.yaml").read_text(encoding="utf-8"))
|
||||
assert spec["module_id"] == "quant_engine"
|
||||
assert spec["authority"]["subject"] == spec["module_id"]
|
||||
assert spec["repository"]["type"] == "research_engine"
|
||||
assert spec["bounded_context"]["domain"] == "quantitative-research-engine"
|
||||
prohibited = " ".join(spec["bounded_context"]["prohibited_responsibilities"]).lower()
|
||||
for term in ("investment advice", "live order", "credentials", "source facts"):
|
||||
assert term in prohibited
|
||||
assert spec["authority"]["revision"] == 4
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["provides"]
|
||||
} == {
|
||||
("researchhub.factor-definition", "1.0.0"),
|
||||
("researchhub.factor-set-ref", "1.0.0"),
|
||||
("researchhub.backtest-run-ref", "1.0.0"),
|
||||
("researchhub.backtest-evidence-manifest", "1.0.0"),
|
||||
("researchhub.portfolio-decision", "1.0.0"),
|
||||
("researchhub.risk-assessment", "1.0.0"),
|
||||
}
|
||||
expected_paths = {
|
||||
"researchhub.factor-definition": "src/quant_engine/factor_contracts.py",
|
||||
"researchhub.factor-set-ref": "src/quant_engine/factor_contracts.py",
|
||||
"researchhub.backtest-run-ref": "src/quant_engine/governed_pipeline.py",
|
||||
"researchhub.backtest-evidence-manifest": "src/quant_engine/artifact.py",
|
||||
"researchhub.portfolio-decision": "src/quant_engine/portfolio_risk_contracts.py",
|
||||
"researchhub.risk-assessment": "src/quant_engine/portfolio_risk_contracts.py",
|
||||
}
|
||||
assert all(item["authority"] == "quant_engine" for item in spec["contracts"]["provides"])
|
||||
assert {
|
||||
item["contract_id"]: item["path"] for item in spec["contracts"]["provides"]
|
||||
} == expected_paths
|
||||
assert {
|
||||
(item["contract_id"], item["version"])
|
||||
for item in spec["contracts"]["consumes"]
|
||||
} == {
|
||||
("researchhub.dataset-snapshot", "1.0.0"),
|
||||
("researchhub.data-foundation", "1.0.0"),
|
||||
}
|
||||
assert all(
|
||||
item["authority"] == "researchhub.data"
|
||||
for item in spec["contracts"]["consumes"]
|
||||
)
|
||||
assert spec["dependencies"] == []
|
||||
capabilities = {item["id"]: item for item in spec["capabilities"]}
|
||||
portfolio_contract = capabilities["portfolio-risk-computation-contracts"]
|
||||
assert portfolio_contract["status"] == "operational"
|
||||
summary = portfolio_contract["summary"].lower()
|
||||
for term in ("receipt", "portfolio decisions", "risk assessments", "without"):
|
||||
assert term in summary
|
||||
for term in ("approval", "maker-checker", "publication", "paper", "live"):
|
||||
assert term in prohibited
|
||||
assert all(
|
||||
command["required"] and not command["network"]
|
||||
for command in spec["verification"]["commands"]
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
test_module_spec_declares_pure_research_engine_boundary()
|
||||
|
||||
@@ -6,8 +6,30 @@ import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
import quant_engine.alpha_factors as alpha_factors_module
|
||||
from quant_engine.factor_contracts import (
|
||||
FactorContractError,
|
||||
FactorInput,
|
||||
ProducerIdentity,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.alpha_factors import (
|
||||
ALPHA158_REGISTRY,
|
||||
ALPHA158_PHASE1_OPERATOR_SPECS,
|
||||
ALPHA158_PHASE2_OPERATOR_SPECS,
|
||||
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE3_FORMULA_SPECS,
|
||||
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE4_FORMULA_SPECS,
|
||||
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE5_FORMULA_SPECS,
|
||||
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256,
|
||||
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
|
||||
ALPHA158_PHASE6_FORMULA_SPECS,
|
||||
alpha_001,
|
||||
alpha_002,
|
||||
alpha_003,
|
||||
@@ -166,6 +188,18 @@ from quant_engine.alpha_factors import (
|
||||
alpha_156,
|
||||
alpha_157,
|
||||
alpha_158,
|
||||
evaluate_phase1_operator,
|
||||
evaluate_phase2_operator,
|
||||
evaluate_phase3_formula,
|
||||
evaluate_phase4_formula,
|
||||
evaluate_phase5_formula,
|
||||
evaluate_phase6_formula,
|
||||
list_phase1_operators,
|
||||
list_phase2_operators,
|
||||
list_phase3_formulas,
|
||||
list_phase4_formulas,
|
||||
list_phase5_formulas,
|
||||
list_phase6_formulas,
|
||||
correlation,
|
||||
covariance,
|
||||
decay_linear,
|
||||
@@ -407,6 +441,50 @@ def test_alpha_registry_required_fields():
|
||||
assert required <= set(meta.keys()), f"{alpha_id} missing fields"
|
||||
|
||||
|
||||
def test_alpha_registry_adapts_to_definition_without_copying_formula_or_inputs():
|
||||
factor_input = FactorInput(
|
||||
"market",
|
||||
"sha256:" + "1" * 64,
|
||||
tuple(ALPHA158_REGISTRY["alpha_005"]["inputs"]),
|
||||
)
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "2" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
assert definition.formula == ALPHA158_REGISTRY["alpha_005"]["formula"]
|
||||
assert definition.inputs[0].required_columns == tuple(
|
||||
ALPHA158_REGISTRY["alpha_005"]["inputs"]
|
||||
)
|
||||
|
||||
incomplete = FactorInput("market", "sha256:" + "1" * 64, ("close",))
|
||||
with pytest.raises(FactorContractError, match="exactly correspond"):
|
||||
factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(incomplete,),
|
||||
implementation_digest="sha256:" + "2" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((incomplete,)),
|
||||
valid_from="2026-01-01T00:00:00Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
|
||||
def test_get_alpha_meta_success():
|
||||
"""已知 alpha_id 返回完整 meta。"""
|
||||
meta = get_alpha_meta("alpha_001")
|
||||
@@ -1232,3 +1310,901 @@ def test_parse_alpha_formula_round_trip_jsonb():
|
||||
serialized = json.dumps(parsed)
|
||||
assert isinstance(serialized, str)
|
||||
assert "ts_rank" in serialized
|
||||
|
||||
|
||||
# ── v1.2.0 Phase 1: deterministic operator dispatch contract ──────────────
|
||||
|
||||
|
||||
def test_phase1_operator_catalog_is_explicit_and_serializable():
|
||||
"""Phase 1 exposes a stable, JSON-friendly catalog for downstream callers."""
|
||||
import json
|
||||
|
||||
expected = {
|
||||
"rank",
|
||||
"delta",
|
||||
"ts_mean",
|
||||
"ts_std",
|
||||
"ts_rank",
|
||||
"correlation",
|
||||
"ts_min",
|
||||
"ts_max",
|
||||
"ts_sum",
|
||||
"decay_linear",
|
||||
}
|
||||
assert set(list_phase1_operators()) == expected
|
||||
assert set(ALPHA158_PHASE1_OPERATOR_SPECS) == expected
|
||||
json.dumps(ALPHA158_PHASE1_OPERATOR_SPECS)
|
||||
for name, spec in ALPHA158_PHASE1_OPERATOR_SPECS.items():
|
||||
assert spec["name"] == name
|
||||
assert isinstance(spec["inputs"], list)
|
||||
assert isinstance(spec["formula"], str)
|
||||
|
||||
|
||||
def test_phase1_unary_operators_preserve_index_and_are_deterministic():
|
||||
values = pd.Series([1.0, 2.0, 3.0, 4.0], index=["a", "b", "c", "d"])
|
||||
|
||||
first = evaluate_phase1_operator("rank", values)
|
||||
second = evaluate_phase1_operator("rank", values)
|
||||
|
||||
pd.testing.assert_series_equal(first, second)
|
||||
assert first.index.equals(values.index)
|
||||
assert first.iloc[-1] == pytest.approx(1.0)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("name", "window"),
|
||||
[
|
||||
("delta", 2),
|
||||
("ts_mean", 2),
|
||||
("ts_std", 2),
|
||||
("ts_rank", 2),
|
||||
("ts_min", 2),
|
||||
("ts_max", 2),
|
||||
("ts_sum", 2),
|
||||
("decay_linear", 2),
|
||||
],
|
||||
)
|
||||
def test_phase1_windowed_operators_require_explicit_window(name: str, window: int):
|
||||
values = pd.Series([1.0, 2.0, 3.0, 4.0])
|
||||
|
||||
result = evaluate_phase1_operator(name, values, window=window)
|
||||
|
||||
assert result.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase1_operator(name, values)
|
||||
with pytest.raises(ValueError, match="positive integer"):
|
||||
evaluate_phase1_operator(name, values, window=1.5) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_phase1_binary_correlation_requires_aligned_secondary_input():
|
||||
values = pd.Series([1.0, 2.0, 3.0, 4.0])
|
||||
other = pd.Series([4.0, 3.0, 2.0, 1.0])
|
||||
|
||||
result = evaluate_phase1_operator("correlation", values, other, window=2)
|
||||
|
||||
assert result.iloc[-1] == pytest.approx(-1.0)
|
||||
with pytest.raises(ValueError, match="secondary"):
|
||||
evaluate_phase1_operator("correlation", values, window=2)
|
||||
|
||||
|
||||
def test_phase1_dispatch_rejects_unknown_or_unused_arguments():
|
||||
values = pd.Series([1.0, 2.0, 3.0])
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase1_operator("unknown", values)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase1_operator("rank", values, window=2)
|
||||
with pytest.raises(ValueError, match="secondary"):
|
||||
evaluate_phase1_operator("rank", values, values)
|
||||
|
||||
|
||||
def test_phase1_dispatch_rejects_window_above_supported_limit():
|
||||
values = pd.Series([1.0, 2.0, 3.0])
|
||||
|
||||
with pytest.raises(ValueError, match="maximum"):
|
||||
evaluate_phase1_operator("ts_mean", values, window=2**63)
|
||||
|
||||
|
||||
# ── v1.2.0 Phase 2: cumulative deterministic operator contract ─────────────
|
||||
|
||||
|
||||
def test_phase2_operator_catalog_is_cumulative_stable_and_serializable():
|
||||
"""Phase 2 exposes all existing building blocks without changing Phase 1."""
|
||||
import json
|
||||
|
||||
phase1 = list_phase1_operators()
|
||||
expected_phase2 = (
|
||||
*phase1,
|
||||
"ts_argmin",
|
||||
"ts_argmax",
|
||||
"product",
|
||||
"returns",
|
||||
"scale",
|
||||
"signed_power",
|
||||
"stddev",
|
||||
"covariance",
|
||||
"log",
|
||||
"abs_series",
|
||||
"sign",
|
||||
"max_pair",
|
||||
"min_pair",
|
||||
"indneutralize",
|
||||
)
|
||||
|
||||
assert list_phase2_operators() == expected_phase2
|
||||
assert tuple(ALPHA158_PHASE2_OPERATOR_SPECS) == expected_phase2
|
||||
assert tuple(ALPHA158_PHASE1_OPERATOR_SPECS) == phase1
|
||||
json.dumps(ALPHA158_PHASE2_OPERATOR_SPECS)
|
||||
for name, spec in ALPHA158_PHASE2_OPERATOR_SPECS.items():
|
||||
assert spec["name"] == name
|
||||
assert isinstance(spec["inputs"], list)
|
||||
assert isinstance(spec["parameters"], list)
|
||||
assert isinstance(spec["formula"], str)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", ["ts_argmin", "ts_argmax", "product", "stddev"])
|
||||
def test_phase2_windowed_unary_dispatch_is_deterministic(name: str):
|
||||
values = pd.Series([3.0, 1.0, 4.0, 2.0], index=["a", "b", "c", "d"])
|
||||
|
||||
first = evaluate_phase2_operator(name, values, window=3)
|
||||
second = evaluate_phase2_operator(name, values, window=3)
|
||||
|
||||
pd.testing.assert_series_equal(first, second)
|
||||
assert first.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase2_operator(name, values)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", ["returns", "scale", "log", "abs_series", "sign"])
|
||||
def test_phase2_unary_dispatch_rejects_unused_arguments(name: str):
|
||||
values = pd.Series([1.0, 2.0, 4.0], index=["a", "b", "c"])
|
||||
|
||||
result = evaluate_phase2_operator(name, values)
|
||||
|
||||
assert result.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="window"):
|
||||
evaluate_phase2_operator(name, values, window=2)
|
||||
with pytest.raises(ValueError, match="secondary"):
|
||||
evaluate_phase2_operator(name, values, secondary=values)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("name", "window"),
|
||||
[("correlation", 2), ("covariance", 2), ("max_pair", None), ("min_pair", None)],
|
||||
)
|
||||
def test_phase2_binary_dispatch_requires_aligned_secondary(name: str, window: int | None):
|
||||
values = pd.Series([1.0, 2.0, 3.0], index=["a", "b", "c"])
|
||||
secondary = pd.Series([3.0, 2.0, 1.0], index=values.index)
|
||||
|
||||
result = evaluate_phase2_operator(name, values, secondary=secondary, window=window)
|
||||
|
||||
assert result.index.equals(values.index)
|
||||
with pytest.raises(ValueError, match="secondary is required"):
|
||||
evaluate_phase2_operator(name, values, window=window)
|
||||
with pytest.raises(ValueError, match="secondary index"):
|
||||
evaluate_phase2_operator(
|
||||
name,
|
||||
values,
|
||||
secondary=secondary.rename(index={"c": "z"}),
|
||||
window=window,
|
||||
)
|
||||
|
||||
|
||||
def test_phase2_signed_power_requires_finite_numeric_exponent():
|
||||
values = pd.Series([-4.0, 0.0, 9.0])
|
||||
|
||||
result = evaluate_phase2_operator("signed_power", values, exponent=0.5)
|
||||
|
||||
pd.testing.assert_series_equal(result, pd.Series([-2.0, 0.0, 3.0]))
|
||||
for exponent in (None, True, float("inf"), float("nan"), "2"):
|
||||
with pytest.raises(ValueError, match="exponent"):
|
||||
evaluate_phase2_operator( # type: ignore[arg-type]
|
||||
"signed_power",
|
||||
values,
|
||||
exponent=exponent,
|
||||
)
|
||||
|
||||
|
||||
def test_phase2_indneutralize_requires_aligned_groups():
|
||||
values = pd.Series([1.0, 3.0, 10.0, 14.0], index=["a", "b", "c", "d"])
|
||||
groups = pd.Series(["x", "x", "y", "y"], index=values.index)
|
||||
|
||||
result = evaluate_phase2_operator("indneutralize", values, groups=groups)
|
||||
|
||||
pd.testing.assert_series_equal(result, pd.Series([-1.0, 1.0, -2.0, 2.0], index=values.index))
|
||||
with pytest.raises(ValueError, match="groups is required"):
|
||||
evaluate_phase2_operator("indneutralize", values)
|
||||
with pytest.raises(ValueError, match="groups index"):
|
||||
evaluate_phase2_operator(
|
||||
"indneutralize",
|
||||
values,
|
||||
groups=groups.rename(index={"d": "z"}),
|
||||
)
|
||||
|
||||
|
||||
def test_phase2_dispatch_validates_primary_series_and_unused_parameters():
|
||||
values = pd.Series([1.0, 2.0, 3.0])
|
||||
|
||||
with pytest.raises(TypeError, match="series must be a pandas Series"):
|
||||
evaluate_phase2_operator("rank", [1.0, 2.0, 3.0]) # type: ignore[arg-type]
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase2_operator("unknown", values)
|
||||
with pytest.raises(ValueError, match="exponent"):
|
||||
evaluate_phase2_operator("rank", values, exponent=2.0)
|
||||
with pytest.raises(ValueError, match="groups"):
|
||||
evaluate_phase2_operator("rank", values, groups=pd.Series(["x", "x", "x"]))
|
||||
with pytest.raises(ValueError, match="maximum"):
|
||||
evaluate_phase2_operator("product", values, window=253)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 3: versioned alpha001-alpha050 formula contract ──────────
|
||||
|
||||
|
||||
def _phase3_market_inputs() -> dict[str, pd.Series]:
|
||||
positions = np.arange(80, dtype=float)
|
||||
index = pd.RangeIndex(len(positions), name="row")
|
||||
open_ = pd.Series(100.0 + positions * 0.2 + np.sin(positions / 4.0), index=index)
|
||||
close = pd.Series(100.5 + positions * 0.18 + np.cos(positions / 5.0), index=index)
|
||||
high = pd.Series(np.maximum(open_, close) + 1.0, index=index)
|
||||
low = pd.Series(np.minimum(open_, close) - 1.0, index=index)
|
||||
volume = pd.Series(1_000.0 + positions**1.3 + 20.0 * np.sin(positions / 3.0), index=index)
|
||||
vwap = (open_ + close + high + low) / 4.0
|
||||
return {
|
||||
"open": open_,
|
||||
"close": close,
|
||||
"high": high,
|
||||
"low": low,
|
||||
"volume": volume,
|
||||
"vwap": vwap,
|
||||
}
|
||||
|
||||
|
||||
def test_phase3_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(1, 51))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase3_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE3_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE3_FORMULA_SPECS.values()
|
||||
) == {"single": 19, "pair": 26, "triple": 4, "quadruple": 1}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE3_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE3_FORMULA_CATALOG_SHA256 == (
|
||||
"9a3360e5ee77a85a35d3c2fdab1efaa531fa0c263a2cb2bb5b965a1fb96fe1bd"
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE3_FORMULA_SPECS, "alpha_001", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE3_FORMULA_SPECS["alpha_001"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_catalog_freezes_callable_signatures_without_rewriting_formulas():
|
||||
import inspect
|
||||
|
||||
legacy_formula_input_differences = {
|
||||
"alpha_011": ("close", "high", "low"),
|
||||
"alpha_035": ("volume",),
|
||||
"alpha_036": ("close",),
|
||||
"alpha_040": ("high", "low"),
|
||||
"alpha_042": ("close",),
|
||||
"alpha_043": ("volume",),
|
||||
}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == legacy_formula_input_differences.get(
|
||||
alpha_id,
|
||||
signature_inputs,
|
||||
)
|
||||
|
||||
assert ALPHA158_PHASE3_FORMULA_SPECS["alpha_035"]["call_inputs"] == (
|
||||
"close",
|
||||
"volume",
|
||||
)
|
||||
assert ALPHA158_REGISTRY["alpha_035"]["inputs"] == ["volume"]
|
||||
|
||||
|
||||
def test_phase3_dispatch_matches_all_existing_alpha001_alpha050_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE3_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase3_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase3_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase3_formula("alpha_051", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
|
||||
evaluate_phase3_formula("alpha_005", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase3_formula(
|
||||
"alpha_005",
|
||||
close=inputs["close"],
|
||||
volume=inputs["volume"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="close must be a pandas Series"):
|
||||
evaluate_phase3_formula( # type: ignore[arg-type]
|
||||
"alpha_005",
|
||||
close=[1.0, 2.0],
|
||||
volume=inputs["volume"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase3_dispatch_rejects_implicit_series_alignment():
|
||||
inputs = _phase3_market_inputs()
|
||||
misaligned_volume = inputs["volume"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="volume index must align with close"):
|
||||
evaluate_phase3_formula(
|
||||
"alpha_005",
|
||||
close=inputs["close"],
|
||||
volume=misaligned_volume,
|
||||
)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 4: versioned alpha051-alpha100 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase4_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(51, 101))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase4_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE4_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE4_FORMULA_SPECS.values()
|
||||
) == {"single": 1, "pair": 27, "triple": 11, "quadruple": 10, "quintuple": 1}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE4_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 == (
|
||||
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE4_FORMULA_SPECS, "alpha_051", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE4_FORMULA_SPECS["alpha_055"]["call_inputs"] == (
|
||||
"open",
|
||||
"high",
|
||||
"low",
|
||||
"volume",
|
||||
"close",
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_dispatch_matches_all_existing_alpha051_alpha100_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase4_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase4_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase4_formula("alpha_050", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*low"):
|
||||
evaluate_phase4_formula("alpha_051", high=inputs["high"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase4_formula(
|
||||
"alpha_051",
|
||||
high=inputs["high"],
|
||||
low=inputs["low"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="high must be a pandas Series"):
|
||||
evaluate_phase4_formula( # type: ignore[arg-type]
|
||||
"alpha_051",
|
||||
high=[1.0, 2.0],
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase4_dispatch_rejects_implicit_series_alignment():
|
||||
inputs = _phase3_market_inputs()
|
||||
misaligned_low = inputs["low"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="low index must align with high"):
|
||||
evaluate_phase4_formula(
|
||||
"alpha_051",
|
||||
high=inputs["high"],
|
||||
low=misaligned_low,
|
||||
)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 5: versioned alpha101-alpha150 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase5_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(101, 151))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase5_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE5_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE5_FORMULA_SPECS.values()
|
||||
) == {"pair": 33, "triple": 14, "quadruple": 3}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE5_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 == (
|
||||
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE5_FORMULA_SPECS, "alpha_101", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"] == (
|
||||
"close",
|
||||
"high",
|
||||
"low",
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_dispatch_matches_all_existing_alpha101_alpha150_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase5_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase5_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase5_formula("alpha_100", close=inputs["close"])
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase5_formula("alpha_151", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*low"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
)
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
low=inputs["low"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="high must be a pandas Series"):
|
||||
evaluate_phase5_formula( # type: ignore[arg-type]
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=[1.0, 2.0],
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_dispatch_rejects_length_and_index_alignment_errors():
|
||||
inputs = _phase3_market_inputs()
|
||||
shorter_low = inputs["low"].iloc[:-1]
|
||||
misaligned_high = inputs["high"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="low length must match close"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=inputs["high"],
|
||||
low=shorter_low,
|
||||
)
|
||||
with pytest.raises(ValueError, match="high index must align with close"):
|
||||
evaluate_phase5_formula(
|
||||
"alpha_101",
|
||||
close=inputs["close"],
|
||||
high=misaligned_high,
|
||||
low=inputs["low"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase5_contract_is_publicly_exported():
|
||||
assert {
|
||||
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE5_FORMULA_SPECS",
|
||||
"list_phase5_formulas",
|
||||
"evaluate_phase5_formula",
|
||||
} <= set(alpha_factors_module.__all__)
|
||||
|
||||
|
||||
# ── Alpha158 Phase 6: versioned alpha151-alpha158 formula contract ──────────
|
||||
|
||||
|
||||
def test_phase6_formula_catalog_is_versioned_exact_and_content_addressed():
|
||||
import hashlib
|
||||
import json
|
||||
from collections import Counter
|
||||
|
||||
expected_ids = tuple(f"alpha_{number:03d}" for number in range(151, 159))
|
||||
expected_fields = {
|
||||
"name",
|
||||
"contract_version",
|
||||
"formula",
|
||||
"category",
|
||||
"complexity",
|
||||
"parameters",
|
||||
"description",
|
||||
"references",
|
||||
"call_inputs",
|
||||
"formula_inputs",
|
||||
"input_category",
|
||||
}
|
||||
|
||||
assert ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION == "1.0.0"
|
||||
assert list_phase6_formulas() == expected_ids
|
||||
assert tuple(ALPHA158_PHASE6_FORMULA_SPECS) == expected_ids
|
||||
assert Counter(
|
||||
spec["input_category"] for spec in ALPHA158_PHASE6_FORMULA_SPECS.values()
|
||||
) == {"pair": 6, "triple": 2}
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
assert set(spec) == expected_fields
|
||||
assert spec["name"] == alpha_id
|
||||
assert spec["contract_version"] == ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION
|
||||
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
|
||||
|
||||
serializable_specs = {
|
||||
alpha_id: {
|
||||
field: list(value) if isinstance(value, tuple) else value
|
||||
for field, value in spec.items()
|
||||
}
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items()
|
||||
}
|
||||
encoded = json.dumps(
|
||||
serializable_specs,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE6_FORMULA_CATALOG_SHA256
|
||||
assert ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 == (
|
||||
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_formula_catalog_is_recursively_immutable():
|
||||
import operator
|
||||
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(ALPHA158_PHASE6_FORMULA_SPECS, "alpha_151", {})
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"],
|
||||
"formula",
|
||||
"changed",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
operator.setitem(
|
||||
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"]["call_inputs"],
|
||||
0,
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_catalog_freezes_callable_and_formula_inputs():
|
||||
import inspect
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
signature_inputs = tuple(
|
||||
"open" if name == "open_" else name
|
||||
for name in inspect.signature(function).parameters
|
||||
)
|
||||
assert spec["call_inputs"] == signature_inputs
|
||||
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
|
||||
|
||||
assert ALPHA158_PHASE6_FORMULA_SPECS["alpha_158"]["call_inputs"] == (
|
||||
"high",
|
||||
"low",
|
||||
"volume",
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_dispatch_matches_all_existing_alpha151_alpha158_functions():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
|
||||
call_inputs = spec["call_inputs"]
|
||||
function = getattr(alpha_factors_module, alpha_id)
|
||||
expected = function(*(inputs[name] for name in call_inputs))
|
||||
actual = evaluate_phase6_formula(
|
||||
alpha_id,
|
||||
**{name: inputs[name] for name in reversed(call_inputs)},
|
||||
)
|
||||
pd.testing.assert_series_equal(actual, expected)
|
||||
|
||||
|
||||
def test_phase6_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
|
||||
inputs = _phase3_market_inputs()
|
||||
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase6_formula("alpha_150", close=inputs["close"])
|
||||
with pytest.raises(KeyError, match="not registered"):
|
||||
evaluate_phase6_formula("alpha_159", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
|
||||
evaluate_phase6_formula("alpha_151", close=inputs["close"])
|
||||
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=inputs["volume"],
|
||||
vwap=inputs["vwap"],
|
||||
)
|
||||
with pytest.raises(TypeError, match="volume must be a pandas Series"):
|
||||
evaluate_phase6_formula( # type: ignore[arg-type]
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=[1.0, 2.0],
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_dispatch_rejects_length_and_index_alignment_errors():
|
||||
inputs = _phase3_market_inputs()
|
||||
shorter_volume = inputs["volume"].iloc[:-1]
|
||||
misaligned_low = inputs["low"].rename(index={79: 80})
|
||||
|
||||
with pytest.raises(ValueError, match="volume length must match close"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_151",
|
||||
close=inputs["close"],
|
||||
volume=shorter_volume,
|
||||
)
|
||||
with pytest.raises(ValueError, match="low index must align with high"):
|
||||
evaluate_phase6_formula(
|
||||
"alpha_158",
|
||||
high=inputs["high"],
|
||||
low=misaligned_low,
|
||||
volume=inputs["volume"],
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_preserves_complete_alpha001_alpha158_formula_body_fingerprint():
|
||||
import ast
|
||||
import hashlib
|
||||
import inspect
|
||||
import json
|
||||
import textwrap
|
||||
|
||||
fingerprints = {}
|
||||
for number in range(1, 159):
|
||||
alpha_id = f"alpha_{number:03d}"
|
||||
source = textwrap.dedent(inspect.getsource(getattr(alpha_factors_module, alpha_id)))
|
||||
node = ast.parse(source).body[0]
|
||||
assert isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef))
|
||||
body = ast.dump(
|
||||
ast.Module(body=node.body, type_ignores=[]),
|
||||
include_attributes=False,
|
||||
)
|
||||
fingerprints[alpha_id] = hashlib.sha256(body.encode()).hexdigest()
|
||||
|
||||
encoded = json.dumps(
|
||||
fingerprints,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
).encode()
|
||||
assert hashlib.sha256(encoded).hexdigest() == (
|
||||
"f3ae807983cf8ca083d0b924c3807ffd84a62a5bd354f9efb1a1a16ca40c7da6"
|
||||
)
|
||||
|
||||
|
||||
def test_phase6_contract_is_publicly_exported():
|
||||
assert {
|
||||
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
|
||||
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
|
||||
"ALPHA158_PHASE6_FORMULA_SPECS",
|
||||
"list_phase6_formulas",
|
||||
"evaluate_phase6_formula",
|
||||
} <= set(alpha_factors_module.__all__)
|
||||
|
||||
@@ -0,0 +1,309 @@
|
||||
"""Stable research-run artifact contracts for downstream persistence."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import date
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
RESEARCH_ARTIFACT_SCHEMA_VERSION,
|
||||
ResearchRunArtifact,
|
||||
build_research_run_artifact,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
|
||||
from quant_engine.risk import CovarianceSnapshot
|
||||
|
||||
|
||||
def _backtest_result() -> FactorBacktestResult:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
scores = pd.DataFrame(
|
||||
{"A": [2.0, 0.0], "B": [1.0, 3.0]},
|
||||
index=dates[:2],
|
||||
)
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
return run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _build(
|
||||
result: FactorBacktestResult,
|
||||
*,
|
||||
parameters: dict[str, object] | None = None,
|
||||
risk_snapshots: dict[date, CovarianceSnapshot] | None = None,
|
||||
) -> ResearchRunArtifact:
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id="run-20260105-a",
|
||||
strategy_id="alpha-top1",
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version="1.0.0",
|
||||
engine_version="1.2.0",
|
||||
code_revision="3b1ad07",
|
||||
data_snapshot_id="qtdb-pro-20260108-v1",
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=parameters or {"top_k": 1, "lag_sessions": 1},
|
||||
benchmark_id="000300.SH",
|
||||
benchmark_returns=benchmark,
|
||||
risk_snapshots=risk_snapshots,
|
||||
)
|
||||
|
||||
|
||||
def test_research_artifact_projects_versioned_queryable_fact_tables() -> None:
|
||||
result = _backtest_result()
|
||||
|
||||
artifact = _build(result)
|
||||
|
||||
assert artifact.schema_version == RESEARCH_ARTIFACT_SCHEMA_VERSION
|
||||
assert artifact.run.loc[0, "run_id"] == "run-20260105-a"
|
||||
assert artifact.run.loc[0, "benchmark_alignment_policy"] == "exact_session_index"
|
||||
assert artifact.nav["run_id"].unique().tolist() == ["run-20260105-a"]
|
||||
assert artifact.nav["pnl_pct"].tolist() == pytest.approx(result.returns.tolist())
|
||||
assert artifact.nav["benchmark_return"].tolist() == pytest.approx(
|
||||
[0.0, 0.01, -0.01, 0.02]
|
||||
)
|
||||
assert artifact.signals.columns.tolist() == [
|
||||
"run_id",
|
||||
"signal_date",
|
||||
"execution_date",
|
||||
"asset_id",
|
||||
"factor_score",
|
||||
"target_weight",
|
||||
]
|
||||
first_signal = artifact.signals[
|
||||
artifact.signals["signal_date"] == result.factor_scores.index[0].date()
|
||||
]
|
||||
assert first_signal.set_index("asset_id").loc["A", "factor_score"] == 2.0
|
||||
assert first_signal.set_index("asset_id").loc["A", "target_weight"] == 1.0
|
||||
assert first_signal["execution_date"].unique().tolist() == [
|
||||
result.schedule.signal_to_execution.iloc[0].date()
|
||||
]
|
||||
assert set(artifact.trades["side"]) == {"buy", "sell"}
|
||||
assert artifact.trades["trade_id"].is_unique
|
||||
assert artifact.trades["trade_id"].str.startswith("run-20260105-a:").all()
|
||||
assert artifact.trades["signal_id"].str.startswith("run-20260105-a:signal:").all()
|
||||
assert {"security", "cash"}.issubset(set(artifact.positions["asset_type"]))
|
||||
assert artifact.positions.groupby("trade_date")["weight"].sum().tolist() == pytest.approx(
|
||||
[1.0, 1.0, 1.0, 1.0]
|
||||
)
|
||||
assert set(artifact.attribution.columns) == {
|
||||
"run_id",
|
||||
"trade_date",
|
||||
"asset_id",
|
||||
"overnight",
|
||||
"intraday",
|
||||
"asset_total",
|
||||
}
|
||||
assert artifact.attribution_daily["residual"].abs().max() < 1e-12
|
||||
assert artifact.risk.empty
|
||||
assert artifact.risk.columns.tolist() == [
|
||||
"run_id",
|
||||
"trade_date",
|
||||
"asset_id",
|
||||
"weight",
|
||||
"marginal_risk",
|
||||
"component_risk",
|
||||
"risk_contribution",
|
||||
"covariance_snapshot_id",
|
||||
"covariance_as_of_date",
|
||||
"risk_measure",
|
||||
"return_frequency",
|
||||
"periods_per_year",
|
||||
]
|
||||
assert artifact.performance.loc[0, "n_trades"] == len(artifact.trades)
|
||||
assert artifact.performance.loc[0, "ir"] == pytest.approx(
|
||||
result.benchmark_stats(pd.Series([0.0, 0.01, -0.01, 0.02], index=result.returns.index))[
|
||||
"information_ratio"
|
||||
]
|
||||
)
|
||||
assert "sortino" in artifact.performance.columns
|
||||
|
||||
|
||||
def test_research_artifact_projects_annualized_risk_from_actual_positions() -> None:
|
||||
result = _backtest_result()
|
||||
trade_date = result.position_weights.index[-1].date()
|
||||
covariance = pd.DataFrame(
|
||||
[[0.0001, 0.00002], [0.00002, 0.0004]],
|
||||
index=["A", "B"],
|
||||
columns=["A", "B"],
|
||||
)
|
||||
snapshot = CovarianceSnapshot(
|
||||
snapshot_id="cov-20260107-v1",
|
||||
as_of_date="2026-01-07",
|
||||
covariance=covariance,
|
||||
return_frequency="1d",
|
||||
periods_per_year=252,
|
||||
data_snapshot_id="qtdb-pro-20260108-v1",
|
||||
)
|
||||
|
||||
artifact = _build(result, risk_snapshots={trade_date: snapshot})
|
||||
|
||||
risk = artifact.risk.set_index("asset_id")
|
||||
expected_weights = result.position_weights.loc[pd.Timestamp(trade_date)]
|
||||
assert artifact.schema_version == "1.1.0"
|
||||
assert risk.index.tolist() == ["A", "B"]
|
||||
assert risk["weight"].tolist() == pytest.approx(expected_weights.tolist())
|
||||
assert risk["covariance_snapshot_id"].unique().tolist() == ["cov-20260107-v1"]
|
||||
assert risk["covariance_as_of_date"].unique().tolist() == [date(2026, 1, 7)]
|
||||
assert risk["risk_measure"].unique().tolist() == ["annualized_volatility"]
|
||||
assert risk["return_frequency"].unique().tolist() == ["1d"]
|
||||
assert risk["periods_per_year"].unique().tolist() == [252]
|
||||
assert risk["component_risk"].sum() == pytest.approx((0.0004 * 252) ** 0.5)
|
||||
assert risk["risk_contribution"].sum() == pytest.approx(1.0)
|
||||
|
||||
|
||||
def test_research_artifact_rejects_risk_from_a_different_data_snapshot() -> None:
|
||||
result = _backtest_result()
|
||||
trade_date = result.position_weights.index[-1].date()
|
||||
covariance = pd.DataFrame(
|
||||
[[0.0001, 0.0], [0.0, 0.0004]],
|
||||
index=["A", "B"],
|
||||
columns=["A", "B"],
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="data lineage differs"):
|
||||
_build(
|
||||
result,
|
||||
risk_snapshots={
|
||||
trade_date: CovarianceSnapshot(
|
||||
snapshot_id="foreign-covariance",
|
||||
as_of_date="2026-01-07",
|
||||
covariance=covariance,
|
||||
return_frequency="1d",
|
||||
periods_per_year=252,
|
||||
data_snapshot_id="different-market-snapshot",
|
||||
)
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def test_research_artifact_rejects_future_or_misaligned_risk_snapshots() -> None:
|
||||
result = _backtest_result()
|
||||
trade_date = result.position_weights.index[-1].date()
|
||||
covariance = pd.DataFrame(
|
||||
[[0.0001, 0.0], [0.0, 0.0004]],
|
||||
index=["A", "B"],
|
||||
columns=["A", "B"],
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="must not be after trade date"):
|
||||
_build(
|
||||
result,
|
||||
risk_snapshots={
|
||||
trade_date: CovarianceSnapshot(
|
||||
snapshot_id="future-covariance",
|
||||
as_of_date="2026-01-09",
|
||||
covariance=covariance,
|
||||
return_frequency="1d",
|
||||
periods_per_year=252,
|
||||
data_snapshot_id="qtdb-pro-20260108-v1",
|
||||
)
|
||||
},
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="same asset labels"):
|
||||
_build(
|
||||
result,
|
||||
risk_snapshots={
|
||||
trade_date: CovarianceSnapshot(
|
||||
snapshot_id="incomplete-universe",
|
||||
as_of_date="2026-01-07",
|
||||
covariance=covariance.loc[["B"], ["B"]],
|
||||
return_frequency="1d",
|
||||
periods_per_year=252,
|
||||
data_snapshot_id="qtdb-pro-20260108-v1",
|
||||
)
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def test_research_artifact_serialization_and_hashes_are_deterministic() -> None:
|
||||
result = _backtest_result()
|
||||
first = _build(result, parameters={"top_k": 1, "lag_sessions": 1})
|
||||
second = _build(result, parameters={"lag_sessions": 1, "top_k": 1})
|
||||
|
||||
assert first.run.loc[0, "config_hash"] == second.run.loc[0, "config_hash"]
|
||||
assert first.content_sha256 == second.content_sha256
|
||||
assert first.manifest() == second.manifest()
|
||||
decoded = json.loads(first.canonical_json())
|
||||
assert decoded["schema_version"] == RESEARCH_ARTIFACT_SCHEMA_VERSION
|
||||
assert decoded["tables"]["nav"][0]["trade_date"] == "2026-01-05"
|
||||
|
||||
leaked_copy = first.nav
|
||||
leaked_copy.loc[0, "nav"] = -999.0
|
||||
assert first.nav.loc[0, "nav"] != -999.0
|
||||
assert first.content_sha256 == second.content_sha256
|
||||
|
||||
|
||||
def test_research_artifact_requires_complete_reproducibility_identity() -> None:
|
||||
result = _backtest_result()
|
||||
|
||||
with pytest.raises(ValueError, match="code_revision"):
|
||||
build_research_run_artifact(
|
||||
result,
|
||||
run_id="run-1",
|
||||
strategy_id="alpha-top1",
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version="1.0.0",
|
||||
engine_version="1.2.0",
|
||||
code_revision="",
|
||||
data_snapshot_id="snapshot-1",
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters={},
|
||||
)
|
||||
|
||||
|
||||
def test_research_artifact_requires_benchmark_identity_and_returns_together() -> None:
|
||||
result = _backtest_result()
|
||||
|
||||
with pytest.raises(ValueError, match="benchmark_id and benchmark_returns"):
|
||||
build_research_run_artifact(
|
||||
result,
|
||||
run_id="run-1",
|
||||
strategy_id="alpha-top1",
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version="1.0.0",
|
||||
engine_version="1.2.0",
|
||||
code_revision="3b1ad07",
|
||||
data_snapshot_id="snapshot-1",
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters={},
|
||||
benchmark_id="000300.SH",
|
||||
)
|
||||
@@ -0,0 +1,118 @@
|
||||
"""Post-execution return attribution contracts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.attribution import DailyReturnAttribution
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.research_pipeline import run_factor_backtest_research
|
||||
|
||||
|
||||
def _zero_cost_config() -> ExecutionConfig:
|
||||
return ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
|
||||
def test_daily_attribution_closes_across_rebalance_and_holding_days() -> None:
|
||||
"""开盘换仓时,隔夜和日内贡献必须来自实际换仓前后持仓。"""
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
scores = pd.DataFrame(
|
||||
{"A": [2.0, 0.0], "B": [1.0, 3.0]},
|
||||
index=dates[:2],
|
||||
)
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=_zero_cost_config(),
|
||||
)
|
||||
|
||||
attribution = result.return_attribution()
|
||||
|
||||
assert isinstance(attribution, DailyReturnAttribution)
|
||||
assert attribution.overnight.loc[dates[2], "A"] == pytest.approx(0.25)
|
||||
assert attribution.intraday.loc[dates[2], "B"] == pytest.approx(-0.125)
|
||||
assert attribution.asset_contributions.loc[dates[3], "B"] == pytest.approx(1 / 6)
|
||||
pd.testing.assert_series_equal(
|
||||
attribution.total_return,
|
||||
result.returns.rename("total_return"),
|
||||
)
|
||||
pd.testing.assert_series_equal(
|
||||
attribution.explained_return + attribution.residual,
|
||||
attribution.total_return,
|
||||
check_names=False,
|
||||
)
|
||||
assert attribution.residual.abs().max() < 1e-12
|
||||
|
||||
|
||||
def test_daily_attribution_reports_execution_cost_separately() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
scores = pd.DataFrame({"A": [1.0]}, index=dates[:1])
|
||||
prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates)
|
||||
config = ExecutionConfig(
|
||||
commission_bps=10,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=10,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
prices,
|
||||
prices,
|
||||
top_k=1,
|
||||
gross_exposure=0.5,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=config,
|
||||
)
|
||||
|
||||
attribution = result.return_attribution()
|
||||
execution = result.execution.daily_executions[1].executions[0]
|
||||
|
||||
assert attribution.asset_contributions.loc[dates[1], "A"] == 0.0
|
||||
assert attribution.transaction_cost.loc[dates[1]] == pytest.approx(
|
||||
-execution.total_cost / 1_000.0
|
||||
)
|
||||
assert attribution.total_return.loc[dates[1]] == pytest.approx(
|
||||
attribution.transaction_cost.loc[dates[1]]
|
||||
)
|
||||
assert attribution.residual.loc[dates[1]] == pytest.approx(0.0, abs=1e-12)
|
||||
|
||||
|
||||
def test_return_attribution_is_empty_for_empty_research_result() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
scores = pd.DataFrame(columns=["A"], index=pd.DatetimeIndex([]), dtype=float)
|
||||
prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
prices,
|
||||
prices,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
)
|
||||
|
||||
attribution = result.return_attribution()
|
||||
|
||||
assert attribution.overnight.empty
|
||||
assert attribution.intraday.empty
|
||||
assert attribution.total_return.empty
|
||||
@@ -0,0 +1,209 @@
|
||||
"""Backtest contract tests for weights, NAV, rebalancing, and benchmarks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.backtest import (
|
||||
BacktestResult,
|
||||
compare_to_benchmark,
|
||||
compute_nav_from_weights,
|
||||
compute_returns_from_nav,
|
||||
rebalance_periodic,
|
||||
run_weight_backtest,
|
||||
weights_to_long_short,
|
||||
)
|
||||
|
||||
|
||||
def test_compute_nav_from_weights_forward_fills_rebalance_weights() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
weights = pd.DataFrame({"A": [0.5], "B": [0.5]}, index=dates[:1])
|
||||
returns = pd.DataFrame({"A": [0.10, 0.00, -0.10], "B": [0.00, 0.10, 0.00]}, index=dates)
|
||||
|
||||
nav = compute_nav_from_weights(weights, returns, initial_capital=100.0)
|
||||
|
||||
expected = pd.Series([105.0, 110.25, 104.7375], index=dates)
|
||||
pd.testing.assert_series_equal(nav, expected)
|
||||
|
||||
|
||||
def test_compute_nav_stays_in_cash_before_first_rebalance() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
weights = pd.DataFrame({"A": [1.0]}, index=dates[1:2])
|
||||
returns = pd.DataFrame({"A": [0.50, 0.10, 0.10]}, index=dates)
|
||||
|
||||
nav = compute_nav_from_weights(weights, returns)
|
||||
|
||||
pd.testing.assert_series_equal(nav, pd.Series([1.0, 1.1, 1.21], index=dates))
|
||||
|
||||
|
||||
def test_compute_nav_ignores_weight_columns_without_returns() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=2, freq="B")
|
||||
weights = pd.DataFrame({"A": [0.5], "MISSING": [0.5]}, index=dates[:1])
|
||||
returns = pd.DataFrame({"A": [0.10, 0.10]}, index=dates)
|
||||
|
||||
nav = compute_nav_from_weights(weights, returns)
|
||||
|
||||
pd.testing.assert_series_equal(nav, pd.Series([1.05, 1.1025], index=dates))
|
||||
|
||||
|
||||
def test_compute_nav_charges_configured_turnover_cost() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=2, freq="B")
|
||||
weights = pd.DataFrame({"A": [1.0]}, index=dates[:1])
|
||||
returns = pd.DataFrame({"A": [0.0, 0.0]}, index=dates)
|
||||
|
||||
nav = compute_nav_from_weights(weights, returns, tc_rate=0.01)
|
||||
|
||||
pd.testing.assert_series_equal(nav, pd.Series([0.995, 0.995], index=dates))
|
||||
|
||||
|
||||
def test_compute_returns_from_nav_preserves_index_and_sets_initial_zero() -> None:
|
||||
nav = pd.Series([100.0, 110.0, 99.0], index=pd.date_range("2026-01-05", periods=3))
|
||||
|
||||
result = compute_returns_from_nav(nav)
|
||||
|
||||
pd.testing.assert_series_equal(result, pd.Series([0.0, 0.1, -0.1], index=nav.index))
|
||||
|
||||
|
||||
def test_rebalance_periodic_maps_weekend_to_previous_trading_day() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=5, freq="B")
|
||||
target = pd.Series({"A": 0.6, "B": 0.4})
|
||||
|
||||
result = rebalance_periodic(target, [pd.Timestamp("2026-01-10")], dates)
|
||||
|
||||
assert result.loc[pd.Timestamp("2026-01-08")].sum() == 0.0
|
||||
pd.testing.assert_series_equal(
|
||||
result.loc[pd.Timestamp("2026-01-09")], target, check_names=False
|
||||
)
|
||||
|
||||
|
||||
def test_rebalance_periodic_accepts_empty_trading_calendar() -> None:
|
||||
target = pd.Series({"A": 1.0})
|
||||
|
||||
result = rebalance_periodic(
|
||||
target,
|
||||
[pd.Timestamp("2026-01-05")],
|
||||
pd.DatetimeIndex([]),
|
||||
)
|
||||
|
||||
assert result.empty
|
||||
assert result.columns.tolist() == ["A"]
|
||||
|
||||
|
||||
def test_weights_to_long_short_allocates_each_leg() -> None:
|
||||
result = weights_to_long_short(["A", "B"], ["C"], long_weight=0.6, short_weight=0.4)
|
||||
|
||||
assert result["A"] == pytest.approx(0.3)
|
||||
assert result["B"] == pytest.approx(0.3)
|
||||
assert result["C"] == pytest.approx(-0.4)
|
||||
assert result.sum() == pytest.approx(0.2)
|
||||
|
||||
|
||||
def test_weights_to_long_short_keeps_explicit_universe() -> None:
|
||||
result = weights_to_long_short(["A"], [], all_tickers=["A", "B"])
|
||||
|
||||
pd.testing.assert_series_equal(result, pd.Series({"A": 0.5, "B": 0.0}))
|
||||
|
||||
|
||||
def test_compare_to_benchmark_returns_report_table() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
strategy = pd.Series([1.0, 1.1, 1.0, 1.2], index=dates)
|
||||
benchmark = pd.Series([1.0, 1.0, 1.05, 1.1], index=dates)
|
||||
|
||||
result = compare_to_benchmark(strategy, benchmark)
|
||||
|
||||
assert result.columns.tolist() == ["策略", "基准"]
|
||||
assert result.loc["n_days", "策略"] == 4
|
||||
assert result.loc["累计收益", "策略"] == pytest.approx(0.2)
|
||||
assert result.loc["累计收益", "基准"] == pytest.approx(0.1)
|
||||
|
||||
|
||||
def test_compare_to_benchmark_rejects_non_overlapping_dates() -> None:
|
||||
strategy = pd.Series([1.0], index=[pd.Timestamp("2026-01-05")])
|
||||
benchmark = pd.Series([1.0], index=[pd.Timestamp("2026-02-05")])
|
||||
|
||||
with pytest.raises(ValueError, match="overlapping dates"):
|
||||
compare_to_benchmark(strategy, benchmark)
|
||||
|
||||
|
||||
# ── 统一回测结果门面 ──────────────────────────────────────
|
||||
|
||||
|
||||
def test_run_weight_backtest_returns_nav_returns_and_input_snapshot() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
weights = pd.DataFrame({"A": [1.0]}, index=dates[:1])
|
||||
stock_returns = pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates)
|
||||
|
||||
result = run_weight_backtest(weights, stock_returns, initial_capital=100.0)
|
||||
|
||||
assert isinstance(result, BacktestResult)
|
||||
pd.testing.assert_series_equal(
|
||||
result.nav,
|
||||
pd.Series([110.0, 99.0, 118.8], index=dates),
|
||||
)
|
||||
pd.testing.assert_series_equal(
|
||||
result.returns,
|
||||
pd.Series([0.0, -0.1, 0.2], index=dates),
|
||||
)
|
||||
pd.testing.assert_frame_equal(result.weights, weights)
|
||||
|
||||
|
||||
def test_backtest_result_stats_reuses_standard_metrics_contract() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
result = run_weight_backtest(
|
||||
pd.DataFrame({"A": [1.0]}, index=dates[:1]),
|
||||
pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates),
|
||||
)
|
||||
|
||||
stats = result.stats(rf=0.02)
|
||||
|
||||
assert stats["n_days"] == 3
|
||||
assert stats["ann_return"] == pytest.approx(
|
||||
(1.0 * 0.9 * 1.2) ** (252 / 3) - 1.0
|
||||
)
|
||||
assert "sharpe" in stats
|
||||
assert stats["drawback"] == stats["max_drawdown"]
|
||||
|
||||
|
||||
def test_backtest_result_builds_benchmark_report() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
benchmark = pd.Series([1.0, 1.05, 1.10], index=dates, name="benchmark")
|
||||
result = run_weight_backtest(
|
||||
pd.DataFrame({"A": [1.0]}, index=dates[:1]),
|
||||
pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates),
|
||||
benchmark_nav=benchmark,
|
||||
)
|
||||
|
||||
report = result.benchmark_report()
|
||||
|
||||
assert report.columns.tolist() == ["策略", "基准"]
|
||||
assert report.loc["累计收益", "基准"] == pytest.approx(0.10)
|
||||
|
||||
|
||||
def test_backtest_result_requires_benchmark_for_comparison() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=2, freq="B")
|
||||
result = run_weight_backtest(
|
||||
pd.DataFrame({"A": [1.0]}, index=dates[:1]),
|
||||
pd.DataFrame({"A": [0.0, 0.0]}, index=dates),
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="benchmark_nav"):
|
||||
result.benchmark_report()
|
||||
|
||||
|
||||
def test_backtest_result_isolated_from_mutated_caller_inputs() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=2, freq="B")
|
||||
weights = pd.DataFrame({"A": [1.0]}, index=dates[:1])
|
||||
benchmark = pd.Series([1.0, 1.1], index=dates)
|
||||
result = run_weight_backtest(
|
||||
weights,
|
||||
pd.DataFrame({"A": [0.0, 0.0]}, index=dates),
|
||||
benchmark_nav=benchmark,
|
||||
)
|
||||
|
||||
weights.iloc[0, 0] = 0.0
|
||||
benchmark.iloc[1] = 99.0
|
||||
|
||||
assert result.weights.iloc[0, 0] == 1.0
|
||||
assert result.benchmark_nav is not None
|
||||
assert result.benchmark_nav.iloc[1] == 1.1
|
||||
@@ -0,0 +1,773 @@
|
||||
"""Backtest run-reference and closed-evidence contract conformance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import replace
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.artifact import (
|
||||
BacktestEvidenceManifest,
|
||||
EvidenceQualification,
|
||||
RESEARCH_ARTIFACT_SCHEMA_VERSION,
|
||||
ResearchRunArtifact,
|
||||
build_backtest_evidence_manifest,
|
||||
build_legacy_backtest_evidence_manifest,
|
||||
build_research_run_artifact,
|
||||
)
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
InputBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
ProducerIdentity,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
)
|
||||
from quant_engine.governed_pipeline import (
|
||||
BacktestContractError,
|
||||
BacktestContractErrorCode,
|
||||
BacktestRun,
|
||||
BacktestRunRef,
|
||||
)
|
||||
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
FACTOR_FIXTURE = ROOT / "tests" / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
BACKTEST_FIXTURE = ROOT / "tests" / "fixtures" / "backtest-evidence-v1.golden.json"
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
CALENDAR_REVISION_ID = "rhcalv1:sha256:1f4ca22557063389badd669cf774bb35234066e847646682dccfe411e252a078"
|
||||
ACTION_REVISION_ID = "rhcav1:sha256:0f947df29f152bfa2c6ab0da7a0d464670c7ad2526cd10f2a7939b42fee5c275"
|
||||
PARAMETERS = {"lag_sessions": 1, "top_k": 1}
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _accepted_authorities(
|
||||
*,
|
||||
factor_evaluation_at: str = "2026-01-03T11:00:00Z",
|
||||
factor_computed_at: str = "2026-01-03T10:15:00Z",
|
||||
factor_artifact_available_at: str = "2026-01-03T10:20:00Z",
|
||||
factor_availability_mode: AvailabilityMode = AvailabilityMode.AS_AVAILABLE,
|
||||
) -> tuple[
|
||||
DatasetSnapshotEnvelope,
|
||||
DataFoundationEnvelope,
|
||||
FactorSetRef,
|
||||
]:
|
||||
fixture = json.loads(FACTOR_FIXTURE.read_text(encoding="utf-8"))
|
||||
snapshot = DatasetSnapshotEnvelope.from_dict(fixture["dataset_snapshot"])
|
||||
foundation = DataFoundationEnvelope.from_dict(fixture["data_foundation"])
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
definition = factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
output_schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
artifact_ref = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
factor_set = FactorSetRef.create(
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
selected_view_ref_ids=(VIEW_REF_ID,),
|
||||
input_bindings=(
|
||||
InputBinding(definition.definition_id, "market", VIEW_REF_ID, VIEW_SCHEMA_DIGEST),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
output_quality=OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
output_coverage=OutputCoverage(
|
||||
"complete", 1, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64
|
||||
),
|
||||
output_schema_bytes=output_schema_bytes,
|
||||
output_content_bytes=output_content_bytes,
|
||||
output_artifact_ref=artifact_ref,
|
||||
availability_mode=factor_availability_mode,
|
||||
evaluation_at=factor_evaluation_at,
|
||||
computed_at=factor_computed_at,
|
||||
artifact_available_at=factor_artifact_available_at,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
actor=ActorIdentity("service", "factor_worker_v1"),
|
||||
correlation_id="research_run_001",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
evidence_scope="synthetic_fixture",
|
||||
decision_eligible=False,
|
||||
)
|
||||
return snapshot, foundation, factor_set
|
||||
|
||||
|
||||
def _config_digest(parameters: dict[str, object] | None = None) -> str:
|
||||
encoded = json.dumps(
|
||||
PARAMETERS if parameters is None else parameters,
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
allow_nan=False,
|
||||
).encode("utf-8")
|
||||
return _sha256(encoded)
|
||||
|
||||
|
||||
def _run_ref(**overrides: Any) -> BacktestRunRef:
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
arguments: dict[str, Any] = {
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"factor_set": factor_set,
|
||||
"universe_digest": "sha256:" + "5" * 64,
|
||||
"trading_calendar_revision_ids": (CALENDAR_REVISION_ID,),
|
||||
"corporate_action_revision_ids": (ACTION_REVISION_ID,),
|
||||
"strategy_id": "alpha-top1",
|
||||
"strategy_version": "1.0.0",
|
||||
"strategy_digest": "sha256:" + "6" * 64,
|
||||
"execution_model_version": "1.0.0",
|
||||
"execution_model_digest": "sha256:" + "7" * 64,
|
||||
"cost_model_version": "1.0.0",
|
||||
"cost_model_digest": "sha256:" + "8" * 64,
|
||||
"random_seed": 7,
|
||||
"code_revision": "d" * 40,
|
||||
"environment_lock_digest": "sha256:" + "9" * 64,
|
||||
"configuration_digest": _config_digest(),
|
||||
"evaluation_at": "2026-01-08T01:00:00Z",
|
||||
"computed_at": "2026-01-08T02:00:00Z",
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return BacktestRunRef.create(**arguments)
|
||||
|
||||
|
||||
def _backtest_result() -> FactorBacktestResult:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
scores = pd.DataFrame({"A": [2.0, 0.0], "B": [1.0, 3.0]}, index=dates[:2])
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]},
|
||||
index=dates,
|
||||
)
|
||||
return run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _artifact(run_ref: BacktestRunRef, *, run_id: str | None = None) -> ResearchRunArtifact:
|
||||
result = _backtest_result()
|
||||
benchmark = pd.Series(
|
||||
[0.0, 0.01, -0.01, 0.02],
|
||||
index=result.returns.index,
|
||||
name="benchmark_return",
|
||||
)
|
||||
return build_research_run_artifact(
|
||||
result,
|
||||
run_id=run_ref.run_id if run_id is None else run_id,
|
||||
strategy_id=run_ref.strategy_id,
|
||||
strategy_name="Alpha Top 1",
|
||||
strategy_version=run_ref.strategy_version,
|
||||
engine_version="1.2.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
data_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
calendar="CN-A",
|
||||
timezone="Asia/Shanghai",
|
||||
started_at="2026-01-08T10:00:00+08:00",
|
||||
finished_at="2026-01-08T10:01:00+08:00",
|
||||
parameters=PARAMETERS,
|
||||
benchmark_id="000300.SH",
|
||||
benchmark_returns=benchmark,
|
||||
)
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[BacktestContractError],
|
||||
code: BacktestContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_backtest_run_ref_is_deterministic_and_binds_only_opaque_authorities() -> None:
|
||||
first = _run_ref()
|
||||
second = _run_ref()
|
||||
|
||||
assert first == second
|
||||
assert first.run_id.startswith("rhbacktestrunv1:sha256:")
|
||||
assert first.replay_spec_digest.startswith("sha256:")
|
||||
assert first.dataset_snapshot_id.startswith("rhdsv1:sha256:")
|
||||
assert first.foundation_id.startswith("rhdfv1:sha256:")
|
||||
assert first.factor_set_id.startswith("rhfactorsetv1:sha256:")
|
||||
assert first.trading_calendar_revision_ids == (CALENDAR_REVISION_ID,)
|
||||
assert first.corporate_action_revision_ids == (ACTION_REVISION_ID,)
|
||||
assert first.replay_parent_run_id is None
|
||||
assert first.replay_attempt == 0
|
||||
assert first.replay_ancestor_run_ids == ()
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
assert BacktestRunRef.from_dict(
|
||||
first.to_dict(),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
) == first
|
||||
forbidden = ("latest", "locator", "uri", "credential", "provider", "broker")
|
||||
assert not any(token in first.to_json().lower() for token in forbidden)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "value"),
|
||||
[
|
||||
("universe_digest", "sha256:" + "a" * 64),
|
||||
("strategy_digest", "sha256:" + "b" * 64),
|
||||
("execution_model_digest", "sha256:" + "c" * 64),
|
||||
("cost_model_digest", "sha256:" + "e" * 64),
|
||||
("random_seed", 8),
|
||||
("code_revision", "e" * 40),
|
||||
("environment_lock_digest", "sha256:" + "f" * 64),
|
||||
("configuration_digest", "sha256:" + "0" * 64),
|
||||
("evaluation_at", "2026-01-08T01:00:01Z"),
|
||||
("computed_at", "2026-01-08T02:00:01Z"),
|
||||
],
|
||||
)
|
||||
def test_every_governed_run_input_mutation_changes_run_identity(
|
||||
field: str,
|
||||
value: object,
|
||||
) -> None:
|
||||
assert _run_ref(**{field: value}).run_id != _run_ref().run_id
|
||||
|
||||
|
||||
def test_backtest_run_ref_rejects_unclosed_upstream_and_unsafe_scalars() -> None:
|
||||
with pytest.raises(BacktestContractError) as wrong_calendar:
|
||||
_run_ref(trading_calendar_revision_ids=())
|
||||
_assert_error(
|
||||
wrong_calendar,
|
||||
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.trading_calendar_revision_ids",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as wrong_action:
|
||||
_run_ref(corporate_action_revision_ids=())
|
||||
_assert_error(
|
||||
wrong_action,
|
||||
BacktestContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.corporate_action_revision_ids",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as bool_seed:
|
||||
_run_ref(random_seed=True)
|
||||
_assert_error(bool_seed, BacktestContractErrorCode.TYPE_ERROR, "$.random_seed")
|
||||
with pytest.raises(BacktestContractError) as bad_revision:
|
||||
_run_ref(code_revision="abc")
|
||||
_assert_error(bad_revision, BacktestContractErrorCode.INVALID_FORMAT, "$.code_revision")
|
||||
with pytest.raises(BacktestContractError) as bad_digest:
|
||||
_run_ref(universe_digest="5" * 64)
|
||||
_assert_error(bad_digest, BacktestContractErrorCode.INVALID_FORMAT, "$.universe_digest")
|
||||
with pytest.raises(BacktestContractError) as lookahead:
|
||||
_run_ref(computed_at="2026-01-08T00:59:59Z")
|
||||
_assert_error(lookahead, BacktestContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
|
||||
with pytest.raises(BacktestContractError) as factor_type:
|
||||
_run_ref(factor_set="rhfactorsetv1:sha256:" + "0" * 64)
|
||||
_assert_error(factor_type, BacktestContractErrorCode.TYPE_ERROR, "$.factor_set")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("factor_times", "expected_path"),
|
||||
[
|
||||
(
|
||||
{
|
||||
"factor_evaluation_at": "2026-01-08T01:00:01Z",
|
||||
"factor_computed_at": "2026-01-08T00:59:59Z",
|
||||
"factor_artifact_available_at": "2026-01-08T01:00:00Z",
|
||||
},
|
||||
"$.evaluation_at",
|
||||
),
|
||||
(
|
||||
{
|
||||
"factor_evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"factor_computed_at": "2026-01-08T01:00:00Z",
|
||||
"factor_artifact_available_at": "2026-01-08T01:00:01Z",
|
||||
"factor_availability_mode": AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
},
|
||||
"$.evaluation_at",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_run_ref_evaluation_closes_factor_pit(
|
||||
factor_times: dict[str, Any],
|
||||
expected_path: str,
|
||||
) -> None:
|
||||
snapshot, foundation, factor_set = _accepted_authorities(**factor_times)
|
||||
|
||||
with pytest.raises(BacktestContractError) as lookahead:
|
||||
_run_ref(
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
)
|
||||
|
||||
_assert_error(
|
||||
lookahead,
|
||||
BacktestContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
expected_path,
|
||||
)
|
||||
|
||||
|
||||
def test_run_ref_rejects_aliases_locators_unsafe_integers_and_invalid_text() -> None:
|
||||
with pytest.raises(BacktestContractError) as mutable_alias:
|
||||
_run_ref(strategy_id="latest")
|
||||
_assert_error(
|
||||
mutable_alias,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as physical_uri:
|
||||
_run_ref(execution_model_version="s3://model-bucket/current")
|
||||
_assert_error(
|
||||
physical_uri,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.execution_model_version",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as unsafe_seed:
|
||||
_run_ref(random_seed=2**53)
|
||||
_assert_error(
|
||||
unsafe_seed,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.random_seed",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as invalid_unicode:
|
||||
_run_ref(strategy_id="\ud800")
|
||||
_assert_error(
|
||||
invalid_unicode,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
|
||||
run_ref = _run_ref()
|
||||
mixed_keys = run_ref.to_dict()
|
||||
mixed_keys[1] = "not-a-contract-key" # type: ignore[index]
|
||||
snapshot, foundation, factor_set = _accepted_authorities()
|
||||
with pytest.raises(BacktestContractError) as invalid_key:
|
||||
BacktestRunRef.from_dict(
|
||||
mixed_keys,
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
factor_set=factor_set,
|
||||
)
|
||||
_assert_error(invalid_key, BacktestContractErrorCode.TYPE_ERROR, "$")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"physical_id",
|
||||
["db.table", "source_alpha", "wind.model", "qtdb_view", "bloomberg-signal"],
|
||||
)
|
||||
def test_run_ref_rejects_physical_terms_in_logical_ids(physical_id: str) -> None:
|
||||
with pytest.raises(BacktestContractError) as physical:
|
||||
_run_ref(strategy_id=physical_id)
|
||||
_assert_error(
|
||||
physical,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_id",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("version", ["1.0.0-.", "1.0.0-foo..bar", "1.0.0-01"])
|
||||
def test_run_ref_requires_strict_semver_prerelease_identifiers(version: str) -> None:
|
||||
with pytest.raises(BacktestContractError) as invalid:
|
||||
_run_ref(strategy_version=version)
|
||||
_assert_error(
|
||||
invalid,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.strategy_version",
|
||||
)
|
||||
|
||||
assert _run_ref(strategy_version="1.0.0-alpha.1").strategy_version == "1.0.0-alpha.1"
|
||||
|
||||
|
||||
def test_replay_lineage_is_acyclic_and_cannot_claim_changed_inputs() -> None:
|
||||
parent = _run_ref()
|
||||
replay = _run_ref(
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="deterministic_reproduction",
|
||||
replay_attempt=1,
|
||||
)
|
||||
|
||||
assert replay.run_id != parent.run_id
|
||||
assert replay.replay_spec_digest == parent.replay_spec_digest
|
||||
assert replay.replay_parent_run_id == parent.run_id
|
||||
assert replay.replay_ancestor_run_ids == (parent.run_id,)
|
||||
|
||||
with pytest.raises(BacktestContractError) as changed_input:
|
||||
_run_ref(
|
||||
universe_digest="sha256:" + "a" * 64,
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="changed_universe",
|
||||
replay_attempt=1,
|
||||
)
|
||||
_assert_error(
|
||||
changed_input,
|
||||
BacktestContractErrorCode.LINEAGE_VIOLATION,
|
||||
"$.replay_spec_digest",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as skipped_attempt:
|
||||
_run_ref(
|
||||
computed_at="2026-01-08T03:00:00Z",
|
||||
parent=parent,
|
||||
replay_reason="skipped_attempt",
|
||||
replay_attempt=2,
|
||||
)
|
||||
_assert_error(
|
||||
skipped_attempt,
|
||||
BacktestContractErrorCode.LINEAGE_VIOLATION,
|
||||
"$.replay_attempt",
|
||||
)
|
||||
|
||||
|
||||
def test_offline_research_manifest_closes_exact_existing_evidence_mapping() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
first = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
second = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
|
||||
assert first == second
|
||||
assert first.manifest_id.startswith("rhbacktestevidencev1:sha256:")
|
||||
assert first.run_id == run_ref.run_id
|
||||
assert first.profile == "offline_research_v1"
|
||||
assert first.qualification is EvidenceQualification.CONTRACT_QUALIFIED
|
||||
mapping = {
|
||||
item.category: tuple(table.logical_name for table in item.tables)
|
||||
for item in first.evidence
|
||||
}
|
||||
assert mapping == {
|
||||
"run": ("run",),
|
||||
"signal": ("signals",),
|
||||
"fill": ("trades",),
|
||||
"position_nav": ("positions", "nav"),
|
||||
"performance": ("performance",),
|
||||
"attribution": ("attribution", "attribution_daily"),
|
||||
"risk_snapshot": ("risk",),
|
||||
"replay": (),
|
||||
}
|
||||
assert "order" not in mapping
|
||||
assert "rejection" not in mapping
|
||||
risk = next(item for item in first.evidence if item.category == "risk_snapshot")
|
||||
assert risk.tables[0].row_count == 0
|
||||
assert risk.tables[0].schema_digest.startswith("sha256:")
|
||||
|
||||
changed_performance = artifact.performance
|
||||
changed_performance.loc[0, "n_days"] += 1
|
||||
changed_artifact = replace(artifact, _performance=changed_performance)
|
||||
changed = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
changed_artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
assert changed.manifest_id != first.manifest_id
|
||||
assert run_ref.run_id == first.run_id == changed.run_id
|
||||
|
||||
|
||||
def test_manifest_rejects_missing_mismatched_or_duplicate_evidence() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
with pytest.raises(BacktestContractError) as wrong_run:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
_artifact(run_ref, run_id="different-run"),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
wrong_run,
|
||||
BacktestContractErrorCode.IDENTITY_MISMATCH,
|
||||
"$.artifact.tables.run.run_id",
|
||||
)
|
||||
missing_signals = replace(artifact, _signals=None) # type: ignore[arg-type]
|
||||
with pytest.raises(BacktestContractError) as missing_table:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
missing_signals,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
missing_table,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.artifact.tables.signals",
|
||||
)
|
||||
with pytest.raises(BacktestContractError) as digest_mismatch:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
expected_table_digests={"performance": "sha256:" + "0" * 64},
|
||||
)
|
||||
_assert_error(
|
||||
digest_mismatch,
|
||||
BacktestContractErrorCode.EVIDENCE_MISMATCH,
|
||||
"$.artifact.tables.performance.content_digest",
|
||||
)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
duplicate = manifest.to_dict()
|
||||
duplicate["evidence"].append(copy.deepcopy(duplicate["evidence"][0]))
|
||||
with pytest.raises(BacktestContractError) as duplicate_category:
|
||||
BacktestEvidenceManifest.from_dict(
|
||||
duplicate,
|
||||
backtest_run_ref=run_ref,
|
||||
artifact=artifact,
|
||||
)
|
||||
_assert_error(
|
||||
duplicate_category,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.evidence[8].category",
|
||||
)
|
||||
|
||||
|
||||
def test_manifest_binds_supported_schema_and_has_collision_free_cell_encoding() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
|
||||
with pytest.raises(BacktestContractError) as unsupported_schema:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, schema_version="999.0.0"),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsupported_schema,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.artifact.schema_version",
|
||||
)
|
||||
|
||||
identities: set[str] = set()
|
||||
for value in (float("nan"), float("inf"), float("-inf")):
|
||||
performance = artifact.performance
|
||||
performance.loc[0, "alpha"] = value
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=performance),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
assert manifest.artifact_schema_version == RESEARCH_ARTIFACT_SCHEMA_VERSION
|
||||
identities.add(manifest.manifest_id)
|
||||
assert len(identities) == 3
|
||||
|
||||
content_digests: set[str] = set()
|
||||
for value in (float("nan"), {"non_finite_float": "nan"}):
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, "alpha"] = value
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=performance),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
performance_entry = next(
|
||||
entry for entry in manifest.evidence if entry.category == "performance"
|
||||
)
|
||||
content_digests.add(performance_entry.tables[0].content_digest)
|
||||
assert len(content_digests) == 2
|
||||
|
||||
unsupported = artifact.performance.astype(object)
|
||||
unsupported.loc[0, "alpha"] = object()
|
||||
with pytest.raises(BacktestContractError) as unsupported_cell:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=unsupported),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsupported_cell,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.artifact.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
invalid_nested_key = artifact.performance.astype(object)
|
||||
invalid_nested_key.at[0, "alpha"] = {"\ud800": "value"}
|
||||
with pytest.raises(BacktestContractError) as invalid_utf8:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=invalid_nested_key),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
invalid_utf8,
|
||||
BacktestContractErrorCode.INVALID_FORMAT,
|
||||
"$.artifact.tables.performance.rows[0].alpha.keys",
|
||||
)
|
||||
|
||||
unsafe_integer = artifact.performance.astype(object)
|
||||
unsafe_integer.loc[0, "alpha"] = 10**5000
|
||||
with pytest.raises(BacktestContractError) as unsafe_cell:
|
||||
build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
replace(artifact, _performance=unsafe_integer),
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
_assert_error(
|
||||
unsafe_cell,
|
||||
BacktestContractErrorCode.INVALID_VALUE,
|
||||
"$.artifact.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
|
||||
def test_artifact_canonical_content_has_typed_collision_free_cell_encoding() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
|
||||
content_hashes: set[str] = set()
|
||||
for value in (
|
||||
float("nan"),
|
||||
float("inf"),
|
||||
float("-inf"),
|
||||
{"non_finite_float": "nan"},
|
||||
):
|
||||
performance = artifact.performance.astype(object)
|
||||
performance.at[0, "alpha"] = value
|
||||
mutated = replace(artifact, _performance=performance)
|
||||
content_hashes.add(mutated.content_sha256)
|
||||
assert "non_finite_float" in mutated.canonical_json()
|
||||
assert len(content_hashes) == 4
|
||||
|
||||
unsupported = artifact.performance.astype(object)
|
||||
unsupported.loc[0, "alpha"] = object()
|
||||
with pytest.raises(BacktestContractError) as unsupported_cell:
|
||||
replace(artifact, _performance=unsupported).canonical_json()
|
||||
_assert_error(
|
||||
unsupported_cell,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.tables.performance.rows[0].alpha",
|
||||
)
|
||||
|
||||
|
||||
def test_legacy_bridge_is_explicit_and_cannot_be_contract_qualified() -> None:
|
||||
run_ref = _run_ref()
|
||||
legacy_run = BacktestRun(
|
||||
run_id="legacy-run-001",
|
||||
dataset_snapshot_id=run_ref.dataset_snapshot_id,
|
||||
factor_version_id="alpha_005@1.0.0",
|
||||
strategy_version_id="alpha-top1@1.0.0",
|
||||
code_revision=run_ref.code_revision,
|
||||
config_hash=_config_digest().removeprefix("sha256:"),
|
||||
created_at=datetime(2026, 1, 8, 2, 0, tzinfo=UTC),
|
||||
)
|
||||
artifact = _artifact(run_ref, run_id=legacy_run.run_id)
|
||||
manifest = build_legacy_backtest_evidence_manifest(
|
||||
legacy_run,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
|
||||
assert manifest.qualification is EvidenceQualification.LEGACY_EXPLORATORY
|
||||
assert manifest.run_id == legacy_run.run_id
|
||||
assert manifest.backtest_run_ref is None
|
||||
assert manifest.to_dict()["run_reference"]["kind"] == "legacy_backtest_run"
|
||||
assert BacktestEvidenceManifest.from_dict(
|
||||
manifest.to_dict(),
|
||||
artifact=artifact,
|
||||
) == manifest
|
||||
with pytest.raises(BacktestContractError) as implicit_promotion:
|
||||
build_backtest_evidence_manifest( # type: ignore[arg-type]
|
||||
legacy_run,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
qualification=EvidenceQualification.CONTRACT_QUALIFIED,
|
||||
)
|
||||
_assert_error(
|
||||
implicit_promotion,
|
||||
BacktestContractErrorCode.TYPE_ERROR,
|
||||
"$.backtest_run_ref",
|
||||
)
|
||||
|
||||
|
||||
def test_golden_contract_and_architecture_boundary() -> None:
|
||||
run_ref = _run_ref()
|
||||
artifact = _artifact(run_ref)
|
||||
manifest = build_backtest_evidence_manifest(
|
||||
run_ref,
|
||||
artifact,
|
||||
artifact_available_at="2026-01-08T02:05:00Z",
|
||||
)
|
||||
golden = json.loads(BACKTEST_FIXTURE.read_text(encoding="utf-8"))
|
||||
|
||||
table_digests = {
|
||||
table.logical_name: table.content_digest
|
||||
for item in manifest.evidence
|
||||
for table in item.tables
|
||||
}
|
||||
assert golden == {
|
||||
"run_id": run_ref.run_id,
|
||||
"replay_spec_digest": run_ref.replay_spec_digest,
|
||||
"manifest_id": manifest.manifest_id,
|
||||
"evidence_digest": manifest.evidence_digest,
|
||||
"table_content_digests": table_digests,
|
||||
}
|
||||
governed_source = (ROOT / "src" / "quant_engine" / "governed_pipeline.py").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
artifact_source = (ROOT / "src" / "quant_engine" / "artifact.py").read_text(
|
||||
encoding="utf-8"
|
||||
)
|
||||
assert "from quant_engine.artifact" not in governed_source
|
||||
assert "BacktestRunRef" in governed_source
|
||||
assert "BacktestEvidenceManifest" not in governed_source
|
||||
assert "BacktestEvidenceManifest" in artifact_source
|
||||
assert not (ROOT / "src" / "quant_engine" / "backtest_contracts.py").exists()
|
||||
@@ -7,10 +7,12 @@ import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.data_adapter import (
|
||||
AssetReturnSnapshot,
|
||||
add_vwap_proxy,
|
||||
apply_adj_factor,
|
||||
load_qtdb_daily,
|
||||
long_to_wide,
|
||||
prepare_asset_return_snapshot,
|
||||
prepare_execution_inputs,
|
||||
prepare_stock_series,
|
||||
rename_tushare_columns,
|
||||
@@ -266,6 +268,17 @@ def test_prepare_execution_inputs_basic(tushare_long: pd.DataFrame) -> None:
|
||||
assert volumes.iloc[0, 0] == pytest.approx(1000.0)
|
||||
|
||||
|
||||
def test_prepare_execution_inputs_can_select_next_session_open_price(
|
||||
tushare_long: pd.DataFrame,
|
||||
) -> None:
|
||||
"""显式 price_col=open 时应生成开盘执行价矩阵。"""
|
||||
renamed = rename_tushare_columns(tushare_long)
|
||||
|
||||
prices, _volumes = prepare_execution_inputs(renamed, price_col="open")
|
||||
|
||||
assert prices.iloc[0, 0] == pytest.approx(10.0)
|
||||
|
||||
|
||||
def test_prepare_execution_inputs_no_volume() -> None:
|
||||
"""无 volume 列 → volumes 全 1.0。"""
|
||||
df = pd.DataFrame(
|
||||
@@ -286,6 +299,177 @@ def test_prepare_execution_inputs_missing_close_raises() -> None:
|
||||
prepare_execution_inputs(df)
|
||||
|
||||
|
||||
def test_prepare_execution_inputs_missing_selected_price_raises() -> None:
|
||||
df = pd.DataFrame({"stock_code": ["A"], "trade_date": ["2024-01-01"], "close": [10.0]})
|
||||
with pytest.raises(ValueError, match="缺 open"):
|
||||
prepare_execution_inputs(df, price_col="open")
|
||||
|
||||
|
||||
# ── prepare_asset_return_snapshot ────────────────────────────
|
||||
|
||||
|
||||
def _daily_prices() -> pd.DataFrame:
|
||||
return pd.DataFrame(
|
||||
{
|
||||
"stock_code": ["B", "A", "B", "A", "B", "A"],
|
||||
"trade_date": [
|
||||
"2024-01-02",
|
||||
"2024-01-01",
|
||||
"2024-01-01",
|
||||
"2024-01-03",
|
||||
"2024-01-03",
|
||||
"2024-01-02",
|
||||
],
|
||||
"close": [18.0, 10.0, 20.0, 12.1, 19.8, 11.0],
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_prepare_asset_return_snapshot_is_stable_and_immutable_by_interface() -> None:
|
||||
snapshot = prepare_asset_return_snapshot(
|
||||
_daily_prices(),
|
||||
source="qtdb_pro.hq_daily",
|
||||
source_snapshot_id="hq-daily:2024-01-03:v1",
|
||||
adjustment="qfq",
|
||||
)
|
||||
|
||||
assert isinstance(snapshot, AssetReturnSnapshot)
|
||||
assert snapshot.data_snapshot_id.startswith("asset-returns-v1:")
|
||||
assert snapshot.source == "qtdb_pro.hq_daily"
|
||||
assert snapshot.source_snapshot_id == "hq-daily:2024-01-03:v1"
|
||||
assert snapshot.price_field == "close"
|
||||
assert snapshot.adjustment == "qfq"
|
||||
assert snapshot.return_method == "simple"
|
||||
assert snapshot.start_date.isoformat() == "2024-01-01"
|
||||
assert snapshot.end_date.isoformat() == "2024-01-03"
|
||||
assert snapshot.sessions == 3
|
||||
assert snapshot.assets == ("A", "B")
|
||||
|
||||
expected = pd.DataFrame(
|
||||
{
|
||||
"A": [np.nan, 0.1, 0.1],
|
||||
"B": [np.nan, -0.1, 0.1],
|
||||
},
|
||||
index=pd.to_datetime(["2024-01-01", "2024-01-02", "2024-01-03"]),
|
||||
)
|
||||
expected.index.name = "trade_date"
|
||||
expected.columns.name = "stock_code"
|
||||
pd.testing.assert_frame_equal(snapshot.returns, expected)
|
||||
|
||||
exposed = snapshot.returns
|
||||
exposed.iloc[1, 0] = 999.0
|
||||
assert snapshot.returns.iloc[1, 0] == pytest.approx(0.1)
|
||||
|
||||
|
||||
def test_asset_return_snapshot_identity_is_order_independent_and_content_addressed() -> None:
|
||||
kwargs = {
|
||||
"source": "qtdb_pro.hq_daily",
|
||||
"source_snapshot_id": "hq-daily:2024-01-03:v1",
|
||||
"adjustment": "none",
|
||||
}
|
||||
baseline = prepare_asset_return_snapshot(_daily_prices(), **kwargs)
|
||||
shuffled = prepare_asset_return_snapshot(
|
||||
_daily_prices().sample(frac=1.0, random_state=7),
|
||||
**kwargs,
|
||||
)
|
||||
changed_prices = _daily_prices().copy()
|
||||
changed_prices.loc[changed_prices["close"] == 12.1, "close"] = 12.2
|
||||
changed_content = prepare_asset_return_snapshot(changed_prices, **kwargs)
|
||||
changed_source = prepare_asset_return_snapshot(
|
||||
_daily_prices(),
|
||||
source="qtdb_pro.hq_daily",
|
||||
source_snapshot_id="hq-daily:2024-01-03:v2",
|
||||
adjustment="none",
|
||||
)
|
||||
|
||||
assert shuffled.data_snapshot_id == baseline.data_snapshot_id
|
||||
assert changed_content.data_snapshot_id != baseline.data_snapshot_id
|
||||
assert changed_source.data_snapshot_id != baseline.data_snapshot_id
|
||||
|
||||
|
||||
def test_prepare_asset_return_snapshot_does_not_fill_missing_prices() -> None:
|
||||
prices = _daily_prices()
|
||||
prices.loc[
|
||||
(prices["stock_code"] == "A") & (prices["trade_date"] == "2024-01-02"),
|
||||
"close",
|
||||
] = np.nan
|
||||
|
||||
snapshot = prepare_asset_return_snapshot(
|
||||
prices,
|
||||
source="qtdb_pro.hq_daily",
|
||||
source_snapshot_id="hq-daily:missing-middle",
|
||||
)
|
||||
|
||||
assert pd.isna(snapshot.returns.loc[pd.Timestamp("2024-01-02"), "A"])
|
||||
assert pd.isna(snapshot.returns.loc[pd.Timestamp("2024-01-03"), "A"])
|
||||
|
||||
|
||||
def test_prepare_asset_return_snapshot_rejects_duplicate_sessions() -> None:
|
||||
duplicate = pd.concat([_daily_prices(), _daily_prices().iloc[[0]]], ignore_index=True)
|
||||
|
||||
with pytest.raises(ValueError, match="duplicate"):
|
||||
prepare_asset_return_snapshot(
|
||||
duplicate,
|
||||
source="qtdb_pro.hq_daily",
|
||||
source_snapshot_id="hq-daily:duplicate",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("invalid_price", [0.0, -1.0, np.inf])
|
||||
def test_prepare_asset_return_snapshot_rejects_invalid_prices(invalid_price: float) -> None:
|
||||
prices = _daily_prices()
|
||||
prices.loc[0, "close"] = invalid_price
|
||||
|
||||
with pytest.raises(ValueError, match="positive finite"):
|
||||
prepare_asset_return_snapshot(
|
||||
prices,
|
||||
source="qtdb_pro.hq_daily",
|
||||
source_snapshot_id="hq-daily:invalid-price",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("source", "source_snapshot_id", "adjustment"),
|
||||
[
|
||||
("", "source-1", "none"),
|
||||
("qtdb_pro.hq_daily", "", "none"),
|
||||
("qtdb_pro.hq_daily", "source-1", ""),
|
||||
],
|
||||
)
|
||||
def test_prepare_asset_return_snapshot_requires_explicit_identity_semantics(
|
||||
source: str,
|
||||
source_snapshot_id: str,
|
||||
adjustment: str,
|
||||
) -> None:
|
||||
with pytest.raises(ValueError, match="must be non-empty"):
|
||||
prepare_asset_return_snapshot(
|
||||
_daily_prices(),
|
||||
source=source,
|
||||
source_snapshot_id=source_snapshot_id,
|
||||
adjustment=adjustment,
|
||||
)
|
||||
|
||||
|
||||
def test_asset_return_snapshot_feeds_reproducible_covariance_lineage() -> None:
|
||||
from quant_engine.risk import estimate_covariance_snapshot
|
||||
|
||||
market_snapshot = prepare_asset_return_snapshot(
|
||||
_daily_prices(),
|
||||
source="qtdb_pro.hq_daily",
|
||||
source_snapshot_id="hq-daily:2024-01-03:v1",
|
||||
)
|
||||
covariance_snapshot = estimate_covariance_snapshot(
|
||||
market_snapshot.returns,
|
||||
as_of_date=market_snapshot.end_date,
|
||||
lookback_sessions=3,
|
||||
min_observations=2,
|
||||
data_snapshot_id=market_snapshot.data_snapshot_id,
|
||||
)
|
||||
|
||||
assert covariance_snapshot.data_snapshot_id == market_snapshot.data_snapshot_id
|
||||
assert covariance_snapshot.snapshot_id.startswith("sample-cov-v1:")
|
||||
|
||||
|
||||
# ── 端到端:长表 → 适配 → alpha158 + execution ──────────────
|
||||
|
||||
|
||||
|
||||
+307
-14
@@ -11,6 +11,7 @@ import pytest
|
||||
from quant_engine.execution import (
|
||||
ExecutionConfig,
|
||||
ExecutionResult,
|
||||
ExecutionSimulationResult,
|
||||
apply_bid_ask_spread,
|
||||
apply_volume_constraint,
|
||||
check_price_limit,
|
||||
@@ -19,7 +20,9 @@ from quant_engine.execution import (
|
||||
compute_realized_pnl,
|
||||
run_end_to_end_poc,
|
||||
simulate_execution,
|
||||
simulate_daily_ledger_with_audit,
|
||||
simulate_multi_day,
|
||||
simulate_multi_day_with_audit,
|
||||
simulate_with_daily_data,
|
||||
total_costs,
|
||||
total_turnover,
|
||||
@@ -327,24 +330,26 @@ def test_simulate_multi_day_length_mismatch_raises():
|
||||
|
||||
|
||||
def test_simulate_multi_day_first_day_value_equals_initial():
|
||||
"""第一天 portfolio_value = initial_cash(无持仓)。"""
|
||||
"""零成本下第一天日末 NAV 等于初始资金。"""
|
||||
signals = [("d1", {"A": 1.0})]
|
||||
prices = [("d1", {"A": 10.0})]
|
||||
positions = simulate_multi_day(signals, prices, 1_000_000.0)
|
||||
# 第一天 NAV = 1_000_000(无持仓),第二天才是调仓后
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
positions = simulate_multi_day(signals, prices, 1_000_000.0, config)
|
||||
assert positions[0].portfolio_value == 1_000_000.0
|
||||
assert positions[0].holdings == {"A": 100_000.0}
|
||||
|
||||
|
||||
def test_simulate_multi_day_holdings_evolution():
|
||||
"""调仓后 holdings 演化。
|
||||
|
||||
注意:positions[i] 是第 i 天 rebalance 之前的快照。
|
||||
所以要看 d2 rebalance 后的 holdings,需要看 positions[2](d3 的快照)。
|
||||
"""
|
||||
"""日末快照应反映当天调仓后的 holdings。"""
|
||||
signals = [
|
||||
("d1", {"A": 0.5, "B": 0.5}),
|
||||
("d2", {"A": 1.0, "B": 0.0}), # 全仓 A
|
||||
("d3", {"A": 1.0, "B": 0.0}), # 第三天的快照才能看到 d2 rebalance 后的 holdings
|
||||
("d3", {"A": 1.0, "B": 0.0}),
|
||||
]
|
||||
prices = [
|
||||
("d1", {"A": 10.0, "B": 20.0}),
|
||||
@@ -352,9 +357,288 @@ def test_simulate_multi_day_holdings_evolution():
|
||||
("d3", {"A": 12.0, "B": 22.0}),
|
||||
]
|
||||
positions = simulate_multi_day(signals, prices, 1_000_000.0)
|
||||
# d3 的 PRE-trade snapshot 应该只有 A(B 在 d2 被平仓)
|
||||
assert "B" not in positions[2].holdings
|
||||
assert "A" in positions[2].holdings
|
||||
assert "B" not in positions[1].holdings
|
||||
assert "A" in positions[1].holdings
|
||||
|
||||
|
||||
def test_simulate_multi_day_with_audit_rebalances_target_weights_by_delta():
|
||||
"""相同目标权重不应在每个交易日重复买入。"""
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
targets = [(date, {"A": 1.0}) for date in ("d1", "d2", "d3")]
|
||||
prices = [(date, {"A": 10.0}) for date in ("d1", "d2", "d3")]
|
||||
|
||||
result = simulate_multi_day_with_audit(targets, prices, 1_000.0, config)
|
||||
|
||||
assert isinstance(result, ExecutionSimulationResult)
|
||||
assert [len(day.executions) for day in result.daily_executions] == [1, 0, 0]
|
||||
assert result.total_turnover == pytest.approx(1_000.0)
|
||||
assert [position.cash for position in result.positions] == pytest.approx([0.0, 0.0, 0.0])
|
||||
assert [position.holdings["A"] for position in result.positions] == pytest.approx(
|
||||
[100.0, 100.0, 100.0]
|
||||
)
|
||||
assert [position.portfolio_value for position in result.positions] == pytest.approx(
|
||||
[1_000.0, 1_000.0, 1_000.0]
|
||||
)
|
||||
|
||||
|
||||
def test_simulate_multi_day_with_audit_records_costs_without_replay():
|
||||
"""成交成本与日末 NAV 应来自同一次状态推进。"""
|
||||
targets = [("d1", {"A": 1.0}), ("d2", {"A": 1.0})]
|
||||
prices = [("d1", {"A": 10.0}), ("d2", {"A": 10.0})]
|
||||
|
||||
result = simulate_multi_day_with_audit(targets, prices, 1_000.0)
|
||||
|
||||
first_day = result.daily_executions[0]
|
||||
assert first_day.nav_before == pytest.approx(1_000.0)
|
||||
assert first_day.nav_after == pytest.approx(result.positions[0].portfolio_value)
|
||||
assert result.total_costs == pytest.approx(sum(r.total_cost for r in first_day.executions))
|
||||
assert result.final_portfolio_value == pytest.approx(1_000.0 - result.total_costs)
|
||||
assert result.daily_executions[1].executions == ()
|
||||
|
||||
|
||||
def test_simulate_multi_day_with_audit_never_spends_more_cash_than_available():
|
||||
"""满仓目标应按可用现金部分成交,不能用负现金隐式加杠杆。"""
|
||||
result = simulate_multi_day_with_audit(
|
||||
[("d1", {"A": 1.0})],
|
||||
[("d1", {"A": 10.0})],
|
||||
1_000.0,
|
||||
)
|
||||
|
||||
execution = result.daily_executions[0].executions[0]
|
||||
assert result.positions[0].cash >= -1e-9
|
||||
assert 0 < execution.partial_fill_pct < 1
|
||||
assert execution.blocked_reason == "insufficient_cash_partial_fill"
|
||||
assert result.final_portfolio_value == pytest.approx(1_000.0 - result.total_costs)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"targets",
|
||||
[
|
||||
{"A": -0.1},
|
||||
{"A": 0.6, "B": 0.5},
|
||||
{"A": float("nan")},
|
||||
],
|
||||
)
|
||||
def test_simulate_multi_day_with_audit_rejects_invalid_long_only_weights(targets):
|
||||
"""多日 A 股目标必须是有限、非负且合计不超过 100% 的权重。"""
|
||||
with pytest.raises(ValueError, match="target weights"):
|
||||
simulate_multi_day_with_audit(
|
||||
[("d1", targets)],
|
||||
[("d1", {"A": 10.0, "B": 10.0})],
|
||||
1_000.0,
|
||||
)
|
||||
|
||||
|
||||
def test_simulate_multi_day_with_audit_requires_price_for_existing_holding():
|
||||
"""已有持仓缺价时无法可信估值,必须失败而不是把市值记为零。"""
|
||||
with pytest.raises(ValueError, match="missing price for held asset A"):
|
||||
simulate_multi_day_with_audit(
|
||||
[("d1", {"A": 1.0}), ("d2", {"A": 1.0})],
|
||||
[("d1", {"A": 10.0}), ("d2", {})],
|
||||
1_000.0,
|
||||
)
|
||||
|
||||
|
||||
def test_simulate_multi_day_with_audit_records_unpriced_target_rejection():
|
||||
"""缺失价格的目标不能吞掉现金,且必须留下拒绝原因。"""
|
||||
result = simulate_multi_day_with_audit(
|
||||
[("d1", {"A": 1.0})],
|
||||
[("d1", {"B": 10.0})],
|
||||
1_000.0,
|
||||
)
|
||||
|
||||
rejection = result.daily_executions[0].executions[0]
|
||||
assert rejection.stock_code == "A"
|
||||
assert rejection.executed_value == 0.0
|
||||
assert rejection.partial_fill_pct == 0.0
|
||||
assert rejection.blocked_reason == "missing_price"
|
||||
assert result.positions[0].cash == 1_000.0
|
||||
assert result.positions[0].holdings == {}
|
||||
|
||||
|
||||
def test_simulate_multi_day_with_audit_requires_matching_dates():
|
||||
"""权重与价格日期错位必须显式失败,不能按位置静默配对。"""
|
||||
with pytest.raises(ValueError, match="dates must match"):
|
||||
simulate_multi_day_with_audit(
|
||||
[("d1", {"A": 1.0})],
|
||||
[("d2", {"A": 10.0})],
|
||||
1_000.0,
|
||||
)
|
||||
|
||||
|
||||
# ── 逐交易日 Ledger:成交时点与估值时点分离 ─────────────────
|
||||
|
||||
|
||||
def test_daily_ledger_marks_every_session_after_sparse_open_execution() -> None:
|
||||
"""下一日开盘成交后,应按每日收盘价持续盯市,而非只记录调仓日。"""
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
result = simulate_daily_ledger_with_audit(
|
||||
target_weights_history=[("d1", {"A": 1.0})],
|
||||
execution_price_history=[("d1", {"A": 10.0})],
|
||||
valuation_price_history=[
|
||||
("d0", {"A": 9.0}),
|
||||
("d1", {"A": 11.0}),
|
||||
("d2", {"A": 12.0}),
|
||||
],
|
||||
initial_cash=1_000.0,
|
||||
config=config,
|
||||
)
|
||||
|
||||
assert [position.date for position in result.positions] == ["d0", "d1", "d2"]
|
||||
assert [position.portfolio_value for position in result.positions] == pytest.approx(
|
||||
[1_000.0, 1_100.0, 1_200.0]
|
||||
)
|
||||
assert [len(day.executions) for day in result.daily_executions] == [0, 1, 0]
|
||||
fill = result.daily_executions[1].executions[0]
|
||||
assert fill.side == "buy"
|
||||
assert fill.quantity == pytest.approx(100.0)
|
||||
assert fill.price == pytest.approx(10.0)
|
||||
pd.testing.assert_series_equal(
|
||||
result.normalized_nav_series,
|
||||
pd.Series([1.0, 1.1, 1.2], index=["d0", "d1", "d2"], dtype=float),
|
||||
)
|
||||
pd.testing.assert_series_equal(
|
||||
result.daily_returns,
|
||||
pd.Series([0.0, 0.1, 1.2 / 1.1 - 1.0], index=["d0", "d1", "d2"]),
|
||||
)
|
||||
|
||||
|
||||
def test_daily_ledger_first_session_cost_reduces_first_return() -> None:
|
||||
"""首个估值日发生交易时,费用必须进入相对初始资金的首日收益。"""
|
||||
result = simulate_daily_ledger_with_audit(
|
||||
target_weights_history=[("d0", {"A": 1.0})],
|
||||
execution_price_history=[("d0", {"A": 10.0})],
|
||||
valuation_price_history=[("d0", {"A": 10.0})],
|
||||
initial_cash=1_000.0,
|
||||
)
|
||||
|
||||
assert result.total_costs > 0
|
||||
assert result.daily_returns.iloc[0] == pytest.approx(
|
||||
result.final_portfolio_value / result.initial_cash - 1.0
|
||||
)
|
||||
assert result.daily_returns.iloc[0] < 0
|
||||
|
||||
|
||||
def test_daily_ledger_nav_is_rebuildable_and_trades_are_projectable() -> None:
|
||||
"""Ledger 必须同时支持现金守恒校验和平台成交表投影。"""
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
result = simulate_daily_ledger_with_audit(
|
||||
target_weights_history=[
|
||||
("d1", {"A": 1.0, "B": 0.0}),
|
||||
("d2", {"A": 0.0, "B": 1.0}),
|
||||
],
|
||||
execution_price_history=[
|
||||
("d1", {"A": 10.0, "B": 20.0}),
|
||||
("d2", {"A": 11.0, "B": 22.0}),
|
||||
],
|
||||
valuation_price_history=[
|
||||
("d0", {"A": 9.0, "B": 19.0}),
|
||||
("d1", {"A": 10.5, "B": 21.0}),
|
||||
("d2", {"A": 12.0, "B": 24.0}),
|
||||
],
|
||||
initial_cash=1_000.0,
|
||||
config=config,
|
||||
)
|
||||
|
||||
close_prices = {
|
||||
"d0": {"A": 9.0, "B": 19.0},
|
||||
"d1": {"A": 10.5, "B": 21.0},
|
||||
"d2": {"A": 12.0, "B": 24.0},
|
||||
}
|
||||
for position in result.positions:
|
||||
rebuilt = position.cash + sum(
|
||||
shares * close_prices[position.date][asset]
|
||||
for asset, shares in position.holdings.items()
|
||||
)
|
||||
assert position.portfolio_value == pytest.approx(rebuilt)
|
||||
|
||||
trades = result.trades_frame
|
||||
assert trades.columns.tolist() == [
|
||||
"trade_date",
|
||||
"ts_code",
|
||||
"side",
|
||||
"qty",
|
||||
"price",
|
||||
"amount",
|
||||
"fee",
|
||||
"slippage",
|
||||
]
|
||||
assert trades["side"].tolist() == ["buy", "sell", "buy"]
|
||||
assert (trades["qty"] > 0).all()
|
||||
|
||||
|
||||
def test_daily_ledger_frame_matches_platform_projection_contract() -> None:
|
||||
"""核心层输出稳定日频投影,但不携带 run_id 或执行数据库写入。"""
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
result = simulate_daily_ledger_with_audit(
|
||||
target_weights_history=[("d1", {"A": 1.0})],
|
||||
execution_price_history=[("d1", {"A": 10.0})],
|
||||
valuation_price_history=[
|
||||
("d0", {"A": 9.0}),
|
||||
("d1", {"A": 11.0}),
|
||||
("d2", {"A": 12.0}),
|
||||
],
|
||||
initial_cash=1_000.0,
|
||||
config=config,
|
||||
)
|
||||
|
||||
ledger = result.ledger_frame
|
||||
|
||||
assert ledger.columns.tolist() == [
|
||||
"trade_date",
|
||||
"portfolio_value",
|
||||
"nav",
|
||||
"pnl",
|
||||
"pnl_pct",
|
||||
"position_value",
|
||||
"cash",
|
||||
"turnover",
|
||||
]
|
||||
assert ledger["trade_date"].tolist() == ["d0", "d1", "d2"]
|
||||
assert ledger["nav"].tolist() == pytest.approx([1.0, 1.1, 1.2])
|
||||
assert ledger["pnl"].tolist() == pytest.approx([0.0, 100.0, 100.0])
|
||||
assert ledger["pnl_pct"].tolist() == pytest.approx([0.0, 0.1, 1.2 / 1.1 - 1.0])
|
||||
assert ledger["position_value"].tolist() == pytest.approx([0.0, 1_100.0, 1_200.0])
|
||||
assert ledger["cash"].tolist() == pytest.approx([1_000.0, 0.0, 0.0])
|
||||
assert ledger["turnover"].tolist() == pytest.approx([0.0, 1.0, 0.0])
|
||||
|
||||
|
||||
def test_daily_ledger_rejects_missing_close_for_held_asset() -> None:
|
||||
"""已有持仓缺少收盘估值价时必须 fail closed。"""
|
||||
with pytest.raises(ValueError, match="missing valuation price for held asset A"):
|
||||
simulate_daily_ledger_with_audit(
|
||||
target_weights_history=[("d0", {"A": 1.0})],
|
||||
execution_price_history=[("d0", {"A": 10.0})],
|
||||
valuation_price_history=[("d0", {"A": 10.0}), ("d1", {})],
|
||||
initial_cash=1_000.0,
|
||||
)
|
||||
|
||||
|
||||
def test_daily_ledger_requires_positive_initial_cash() -> None:
|
||||
"""可信收益曲线需要正初始资金作为归一化基准。"""
|
||||
with pytest.raises(ValueError, match="initial_cash must be positive"):
|
||||
simulate_daily_ledger_with_audit([], [], [], initial_cash=0.0)
|
||||
|
||||
|
||||
# ── v1.2.0 Phase 1:端到端 POC(run_end_to_end_poc) ─────
|
||||
@@ -449,6 +733,15 @@ def test_run_end_to_end_poc_costs_recorded():
|
||||
result = run_end_to_end_poc(signals, prices, 1_000_000.0)
|
||||
assert result["total_costs"] > 0
|
||||
assert result["total_turnover"] > 0
|
||||
executions = [
|
||||
execution
|
||||
for daily in result["daily_executions"]
|
||||
for execution in daily.executions
|
||||
]
|
||||
assert result["total_costs"] == pytest.approx(sum(item.total_cost for item in executions))
|
||||
assert result["total_turnover"] == pytest.approx(
|
||||
sum(item.executed_value for item in executions)
|
||||
)
|
||||
|
||||
|
||||
# ── v1.2.0 Phase 2: T+1 / 涨跌停 / 部分成交 / 买卖价差 ─────
|
||||
@@ -741,8 +1034,8 @@ def test_compute_realized_pnl_sell_realizes():
|
||||
target_weights_history=targets,
|
||||
)
|
||||
pnl_list = compute_realized_pnl(positions)
|
||||
# 第三天(卖出兑现)应有 realized 正利润(cash 从 -800 → 2M = +2M)
|
||||
assert pnl_list[2].realized_pnl > 0
|
||||
# 第二天日末快照已包含当日卖出,现金流入应在当天反映。
|
||||
assert pnl_list[1].realized_pnl > 0
|
||||
|
||||
|
||||
# ── O3: end-to-end 端到端测试(集成多个函数) ──────────────
|
||||
|
||||
@@ -0,0 +1,938 @@
|
||||
"""Versioned factor-definition and factor-set contract conformance."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import FrozenInstanceError
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_contracts import (
|
||||
ActorIdentity,
|
||||
AvailabilityMode,
|
||||
ContractErrorCode,
|
||||
Causation,
|
||||
DataFoundationEnvelope,
|
||||
DatasetSnapshotEnvelope,
|
||||
FactorContractError,
|
||||
FactorDefinition,
|
||||
FactorInput,
|
||||
FactorSetRef,
|
||||
HistoricalAvailability,
|
||||
InputBinding,
|
||||
LegacyFactorBinding,
|
||||
OutputArtifactRef,
|
||||
OutputCoverage,
|
||||
OutputQuality,
|
||||
OutputQualityCheck,
|
||||
PayloadValidation,
|
||||
ProducerIdentity,
|
||||
TypedParameter,
|
||||
ViewAvailability,
|
||||
canonical_json_bytes,
|
||||
factor_definition_from_alpha158,
|
||||
factor_input_schema_digest,
|
||||
validate_factor_catalog,
|
||||
)
|
||||
from quant_engine.governed_pipeline import (
|
||||
FactorVersion,
|
||||
bind_legacy_factor,
|
||||
project_legacy_factor,
|
||||
)
|
||||
|
||||
|
||||
FIXTURE_PATH = Path(__file__).parent / "fixtures" / "factor-contracts-v1.golden.json"
|
||||
VIEW_REF_ID = "rhviewrefv1:sha256:bf776bcd26d940fafde1d650776a5505fb3fe8b5b068c351622bf2c42385629c"
|
||||
VIEW_SCHEMA_DIGEST = "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"
|
||||
|
||||
|
||||
def _golden() -> dict[str, Any]:
|
||||
loaded = json.loads(FIXTURE_PATH.read_text(encoding="utf-8"))
|
||||
assert isinstance(loaded, dict)
|
||||
return loaded
|
||||
|
||||
|
||||
def _sha256(value: bytes) -> str:
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()}"
|
||||
|
||||
|
||||
def _reidentify(item: dict[str, Any], field: str, prefix: str) -> None:
|
||||
payload = {key: value for key, value in item.items() if key != field}
|
||||
item[field] = f"{prefix}{hashlib.sha256(canonical_json_bytes(payload)).hexdigest()}"
|
||||
|
||||
|
||||
def _snapshot_and_foundation(
|
||||
fixture: dict[str, Any] | None = None,
|
||||
) -> tuple[DatasetSnapshotEnvelope, DataFoundationEnvelope]:
|
||||
source = _golden() if fixture is None else fixture
|
||||
return (
|
||||
DatasetSnapshotEnvelope.from_dict(source["dataset_snapshot"]),
|
||||
DataFoundationEnvelope.from_dict(source["data_foundation"]),
|
||||
)
|
||||
|
||||
|
||||
def _definition(
|
||||
*,
|
||||
inputs: tuple[FactorInput, ...] | None = None,
|
||||
**overrides: Any,
|
||||
) -> FactorDefinition:
|
||||
factor_inputs = inputs or (
|
||||
FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume")),
|
||||
)
|
||||
arguments: dict[str, Any] = {
|
||||
"factor_id": "alpha_005",
|
||||
"version": "1.0.0",
|
||||
"formula": "correlation(close, volume, 10)",
|
||||
"parameters": {},
|
||||
"implementation_digest": "sha256:" + "1" * 64,
|
||||
"input_schema_digest": factor_input_schema_digest(factor_inputs),
|
||||
"inputs": factor_inputs,
|
||||
"valid_from": "2026-01-01T00:00:00.000000Z",
|
||||
"valid_until": "2027-01-01T00:00:00Z",
|
||||
"warmup_sessions": 10,
|
||||
"lag_sessions": 1,
|
||||
"producer": ProducerIdentity("quant_engine", "1.0.0"),
|
||||
"code_revision": "c" * 40,
|
||||
}
|
||||
arguments.update(overrides)
|
||||
return FactorDefinition.create(**arguments)
|
||||
|
||||
|
||||
def _golden_definition() -> FactorDefinition:
|
||||
factor_input = FactorInput("market", VIEW_SCHEMA_DIGEST, ("close", "volume"))
|
||||
return factor_definition_from_alpha158(
|
||||
"alpha_005",
|
||||
version="1.0.0",
|
||||
parameters={},
|
||||
inputs=(factor_input,),
|
||||
implementation_digest="sha256:" + "1" * 64,
|
||||
input_schema_digest=factor_input_schema_digest((factor_input,)),
|
||||
valid_from="2026-01-01T00:00:00.000000Z",
|
||||
valid_until="2027-01-01T00:00:00Z",
|
||||
warmup_sessions=10,
|
||||
lag_sessions=1,
|
||||
producer=ProducerIdentity("quant_engine", "1.0.0"),
|
||||
code_revision="c" * 40,
|
||||
)
|
||||
|
||||
|
||||
def _factor_set_arguments(
|
||||
*,
|
||||
fixture: dict[str, Any] | None = None,
|
||||
snapshot: DatasetSnapshotEnvelope | None = None,
|
||||
foundation: DataFoundationEnvelope | None = None,
|
||||
definition: FactorDefinition | None = None,
|
||||
) -> dict[str, Any]:
|
||||
source = _golden() if fixture is None else fixture
|
||||
if snapshot is None or foundation is None:
|
||||
parsed_snapshot, parsed_foundation = _snapshot_and_foundation(source)
|
||||
snapshot = snapshot or parsed_snapshot
|
||||
foundation = foundation or parsed_foundation
|
||||
selected_definition = definition or _golden_definition()
|
||||
output_schema_bytes = canonical_json_bytes(source["output_schema"])
|
||||
output_content_bytes = canonical_json_bytes(source["output_content"])
|
||||
artifact = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(output_schema_bytes),
|
||||
content_digest=_sha256(output_content_bytes),
|
||||
)
|
||||
return {
|
||||
"definitions": (selected_definition,),
|
||||
"dataset_snapshot": snapshot,
|
||||
"foundation": foundation,
|
||||
"selected_view_ref_ids": (VIEW_REF_ID,),
|
||||
"input_bindings": (
|
||||
InputBinding(
|
||||
selected_definition.definition_id,
|
||||
"market",
|
||||
VIEW_REF_ID,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
"view_availability": (
|
||||
ViewAvailability(VIEW_REF_ID, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
"output_quality": OutputQuality(
|
||||
"passed",
|
||||
(OutputQualityCheck("finite_values", "passed", "sha256:" + "3" * 64),),
|
||||
),
|
||||
"output_coverage": OutputCoverage(
|
||||
"complete",
|
||||
1,
|
||||
1,
|
||||
"row",
|
||||
"alpha_005.cn_a",
|
||||
"sha256:" + "4" * 64,
|
||||
),
|
||||
"output_schema_bytes": output_schema_bytes,
|
||||
"output_content_bytes": output_content_bytes,
|
||||
"output_artifact_ref": artifact,
|
||||
"availability_mode": AvailabilityMode.AS_AVAILABLE,
|
||||
"evaluation_at": "2026-01-03T11:00:00Z",
|
||||
"computed_at": "2026-01-03T10:15:00Z",
|
||||
"artifact_available_at": "2026-01-03T10:20:00Z",
|
||||
"producer": ProducerIdentity("quant_engine", "1.0.0"),
|
||||
"code_revision": "c" * 40,
|
||||
"actor": ActorIdentity("service", "factor_worker_v1"),
|
||||
"correlation_id": "research_run_001",
|
||||
"causation": Causation("foundation", foundation.foundation_id),
|
||||
"evidence_scope": "synthetic_fixture",
|
||||
"decision_eligible": False,
|
||||
}
|
||||
|
||||
|
||||
def _factor_set(**overrides: Any) -> FactorSetRef:
|
||||
arguments = _factor_set_arguments()
|
||||
arguments.update(overrides)
|
||||
return FactorSetRef.create(**arguments)
|
||||
|
||||
|
||||
def _assert_error(
|
||||
error: pytest.ExceptionInfo[FactorContractError],
|
||||
code: ContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
assert error.value.code is code
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def _mutate_artifact_schema_binding(value: dict[str, Any]) -> None:
|
||||
artifact = value["output_artifact_ref"]
|
||||
artifact["schema_digest"] = "sha256:" + "0" * 64
|
||||
_reidentify(artifact, "artifact_id", "rhfactoroutputv1:sha256:")
|
||||
|
||||
|
||||
def test_golden_contracts_are_content_addressed_round_trippable_and_deeply_immutable() -> None:
|
||||
fixture = _golden()
|
||||
original_snapshot = copy.deepcopy(fixture["dataset_snapshot"])
|
||||
original_foundation = copy.deepcopy(fixture["data_foundation"])
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
definition = _golden_definition()
|
||||
factor_set = FactorSetRef.create(**_factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition))
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id="factor:demo-momentum",
|
||||
legacy_version="1.0.0",
|
||||
legacy_definition_sha256="b" * 64,
|
||||
legacy_dataset_schema_version="1.0.0",
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
|
||||
assert snapshot.pit_cutoff == "2026-01-02T07:01:00Z"
|
||||
assert foundation.pit_cutoff == factor_set.pit_cutoff == "2026-01-03T00:00:00Z"
|
||||
assert snapshot.pit_cutoff != foundation.pit_cutoff
|
||||
assert definition.definition_id == fixture["expected"]["definition_id"]
|
||||
assert definition.input_schema_digest == fixture["expected"]["input_schema_digest"]
|
||||
assert factor_set.factor_set_id == fixture["expected"]["factor_set_id"]
|
||||
assert factor_set.output_artifact_ref.artifact_id == fixture["expected"]["output_artifact_id"]
|
||||
assert binding.binding_id == fixture["expected"]["legacy_binding_id"]
|
||||
assert not definition.to_json().endswith("\n")
|
||||
assert not factor_set.to_json().endswith("\n")
|
||||
assert FactorDefinition.from_json(definition.to_json()) == definition
|
||||
|
||||
reparsed = FactorSetRef.from_json(
|
||||
factor_set.to_json(),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
output_schema_bytes=canonical_json_bytes(fixture["output_schema"]),
|
||||
output_content_bytes=canonical_json_bytes(fixture["output_content"]),
|
||||
)
|
||||
reference_only = FactorSetRef.from_json(
|
||||
factor_set.to_json(),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
assert reparsed.factor_set_id == factor_set.factor_set_id
|
||||
assert reparsed.payload_validation is PayloadValidation.PAYLOAD_REVALIDATED
|
||||
assert reference_only.payload_validation is PayloadValidation.REFERENCE_ONLY
|
||||
|
||||
fixture["dataset_snapshot"]["descriptor"]["dataset"]["dimensions"].append("forbidden")
|
||||
fixture["data_foundation"]["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
|
||||
assert snapshot.to_dict() == original_snapshot
|
||||
assert foundation.to_dict() == original_foundation
|
||||
returned = snapshot.to_dict()
|
||||
returned["descriptor"]["dataset"]["dimensions"].append("also_forbidden")
|
||||
assert snapshot.to_dict() == original_snapshot
|
||||
with pytest.raises(FrozenInstanceError):
|
||||
snapshot.snapshot_id = "rhdsv1:sha256:" + "0" * 64 # type: ignore[misc]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("variant", ["whitespace", "key_order"])
|
||||
def test_contract_decoders_reject_non_canonical_json(variant: str) -> None:
|
||||
fixture = _golden()
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
definition = _golden_definition()
|
||||
factor_set = FactorSetRef.create(
|
||||
**_factor_set_arguments(
|
||||
fixture=fixture,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
definition=definition,
|
||||
)
|
||||
)
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id="factor:demo-momentum",
|
||||
legacy_version="1.0.0",
|
||||
legacy_definition_sha256="b" * 64,
|
||||
legacy_dataset_schema_version="1.0.0",
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
|
||||
def non_canonical(value: str) -> str:
|
||||
if variant == "whitespace":
|
||||
return value + "\n"
|
||||
loaded = json.loads(value)
|
||||
reversed_items = dict(reversed(tuple(loaded.items())))
|
||||
return json.dumps(reversed_items, ensure_ascii=False, separators=(",", ":"))
|
||||
|
||||
decoders = (
|
||||
lambda value: FactorDefinition.from_json(value),
|
||||
lambda value: FactorSetRef.from_json(
|
||||
value,
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
),
|
||||
lambda value: LegacyFactorBinding.from_json(value, definition=definition),
|
||||
)
|
||||
for decoder, encoded in zip(
|
||||
decoders,
|
||||
(definition.to_json(), factor_set.to_json(), binding.to_json()),
|
||||
strict=True,
|
||||
):
|
||||
with pytest.raises(FactorContractError) as exc_info:
|
||||
decoder(non_canonical(encoded))
|
||||
assert exc_info.value.code is ContractErrorCode.INVALID_FORMAT
|
||||
assert exc_info.value.path == "$"
|
||||
|
||||
|
||||
def test_factor_definition_identity_is_order_independent_where_semantics_are_unordered() -> None:
|
||||
first_input = FactorInput("prices", "sha256:" + "6" * 64, ("close",))
|
||||
second_input = FactorInput("volumes", "sha256:" + "7" * 64, ("volume",))
|
||||
inputs = (first_input, second_input)
|
||||
parameters_a = {
|
||||
"window": TypedParameter("integer", 10),
|
||||
"weights": TypedParameter("json", {"fast": [1, 2], "slow": [3, 4]}),
|
||||
}
|
||||
parameters_b = {
|
||||
"weights": TypedParameter("json", {"slow": [3, 4], "fast": [1, 2]}),
|
||||
"window": TypedParameter("integer", 10),
|
||||
}
|
||||
first = _definition(
|
||||
inputs=inputs,
|
||||
parameters=parameters_a,
|
||||
input_schema_digest=factor_input_schema_digest(inputs),
|
||||
)
|
||||
second = _definition(
|
||||
inputs=tuple(reversed(inputs)),
|
||||
parameters=parameters_b,
|
||||
input_schema_digest=factor_input_schema_digest(tuple(reversed(inputs))),
|
||||
)
|
||||
assert first.definition_id == second.definition_id
|
||||
assert first.to_json() == second.to_json()
|
||||
|
||||
semantic_changes = (
|
||||
_definition(factor_id="alpha_006"),
|
||||
_definition(version="1.0.1"),
|
||||
_definition(formula="correlation(close, volume, 11)"),
|
||||
_definition(parameters={"window": TypedParameter("integer", 10)}),
|
||||
_definition(implementation_digest="sha256:" + "9" * 64),
|
||||
_definition(valid_until="2027-01-02T00:00:00Z"),
|
||||
_definition(warmup_sessions=11),
|
||||
_definition(lag_sessions=2),
|
||||
_definition(producer=ProducerIdentity("quant_engine", "1.0.1")),
|
||||
_definition(code_revision="d" * 40),
|
||||
)
|
||||
assert all(changed.definition_id != _golden_definition().definition_id for changed in semantic_changes)
|
||||
assert len({changed.definition_id for changed in semantic_changes}) == len(semantic_changes)
|
||||
|
||||
|
||||
def test_parameter_types_decimal_profile_and_detached_nested_values_are_strict() -> None:
|
||||
nested = {"ordered": [1, {"flag": True}]}
|
||||
parameter = TypedParameter("json", nested)
|
||||
nested["ordered"].append(2)
|
||||
definition = _definition(parameters={"payload": parameter})
|
||||
assert definition.to_dict()["parameters"]["payload"]["value"] == {
|
||||
"ordered": [1, {"flag": True}]
|
||||
}
|
||||
integer_definition = _definition(parameters={"value": TypedParameter("integer", 1)})
|
||||
string_definition = _definition(parameters={"value": TypedParameter("string", "1")})
|
||||
assert integer_definition.definition_id != string_definition.definition_id
|
||||
|
||||
for parameter_type, value, code in (
|
||||
("decimal", "1.0", ContractErrorCode.INVALID_FORMAT),
|
||||
("decimal", "1e3", ContractErrorCode.INVALID_FORMAT),
|
||||
("decimal", "-0", ContractErrorCode.INVALID_FORMAT),
|
||||
("integer", True, ContractErrorCode.TYPE_ERROR),
|
||||
("json", 1.5, ContractErrorCode.TYPE_ERROR),
|
||||
("json", {"é": "bad-key"}, ContractErrorCode.INVALID_FORMAT),
|
||||
("json", 9_007_199_254_740_992, ContractErrorCode.INVALID_VALUE),
|
||||
):
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
TypedParameter(parameter_type, value)
|
||||
assert error.value.code is code
|
||||
assert TypedParameter("decimal", "10.25").to_dict()["value"] == "10.25"
|
||||
|
||||
|
||||
def test_catalog_rejects_duplicate_and_overlapping_logical_validity_but_allows_adjacency() -> None:
|
||||
base = _golden_definition()
|
||||
adjacent = _definition(valid_from="2027-01-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
|
||||
assert len(validate_factor_catalog((adjacent, base))) == 2
|
||||
with pytest.raises(FactorContractError) as duplicate:
|
||||
validate_factor_catalog((base, base))
|
||||
_assert_error(duplicate, ContractErrorCode.INVALID_VALUE, "$.definitions")
|
||||
overlapping = _definition(valid_from="2026-06-01T00:00:00Z", valid_until="2028-01-01T00:00:00Z")
|
||||
with pytest.raises(FactorContractError) as overlap:
|
||||
validate_factor_catalog((base, overlapping))
|
||||
_assert_error(overlap, ContractErrorCode.TIME_ORDER_VIOLATION, "$.definitions")
|
||||
|
||||
|
||||
def test_upstream_contracts_reject_unknown_fields_identity_forgery_and_unqualified_input() -> None:
|
||||
unknown = _golden()["dataset_snapshot"]
|
||||
unknown["provider"] = "forbidden"
|
||||
with pytest.raises(FactorContractError) as unknown_error:
|
||||
DatasetSnapshotEnvelope.from_dict(unknown)
|
||||
_assert_error(unknown_error, ContractErrorCode.UNKNOWN_FIELD, "$.provider")
|
||||
|
||||
forged = _golden()["data_foundation"]
|
||||
forged["standardized_views"][0]["schema_digest"] = "sha256:" + "0" * 64
|
||||
with pytest.raises(FactorContractError) as forged_error:
|
||||
DataFoundationEnvelope.from_dict(forged)
|
||||
assert forged_error.value.code is ContractErrorCode.IDENTITY_MISMATCH
|
||||
assert forged_error.value.path.endswith("view_ref_id")
|
||||
|
||||
rejected_source = _golden()
|
||||
rejected_source["dataset_snapshot"]["descriptor"]["qualification"]["status"] = "rejected"
|
||||
_reidentify(rejected_source["dataset_snapshot"], "snapshot_id", "rhdsv1:sha256:")
|
||||
rejected_snapshot = DatasetSnapshotEnvelope.from_dict(rejected_source["dataset_snapshot"])
|
||||
_, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as rejected_error:
|
||||
FactorSetRef.create(
|
||||
**_factor_set_arguments(snapshot=rejected_snapshot, foundation=foundation)
|
||||
)
|
||||
_assert_error(
|
||||
rejected_error,
|
||||
ContractErrorCode.QUALIFICATION_REJECTED,
|
||||
"$.dataset_snapshot.descriptor.qualification",
|
||||
)
|
||||
|
||||
|
||||
def test_foundation_rejects_future_knowledge_and_per_view_calendar_borrowing() -> None:
|
||||
future = _golden()["data_foundation"]
|
||||
action = future["corporate_action_revisions"][0]
|
||||
old_action_id = action["action_revision_id"]
|
||||
action["knowledge_time"] = "2026-01-03T00:00:01Z"
|
||||
_reidentify(action, "action_revision_id", "rhcav1:sha256:")
|
||||
future["standardized_views"][0]["corporate_action_revision_ids"] = [action["action_revision_id"]]
|
||||
lineage = next(item for item in future["revision_lineage"] if item["revision_id"] == old_action_id)
|
||||
lineage["revision_id"] = action["action_revision_id"]
|
||||
lineage["knowledge_time"] = action["knowledge_time"]
|
||||
_reidentify(future["standardized_views"][0], "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(future, "foundation_id", "rhdfv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as future_error:
|
||||
DataFoundationEnvelope.from_dict(future)
|
||||
_assert_error(
|
||||
future_error,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
"$.revision_lineage.knowledge_time",
|
||||
)
|
||||
|
||||
uncovered = _golden()["data_foundation"]
|
||||
original_route_id = uncovered["instrument_routes"][0]["route_revision_id"]
|
||||
second_calendar = copy.deepcopy(uncovered["trading_calendar_revisions"][0])
|
||||
second_calendar["calendar_id"] = "rhcalendar:99990000111122223333444455556666"
|
||||
_reidentify(second_calendar, "calendar_revision_id", "rhcalv1:sha256:")
|
||||
uncovered["trading_calendar_revisions"].append(second_calendar)
|
||||
route = uncovered["instrument_routes"][0]
|
||||
route["calendar_id"] = second_calendar["calendar_id"]
|
||||
_reidentify(route, "route_revision_id", "rhroutev1:sha256:")
|
||||
route_lineage = next(item for item in uncovered["revision_lineage"] if item["revision_id"] == original_route_id)
|
||||
route_lineage["revision_id"] = route["route_revision_id"]
|
||||
uncovered["revision_lineage"].append(
|
||||
{
|
||||
"revision_kind": "trading_calendar",
|
||||
"revision_id": second_calendar["calendar_revision_id"],
|
||||
"revision_number": 1,
|
||||
"knowledge_time": second_calendar["knowledge_time"],
|
||||
"evidence_digest": second_calendar["evidence_digest"],
|
||||
}
|
||||
)
|
||||
view = uncovered["standardized_views"][0]
|
||||
view["instrument_route_revision_ids"] = [route["route_revision_id"]]
|
||||
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(uncovered, "foundation_id", "rhdfv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as calendar_error:
|
||||
DataFoundationEnvelope.from_dict(uncovered)
|
||||
assert calendar_error.value.code is ContractErrorCode.INPUT_CLOSURE_VIOLATION
|
||||
assert "selected route calendar" in calendar_error.value.detail
|
||||
|
||||
|
||||
def _replay_fixture() -> dict[str, Any]:
|
||||
fixture = _golden()
|
||||
snapshot = fixture["dataset_snapshot"]
|
||||
snapshot["descriptor"]["published_at"] = "2026-01-04T00:00:00Z"
|
||||
_reidentify(snapshot, "snapshot_id", "rhdsv1:sha256:")
|
||||
foundation = fixture["data_foundation"]
|
||||
foundation["dataset_snapshot_id"] = snapshot["snapshot_id"]
|
||||
for view in foundation["standardized_views"]:
|
||||
view["dataset_snapshot_id"] = snapshot["snapshot_id"]
|
||||
_reidentify(view, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
|
||||
return fixture
|
||||
|
||||
|
||||
def test_as_available_and_retrospective_replay_keep_distinct_time_claims() -> None:
|
||||
as_available = _factor_set()
|
||||
assert as_available.historical_availability is HistoricalAvailability.DECLARED_AS_AVAILABLE
|
||||
|
||||
replay_source = _replay_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(replay_source)
|
||||
replay_view_id = next(iter(foundation.views))
|
||||
arguments = _factor_set_arguments(
|
||||
fixture=replay_source,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
arguments.update(
|
||||
selected_view_ref_ids=(replay_view_id,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
arguments["definitions"][0].definition_id,
|
||||
"market",
|
||||
replay_view_id,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(replay_view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
computed_at="2026-01-04T00:20:00Z",
|
||||
artifact_available_at="2026-01-04T00:25:00Z",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
replay = FactorSetRef.create(**arguments)
|
||||
assert replay.evaluation_at == "2026-01-03T11:00:00Z"
|
||||
assert replay.computed_at == "2026-01-04T00:20:00Z"
|
||||
assert replay.historical_availability is HistoricalAvailability.NOT_ESTABLISHED
|
||||
|
||||
replay_source_args = _factor_set_arguments(
|
||||
fixture=replay_source,
|
||||
snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
replay_source_args.update(
|
||||
selected_view_ref_ids=(replay_view_id,),
|
||||
input_bindings=(
|
||||
InputBinding(
|
||||
replay_source_args["definitions"][0].definition_id,
|
||||
"market",
|
||||
replay_view_id,
|
||||
VIEW_SCHEMA_DIGEST,
|
||||
),
|
||||
),
|
||||
view_availability=(
|
||||
ViewAvailability(replay_view_id, "2026-01-02T23:50:00Z", "sha256:" + "2" * 64),
|
||||
),
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
with pytest.raises(FactorContractError) as late_publication:
|
||||
FactorSetRef.create(**replay_source_args)
|
||||
_assert_error(
|
||||
late_publication,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
"$.dataset_snapshot.descriptor.published_at",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("overrides", "path"),
|
||||
[
|
||||
({"view_availability": (ViewAvailability(VIEW_REF_ID, "2026-01-03T00:00:01Z", "sha256:" + "2" * 64),)}, "$.view_availability"),
|
||||
({"computed_at": "2026-01-02T23:40:00Z"}, "$.computed_at"),
|
||||
({"artifact_available_at": "2026-01-03T10:14:00Z"}, "$.artifact_available_at"),
|
||||
({"artifact_available_at": "2026-01-03T11:00:01Z"}, "$.artifact_available_at"),
|
||||
({"evaluation_at": "2026-01-03T11:00:00"}, "$.evaluation_at"),
|
||||
],
|
||||
)
|
||||
def test_as_available_time_failures_are_typed(overrides: dict[str, Any], path: str) -> None:
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
_factor_set(**overrides)
|
||||
assert error.value.code in {
|
||||
ContractErrorCode.INVALID_FORMAT,
|
||||
ContractErrorCode.TIME_ORDER_VIOLATION,
|
||||
}
|
||||
assert error.value.path == path
|
||||
|
||||
|
||||
def test_replay_rejects_backdating_and_historical_availability_promotion() -> None:
|
||||
source = _replay_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(source)
|
||||
view_id = next(iter(foundation.views))
|
||||
arguments = _factor_set_arguments(fixture=source, snapshot=snapshot, foundation=foundation)
|
||||
definition = arguments["definitions"][0]
|
||||
arguments.update(
|
||||
selected_view_ref_ids=(view_id,),
|
||||
input_bindings=(InputBinding(definition.definition_id, "market", view_id, VIEW_SCHEMA_DIGEST),),
|
||||
view_availability=(ViewAvailability(view_id, "2026-01-04T00:10:00Z", "sha256:" + "2" * 64),),
|
||||
availability_mode=AvailabilityMode.RETROSPECTIVE_REPLAY,
|
||||
computed_at="2026-01-04T00:20:00Z",
|
||||
artifact_available_at="2026-01-04T00:25:00Z",
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
replay = FactorSetRef.create(**arguments)
|
||||
promoted = replay.to_dict()
|
||||
promoted["historical_availability"] = "declared_as_available"
|
||||
_reidentify(promoted, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as promotion_error:
|
||||
FactorSetRef.from_dict(
|
||||
promoted,
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
promotion_error,
|
||||
ContractErrorCode.READINESS_ESCALATION,
|
||||
"$.historical_availability",
|
||||
)
|
||||
arguments["computed_at"] = "2026-01-03T11:30:00Z"
|
||||
with pytest.raises(FactorContractError) as backdated_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(backdated_error, ContractErrorCode.TIME_ORDER_VIOLATION, "$.computed_at")
|
||||
|
||||
|
||||
def _multi_view_fixture() -> tuple[dict[str, Any], str]:
|
||||
fixture = _golden()
|
||||
foundation = fixture["data_foundation"]
|
||||
second = copy.deepcopy(foundation["standardized_views"][0])
|
||||
second["view_id"] = "rhview:11111111222222223333333344444444"
|
||||
second["schema_digest"] = "sha256:" + "6" * 64
|
||||
second["content_digest"] = "sha256:" + "7" * 64
|
||||
second["transformation_digest"] = "sha256:" + "8" * 64
|
||||
_reidentify(second, "view_ref_id", "rhviewrefv1:sha256:")
|
||||
foundation["standardized_views"].append(second)
|
||||
_reidentify(foundation, "foundation_id", "rhdfv1:sha256:")
|
||||
return fixture, second["view_ref_id"]
|
||||
|
||||
|
||||
def test_multi_input_mapping_requires_exact_consumption_closure_and_is_order_independent() -> None:
|
||||
fixture, second_view_id = _multi_view_fixture()
|
||||
snapshot, foundation = _snapshot_and_foundation(fixture)
|
||||
inputs = (
|
||||
FactorInput("prices", VIEW_SCHEMA_DIGEST, ("close",)),
|
||||
FactorInput("volumes", "sha256:" + "6" * 64, ("volume",)),
|
||||
)
|
||||
definition = _definition(
|
||||
inputs=inputs,
|
||||
formula="correlation(close, volume, 10)",
|
||||
input_schema_digest=factor_input_schema_digest(inputs),
|
||||
)
|
||||
first_binding = InputBinding(definition.definition_id, "prices", VIEW_REF_ID, VIEW_SCHEMA_DIGEST)
|
||||
second_binding = InputBinding(definition.definition_id, "volumes", second_view_id, "sha256:" + "6" * 64)
|
||||
first_availability = ViewAvailability(VIEW_REF_ID, "2026-01-02T23:40:00Z", "sha256:" + "2" * 64)
|
||||
second_availability = ViewAvailability(second_view_id, "2026-01-02T23:50:00Z", "sha256:" + "6" * 64)
|
||||
base = _factor_set_arguments(fixture=fixture, snapshot=snapshot, foundation=foundation, definition=definition)
|
||||
base.update(
|
||||
selected_view_ref_ids=(VIEW_REF_ID, second_view_id),
|
||||
input_bindings=(first_binding, second_binding),
|
||||
view_availability=(first_availability, second_availability),
|
||||
causation=Causation("foundation", foundation.foundation_id),
|
||||
)
|
||||
first = FactorSetRef.create(**base)
|
||||
reordered = dict(base)
|
||||
reordered.update(
|
||||
selected_view_ref_ids=(second_view_id, VIEW_REF_ID),
|
||||
input_bindings=(second_binding, first_binding),
|
||||
view_availability=(second_availability, first_availability),
|
||||
)
|
||||
assert FactorSetRef.create(**reordered).factor_set_id == first.factor_set_id
|
||||
|
||||
for invalid_bindings, invalid_views in (
|
||||
((first_binding,), (VIEW_REF_ID, second_view_id)),
|
||||
((first_binding, second_binding), (VIEW_REF_ID,)),
|
||||
((first_binding, second_binding), (VIEW_REF_ID, second_view_id, VIEW_REF_ID)),
|
||||
):
|
||||
invalid = dict(base)
|
||||
invalid.update(input_bindings=invalid_bindings, selected_view_ref_ids=invalid_views)
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.create(**invalid)
|
||||
assert error.value.code in {
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
ContractErrorCode.INVALID_VALUE,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("mutate", "code", "path"),
|
||||
[
|
||||
(lambda value: value["producer"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.producer.id"),
|
||||
(lambda value: value["producer"].pop("version"), ContractErrorCode.MISSING_FIELD, "$.producer.version"),
|
||||
(lambda value: value["producer"].update(id="other_engine"), ContractErrorCode.LINEAGE_VIOLATION, "$.producer.id"),
|
||||
(lambda value: value["producer"].update(version="latest"), ContractErrorCode.INVALID_FORMAT, "$.producer.version"),
|
||||
(lambda value: value.update(code_revision="bad"), ContractErrorCode.INVALID_FORMAT, "$.code_revision"),
|
||||
(lambda value: value["actor"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.actor.kind"),
|
||||
(lambda value: value["actor"].pop("id"), ContractErrorCode.MISSING_FIELD, "$.actor.id"),
|
||||
(lambda value: value["actor"].update(kind="robot"), ContractErrorCode.INVALID_VALUE, "$.actor.kind"),
|
||||
(lambda value: value["actor"].update(id="latest"), ContractErrorCode.INVALID_VALUE, "$.actor.id"),
|
||||
(lambda value: value.pop("correlation_id"), ContractErrorCode.MISSING_FIELD, "$.correlation_id"),
|
||||
(lambda value: value.update(correlation_id="latest"), ContractErrorCode.INVALID_VALUE, "$.correlation_id"),
|
||||
(lambda value: value["causation"].pop("kind"), ContractErrorCode.MISSING_FIELD, "$.causation.kind"),
|
||||
(lambda value: value["causation"].update(kind="run"), ContractErrorCode.INVALID_VALUE, "$.causation.kind"),
|
||||
(lambda value: value["causation"].update(id="rhdfv1:sha256:" + "0" * 64), ContractErrorCode.LINEAGE_VIOLATION, "$.causation.id"),
|
||||
(lambda value: value.pop("output_artifact_ref"), ContractErrorCode.MISSING_FIELD, "$.output_artifact_ref"),
|
||||
(lambda value: value["output_artifact_ref"].update(artifact_id="rhfactoroutputv1:sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.output_artifact_ref.artifact_id"),
|
||||
(_mutate_artifact_schema_binding, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref"),
|
||||
(lambda value: value.update(availability_mode="implicit_fallback"), ContractErrorCode.INVALID_VALUE, "$.availability_mode"),
|
||||
(lambda value: value.pop("computed_at"), ContractErrorCode.MISSING_FIELD, "$.computed_at"),
|
||||
(lambda value: value.update(decision_eligible=True), ContractErrorCode.READINESS_ESCALATION, "$.decision_eligible"),
|
||||
(lambda value: value.update(evidence_scope="real_data"), ContractErrorCode.READINESS_ESCALATION, "$.evidence_scope"),
|
||||
(lambda value: value["upstream_evidence"].update(qualification_evidence_digest="sha256:" + "0" * 64), ContractErrorCode.IDENTITY_MISMATCH, "$.upstream_evidence"),
|
||||
],
|
||||
)
|
||||
def test_lineage_artifact_and_readiness_fields_have_independent_typed_negatives(
|
||||
mutate: Callable[[dict[str, Any]], Any],
|
||||
code: ContractErrorCode,
|
||||
path: str,
|
||||
) -> None:
|
||||
factor_set = _factor_set()
|
||||
value = factor_set.to_dict()
|
||||
mutate(value)
|
||||
if "factor_set_id" in value:
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(error, code, path)
|
||||
|
||||
|
||||
def test_output_schema_content_bytes_cannot_be_swapped_or_forged() -> None:
|
||||
factor_set = _factor_set()
|
||||
fixture = _golden()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
schema_bytes = canonical_json_bytes(fixture["output_schema"])
|
||||
content_bytes = canonical_json_bytes(fixture["output_content"])
|
||||
with pytest.raises(FactorContractError) as swapped:
|
||||
FactorSetRef.from_dict(
|
||||
factor_set.to_dict(),
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
output_schema_bytes=content_bytes,
|
||||
output_content_bytes=schema_bytes,
|
||||
)
|
||||
_assert_error(swapped, ContractErrorCode.ARTIFACT_MISMATCH, "$.output_artifact_ref")
|
||||
with pytest.raises(FactorContractError) as noncanonical:
|
||||
FactorSetRef.create(
|
||||
**{
|
||||
**_factor_set_arguments(),
|
||||
"output_schema_bytes": json.dumps(fixture["output_schema"], indent=2).encode(),
|
||||
}
|
||||
)
|
||||
_assert_error(noncanonical, ContractErrorCode.INVALID_FORMAT, "$.output_schema_bytes")
|
||||
|
||||
|
||||
def test_unsuccessful_output_quality_or_coverage_cannot_form_a_factor_set() -> None:
|
||||
with pytest.raises(FactorContractError) as failed_quality:
|
||||
_factor_set(
|
||||
output_quality=OutputQuality(
|
||||
"failed",
|
||||
(OutputQualityCheck("finite_values", "failed", "sha256:" + "3" * 64),),
|
||||
)
|
||||
)
|
||||
_assert_error(failed_quality, ContractErrorCode.INVALID_VALUE, "$.output_quality")
|
||||
|
||||
for coverage in (
|
||||
OutputCoverage("incomplete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
|
||||
OutputCoverage("complete", 2, 1, "row", "alpha_005.cn_a", "sha256:" + "4" * 64),
|
||||
):
|
||||
with pytest.raises(FactorContractError) as incomplete:
|
||||
_factor_set(output_coverage=coverage)
|
||||
_assert_error(incomplete, ContractErrorCode.INVALID_VALUE, "$.output_coverage")
|
||||
|
||||
|
||||
def test_external_snapshot_definition_and_view_references_cannot_be_substituted() -> None:
|
||||
factor_set = _factor_set()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
value = factor_set.to_dict()
|
||||
value["dataset_snapshot_id"] = "rhdsv1:sha256:" + "0" * 64
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as snapshot_error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
snapshot_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.dataset_snapshot_id",
|
||||
)
|
||||
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = ["rhfactorv1:sha256:" + "0" * 64]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
with pytest.raises(FactorContractError) as definition_error:
|
||||
FactorSetRef.from_dict(
|
||||
value,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(
|
||||
definition_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.definition_ids",
|
||||
)
|
||||
|
||||
arguments = _factor_set_arguments()
|
||||
arguments["selected_view_ref_ids"] = ("rhviewrefv1:sha256:" + "0" * 64,)
|
||||
with pytest.raises(FactorContractError) as view_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(
|
||||
view_error,
|
||||
ContractErrorCode.INPUT_CLOSURE_VIOLATION,
|
||||
"$.selected_view_ref_ids",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"invalid_definition_id",
|
||||
[
|
||||
{"unexpected": "object"},
|
||||
["array"],
|
||||
42,
|
||||
True,
|
||||
None,
|
||||
],
|
||||
)
|
||||
def test_factor_set_ref_definition_ids_reject_non_string_types(
|
||||
invalid_definition_id: Any,
|
||||
) -> None:
|
||||
factor_set = _factor_set()
|
||||
definition = _golden_definition()
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = [definition.definition_id, invalid_definition_id]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_json(
|
||||
canonical_json_bytes(value),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
|
||||
_assert_error(error, ContractErrorCode.TYPE_ERROR, "$.definition_ids[1]")
|
||||
|
||||
|
||||
def test_factor_set_ref_definition_ids_still_reject_duplicate_strings() -> None:
|
||||
factor_set = _factor_set()
|
||||
definition = _golden_definition()
|
||||
value = factor_set.to_dict()
|
||||
value["definition_ids"] = [definition.definition_id, definition.definition_id]
|
||||
_reidentify(value, "factor_set_id", "rhfactorsetv1:sha256:")
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
|
||||
with pytest.raises(FactorContractError) as error:
|
||||
FactorSetRef.from_json(
|
||||
canonical_json_bytes(value),
|
||||
definitions=(definition,),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
|
||||
_assert_error(error, ContractErrorCode.INVALID_VALUE, "$.definition_ids")
|
||||
|
||||
|
||||
def test_factor_set_parent_requires_exact_identity_and_correlation() -> None:
|
||||
parent = _factor_set()
|
||||
child_arguments = _factor_set_arguments()
|
||||
child_arguments.update(
|
||||
output_content_bytes=canonical_json_bytes({"rows": [{"value": "0.250"}]}),
|
||||
causation=Causation("factor_set", parent.factor_set_id),
|
||||
parent=parent,
|
||||
)
|
||||
child_arguments["output_artifact_ref"] = OutputArtifactRef.create(
|
||||
schema_digest=_sha256(child_arguments["output_schema_bytes"]),
|
||||
content_digest=_sha256(child_arguments["output_content_bytes"]),
|
||||
)
|
||||
child = FactorSetRef.create(**child_arguments)
|
||||
assert child.causation.id == parent.factor_set_id
|
||||
missing_parent = child.to_dict()
|
||||
snapshot, foundation = _snapshot_and_foundation()
|
||||
with pytest.raises(FactorContractError) as missing_error:
|
||||
FactorSetRef.from_dict(
|
||||
missing_parent,
|
||||
definitions=(_golden_definition(),),
|
||||
dataset_snapshot=snapshot,
|
||||
foundation=foundation,
|
||||
)
|
||||
_assert_error(missing_error, ContractErrorCode.LINEAGE_VIOLATION, "$.causation")
|
||||
wrong_correlation = dict(child_arguments)
|
||||
wrong_correlation["correlation_id"] = "different_run"
|
||||
with pytest.raises(FactorContractError) as correlation_error:
|
||||
FactorSetRef.create(**wrong_correlation)
|
||||
_assert_error(correlation_error, ContractErrorCode.LINEAGE_VIOLATION, "$.correlation_id")
|
||||
|
||||
|
||||
def test_legacy_bridge_is_explicit_lossy_and_preserves_all_four_historical_fields() -> None:
|
||||
definition = _golden_definition()
|
||||
legacy = FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="1.0.0",
|
||||
)
|
||||
binding = LegacyFactorBinding.create(
|
||||
definition=definition,
|
||||
legacy_factor_id=legacy.factor_id,
|
||||
legacy_version=legacy.version,
|
||||
legacy_definition_sha256=legacy.definition_sha256,
|
||||
legacy_dataset_schema_version=legacy.dataset_schema_version,
|
||||
canonical_input_schema_digest=definition.input_schema_digest,
|
||||
correspondence_evidence_digest="sha256:" + "5" * 64,
|
||||
)
|
||||
assert bind_legacy_factor(legacy, definition, binding) is definition
|
||||
assert project_legacy_factor(definition, binding) == legacy
|
||||
assert legacy.version_id == "factor:demo-momentum@1.0.0"
|
||||
assert legacy.definition_sha256 != definition.definition_id.rsplit(":", maxsplit=1)[-1]
|
||||
assert LegacyFactorBinding.from_json(binding.to_json(), definition=definition) == binding
|
||||
|
||||
mismatched = FactorVersion(
|
||||
factor_id="factor:different",
|
||||
version=legacy.version,
|
||||
definition_sha256=legacy.definition_sha256,
|
||||
dataset_schema_version=legacy.dataset_schema_version,
|
||||
)
|
||||
with pytest.raises(FactorContractError) as mismatch_error:
|
||||
bind_legacy_factor(mismatched, definition, binding)
|
||||
_assert_error(mismatch_error, ContractErrorCode.LEGACY_BINDING_MISMATCH, "$.binding")
|
||||
|
||||
|
||||
def test_bare_legacy_factor_or_id_cannot_enter_factor_set_contract() -> None:
|
||||
legacy = FactorVersion("factor:demo-momentum", "1.0.0", "b" * 64, "1.0.0")
|
||||
arguments = _factor_set_arguments()
|
||||
arguments["definitions"] = (legacy,)
|
||||
with pytest.raises(FactorContractError) as legacy_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(legacy_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
|
||||
arguments["definitions"] = (legacy.version_id,)
|
||||
with pytest.raises(FactorContractError) as id_error:
|
||||
FactorSetRef.create(**arguments)
|
||||
_assert_error(id_error, ContractErrorCode.TYPE_ERROR, "$.definitions[0]")
|
||||
@@ -0,0 +1,173 @@
|
||||
"""Contracts for reusable factor diagnostics and transformations."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.factor_library import (
|
||||
annualized_sharpe,
|
||||
apply_factor_direction,
|
||||
cross_sectional_momentum,
|
||||
cross_sectional_pct_rank,
|
||||
cross_sectional_rank_with_direction,
|
||||
ic_summary,
|
||||
jb_test,
|
||||
kurtosis,
|
||||
ols_regress,
|
||||
rolling_annual_vol,
|
||||
rolling_zscore,
|
||||
skewness,
|
||||
spearman_ic,
|
||||
time_series_momentum,
|
||||
turnover,
|
||||
winsorize,
|
||||
)
|
||||
|
||||
|
||||
def test_turnover_supports_one_way_and_round_trip_conventions() -> None:
|
||||
weights = pd.DataFrame({"A": [1.0, 0.0], "B": [0.0, 1.0]})
|
||||
|
||||
pd.testing.assert_series_equal(turnover(weights), pd.Series([1.0], index=[1]))
|
||||
pd.testing.assert_series_equal(
|
||||
turnover(weights, divide_by_two=False), pd.Series([2.0], index=[1])
|
||||
)
|
||||
assert turnover(weights.iloc[:1]).empty
|
||||
|
||||
|
||||
def test_ic_functions_measure_monotonic_relationship() -> None:
|
||||
factor = pd.Series([1.0, 2.0, 3.0, 4.0])
|
||||
forward = pd.Series([10.0, 20.0, 30.0, 40.0])
|
||||
|
||||
assert spearman_ic(factor, forward) == pytest.approx(1.0)
|
||||
result = ic_summary(factor, forward, periods=(1,), method="pearson")
|
||||
assert result.loc[1, "ic_mean"] == pytest.approx(1.0)
|
||||
assert result.loc[1, "n"] == 4
|
||||
|
||||
|
||||
def test_ic_summary_rejects_unknown_method() -> None:
|
||||
with pytest.raises(ValueError, match="not supported"):
|
||||
ic_summary(pd.Series([1, 2, 3]), pd.Series([1, 2, 3]), method="kendall")
|
||||
|
||||
|
||||
def test_winsorize_clips_tails_and_preserves_nan() -> None:
|
||||
values = pd.Series([0.0, 1.0, 2.0, 100.0, np.nan])
|
||||
|
||||
result = winsorize(values, lower=0.25, upper=0.75)
|
||||
|
||||
assert result.iloc[0] == pytest.approx(0.75)
|
||||
assert result.iloc[3] == pytest.approx(26.5)
|
||||
assert pd.isna(result.iloc[4])
|
||||
|
||||
|
||||
def test_distribution_diagnostics_handle_short_samples() -> None:
|
||||
assert np.isnan(skewness(pd.Series([1.0, 2.0])))
|
||||
assert np.isnan(kurtosis(pd.Series([1.0, 2.0, 3.0])))
|
||||
jb, p_value = jb_test(pd.Series(range(7), dtype=float))
|
||||
assert np.isnan(jb)
|
||||
assert np.isnan(p_value)
|
||||
|
||||
|
||||
def test_distribution_diagnostics_return_finite_values() -> None:
|
||||
values = pd.Series([-2.0, -1.0, -0.5, 0.0, 0.25, 0.75, 1.0, 3.0])
|
||||
|
||||
assert np.isfinite(skewness(values))
|
||||
assert np.isfinite(kurtosis(values))
|
||||
jb, p_value = jb_test(values)
|
||||
assert jb >= 0
|
||||
assert 0 <= p_value <= 1
|
||||
|
||||
|
||||
def test_ols_recovers_linear_coefficients_and_residual_index() -> None:
|
||||
index = pd.date_range("2026-01-01", periods=8)
|
||||
factor = pd.Series(np.arange(8, dtype=float), index=index, name="factor")
|
||||
target = 1.5 + 2.0 * factor
|
||||
|
||||
result = ols_regress(target, factor)
|
||||
|
||||
assert result.alpha == pytest.approx(1.5)
|
||||
assert result.beta["factor"] == pytest.approx(2.0)
|
||||
assert result.r_squared == pytest.approx(1.0)
|
||||
assert result.n == 8
|
||||
assert result.resid.index.equals(index)
|
||||
|
||||
|
||||
def test_ols_handles_collinear_factors_without_crashing() -> None:
|
||||
x = pd.DataFrame({"a": np.arange(8, dtype=float), "b": np.arange(8, dtype=float)})
|
||||
y = pd.Series(1.0 + x["a"])
|
||||
|
||||
result = ols_regress(y, x)
|
||||
|
||||
assert result.n == 8
|
||||
assert np.isfinite(result.beta).all()
|
||||
np.testing.assert_allclose(result.resid, 0.0, atol=1e-12)
|
||||
|
||||
|
||||
def test_ols_short_sample_returns_empty_estimate() -> None:
|
||||
result = ols_regress(pd.Series([1.0, 2.0]), pd.Series([1.0, 2.0], name="x"))
|
||||
|
||||
assert np.isnan(result.alpha)
|
||||
assert result.beta.empty
|
||||
assert result.n == 2
|
||||
|
||||
|
||||
def test_momentum_and_rolling_transforms_match_manual_values() -> None:
|
||||
prices = pd.DataFrame({"A": [100.0, 110.0, 121.0, 133.1]})
|
||||
momentum = cross_sectional_momentum(prices, lookback=2, skip=0)
|
||||
assert momentum.iloc[2, 0] == pytest.approx(0.21)
|
||||
|
||||
returns = pd.Series([0.1, 0.1, -0.5, -0.5])
|
||||
pd.testing.assert_series_equal(
|
||||
time_series_momentum(returns, lookback=2),
|
||||
pd.Series([0, 1, -1, -1]),
|
||||
)
|
||||
|
||||
values = pd.Series([1.0, 2.0, 3.0])
|
||||
zscore = rolling_zscore(values, window=3)
|
||||
assert zscore.iloc[-1] == pytest.approx(1.0)
|
||||
annual_vol = rolling_annual_vol(returns, window=2, min_periods=2, trading_days=4)
|
||||
assert annual_vol.iloc[1] == pytest.approx(0.0)
|
||||
|
||||
|
||||
def test_rank_helpers_support_global_and_grouped_ranking() -> None:
|
||||
frame = pd.DataFrame(
|
||||
{"factor": [3.0, 1.0, 2.0, 4.0], "industry": ["x", "x", "y", "y"]}
|
||||
)
|
||||
|
||||
global_rank = cross_sectional_pct_rank(frame, "factor", ascending=True)
|
||||
grouped_rank = cross_sectional_pct_rank(
|
||||
frame, "factor", group_col="industry", ascending=True
|
||||
)
|
||||
|
||||
assert global_rank.tolist() == [0.75, 0.25, 0.5, 1.0]
|
||||
assert grouped_rank.tolist() == [1.0, 0.5, 0.5, 1.0]
|
||||
assert cross_sectional_pct_rank(frame, "missing").empty
|
||||
|
||||
|
||||
def test_factor_direction_and_directional_rank() -> None:
|
||||
pe = pd.Series([10.0, 20.0], name="pe_ttm")
|
||||
pd.testing.assert_series_equal(apply_factor_direction(pe), -pe)
|
||||
|
||||
frame = pd.DataFrame({"pe_ttm": [10.0, 20.0], "roe": [0.1, 0.2]})
|
||||
assert cross_sectional_rank_with_direction(frame, "pe_ttm").tolist() == [1.0, 0.5]
|
||||
assert cross_sectional_rank_with_direction(frame, "roe").tolist() == [0.5, 1.0]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("direction", ["sideways", "", "REVERSE"])
|
||||
def test_factor_direction_rejects_unknown_values(direction: str) -> None:
|
||||
factor = pd.Series([1.0, 2.0], name="roe")
|
||||
|
||||
with pytest.raises(ValueError, match="direction"):
|
||||
apply_factor_direction(factor, direction=direction)
|
||||
with pytest.raises(ValueError, match="direction"):
|
||||
cross_sectional_rank_with_direction(
|
||||
pd.DataFrame({"roe": factor}), "roe", direction=direction
|
||||
)
|
||||
|
||||
|
||||
def test_annualized_sharpe_handles_empty_and_nonzero_returns() -> None:
|
||||
assert annualized_sharpe(pd.Series(dtype=float)) == 0.0
|
||||
returns = pd.Series([0.01, -0.01, 0.02, 0.0])
|
||||
expected = returns.mean() * 252 / (returns.std() * np.sqrt(252))
|
||||
assert annualized_sharpe(returns) == pytest.approx(expected)
|
||||
@@ -0,0 +1,338 @@
|
||||
"""Governed Personal Quant OS vertical-slice contracts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.governed_pipeline import (
|
||||
DatasetSnapshot,
|
||||
FactorVersion,
|
||||
PaperOrderIntent,
|
||||
RiskDecisionStatus,
|
||||
RiskPolicy,
|
||||
StrategyStage,
|
||||
StrategyVersion,
|
||||
create_paper_order_intent,
|
||||
run_governed_factor_slice,
|
||||
)
|
||||
|
||||
|
||||
def _calendar() -> pd.DatetimeIndex:
|
||||
return pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
|
||||
|
||||
def _scores() -> pd.DataFrame:
|
||||
dates = _calendar()
|
||||
return pd.DataFrame(
|
||||
{"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]},
|
||||
index=dates[:2],
|
||||
)
|
||||
|
||||
|
||||
def _prices() -> tuple[pd.DataFrame, pd.DataFrame]:
|
||||
dates = _calendar()
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 10.2, 10.4], "B": [20.0, 20.0, 20.5, 21.0], "C": [30.0, 30.0, 30.0, 30.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = opens * 1.01
|
||||
return opens, closes
|
||||
|
||||
|
||||
def _snapshot() -> DatasetSnapshot:
|
||||
return DatasetSnapshot(
|
||||
snapshot_id="dataset:cn-a-daily-20260108-v1",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 7, tzinfo=UTC),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 8, 5, tzinfo=UTC),
|
||||
)
|
||||
|
||||
|
||||
def _factor() -> FactorVersion:
|
||||
return FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="1.0.0",
|
||||
)
|
||||
|
||||
|
||||
def _strategy() -> StrategyVersion:
|
||||
return StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id="factor:demo-momentum@1.0.0",
|
||||
stage=StrategyStage.APPROVED,
|
||||
)
|
||||
|
||||
|
||||
def _execution_config() -> ExecutionConfig:
|
||||
return ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
|
||||
def test_governed_slice_is_reproducible_and_creates_only_paper_intent() -> None:
|
||||
opens, closes = _prices()
|
||||
created_at = datetime(2026, 1, 9, 1, tzinfo=UTC)
|
||||
policy = RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
)
|
||||
|
||||
result = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=policy,
|
||||
code_revision="c" * 40,
|
||||
created_at=created_at,
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
assert result.backtest_run.dataset_snapshot_id == _snapshot().snapshot_id
|
||||
assert result.backtest_run.factor_version_id == _factor().version_id
|
||||
assert result.backtest_run.strategy_version_id == _strategy().version_id
|
||||
assert result.backtest_run.code_revision == "c" * 40
|
||||
assert len(result.backtest_run.config_hash) == 64
|
||||
assert result.portfolio_target.backtest_run_id == result.backtest_run.run_id
|
||||
assert result.risk_decision.status is RiskDecisionStatus.APPROVED
|
||||
assert result.risk_decision.portfolio_target_id == result.portfolio_target.target_id
|
||||
assert result.order_intent is not None
|
||||
assert result.order_intent.environment == "paper"
|
||||
assert result.order_intent.risk_decision_id == result.risk_decision.decision_id
|
||||
assert result.order_intent.portfolio_target_id == result.portfolio_target.target_id
|
||||
assert result.factor_version.version_id == "factor:demo-momentum@1.0.0"
|
||||
assert result.factor_version.definition_sha256 == "b" * 64
|
||||
assert result.strategy_version.version_id == "strategy:demo-top2@1.0.0"
|
||||
assert result.backtest_run.run_id == (
|
||||
"backtest-run:e74403571f6a73c98b380220748957a422518c0bed883fabc4b93ebe13f05a37"
|
||||
)
|
||||
assert result.backtest_run.config_hash == (
|
||||
"40a3c804a2dc940161a626d1a5d25817c13463e685005e37fc48c41d1e20b87b"
|
||||
)
|
||||
assert result.portfolio_target.target_id == (
|
||||
"portfolio-target:ab2d398489aa9a292ee155a1098e9340beac0a924874cdaeb5b7b4d379ac9ce8"
|
||||
)
|
||||
assert result.risk_decision.decision_id == (
|
||||
"risk-decision:95924926bd327e44beeb15a63f47d14c5e77b97533fc3227fba2eda68e9b423d"
|
||||
)
|
||||
assert result.order_intent.intent_id == (
|
||||
"order-intent:73c349086c2c05b68424ace9286896eb21507b9080da5ff9b96b58c88d8f6ac1"
|
||||
)
|
||||
|
||||
repeated = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=policy,
|
||||
code_revision="c" * 40,
|
||||
created_at=created_at,
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
assert repeated.backtest_run.run_id == result.backtest_run.run_id
|
||||
assert repeated.portfolio_target.target_id == result.portfolio_target.target_id
|
||||
assert repeated.risk_decision.decision_id == result.risk_decision.decision_id
|
||||
assert repeated.order_intent == result.order_intent
|
||||
|
||||
|
||||
def test_risk_rejection_blocks_order_intent() -> None:
|
||||
opens, closes = _prices()
|
||||
result = run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=_strategy(),
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:no-concentration@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.4,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
assert result.risk_decision.status is RiskDecisionStatus.REJECTED
|
||||
assert any("single-asset weight" in reason for reason in result.risk_decision.reasons)
|
||||
assert result.order_intent is None
|
||||
with pytest.raises(ValueError, match="approved risk decision"):
|
||||
create_paper_order_intent(result.portfolio_target, result.risk_decision)
|
||||
with pytest.raises(ValueError, match="approved risk decision"):
|
||||
PaperOrderIntent(result.portfolio_target, result.risk_decision)
|
||||
|
||||
|
||||
def test_dataset_snapshot_requires_point_in_time_ordering_and_aware_times() -> None:
|
||||
with pytest.raises(ValueError, match="timezone-aware"):
|
||||
DatasetSnapshot(
|
||||
snapshot_id="dataset:invalid",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 7),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="effective_at <= available_at <= ingested_at"):
|
||||
DatasetSnapshot(
|
||||
snapshot_id="dataset:invalid",
|
||||
schema_version="1.0.0",
|
||||
content_sha256="a" * 64,
|
||||
effective_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
|
||||
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
|
||||
ingested_at=datetime(2026, 1, 8, 10, tzinfo=UTC),
|
||||
)
|
||||
|
||||
|
||||
def test_strategy_factor_lineage_must_match() -> None:
|
||||
opens, closes = _prices()
|
||||
mismatched = StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id="factor:other@1.0.0",
|
||||
stage=StrategyStage.APPROVED,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="factor lineage"):
|
||||
run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=mismatched,
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
|
||||
def test_governed_slice_requires_matching_schema_and_snapshot_available_by_run_time() -> None:
|
||||
opens, closes = _prices()
|
||||
common = {
|
||||
"factor_scores": _scores(),
|
||||
"execution_prices": opens,
|
||||
"valuation_prices": closes,
|
||||
"strategy_version": _strategy(),
|
||||
"risk_policy": RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
"code_revision": "c" * 40,
|
||||
"top_k": 2,
|
||||
"execution_price_field": "open",
|
||||
"valuation_price_field": "close",
|
||||
"execution_config": _execution_config(),
|
||||
}
|
||||
|
||||
with pytest.raises(ValueError, match="dataset schema"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=FactorVersion(
|
||||
factor_id="factor:demo-momentum",
|
||||
version="1.0.0",
|
||||
definition_sha256="b" * 64,
|
||||
dataset_schema_version="2.0.0",
|
||||
),
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
**common,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="available before the research run"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
created_at=datetime(2026, 1, 8, 7, 30, tzinfo=UTC),
|
||||
**common,
|
||||
)
|
||||
|
||||
future_scores = _scores()
|
||||
future_scores.index = pd.date_range("2026-01-12", periods=2, freq="B")
|
||||
with pytest.raises(ValueError, match="future decision dates"):
|
||||
run_governed_factor_slice(
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
factor_scores=future_scores,
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
strategy_version=common["strategy_version"],
|
||||
risk_policy=common["risk_policy"],
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
|
||||
|
||||
def test_paper_intent_requires_approved_strategy_stage() -> None:
|
||||
opens, closes = _prices()
|
||||
validated = StrategyVersion(
|
||||
strategy_id="strategy:demo-top2",
|
||||
version="1.0.0",
|
||||
factor_version_id=_factor().version_id,
|
||||
stage=StrategyStage.VALIDATED,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="Approved or Paper"):
|
||||
run_governed_factor_slice(
|
||||
factor_scores=_scores(),
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
dataset_snapshot=_snapshot(),
|
||||
factor_version=_factor(),
|
||||
strategy_version=validated,
|
||||
risk_policy=RiskPolicy(
|
||||
policy_id="risk:paper-default@1.0.0",
|
||||
max_gross_exposure=1.0,
|
||||
max_single_asset_weight=0.6,
|
||||
max_positions=10,
|
||||
),
|
||||
code_revision="c" * 40,
|
||||
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
|
||||
top_k=2,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
execution_config=_execution_config(),
|
||||
)
|
||||
@@ -0,0 +1,163 @@
|
||||
"""Mathematical contracts for the standard performance metrics."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.metrics import (
|
||||
TRADING_DAYS_PER_YEAR,
|
||||
annualized_return,
|
||||
annualized_volatility,
|
||||
benchmark_summary,
|
||||
calmar_ratio,
|
||||
max_drawdown,
|
||||
sharpe_ratio,
|
||||
sortino_ratio,
|
||||
summary,
|
||||
win_rate,
|
||||
)
|
||||
|
||||
|
||||
def test_annualized_return_uses_compounded_simple_returns() -> None:
|
||||
returns = pd.Series([0.10, -0.10])
|
||||
expected = 0.99 ** (TRADING_DAYS_PER_YEAR / 2) - 1.0
|
||||
|
||||
assert annualized_return(returns) == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_annualized_volatility_uses_sample_standard_deviation() -> None:
|
||||
returns = pd.Series([0.01, 0.03, 0.02])
|
||||
|
||||
assert annualized_volatility(returns) == pytest.approx(
|
||||
returns.std() * np.sqrt(TRADING_DAYS_PER_YEAR)
|
||||
)
|
||||
|
||||
|
||||
def test_sharpe_ratio_subtracts_annual_risk_free_rate() -> None:
|
||||
returns = pd.Series([0.01, -0.005, 0.02, 0.0])
|
||||
|
||||
result = sharpe_ratio(returns, rf=0.02)
|
||||
|
||||
assert result == pytest.approx(
|
||||
(annualized_return(returns) - 0.02) / annualized_volatility(returns)
|
||||
)
|
||||
|
||||
|
||||
def test_zero_volatility_metrics_return_zero() -> None:
|
||||
returns = pd.Series([0.0, 0.0, 0.0])
|
||||
|
||||
assert sharpe_ratio(returns) == 0.0
|
||||
assert sortino_ratio(returns) == 0.0
|
||||
assert calmar_ratio(returns) == 0.0
|
||||
|
||||
|
||||
def test_sortino_ratio_uses_all_sessions_for_downside_deviation() -> None:
|
||||
returns = pd.Series([0.02, -0.01, 0.0, -0.03])
|
||||
downside = np.minimum(returns.to_numpy(), 0.0)
|
||||
downside_deviation = np.sqrt(np.mean(np.square(downside))) * np.sqrt(
|
||||
TRADING_DAYS_PER_YEAR
|
||||
)
|
||||
|
||||
assert sortino_ratio(returns) == pytest.approx(
|
||||
annualized_return(returns) / downside_deviation
|
||||
)
|
||||
|
||||
|
||||
def test_max_drawdown_includes_loss_from_initial_capital() -> None:
|
||||
returns = pd.Series([-0.20, 0.0])
|
||||
|
||||
assert max_drawdown(returns) == pytest.approx(-0.20)
|
||||
|
||||
|
||||
def test_max_drawdown_tracks_peak_to_trough_loss() -> None:
|
||||
returns = pd.Series([0.10, -0.20, 0.05])
|
||||
|
||||
assert max_drawdown(returns) == pytest.approx(-0.20)
|
||||
|
||||
|
||||
def test_metrics_clean_nan_and_infinite_values() -> None:
|
||||
returns = pd.Series([0.10, np.nan, np.inf, -0.05, -np.inf])
|
||||
|
||||
assert win_rate(returns) == 0.5
|
||||
assert summary(returns)["n_days"] == 2
|
||||
|
||||
|
||||
def test_summary_aliases_match_canonical_fields() -> None:
|
||||
result = summary(pd.Series([0.01, -0.02, 0.03]))
|
||||
|
||||
assert result["annual_yield"] == result["ann_return"]
|
||||
assert result["annual_sd"] == result["ann_volatility"]
|
||||
assert result["drawback"] == result["max_drawdown"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"metric",
|
||||
[
|
||||
annualized_return,
|
||||
annualized_volatility,
|
||||
sharpe_ratio,
|
||||
sortino_ratio,
|
||||
max_drawdown,
|
||||
calmar_ratio,
|
||||
win_rate,
|
||||
],
|
||||
)
|
||||
def test_metrics_reject_non_series_input(metric) -> None:
|
||||
with pytest.raises(TypeError, match=r"expected pd\.Series"):
|
||||
metric([0.01, 0.02])
|
||||
|
||||
|
||||
def test_short_and_empty_series_return_zero() -> None:
|
||||
assert annualized_return(pd.Series(dtype=float)) == 0.0
|
||||
assert annualized_volatility(pd.Series([0.01])) == 0.0
|
||||
assert max_drawdown(pd.Series([0.01])) == 0.0
|
||||
assert win_rate(pd.Series(dtype=float)) == 0.0
|
||||
|
||||
|
||||
def test_benchmark_summary_uses_aligned_active_returns_and_regression() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
benchmark = pd.Series([-0.01, 0.0, 0.01, 0.02], index=dates)
|
||||
portfolio = 0.001 + 1.5 * benchmark
|
||||
active = portfolio - benchmark
|
||||
|
||||
result = benchmark_summary(portfolio, benchmark)
|
||||
|
||||
assert result["n_observations"] == 4
|
||||
assert result["tracking_error"] == pytest.approx(
|
||||
active.std() * np.sqrt(TRADING_DAYS_PER_YEAR)
|
||||
)
|
||||
assert result["information_ratio"] == pytest.approx(
|
||||
active.mean() / active.std() * np.sqrt(TRADING_DAYS_PER_YEAR)
|
||||
)
|
||||
assert result["beta"] == pytest.approx(1.5)
|
||||
assert result["alpha"] == pytest.approx(1.001**TRADING_DAYS_PER_YEAR - 1.0)
|
||||
|
||||
|
||||
def test_benchmark_summary_rejects_silent_calendar_alignment() -> None:
|
||||
portfolio = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-05", periods=2))
|
||||
benchmark = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-06", periods=2))
|
||||
|
||||
with pytest.raises(ValueError, match="matching indexes"):
|
||||
benchmark_summary(portfolio, benchmark)
|
||||
|
||||
|
||||
def test_benchmark_summary_rejects_missing_observations() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=2)
|
||||
portfolio = pd.Series([0.01, np.nan], index=dates)
|
||||
benchmark = pd.Series([0.0, 0.01], index=dates)
|
||||
|
||||
with pytest.raises(ValueError, match="finite"):
|
||||
benchmark_summary(portfolio, benchmark)
|
||||
|
||||
|
||||
def test_benchmark_summary_marks_constant_benchmark_regression_unestimable() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3)
|
||||
portfolio = pd.Series([0.01, -0.01, 0.02], index=dates)
|
||||
benchmark = pd.Series([0.0, 0.0, 0.0], index=dates)
|
||||
|
||||
result = benchmark_summary(portfolio, benchmark)
|
||||
|
||||
assert np.isnan(result["alpha"])
|
||||
assert np.isnan(result["beta"])
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Factor-score portfolio construction and backtest integration contracts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.backtest import run_weight_backtest
|
||||
from quant_engine.portfolio_construction import (
|
||||
equal_weight,
|
||||
scores_to_target_weights,
|
||||
scores_to_weight_table,
|
||||
select_top_k,
|
||||
)
|
||||
|
||||
|
||||
def test_select_top_k_ignores_nan_and_breaks_ties_by_input_order() -> None:
|
||||
scores = pd.Series([1.0, 1.0, np.nan, 0.5], index=["B", "A", "C", "D"])
|
||||
|
||||
selected = select_top_k(scores, top_k=2)
|
||||
|
||||
assert selected.tolist() == ["B", "A"]
|
||||
|
||||
|
||||
def test_select_top_k_can_select_lowest_scores() -> None:
|
||||
scores = pd.Series([3.0, 1.0, 2.0], index=["A", "B", "C"])
|
||||
|
||||
selected = select_top_k(scores, top_k=2, largest=False)
|
||||
|
||||
assert selected.tolist() == ["B", "C"]
|
||||
|
||||
|
||||
def test_equal_weight_allocates_requested_gross_exposure() -> None:
|
||||
result = equal_weight(pd.Index(["A", "B", "C"]), gross_exposure=0.9)
|
||||
|
||||
pd.testing.assert_series_equal(
|
||||
result,
|
||||
pd.Series([0.3, 0.3, 0.3], index=["A", "B", "C"], name="weight"),
|
||||
)
|
||||
|
||||
|
||||
def test_equal_weight_returns_empty_float_series_for_no_assets() -> None:
|
||||
result = equal_weight(pd.Index([], dtype=object))
|
||||
|
||||
assert result.empty
|
||||
assert result.dtype == float
|
||||
assert result.name == "weight"
|
||||
|
||||
|
||||
def test_scores_to_target_weights_keeps_full_universe_with_zero_for_unselected() -> None:
|
||||
scores = pd.Series([0.2, 0.8, 0.5], index=["A", "B", "C"])
|
||||
|
||||
result = scores_to_target_weights(scores, top_k=2)
|
||||
|
||||
pd.testing.assert_series_equal(
|
||||
result,
|
||||
pd.Series([0.0, 0.5, 0.5], index=scores.index, name="weight"),
|
||||
)
|
||||
|
||||
|
||||
def test_scores_to_target_weights_divides_exposure_over_available_scores() -> None:
|
||||
scores = pd.Series([1.0, np.nan, 0.5], index=["A", "B", "C"])
|
||||
|
||||
result = scores_to_target_weights(scores, top_k=5, gross_exposure=0.8)
|
||||
|
||||
pd.testing.assert_series_equal(
|
||||
result,
|
||||
pd.Series([0.4, 0.0, 0.4], index=scores.index, name="weight"),
|
||||
)
|
||||
|
||||
|
||||
def test_scores_to_weight_table_constructs_each_rebalance_independently() -> None:
|
||||
dates = pd.to_datetime(["2026-01-05", "2026-01-07"])
|
||||
scores = pd.DataFrame(
|
||||
{"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]},
|
||||
index=dates,
|
||||
)
|
||||
|
||||
result = scores_to_weight_table(scores, top_k=2)
|
||||
|
||||
expected = pd.DataFrame(
|
||||
{"A": [0.5, 0.0], "B": [0.5, 0.5], "C": [0.0, 0.5]},
|
||||
index=dates,
|
||||
)
|
||||
pd.testing.assert_frame_equal(result, expected)
|
||||
|
||||
changed_future = scores.copy()
|
||||
changed_future.iloc[1] = [100.0, -100.0, 0.0]
|
||||
changed_result = scores_to_weight_table(changed_future, top_k=2)
|
||||
pd.testing.assert_series_equal(result.iloc[0], changed_result.iloc[0])
|
||||
|
||||
|
||||
def test_effective_holding_weights_flow_into_weight_backtest() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=3, freq="B")
|
||||
effective_weights = pd.DataFrame(
|
||||
{"A": [1.0, 0.0], "B": [0.0, 1.0]},
|
||||
index=dates[[0, 2]],
|
||||
)
|
||||
stock_returns = pd.DataFrame(
|
||||
{"A": [0.10, 0.0, 0.0], "B": [0.0, 0.0, 0.20]},
|
||||
index=dates,
|
||||
)
|
||||
|
||||
result = run_weight_backtest(effective_weights, stock_returns)
|
||||
|
||||
pd.testing.assert_series_equal(result.nav, pd.Series([1.1, 1.1, 1.32], index=dates))
|
||||
pd.testing.assert_frame_equal(result.weights, effective_weights)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("top_k", [0, -1])
|
||||
def test_portfolio_construction_rejects_non_positive_top_k(top_k: int) -> None:
|
||||
scores = pd.Series([1.0], index=["A"])
|
||||
|
||||
with pytest.raises(ValueError, match="top_k must be positive"):
|
||||
select_top_k(scores, top_k=top_k)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("gross_exposure", [-0.1, np.inf, np.nan])
|
||||
def test_equal_weight_rejects_invalid_gross_exposure(gross_exposure: float) -> None:
|
||||
with pytest.raises(ValueError, match="gross_exposure"):
|
||||
equal_weight(pd.Index(["A"]), gross_exposure=gross_exposure)
|
||||
|
||||
|
||||
def test_portfolio_construction_rejects_duplicate_assets() -> None:
|
||||
duplicate_scores = pd.Series([1.0, 2.0], index=["A", "A"])
|
||||
|
||||
with pytest.raises(ValueError, match="unique asset labels"):
|
||||
scores_to_target_weights(duplicate_scores, top_k=1)
|
||||
|
||||
|
||||
def test_weight_table_rejects_duplicate_rebalance_dates() -> None:
|
||||
duplicate_date = pd.Timestamp("2026-01-05")
|
||||
scores = pd.DataFrame(
|
||||
{"A": [1.0, 2.0]},
|
||||
index=[duplicate_date, duplicate_date],
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="unique rebalance dates"):
|
||||
scores_to_weight_table(scores, top_k=1)
|
||||
|
||||
|
||||
def test_weight_table_rejects_unsorted_rebalance_dates() -> None:
|
||||
scores = pd.DataFrame(
|
||||
{"A": [1.0, 2.0]},
|
||||
index=pd.to_datetime(["2026-01-07", "2026-01-05"]),
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="chronological order"):
|
||||
scores_to_weight_table(scores, top_k=1)
|
||||
|
||||
|
||||
def test_weight_table_rejects_non_numeric_scores() -> None:
|
||||
scores = pd.DataFrame({"A": ["high"], "B": ["low"]})
|
||||
|
||||
with pytest.raises(TypeError, match="numeric"):
|
||||
scores_to_weight_table(scores, top_k=1)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,336 @@
|
||||
"""No-lookahead factor-score to execution-audit integration contracts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.execution import ExecutionConfig
|
||||
from quant_engine.research_pipeline import (
|
||||
FactorBacktestResult,
|
||||
FactorExecutionResult,
|
||||
TargetWeightSchedule,
|
||||
run_factor_backtest_research,
|
||||
run_factor_execution_research,
|
||||
schedule_target_weights,
|
||||
)
|
||||
|
||||
|
||||
def _calendar() -> pd.DatetimeIndex:
|
||||
return pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
|
||||
|
||||
def _factor_scores() -> pd.DataFrame:
|
||||
dates = _calendar()
|
||||
return pd.DataFrame(
|
||||
{"A": [2.0, 0.0], "B": [1.0, 3.0]},
|
||||
index=dates[:2],
|
||||
)
|
||||
|
||||
|
||||
def _next_session_open_prices() -> pd.DataFrame:
|
||||
dates = _calendar()
|
||||
return pd.DataFrame(
|
||||
{"A": [1.0, 10.0, 10.0, 10.0], "B": [1.0, 10.0, 20.0, 20.0]},
|
||||
index=dates,
|
||||
)
|
||||
|
||||
|
||||
def test_schedule_target_weights_maps_signal_to_next_trading_session() -> None:
|
||||
dates = _calendar()
|
||||
decision_weights = pd.DataFrame(
|
||||
{"A": [1.0, 0.0], "B": [0.0, 1.0]},
|
||||
index=dates[:2],
|
||||
)
|
||||
|
||||
schedule = schedule_target_weights(decision_weights, dates, lag_sessions=1)
|
||||
|
||||
assert isinstance(schedule, TargetWeightSchedule)
|
||||
assert schedule.lag_sessions == 1
|
||||
pd.testing.assert_series_equal(
|
||||
schedule.signal_to_execution,
|
||||
pd.Series(dates[1:3], index=dates[:2], name="execution_date"),
|
||||
)
|
||||
expected = decision_weights.copy()
|
||||
expected.index = dates[1:3]
|
||||
expected.index.name = "execution_date"
|
||||
pd.testing.assert_frame_equal(schedule.execution_weights, expected)
|
||||
assert (schedule.execution_weights.index > schedule.signal_to_execution.index).all()
|
||||
|
||||
|
||||
def test_factor_execution_research_uses_next_session_prices() -> None:
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
result = run_factor_execution_research(
|
||||
_factor_scores(),
|
||||
_next_session_open_prices(),
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
initial_cash=1_000.0,
|
||||
config=config,
|
||||
)
|
||||
|
||||
assert isinstance(result, FactorExecutionResult)
|
||||
assert result.execution_price_field == "open"
|
||||
assert result.execution.daily_executions[0].date == str(_calendar()[1])
|
||||
assert result.execution.positions[0].holdings == {"A": 100.0}
|
||||
assert result.execution.positions[1].holdings == {"B": 50.0}
|
||||
assert result.execution.final_portfolio_value == pytest.approx(1_000.0)
|
||||
|
||||
|
||||
def test_factor_execution_result_snapshots_research_inputs() -> None:
|
||||
scores = _factor_scores()
|
||||
prices = _next_session_open_prices()
|
||||
|
||||
result = run_factor_execution_research(
|
||||
scores,
|
||||
prices,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
)
|
||||
scores.iloc[0, 0] = -999.0
|
||||
prices.iloc[1, 0] = 999.0
|
||||
|
||||
assert result.factor_scores.iloc[0, 0] == 2.0
|
||||
assert result.execution_prices.loc[_calendar()[1], "A"] == 10.0
|
||||
assert result.execution.positions[0].holdings["A"] < 200_000.0
|
||||
|
||||
|
||||
@pytest.mark.parametrize("lag_sessions", [0, -1, True])
|
||||
def test_schedule_target_weights_requires_positive_integer_lag(lag_sessions: int) -> None:
|
||||
with pytest.raises(ValueError, match="lag_sessions"):
|
||||
schedule_target_weights(
|
||||
pd.DataFrame({"A": [1.0]}, index=_calendar()[:1]),
|
||||
_calendar(),
|
||||
lag_sessions=lag_sessions,
|
||||
)
|
||||
|
||||
|
||||
def test_schedule_target_weights_rejects_signal_outside_trading_calendar() -> None:
|
||||
weekend = pd.Timestamp("2026-01-10")
|
||||
with pytest.raises(ValueError, match="signal dates must be trading sessions"):
|
||||
schedule_target_weights(
|
||||
pd.DataFrame({"A": [1.0]}, index=[weekend]),
|
||||
_calendar(),
|
||||
)
|
||||
|
||||
|
||||
def test_schedule_target_weights_rejects_missing_future_execution_session() -> None:
|
||||
dates = _calendar()
|
||||
with pytest.raises(ValueError, match="future execution session"):
|
||||
schedule_target_weights(
|
||||
pd.DataFrame({"A": [1.0]}, index=dates[-1:]),
|
||||
dates,
|
||||
)
|
||||
|
||||
|
||||
def test_factor_execution_research_requires_explicit_price_field() -> None:
|
||||
with pytest.raises(ValueError, match="execution_price_field"):
|
||||
run_factor_execution_research(
|
||||
_factor_scores(),
|
||||
_next_session_open_prices(),
|
||||
top_k=1,
|
||||
execution_price_field="",
|
||||
)
|
||||
|
||||
|
||||
def test_factor_execution_research_accepts_empty_scores() -> None:
|
||||
scores = pd.DataFrame(columns=["A", "B"], index=pd.DatetimeIndex([]), dtype=float)
|
||||
|
||||
result = run_factor_execution_research(
|
||||
scores,
|
||||
_next_session_open_prices(),
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
)
|
||||
|
||||
assert result.schedule.execution_weights.empty
|
||||
assert result.execution.positions == ()
|
||||
|
||||
|
||||
def test_factor_backtest_research_runs_signal_to_daily_performance_without_lookahead() -> None:
|
||||
"""信号日保持现金,下一日开盘成交后才参与当日收盘收益。"""
|
||||
dates = _calendar()
|
||||
scores = pd.DataFrame({"A": [2.0], "B": [1.0]}, index=dates[:1])
|
||||
opens = pd.DataFrame(
|
||||
{"A": [1.0, 10.0, 10.0, 10.0], "B": [1.0, 20.0, 20.0, 20.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [500.0, 11.0, 12.0, 12.0], "B": [500.0, 20.0, 20.0, 20.0]},
|
||||
index=dates,
|
||||
)
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=config,
|
||||
)
|
||||
|
||||
assert isinstance(result, FactorBacktestResult)
|
||||
assert result.execution_price_field == "open"
|
||||
assert result.valuation_price_field == "close"
|
||||
pd.testing.assert_series_equal(
|
||||
result.nav,
|
||||
pd.Series([1.0, 1.1, 1.2, 1.2], index=dates, name="nav"),
|
||||
)
|
||||
pd.testing.assert_series_equal(
|
||||
result.returns,
|
||||
pd.Series([0.0, 0.1, 1.2 / 1.1 - 1.0, 0.0], index=dates, name="returns"),
|
||||
)
|
||||
assert result.stats()["n_days"] == 4
|
||||
assert result.execution.daily_executions[0].executions == ()
|
||||
assert result.execution.daily_executions[1].executions[0].price == 10.0
|
||||
|
||||
|
||||
def test_factor_backtest_result_snapshots_both_price_semantics() -> None:
|
||||
scores = pd.DataFrame({"A": [1.0]}, index=_calendar()[:1])
|
||||
opens = pd.DataFrame({"A": [10.0, 10.0, 10.0, 10.0]}, index=_calendar())
|
||||
closes = pd.DataFrame({"A": [10.0, 11.0, 12.0, 13.0]}, index=_calendar())
|
||||
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
)
|
||||
opens.iloc[1, 0] = 999.0
|
||||
closes.iloc[1, 0] = 999.0
|
||||
|
||||
assert result.execution_prices.iloc[1, 0] == 10.0
|
||||
assert result.valuation_prices.iloc[1, 0] == 11.0
|
||||
|
||||
|
||||
def test_factor_backtest_research_requires_matching_daily_calendars() -> None:
|
||||
scores = pd.DataFrame({"A": [1.0]}, index=_calendar()[:1])
|
||||
opens = pd.DataFrame({"A": [10.0, 10.0, 10.0, 10.0]}, index=_calendar())
|
||||
closes = pd.DataFrame({"A": [10.0, 11.0, 12.0]}, index=_calendar()[:3])
|
||||
|
||||
with pytest.raises(ValueError, match="matching trading calendars"):
|
||||
run_factor_backtest_research(
|
||||
scores,
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
)
|
||||
|
||||
|
||||
def test_factor_backtest_starts_at_first_signal_instead_of_price_warmup() -> None:
|
||||
"""因子预热行情不能作为空仓日混入研究绩效区间。"""
|
||||
dates = pd.date_range("2026-01-05", periods=5, freq="B")
|
||||
scores = pd.DataFrame({"A": [1.0]}, index=dates[2:3])
|
||||
opens = pd.DataFrame({"A": [1.0, 1.0, 1.0, 10.0, 10.0]}, index=dates)
|
||||
closes = pd.DataFrame({"A": [100.0, 200.0, 300.0, 11.0, 12.0]}, index=dates)
|
||||
config = ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
)
|
||||
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
execution_prices=opens,
|
||||
valuation_prices=closes,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=config,
|
||||
)
|
||||
|
||||
assert result.nav.index.equals(dates[2:])
|
||||
pd.testing.assert_series_equal(
|
||||
result.nav,
|
||||
pd.Series([1.0, 1.1, 1.2], index=dates[2:], name="nav"),
|
||||
)
|
||||
assert result.stats()["n_days"] == 3
|
||||
|
||||
|
||||
def test_factor_backtest_exposes_net_benchmark_metrics() -> None:
|
||||
dates = _calendar()
|
||||
scores = pd.DataFrame({"A": [1.0]}, index=dates[:1])
|
||||
prices = pd.DataFrame({"A": [10.0, 10.0, 11.0, 11.0]}, index=dates)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
prices,
|
||||
prices,
|
||||
top_k=1,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
benchmark = pd.Series([0.0, 0.01, -0.01, 0.0], index=dates)
|
||||
|
||||
relative = result.benchmark_stats(benchmark)
|
||||
|
||||
assert relative["n_observations"] == len(result.returns)
|
||||
assert relative["tracking_error"] > 0
|
||||
|
||||
|
||||
def test_factor_backtest_projects_actual_close_weights_from_ledger() -> None:
|
||||
dates = _calendar()
|
||||
scores = pd.DataFrame({"A": [1.0], "B": [0.0]}, index=dates[:1])
|
||||
opens = pd.DataFrame(
|
||||
{"A": [10.0, 10.0, 10.0, 10.0], "B": [20.0, 20.0, 20.0, 20.0]},
|
||||
index=dates,
|
||||
)
|
||||
closes = pd.DataFrame(
|
||||
{"A": [10.0, 11.0, 12.0, 12.0], "B": [20.0, 20.0, 20.0, 20.0]},
|
||||
index=dates,
|
||||
)
|
||||
result = run_factor_backtest_research(
|
||||
scores,
|
||||
opens,
|
||||
closes,
|
||||
top_k=1,
|
||||
gross_exposure=0.5,
|
||||
execution_price_field="open",
|
||||
valuation_price_field="close",
|
||||
initial_cash=1_000.0,
|
||||
config=ExecutionConfig(
|
||||
commission_bps=0,
|
||||
stamp_tax_bps=0,
|
||||
slippage_bps=0,
|
||||
min_trade_amount=0,
|
||||
),
|
||||
)
|
||||
|
||||
weights = result.position_weights
|
||||
cash = result.cash_weights
|
||||
|
||||
assert weights.index.equals(result.nav.index)
|
||||
assert weights.columns.tolist() == ["A", "B"]
|
||||
assert weights.loc[dates[0]].sum() == 0.0
|
||||
assert cash.loc[dates[0]] == 1.0
|
||||
assert weights.loc[dates[1], "A"] == pytest.approx(550.0 / 1_050.0)
|
||||
pd.testing.assert_series_equal(
|
||||
weights.sum(axis=1) + cash,
|
||||
pd.Series(1.0, index=dates),
|
||||
check_names=False,
|
||||
)
|
||||
@@ -0,0 +1,270 @@
|
||||
"""Risk contribution contracts and validation tests."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
from quant_engine.risk import (
|
||||
ComponentRiskResult,
|
||||
CovarianceSnapshot,
|
||||
component_var,
|
||||
estimate_covariance_snapshot,
|
||||
labeled_component_risk,
|
||||
marginal_risk_contribution,
|
||||
risk_contribution,
|
||||
)
|
||||
|
||||
|
||||
def test_estimate_covariance_snapshot_is_complete_case_and_reproducible() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=6, freq="B")
|
||||
returns = pd.DataFrame(
|
||||
{
|
||||
"A": [0.01, 0.02, 0.03, 0.04, 0.05, 99.0],
|
||||
"B": [0.02, 0.01, np.nan, 0.03, 0.04, -99.0],
|
||||
},
|
||||
index=dates,
|
||||
)
|
||||
as_of = dates[4]
|
||||
|
||||
snapshot = estimate_covariance_snapshot(
|
||||
returns,
|
||||
as_of_date=as_of,
|
||||
lookback_sessions=4,
|
||||
min_observations=3,
|
||||
data_snapshot_id="market-returns-20260109-v1",
|
||||
return_frequency="1d",
|
||||
periods_per_year=252,
|
||||
)
|
||||
|
||||
expected_window = returns.loc[:as_of].tail(4)
|
||||
expected = expected_window.dropna(how="any").cov()
|
||||
pd.testing.assert_frame_equal(snapshot.covariance, expected)
|
||||
assert snapshot.snapshot_id.startswith("sample-cov-v1:")
|
||||
assert snapshot.as_of_date == as_of.date()
|
||||
assert snapshot.method == "sample"
|
||||
assert snapshot.window_start_date == expected_window.index[0].date()
|
||||
assert snapshot.window_end_date == as_of.date()
|
||||
assert snapshot.observations == 3
|
||||
assert snapshot.lookback_sessions == 4
|
||||
assert snapshot.missing_policy == "complete_case"
|
||||
assert snapshot.data_snapshot_id == "market-returns-20260109-v1"
|
||||
assert len(snapshot.input_sha256) == 64
|
||||
|
||||
future_changed = returns.copy()
|
||||
future_changed.loc[dates[-1], :] = [1_000_000.0, -1_000_000.0]
|
||||
repeated = estimate_covariance_snapshot(
|
||||
future_changed,
|
||||
as_of_date=as_of,
|
||||
lookback_sessions=4,
|
||||
min_observations=3,
|
||||
data_snapshot_id="market-returns-20260109-v1",
|
||||
return_frequency="1d",
|
||||
periods_per_year=252,
|
||||
)
|
||||
assert repeated.snapshot_id == snapshot.snapshot_id
|
||||
pd.testing.assert_frame_equal(repeated.covariance, snapshot.covariance)
|
||||
|
||||
|
||||
def test_covariance_snapshot_identity_captures_data_and_estimator_contract() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
returns = pd.DataFrame(
|
||||
{"A": [0.01, 0.02, -0.01, 0.03], "B": [0.02, -0.01, 0.01, 0.04]},
|
||||
index=dates,
|
||||
)
|
||||
base = estimate_covariance_snapshot(
|
||||
returns,
|
||||
as_of_date=dates[-1],
|
||||
lookback_sessions=4,
|
||||
min_observations=3,
|
||||
data_snapshot_id="snapshot-a",
|
||||
)
|
||||
different_source = estimate_covariance_snapshot(
|
||||
returns,
|
||||
as_of_date=dates[-1],
|
||||
lookback_sessions=4,
|
||||
min_observations=3,
|
||||
data_snapshot_id="snapshot-b",
|
||||
)
|
||||
|
||||
assert base.snapshot_id != different_source.snapshot_id
|
||||
assert base.covariance.equals(different_source.covariance)
|
||||
|
||||
|
||||
def test_estimate_covariance_snapshot_rejects_ambiguous_or_insufficient_history() -> None:
|
||||
dates = pd.date_range("2026-01-05", periods=4, freq="B")
|
||||
returns = pd.DataFrame(
|
||||
{"A": [0.01, np.nan, 0.03, 0.04], "B": [0.02, 0.01, np.nan, 0.03]},
|
||||
index=dates,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="complete observations"):
|
||||
estimate_covariance_snapshot(
|
||||
returns,
|
||||
as_of_date=dates[-1],
|
||||
lookback_sessions=4,
|
||||
min_observations=3,
|
||||
data_snapshot_id="snapshot-a",
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="strictly increasing"):
|
||||
estimate_covariance_snapshot(
|
||||
returns.iloc[::-1],
|
||||
as_of_date=dates[-1],
|
||||
lookback_sessions=4,
|
||||
min_observations=2,
|
||||
data_snapshot_id="snapshot-a",
|
||||
)
|
||||
|
||||
|
||||
def test_covariance_snapshot_is_validated_and_immutable_by_interface() -> None:
|
||||
covariance = pd.DataFrame(
|
||||
[[0.04, 0.01], [0.01, 0.09]],
|
||||
index=["A", "B"],
|
||||
columns=["A", "B"],
|
||||
)
|
||||
snapshot = CovarianceSnapshot(
|
||||
snapshot_id="cov-20260107-v1",
|
||||
as_of_date="2026-01-07",
|
||||
covariance=covariance,
|
||||
return_frequency="1d",
|
||||
periods_per_year=252,
|
||||
)
|
||||
|
||||
covariance.loc["A", "A"] = 999.0
|
||||
leaked_copy = snapshot.covariance
|
||||
leaked_copy.loc["B", "B"] = 999.0
|
||||
|
||||
assert snapshot.as_of_date == pd.Timestamp("2026-01-07").date()
|
||||
assert snapshot.covariance.loc["A", "A"] == pytest.approx(0.04)
|
||||
assert snapshot.covariance.loc["B", "B"] == pytest.approx(0.09)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("kwargs", "message"),
|
||||
[
|
||||
({"snapshot_id": ""}, "snapshot_id"),
|
||||
({"return_frequency": ""}, "return_frequency"),
|
||||
({"periods_per_year": 0}, "periods_per_year"),
|
||||
],
|
||||
)
|
||||
def test_covariance_snapshot_rejects_incomplete_identity(
|
||||
kwargs: dict[str, object],
|
||||
message: str,
|
||||
) -> None:
|
||||
values: dict[str, object] = {
|
||||
"snapshot_id": "cov-20260107-v1",
|
||||
"as_of_date": "2026-01-07",
|
||||
"covariance": pd.DataFrame([[0.04]], index=["A"], columns=["A"]),
|
||||
"return_frequency": "1d",
|
||||
"periods_per_year": 252,
|
||||
}
|
||||
values.update(kwargs)
|
||||
|
||||
with pytest.raises((TypeError, ValueError), match=message):
|
||||
CovarianceSnapshot(**values)
|
||||
|
||||
|
||||
def test_risk_contribution_sums_to_one_for_positive_portfolio_variance() -> None:
|
||||
weights = np.array([0.5, 0.5])
|
||||
covariance = np.diag([1.0, 4.0])
|
||||
|
||||
result = risk_contribution(weights, covariance)
|
||||
|
||||
np.testing.assert_allclose(result, [0.2, 0.8])
|
||||
assert result.sum() == pytest.approx(1.0)
|
||||
|
||||
|
||||
def test_zero_variance_portfolio_falls_back_to_equal_contribution() -> None:
|
||||
result = risk_contribution(np.array([0.2, 0.3, 0.5]), np.zeros((3, 3)))
|
||||
|
||||
np.testing.assert_allclose(result, np.full(3, 1 / 3))
|
||||
|
||||
|
||||
def test_marginal_and_component_risk_follow_matrix_identities() -> None:
|
||||
weights = np.array([0.25, 0.75])
|
||||
covariance = np.array([[0.04, 0.01], [0.01, 0.09]])
|
||||
|
||||
marginal = marginal_risk_contribution(weights, covariance)
|
||||
component = component_var(weights, covariance)
|
||||
|
||||
np.testing.assert_allclose(marginal, covariance @ weights)
|
||||
np.testing.assert_allclose(component, weights * marginal)
|
||||
assert component.sum() == pytest.approx(weights @ covariance @ weights)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"function",
|
||||
[risk_contribution, marginal_risk_contribution, component_var],
|
||||
)
|
||||
def test_risk_functions_reject_covariance_shape_mismatch(function) -> None:
|
||||
with pytest.raises(ValueError, match="does not match weights length"):
|
||||
function(np.array([0.5, 0.5]), np.eye(3))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"function",
|
||||
[risk_contribution, marginal_risk_contribution, component_var],
|
||||
)
|
||||
def test_risk_functions_reject_empty_portfolio(function) -> None:
|
||||
with pytest.raises(ValueError, match="at least one asset"):
|
||||
function(np.array([]), np.empty((0, 0)))
|
||||
|
||||
|
||||
def test_labeled_component_risk_aligns_covariance_and_closes_to_volatility() -> None:
|
||||
weights = pd.Series({"A": 0.25, "B": 0.75}, name="weight")
|
||||
covariance = pd.DataFrame(
|
||||
[[0.09, 0.01], [0.01, 0.04]],
|
||||
index=["B", "A"],
|
||||
columns=["B", "A"],
|
||||
)
|
||||
|
||||
result = labeled_component_risk(weights, covariance)
|
||||
|
||||
aligned = covariance.reindex(index=weights.index, columns=weights.index)
|
||||
expected_volatility = float(np.sqrt(weights @ aligned @ weights))
|
||||
assert isinstance(result, ComponentRiskResult)
|
||||
assert result.component.index.tolist() == ["A", "B"]
|
||||
assert result.portfolio_volatility == pytest.approx(expected_volatility)
|
||||
assert result.component.sum() == pytest.approx(expected_volatility)
|
||||
assert result.percentage.sum() == pytest.approx(1.0)
|
||||
|
||||
|
||||
def test_component_risk_groups_actual_asset_contributions_by_label() -> None:
|
||||
weights = pd.Series({"A": 0.2, "B": 0.3, "C": 0.5})
|
||||
covariance = pd.DataFrame(np.diag([0.04, 0.09, 0.16]), index=weights.index, columns=weights.index)
|
||||
groups = pd.Series({"C": "growth", "A": "value", "B": "value"})
|
||||
|
||||
result = labeled_component_risk(weights, covariance)
|
||||
grouped = result.grouped_component(groups)
|
||||
|
||||
assert grouped.index.tolist() == ["growth", "value"]
|
||||
assert grouped.loc["value"] == pytest.approx(
|
||||
result.component.loc["A"] + result.component.loc["B"]
|
||||
)
|
||||
assert grouped.sum() == pytest.approx(result.portfolio_volatility)
|
||||
|
||||
|
||||
def test_labeled_component_risk_rejects_asset_label_mismatch() -> None:
|
||||
weights = pd.Series({"A": 0.5, "B": 0.5})
|
||||
covariance = pd.DataFrame(np.eye(2), index=["A", "C"], columns=["A", "C"])
|
||||
|
||||
with pytest.raises(ValueError, match="same asset labels"):
|
||||
labeled_component_risk(weights, covariance)
|
||||
|
||||
|
||||
def test_labeled_component_risk_rejects_invalid_covariance() -> None:
|
||||
weights = pd.Series({"A": 0.5, "B": 0.5})
|
||||
asymmetric = pd.DataFrame([[1.0, 0.2], [0.1, 1.0]], index=weights.index, columns=weights.index)
|
||||
|
||||
with pytest.raises(ValueError, match="symmetric"):
|
||||
labeled_component_risk(weights, asymmetric)
|
||||
|
||||
|
||||
def test_labeled_component_risk_rejects_zero_variance_portfolio() -> None:
|
||||
weights = pd.Series({"A": 0.5, "B": 0.5})
|
||||
covariance = pd.DataFrame(np.zeros((2, 2)), index=weights.index, columns=weights.index)
|
||||
|
||||
with pytest.raises(ValueError, match="positive portfolio variance"):
|
||||
labeled_component_risk(weights, covariance)
|
||||
Reference in New Issue
Block a user