diff --git a/README.md b/README.md index 7714123..3ac93af 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ | 仓库 | 角色 | |---|---| -| `quant_engine` | **纯回测核心**(alpha + execution + indicators + data_adapter + backtest + metrics) | +| `quant_engine` | **纯研究核心**(alpha + execution + ledger + attribution + risk + metrics) | | `research_results` | 业务集成(47 个 proj 调度 + 注册 + 平台对接) | | `tushare2db_pro_aoge` | 数据层(行情 ELT) | | `research_platform` | 展示层(FastAPI + Next.js) | @@ -19,14 +19,18 @@ ## 模块 - `alpha_factors` — 158 alpha 公式 + 24 基础算子(移植自 qlib alpha158) -- `execution` — 执行仿真(成本/滑点/T+1/涨跌停/部分成交/价差)+ 多日 NAV + PnL 拆解(借鉴 hikyuu 部件化思想) +- `execution` — A 股长仓执行仿真(成本/滑点/现金约束)+ 稀疏调仓/完整交易日 Ledger + 可投影成交与 NAV 审计;T+1、涨跌停、成交量与价差提供独立约束函数 - `indicators` — 50+ 技术指标(MACD / KDJ / 布林 / ATR / ADX / 等) - `data_adapter` — 桥接 qtdb_pro 长表与新模块(rename / long-wide / 复权 / vwap 代理) - `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark) -- `metrics` — 绩效(年化收益 / 波动率 / Sharpe / 最大回撤 / Calmar) +- `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表 +- `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排) +- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表与 manifest +- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计 +- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效 - `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test) - `portfolio_decomp` — 组合分解(risk_parity / mean_variance / 因子归因) -- `risk` — 风险指标(边际 / 风险贡献) +- `risk` — ndarray 低层风险公式 + 标签安全、可分组的 Euler 成分风险分解 - `perf_stats` — 详细绩效(与 metrics 并存) - `logging` — 统一 logger(标准库 + 可选 loguru) @@ -58,8 +62,13 @@ ruff check src/ tests/ # lint ```python from quant_engine.alpha_factors import alpha_001, alpha_005, ALPHA158_REGISTRY from quant_engine.execution import ( - ExecutionConfig, simulate_with_daily_data, compute_realized_pnl, + ExecutionConfig, simulate_daily_ledger_with_audit, + simulate_multi_day_with_audit, simulate_with_daily_data, ) +from quant_engine.research_pipeline import ( + run_factor_backtest_research, run_factor_execution_research, +) +from quant_engine.backtest import run_weight_backtest from quant_engine.indicators import macd, bollinger, kdj from quant_engine.data_adapter import ( long_to_wide, wide_to_long, rename_tushare_columns, @@ -70,8 +79,116 @@ from quant_engine.data_adapter import ( # 端到端:qtdb_pro 长表 → 适配 → alpha158 → execution df = load_qtdb_daily(["000001.SZ"], "2024-01-01", with_adj=True) -prices, volumes = prepare_execution_inputs(df) -result = simulate_with_daily_data(prices, initial_cash=1_000_000.0) +close_prices, volumes = prepare_execution_inputs(df) +open_prices, _ = prepare_execution_inputs(df, price_col="open") +result = simulate_with_daily_data(close_prices, initial_cash=1_000_000.0) + +# 已正确滞后的目标权重 → 现金约束执行 → 唯一来源的成交/拒绝/日末持仓/NAV +execution = simulate_multi_day_with_audit( + target_weights_history=[ + ("2024-01-02", {"000001.SZ": 1.0}), + ("2024-01-03", {"000001.SZ": 1.0}), + ], + price_history=[ + ("2024-01-02", {"000001.SZ": 10.0}), + ("2024-01-03", {"000001.SZ": 10.5}), + ], + initial_cash=1_000_000.0, + config=ExecutionConfig(), +) +print(execution.nav_series) +print(execution.daily_executions) + +# 多期因子分数(必须是 point-in-time 数据)→ Top-K → 下一交易日 open 执行 +factor_execution = run_factor_execution_research( + factor_scores, + top_k=20, + execution_prices=open_prices, + execution_price_field="open", + initial_cash=1_000_000.0, +) + +# 推荐研究入口:同一交易日历上显式区分 open 成交和 close 估值。 +# 因子日保持现金,下一交易日成交后的真实持仓才参与当日收盘收益。 +factor_backtest = run_factor_backtest_research( + factor_scores, + top_k=20, + execution_prices=open_prices, + valuation_prices=close_prices, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000_000.0, + config=ExecutionConfig(), +) +print(factor_backtest.nav) +print(factor_backtest.returns) +print(factor_backtest.stats()) +print(factor_backtest.execution.ledger_frame) +print(factor_backtest.execution.trades_frame) +print(factor_backtest.position_weights) # 实际日末资产权重 +print(factor_backtest.cash_weights) + +# 所有分析都以实际成交后的 Ledger 为事实源,不直接使用目标权重伪造结果。 +attribution = factor_backtest.return_attribution() +print(attribution.asset_contributions) +print(attribution.transaction_cost) +print(attribution.residual) # 应接近 0;否则说明贡献未闭合到账本收益 + +# benchmark_returns 必须与成本后 factor_backtest.returns 使用完全相同的日期索引。 +print(factor_backtest.benchmark_stats(benchmark_returns)) + +# 下游稳定交付:显式提供代码版本、数据快照和时区,不在核心层写数据库。 +from quant_engine.artifact import build_research_run_artifact +from quant_engine.data_adapter import prepare_asset_return_snapshot +from quant_engine.risk import estimate_covariance_snapshot + +risk_date = factor_backtest.position_weights.index[-1].date() +market_snapshot = prepare_asset_return_snapshot( + qtdb_daily_long, + source="qtdb_pro.hq_daily", + source_snapshot_id="", + adjustment="qfq", +) +risk_snapshot = estimate_covariance_snapshot( + market_snapshot.returns, + as_of_date=risk_date, + lookback_sessions=252, + min_observations=120, + data_snapshot_id=market_snapshot.data_snapshot_id, +) + +artifact = build_research_run_artifact( + factor_backtest, + run_id="research-run-001", + strategy_id="alpha-top20", + strategy_name="Alpha Top 20", + strategy_version="1.0.0", + engine_version="1.2.0", + code_revision="", + data_snapshot_id=market_snapshot.data_snapshot_id, + calendar="CN-A", + timezone="Asia/Shanghai", + started_at="2026-08-21T10:00:00+08:00", + finished_at="2026-08-21T10:01:00+08:00", + parameters={"top_k": 20, "lag_sessions": 1}, + benchmark_id="000300.SH", + benchmark_returns=benchmark_returns, + risk_snapshots={risk_date: risk_snapshot}, +) +print(artifact.manifest()) + +# run_weight_backtest 是低层算子:只接受收益区间开始前已经生效的持仓权重。 +# 不要把 signal-date 的 factor_scores/decision_weights 直接传给它。 +backtest = run_weight_backtest( + weights=effective_holding_weights, + stock_returns=daily_returns, + initial_capital=1_000_000.0, + benchmark_nav=benchmark_nav, +) +print(factor_execution.schedule.signal_to_execution) +print(factor_execution.execution.daily_executions) +print(backtest.stats()) +print(backtest.benchmark_report()) ``` ## 与 research_results 的关系 diff --git a/docs/OPEN_SOURCE_REFERENCES.md b/docs/OPEN_SOURCE_REFERENCES.md new file mode 100644 index 0000000..645184c --- /dev/null +++ b/docs/OPEN_SOURCE_REFERENCES.md @@ -0,0 +1,64 @@ +# Open-source design references + +本项目采用“借鉴稳定语义、保留轻量实现”的策略。引入新量化能力前先检查成熟 +开源案例;除非维护成本和许可证收益明确优于本地小型实现,否则不增加框架级依赖。 + +## 2026-08-21:成交后归因与相对绩效 + +| 项目 | 借鉴内容 | 当前决策 | +|---|---|---| +| [Qlib](https://github.com/microsoft/qlib) | 信号时间与交易时间分离、成本前后超额收益分开报告 | 借鉴语义;不引入完整框架 | +| [Zipline](https://github.com/quantopian/zipline) | Ledger / transaction / portfolio value 状态模型 | 以现有 `ExecutionSimulationResult` 承担事实源 | +| [empyrical](https://github.com/quantopian/empyrical) | beta 协方差口径、alpha 几何年化、年化因子 | 移植小型公式;不增加老旧运行时依赖 | +| [Riskfolio-Lib](https://github.com/dcajasn/Riskfolio-Lib) | Euler component risk 与分组/因子风险贡献 | 只实现当前需要的 pandas/numpy 标签安全封装 | +| [PyPortfolioOpt](https://github.com/PyPortfolio/PyPortfolioOpt) | 协方差估计与优化器解耦 | 留作未来风险模型适配器参考 | + +当前核心不新增依赖。逐日收益归因必须从实际换仓前后持仓、成交记录、执行价和 +收盘估值推导;因子分数与目标权重只是意图,不能作为成交后归因事实源。 + +## 2026-08-21:研究运行工件 + +- 借鉴 [Qlib Recorder / RecordTemplate](https://github.com/microsoft/qlib/blob/main/qlib/workflow/record_temp.py) + 将 signal、portfolio analysis 和 risk analysis 分成稳定事实,但不引入 Qlib 运行时; +- 借鉴 [MLflow Tracking](https://mlflow.org/docs/latest/tracking/) 的 run / params / + metrics / artifacts 分层,但 MLflow 只保留为未来可选 exporter; +- HTML、PNG 和 tearsheet 是可再生展示物,不能替代 NAV、成交、持仓、归因和绩效事实。 + +因此 `ResearchRunArtifact` 使用显式 `schema_version`、`config_hash`、代码版本和数据 +快照身份,并提供确定性 JSON / SHA-256 manifest;核心层仍不写数据库或 artifact store。 + +schema `1.1.0` 将 Qlib 的独立 risk-analysis artifact 思路与 Riskfolio-Lib 的 Euler +component-risk 语义结合,但只保留本项目需要的轻量合同:协方差快照必须声明 +`snapshot_id`、`as_of_date`、收益频率和年化期数;风险从成交后的实际日末持仓计算, +component risk 闭合到年化组合波动,percentage contribution 闭合到 1。未来日期、资产 +标签不完整和零方差组合都直接失败,不以默认值伪造结果。 + +## 2026-08-21:协方差快照估计 + +| 项目 | 借鉴内容 | 当前决策 | +|---|---|---| +| [PyPortfolioOpt risk models](https://github.com/PyPortfolio/PyPortfolioOpt/blob/main/pypfopt/risk_models.py) | 将收益输入、协方差估计器和组合优化解耦;sample / EWM / shrinkage 使用统一标签输出 | 借鉴可替换估计器边界,不引入完整包 | +| [scikit-learn covariance](https://github.com/scikit-learn/scikit-learn/blob/main/sklearn/covariance/_shrunk_covariance.py) | 维护成熟的 Ledoit–Wolf / OAS shrinkage 实现 | 未来作为可选 adapter;不复制统计公式 | +| [Qlib structured risk model](https://github.com/microsoft/qlib/blob/main/qlib/model/riskmodel/structured.py) | PCA/FA 结构化协方差和固定随机状态 | 留作因子风险模型阶段,不进入当前 baseline | + +当前 `estimate_covariance_snapshot` 只编排 pandas 的 sample covariance:先按 `as_of_date` +截断,再取固定 session 窗口,使用 complete-case 行并拒绝历史不足;禁止 pandas 默认的 +pairwise 样本集合产生含义不一致的矩阵。snapshot ID 对窗口数据、缺失掩码、上游数据 +快照身份和估计参数做 SHA-256,追加未来数据不会改变历史快照。 + +市场适配层现以 `AssetReturnSnapshot` 固化 simple-return 输入:上游 ingestion snapshot ID、 +数据源、价格字段、复权口径、规范化价格值和缺失掩码共同形成内容寻址 ID;不前向填充 +停牌/缺失价格。该 ID 同时传入协方差快照和研究运行工件,避免同一研究链出现两套数据 +身份。 + +可选 shrinkage adapter 的评估结论是“保留边界,暂不实现”:当前运行依赖没有声明 +scikit-learn,本切片也不修改版本或锁文件。未来只有在依赖治理接受后,才以延迟导入 +直接调用 scikit-learn 的 `LedoitWolf` / `OAS`,并让估计器名称、库版本与参数进入 +snapshot identity;不复制成熟统计公式,也不让环境中偶然存在的包改变 baseline 行为。 + +## hikyuu 的定位 + +[hikyuu](https://github.com/fasiondog/hikyuu) 的 SG / MM / CN / PG 部件化思想、 +A 股交易约束和系统组合方式仍有借鉴价值;但其完整 C++/Python 运行时、对象模型和 +数据体系不适合作为本项目核心依赖。当前原则是按真实研究链路吸收边界设计,不复制 +其框架层级,也不为了“架构完整”预先建设尚无端到端需求的抽象。 diff --git a/docs/handoff/2026-08-21-ledger-attribution.md b/docs/handoff/2026-08-21-ledger-attribution.md new file mode 100644 index 0000000..4b6ef8a --- /dev/null +++ b/docs/handoff/2026-08-21-ledger-attribution.md @@ -0,0 +1,42 @@ +# Ledger-backed attribution handoff + +## Goal + +在 `ExecutionSimulationResult` 日频 Ledger 之上增加轻量、可审计的成交后分析层: + +- 逐日隔夜 / 日内资产收益贡献; +- 佣金、印花税、滑点成本独立贡献; +- 贡献闭合到成本后日收益并显式暴露 residual; +- 严格日期对齐的 TE / IR / alpha / beta; +- 标签安全且可分组的 Euler component risk。 +- 从 Ledger 股数和收盘估值投影的实际资产 / 现金权重。 + +## Branch stack + +- 当前:`codex/ledger-attribution-20260821` +- 基线:`codex/post-execution-ledger-20260821` +- 再下层:`codex/core-contracts-20260821`(PR #2,尚待用户确认合并) + +本分支不得直接合并到 `main`。应按上述顺序逐层审阅;未经用户明确确认,不得合并 +L2 PR。 + +## Open-source decision + +调研结论记录在 `docs/OPEN_SOURCE_REFERENCES.md`。Qlib、Zipline、empyrical、 +Riskfolio-Lib 和 PyPortfolioOpt 只作为时间语义、Ledger、相对指标与 Euler 风险贡献 +的设计参考;本阶段没有新增运行时依赖。 + +## Verification + +- `pytest -q --cov=src --cov-report=term-missing`: 514 passed,9 个既有 SciPy warning,91% coverage; +- `mypy --strict src/`: 15 source files passed; +- 变更范围 `ruff check`: passed; +- 全仓 Ruff:仅 13 个既有 `tests/governance/*` PT009; +- workspace verify/status:passed,预期提示 quant_engine 非 main; +- global Gitea workflow check:passed,23 个无关仓库 warning。 + +## Next action + +先按堆叠顺序审阅 PR。基础 Ledger 分支完成后,再将本分支 rebase 到其最终提交, +运行唯一一次 `ship --ready`;随后将稳定输出适配到 `research_results` 与 +`research_platform`,不要在核心层直接写数据库。 diff --git a/docs/handoff/2026-08-21-post-execution-ledger.md b/docs/handoff/2026-08-21-post-execution-ledger.md new file mode 100644 index 0000000..588307c --- /dev/null +++ b/docs/handoff/2026-08-21-post-execution-ledger.md @@ -0,0 +1,33 @@ +# Post-execution daily Ledger handoff + +## 状态 + +- 分支:`codex/post-execution-ledger-20260821` +- 基线:`codex/core-contracts-20260821`(PR #2,尚未获用户确认合并) +- 本分支不得直接合并到 `main`;先等待 PR #2 合并,再整理基线并创建独立 PR。 +- 无账户、券商、数据库或实盘副作用。 + +## 已完成 + +- 新增稀疏调仓、完整交易日估值的 `simulate_daily_ledger_with_audit()`。 +- 显式分离 execution price 与 valuation price,支持下一日 open 成交、当日 close 估值。 +- 成交记录补齐 `side / quantity / price`,并提供 `trades_frame`。 +- 提供平台中立的 `ledger_frame`,不携带 `run_id`,不写数据库。 +- 新增 `run_factor_backtest_research()`:PIT 因子、下一交易日执行、日频 NAV、首日成本收益和标准绩效。 +- 研究区间从首条有效信号日开始,排除因子预热行情对绩效的稀释。 + +## 验证 + +- `pytest -q --cov=src --cov-report=term-missing`:500 passed,total coverage 91%。 +- `mypy --strict src/`:14 source files passed。 +- 本阶段文件 scoped Ruff:passed。 +- 全仓 Ruff:仅既有 governance tests 的 13 个 PT009 基线问题。 +- workspace verify/status:通过;仅提示功能分支不是引导基线 `main`。 +- 全局 Gitea workflow check:通过,23 个既有警告。 + +## 继续步骤 + +1. 获得用户对 PR #2 的明确合并确认并按 L2 流程合并。 +2. 将本分支整理到更新后的 `main`,重新运行相同全量验证。 +3. 为 Ledger 阶段创建独立 PR,执行唯一一次最终 `ship --ready`,等待用户确认合并。 +4. 后续在 `research_results` 增加业务投影适配器,再由 `research_platform` 持久化和展示;核心层继续保持无数据库写入。 diff --git a/docs/handoff/2026-08-21-research-artifact-contract.md b/docs/handoff/2026-08-21-research-artifact-contract.md new file mode 100644 index 0000000..9ec6a8f --- /dev/null +++ b/docs/handoff/2026-08-21-research-artifact-contract.md @@ -0,0 +1,57 @@ +# Research artifact contract handoff + +## Goal + +把完整可信研究链固化成存储中立、版本化、确定性的 `ResearchRunArtifact`,供 +`research_results` 持久化和 `research_platform` 查询: + +- run identity / schema version / config hash / code revision / data snapshot; +- signal scores / decision weights / signal-to-execution mapping; +- NAV / returns / benchmark / costs; +- trades / realized positions / cash; +- asset and daily return attribution; +- performance including Sortino / TE / IR / alpha / beta; +- reproducible covariance snapshots and annualized Euler component-risk facts; +- canonical JSON / SHA-256 manifest。 + +## Branch stack + +- 当前:`codex/research-artifact-contract-20260821` +- 基线:`codex/ledger-attribution-20260821`(Draft PR #4) +- 下层:Draft PR #3 → Ready PR #2 → `main` + +不得绕过堆叠顺序直接合并到 `main`。 + +## Verification + +- `pytest -q`: 540 passed,9 个既有 SciPy warning; +- data-adapter focused coverage 77%(包含未连接真实 ClickHouse 的 I/O 便捷函数); +- `mypy --strict src/`: 16 source files passed; +- changed-scope Ruff: passed; +- no runtime dependency added; +- no database, network, broker or filesystem write side effect in artifact builder。 +- 三仓隔离 ClickHouse 黄金链路通过:市场价格 → return snapshot → covariance → artifact → + publisher → reader;使用随机 localhost 端口、tmpfs 和自动容器清理。 + +## Current risk contract + +- artifact schema:`1.1.0`; +- `CovarianceSnapshot` 对输入矩阵深拷贝并显式记录截至日、频率和年化期数; +- `risk_snapshots` 按研究交易日映射,可只生成需要的风险观察日; +- 使用成交后实际持仓,不包含现金风险资产;协方差资产标签必须与研究资产全集一致; +- `covariance_as_of_date` 不得晚于 `trade_date`;无正组合方差时拒绝产物。 +- `estimate_covariance_snapshot` 从显式数据快照的日收益生成无前视、complete-case、 + SHA-256 可复现的 per-period sample covariance;不包含 I/O 或未来行。 +- `prepare_asset_return_snapshot` 从规范化长表行情生成不前向填充的 simple daily returns; + 显式 ingestion snapshot ID、源/字段/复权口径、价格值和缺失掩码共同形成 + `asset-returns-v1:`,并把同一 ID 传给 covariance 与 run artifact。 +- artifact builder fail closed:每个 `CovarianceSnapshot.data_snapshot_id` 必须与 run 级 + `data_snapshot_id` 完全一致,禁止把其他行情快照的风险分解静默发布到当前研究运行。 +- shrinkage 适配器本轮不实现:scikit-learn 尚非声明依赖,未来只允许薄适配 + `LedoitWolf` / `OAS`,不复制公式、不依赖环境偶然安装状态。 + +## Next action + +保持 Draft PR #5,不绕过堆叠顺序合并;下游 `research_results` / `research_platform` +继续在现有 Draft 分支消费同一数据 lineage。下一阶段优先把 ingestion snapshot ID 从 +真实 ELT 元数据接入调用方,再在依赖治理通过后单独交付可选 shrinkage adapter。 diff --git a/src/quant_engine/artifact.py b/src/quant_engine/artifact.py new file mode 100644 index 0000000..2d2b8f8 --- /dev/null +++ b/src/quant_engine/artifact.py @@ -0,0 +1,580 @@ +"""Versioned, deterministic research-run artifacts for downstream adapters. + +This module is deliberately storage-neutral. It snapshots a completed +``FactorBacktestResult`` into queryable fact tables but never writes a database, +starts a service, or talks to a broker. ``research_results`` owns persistence; +``research_platform`` owns read models and presentation. +""" + +from __future__ import annotations + +import hashlib +import json +import math +from collections.abc import Mapping +from dataclasses import dataclass +from datetime import date, datetime +from typing import Any + +import numpy as np +import pandas as pd + +from quant_engine.research_pipeline import FactorBacktestResult +from quant_engine.risk import CovarianceSnapshot, labeled_component_risk + +RESEARCH_ARTIFACT_SCHEMA_VERSION = "1.1.0" + +RISK_COLUMNS = [ + "run_id", + "trade_date", + "asset_id", + "weight", + "marginal_risk", + "component_risk", + "risk_contribution", + "covariance_snapshot_id", + "covariance_as_of_date", + "risk_measure", + "return_frequency", + "periods_per_year", +] + +__all__ = [ + "RESEARCH_ARTIFACT_SCHEMA_VERSION", + "ResearchRunArtifact", + "build_research_run_artifact", +] + + +def _frame_copy(frame: pd.DataFrame) -> pd.DataFrame: + return frame.copy(deep=True) + + +@dataclass(frozen=True, slots=True, eq=False) +class ResearchRunArtifact: + """Immutable-by-interface snapshot of one completed research run.""" + + schema_version: str + _run: pd.DataFrame + _signals: pd.DataFrame + _nav: pd.DataFrame + _trades: pd.DataFrame + _positions: pd.DataFrame + _attribution: pd.DataFrame + _attribution_daily: pd.DataFrame + _risk: pd.DataFrame + _performance: pd.DataFrame + + @property + def run(self) -> pd.DataFrame: + return _frame_copy(self._run) + + @property + def nav(self) -> pd.DataFrame: + return _frame_copy(self._nav) + + @property + def signals(self) -> pd.DataFrame: + return _frame_copy(self._signals) + + @property + def trades(self) -> pd.DataFrame: + return _frame_copy(self._trades) + + @property + def positions(self) -> pd.DataFrame: + return _frame_copy(self._positions) + + @property + def attribution(self) -> pd.DataFrame: + return _frame_copy(self._attribution) + + @property + def attribution_daily(self) -> pd.DataFrame: + return _frame_copy(self._attribution_daily) + + @property + def risk(self) -> pd.DataFrame: + return _frame_copy(self._risk) + + @property + def performance(self) -> pd.DataFrame: + return _frame_copy(self._performance) + + def table_frames(self) -> Mapping[str, pd.DataFrame]: + """Return isolated table snapshots keyed by stable logical table name.""" + return { + "run": self.run, + "signals": self.signals, + "nav": self.nav, + "trades": self.trades, + "positions": self.positions, + "attribution": self.attribution, + "attribution_daily": self.attribution_daily, + "risk": self.risk, + "performance": self.performance, + } + + def canonical_json(self) -> str: + """Serialize tables deterministically for checksums and artifact storage.""" + payload = { + "schema_version": self.schema_version, + "tables": { + name: _frame_records(frame) + for name, frame in self._internal_table_frames().items() + }, + } + return json.dumps( + payload, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ) + + @property + def content_sha256(self) -> str: + return hashlib.sha256(self.canonical_json().encode("utf-8")).hexdigest() + + def manifest(self) -> Mapping[str, object]: + """Return a compact immutable identity and row-count manifest.""" + return { + "schema_version": self.schema_version, + "run_id": str(self._run.at[0, "run_id"]), + "config_hash": str(self._run.at[0, "config_hash"]), + "content_sha256": self.content_sha256, + "tables": { + name: len(frame) for name, frame in self._internal_table_frames().items() + }, + } + + def _internal_table_frames(self) -> Mapping[str, pd.DataFrame]: + return { + "run": self._run, + "signals": self._signals, + "nav": self._nav, + "trades": self._trades, + "positions": self._positions, + "attribution": self._attribution, + "attribution_daily": self._attribution_daily, + "risk": self._risk, + "performance": self._performance, + } + + +def _required_text(value: str, name: str, *, max_length: int | None = None) -> str: + normalized = value.strip() + if not normalized: + raise ValueError(f"{name} must be non-empty") + if max_length is not None and len(normalized) > max_length: + raise ValueError(f"{name} must contain at most {max_length} characters") + return normalized + + +def _aware_timestamp(value: str | pd.Timestamp, name: str) -> pd.Timestamp: + try: + timestamp = pd.Timestamp(value) + except (TypeError, ValueError) as error: + raise ValueError(f"{name} must be a valid timestamp") from error + if timestamp.tzinfo is None: + raise ValueError(f"{name} must include a timezone") + return timestamp + + +def _json_value(value: object) -> object: + if value is None or isinstance(value, str | bool | int): + return value + if isinstance(value, float): + return value if math.isfinite(value) else None + if isinstance(value, np.generic): + return _json_value(value.item()) + if isinstance(value, pd.Timestamp): + return value.isoformat() + if isinstance(value, datetime): + return value.isoformat() + if isinstance(value, date): + return value.isoformat() + if isinstance(value, Mapping): + return { + str(key): _json_value(item) + for key, item in sorted(value.items(), key=lambda pair: str(pair[0])) + } + if isinstance(value, list | tuple): + return [_json_value(item) for item in value] + raise TypeError(f"value of type {type(value).__name__} is not JSON serializable") + + +def _canonical_mapping_json(values: Mapping[str, object]) -> str: + normalized = _json_value(values) + return json.dumps( + normalized, + ensure_ascii=False, + sort_keys=True, + separators=(",", ":"), + allow_nan=False, + ) + + +def _frame_records(frame: pd.DataFrame) -> list[dict[str, object]]: + return [ + {str(key): _json_value(value) for key, value in row.items()} + for row in frame.to_dict(orient="records") + ] + + +def _build_nav( + result: FactorBacktestResult, + run_id: str, + benchmark_returns: pd.Series | None, +) -> pd.DataFrame: + nav = result.execution.ledger_frame.copy(deep=True) + nav.insert(0, "run_id", run_id) + nav["trade_date"] = pd.to_datetime(nav["trade_date"]).dt.date + nav["total_cost"] = [ + sum(execution.total_cost for execution in daily.executions) + for daily in result.execution.daily_executions + ] + if benchmark_returns is None: + nav["benchmark_nav"] = np.nan + nav["benchmark_return"] = np.nan + nav["excess_ret"] = np.nan + else: + benchmark = benchmark_returns.astype(float, copy=True) + nav["benchmark_nav"] = (1.0 + benchmark).cumprod().to_numpy() + nav["benchmark_return"] = benchmark.to_numpy() + nav["excess_ret"] = result.returns.to_numpy() - benchmark.to_numpy() + return nav + + +def _build_signals(result: FactorBacktestResult, run_id: str) -> pd.DataFrame: + columns = [ + "run_id", + "signal_date", + "execution_date", + "asset_id", + "factor_score", + "target_weight", + ] + rows: list[dict[str, object]] = [] + for signal_date, scores in result.factor_scores.iterrows(): + execution_date = pd.Timestamp(result.schedule.signal_to_execution.at[signal_date]).date() + for asset, score in scores.items(): + rows.append( + { + "run_id": run_id, + "signal_date": pd.Timestamp(signal_date).date(), + "execution_date": execution_date, + "asset_id": asset, + "factor_score": float(score), + "target_weight": float( + result.schedule.decision_weights.at[signal_date, asset] + ), + } + ) + return pd.DataFrame(rows, columns=columns) + + +def _build_trades(result: FactorBacktestResult, run_id: str) -> pd.DataFrame: + trades = result.execution.trades_frame.copy(deep=True) + trades.insert(0, "run_id", run_id) + trades["trade_date"] = pd.to_datetime(trades["trade_date"]).dt.date + trades.insert( + 1, + "trade_id", + [f"{run_id}:{sequence:08d}" for sequence in range(1, len(trades) + 1)], + ) + signal_by_execution = { + pd.Timestamp(execution_date).date(): pd.Timestamp(signal_date).date() + for signal_date, execution_date in result.schedule.signal_to_execution.items() + } + trades["signal_id"] = [ + f"{run_id}:signal:{signal_by_execution[trade_date].isoformat()}" + for trade_date in trades["trade_date"] + ] + trades["total_cost"] = trades["fee"] + trades["slippage"] + return trades + + +def _build_positions(result: FactorBacktestResult, run_id: str) -> pd.DataFrame: + columns = [ + "run_id", + "trade_date", + "asset_id", + "asset_type", + "quantity", + "mark_price", + "market_value", + "weight", + ] + rows: list[dict[str, object]] = [] + weights = result.position_weights + cash_weights = result.cash_weights + for date_value, position in zip( + result.valuation_prices.index, + result.execution.positions, + strict=True, + ): + session_date = pd.Timestamp(date_value).date() + for asset, quantity in position.holdings.items(): + mark_price = float(result.valuation_prices.at[date_value, asset]) + rows.append( + { + "run_id": run_id, + "trade_date": session_date, + "asset_id": asset, + "asset_type": "security", + "quantity": quantity, + "mark_price": mark_price, + "market_value": quantity * mark_price, + "weight": float(weights.at[date_value, asset]), + } + ) + rows.append( + { + "run_id": run_id, + "trade_date": session_date, + "asset_id": "CASH", + "asset_type": "cash", + "quantity": position.cash, + "mark_price": 1.0, + "market_value": position.cash, + "weight": float(cash_weights.at[date_value]), + } + ) + return pd.DataFrame(rows, columns=columns) + + +def _build_attribution( + result: FactorBacktestResult, + run_id: str, +) -> tuple[pd.DataFrame, pd.DataFrame]: + contribution = result.return_attribution() + rows: list[dict[str, object]] = [] + for date_value in contribution.overnight.index: + for asset in contribution.overnight.columns: + overnight = float(contribution.overnight.at[date_value, asset]) + intraday = float(contribution.intraday.at[date_value, asset]) + rows.append( + { + "run_id": run_id, + "trade_date": pd.Timestamp(date_value).date(), + "asset_id": asset, + "overnight": overnight, + "intraday": intraday, + "asset_total": overnight + intraday, + } + ) + daily = pd.DataFrame( + { + "run_id": run_id, + "trade_date": contribution.total_return.index.date, + "transaction_cost": contribution.transaction_cost.to_numpy(), + "explained_return": contribution.explained_return.to_numpy(), + "residual": contribution.residual.to_numpy(), + "total_return": contribution.total_return.to_numpy(), + } + ) + return pd.DataFrame(rows), daily + + +def _risk_trade_date(value: object) -> date: + try: + timestamp = pd.Timestamp(value) + except (TypeError, ValueError) as error: + raise ValueError("risk snapshot keys must be valid trade dates") from error + if pd.isna(timestamp): + raise ValueError("risk snapshot keys must be valid trade dates") + return date(int(timestamp.year), int(timestamp.month), int(timestamp.day)) + + +def _build_risk( + result: FactorBacktestResult, + run_id: str, + data_snapshot_id: str, + risk_snapshots: Mapping[object, CovarianceSnapshot] | None, +) -> pd.DataFrame: + if risk_snapshots is None: + return pd.DataFrame(columns=RISK_COLUMNS) + if not isinstance(risk_snapshots, Mapping): + raise TypeError("risk_snapshots must be a mapping") + + session_by_date = { + pd.Timestamp(session).date(): session for session in result.position_weights.index + } + normalized: dict[date, CovarianceSnapshot] = {} + for raw_trade_date, snapshot in risk_snapshots.items(): + trade_date = _risk_trade_date(raw_trade_date) + if trade_date in normalized: + raise ValueError(f"duplicate risk snapshot trade date: {trade_date}") + if trade_date not in session_by_date: + raise ValueError(f"risk snapshot trade date {trade_date} must be a result session") + if not isinstance(snapshot, CovarianceSnapshot): + raise TypeError("risk snapshot values must be CovarianceSnapshot instances") + if snapshot.as_of_date > trade_date: + raise ValueError( + f"covariance as_of_date {snapshot.as_of_date} must not be after trade date " + f"{trade_date}" + ) + if snapshot.data_snapshot_id != data_snapshot_id: + raise ValueError("covariance snapshot data lineage differs from research run") + normalized[trade_date] = snapshot + + weights_by_date = result.position_weights + rows: list[dict[str, object]] = [] + for trade_date in sorted(normalized): + snapshot = normalized[trade_date] + session = session_by_date[trade_date] + weights = weights_by_date.loc[session].astype(float, copy=True) + annualized_covariance = snapshot.covariance * snapshot.periods_per_year + decomposition = labeled_component_risk(weights, annualized_covariance) + for asset_id in weights.index: + rows.append( + { + "run_id": run_id, + "trade_date": trade_date, + "asset_id": asset_id, + "weight": float(weights.loc[asset_id]), + "marginal_risk": float(decomposition.marginal.loc[asset_id]), + "component_risk": float(decomposition.component.loc[asset_id]), + "risk_contribution": float(decomposition.percentage.loc[asset_id]), + "covariance_snapshot_id": snapshot.snapshot_id, + "covariance_as_of_date": snapshot.as_of_date, + "risk_measure": "annualized_volatility", + "return_frequency": snapshot.return_frequency, + "periods_per_year": snapshot.periods_per_year, + } + ) + return pd.DataFrame(rows, columns=RISK_COLUMNS) + + +def _build_performance( + result: FactorBacktestResult, + run_id: str, + benchmark_returns: pd.Series | None, +) -> pd.DataFrame: + stats = result.stats() + relative = ( + result.benchmark_stats(benchmark_returns) + if benchmark_returns is not None + else { + "tracking_error": float("nan"), + "information_ratio": float("nan"), + "alpha": float("nan"), + "beta": float("nan"), + } + ) + total_return = float(result.nav.iloc[-1] - 1.0) + return pd.DataFrame( + [ + { + "run_id": run_id, + "total_ret": total_return, + "ann_ret": stats["ann_return"], + "ann_volatility": stats["ann_volatility"], + "sharpe": stats["sharpe"], + "sortino": stats["sortino"], + "max_dd": stats["max_drawdown"], + "calmar": stats["calmar"], + "win_rate": stats["win_rate"], + "tracking_error": relative["tracking_error"], + "ir": relative["information_ratio"], + "alpha": relative["alpha"], + "beta": relative["beta"], + "n_trades": len(result.execution.trades_frame), + "n_days": len(result.returns), + } + ] + ) + + +def build_research_run_artifact( + result: FactorBacktestResult, + *, + run_id: str, + strategy_id: str, + strategy_name: str, + strategy_version: str, + engine_version: str, + code_revision: str, + data_snapshot_id: str, + calendar: str, + timezone: str, + started_at: str | pd.Timestamp, + finished_at: str | pd.Timestamp, + parameters: Mapping[str, object], + benchmark_id: str | None = None, + benchmark_returns: pd.Series | None = None, + risk_snapshots: Mapping[object, CovarianceSnapshot] | None = None, +) -> ResearchRunArtifact: + """Snapshot one successful factor backtest into schema-versioned fact tables.""" + if not isinstance(result, FactorBacktestResult): + raise TypeError("result must be a FactorBacktestResult") + if result.nav.empty: + raise ValueError("result must contain at least one research session") + normalized_run_id = _required_text(run_id, "run_id", max_length=64) + normalized_strategy_id = _required_text(strategy_id, "strategy_id") + normalized_strategy_name = _required_text(strategy_name, "strategy_name") + normalized_strategy_version = _required_text(strategy_version, "strategy_version") + normalized_engine_version = _required_text(engine_version, "engine_version") + normalized_code_revision = _required_text(code_revision, "code_revision") + normalized_snapshot = _required_text(data_snapshot_id, "data_snapshot_id") + normalized_calendar = _required_text(calendar, "calendar") + normalized_timezone = _required_text(timezone, "timezone") + if not isinstance(parameters, Mapping): + raise TypeError("parameters must be a mapping") + + started = _aware_timestamp(started_at, "started_at") + finished = _aware_timestamp(finished_at, "finished_at") + if finished < started: + raise ValueError("finished_at must not precede started_at") + if (benchmark_id is None) != (benchmark_returns is None): + raise ValueError("benchmark_id and benchmark_returns must be provided together") + normalized_benchmark = "" + if benchmark_id is not None: + normalized_benchmark = _required_text(benchmark_id, "benchmark_id") + result.benchmark_stats(benchmark_returns) + + params_json = _canonical_mapping_json(parameters) + config_hash = hashlib.sha256(params_json.encode("utf-8")).hexdigest() + run = pd.DataFrame( + [ + { + "schema_version": RESEARCH_ARTIFACT_SCHEMA_VERSION, + "run_id": normalized_run_id, + "strategy_id": normalized_strategy_id, + "strategy_name": normalized_strategy_name, + "strategy_version": normalized_strategy_version, + "engine_version": normalized_engine_version, + "code_revision": normalized_code_revision, + "config_hash": config_hash, + "data_snapshot_id": normalized_snapshot, + "benchmark_id": normalized_benchmark, + "benchmark_alignment_policy": ( + "exact_session_index" if benchmark_returns is not None else "none" + ), + "frequency": "1d", + "calendar": normalized_calendar, + "timezone": normalized_timezone, + "initial_capital": result.execution.initial_cash, + "start_date": result.nav.index[0].date(), + "end_date": result.nav.index[-1].date(), + "status": "success", + "started_at": started, + "finished_at": finished, + "params_json": params_json, + } + ] + ) + attribution, attribution_daily = _build_attribution(result, normalized_run_id) + return ResearchRunArtifact( + schema_version=RESEARCH_ARTIFACT_SCHEMA_VERSION, + _run=run, + _signals=_build_signals(result, normalized_run_id), + _nav=_build_nav(result, normalized_run_id, benchmark_returns), + _trades=_build_trades(result, normalized_run_id), + _positions=_build_positions(result, normalized_run_id), + _attribution=attribution, + _attribution_daily=attribution_daily, + _risk=_build_risk(result, normalized_run_id, normalized_snapshot, risk_snapshots), + _performance=_build_performance(result, normalized_run_id, benchmark_returns), + ) diff --git a/src/quant_engine/attribution.py b/src/quant_engine/attribution.py new file mode 100644 index 0000000..cb0c2c8 --- /dev/null +++ b/src/quant_engine/attribution.py @@ -0,0 +1,141 @@ +"""Post-execution daily return attribution derived from the portfolio ledger. + +The ledger is the source of truth: previous-close holdings explain overnight +PnL, current-close holdings explain intraday PnL, and actual execution costs +remain a separate contribution. Target weights and factor scores are not +accepted here because they are intentions rather than realized positions. +""" + +from __future__ import annotations + +import math +from dataclasses import dataclass + +import pandas as pd + +from quant_engine.execution import ExecutionSimulationResult + +__all__ = ["DailyReturnAttribution", "compute_daily_return_attribution"] + + +@dataclass(frozen=True, slots=True, eq=False) +class DailyReturnAttribution: + """Auditable decomposition of each net portfolio return.""" + + overnight: pd.DataFrame + intraday: pd.DataFrame + transaction_cost: pd.Series + residual: pd.Series + total_return: pd.Series + + @property + def asset_contributions(self) -> pd.DataFrame: + """Return the combined overnight and intraday contribution by asset.""" + return self.overnight + self.intraday + + @property + def explained_return(self) -> pd.Series: + """Return asset contributions plus execution costs, before residual.""" + explained = self.asset_contributions.sum(axis=1) + self.transaction_cost + return explained.rename("explained_return") + + +def _validate_prices( + execution: ExecutionSimulationResult, + execution_prices: pd.DataFrame, + valuation_prices: pd.DataFrame, +) -> pd.DatetimeIndex: + if not isinstance(execution_prices, pd.DataFrame): + raise TypeError("execution_prices must be a pandas DataFrame") + if not isinstance(valuation_prices, pd.DataFrame): + raise TypeError("valuation_prices must be a pandas DataFrame") + if not isinstance(execution_prices.index, pd.DatetimeIndex): + raise TypeError("execution_prices must use a DatetimeIndex") + if not execution_prices.index.equals(valuation_prices.index): + raise ValueError("execution and valuation prices must use matching trading calendars") + if not execution_prices.columns.equals(valuation_prices.columns): + raise ValueError("execution and valuation prices must use matching asset labels") + + ledger_index = pd.DatetimeIndex(pd.Timestamp(position.date) for position in execution.positions) + if not ledger_index.equals(execution_prices.index): + raise ValueError("ledger and price histories must use matching trading calendars") + if len(execution.positions) != len(execution.daily_executions): + raise ValueError("ledger positions and executions must have matching lengths") + return execution_prices.index.copy() + + +def _price_for_held_asset( + prices: pd.DataFrame, + date: pd.Timestamp, + asset: str, + stage: str, +) -> float: + if asset not in prices.columns: + raise ValueError(f"missing {stage} price for held asset {asset} on {date}") + price = float(prices.at[date, asset]) + if not math.isfinite(price) or price <= 0: + raise ValueError(f"invalid {stage} price for held asset {asset} on {date}") + return price + + +def compute_daily_return_attribution( + execution: ExecutionSimulationResult, + execution_prices: pd.DataFrame, + valuation_prices: pd.DataFrame, +) -> DailyReturnAttribution: + """Decompose net daily returns using realized pre/post-execution holdings. + + For each session, previous-close shares earn the move from the previous + close to the current execution price; current-close shares earn the move + from execution price to current close. Actual commissions, stamp tax and + slippage are divided by the same previous NAV denominator. ``residual`` + exposes any failure of those components to close to the ledger return. + """ + index = _validate_prices(execution, execution_prices, valuation_prices) + columns = execution_prices.columns.copy() + overnight = pd.DataFrame(0.0, index=index.copy(), columns=columns) + intraday = pd.DataFrame(0.0, index=index.copy(), columns=columns) + cost = pd.Series(0.0, index=index.copy(), name="transaction_cost") + + previous_holdings: dict[str, float] = {} + previous_nav = execution.initial_cash + for row_number, (date, position, daily) in enumerate( + zip(index, execution.positions, execution.daily_executions, strict=True) + ): + if previous_nav <= 0 or not math.isfinite(previous_nav): + raise ValueError(f"previous portfolio value must be positive and finite on {date}") + + for asset, shares in previous_holdings.items(): + execution_price = _price_for_held_asset( + execution_prices, date, asset, "execution" + ) + previous_close = _price_for_held_asset( + valuation_prices, index[row_number - 1], asset, "previous valuation" + ) + overnight.at[date, asset] = shares * (execution_price - previous_close) / previous_nav + + for asset, shares in position.holdings.items(): + execution_price = _price_for_held_asset( + execution_prices, date, asset, "execution" + ) + close_price = _price_for_held_asset(valuation_prices, date, asset, "valuation") + intraday.at[date, asset] = shares * (close_price - execution_price) / previous_nav + + cost.at[date] = -sum(item.total_cost for item in daily.executions) / previous_nav + previous_holdings = position.holdings + previous_nav = position.portfolio_value + + total_return = pd.Series( + execution.daily_returns.to_numpy(copy=True), + index=index.copy(), + name="total_return", + ) + explained = (overnight + intraday).sum(axis=1) + cost + residual = (total_return - explained).rename("residual") + return DailyReturnAttribution( + overnight=overnight, + intraday=intraday, + transaction_cost=cost, + residual=residual, + total_return=total_return, + ) diff --git a/src/quant_engine/backtest.py b/src/quant_engine/backtest.py index 6becddb..46058c2 100644 --- a/src/quant_engine/backtest.py +++ b/src/quant_engine/backtest.py @@ -13,29 +13,28 @@ ```python from quant_engine.backtest import ( - compute_nav_from_weights, # 调仓表 → 净值 - rebalance_table, # 周期性再平衡 - compare_to_benchmark, # 策略 vs 基准 + rebalance_periodic, # 周期性再平衡 + run_weight_backtest, # 权重 → 统一结果对象 weights_to_long_short, # 多空组合 ) -# 1. 调仓表 → 净值 -nav = compute_nav_from_weights( +# 调仓表 → 净值、收益、绩效与基准报告 +rebalance_table = rebalance_periodic(target_weights, rebalance_dates, returns.index) +result = run_weight_backtest( weights=rebalance_table, # 每周/每月调仓 stock_returns=returns, # 个股日收益 initial_capital=1.0, + benchmark_nav=benchmark_nav, ) - -# 2. 跟基准比 -result = compare_to_benchmark(nav, benchmark_nav) -print(result.summary()) +print(result.stats()) +print(result.benchmark_report()) ``` """ from __future__ import annotations -from pathlib import Path from collections.abc import Mapping, Sequence +from dataclasses import dataclass import numpy as np import pandas as pd @@ -46,6 +45,26 @@ from quant_engine.metrics import summary as metrics_summary logger = get_logger(__name__) +@dataclass(frozen=True, slots=True, eq=False) +class BacktestResult: + """一次权重回测的稳定结果快照。""" + + nav: pd.Series + returns: pd.Series + weights: pd.DataFrame + benchmark_nav: pd.Series | None = None + + def stats(self, rf: float = 0.0) -> Mapping[str, float]: + """返回标准绩效指标。""" + return metrics_summary(self.returns, rf) + + def benchmark_report(self, rf: float = 0.0) -> pd.DataFrame: + """返回策略与基准的对比报告。""" + if self.benchmark_nav is None: + raise ValueError("benchmark_nav is required for benchmark comparison") + return compare_to_benchmark(self.nav, self.benchmark_nav, rf) + + # ── 调仓表 → 净值 ────────────────────────────────────── @@ -57,8 +76,9 @@ def compute_nav_from_weights( ) -> pd.Series: """从调仓表(日期 × 股票权重)+ 个股日收益 → 净值曲线。 - 假设:在调仓日之间权重不变(**前向填充**)。 - 调仓日的权重 = `weights.loc[rebalance_date]`。 + 假设:输入是该收益测量区间开始前已经生效的持仓权重,并在调仓日之间 + 保持不变(**前向填充**)。本函数不会把信号日自动解释为执行日;因子分数 + 应先经交易日历调度和实际执行时点处理,避免把同一时点未知的收益计入。 Args: weights: 调仓日 × 股票代码 的权重 DataFrame(**0~1**,行和 ≤ 1) @@ -113,6 +133,29 @@ def compute_returns_from_nav(nav: pd.Series) -> pd.Series: return nav.pct_change().fillna(0.0) +def run_weight_backtest( + weights: pd.DataFrame, + stock_returns: pd.DataFrame, + initial_capital: float = 1.0, + tc_rate: float = 0.0, + benchmark_nav: pd.Series | None = None, +) -> BacktestResult: + """执行权重回测并返回隔离于调用方输入的结果快照。""" + weights_snapshot = weights.copy(deep=True) + nav = compute_nav_from_weights( + weights=weights_snapshot, + stock_returns=stock_returns, + initial_capital=initial_capital, + tc_rate=tc_rate, + ) + return BacktestResult( + nav=nav, + returns=compute_returns_from_nav(nav), + weights=weights_snapshot, + benchmark_nav=None if benchmark_nav is None else benchmark_nav.copy(deep=True), + ) + + # ── 调仓工具 ────────────────────────────────────── @@ -131,6 +174,9 @@ def rebalance_periodic( Returns: 调仓表 DataFrame(all_dates × 股票代码) """ + if all_dates.empty: + return pd.DataFrame(index=all_dates, columns=target_weights.index, dtype=float) + table = pd.DataFrame(0.0, index=all_dates, columns=target_weights.index) for date in rebalance_dates: if date not in all_dates: @@ -201,6 +247,8 @@ def compare_to_benchmark( """ # 对齐 index common = strategy_nav.index.intersection(benchmark_nav.index) + if common.empty: + raise ValueError("strategy and benchmark must have overlapping dates") s = strategy_nav.loc[common] b = benchmark_nav.loc[common] diff --git a/src/quant_engine/data_adapter.py b/src/quant_engine/data_adapter.py index 9181700..28b3496 100644 --- a/src/quant_engine/data_adapter.py +++ b/src/quant_engine/data_adapter.py @@ -6,22 +6,27 @@ - execution.py 需要**宽表**(date × stock_code)prices / volumes - Tushare 字段命名:`ts_code / vol(手) / amount(千元) / pct_chg`,且**无 vwap 字段** -本模块提供 6 个纯函数,让新模块直接吃 qtdb_pro 真实数据: +本模块提供可组合的数据适配函数,让新模块直接吃 qtdb_pro 真实数据: 1. `long_to_wide()` — 长表 → 宽表(date × stock_code) 2. `wide_to_long()` — 宽表 → 长表 3. `rename_tushare_columns()` — 列名映射(ts_code→stock_code, vol→volume 等) 4. `add_vwap_proxy()` — vwap 代理(Tushare 无 vwap 字段) 5. `apply_adj_factor()` — 复权(hq_daily × hq_adj_factor 前复权) 6. `prepare_stock_series()` — 单股提取(alpha_factors 输入) -7. `prepare_execution_inputs()` — execution 输入(prices + volumes 宽表) -8. `load_qtdb_daily()` — 便捷加载(qtdb_pro.hq_daily + 可选复权) +7. `prepare_asset_return_snapshot()` — 带稳定 lineage 的资产日收益 +8. `prepare_execution_inputs()` — execution 输入(prices + volumes 宽表) +9. `load_qtdb_daily()` — 便捷加载(qtdb_pro.hq_daily + 可选复权) 全部纯 pandas/numpy,零新依赖,mypy strict 兼容。 """ from __future__ import annotations +import hashlib +import json from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from datetime import date from typing import Any import numpy as np @@ -32,12 +37,14 @@ from quant_engine.logging import get_logger logger = get_logger(__name__) __all__ = [ + "AssetReturnSnapshot", "long_to_wide", "wide_to_long", "rename_tushare_columns", "add_vwap_proxy", "apply_adj_factor", "prepare_stock_series", + "prepare_asset_return_snapshot", "prepare_execution_inputs", "load_qtdb_daily", ] @@ -57,6 +64,107 @@ TUSHARE_RENAME: dict[str, str] = { } +@dataclass(frozen=True, slots=True, init=False, eq=False) +class AssetReturnSnapshot: + """Immutable-by-interface daily return matrix with reproducible lineage.""" + + data_snapshot_id: str + source: str + source_snapshot_id: str + price_field: str + adjustment: str + return_method: str + start_date: date + end_date: date + sessions: int + assets: tuple[str, ...] + _returns: pd.DataFrame + + def __init__( + self, + *, + data_snapshot_id: str, + source: str, + source_snapshot_id: str, + price_field: str, + adjustment: str, + return_method: str, + start_date: date, + end_date: date, + assets: tuple[str, ...], + returns: pd.DataFrame, + ) -> None: + for value, name in ( + (data_snapshot_id, "data_snapshot_id"), + (source, "source"), + (source_snapshot_id, "source_snapshot_id"), + (price_field, "price_field"), + (adjustment, "adjustment"), + (return_method, "return_method"), + ): + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{name} must be non-empty") + if returns.empty or not isinstance(returns.index, pd.DatetimeIndex): + raise ValueError("returns must contain a DatetimeIndex and at least one session") + if tuple(returns.columns) != assets: + raise ValueError("assets must match returns columns") + if start_date > end_date: + raise ValueError("start_date must not be after end_date") + + object.__setattr__(self, "data_snapshot_id", data_snapshot_id.strip()) + object.__setattr__(self, "source", source.strip()) + object.__setattr__(self, "source_snapshot_id", source_snapshot_id.strip()) + object.__setattr__(self, "price_field", price_field.strip()) + object.__setattr__(self, "adjustment", adjustment.strip()) + object.__setattr__(self, "return_method", return_method.strip()) + object.__setattr__(self, "start_date", start_date) + object.__setattr__(self, "end_date", end_date) + object.__setattr__(self, "sessions", len(returns)) + object.__setattr__(self, "assets", assets) + object.__setattr__(self, "_returns", returns.copy(deep=True)) + + @property + def returns(self) -> pd.DataFrame: + """Return an isolated copy so callers cannot mutate the snapshot.""" + return self._returns.copy(deep=True) + + +def _non_empty(value: str, name: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{name} must be non-empty") + return value.strip() + + +def _asset_return_snapshot_id( + prices: pd.DataFrame, + *, + source: str, + source_snapshot_id: str, + price_field: str, + adjustment: str, +) -> str: + values = prices.to_numpy(dtype=float, copy=True) + missing = np.isnan(values) + normalized = np.where(missing, 0.0, values).astype(" AssetReturnSnapshot: + """Build deterministic simple daily returns from a long market-price table. + + ``source_snapshot_id`` must identify the upstream ingestion snapshot. The + resulting ID additionally fingerprints canonical price values and their + missing mask, so changed contents cannot retain the same downstream identity. + Missing prices are never forward-filled. + """ + normalized_source = _non_empty(source, "source") + normalized_source_snapshot_id = _non_empty( + source_snapshot_id, + "source_snapshot_id", + ) + normalized_price_col = _non_empty(price_col, "price_col") + normalized_adjustment = _non_empty(adjustment, "adjustment") + if not isinstance(df, pd.DataFrame): + raise TypeError("df must be a pandas DataFrame") + if df.empty: + raise ValueError("df must contain market prices") + required = {date_col, stock_col, normalized_price_col} + missing_columns = sorted(required.difference(df.columns)) + if missing_columns: + raise ValueError(f"prepare_asset_return_snapshot: missing columns={missing_columns}") + + market = df[[date_col, stock_col, normalized_price_col]].copy() + if any(not isinstance(asset, str) or not asset.strip() for asset in market[stock_col]): + raise ValueError("asset labels must be non-empty strings") + market[stock_col] = market[stock_col].str.strip() + try: + normalized_dates = pd.to_datetime(market[date_col], errors="raise") + except (TypeError, ValueError) as error: + raise ValueError("trade dates must be valid dates") from error + if normalized_dates.isna().any(): + raise ValueError("trade dates must be valid dates") + market[date_col] = normalized_dates.dt.normalize() + if market.duplicated(subset=[date_col, stock_col]).any(): + raise ValueError("duplicate asset/session prices are not allowed") + + try: + market[normalized_price_col] = pd.to_numeric( + market[normalized_price_col], + errors="raise", + ) + except (TypeError, ValueError) as error: + raise ValueError("prices must be numeric") from error + observed_prices = market[normalized_price_col].dropna().to_numpy(dtype=float) + if observed_prices.size == 0 or not np.isfinite(observed_prices).all(): + raise ValueError("prices must contain positive finite observations") + if (observed_prices <= 0.0).any(): + raise ValueError("prices must contain positive finite observations") + + prices = market.pivot( + index=date_col, + columns=stock_col, + values=normalized_price_col, + ).sort_index() + prices = prices.reindex(sorted(str(asset) for asset in prices.columns), axis="columns") + prices = prices.astype(float) + if len(prices) < 2: + raise ValueError("market prices must contain at least two sessions") + returns = prices.pct_change(fill_method=None) + assets = tuple(str(asset) for asset in prices.columns) + snapshot_id = _asset_return_snapshot_id( + prices, + source=normalized_source, + source_snapshot_id=normalized_source_snapshot_id, + price_field=normalized_price_col, + adjustment=normalized_adjustment, + ) + return AssetReturnSnapshot( + data_snapshot_id=snapshot_id, + source=normalized_source, + source_snapshot_id=normalized_source_snapshot_id, + price_field=normalized_price_col, + adjustment=normalized_adjustment, + return_method="simple", + start_date=prices.index[0].date(), + end_date=prices.index[-1].date(), + assets=assets, + returns=returns, + ) + + def prepare_execution_inputs( df: pd.DataFrame, stock_col: str = "stock_code", date_col: str = "trade_date", + *, + price_col: str = "close", ) -> tuple[pd.DataFrame, pd.DataFrame]: """长表行情 → execution 输入(prices + volumes 宽表)。 @@ -294,10 +496,11 @@ def prepare_execution_inputs( df: 长表行情(含 close / volume 列,Tushare rename 后) stock_col: 股票代码列名 date_col: 日期列名 + price_col: 执行价字段,默认 close;防前视研究可显式选择下一交易日 open Returns: (prices_wide, volumes_wide): - - prices_wide: date × stock_code,值=close + - prices_wide: date × stock_code,值=price_col - volumes_wide: date × stock_code,值=volume(若无 volume 列则全 1.0) Examples: @@ -313,9 +516,9 @@ def prepare_execution_inputs( """ if df.empty: return pd.DataFrame(), pd.DataFrame() - if "close" not in df.columns: - raise ValueError(f"prepare_execution_inputs: 缺 close 列,实际列={list(df.columns)}") - prices = long_to_wide(df, value_col="close", date_col=date_col, stock_col=stock_col) + if price_col not in df.columns: + raise ValueError(f"prepare_execution_inputs: 缺 {price_col} 列,实际列={list(df.columns)}") + prices = long_to_wide(df, value_col=price_col, date_col=date_col, stock_col=stock_col) if "volume" in df.columns: volumes = long_to_wide(df, value_col="volume", date_col=date_col, stock_col=stock_col) else: diff --git a/src/quant_engine/execution.py b/src/quant_engine/execution.py index 279c030..9fd431e 100644 --- a/src/quant_engine/execution.py +++ b/src/quant_engine/execution.py @@ -10,7 +10,9 @@ 借鉴 hikyuu SG/MM/CN/PG 部件化思想(不引入 hikyuu 框架): - ExecutionConfig:佣金 + 印花税 + 滑点 + 最小交易额 + 止损/止盈阈值 - simulate_execution():从目标权重 → 实际成交金额(应用成本/滑点) -- simulate_multi_day():多日组合仿真(NAV 序列 + 调仓记录) +- simulate_daily_ledger_with_audit():稀疏调仓 + 完整交易日收盘估值 Ledger +- simulate_multi_day_with_audit():目标权重差额调仓(成交/拒绝/持仓/NAV) +- simulate_multi_day():兼容的多日日末持仓快照入口 - check_stop_loss_take_profit():止损/止盈触发判定 - run_end_to_end_poc():signal → 调仓 → 执行 → NAV 端到端 POC @@ -19,8 +21,9 @@ from __future__ import annotations +import math from collections.abc import Mapping -from dataclasses import dataclass +from dataclasses import dataclass, replace from typing import Any import pandas as pd @@ -101,6 +104,9 @@ class ExecutionResult: net_cash_flow: float # 净现金流(买入为负,卖出为正) partial_fill_pct: float = 1.0 # 实际成交占目标的比例(1.0 = 全部成交) blocked_reason: str = "" # 阻塞原因(如涨跌停停牌) + side: str = "" # buy / sell;未成交记录也保留目标方向 + quantity: float = 0.0 # 实际成交股数 + price: float = 0.0 # 未含滑点的参考执行价 def _apply_costs( @@ -344,19 +350,458 @@ class DailyExecution: """单日执行记录。""" date: str - executions: list[ExecutionResult] + executions: tuple[ExecutionResult, ...] nav_before: float nav_after: float rebalance_triggered: bool -def simulate_multi_day( +@dataclass(frozen=True) +class ExecutionSimulationResult: + """单次多日仿真的持仓与执行审计结果。""" + + initial_cash: float + positions: tuple[DailyPosition, ...] + daily_executions: tuple[DailyExecution, ...] + + @property + def nav_series(self) -> pd.Series: + """返回按日期索引的日末 NAV 副本。""" + return pd.Series( + [position.portfolio_value for position in self.positions], + index=[position.date for position in self.positions], + dtype=float, + ) + + @property + def normalized_nav_series(self) -> pd.Series: + """返回以初始资金为 1 的净值曲线副本。""" + nav = self.nav_series + if self.initial_cash == 0: + return pd.Series(0.0, index=nav.index, dtype=float) + return nav / self.initial_cash + + @property + def daily_returns(self) -> pd.Series: + """返回逐日收益;首日相对初始资金计算,保留首日交易成本。""" + nav = self.nav_series + if nav.empty: + return nav + returns = nav.pct_change() + returns.iloc[0] = ( + nav.iloc[0] / self.initial_cash - 1.0 if self.initial_cash != 0 else 0.0 + ) + return returns.fillna(0.0) + + @property + def trades_frame(self) -> pd.DataFrame: + """返回可投影到平台成交明细的实际成交表,不包含纯拒绝记录。""" + columns = [ + "trade_date", + "ts_code", + "side", + "qty", + "price", + "amount", + "fee", + "slippage", + ] + rows = [ + { + "trade_date": daily.date, + "ts_code": execution.stock_code, + "side": execution.side, + "qty": execution.quantity, + "price": execution.price, + "amount": execution.executed_value, + "fee": execution.commission + execution.stamp_tax, + "slippage": execution.slippage_cost, + } + for daily in self.daily_executions + for execution in daily.executions + if execution.quantity > 0 + ] + return pd.DataFrame(rows, columns=columns) + + @property + def ledger_frame(self) -> pd.DataFrame: + """返回稳定的日频 Ledger 投影,不附加运行元数据或写数据库。""" + columns = [ + "trade_date", + "portfolio_value", + "nav", + "pnl", + "pnl_pct", + "position_value", + "cash", + "turnover", + ] + previous_value = self.initial_cash + rows: list[dict[str, float | str]] = [] + daily_returns = self.daily_returns + for index, (position, daily) in enumerate( + zip(self.positions, self.daily_executions, strict=True) + ): + daily_turnover = sum( + execution.executed_value + for execution in daily.executions + if execution.quantity > 0 + ) + turnover_rate = daily_turnover / daily.nav_before if daily.nav_before > 0 else 0.0 + rows.append( + { + "trade_date": position.date, + "portfolio_value": position.portfolio_value, + "nav": ( + position.portfolio_value / self.initial_cash + if self.initial_cash != 0 + else 0.0 + ), + "pnl": position.portfolio_value - previous_value, + "pnl_pct": float(daily_returns.iloc[index]), + "position_value": position.portfolio_value - position.cash, + "cash": position.cash, + "turnover": turnover_rate, + } + ) + previous_value = position.portfolio_value + return pd.DataFrame(rows, columns=columns) + + @property + def total_costs(self) -> float: + """汇总实际成交产生的成本。""" + return sum( + execution.total_cost + for daily in self.daily_executions + for execution in daily.executions + ) + + @property + def total_turnover(self) -> float: + """汇总实际成交金额。""" + return sum( + execution.executed_value + for daily in self.daily_executions + for execution in daily.executions + ) + + @property + def total_rebalances(self) -> int: + """返回至少有一笔实际成交的调仓日数量。""" + return sum(daily.rebalance_triggered for daily in self.daily_executions) + + @property + def final_portfolio_value(self) -> float: + """返回最后一个日末 NAV;空输入时返回初始资金。""" + if not self.positions: + return self.initial_cash + return self.positions[-1].portfolio_value + + @property + def return_pct(self) -> float: + """返回相对初始资金的百分比收益。""" + if self.initial_cash == 0: + return 0.0 + return (self.final_portfolio_value / self.initial_cash - 1.0) * 100.0 + + +def _blocked_execution(stock_code: str, target_value: float, reason: str) -> ExecutionResult: + """构造未成交但可审计的执行记录。""" + return ExecutionResult( + stock_code=stock_code, + target_value=target_value, + executed_value=0.0, + commission=0.0, + stamp_tax=0.0, + slippage_cost=0.0, + total_cost=0.0, + net_cash_flow=0.0, + partial_fill_pct=0.0, + blocked_reason=reason, + side="buy" if target_value > 0 else "sell" if target_value < 0 else "", + ) + + +def _validate_target_weights(date: str, targets: Mapping[str, float]) -> dict[str, float]: + """校验并复制单日长仓目标权重。""" + normalized: dict[str, float] = {} + for stock_code, raw_weight in targets.items(): + try: + weight = float(raw_weight) + except (TypeError, ValueError) as error: + raise ValueError(f"target weights on {date!r} must be numeric") from error + if not math.isfinite(weight) or weight < 0: + raise ValueError(f"target weights on {date!r} must be finite and non-negative") + normalized[stock_code] = weight + if sum(normalized.values()) > 1.0 + 1e-12: + raise ValueError(f"target weights on {date!r} must sum to at most 1.0") + return normalized + + +def _partially_fill_buy( + desired: ExecutionResult, + fill_pct: float, + config: ExecutionConfig, +) -> ExecutionResult: + """按同一比例缩放买入,保留原始目标金额供审计。""" + actual_target_value = desired.target_value * fill_pct + executed_value, commission, stamp_tax, slippage_cost = _apply_costs( + actual_target_value, + True, + config, + ) + total_cost = commission + stamp_tax + slippage_cost + return ExecutionResult( + stock_code=desired.stock_code, + target_value=desired.target_value, + executed_value=executed_value, + commission=commission, + stamp_tax=stamp_tax, + slippage_cost=slippage_cost, + total_cost=total_cost, + net_cash_flow=-(executed_value + commission + stamp_tax), + partial_fill_pct=fill_pct, + blocked_reason="insufficient_cash_partial_fill", + ) + + +def _rebalance_at_prices( + date: str, + targets: Mapping[str, float], + prices: Mapping[str, float], + cash: float, + holdings: dict[str, float], + config: ExecutionConfig, +) -> tuple[float, tuple[ExecutionResult, ...], float, float]: + """在单一执行时点按目标权重差额调仓,并原地更新 holdings。""" + normalized_targets = _validate_target_weights(date, targets) + for held_code in holdings: + held_price = prices.get(held_code) + if held_price is None or not math.isfinite(held_price) or held_price <= 0: + raise ValueError(f"missing price for held asset {held_code} on {date!r}") + + nav_before = cash + sum( + shares * prices.get(stock_code, 0.0) + for stock_code, shares in holdings.items() + ) + effective_targets = dict.fromkeys(holdings, 0.0) + effective_targets.update(normalized_targets) + buy_weights: dict[str, float] = {} + sell_weights: dict[str, float] = {} + rejected: list[ExecutionResult] = [] + + for stock_code, target_weight in effective_targets.items(): + price = prices.get(stock_code) + target_value = float(target_weight) * nav_before + if price is None or not math.isfinite(price) or price <= 0: + if target_value != 0 or holdings.get(stock_code, 0.0) != 0: + rejected.append(_blocked_execution(stock_code, target_value, "missing_price")) + continue + + current_value = holdings.get(stock_code, 0.0) * price + trade_value = target_value - current_value + if abs(trade_value) < config.min_trade_amount or math.isclose( + trade_value, 0.0, abs_tol=1e-12 + ): + continue + if nav_before == 0: + rejected.append(_blocked_execution(stock_code, trade_value, "zero_nav")) + continue + destination = buy_weights if trade_value > 0 else sell_weights + destination[stock_code] = trade_value / nav_before + + sell_executions = simulate_execution(sell_weights, nav_before, config) + filled: list[ExecutionResult] = [] + for raw_execution in sell_executions: + price = prices[raw_execution.stock_code] + quantity = abs(raw_execution.target_value) / price + execution = replace( + raw_execution, + side="sell", + quantity=quantity, + price=price, + ) + held = holdings.get(execution.stock_code, 0.0) + holdings[execution.stock_code] = max(0.0, held - quantity) + if holdings[execution.stock_code] < 1e-6: + del holdings[execution.stock_code] + cash += execution.net_cash_flow + filled.append(execution) + + desired_buys = simulate_execution(buy_weights, nav_before, config) + required_cash = sum(-execution.net_cash_flow for execution in desired_buys) + buy_fill_pct = min(1.0, max(cash, 0.0) / required_cash) if required_cash > 0 else 1.0 + for desired in desired_buys: + if buy_fill_pct == 0: + rejected.append( + _blocked_execution(desired.stock_code, desired.target_value, "insufficient_cash") + ) + continue + raw_execution = ( + desired + if buy_fill_pct == 1.0 + else _partially_fill_buy(desired, buy_fill_pct, config) + ) + price = prices[raw_execution.stock_code] + quantity = abs(raw_execution.target_value) * raw_execution.partial_fill_pct / price + execution = replace( + raw_execution, + side="buy", + quantity=quantity, + price=price, + ) + holdings[execution.stock_code] = holdings.get(execution.stock_code, 0.0) + quantity + cash += execution.net_cash_flow + if math.isclose(cash, 0.0, abs_tol=1e-9): + cash = 0.0 + filled.append(execution) + + executions = (*filled, *rejected) + nav_after = cash + sum( + shares * prices.get(stock_code, 0.0) + for stock_code, shares in holdings.items() + ) + return cash, executions, nav_before, nav_after + + +def _validate_sparse_daily_histories( + target_weights_history: list[tuple[str, dict[str, float]]], + execution_price_history: list[tuple[str, dict[str, float]]], + valuation_price_history: list[tuple[str, dict[str, float]]], +) -> tuple[ + dict[str, dict[str, float]], + dict[str, dict[str, float]], + list[tuple[str, dict[str, float]]], +]: + """校验稀疏调仓与完整估值日历,并隔离调用方可变输入。""" + target_dates = [date for date, _ in target_weights_history] + execution_dates = [date for date, _ in execution_price_history] + valuation_dates = [date for date, _ in valuation_price_history] + if len(set(target_dates)) != len(target_dates): + raise ValueError("target_weights_history must contain unique dates") + if len(set(execution_dates)) != len(execution_dates): + raise ValueError("execution_price_history must contain unique dates") + if len(set(valuation_dates)) != len(valuation_dates): + raise ValueError("valuation_price_history must contain unique dates") + if execution_dates != target_dates: + raise ValueError("execution price dates must exactly match target weight dates") + + valuation_positions = {date: index for index, date in enumerate(valuation_dates)} + missing_dates = [date for date in target_dates if date not in valuation_positions] + if missing_dates: + raise ValueError(f"target dates must belong to valuation calendar: {missing_dates}") + positions = [valuation_positions[date] for date in target_dates] + if positions != sorted(positions): + raise ValueError("target weights must follow valuation calendar order") + + targets = {date: dict(values) for date, values in target_weights_history} + execution_prices = {date: dict(values) for date, values in execution_price_history} + valuation_prices = [(date, dict(values)) for date, values in valuation_price_history] + return targets, execution_prices, valuation_prices + + +def _simulate_daily_ledger( + target_weights_history: list[tuple[str, dict[str, float]]], + execution_price_history: list[tuple[str, dict[str, float]]], + valuation_price_history: list[tuple[str, dict[str, float]]], + initial_cash: float, + config: ExecutionConfig, +) -> ExecutionSimulationResult: + targets_by_date, execution_prices_by_date, valuation_history = ( + _validate_sparse_daily_histories( + target_weights_history, + execution_price_history, + valuation_price_history, + ) + ) + cash = initial_cash + holdings: dict[str, float] = {} + positions: list[DailyPosition] = [] + daily_executions: list[DailyExecution] = [] + + for date, valuation_prices in valuation_history: + targets = targets_by_date.get(date) + if targets is None: + executions: tuple[ExecutionResult, ...] = () + nav_before = 0.0 + nav_after = 0.0 + rebalance_triggered = False + else: + cash, executions, nav_before, nav_after = _rebalance_at_prices( + date, + targets, + execution_prices_by_date[date], + cash, + holdings, + config, + ) + rebalance_triggered = any(execution.quantity > 0 for execution in executions) + + for held_code in holdings: + valuation_price = valuation_prices.get(held_code) + if ( + valuation_price is None + or not math.isfinite(valuation_price) + or valuation_price <= 0 + ): + raise ValueError( + f"missing valuation price for held asset {held_code} on {date!r}" + ) + portfolio_value = cash + sum( + shares * valuation_prices[stock_code] + for stock_code, shares in holdings.items() + ) + if targets is None: + nav_before = portfolio_value + nav_after = portfolio_value + positions.append(DailyPosition(date, cash, dict(holdings), portfolio_value)) + daily_executions.append( + DailyExecution( + date=date, + executions=executions, + nav_before=nav_before, + nav_after=nav_after, + rebalance_triggered=rebalance_triggered, + ) + ) + + return ExecutionSimulationResult( + initial_cash=initial_cash, + positions=tuple(positions), + daily_executions=tuple(daily_executions), + ) + + +def simulate_daily_ledger_with_audit( + target_weights_history: list[tuple[str, dict[str, float]]], + execution_price_history: list[tuple[str, dict[str, float]]], + valuation_price_history: list[tuple[str, dict[str, float]]], + initial_cash: float, + config: ExecutionConfig | None = None, +) -> ExecutionSimulationResult: + """以稀疏调仓和完整日历运行成交后持仓 Ledger。 + + 执行价只用于调仓日现金与股数变化,估值价用于每个交易日日末 NAV;二者 + 显式分离,从而支持“下一日 open 成交、同日 close 估值”的无前视研究。 + """ + if not math.isfinite(initial_cash) or initial_cash <= 0: + raise ValueError(f"initial_cash must be positive and finite, got {initial_cash}") + return _simulate_daily_ledger( + target_weights_history, + execution_price_history, + valuation_price_history, + initial_cash, + ExecutionConfig() if config is None else config, + ) + + +def simulate_multi_day_with_audit( target_weights_history: list[tuple[str, dict[str, float]]], price_history: list[tuple[str, dict[str, float]]], initial_cash: float, config: ExecutionConfig | None = None, -) -> list[DailyPosition]: - """多日组合仿真(NAV 序列)。 +) -> ExecutionSimulationResult: + """按目标权重差额推进组合,并返回唯一事实来源的审计结果。 Args: target_weights_history: [(date, {stock_code: target_weight})] @@ -365,97 +810,45 @@ def simulate_multi_day( config: 执行配置 Returns: - DailyPosition 列表(每日 NAV 快照)。 + 日末持仓快照与逐日成交记录组成的结构化审计结果。 Note: - 调仓频率 = target_weights_history 的频率(每日 / 每周 / 每月都行) - - 每日先按当日 close 估值,再按当日 target 调仓(下一交易日生效) - - 此处简化:调仓使用当日 close 价格 + - 每日先按当日 close 估值,再交易“目标市值 - 当前市值”的差额 + - 此处简化为当日 close 成交;调用方必须传入已正确滞后的目标权重 """ - if config is None: - config = ExecutionConfig() + if not math.isfinite(initial_cash) or initial_cash < 0: + raise ValueError(f"initial_cash must be finite and non-negative, got {initial_cash}") if len(target_weights_history) != len(price_history): raise ValueError("target_weights_history and price_history must have same length") - if not target_weights_history: - return [] - cash = initial_cash - holdings: dict[str, float] = {} - positions: list[DailyPosition] = [] - for (date, targets), (_, prices) in zip(target_weights_history, price_history, strict=True): - # 1) 先按当日收盘价估值 - portfolio_value = cash + sum( - shares * prices.get(code, 0.0) for code, shares in holdings.items() - ) - positions.append( - DailyPosition( - date=date, - cash=cash, - holdings=dict(holdings), - portfolio_value=portfolio_value, + for (date, _), (price_date, _) in zip(target_weights_history, price_history, strict=True): + if date != price_date: + raise ValueError( + f"target and price dates must match, got {date!r} and {price_date!r}" ) - ) - # 2) 计算 effective_targets(包含需要平仓的零权重) - effective_targets: dict[str, float] = dict(targets) - for held_code in holdings: - if held_code not in effective_targets: - effective_targets[held_code] = 0.0 - # 3) 调仓(只对非零目标调用 simulate_execution) - non_zero_targets = {k: v for k, v in effective_targets.items() if v != 0} - results = simulate_execution(non_zero_targets, portfolio_value, config) - # 4) 处理零目标(平仓):构造 ExecutionResult,shares = held(全部卖出) - for stock_code, weight in effective_targets.items(): - if weight == 0 and stock_code in holdings and holdings[stock_code] > 0: - price = prices.get(stock_code, 0.0) - if price > 0: - held = holdings[stock_code] - # 全部卖出:target_shares = held - # executed_value = held * price(考虑滑点) - slippage_factor = 1.0 - config.slippage_bps / 10000.0 - target_value = -held * price - executed_value = target_value * slippage_factor - commission = abs(executed_value) * config.commission_bps / 10000.0 - stamp_tax = abs(executed_value) * config.stamp_tax_bps / 10000.0 - slippage_cost = abs(executed_value - target_value) - # 标记净卖出 shares = held - results.append( - ExecutionResult( - stock_code=stock_code, - target_value=target_value, - executed_value=executed_value, - commission=commission, - stamp_tax=stamp_tax, - slippage_cost=slippage_cost, - total_cost=commission + stamp_tax + slippage_cost, - net_cash_flow=executed_value - commission - stamp_tax, - ) - ) - # 5) 应用执行结果到持仓 - for r in results: - cost = r.executed_value + r.commission + r.stamp_tax - proceeds = r.executed_value - r.commission - r.stamp_tax - price = prices.get(r.stock_code, 0.0) - if r.target_value > 0: - # 买入:shares = 正数 executed_value / price,cash 减少 cost - shares = r.executed_value / price if price > 0 else 0.0 - holdings[r.stock_code] = holdings.get(r.stock_code, 0.0) + shares - cash -= cost - else: - # 卖出:cash 增加 proceeds 的绝对值(proceeds 本是负的) - held = holdings.get(r.stock_code, 0.0) - if held > 0: - # 如果是 zero-target 触发的全卖(target_value 与持仓市值近似),全部卖出 - if abs(r.target_value) >= held * price * 0.95: - sell_shares = held - else: - target_shares = abs(r.executed_value) / price if price > 0 else held - sell_shares = min(held, target_shares) - holdings[r.stock_code] = held - sell_shares - if holdings[r.stock_code] < 1e-6: - del holdings[r.stock_code] - # proceeds 是负的(target_value 负),cash += proceeds 实际是减去 - # 但卖出是现金流入,所以应该 cash += abs(proceeds) - cash += abs(proceeds) - return positions + return _simulate_daily_ledger( + target_weights_history, + price_history, + price_history, + initial_cash, + ExecutionConfig() if config is None else config, + ) + + +def simulate_multi_day( + target_weights_history: list[tuple[str, dict[str, float]]], + price_history: list[tuple[str, dict[str, float]]], + initial_cash: float, + config: ExecutionConfig | None = None, +) -> list[DailyPosition]: + """兼容入口:返回多日仿真的日末持仓快照。""" + result = simulate_multi_day_with_audit( + target_weights_history, + price_history, + initial_cash, + config, + ) + return list(result.positions) def run_end_to_end_poc( @@ -486,64 +879,16 @@ def run_end_to_end_poc( config = ExecutionConfig() if len(signals) != len(prices): raise ValueError("signals and prices must have same length") - positions = simulate_multi_day(signals, prices, initial_cash, config) - nav_series = pd.Series( - [p.portfolio_value for p in positions], index=[p.date for p in positions] - ) - # 计算 total_costs / total_turnover(重放所有执行) - total_cost_acc = 0.0 - total_turnover_acc = 0.0 - rebalance_count = 0 - cash = initial_cash - holdings: dict[str, float] = {} - for (date, targets), (_, price_map) in zip(signals, prices, strict=True): - portfolio_value = cash + sum( - shares * price_map.get(code, 0.0) for code, shares in holdings.items() - ) - if targets: - rebalance_count += 1 - # 自动平仓:持仓但不在 target 中的股票 - effective_targets: dict[str, float] = dict(targets) - for held_code in holdings: - if held_code not in effective_targets: - effective_targets[held_code] = 0.0 - results = simulate_execution(effective_targets, portfolio_value, config) - total_cost_acc += total_costs(results) - total_turnover_acc += total_turnover(results) - for r in results: - cost = r.executed_value + r.commission + r.stamp_tax - proceeds = r.executed_value - r.commission - r.stamp_tax - if r.target_value > 0: - shares = ( - r.executed_value / price_map[r.stock_code] - if price_map[r.stock_code] > 0 - else 0.0 - ) - holdings[r.stock_code] = holdings.get(r.stock_code, 0.0) + shares - cash -= cost - else: - held = holdings.get(r.stock_code, 0.0) - if held > 0: - sell_shares = min( - held, - abs(r.executed_value / price_map[r.stock_code]) - if price_map[r.stock_code] > 0 - else held, - ) - holdings[r.stock_code] = held - sell_shares - if holdings[r.stock_code] < 1e-6: - del holdings[r.stock_code] - cash += proceeds + audit = simulate_multi_day_with_audit(signals, prices, initial_cash, config) return { - "positions": positions, - "nav_series": nav_series, - "total_costs": total_cost_acc, - "total_turnover": total_turnover_acc, - "total_rebalances": rebalance_count, - "final_portfolio_value": nav_series.iloc[-1] if len(nav_series) > 0 else initial_cash, - "return_pct": ((nav_series.iloc[-1] / initial_cash) - 1) * 100 - if len(nav_series) > 0 - else 0.0, + "positions": list(audit.positions), + "daily_executions": list(audit.daily_executions), + "nav_series": audit.nav_series, + "total_costs": audit.total_costs, + "total_turnover": audit.total_turnover, + "total_rebalances": audit.total_rebalances, + "final_portfolio_value": audit.final_portfolio_value, + "return_pct": audit.return_pct, } @@ -667,7 +1012,10 @@ __all__ = [ "apply_bid_ask_spread", "DailyPosition", "DailyExecution", + "ExecutionSimulationResult", + "simulate_daily_ledger_with_audit", "simulate_multi_day", + "simulate_multi_day_with_audit", "run_end_to_end_poc", "DailyPnL", "simulate_with_daily_data", diff --git a/src/quant_engine/factor_library.py b/src/quant_engine/factor_library.py index 4fa4c59..2de5538 100644 --- a/src/quant_engine/factor_library.py +++ b/src/quant_engine/factor_library.py @@ -312,8 +312,8 @@ def ols_regress( ss_tot = float(((y_arr - y_arr.mean()) ** 2).sum()) r_sq = 1.0 - ss_res / ss_tot if ss_tot > 0 else np.nan sigma2 = ss_res / max(n - k, 1) - # 协方差矩阵 = sigma2 * (X'X)^-1 - xtx_inv = np.linalg.inv(x_arr.T @ x_arr) if sigma2 > 0 else np.full((k, k), np.nan) + # 广义协方差矩阵 = sigma2 * (X'X)^+,伪逆兼容共线因子。 + xtx_inv = np.linalg.pinv(x_arr.T @ x_arr) if sigma2 > 0 else np.full((k, k), np.nan) se = np.sqrt(np.diag(xtx_inv) * sigma2) t_vals = coef / se if sigma2 > 0 else np.full_like(coef, np.nan) if add_constant: @@ -513,6 +513,8 @@ def apply_factor_direction( Returns: 方向调整后的因子(同向 = 越大越好) """ + if direction not in {"auto", "forward", "reverse"}: + raise ValueError(f"direction={direction!r} not supported (auto / forward / reverse)") if factor.empty: return factor.copy() if direction == "auto": @@ -541,6 +543,8 @@ def cross_sectional_rank_with_direction( Returns: pd.Series(百分位排名 [0, 1],越大越优) """ + if direction not in {"auto", "forward", "reverse"}: + raise ValueError(f"direction={direction!r} not supported (auto / forward / reverse)") if df.empty or factor_col not in df.columns: return pd.Series(dtype=float) factor = df[factor_col] diff --git a/src/quant_engine/metrics.py b/src/quant_engine/metrics.py index 270ceea..3557214 100644 --- a/src/quant_engine/metrics.py +++ b/src/quant_engine/metrics.py @@ -65,13 +65,28 @@ def sharpe_ratio(r: pd.Series, rf: float = 0.0) -> float: return (annualized_return(r) - rf) / vol +def sortino_ratio(r: pd.Series, rf: float = 0.0) -> float: + """Sortino = (年化收益 - rf) / 年化下行偏差。""" + r = _clean(r) + if len(r) < 2: + return 0.0 + downside = np.minimum(r.to_numpy(dtype=float), 0.0) + downside_deviation = float( + np.sqrt(np.mean(np.square(downside))) * np.sqrt(TRADING_DAYS_PER_YEAR) + ) + if downside_deviation == 0: + return 0.0 + return (annualized_return(r) - rf) / downside_deviation + + def max_drawdown(r: pd.Series) -> float: """最大回撤(负数)。例如 -0.2 表示最大亏 20%。""" r = _clean(r) if len(r) < 2: return 0.0 nav = (1 + r).cumprod() - peak = nav.cummax() + # 初始资金净值为 1;否则首个观测日的亏损会被误当成新的历史高点。 + peak = nav.cummax().clip(lower=1.0) drawdown = (nav - peak) / peak return float(drawdown.min()) @@ -119,6 +134,7 @@ def summary(r: pd.Series, rf: float = 0.0) -> Mapping[str, float]: "ann_return": ann_ret, "ann_volatility": ann_vol, "sharpe": sharpe_ratio(r, rf), + "sortino": sortino_ratio(r, rf), "max_drawdown": mdd, "calmar": calmar_ratio(r), "win_rate": win_rate(r), @@ -129,6 +145,62 @@ def summary(r: pd.Series, rf: float = 0.0) -> Mapping[str, float]: } +def benchmark_summary( + portfolio_returns: pd.Series, + benchmark_returns: pd.Series, + *, + risk_free_daily: float = 0.0, + annualization: int = TRADING_DAYS_PER_YEAR, +) -> Mapping[str, float]: + """计算成本后组合相对基准的严格对齐绩效。 + + 与通用 ``summary`` 不同,本函数拒绝静默清洗或日期 inner join。alpha + 使用日频回归截距的几何年化;基准方差不足时 alpha/beta 为 NaN,明确 + 表示回归不可估计。 + """ + portfolio, benchmark = _validate_benchmark_inputs( + portfolio_returns, + benchmark_returns, + ) + if isinstance(annualization, bool) or not isinstance(annualization, int): + raise TypeError("annualization must be an integer") + if annualization <= 0: + raise ValueError("annualization must be positive") + if not np.isfinite(risk_free_daily): + raise ValueError("risk_free_daily must be finite") + + active = portfolio - benchmark + active_std = float(active.std()) + tracking_error = active_std * float(np.sqrt(annualization)) + information_ratio = ( + float(active.mean()) / active_std * float(np.sqrt(annualization)) + if active_std >= 1e-30 + else float("nan") + ) + + adjusted_portfolio = portfolio - risk_free_daily + adjusted_benchmark = benchmark - risk_free_daily + benchmark_variance = float(adjusted_benchmark.var()) + if benchmark_variance < 1e-30: + beta = float("nan") + alpha = float("nan") + else: + beta = float(adjusted_portfolio.cov(adjusted_benchmark) / benchmark_variance) + alpha_daily = float((adjusted_portfolio - beta * adjusted_benchmark).mean()) + alpha = ( + float((1.0 + alpha_daily) ** annualization - 1.0) + if alpha_daily > -1.0 + else float("nan") + ) + return { + "n_observations": len(portfolio), + "tracking_error": tracking_error, + "information_ratio": information_ratio, + "alpha": alpha, + "beta": beta, + } + + # ── 内部 ────────────────────────────────────── @@ -137,3 +209,29 @@ def _clean(r: pd.Series) -> pd.Series: if not isinstance(r, pd.Series): raise TypeError(f"expected pd.Series, got {type(r).__name__}") return r.replace([np.inf, -np.inf], np.nan).dropna() + + +def _validate_benchmark_inputs( + portfolio_returns: pd.Series, + benchmark_returns: pd.Series, +) -> tuple[pd.Series, pd.Series]: + if not isinstance(portfolio_returns, pd.Series): + raise TypeError("portfolio_returns must be a pandas Series") + if not isinstance(benchmark_returns, pd.Series): + raise TypeError("benchmark_returns must be a pandas Series") + if not portfolio_returns.index.equals(benchmark_returns.index): + raise ValueError("portfolio and benchmark returns must use matching indexes") + if not portfolio_returns.index.is_unique: + raise ValueError("portfolio and benchmark indexes must be unique") + if len(portfolio_returns) < 2: + raise ValueError("benchmark metrics require at least two observations") + + portfolio = portfolio_returns.astype(float, copy=True) + benchmark = benchmark_returns.astype(float, copy=True) + if not np.isfinite(portfolio.to_numpy()).all() or not np.isfinite( + benchmark.to_numpy() + ).all(): + raise ValueError("portfolio and benchmark returns must be finite") + if (portfolio < -1.0).any() or (benchmark < -1.0).any(): + raise ValueError("simple returns cannot be less than -1") + return portfolio, benchmark diff --git a/src/quant_engine/portfolio_construction.py b/src/quant_engine/portfolio_construction.py new file mode 100644 index 0000000..a58cad4 --- /dev/null +++ b/src/quant_engine/portfolio_construction.py @@ -0,0 +1,104 @@ +"""因子分数到目标权重的轻量组合构建闭环。""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +from pandas.api.types import is_numeric_dtype + +__all__ = [ + "select_top_k", + "equal_weight", + "scores_to_target_weights", + "scores_to_weight_table", +] + + +def _validate_top_k(top_k: int) -> None: + if isinstance(top_k, bool) or not isinstance(top_k, int) or top_k <= 0: + raise ValueError("top_k must be positive") + + +def _validate_gross_exposure(gross_exposure: float) -> None: + if not np.isfinite(gross_exposure) or gross_exposure < 0: + raise ValueError("gross_exposure must be finite and non-negative") + + +def _validate_score_series(scores: pd.Series) -> None: + if not isinstance(scores, pd.Series): + raise TypeError(f"scores must be a pandas Series, got {type(scores).__name__}") + if not scores.index.is_unique: + raise ValueError("scores must contain unique asset labels") + if not is_numeric_dtype(scores.dtype): + raise TypeError("scores must contain numeric values") + + +def select_top_k(scores: pd.Series, top_k: int, *, largest: bool = True) -> pd.Index: + """稳定选择最高或最低的 K 个有效因子分数。""" + _validate_top_k(top_k) + _validate_score_series(scores) + valid_scores = scores.dropna() + ordered = valid_scores.sort_values(ascending=not largest, kind="mergesort") + return ordered.iloc[:top_k].index.copy() + + +def equal_weight(assets: pd.Index, *, gross_exposure: float = 1.0) -> pd.Series: + """在已选资产间等权分配指定总敞口。""" + _validate_gross_exposure(gross_exposure) + if not assets.is_unique: + raise ValueError("assets must contain unique asset labels") + if assets.empty: + return pd.Series(index=assets.copy(), dtype=float, name="weight") + weight = gross_exposure / len(assets) + return pd.Series(weight, index=assets.copy(), dtype=float, name="weight") + + +def scores_to_target_weights( + scores: pd.Series, + top_k: int, + *, + gross_exposure: float = 1.0, + largest: bool = True, +) -> pd.Series: + """把单期因子分数转换为完整股票池目标权重。""" + _validate_score_series(scores) + selected = select_top_k(scores, top_k, largest=largest) + selected_weights = equal_weight(selected, gross_exposure=gross_exposure) + result = pd.Series(0.0, index=scores.index.copy(), dtype=float, name="weight") + result.loc[selected_weights.index] = selected_weights + return result + + +def scores_to_weight_table( + scores: pd.DataFrame, + top_k: int, + *, + gross_exposure: float = 1.0, + largest: bool = True, +) -> pd.DataFrame: + """逐调仓日独立构建目标权重表,避免使用未来分数。""" + if not isinstance(scores, pd.DataFrame): + raise TypeError(f"scores must be a pandas DataFrame, got {type(scores).__name__}") + _validate_top_k(top_k) + _validate_gross_exposure(gross_exposure) + if not scores.index.is_unique: + raise ValueError("scores must contain unique rebalance dates") + if not scores.index.is_monotonic_increasing: + raise ValueError("scores rebalance dates must be in chronological order") + if not scores.columns.is_unique: + raise ValueError("scores must contain unique asset labels") + if not all(is_numeric_dtype(dtype) for dtype in scores.dtypes): + raise TypeError("scores must contain numeric values") + if scores.empty: + return pd.DataFrame(index=scores.index.copy(), columns=scores.columns.copy(), dtype=float) + + rows = [ + scores_to_target_weights( + row, + top_k, + gross_exposure=gross_exposure, + largest=largest, + ).to_numpy() + for _, row in scores.iterrows() + ] + return pd.DataFrame(rows, index=scores.index.copy(), columns=scores.columns.copy(), dtype=float) diff --git a/src/quant_engine/research_pipeline.py b/src/quant_engine/research_pipeline.py new file mode 100644 index 0000000..bfdfb11 --- /dev/null +++ b/src/quant_engine/research_pipeline.py @@ -0,0 +1,401 @@ +"""可信研究链路:因子分数经交易日历滞后后进入执行与日频 Ledger。 + +本模块只编排现有组合构建与执行组件,不连接账户、券商或实盘订单。 +时间契约借鉴 Qlib 的 prediction/trade time 分离与 Backtrader 的 next-bar +执行语义:signal_date 上形成的目标权重,默认最早在下一交易时点执行。 +完整回测链路进一步分离 execution price 与日末 valuation price,非调仓日也 +持续盯市,并从真实成交后持仓派生日收益和绩效。 +""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass + +import numpy as np +import pandas as pd +from pandas.api.types import is_numeric_dtype + +from quant_engine.attribution import DailyReturnAttribution, compute_daily_return_attribution +from quant_engine.execution import ( + ExecutionConfig, + ExecutionSimulationResult, + simulate_daily_ledger_with_audit, + simulate_multi_day_with_audit, +) +from quant_engine.metrics import benchmark_summary, summary as metrics_summary +from quant_engine.portfolio_construction import scores_to_weight_table + +__all__ = [ + "TargetWeightSchedule", + "FactorExecutionResult", + "FactorBacktestResult", + "schedule_target_weights", + "run_factor_execution_research", + "run_factor_backtest_research", +] + + +@dataclass(frozen=True, slots=True, eq=False) +class TargetWeightSchedule: + """保留决策时间和执行时间的目标权重调度快照。""" + + decision_weights: pd.DataFrame + signal_to_execution: pd.Series + execution_weights: pd.DataFrame + lag_sessions: int + + +@dataclass(frozen=True, slots=True, eq=False) +class FactorExecutionResult: + """因子到执行审计的一次可复现研究结果。""" + + factor_scores: pd.DataFrame + execution_prices: pd.DataFrame + schedule: TargetWeightSchedule + execution_price_field: str + execution: ExecutionSimulationResult + + +@dataclass(frozen=True, slots=True, eq=False) +class FactorBacktestResult: + """因子、成交后日频 Ledger 与绩效的一次可复现快照。""" + + factor_scores: pd.DataFrame + execution_prices: pd.DataFrame + valuation_prices: pd.DataFrame + schedule: TargetWeightSchedule + execution_price_field: str + valuation_price_field: str + execution: ExecutionSimulationResult + + @property + def nav(self) -> pd.Series: + """返回以初始资金归一化为 1 的日频 NAV。""" + return pd.Series( + self.execution.normalized_nav_series.to_numpy(copy=True), + index=self.valuation_prices.index.copy(), + name="nav", + ) + + @property + def returns(self) -> pd.Series: + """返回包含首日成本影响的日频收益。""" + return pd.Series( + self.execution.daily_returns.to_numpy(copy=True), + index=self.valuation_prices.index.copy(), + name="returns", + ) + + @property + def position_weights(self) -> pd.DataFrame: + """按日末实际股数、收盘估值和账本 NAV 投影资产权重。""" + weights = pd.DataFrame( + 0.0, + index=self.valuation_prices.index.copy(), + columns=self.valuation_prices.columns.copy(), + ) + for date, position in zip( + self.valuation_prices.index, + self.execution.positions, + strict=True, + ): + if position.portfolio_value <= 0: + raise ValueError(f"portfolio value must be positive on {date}") + for asset, shares in position.holdings.items(): + weights.at[date, asset] = ( + shares * float(self.valuation_prices.at[date, asset]) + / position.portfolio_value + ) + return weights + + @property + def cash_weights(self) -> pd.Series: + """返回与实际资产权重使用同一日末 NAV 分母的现金权重。""" + values = [] + for date, position in zip( + self.valuation_prices.index, + self.execution.positions, + strict=True, + ): + if position.portfolio_value <= 0: + raise ValueError(f"portfolio value must be positive on {date}") + values.append(position.cash / position.portfolio_value) + return pd.Series( + values, + index=self.valuation_prices.index.copy(), + dtype=float, + name="cash_weight", + ) + + def stats(self, rf: float = 0.0) -> Mapping[str, float]: + """复用标准绩效口径计算指标。""" + return metrics_summary(self.returns, rf) + + def return_attribution(self) -> DailyReturnAttribution: + """从实际成交后持仓与账本生成逐日净收益归因。""" + return compute_daily_return_attribution( + self.execution, + self.execution_prices, + self.valuation_prices, + ) + + def benchmark_stats(self, benchmark_returns: pd.Series) -> Mapping[str, float]: + """计算成本后日收益相对同日基准的 TE、IR、alpha 与 beta。""" + return benchmark_summary(self.returns, benchmark_returns) + + +def _validate_datetime_index(index: pd.Index, name: str) -> pd.DatetimeIndex: + if not isinstance(index, pd.DatetimeIndex): + raise TypeError(f"{name} must use a DatetimeIndex") + if not index.is_unique: + raise ValueError(f"{name} must contain unique sessions") + if not index.is_monotonic_increasing: + raise ValueError(f"{name} must be in chronological order") + return index + + +def _validate_decision_weights(decision_weights: pd.DataFrame) -> None: + if not isinstance(decision_weights, pd.DataFrame): + raise TypeError( + f"decision_weights must be a pandas DataFrame, got {type(decision_weights).__name__}" + ) + _validate_datetime_index(decision_weights.index, "decision_weights index") + if not decision_weights.columns.is_unique: + raise ValueError("decision_weights must contain unique asset labels") + if not all(is_numeric_dtype(dtype) for dtype in decision_weights.dtypes): + raise TypeError("decision_weights must contain numeric values") + values = decision_weights.to_numpy(dtype=float) + if not np.isfinite(values).all() or (values < 0).any(): + raise ValueError("decision_weights must be finite and non-negative") + if (decision_weights.sum(axis=1) > 1.0 + 1e-12).any(): + raise ValueError("decision_weights rows must sum to at most 1.0") + + +def _validate_execution_prices(execution_prices: pd.DataFrame) -> pd.DatetimeIndex: + if not isinstance(execution_prices, pd.DataFrame): + raise TypeError( + f"execution_prices must be a pandas DataFrame, got {type(execution_prices).__name__}" + ) + calendar = _validate_datetime_index(execution_prices.index, "execution_prices index") + if not execution_prices.columns.is_unique: + raise ValueError("execution_prices must contain unique asset labels") + if not all(is_numeric_dtype(dtype) for dtype in execution_prices.dtypes): + raise TypeError("execution_prices must contain numeric values") + return calendar + + +def schedule_target_weights( + decision_weights: pd.DataFrame, + trading_calendar: pd.DatetimeIndex, + *, + lag_sessions: int = 1, +) -> TargetWeightSchedule: + """将信号日目标权重映射到后续真实交易日,不做整数行盲移位。 + + 所有信号日必须属于 ``trading_calendar``,且日历必须包含每个信号对应的 + 未来执行日;无法执行的末尾信号会显式失败,避免被静默丢弃。 + """ + _validate_decision_weights(decision_weights) + calendar = _validate_datetime_index(trading_calendar, "trading_calendar") + if isinstance(lag_sessions, bool) or not isinstance(lag_sessions, int) or lag_sessions <= 0: + raise ValueError("lag_sessions must be a positive integer") + + decision_snapshot = decision_weights.copy(deep=True) + if decision_snapshot.empty: + execution_weights = decision_snapshot.copy(deep=True) + execution_weights.index = pd.DatetimeIndex([], name="execution_date") + mapping = pd.Series( + calendar[:0], + index=decision_snapshot.index.copy(), + name="execution_date", + ) + return TargetWeightSchedule( + decision_weights=decision_snapshot, + signal_to_execution=mapping, + execution_weights=execution_weights, + lag_sessions=lag_sessions, + ) + + signal_positions = calendar.get_indexer(decision_snapshot.index) + if (signal_positions < 0).any(): + missing = decision_snapshot.index[signal_positions < 0] + raise ValueError( + "signal dates must be trading sessions; missing=" + + ", ".join(str(date) for date in missing) + ) + + execution_positions = signal_positions + lag_sessions + if (execution_positions >= len(calendar)).any(): + unavailable = decision_snapshot.index[execution_positions >= len(calendar)] + raise ValueError( + "trading_calendar lacks a future execution session for signal dates: " + + ", ".join(str(date) for date in unavailable) + ) + + execution_dates = calendar.take(execution_positions) + signal_to_execution = pd.Series( + execution_dates, + index=decision_snapshot.index.copy(), + name="execution_date", + ) + execution_weights = decision_snapshot.copy(deep=True) + execution_weights.index = pd.DatetimeIndex(execution_dates, name="execution_date") + return TargetWeightSchedule( + decision_weights=decision_snapshot, + signal_to_execution=signal_to_execution, + execution_weights=execution_weights, + lag_sessions=lag_sessions, + ) + + +def run_factor_execution_research( + factor_scores: pd.DataFrame, + execution_prices: pd.DataFrame, + *, + top_k: int, + execution_price_field: str, + lag_sessions: int = 1, + gross_exposure: float = 1.0, + largest: bool = True, + initial_cash: float = 1_000_000.0, + config: ExecutionConfig | None = None, +) -> FactorExecutionResult: + """运行因子分数 → 目标权重 → 下一交易时点 → 执行审计链路。 + + ``execution_prices`` 必须代表实际拟执行时点的价格矩阵,例如日频研究中 + signal 日收盘生成分数后使用下一交易日 ``open``。价格字段名称被保存在 + 结果元数据中,但函数不会猜测或重写价格语义。 + """ + price_field = execution_price_field.strip() + if not price_field: + raise ValueError("execution_price_field must be non-empty") + calendar = _validate_execution_prices(execution_prices) + + factor_snapshot = factor_scores.copy(deep=True) + decision_weights = scores_to_weight_table( + factor_snapshot, + top_k, + gross_exposure=gross_exposure, + largest=largest, + ) + schedule = schedule_target_weights( + decision_weights, + calendar, + lag_sessions=lag_sessions, + ) + price_snapshot = execution_prices.copy(deep=True) + + target_history: list[tuple[str, dict[str, float]]] = [] + price_history: list[tuple[str, dict[str, float]]] = [] + for execution_date, weights in schedule.execution_weights.iterrows(): + date_label = str(pd.Timestamp(execution_date)) + target_history.append( + (date_label, {asset: float(weight) for asset, weight in weights.items()}) + ) + prices = price_snapshot.loc[execution_date] + price_history.append( + (date_label, {asset: float(price) for asset, price in prices.items()}) + ) + + execution = simulate_multi_day_with_audit( + target_history, + price_history, + initial_cash, + config, + ) + return FactorExecutionResult( + factor_scores=factor_snapshot, + execution_prices=price_snapshot, + schedule=schedule, + execution_price_field=price_field, + execution=execution, + ) + + +def run_factor_backtest_research( + factor_scores: pd.DataFrame, + execution_prices: pd.DataFrame, + valuation_prices: pd.DataFrame, + *, + top_k: int, + execution_price_field: str, + valuation_price_field: str, + lag_sessions: int = 1, + gross_exposure: float = 1.0, + largest: bool = True, + initial_cash: float = 1_000_000.0, + config: ExecutionConfig | None = None, +) -> FactorBacktestResult: + """运行 PIT 因子到成交后日频 Ledger、收益与绩效的可信研究链路。""" + execution_field = execution_price_field.strip() + valuation_field = valuation_price_field.strip() + if not execution_field: + raise ValueError("execution_price_field must be non-empty") + if not valuation_field: + raise ValueError("valuation_price_field must be non-empty") + + execution_calendar = _validate_execution_prices(execution_prices) + valuation_calendar = _validate_execution_prices(valuation_prices) + if not execution_calendar.equals(valuation_calendar): + raise ValueError("execution and valuation prices must use matching trading calendars") + if not execution_prices.columns.equals(valuation_prices.columns): + raise ValueError("execution and valuation prices must use matching asset labels") + + factor_snapshot = factor_scores.copy(deep=True) + execution_snapshot = execution_prices.copy(deep=True) + valuation_snapshot = valuation_prices.copy(deep=True) + decision_weights = scores_to_weight_table( + factor_snapshot, + top_k, + gross_exposure=gross_exposure, + largest=largest, + ) + schedule = schedule_target_weights( + decision_weights, + execution_calendar, + lag_sessions=lag_sessions, + ) + if decision_weights.empty: + execution_window = execution_snapshot.iloc[:0].copy() + valuation_window = valuation_snapshot.iloc[:0].copy() + else: + research_start = decision_weights.index[0] + execution_window = execution_snapshot.loc[research_start:].copy() + valuation_window = valuation_snapshot.loc[research_start:].copy() + + target_history: list[tuple[str, dict[str, float]]] = [] + execution_history: list[tuple[str, dict[str, float]]] = [] + for execution_date, weights in schedule.execution_weights.iterrows(): + date_label = str(pd.Timestamp(execution_date)) + target_history.append( + (date_label, {asset: float(weight) for asset, weight in weights.items()}) + ) + prices = execution_window.loc[execution_date] + execution_history.append( + (date_label, {asset: float(price) for asset, price in prices.items()}) + ) + + valuation_history = [ + ( + str(pd.Timestamp(valuation_date)), + {asset: float(price) for asset, price in prices.items()}, + ) + for valuation_date, prices in valuation_window.iterrows() + ] + execution = simulate_daily_ledger_with_audit( + target_history, + execution_history, + valuation_history, + initial_cash, + config, + ) + return FactorBacktestResult( + factor_scores=factor_snapshot, + execution_prices=execution_window, + valuation_prices=valuation_window, + schedule=schedule, + execution_price_field=execution_field, + valuation_price_field=valuation_field, + execution=execution, + ) diff --git a/src/quant_engine/risk.py b/src/quant_engine/risk.py index 2ca6a69..748ad36 100644 --- a/src/quant_engine/risk.py +++ b/src/quant_engine/risk.py @@ -5,10 +5,310 @@ from __future__ import annotations -import numpy as np -from numpy.typing import NDArray +import hashlib +import json +from dataclasses import dataclass +from datetime import date from typing import Any +import numpy as np +import pandas as pd +from numpy.typing import NDArray + +__all__ = [ + "ComponentRiskResult", + "CovarianceSnapshot", + "component_var", + "estimate_covariance_snapshot", + "labeled_component_risk", + "marginal_risk_contribution", + "risk_contribution", +] + + +@dataclass(frozen=True, slots=True, init=False, eq=False) +class CovarianceSnapshot: + """Immutable-by-interface covariance input with explicit time semantics.""" + + snapshot_id: str + as_of_date: date + _covariance: pd.DataFrame + return_frequency: str + periods_per_year: int + method: str + window_start_date: date | None + window_end_date: date | None + observations: int | None + lookback_sessions: int | None + missing_policy: str + data_snapshot_id: str + input_sha256: str + + def __init__( + self, + *, + snapshot_id: str, + as_of_date: str | date | pd.Timestamp, + covariance: pd.DataFrame, + return_frequency: str, + periods_per_year: int, + method: str = "provided", + window_start_date: str | date | pd.Timestamp | None = None, + window_end_date: str | date | pd.Timestamp | None = None, + observations: int | None = None, + lookback_sessions: int | None = None, + missing_policy: str = "provided", + data_snapshot_id: str = "", + input_sha256: str = "", + ) -> None: + if not isinstance(snapshot_id, str) or not snapshot_id.strip(): + raise ValueError("snapshot_id must be non-empty") + if not isinstance(return_frequency, str) or not return_frequency.strip(): + raise ValueError("return_frequency must be non-empty") + if isinstance(periods_per_year, bool) or not isinstance(periods_per_year, int): + raise TypeError("periods_per_year must be an integer") + if periods_per_year <= 0: + raise ValueError("periods_per_year must be positive") + if not isinstance(covariance, pd.DataFrame): + raise TypeError("covariance must be a pandas DataFrame") + if covariance.empty: + raise ValueError("covariance must contain at least one asset") + if not isinstance(method, str) or not method.strip(): + raise ValueError("method must be non-empty") + if not isinstance(missing_policy, str) or not missing_policy.strip(): + raise ValueError("missing_policy must be non-empty") + for value, name in ( + (observations, "observations"), + (lookback_sessions, "lookback_sessions"), + ): + if value is not None and ( + isinstance(value, bool) or not isinstance(value, int) or value <= 0 + ): + raise ValueError(f"{name} must be a positive integer when provided") + if input_sha256 and ( + len(input_sha256) != 64 + or any(character not in "0123456789abcdef" for character in input_sha256) + ): + raise ValueError("input_sha256 must be a lowercase SHA-256 digest") + + normalized_as_of = _normalized_date(as_of_date, "as_of_date") + normalized_window_start = ( + None + if window_start_date is None + else _normalized_date(window_start_date, "window_start_date") + ) + normalized_window_end = ( + None + if window_end_date is None + else _normalized_date(window_end_date, "window_end_date") + ) + if (normalized_window_start is None) != (normalized_window_end is None): + raise ValueError("window_start_date and window_end_date must be provided together") + if ( + normalized_window_start is not None + and normalized_window_end is not None + and normalized_window_start > normalized_window_end + ): + raise ValueError("window_start_date must not be after window_end_date") + if normalized_window_end is not None and normalized_window_end > normalized_as_of: + raise ValueError("window_end_date must not be after as_of_date") + + object.__setattr__(self, "snapshot_id", snapshot_id.strip()) + object.__setattr__(self, "as_of_date", normalized_as_of) + object.__setattr__(self, "_covariance", covariance.copy(deep=True)) + object.__setattr__(self, "return_frequency", return_frequency.strip()) + object.__setattr__(self, "periods_per_year", periods_per_year) + object.__setattr__(self, "method", method.strip()) + object.__setattr__(self, "window_start_date", normalized_window_start) + object.__setattr__(self, "window_end_date", normalized_window_end) + object.__setattr__(self, "observations", observations) + object.__setattr__(self, "lookback_sessions", lookback_sessions) + object.__setattr__(self, "missing_policy", missing_policy.strip()) + object.__setattr__(self, "data_snapshot_id", data_snapshot_id.strip()) + object.__setattr__(self, "input_sha256", input_sha256) + + @property + def covariance(self) -> pd.DataFrame: + """Return an isolated copy so callers cannot mutate the snapshot.""" + return self._covariance.copy(deep=True) + + +def _normalized_date(value: object, name: str) -> date: + try: + timestamp = pd.Timestamp(value) + except (TypeError, ValueError) as error: + raise ValueError(f"{name} must be a valid date") from error + if pd.isna(timestamp): + raise ValueError(f"{name} must be a valid date") + return date(int(timestamp.year), int(timestamp.month), int(timestamp.day)) + + +def _positive_integer(value: int, name: str, *, minimum: int = 1) -> int: + if isinstance(value, bool) or not isinstance(value, int) or value < minimum: + raise ValueError(f"{name} must be an integer of at least {minimum}") + return value + + +def _input_fingerprint(window: pd.DataFrame, session_dates: list[date]) -> str: + values = window.to_numpy(dtype=float, copy=True) + missing = np.isnan(values) + normalized = np.where(missing, 0.0, values).astype(" CovarianceSnapshot: + """Estimate a deterministic per-period sample covariance without look-ahead. + + The selected lookback window is truncated at ``as_of_date`` before any + calculation. Rows containing a missing asset return are removed as complete + cases, preventing pairwise sample sets from producing an ambiguous matrix. + """ + if not isinstance(asset_returns, pd.DataFrame): + raise TypeError("asset_returns must be a pandas DataFrame") + if asset_returns.empty or asset_returns.shape[1] == 0: + raise ValueError("asset_returns must contain observations and assets") + if not isinstance(asset_returns.index, pd.DatetimeIndex): + raise TypeError("asset_returns index must be a DatetimeIndex") + if not asset_returns.index.is_unique or not asset_returns.index.is_monotonic_increasing: + raise ValueError("asset_returns index must be unique and strictly increasing") + if not asset_returns.columns.is_unique: + raise ValueError("asset_returns must contain unique asset labels") + if any(not isinstance(asset, str) or not asset.strip() for asset in asset_returns.columns): + raise ValueError("asset_returns asset labels must be non-empty strings") + + lookback = _positive_integer(lookback_sessions, "lookback_sessions") + minimum = _positive_integer(min_observations, "min_observations", minimum=2) + if minimum > lookback: + raise ValueError("min_observations must not exceed lookback_sessions") + normalized_data_snapshot_id = data_snapshot_id.strip() + if not normalized_data_snapshot_id: + raise ValueError("data_snapshot_id must be non-empty") + normalized_as_of = _normalized_date(as_of_date, "as_of_date") + + returns = asset_returns.astype(float, copy=True) + values = returns.to_numpy() + if np.isinf(values).any(): + raise ValueError("asset_returns must not contain infinite values") + session_dates = [ + _normalized_date(index_value, "asset_returns index") for index_value in returns.index + ] + if len(set(session_dates)) != len(session_dates): + raise ValueError("asset_returns must contain at most one observation per session date") + historical_mask = [session <= normalized_as_of for session in session_dates] + window = returns.loc[historical_mask].tail(lookback) + if window.empty: + raise ValueError("asset_returns contain no observations on or before as_of_date") + window_dates = [ + _normalized_date(index_value, "asset_returns index") for index_value in window.index + ] + complete = window.dropna(axis=0, how="any") + if len(complete) < minimum: + raise ValueError( + f"complete observations must be at least {minimum}; received {len(complete)}" + ) + + covariance = complete.cov(ddof=1) + covariance_values = covariance.to_numpy() + if not np.isfinite(covariance_values).all(): + raise ValueError("sample covariance must be finite") + input_sha256 = _input_fingerprint(window, window_dates) + identity = { + "as_of_date": normalized_as_of.isoformat(), + "assets": list(returns.columns), + "data_snapshot_id": normalized_data_snapshot_id, + "estimator": "sample-cov-v1", + "input_sha256": input_sha256, + "lookback_sessions": lookback, + "min_observations": minimum, + "missing_policy": "complete_case", + "observations": len(complete), + "periods_per_year": periods_per_year, + "return_frequency": return_frequency, + "window_end_date": window_dates[-1].isoformat(), + "window_start_date": window_dates[0].isoformat(), + } + identity_bytes = json.dumps( + identity, + sort_keys=True, + separators=(",", ":"), + ).encode("utf-8") + digest = hashlib.sha256(identity_bytes) + digest.update(covariance_values.astype(" pd.Series: + """Aggregate asset component risk by an explicitly aligned label series.""" + if not isinstance(groups, pd.Series): + raise TypeError("groups must be a pandas Series") + if not groups.index.is_unique: + raise ValueError("groups must contain unique asset labels") + if not self.component.index.difference(groups.index).empty or not groups.index.difference( + self.component.index + ).empty: + raise ValueError("groups and component risk must use the same asset labels") + aligned = groups.reindex(self.component.index) + if aligned.isna().any(): + raise ValueError("groups must contain a non-missing label for every asset") + grouped = self.component.groupby(aligned, sort=True).sum() + grouped.name = "component_risk" + return grouped + + +def _validate_inputs(weights: NDArray[Any], cov: NDArray[Any]) -> tuple[NDArray[Any], NDArray[Any]]: + """Normalize a portfolio vector and its covariance matrix.""" + w = np.asarray(weights, dtype=float).ravel() + covariance = np.asarray(cov, dtype=float) + k = w.size + if k == 0: + raise ValueError("weights must contain at least one asset") + if covariance.shape != (k, k): + raise ValueError(f"cov shape {covariance.shape} does not match weights length {k}") + return w, covariance + def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]: """风险贡献率 (RC_i): w_i * (Σw)_i / w'Σw。 @@ -25,11 +325,8 @@ def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]: Returns: RC: 风险贡献向量 (k,), Σ=1 """ - w = np.asarray(weights, dtype=float).ravel() - cov = np.asarray(cov, dtype=float) + w, cov = _validate_inputs(weights, cov) k = w.size - if cov.shape != (k, k): - raise ValueError(f"cov 形状 {cov.shape} 与 weights 长度 {k} 不匹配") port_var = float(w @ cov @ w) if port_var <= 0: @@ -41,13 +338,78 @@ def risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]: def marginal_risk_contribution(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]: """边际风险贡献 (MRC_i): (Σw)_i。""" - w = np.asarray(weights, dtype=float).ravel() - cov = np.asarray(cov, dtype=float) + w, cov = _validate_inputs(weights, cov) return cov @ w # type: ignore[no-any-return] def component_var(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]: """成分方差: w_i · (Σw)_i; 与 RC 的关系 RC_i = CV_i / w'Σw。""" - w = np.asarray(weights, dtype=float).ravel() - cov = np.asarray(cov, dtype=float) + w, cov = _validate_inputs(weights, cov) return w * (cov @ w) # type: ignore[no-any-return] + + +def labeled_component_risk( + weights: pd.Series, + covariance: pd.DataFrame, +) -> ComponentRiskResult: + """Return a label-safe Euler decomposition that sums to portfolio volatility. + + The covariance matrix may use a different asset order, but its row and + column label sets must exactly match ``weights``. Invalid or indefinite + covariance input is rejected instead of silently producing misleading risk + percentages. + """ + if not isinstance(weights, pd.Series): + raise TypeError("weights must be a pandas Series") + if not isinstance(covariance, pd.DataFrame): + raise TypeError("covariance must be a pandas DataFrame") + if weights.empty: + raise ValueError("weights must contain at least one asset") + if not weights.index.is_unique: + raise ValueError("weights must contain unique asset labels") + if not covariance.index.is_unique or not covariance.columns.is_unique: + raise ValueError("covariance must contain unique asset labels") + if not weights.index.difference(covariance.index).empty or not covariance.index.difference( + weights.index + ).empty: + raise ValueError("weights and covariance must use the same asset labels") + if not weights.index.difference(covariance.columns).empty or not covariance.columns.difference( + weights.index + ).empty: + raise ValueError("weights and covariance must use the same asset labels") + + aligned_weights = weights.astype(float, copy=True) + aligned_covariance = covariance.reindex( + index=weights.index, + columns=weights.index, + ).astype(float, copy=True) + weight_values = aligned_weights.to_numpy() + covariance_values = aligned_covariance.to_numpy() + if not np.isfinite(weight_values).all(): + raise ValueError("weights must be finite") + if not np.isfinite(covariance_values).all(): + raise ValueError("covariance must be finite") + if not np.allclose(covariance_values, covariance_values.T, rtol=1e-10, atol=1e-12): + raise ValueError("covariance must be symmetric") + eigenvalues = np.linalg.eigvalsh(covariance_values) + scale = max(1.0, float(np.max(np.abs(eigenvalues)))) + if float(eigenvalues.min()) < -1e-10 * scale: + raise ValueError("covariance must be positive semidefinite") + + portfolio_variance = float(weight_values @ covariance_values @ weight_values) + if portfolio_variance <= 0 or not np.isfinite(portfolio_variance): + raise ValueError("weights and covariance must produce positive portfolio variance") + portfolio_volatility = float(np.sqrt(portfolio_variance)) + marginal_values = covariance_values @ weight_values / portfolio_volatility + component_values = weight_values * marginal_values + percentage_values = component_values / portfolio_volatility + return ComponentRiskResult( + portfolio_volatility=portfolio_volatility, + marginal=pd.Series(marginal_values, index=weights.index.copy(), name="marginal_risk"), + component=pd.Series(component_values, index=weights.index.copy(), name="component_risk"), + percentage=pd.Series( + percentage_values, + index=weights.index.copy(), + name="risk_contribution", + ), + ) diff --git a/tests/test_artifact.py b/tests/test_artifact.py new file mode 100644 index 0000000..d208d6c --- /dev/null +++ b/tests/test_artifact.py @@ -0,0 +1,309 @@ +"""Stable research-run artifact contracts for downstream persistence.""" + +from __future__ import annotations + +import json +from datetime import date + +import pandas as pd +import pytest + +from quant_engine.artifact import ( + RESEARCH_ARTIFACT_SCHEMA_VERSION, + ResearchRunArtifact, + build_research_run_artifact, +) +from quant_engine.execution import ExecutionConfig +from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research +from quant_engine.risk import CovarianceSnapshot + + +def _backtest_result() -> FactorBacktestResult: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + scores = pd.DataFrame( + {"A": [2.0, 0.0], "B": [1.0, 3.0]}, + index=dates[:2], + ) + opens = pd.DataFrame( + {"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]}, + index=dates, + ) + return run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ), + ) + + +def _build( + result: FactorBacktestResult, + *, + parameters: dict[str, object] | None = None, + risk_snapshots: dict[date, CovarianceSnapshot] | None = None, +) -> ResearchRunArtifact: + benchmark = pd.Series( + [0.0, 0.01, -0.01, 0.02], + index=result.returns.index, + name="benchmark_return", + ) + return build_research_run_artifact( + result, + run_id="run-20260105-a", + strategy_id="alpha-top1", + strategy_name="Alpha Top 1", + strategy_version="1.0.0", + engine_version="1.2.0", + code_revision="3b1ad07", + data_snapshot_id="qtdb-pro-20260108-v1", + calendar="CN-A", + timezone="Asia/Shanghai", + started_at="2026-01-08T10:00:00+08:00", + finished_at="2026-01-08T10:01:00+08:00", + parameters=parameters or {"top_k": 1, "lag_sessions": 1}, + benchmark_id="000300.SH", + benchmark_returns=benchmark, + risk_snapshots=risk_snapshots, + ) + + +def test_research_artifact_projects_versioned_queryable_fact_tables() -> None: + result = _backtest_result() + + artifact = _build(result) + + assert artifact.schema_version == RESEARCH_ARTIFACT_SCHEMA_VERSION + assert artifact.run.loc[0, "run_id"] == "run-20260105-a" + assert artifact.run.loc[0, "benchmark_alignment_policy"] == "exact_session_index" + assert artifact.nav["run_id"].unique().tolist() == ["run-20260105-a"] + assert artifact.nav["pnl_pct"].tolist() == pytest.approx(result.returns.tolist()) + assert artifact.nav["benchmark_return"].tolist() == pytest.approx( + [0.0, 0.01, -0.01, 0.02] + ) + assert artifact.signals.columns.tolist() == [ + "run_id", + "signal_date", + "execution_date", + "asset_id", + "factor_score", + "target_weight", + ] + first_signal = artifact.signals[ + artifact.signals["signal_date"] == result.factor_scores.index[0].date() + ] + assert first_signal.set_index("asset_id").loc["A", "factor_score"] == 2.0 + assert first_signal.set_index("asset_id").loc["A", "target_weight"] == 1.0 + assert first_signal["execution_date"].unique().tolist() == [ + result.schedule.signal_to_execution.iloc[0].date() + ] + assert set(artifact.trades["side"]) == {"buy", "sell"} + assert artifact.trades["trade_id"].is_unique + assert artifact.trades["trade_id"].str.startswith("run-20260105-a:").all() + assert artifact.trades["signal_id"].str.startswith("run-20260105-a:signal:").all() + assert {"security", "cash"}.issubset(set(artifact.positions["asset_type"])) + assert artifact.positions.groupby("trade_date")["weight"].sum().tolist() == pytest.approx( + [1.0, 1.0, 1.0, 1.0] + ) + assert set(artifact.attribution.columns) == { + "run_id", + "trade_date", + "asset_id", + "overnight", + "intraday", + "asset_total", + } + assert artifact.attribution_daily["residual"].abs().max() < 1e-12 + assert artifact.risk.empty + assert artifact.risk.columns.tolist() == [ + "run_id", + "trade_date", + "asset_id", + "weight", + "marginal_risk", + "component_risk", + "risk_contribution", + "covariance_snapshot_id", + "covariance_as_of_date", + "risk_measure", + "return_frequency", + "periods_per_year", + ] + assert artifact.performance.loc[0, "n_trades"] == len(artifact.trades) + assert artifact.performance.loc[0, "ir"] == pytest.approx( + result.benchmark_stats(pd.Series([0.0, 0.01, -0.01, 0.02], index=result.returns.index))[ + "information_ratio" + ] + ) + assert "sortino" in artifact.performance.columns + + +def test_research_artifact_projects_annualized_risk_from_actual_positions() -> None: + result = _backtest_result() + trade_date = result.position_weights.index[-1].date() + covariance = pd.DataFrame( + [[0.0001, 0.00002], [0.00002, 0.0004]], + index=["A", "B"], + columns=["A", "B"], + ) + snapshot = CovarianceSnapshot( + snapshot_id="cov-20260107-v1", + as_of_date="2026-01-07", + covariance=covariance, + return_frequency="1d", + periods_per_year=252, + data_snapshot_id="qtdb-pro-20260108-v1", + ) + + artifact = _build(result, risk_snapshots={trade_date: snapshot}) + + risk = artifact.risk.set_index("asset_id") + expected_weights = result.position_weights.loc[pd.Timestamp(trade_date)] + assert artifact.schema_version == "1.1.0" + assert risk.index.tolist() == ["A", "B"] + assert risk["weight"].tolist() == pytest.approx(expected_weights.tolist()) + assert risk["covariance_snapshot_id"].unique().tolist() == ["cov-20260107-v1"] + assert risk["covariance_as_of_date"].unique().tolist() == [date(2026, 1, 7)] + assert risk["risk_measure"].unique().tolist() == ["annualized_volatility"] + assert risk["return_frequency"].unique().tolist() == ["1d"] + assert risk["periods_per_year"].unique().tolist() == [252] + assert risk["component_risk"].sum() == pytest.approx((0.0004 * 252) ** 0.5) + assert risk["risk_contribution"].sum() == pytest.approx(1.0) + + +def test_research_artifact_rejects_risk_from_a_different_data_snapshot() -> None: + result = _backtest_result() + trade_date = result.position_weights.index[-1].date() + covariance = pd.DataFrame( + [[0.0001, 0.0], [0.0, 0.0004]], + index=["A", "B"], + columns=["A", "B"], + ) + + with pytest.raises(ValueError, match="data lineage differs"): + _build( + result, + risk_snapshots={ + trade_date: CovarianceSnapshot( + snapshot_id="foreign-covariance", + as_of_date="2026-01-07", + covariance=covariance, + return_frequency="1d", + periods_per_year=252, + data_snapshot_id="different-market-snapshot", + ) + }, + ) + + +def test_research_artifact_rejects_future_or_misaligned_risk_snapshots() -> None: + result = _backtest_result() + trade_date = result.position_weights.index[-1].date() + covariance = pd.DataFrame( + [[0.0001, 0.0], [0.0, 0.0004]], + index=["A", "B"], + columns=["A", "B"], + ) + + with pytest.raises(ValueError, match="must not be after trade date"): + _build( + result, + risk_snapshots={ + trade_date: CovarianceSnapshot( + snapshot_id="future-covariance", + as_of_date="2026-01-09", + covariance=covariance, + return_frequency="1d", + periods_per_year=252, + data_snapshot_id="qtdb-pro-20260108-v1", + ) + }, + ) + + with pytest.raises(ValueError, match="same asset labels"): + _build( + result, + risk_snapshots={ + trade_date: CovarianceSnapshot( + snapshot_id="incomplete-universe", + as_of_date="2026-01-07", + covariance=covariance.loc[["B"], ["B"]], + return_frequency="1d", + periods_per_year=252, + data_snapshot_id="qtdb-pro-20260108-v1", + ) + }, + ) + + +def test_research_artifact_serialization_and_hashes_are_deterministic() -> None: + result = _backtest_result() + first = _build(result, parameters={"top_k": 1, "lag_sessions": 1}) + second = _build(result, parameters={"lag_sessions": 1, "top_k": 1}) + + assert first.run.loc[0, "config_hash"] == second.run.loc[0, "config_hash"] + assert first.content_sha256 == second.content_sha256 + assert first.manifest() == second.manifest() + decoded = json.loads(first.canonical_json()) + assert decoded["schema_version"] == RESEARCH_ARTIFACT_SCHEMA_VERSION + assert decoded["tables"]["nav"][0]["trade_date"] == "2026-01-05" + + leaked_copy = first.nav + leaked_copy.loc[0, "nav"] = -999.0 + assert first.nav.loc[0, "nav"] != -999.0 + assert first.content_sha256 == second.content_sha256 + + +def test_research_artifact_requires_complete_reproducibility_identity() -> None: + result = _backtest_result() + + with pytest.raises(ValueError, match="code_revision"): + build_research_run_artifact( + result, + run_id="run-1", + strategy_id="alpha-top1", + strategy_name="Alpha Top 1", + strategy_version="1.0.0", + engine_version="1.2.0", + code_revision="", + data_snapshot_id="snapshot-1", + calendar="CN-A", + timezone="Asia/Shanghai", + started_at="2026-01-08T10:00:00+08:00", + finished_at="2026-01-08T10:01:00+08:00", + parameters={}, + ) + + +def test_research_artifact_requires_benchmark_identity_and_returns_together() -> None: + result = _backtest_result() + + with pytest.raises(ValueError, match="benchmark_id and benchmark_returns"): + build_research_run_artifact( + result, + run_id="run-1", + strategy_id="alpha-top1", + strategy_name="Alpha Top 1", + strategy_version="1.0.0", + engine_version="1.2.0", + code_revision="3b1ad07", + data_snapshot_id="snapshot-1", + calendar="CN-A", + timezone="Asia/Shanghai", + started_at="2026-01-08T10:00:00+08:00", + finished_at="2026-01-08T10:01:00+08:00", + parameters={}, + benchmark_id="000300.SH", + ) diff --git a/tests/test_attribution.py b/tests/test_attribution.py new file mode 100644 index 0000000..99eccce --- /dev/null +++ b/tests/test_attribution.py @@ -0,0 +1,118 @@ +"""Post-execution return attribution contracts.""" + +from __future__ import annotations + +import pandas as pd +import pytest + +from quant_engine.attribution import DailyReturnAttribution +from quant_engine.execution import ExecutionConfig +from quant_engine.research_pipeline import run_factor_backtest_research + + +def _zero_cost_config() -> ExecutionConfig: + return ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + + +def test_daily_attribution_closes_across_rebalance_and_holding_days() -> None: + """开盘换仓时,隔夜和日内贡献必须来自实际换仓前后持仓。""" + dates = pd.date_range("2026-01-05", periods=4, freq="B") + scores = pd.DataFrame( + {"A": [2.0, 0.0], "B": [1.0, 3.0]}, + index=dates[:2], + ) + opens = pd.DataFrame( + {"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]}, + index=dates, + ) + result = run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=_zero_cost_config(), + ) + + attribution = result.return_attribution() + + assert isinstance(attribution, DailyReturnAttribution) + assert attribution.overnight.loc[dates[2], "A"] == pytest.approx(0.25) + assert attribution.intraday.loc[dates[2], "B"] == pytest.approx(-0.125) + assert attribution.asset_contributions.loc[dates[3], "B"] == pytest.approx(1 / 6) + pd.testing.assert_series_equal( + attribution.total_return, + result.returns.rename("total_return"), + ) + pd.testing.assert_series_equal( + attribution.explained_return + attribution.residual, + attribution.total_return, + check_names=False, + ) + assert attribution.residual.abs().max() < 1e-12 + + +def test_daily_attribution_reports_execution_cost_separately() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + scores = pd.DataFrame({"A": [1.0]}, index=dates[:1]) + prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates) + config = ExecutionConfig( + commission_bps=10, + stamp_tax_bps=0, + slippage_bps=10, + min_trade_amount=0, + ) + result = run_factor_backtest_research( + scores, + prices, + prices, + top_k=1, + gross_exposure=0.5, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=config, + ) + + attribution = result.return_attribution() + execution = result.execution.daily_executions[1].executions[0] + + assert attribution.asset_contributions.loc[dates[1], "A"] == 0.0 + assert attribution.transaction_cost.loc[dates[1]] == pytest.approx( + -execution.total_cost / 1_000.0 + ) + assert attribution.total_return.loc[dates[1]] == pytest.approx( + attribution.transaction_cost.loc[dates[1]] + ) + assert attribution.residual.loc[dates[1]] == pytest.approx(0.0, abs=1e-12) + + +def test_return_attribution_is_empty_for_empty_research_result() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + scores = pd.DataFrame(columns=["A"], index=pd.DatetimeIndex([]), dtype=float) + prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates) + result = run_factor_backtest_research( + scores, + prices, + prices, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + ) + + attribution = result.return_attribution() + + assert attribution.overnight.empty + assert attribution.intraday.empty + assert attribution.total_return.empty diff --git a/tests/test_backtest.py b/tests/test_backtest.py new file mode 100644 index 0000000..9f8d4a8 --- /dev/null +++ b/tests/test_backtest.py @@ -0,0 +1,209 @@ +"""Backtest contract tests for weights, NAV, rebalancing, and benchmarks.""" + +from __future__ import annotations + +import pandas as pd +import pytest + +from quant_engine.backtest import ( + BacktestResult, + compare_to_benchmark, + compute_nav_from_weights, + compute_returns_from_nav, + rebalance_periodic, + run_weight_backtest, + weights_to_long_short, +) + + +def test_compute_nav_from_weights_forward_fills_rebalance_weights() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + weights = pd.DataFrame({"A": [0.5], "B": [0.5]}, index=dates[:1]) + returns = pd.DataFrame({"A": [0.10, 0.00, -0.10], "B": [0.00, 0.10, 0.00]}, index=dates) + + nav = compute_nav_from_weights(weights, returns, initial_capital=100.0) + + expected = pd.Series([105.0, 110.25, 104.7375], index=dates) + pd.testing.assert_series_equal(nav, expected) + + +def test_compute_nav_stays_in_cash_before_first_rebalance() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + weights = pd.DataFrame({"A": [1.0]}, index=dates[1:2]) + returns = pd.DataFrame({"A": [0.50, 0.10, 0.10]}, index=dates) + + nav = compute_nav_from_weights(weights, returns) + + pd.testing.assert_series_equal(nav, pd.Series([1.0, 1.1, 1.21], index=dates)) + + +def test_compute_nav_ignores_weight_columns_without_returns() -> None: + dates = pd.date_range("2026-01-05", periods=2, freq="B") + weights = pd.DataFrame({"A": [0.5], "MISSING": [0.5]}, index=dates[:1]) + returns = pd.DataFrame({"A": [0.10, 0.10]}, index=dates) + + nav = compute_nav_from_weights(weights, returns) + + pd.testing.assert_series_equal(nav, pd.Series([1.05, 1.1025], index=dates)) + + +def test_compute_nav_charges_configured_turnover_cost() -> None: + dates = pd.date_range("2026-01-05", periods=2, freq="B") + weights = pd.DataFrame({"A": [1.0]}, index=dates[:1]) + returns = pd.DataFrame({"A": [0.0, 0.0]}, index=dates) + + nav = compute_nav_from_weights(weights, returns, tc_rate=0.01) + + pd.testing.assert_series_equal(nav, pd.Series([0.995, 0.995], index=dates)) + + +def test_compute_returns_from_nav_preserves_index_and_sets_initial_zero() -> None: + nav = pd.Series([100.0, 110.0, 99.0], index=pd.date_range("2026-01-05", periods=3)) + + result = compute_returns_from_nav(nav) + + pd.testing.assert_series_equal(result, pd.Series([0.0, 0.1, -0.1], index=nav.index)) + + +def test_rebalance_periodic_maps_weekend_to_previous_trading_day() -> None: + dates = pd.date_range("2026-01-05", periods=5, freq="B") + target = pd.Series({"A": 0.6, "B": 0.4}) + + result = rebalance_periodic(target, [pd.Timestamp("2026-01-10")], dates) + + assert result.loc[pd.Timestamp("2026-01-08")].sum() == 0.0 + pd.testing.assert_series_equal( + result.loc[pd.Timestamp("2026-01-09")], target, check_names=False + ) + + +def test_rebalance_periodic_accepts_empty_trading_calendar() -> None: + target = pd.Series({"A": 1.0}) + + result = rebalance_periodic( + target, + [pd.Timestamp("2026-01-05")], + pd.DatetimeIndex([]), + ) + + assert result.empty + assert result.columns.tolist() == ["A"] + + +def test_weights_to_long_short_allocates_each_leg() -> None: + result = weights_to_long_short(["A", "B"], ["C"], long_weight=0.6, short_weight=0.4) + + assert result["A"] == pytest.approx(0.3) + assert result["B"] == pytest.approx(0.3) + assert result["C"] == pytest.approx(-0.4) + assert result.sum() == pytest.approx(0.2) + + +def test_weights_to_long_short_keeps_explicit_universe() -> None: + result = weights_to_long_short(["A"], [], all_tickers=["A", "B"]) + + pd.testing.assert_series_equal(result, pd.Series({"A": 0.5, "B": 0.0})) + + +def test_compare_to_benchmark_returns_report_table() -> None: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + strategy = pd.Series([1.0, 1.1, 1.0, 1.2], index=dates) + benchmark = pd.Series([1.0, 1.0, 1.05, 1.1], index=dates) + + result = compare_to_benchmark(strategy, benchmark) + + assert result.columns.tolist() == ["策略", "基准"] + assert result.loc["n_days", "策略"] == 4 + assert result.loc["累计收益", "策略"] == pytest.approx(0.2) + assert result.loc["累计收益", "基准"] == pytest.approx(0.1) + + +def test_compare_to_benchmark_rejects_non_overlapping_dates() -> None: + strategy = pd.Series([1.0], index=[pd.Timestamp("2026-01-05")]) + benchmark = pd.Series([1.0], index=[pd.Timestamp("2026-02-05")]) + + with pytest.raises(ValueError, match="overlapping dates"): + compare_to_benchmark(strategy, benchmark) + + +# ── 统一回测结果门面 ────────────────────────────────────── + + +def test_run_weight_backtest_returns_nav_returns_and_input_snapshot() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + weights = pd.DataFrame({"A": [1.0]}, index=dates[:1]) + stock_returns = pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates) + + result = run_weight_backtest(weights, stock_returns, initial_capital=100.0) + + assert isinstance(result, BacktestResult) + pd.testing.assert_series_equal( + result.nav, + pd.Series([110.0, 99.0, 118.8], index=dates), + ) + pd.testing.assert_series_equal( + result.returns, + pd.Series([0.0, -0.1, 0.2], index=dates), + ) + pd.testing.assert_frame_equal(result.weights, weights) + + +def test_backtest_result_stats_reuses_standard_metrics_contract() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + result = run_weight_backtest( + pd.DataFrame({"A": [1.0]}, index=dates[:1]), + pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates), + ) + + stats = result.stats(rf=0.02) + + assert stats["n_days"] == 3 + assert stats["ann_return"] == pytest.approx( + (1.0 * 0.9 * 1.2) ** (252 / 3) - 1.0 + ) + assert "sharpe" in stats + assert stats["drawback"] == stats["max_drawdown"] + + +def test_backtest_result_builds_benchmark_report() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + benchmark = pd.Series([1.0, 1.05, 1.10], index=dates, name="benchmark") + result = run_weight_backtest( + pd.DataFrame({"A": [1.0]}, index=dates[:1]), + pd.DataFrame({"A": [0.10, -0.10, 0.20]}, index=dates), + benchmark_nav=benchmark, + ) + + report = result.benchmark_report() + + assert report.columns.tolist() == ["策略", "基准"] + assert report.loc["累计收益", "基准"] == pytest.approx(0.10) + + +def test_backtest_result_requires_benchmark_for_comparison() -> None: + dates = pd.date_range("2026-01-05", periods=2, freq="B") + result = run_weight_backtest( + pd.DataFrame({"A": [1.0]}, index=dates[:1]), + pd.DataFrame({"A": [0.0, 0.0]}, index=dates), + ) + + with pytest.raises(ValueError, match="benchmark_nav"): + result.benchmark_report() + + +def test_backtest_result_isolated_from_mutated_caller_inputs() -> None: + dates = pd.date_range("2026-01-05", periods=2, freq="B") + weights = pd.DataFrame({"A": [1.0]}, index=dates[:1]) + benchmark = pd.Series([1.0, 1.1], index=dates) + result = run_weight_backtest( + weights, + pd.DataFrame({"A": [0.0, 0.0]}, index=dates), + benchmark_nav=benchmark, + ) + + weights.iloc[0, 0] = 0.0 + benchmark.iloc[1] = 99.0 + + assert result.weights.iloc[0, 0] == 1.0 + assert result.benchmark_nav is not None + assert result.benchmark_nav.iloc[1] == 1.1 diff --git a/tests/test_data_adapter.py b/tests/test_data_adapter.py index 0f101d9..87b5af5 100644 --- a/tests/test_data_adapter.py +++ b/tests/test_data_adapter.py @@ -7,10 +7,12 @@ import pandas as pd import pytest from quant_engine.data_adapter import ( + AssetReturnSnapshot, add_vwap_proxy, apply_adj_factor, load_qtdb_daily, long_to_wide, + prepare_asset_return_snapshot, prepare_execution_inputs, prepare_stock_series, rename_tushare_columns, @@ -266,6 +268,17 @@ def test_prepare_execution_inputs_basic(tushare_long: pd.DataFrame) -> None: assert volumes.iloc[0, 0] == pytest.approx(1000.0) +def test_prepare_execution_inputs_can_select_next_session_open_price( + tushare_long: pd.DataFrame, +) -> None: + """显式 price_col=open 时应生成开盘执行价矩阵。""" + renamed = rename_tushare_columns(tushare_long) + + prices, _volumes = prepare_execution_inputs(renamed, price_col="open") + + assert prices.iloc[0, 0] == pytest.approx(10.0) + + def test_prepare_execution_inputs_no_volume() -> None: """无 volume 列 → volumes 全 1.0。""" df = pd.DataFrame( @@ -286,6 +299,177 @@ def test_prepare_execution_inputs_missing_close_raises() -> None: prepare_execution_inputs(df) +def test_prepare_execution_inputs_missing_selected_price_raises() -> None: + df = pd.DataFrame({"stock_code": ["A"], "trade_date": ["2024-01-01"], "close": [10.0]}) + with pytest.raises(ValueError, match="缺 open"): + prepare_execution_inputs(df, price_col="open") + + +# ── prepare_asset_return_snapshot ──────────────────────────── + + +def _daily_prices() -> pd.DataFrame: + return pd.DataFrame( + { + "stock_code": ["B", "A", "B", "A", "B", "A"], + "trade_date": [ + "2024-01-02", + "2024-01-01", + "2024-01-01", + "2024-01-03", + "2024-01-03", + "2024-01-02", + ], + "close": [18.0, 10.0, 20.0, 12.1, 19.8, 11.0], + } + ) + + +def test_prepare_asset_return_snapshot_is_stable_and_immutable_by_interface() -> None: + snapshot = prepare_asset_return_snapshot( + _daily_prices(), + source="qtdb_pro.hq_daily", + source_snapshot_id="hq-daily:2024-01-03:v1", + adjustment="qfq", + ) + + assert isinstance(snapshot, AssetReturnSnapshot) + assert snapshot.data_snapshot_id.startswith("asset-returns-v1:") + assert snapshot.source == "qtdb_pro.hq_daily" + assert snapshot.source_snapshot_id == "hq-daily:2024-01-03:v1" + assert snapshot.price_field == "close" + assert snapshot.adjustment == "qfq" + assert snapshot.return_method == "simple" + assert snapshot.start_date.isoformat() == "2024-01-01" + assert snapshot.end_date.isoformat() == "2024-01-03" + assert snapshot.sessions == 3 + assert snapshot.assets == ("A", "B") + + expected = pd.DataFrame( + { + "A": [np.nan, 0.1, 0.1], + "B": [np.nan, -0.1, 0.1], + }, + index=pd.to_datetime(["2024-01-01", "2024-01-02", "2024-01-03"]), + ) + expected.index.name = "trade_date" + expected.columns.name = "stock_code" + pd.testing.assert_frame_equal(snapshot.returns, expected) + + exposed = snapshot.returns + exposed.iloc[1, 0] = 999.0 + assert snapshot.returns.iloc[1, 0] == pytest.approx(0.1) + + +def test_asset_return_snapshot_identity_is_order_independent_and_content_addressed() -> None: + kwargs = { + "source": "qtdb_pro.hq_daily", + "source_snapshot_id": "hq-daily:2024-01-03:v1", + "adjustment": "none", + } + baseline = prepare_asset_return_snapshot(_daily_prices(), **kwargs) + shuffled = prepare_asset_return_snapshot( + _daily_prices().sample(frac=1.0, random_state=7), + **kwargs, + ) + changed_prices = _daily_prices().copy() + changed_prices.loc[changed_prices["close"] == 12.1, "close"] = 12.2 + changed_content = prepare_asset_return_snapshot(changed_prices, **kwargs) + changed_source = prepare_asset_return_snapshot( + _daily_prices(), + source="qtdb_pro.hq_daily", + source_snapshot_id="hq-daily:2024-01-03:v2", + adjustment="none", + ) + + assert shuffled.data_snapshot_id == baseline.data_snapshot_id + assert changed_content.data_snapshot_id != baseline.data_snapshot_id + assert changed_source.data_snapshot_id != baseline.data_snapshot_id + + +def test_prepare_asset_return_snapshot_does_not_fill_missing_prices() -> None: + prices = _daily_prices() + prices.loc[ + (prices["stock_code"] == "A") & (prices["trade_date"] == "2024-01-02"), + "close", + ] = np.nan + + snapshot = prepare_asset_return_snapshot( + prices, + source="qtdb_pro.hq_daily", + source_snapshot_id="hq-daily:missing-middle", + ) + + assert pd.isna(snapshot.returns.loc[pd.Timestamp("2024-01-02"), "A"]) + assert pd.isna(snapshot.returns.loc[pd.Timestamp("2024-01-03"), "A"]) + + +def test_prepare_asset_return_snapshot_rejects_duplicate_sessions() -> None: + duplicate = pd.concat([_daily_prices(), _daily_prices().iloc[[0]]], ignore_index=True) + + with pytest.raises(ValueError, match="duplicate"): + prepare_asset_return_snapshot( + duplicate, + source="qtdb_pro.hq_daily", + source_snapshot_id="hq-daily:duplicate", + ) + + +@pytest.mark.parametrize("invalid_price", [0.0, -1.0, np.inf]) +def test_prepare_asset_return_snapshot_rejects_invalid_prices(invalid_price: float) -> None: + prices = _daily_prices() + prices.loc[0, "close"] = invalid_price + + with pytest.raises(ValueError, match="positive finite"): + prepare_asset_return_snapshot( + prices, + source="qtdb_pro.hq_daily", + source_snapshot_id="hq-daily:invalid-price", + ) + + +@pytest.mark.parametrize( + ("source", "source_snapshot_id", "adjustment"), + [ + ("", "source-1", "none"), + ("qtdb_pro.hq_daily", "", "none"), + ("qtdb_pro.hq_daily", "source-1", ""), + ], +) +def test_prepare_asset_return_snapshot_requires_explicit_identity_semantics( + source: str, + source_snapshot_id: str, + adjustment: str, +) -> None: + with pytest.raises(ValueError, match="must be non-empty"): + prepare_asset_return_snapshot( + _daily_prices(), + source=source, + source_snapshot_id=source_snapshot_id, + adjustment=adjustment, + ) + + +def test_asset_return_snapshot_feeds_reproducible_covariance_lineage() -> None: + from quant_engine.risk import estimate_covariance_snapshot + + market_snapshot = prepare_asset_return_snapshot( + _daily_prices(), + source="qtdb_pro.hq_daily", + source_snapshot_id="hq-daily:2024-01-03:v1", + ) + covariance_snapshot = estimate_covariance_snapshot( + market_snapshot.returns, + as_of_date=market_snapshot.end_date, + lookback_sessions=3, + min_observations=2, + data_snapshot_id=market_snapshot.data_snapshot_id, + ) + + assert covariance_snapshot.data_snapshot_id == market_snapshot.data_snapshot_id + assert covariance_snapshot.snapshot_id.startswith("sample-cov-v1:") + + # ── 端到端:长表 → 适配 → alpha158 + execution ────────────── diff --git a/tests/test_execution.py b/tests/test_execution.py index d244ad2..aebda84 100644 --- a/tests/test_execution.py +++ b/tests/test_execution.py @@ -11,6 +11,7 @@ import pytest from quant_engine.execution import ( ExecutionConfig, ExecutionResult, + ExecutionSimulationResult, apply_bid_ask_spread, apply_volume_constraint, check_price_limit, @@ -19,7 +20,9 @@ from quant_engine.execution import ( compute_realized_pnl, run_end_to_end_poc, simulate_execution, + simulate_daily_ledger_with_audit, simulate_multi_day, + simulate_multi_day_with_audit, simulate_with_daily_data, total_costs, total_turnover, @@ -327,24 +330,26 @@ def test_simulate_multi_day_length_mismatch_raises(): def test_simulate_multi_day_first_day_value_equals_initial(): - """第一天 portfolio_value = initial_cash(无持仓)。""" + """零成本下第一天日末 NAV 等于初始资金。""" signals = [("d1", {"A": 1.0})] prices = [("d1", {"A": 10.0})] - positions = simulate_multi_day(signals, prices, 1_000_000.0) - # 第一天 NAV = 1_000_000(无持仓),第二天才是调仓后 + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + positions = simulate_multi_day(signals, prices, 1_000_000.0, config) assert positions[0].portfolio_value == 1_000_000.0 + assert positions[0].holdings == {"A": 100_000.0} def test_simulate_multi_day_holdings_evolution(): - """调仓后 holdings 演化。 - - 注意:positions[i] 是第 i 天 rebalance 之前的快照。 - 所以要看 d2 rebalance 后的 holdings,需要看 positions[2](d3 的快照)。 - """ + """日末快照应反映当天调仓后的 holdings。""" signals = [ ("d1", {"A": 0.5, "B": 0.5}), ("d2", {"A": 1.0, "B": 0.0}), # 全仓 A - ("d3", {"A": 1.0, "B": 0.0}), # 第三天的快照才能看到 d2 rebalance 后的 holdings + ("d3", {"A": 1.0, "B": 0.0}), ] prices = [ ("d1", {"A": 10.0, "B": 20.0}), @@ -352,9 +357,288 @@ def test_simulate_multi_day_holdings_evolution(): ("d3", {"A": 12.0, "B": 22.0}), ] positions = simulate_multi_day(signals, prices, 1_000_000.0) - # d3 的 PRE-trade snapshot 应该只有 A(B 在 d2 被平仓) - assert "B" not in positions[2].holdings - assert "A" in positions[2].holdings + assert "B" not in positions[1].holdings + assert "A" in positions[1].holdings + + +def test_simulate_multi_day_with_audit_rebalances_target_weights_by_delta(): + """相同目标权重不应在每个交易日重复买入。""" + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + targets = [(date, {"A": 1.0}) for date in ("d1", "d2", "d3")] + prices = [(date, {"A": 10.0}) for date in ("d1", "d2", "d3")] + + result = simulate_multi_day_with_audit(targets, prices, 1_000.0, config) + + assert isinstance(result, ExecutionSimulationResult) + assert [len(day.executions) for day in result.daily_executions] == [1, 0, 0] + assert result.total_turnover == pytest.approx(1_000.0) + assert [position.cash for position in result.positions] == pytest.approx([0.0, 0.0, 0.0]) + assert [position.holdings["A"] for position in result.positions] == pytest.approx( + [100.0, 100.0, 100.0] + ) + assert [position.portfolio_value for position in result.positions] == pytest.approx( + [1_000.0, 1_000.0, 1_000.0] + ) + + +def test_simulate_multi_day_with_audit_records_costs_without_replay(): + """成交成本与日末 NAV 应来自同一次状态推进。""" + targets = [("d1", {"A": 1.0}), ("d2", {"A": 1.0})] + prices = [("d1", {"A": 10.0}), ("d2", {"A": 10.0})] + + result = simulate_multi_day_with_audit(targets, prices, 1_000.0) + + first_day = result.daily_executions[0] + assert first_day.nav_before == pytest.approx(1_000.0) + assert first_day.nav_after == pytest.approx(result.positions[0].portfolio_value) + assert result.total_costs == pytest.approx(sum(r.total_cost for r in first_day.executions)) + assert result.final_portfolio_value == pytest.approx(1_000.0 - result.total_costs) + assert result.daily_executions[1].executions == () + + +def test_simulate_multi_day_with_audit_never_spends_more_cash_than_available(): + """满仓目标应按可用现金部分成交,不能用负现金隐式加杠杆。""" + result = simulate_multi_day_with_audit( + [("d1", {"A": 1.0})], + [("d1", {"A": 10.0})], + 1_000.0, + ) + + execution = result.daily_executions[0].executions[0] + assert result.positions[0].cash >= -1e-9 + assert 0 < execution.partial_fill_pct < 1 + assert execution.blocked_reason == "insufficient_cash_partial_fill" + assert result.final_portfolio_value == pytest.approx(1_000.0 - result.total_costs) + + +@pytest.mark.parametrize( + "targets", + [ + {"A": -0.1}, + {"A": 0.6, "B": 0.5}, + {"A": float("nan")}, + ], +) +def test_simulate_multi_day_with_audit_rejects_invalid_long_only_weights(targets): + """多日 A 股目标必须是有限、非负且合计不超过 100% 的权重。""" + with pytest.raises(ValueError, match="target weights"): + simulate_multi_day_with_audit( + [("d1", targets)], + [("d1", {"A": 10.0, "B": 10.0})], + 1_000.0, + ) + + +def test_simulate_multi_day_with_audit_requires_price_for_existing_holding(): + """已有持仓缺价时无法可信估值,必须失败而不是把市值记为零。""" + with pytest.raises(ValueError, match="missing price for held asset A"): + simulate_multi_day_with_audit( + [("d1", {"A": 1.0}), ("d2", {"A": 1.0})], + [("d1", {"A": 10.0}), ("d2", {})], + 1_000.0, + ) + + +def test_simulate_multi_day_with_audit_records_unpriced_target_rejection(): + """缺失价格的目标不能吞掉现金,且必须留下拒绝原因。""" + result = simulate_multi_day_with_audit( + [("d1", {"A": 1.0})], + [("d1", {"B": 10.0})], + 1_000.0, + ) + + rejection = result.daily_executions[0].executions[0] + assert rejection.stock_code == "A" + assert rejection.executed_value == 0.0 + assert rejection.partial_fill_pct == 0.0 + assert rejection.blocked_reason == "missing_price" + assert result.positions[0].cash == 1_000.0 + assert result.positions[0].holdings == {} + + +def test_simulate_multi_day_with_audit_requires_matching_dates(): + """权重与价格日期错位必须显式失败,不能按位置静默配对。""" + with pytest.raises(ValueError, match="dates must match"): + simulate_multi_day_with_audit( + [("d1", {"A": 1.0})], + [("d2", {"A": 10.0})], + 1_000.0, + ) + + +# ── 逐交易日 Ledger:成交时点与估值时点分离 ───────────────── + + +def test_daily_ledger_marks_every_session_after_sparse_open_execution() -> None: + """下一日开盘成交后,应按每日收盘价持续盯市,而非只记录调仓日。""" + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + + result = simulate_daily_ledger_with_audit( + target_weights_history=[("d1", {"A": 1.0})], + execution_price_history=[("d1", {"A": 10.0})], + valuation_price_history=[ + ("d0", {"A": 9.0}), + ("d1", {"A": 11.0}), + ("d2", {"A": 12.0}), + ], + initial_cash=1_000.0, + config=config, + ) + + assert [position.date for position in result.positions] == ["d0", "d1", "d2"] + assert [position.portfolio_value for position in result.positions] == pytest.approx( + [1_000.0, 1_100.0, 1_200.0] + ) + assert [len(day.executions) for day in result.daily_executions] == [0, 1, 0] + fill = result.daily_executions[1].executions[0] + assert fill.side == "buy" + assert fill.quantity == pytest.approx(100.0) + assert fill.price == pytest.approx(10.0) + pd.testing.assert_series_equal( + result.normalized_nav_series, + pd.Series([1.0, 1.1, 1.2], index=["d0", "d1", "d2"], dtype=float), + ) + pd.testing.assert_series_equal( + result.daily_returns, + pd.Series([0.0, 0.1, 1.2 / 1.1 - 1.0], index=["d0", "d1", "d2"]), + ) + + +def test_daily_ledger_first_session_cost_reduces_first_return() -> None: + """首个估值日发生交易时,费用必须进入相对初始资金的首日收益。""" + result = simulate_daily_ledger_with_audit( + target_weights_history=[("d0", {"A": 1.0})], + execution_price_history=[("d0", {"A": 10.0})], + valuation_price_history=[("d0", {"A": 10.0})], + initial_cash=1_000.0, + ) + + assert result.total_costs > 0 + assert result.daily_returns.iloc[0] == pytest.approx( + result.final_portfolio_value / result.initial_cash - 1.0 + ) + assert result.daily_returns.iloc[0] < 0 + + +def test_daily_ledger_nav_is_rebuildable_and_trades_are_projectable() -> None: + """Ledger 必须同时支持现金守恒校验和平台成交表投影。""" + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + result = simulate_daily_ledger_with_audit( + target_weights_history=[ + ("d1", {"A": 1.0, "B": 0.0}), + ("d2", {"A": 0.0, "B": 1.0}), + ], + execution_price_history=[ + ("d1", {"A": 10.0, "B": 20.0}), + ("d2", {"A": 11.0, "B": 22.0}), + ], + valuation_price_history=[ + ("d0", {"A": 9.0, "B": 19.0}), + ("d1", {"A": 10.5, "B": 21.0}), + ("d2", {"A": 12.0, "B": 24.0}), + ], + initial_cash=1_000.0, + config=config, + ) + + close_prices = { + "d0": {"A": 9.0, "B": 19.0}, + "d1": {"A": 10.5, "B": 21.0}, + "d2": {"A": 12.0, "B": 24.0}, + } + for position in result.positions: + rebuilt = position.cash + sum( + shares * close_prices[position.date][asset] + for asset, shares in position.holdings.items() + ) + assert position.portfolio_value == pytest.approx(rebuilt) + + trades = result.trades_frame + assert trades.columns.tolist() == [ + "trade_date", + "ts_code", + "side", + "qty", + "price", + "amount", + "fee", + "slippage", + ] + assert trades["side"].tolist() == ["buy", "sell", "buy"] + assert (trades["qty"] > 0).all() + + +def test_daily_ledger_frame_matches_platform_projection_contract() -> None: + """核心层输出稳定日频投影,但不携带 run_id 或执行数据库写入。""" + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + result = simulate_daily_ledger_with_audit( + target_weights_history=[("d1", {"A": 1.0})], + execution_price_history=[("d1", {"A": 10.0})], + valuation_price_history=[ + ("d0", {"A": 9.0}), + ("d1", {"A": 11.0}), + ("d2", {"A": 12.0}), + ], + initial_cash=1_000.0, + config=config, + ) + + ledger = result.ledger_frame + + assert ledger.columns.tolist() == [ + "trade_date", + "portfolio_value", + "nav", + "pnl", + "pnl_pct", + "position_value", + "cash", + "turnover", + ] + assert ledger["trade_date"].tolist() == ["d0", "d1", "d2"] + assert ledger["nav"].tolist() == pytest.approx([1.0, 1.1, 1.2]) + assert ledger["pnl"].tolist() == pytest.approx([0.0, 100.0, 100.0]) + assert ledger["pnl_pct"].tolist() == pytest.approx([0.0, 0.1, 1.2 / 1.1 - 1.0]) + assert ledger["position_value"].tolist() == pytest.approx([0.0, 1_100.0, 1_200.0]) + assert ledger["cash"].tolist() == pytest.approx([1_000.0, 0.0, 0.0]) + assert ledger["turnover"].tolist() == pytest.approx([0.0, 1.0, 0.0]) + + +def test_daily_ledger_rejects_missing_close_for_held_asset() -> None: + """已有持仓缺少收盘估值价时必须 fail closed。""" + with pytest.raises(ValueError, match="missing valuation price for held asset A"): + simulate_daily_ledger_with_audit( + target_weights_history=[("d0", {"A": 1.0})], + execution_price_history=[("d0", {"A": 10.0})], + valuation_price_history=[("d0", {"A": 10.0}), ("d1", {})], + initial_cash=1_000.0, + ) + + +def test_daily_ledger_requires_positive_initial_cash() -> None: + """可信收益曲线需要正初始资金作为归一化基准。""" + with pytest.raises(ValueError, match="initial_cash must be positive"): + simulate_daily_ledger_with_audit([], [], [], initial_cash=0.0) # ── v1.2.0 Phase 1:端到端 POC(run_end_to_end_poc) ───── @@ -449,6 +733,15 @@ def test_run_end_to_end_poc_costs_recorded(): result = run_end_to_end_poc(signals, prices, 1_000_000.0) assert result["total_costs"] > 0 assert result["total_turnover"] > 0 + executions = [ + execution + for daily in result["daily_executions"] + for execution in daily.executions + ] + assert result["total_costs"] == pytest.approx(sum(item.total_cost for item in executions)) + assert result["total_turnover"] == pytest.approx( + sum(item.executed_value for item in executions) + ) # ── v1.2.0 Phase 2: T+1 / 涨跌停 / 部分成交 / 买卖价差 ───── @@ -741,8 +1034,8 @@ def test_compute_realized_pnl_sell_realizes(): target_weights_history=targets, ) pnl_list = compute_realized_pnl(positions) - # 第三天(卖出兑现)应有 realized 正利润(cash 从 -800 → 2M = +2M) - assert pnl_list[2].realized_pnl > 0 + # 第二天日末快照已包含当日卖出,现金流入应在当天反映。 + assert pnl_list[1].realized_pnl > 0 # ── O3: end-to-end 端到端测试(集成多个函数) ────────────── diff --git a/tests/test_factor_library.py b/tests/test_factor_library.py new file mode 100644 index 0000000..25591fe --- /dev/null +++ b/tests/test_factor_library.py @@ -0,0 +1,173 @@ +"""Contracts for reusable factor diagnostics and transformations.""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from quant_engine.factor_library import ( + annualized_sharpe, + apply_factor_direction, + cross_sectional_momentum, + cross_sectional_pct_rank, + cross_sectional_rank_with_direction, + ic_summary, + jb_test, + kurtosis, + ols_regress, + rolling_annual_vol, + rolling_zscore, + skewness, + spearman_ic, + time_series_momentum, + turnover, + winsorize, +) + + +def test_turnover_supports_one_way_and_round_trip_conventions() -> None: + weights = pd.DataFrame({"A": [1.0, 0.0], "B": [0.0, 1.0]}) + + pd.testing.assert_series_equal(turnover(weights), pd.Series([1.0], index=[1])) + pd.testing.assert_series_equal( + turnover(weights, divide_by_two=False), pd.Series([2.0], index=[1]) + ) + assert turnover(weights.iloc[:1]).empty + + +def test_ic_functions_measure_monotonic_relationship() -> None: + factor = pd.Series([1.0, 2.0, 3.0, 4.0]) + forward = pd.Series([10.0, 20.0, 30.0, 40.0]) + + assert spearman_ic(factor, forward) == pytest.approx(1.0) + result = ic_summary(factor, forward, periods=(1,), method="pearson") + assert result.loc[1, "ic_mean"] == pytest.approx(1.0) + assert result.loc[1, "n"] == 4 + + +def test_ic_summary_rejects_unknown_method() -> None: + with pytest.raises(ValueError, match="not supported"): + ic_summary(pd.Series([1, 2, 3]), pd.Series([1, 2, 3]), method="kendall") + + +def test_winsorize_clips_tails_and_preserves_nan() -> None: + values = pd.Series([0.0, 1.0, 2.0, 100.0, np.nan]) + + result = winsorize(values, lower=0.25, upper=0.75) + + assert result.iloc[0] == pytest.approx(0.75) + assert result.iloc[3] == pytest.approx(26.5) + assert pd.isna(result.iloc[4]) + + +def test_distribution_diagnostics_handle_short_samples() -> None: + assert np.isnan(skewness(pd.Series([1.0, 2.0]))) + assert np.isnan(kurtosis(pd.Series([1.0, 2.0, 3.0]))) + jb, p_value = jb_test(pd.Series(range(7), dtype=float)) + assert np.isnan(jb) + assert np.isnan(p_value) + + +def test_distribution_diagnostics_return_finite_values() -> None: + values = pd.Series([-2.0, -1.0, -0.5, 0.0, 0.25, 0.75, 1.0, 3.0]) + + assert np.isfinite(skewness(values)) + assert np.isfinite(kurtosis(values)) + jb, p_value = jb_test(values) + assert jb >= 0 + assert 0 <= p_value <= 1 + + +def test_ols_recovers_linear_coefficients_and_residual_index() -> None: + index = pd.date_range("2026-01-01", periods=8) + factor = pd.Series(np.arange(8, dtype=float), index=index, name="factor") + target = 1.5 + 2.0 * factor + + result = ols_regress(target, factor) + + assert result.alpha == pytest.approx(1.5) + assert result.beta["factor"] == pytest.approx(2.0) + assert result.r_squared == pytest.approx(1.0) + assert result.n == 8 + assert result.resid.index.equals(index) + + +def test_ols_handles_collinear_factors_without_crashing() -> None: + x = pd.DataFrame({"a": np.arange(8, dtype=float), "b": np.arange(8, dtype=float)}) + y = pd.Series(1.0 + x["a"]) + + result = ols_regress(y, x) + + assert result.n == 8 + assert np.isfinite(result.beta).all() + np.testing.assert_allclose(result.resid, 0.0, atol=1e-12) + + +def test_ols_short_sample_returns_empty_estimate() -> None: + result = ols_regress(pd.Series([1.0, 2.0]), pd.Series([1.0, 2.0], name="x")) + + assert np.isnan(result.alpha) + assert result.beta.empty + assert result.n == 2 + + +def test_momentum_and_rolling_transforms_match_manual_values() -> None: + prices = pd.DataFrame({"A": [100.0, 110.0, 121.0, 133.1]}) + momentum = cross_sectional_momentum(prices, lookback=2, skip=0) + assert momentum.iloc[2, 0] == pytest.approx(0.21) + + returns = pd.Series([0.1, 0.1, -0.5, -0.5]) + pd.testing.assert_series_equal( + time_series_momentum(returns, lookback=2), + pd.Series([0, 1, -1, -1]), + ) + + values = pd.Series([1.0, 2.0, 3.0]) + zscore = rolling_zscore(values, window=3) + assert zscore.iloc[-1] == pytest.approx(1.0) + annual_vol = rolling_annual_vol(returns, window=2, min_periods=2, trading_days=4) + assert annual_vol.iloc[1] == pytest.approx(0.0) + + +def test_rank_helpers_support_global_and_grouped_ranking() -> None: + frame = pd.DataFrame( + {"factor": [3.0, 1.0, 2.0, 4.0], "industry": ["x", "x", "y", "y"]} + ) + + global_rank = cross_sectional_pct_rank(frame, "factor", ascending=True) + grouped_rank = cross_sectional_pct_rank( + frame, "factor", group_col="industry", ascending=True + ) + + assert global_rank.tolist() == [0.75, 0.25, 0.5, 1.0] + assert grouped_rank.tolist() == [1.0, 0.5, 0.5, 1.0] + assert cross_sectional_pct_rank(frame, "missing").empty + + +def test_factor_direction_and_directional_rank() -> None: + pe = pd.Series([10.0, 20.0], name="pe_ttm") + pd.testing.assert_series_equal(apply_factor_direction(pe), -pe) + + frame = pd.DataFrame({"pe_ttm": [10.0, 20.0], "roe": [0.1, 0.2]}) + assert cross_sectional_rank_with_direction(frame, "pe_ttm").tolist() == [1.0, 0.5] + assert cross_sectional_rank_with_direction(frame, "roe").tolist() == [0.5, 1.0] + + +@pytest.mark.parametrize("direction", ["sideways", "", "REVERSE"]) +def test_factor_direction_rejects_unknown_values(direction: str) -> None: + factor = pd.Series([1.0, 2.0], name="roe") + + with pytest.raises(ValueError, match="direction"): + apply_factor_direction(factor, direction=direction) + with pytest.raises(ValueError, match="direction"): + cross_sectional_rank_with_direction( + pd.DataFrame({"roe": factor}), "roe", direction=direction + ) + + +def test_annualized_sharpe_handles_empty_and_nonzero_returns() -> None: + assert annualized_sharpe(pd.Series(dtype=float)) == 0.0 + returns = pd.Series([0.01, -0.01, 0.02, 0.0]) + expected = returns.mean() * 252 / (returns.std() * np.sqrt(252)) + assert annualized_sharpe(returns) == pytest.approx(expected) diff --git a/tests/test_metrics.py b/tests/test_metrics.py new file mode 100644 index 0000000..cb56d11 --- /dev/null +++ b/tests/test_metrics.py @@ -0,0 +1,163 @@ +"""Mathematical contracts for the standard performance metrics.""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from quant_engine.metrics import ( + TRADING_DAYS_PER_YEAR, + annualized_return, + annualized_volatility, + benchmark_summary, + calmar_ratio, + max_drawdown, + sharpe_ratio, + sortino_ratio, + summary, + win_rate, +) + + +def test_annualized_return_uses_compounded_simple_returns() -> None: + returns = pd.Series([0.10, -0.10]) + expected = 0.99 ** (TRADING_DAYS_PER_YEAR / 2) - 1.0 + + assert annualized_return(returns) == pytest.approx(expected) + + +def test_annualized_volatility_uses_sample_standard_deviation() -> None: + returns = pd.Series([0.01, 0.03, 0.02]) + + assert annualized_volatility(returns) == pytest.approx( + returns.std() * np.sqrt(TRADING_DAYS_PER_YEAR) + ) + + +def test_sharpe_ratio_subtracts_annual_risk_free_rate() -> None: + returns = pd.Series([0.01, -0.005, 0.02, 0.0]) + + result = sharpe_ratio(returns, rf=0.02) + + assert result == pytest.approx( + (annualized_return(returns) - 0.02) / annualized_volatility(returns) + ) + + +def test_zero_volatility_metrics_return_zero() -> None: + returns = pd.Series([0.0, 0.0, 0.0]) + + assert sharpe_ratio(returns) == 0.0 + assert sortino_ratio(returns) == 0.0 + assert calmar_ratio(returns) == 0.0 + + +def test_sortino_ratio_uses_all_sessions_for_downside_deviation() -> None: + returns = pd.Series([0.02, -0.01, 0.0, -0.03]) + downside = np.minimum(returns.to_numpy(), 0.0) + downside_deviation = np.sqrt(np.mean(np.square(downside))) * np.sqrt( + TRADING_DAYS_PER_YEAR + ) + + assert sortino_ratio(returns) == pytest.approx( + annualized_return(returns) / downside_deviation + ) + + +def test_max_drawdown_includes_loss_from_initial_capital() -> None: + returns = pd.Series([-0.20, 0.0]) + + assert max_drawdown(returns) == pytest.approx(-0.20) + + +def test_max_drawdown_tracks_peak_to_trough_loss() -> None: + returns = pd.Series([0.10, -0.20, 0.05]) + + assert max_drawdown(returns) == pytest.approx(-0.20) + + +def test_metrics_clean_nan_and_infinite_values() -> None: + returns = pd.Series([0.10, np.nan, np.inf, -0.05, -np.inf]) + + assert win_rate(returns) == 0.5 + assert summary(returns)["n_days"] == 2 + + +def test_summary_aliases_match_canonical_fields() -> None: + result = summary(pd.Series([0.01, -0.02, 0.03])) + + assert result["annual_yield"] == result["ann_return"] + assert result["annual_sd"] == result["ann_volatility"] + assert result["drawback"] == result["max_drawdown"] + + +@pytest.mark.parametrize( + "metric", + [ + annualized_return, + annualized_volatility, + sharpe_ratio, + sortino_ratio, + max_drawdown, + calmar_ratio, + win_rate, + ], +) +def test_metrics_reject_non_series_input(metric) -> None: + with pytest.raises(TypeError, match=r"expected pd\.Series"): + metric([0.01, 0.02]) + + +def test_short_and_empty_series_return_zero() -> None: + assert annualized_return(pd.Series(dtype=float)) == 0.0 + assert annualized_volatility(pd.Series([0.01])) == 0.0 + assert max_drawdown(pd.Series([0.01])) == 0.0 + assert win_rate(pd.Series(dtype=float)) == 0.0 + + +def test_benchmark_summary_uses_aligned_active_returns_and_regression() -> None: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + benchmark = pd.Series([-0.01, 0.0, 0.01, 0.02], index=dates) + portfolio = 0.001 + 1.5 * benchmark + active = portfolio - benchmark + + result = benchmark_summary(portfolio, benchmark) + + assert result["n_observations"] == 4 + assert result["tracking_error"] == pytest.approx( + active.std() * np.sqrt(TRADING_DAYS_PER_YEAR) + ) + assert result["information_ratio"] == pytest.approx( + active.mean() / active.std() * np.sqrt(TRADING_DAYS_PER_YEAR) + ) + assert result["beta"] == pytest.approx(1.5) + assert result["alpha"] == pytest.approx(1.001**TRADING_DAYS_PER_YEAR - 1.0) + + +def test_benchmark_summary_rejects_silent_calendar_alignment() -> None: + portfolio = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-05", periods=2)) + benchmark = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-06", periods=2)) + + with pytest.raises(ValueError, match="matching indexes"): + benchmark_summary(portfolio, benchmark) + + +def test_benchmark_summary_rejects_missing_observations() -> None: + dates = pd.date_range("2026-01-05", periods=2) + portfolio = pd.Series([0.01, np.nan], index=dates) + benchmark = pd.Series([0.0, 0.01], index=dates) + + with pytest.raises(ValueError, match="finite"): + benchmark_summary(portfolio, benchmark) + + +def test_benchmark_summary_marks_constant_benchmark_regression_unestimable() -> None: + dates = pd.date_range("2026-01-05", periods=3) + portfolio = pd.Series([0.01, -0.01, 0.02], index=dates) + benchmark = pd.Series([0.0, 0.0, 0.0], index=dates) + + result = benchmark_summary(portfolio, benchmark) + + assert np.isnan(result["alpha"]) + assert np.isnan(result["beta"]) diff --git a/tests/test_portfolio_construction.py b/tests/test_portfolio_construction.py new file mode 100644 index 0000000..b12a345 --- /dev/null +++ b/tests/test_portfolio_construction.py @@ -0,0 +1,157 @@ +"""Factor-score portfolio construction and backtest integration contracts.""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from quant_engine.backtest import run_weight_backtest +from quant_engine.portfolio_construction import ( + equal_weight, + scores_to_target_weights, + scores_to_weight_table, + select_top_k, +) + + +def test_select_top_k_ignores_nan_and_breaks_ties_by_input_order() -> None: + scores = pd.Series([1.0, 1.0, np.nan, 0.5], index=["B", "A", "C", "D"]) + + selected = select_top_k(scores, top_k=2) + + assert selected.tolist() == ["B", "A"] + + +def test_select_top_k_can_select_lowest_scores() -> None: + scores = pd.Series([3.0, 1.0, 2.0], index=["A", "B", "C"]) + + selected = select_top_k(scores, top_k=2, largest=False) + + assert selected.tolist() == ["B", "C"] + + +def test_equal_weight_allocates_requested_gross_exposure() -> None: + result = equal_weight(pd.Index(["A", "B", "C"]), gross_exposure=0.9) + + pd.testing.assert_series_equal( + result, + pd.Series([0.3, 0.3, 0.3], index=["A", "B", "C"], name="weight"), + ) + + +def test_equal_weight_returns_empty_float_series_for_no_assets() -> None: + result = equal_weight(pd.Index([], dtype=object)) + + assert result.empty + assert result.dtype == float + assert result.name == "weight" + + +def test_scores_to_target_weights_keeps_full_universe_with_zero_for_unselected() -> None: + scores = pd.Series([0.2, 0.8, 0.5], index=["A", "B", "C"]) + + result = scores_to_target_weights(scores, top_k=2) + + pd.testing.assert_series_equal( + result, + pd.Series([0.0, 0.5, 0.5], index=scores.index, name="weight"), + ) + + +def test_scores_to_target_weights_divides_exposure_over_available_scores() -> None: + scores = pd.Series([1.0, np.nan, 0.5], index=["A", "B", "C"]) + + result = scores_to_target_weights(scores, top_k=5, gross_exposure=0.8) + + pd.testing.assert_series_equal( + result, + pd.Series([0.4, 0.0, 0.4], index=scores.index, name="weight"), + ) + + +def test_scores_to_weight_table_constructs_each_rebalance_independently() -> None: + dates = pd.to_datetime(["2026-01-05", "2026-01-07"]) + scores = pd.DataFrame( + {"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]}, + index=dates, + ) + + result = scores_to_weight_table(scores, top_k=2) + + expected = pd.DataFrame( + {"A": [0.5, 0.0], "B": [0.5, 0.5], "C": [0.0, 0.5]}, + index=dates, + ) + pd.testing.assert_frame_equal(result, expected) + + changed_future = scores.copy() + changed_future.iloc[1] = [100.0, -100.0, 0.0] + changed_result = scores_to_weight_table(changed_future, top_k=2) + pd.testing.assert_series_equal(result.iloc[0], changed_result.iloc[0]) + + +def test_effective_holding_weights_flow_into_weight_backtest() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + effective_weights = pd.DataFrame( + {"A": [1.0, 0.0], "B": [0.0, 1.0]}, + index=dates[[0, 2]], + ) + stock_returns = pd.DataFrame( + {"A": [0.10, 0.0, 0.0], "B": [0.0, 0.0, 0.20]}, + index=dates, + ) + + result = run_weight_backtest(effective_weights, stock_returns) + + pd.testing.assert_series_equal(result.nav, pd.Series([1.1, 1.1, 1.32], index=dates)) + pd.testing.assert_frame_equal(result.weights, effective_weights) + + +@pytest.mark.parametrize("top_k", [0, -1]) +def test_portfolio_construction_rejects_non_positive_top_k(top_k: int) -> None: + scores = pd.Series([1.0], index=["A"]) + + with pytest.raises(ValueError, match="top_k must be positive"): + select_top_k(scores, top_k=top_k) + + +@pytest.mark.parametrize("gross_exposure", [-0.1, np.inf, np.nan]) +def test_equal_weight_rejects_invalid_gross_exposure(gross_exposure: float) -> None: + with pytest.raises(ValueError, match="gross_exposure"): + equal_weight(pd.Index(["A"]), gross_exposure=gross_exposure) + + +def test_portfolio_construction_rejects_duplicate_assets() -> None: + duplicate_scores = pd.Series([1.0, 2.0], index=["A", "A"]) + + with pytest.raises(ValueError, match="unique asset labels"): + scores_to_target_weights(duplicate_scores, top_k=1) + + +def test_weight_table_rejects_duplicate_rebalance_dates() -> None: + duplicate_date = pd.Timestamp("2026-01-05") + scores = pd.DataFrame( + {"A": [1.0, 2.0]}, + index=[duplicate_date, duplicate_date], + ) + + with pytest.raises(ValueError, match="unique rebalance dates"): + scores_to_weight_table(scores, top_k=1) + + +def test_weight_table_rejects_unsorted_rebalance_dates() -> None: + scores = pd.DataFrame( + {"A": [1.0, 2.0]}, + index=pd.to_datetime(["2026-01-07", "2026-01-05"]), + ) + + with pytest.raises(ValueError, match="chronological order"): + scores_to_weight_table(scores, top_k=1) + + +def test_weight_table_rejects_non_numeric_scores() -> None: + scores = pd.DataFrame({"A": ["high"], "B": ["low"]}) + + with pytest.raises(TypeError, match="numeric"): + scores_to_weight_table(scores, top_k=1) diff --git a/tests/test_research_pipeline.py b/tests/test_research_pipeline.py new file mode 100644 index 0000000..310e245 --- /dev/null +++ b/tests/test_research_pipeline.py @@ -0,0 +1,336 @@ +"""No-lookahead factor-score to execution-audit integration contracts.""" + +from __future__ import annotations + +import pandas as pd +import pytest + +from quant_engine.execution import ExecutionConfig +from quant_engine.research_pipeline import ( + FactorBacktestResult, + FactorExecutionResult, + TargetWeightSchedule, + run_factor_backtest_research, + run_factor_execution_research, + schedule_target_weights, +) + + +def _calendar() -> pd.DatetimeIndex: + return pd.date_range("2026-01-05", periods=4, freq="B") + + +def _factor_scores() -> pd.DataFrame: + dates = _calendar() + return pd.DataFrame( + {"A": [2.0, 0.0], "B": [1.0, 3.0]}, + index=dates[:2], + ) + + +def _next_session_open_prices() -> pd.DataFrame: + dates = _calendar() + return pd.DataFrame( + {"A": [1.0, 10.0, 10.0, 10.0], "B": [1.0, 10.0, 20.0, 20.0]}, + index=dates, + ) + + +def test_schedule_target_weights_maps_signal_to_next_trading_session() -> None: + dates = _calendar() + decision_weights = pd.DataFrame( + {"A": [1.0, 0.0], "B": [0.0, 1.0]}, + index=dates[:2], + ) + + schedule = schedule_target_weights(decision_weights, dates, lag_sessions=1) + + assert isinstance(schedule, TargetWeightSchedule) + assert schedule.lag_sessions == 1 + pd.testing.assert_series_equal( + schedule.signal_to_execution, + pd.Series(dates[1:3], index=dates[:2], name="execution_date"), + ) + expected = decision_weights.copy() + expected.index = dates[1:3] + expected.index.name = "execution_date" + pd.testing.assert_frame_equal(schedule.execution_weights, expected) + assert (schedule.execution_weights.index > schedule.signal_to_execution.index).all() + + +def test_factor_execution_research_uses_next_session_prices() -> None: + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + + result = run_factor_execution_research( + _factor_scores(), + _next_session_open_prices(), + top_k=1, + execution_price_field="open", + initial_cash=1_000.0, + config=config, + ) + + assert isinstance(result, FactorExecutionResult) + assert result.execution_price_field == "open" + assert result.execution.daily_executions[0].date == str(_calendar()[1]) + assert result.execution.positions[0].holdings == {"A": 100.0} + assert result.execution.positions[1].holdings == {"B": 50.0} + assert result.execution.final_portfolio_value == pytest.approx(1_000.0) + + +def test_factor_execution_result_snapshots_research_inputs() -> None: + scores = _factor_scores() + prices = _next_session_open_prices() + + result = run_factor_execution_research( + scores, + prices, + top_k=1, + execution_price_field="open", + ) + scores.iloc[0, 0] = -999.0 + prices.iloc[1, 0] = 999.0 + + assert result.factor_scores.iloc[0, 0] == 2.0 + assert result.execution_prices.loc[_calendar()[1], "A"] == 10.0 + assert result.execution.positions[0].holdings["A"] < 200_000.0 + + +@pytest.mark.parametrize("lag_sessions", [0, -1, True]) +def test_schedule_target_weights_requires_positive_integer_lag(lag_sessions: int) -> None: + with pytest.raises(ValueError, match="lag_sessions"): + schedule_target_weights( + pd.DataFrame({"A": [1.0]}, index=_calendar()[:1]), + _calendar(), + lag_sessions=lag_sessions, + ) + + +def test_schedule_target_weights_rejects_signal_outside_trading_calendar() -> None: + weekend = pd.Timestamp("2026-01-10") + with pytest.raises(ValueError, match="signal dates must be trading sessions"): + schedule_target_weights( + pd.DataFrame({"A": [1.0]}, index=[weekend]), + _calendar(), + ) + + +def test_schedule_target_weights_rejects_missing_future_execution_session() -> None: + dates = _calendar() + with pytest.raises(ValueError, match="future execution session"): + schedule_target_weights( + pd.DataFrame({"A": [1.0]}, index=dates[-1:]), + dates, + ) + + +def test_factor_execution_research_requires_explicit_price_field() -> None: + with pytest.raises(ValueError, match="execution_price_field"): + run_factor_execution_research( + _factor_scores(), + _next_session_open_prices(), + top_k=1, + execution_price_field="", + ) + + +def test_factor_execution_research_accepts_empty_scores() -> None: + scores = pd.DataFrame(columns=["A", "B"], index=pd.DatetimeIndex([]), dtype=float) + + result = run_factor_execution_research( + scores, + _next_session_open_prices(), + top_k=1, + execution_price_field="open", + ) + + assert result.schedule.execution_weights.empty + assert result.execution.positions == () + + +def test_factor_backtest_research_runs_signal_to_daily_performance_without_lookahead() -> None: + """信号日保持现金,下一日开盘成交后才参与当日收盘收益。""" + dates = _calendar() + scores = pd.DataFrame({"A": [2.0], "B": [1.0]}, index=dates[:1]) + opens = pd.DataFrame( + {"A": [1.0, 10.0, 10.0, 10.0], "B": [1.0, 20.0, 20.0, 20.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [500.0, 11.0, 12.0, 12.0], "B": [500.0, 20.0, 20.0, 20.0]}, + index=dates, + ) + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + + result = run_factor_backtest_research( + scores, + execution_prices=opens, + valuation_prices=closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=config, + ) + + assert isinstance(result, FactorBacktestResult) + assert result.execution_price_field == "open" + assert result.valuation_price_field == "close" + pd.testing.assert_series_equal( + result.nav, + pd.Series([1.0, 1.1, 1.2, 1.2], index=dates, name="nav"), + ) + pd.testing.assert_series_equal( + result.returns, + pd.Series([0.0, 0.1, 1.2 / 1.1 - 1.0, 0.0], index=dates, name="returns"), + ) + assert result.stats()["n_days"] == 4 + assert result.execution.daily_executions[0].executions == () + assert result.execution.daily_executions[1].executions[0].price == 10.0 + + +def test_factor_backtest_result_snapshots_both_price_semantics() -> None: + scores = pd.DataFrame({"A": [1.0]}, index=_calendar()[:1]) + opens = pd.DataFrame({"A": [10.0, 10.0, 10.0, 10.0]}, index=_calendar()) + closes = pd.DataFrame({"A": [10.0, 11.0, 12.0, 13.0]}, index=_calendar()) + + result = run_factor_backtest_research( + scores, + execution_prices=opens, + valuation_prices=closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + ) + opens.iloc[1, 0] = 999.0 + closes.iloc[1, 0] = 999.0 + + assert result.execution_prices.iloc[1, 0] == 10.0 + assert result.valuation_prices.iloc[1, 0] == 11.0 + + +def test_factor_backtest_research_requires_matching_daily_calendars() -> None: + scores = pd.DataFrame({"A": [1.0]}, index=_calendar()[:1]) + opens = pd.DataFrame({"A": [10.0, 10.0, 10.0, 10.0]}, index=_calendar()) + closes = pd.DataFrame({"A": [10.0, 11.0, 12.0]}, index=_calendar()[:3]) + + with pytest.raises(ValueError, match="matching trading calendars"): + run_factor_backtest_research( + scores, + execution_prices=opens, + valuation_prices=closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + ) + + +def test_factor_backtest_starts_at_first_signal_instead_of_price_warmup() -> None: + """因子预热行情不能作为空仓日混入研究绩效区间。""" + dates = pd.date_range("2026-01-05", periods=5, freq="B") + scores = pd.DataFrame({"A": [1.0]}, index=dates[2:3]) + opens = pd.DataFrame({"A": [1.0, 1.0, 1.0, 10.0, 10.0]}, index=dates) + closes = pd.DataFrame({"A": [100.0, 200.0, 300.0, 11.0, 12.0]}, index=dates) + config = ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + + result = run_factor_backtest_research( + scores, + execution_prices=opens, + valuation_prices=closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=config, + ) + + assert result.nav.index.equals(dates[2:]) + pd.testing.assert_series_equal( + result.nav, + pd.Series([1.0, 1.1, 1.2], index=dates[2:], name="nav"), + ) + assert result.stats()["n_days"] == 3 + + +def test_factor_backtest_exposes_net_benchmark_metrics() -> None: + dates = _calendar() + scores = pd.DataFrame({"A": [1.0]}, index=dates[:1]) + prices = pd.DataFrame({"A": [10.0, 10.0, 11.0, 11.0]}, index=dates) + result = run_factor_backtest_research( + scores, + prices, + prices, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + config=ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ), + ) + benchmark = pd.Series([0.0, 0.01, -0.01, 0.0], index=dates) + + relative = result.benchmark_stats(benchmark) + + assert relative["n_observations"] == len(result.returns) + assert relative["tracking_error"] > 0 + + +def test_factor_backtest_projects_actual_close_weights_from_ledger() -> None: + dates = _calendar() + scores = pd.DataFrame({"A": [1.0], "B": [0.0]}, index=dates[:1]) + opens = pd.DataFrame( + {"A": [10.0, 10.0, 10.0, 10.0], "B": [20.0, 20.0, 20.0, 20.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [10.0, 11.0, 12.0, 12.0], "B": [20.0, 20.0, 20.0, 20.0]}, + index=dates, + ) + result = run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + gross_exposure=0.5, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ), + ) + + weights = result.position_weights + cash = result.cash_weights + + assert weights.index.equals(result.nav.index) + assert weights.columns.tolist() == ["A", "B"] + assert weights.loc[dates[0]].sum() == 0.0 + assert cash.loc[dates[0]] == 1.0 + assert weights.loc[dates[1], "A"] == pytest.approx(550.0 / 1_050.0) + pd.testing.assert_series_equal( + weights.sum(axis=1) + cash, + pd.Series(1.0, index=dates), + check_names=False, + ) diff --git a/tests/test_risk.py b/tests/test_risk.py new file mode 100644 index 0000000..8160e6e --- /dev/null +++ b/tests/test_risk.py @@ -0,0 +1,270 @@ +"""Risk contribution contracts and validation tests.""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from quant_engine.risk import ( + ComponentRiskResult, + CovarianceSnapshot, + component_var, + estimate_covariance_snapshot, + labeled_component_risk, + marginal_risk_contribution, + risk_contribution, +) + + +def test_estimate_covariance_snapshot_is_complete_case_and_reproducible() -> None: + dates = pd.date_range("2026-01-05", periods=6, freq="B") + returns = pd.DataFrame( + { + "A": [0.01, 0.02, 0.03, 0.04, 0.05, 99.0], + "B": [0.02, 0.01, np.nan, 0.03, 0.04, -99.0], + }, + index=dates, + ) + as_of = dates[4] + + snapshot = estimate_covariance_snapshot( + returns, + as_of_date=as_of, + lookback_sessions=4, + min_observations=3, + data_snapshot_id="market-returns-20260109-v1", + return_frequency="1d", + periods_per_year=252, + ) + + expected_window = returns.loc[:as_of].tail(4) + expected = expected_window.dropna(how="any").cov() + pd.testing.assert_frame_equal(snapshot.covariance, expected) + assert snapshot.snapshot_id.startswith("sample-cov-v1:") + assert snapshot.as_of_date == as_of.date() + assert snapshot.method == "sample" + assert snapshot.window_start_date == expected_window.index[0].date() + assert snapshot.window_end_date == as_of.date() + assert snapshot.observations == 3 + assert snapshot.lookback_sessions == 4 + assert snapshot.missing_policy == "complete_case" + assert snapshot.data_snapshot_id == "market-returns-20260109-v1" + assert len(snapshot.input_sha256) == 64 + + future_changed = returns.copy() + future_changed.loc[dates[-1], :] = [1_000_000.0, -1_000_000.0] + repeated = estimate_covariance_snapshot( + future_changed, + as_of_date=as_of, + lookback_sessions=4, + min_observations=3, + data_snapshot_id="market-returns-20260109-v1", + return_frequency="1d", + periods_per_year=252, + ) + assert repeated.snapshot_id == snapshot.snapshot_id + pd.testing.assert_frame_equal(repeated.covariance, snapshot.covariance) + + +def test_covariance_snapshot_identity_captures_data_and_estimator_contract() -> None: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + returns = pd.DataFrame( + {"A": [0.01, 0.02, -0.01, 0.03], "B": [0.02, -0.01, 0.01, 0.04]}, + index=dates, + ) + base = estimate_covariance_snapshot( + returns, + as_of_date=dates[-1], + lookback_sessions=4, + min_observations=3, + data_snapshot_id="snapshot-a", + ) + different_source = estimate_covariance_snapshot( + returns, + as_of_date=dates[-1], + lookback_sessions=4, + min_observations=3, + data_snapshot_id="snapshot-b", + ) + + assert base.snapshot_id != different_source.snapshot_id + assert base.covariance.equals(different_source.covariance) + + +def test_estimate_covariance_snapshot_rejects_ambiguous_or_insufficient_history() -> None: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + returns = pd.DataFrame( + {"A": [0.01, np.nan, 0.03, 0.04], "B": [0.02, 0.01, np.nan, 0.03]}, + index=dates, + ) + + with pytest.raises(ValueError, match="complete observations"): + estimate_covariance_snapshot( + returns, + as_of_date=dates[-1], + lookback_sessions=4, + min_observations=3, + data_snapshot_id="snapshot-a", + ) + + with pytest.raises(ValueError, match="strictly increasing"): + estimate_covariance_snapshot( + returns.iloc[::-1], + as_of_date=dates[-1], + lookback_sessions=4, + min_observations=2, + data_snapshot_id="snapshot-a", + ) + + +def test_covariance_snapshot_is_validated_and_immutable_by_interface() -> None: + covariance = pd.DataFrame( + [[0.04, 0.01], [0.01, 0.09]], + index=["A", "B"], + columns=["A", "B"], + ) + snapshot = CovarianceSnapshot( + snapshot_id="cov-20260107-v1", + as_of_date="2026-01-07", + covariance=covariance, + return_frequency="1d", + periods_per_year=252, + ) + + covariance.loc["A", "A"] = 999.0 + leaked_copy = snapshot.covariance + leaked_copy.loc["B", "B"] = 999.0 + + assert snapshot.as_of_date == pd.Timestamp("2026-01-07").date() + assert snapshot.covariance.loc["A", "A"] == pytest.approx(0.04) + assert snapshot.covariance.loc["B", "B"] == pytest.approx(0.09) + + +@pytest.mark.parametrize( + ("kwargs", "message"), + [ + ({"snapshot_id": ""}, "snapshot_id"), + ({"return_frequency": ""}, "return_frequency"), + ({"periods_per_year": 0}, "periods_per_year"), + ], +) +def test_covariance_snapshot_rejects_incomplete_identity( + kwargs: dict[str, object], + message: str, +) -> None: + values: dict[str, object] = { + "snapshot_id": "cov-20260107-v1", + "as_of_date": "2026-01-07", + "covariance": pd.DataFrame([[0.04]], index=["A"], columns=["A"]), + "return_frequency": "1d", + "periods_per_year": 252, + } + values.update(kwargs) + + with pytest.raises((TypeError, ValueError), match=message): + CovarianceSnapshot(**values) + + +def test_risk_contribution_sums_to_one_for_positive_portfolio_variance() -> None: + weights = np.array([0.5, 0.5]) + covariance = np.diag([1.0, 4.0]) + + result = risk_contribution(weights, covariance) + + np.testing.assert_allclose(result, [0.2, 0.8]) + assert result.sum() == pytest.approx(1.0) + + +def test_zero_variance_portfolio_falls_back_to_equal_contribution() -> None: + result = risk_contribution(np.array([0.2, 0.3, 0.5]), np.zeros((3, 3))) + + np.testing.assert_allclose(result, np.full(3, 1 / 3)) + + +def test_marginal_and_component_risk_follow_matrix_identities() -> None: + weights = np.array([0.25, 0.75]) + covariance = np.array([[0.04, 0.01], [0.01, 0.09]]) + + marginal = marginal_risk_contribution(weights, covariance) + component = component_var(weights, covariance) + + np.testing.assert_allclose(marginal, covariance @ weights) + np.testing.assert_allclose(component, weights * marginal) + assert component.sum() == pytest.approx(weights @ covariance @ weights) + + +@pytest.mark.parametrize( + "function", + [risk_contribution, marginal_risk_contribution, component_var], +) +def test_risk_functions_reject_covariance_shape_mismatch(function) -> None: + with pytest.raises(ValueError, match="does not match weights length"): + function(np.array([0.5, 0.5]), np.eye(3)) + + +@pytest.mark.parametrize( + "function", + [risk_contribution, marginal_risk_contribution, component_var], +) +def test_risk_functions_reject_empty_portfolio(function) -> None: + with pytest.raises(ValueError, match="at least one asset"): + function(np.array([]), np.empty((0, 0))) + + +def test_labeled_component_risk_aligns_covariance_and_closes_to_volatility() -> None: + weights = pd.Series({"A": 0.25, "B": 0.75}, name="weight") + covariance = pd.DataFrame( + [[0.09, 0.01], [0.01, 0.04]], + index=["B", "A"], + columns=["B", "A"], + ) + + result = labeled_component_risk(weights, covariance) + + aligned = covariance.reindex(index=weights.index, columns=weights.index) + expected_volatility = float(np.sqrt(weights @ aligned @ weights)) + assert isinstance(result, ComponentRiskResult) + assert result.component.index.tolist() == ["A", "B"] + assert result.portfolio_volatility == pytest.approx(expected_volatility) + assert result.component.sum() == pytest.approx(expected_volatility) + assert result.percentage.sum() == pytest.approx(1.0) + + +def test_component_risk_groups_actual_asset_contributions_by_label() -> None: + weights = pd.Series({"A": 0.2, "B": 0.3, "C": 0.5}) + covariance = pd.DataFrame(np.diag([0.04, 0.09, 0.16]), index=weights.index, columns=weights.index) + groups = pd.Series({"C": "growth", "A": "value", "B": "value"}) + + result = labeled_component_risk(weights, covariance) + grouped = result.grouped_component(groups) + + assert grouped.index.tolist() == ["growth", "value"] + assert grouped.loc["value"] == pytest.approx( + result.component.loc["A"] + result.component.loc["B"] + ) + assert grouped.sum() == pytest.approx(result.portfolio_volatility) + + +def test_labeled_component_risk_rejects_asset_label_mismatch() -> None: + weights = pd.Series({"A": 0.5, "B": 0.5}) + covariance = pd.DataFrame(np.eye(2), index=["A", "C"], columns=["A", "C"]) + + with pytest.raises(ValueError, match="same asset labels"): + labeled_component_risk(weights, covariance) + + +def test_labeled_component_risk_rejects_invalid_covariance() -> None: + weights = pd.Series({"A": 0.5, "B": 0.5}) + asymmetric = pd.DataFrame([[1.0, 0.2], [0.1, 1.0]], index=weights.index, columns=weights.index) + + with pytest.raises(ValueError, match="symmetric"): + labeled_component_risk(weights, asymmetric) + + +def test_labeled_component_risk_rejects_zero_variance_portfolio() -> None: + weights = pd.Series({"A": 0.5, "B": 0.5}) + covariance = pd.DataFrame(np.zeros((2, 2)), index=weights.index, columns=weights.index) + + with pytest.raises(ValueError, match="positive portfolio variance"): + labeled_component_risk(weights, covariance)