diff --git a/README.md b/README.md index d16bdbd..2a3cf2d 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ | 仓库 | 角色 | |---|---| -| `quant_engine` | **纯回测核心**(alpha + execution + indicators + data_adapter + backtest + metrics) | +| `quant_engine` | **纯研究核心**(alpha + execution + ledger + attribution + risk + metrics) | | `research_results` | 业务集成(47 个 proj 调度 + 注册 + 平台对接) | | `tushare2db_pro_aoge` | 数据层(行情 ELT) | | `research_platform` | 展示层(FastAPI + Next.js) | @@ -25,10 +25,11 @@ - `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark) - `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表 - `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排) -- `metrics` — 绩效(年化收益 / 波动率 / Sharpe / 最大回撤 / Calmar) +- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计 +- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效 - `factor_library` — 通用方法(turnover / winsorize / IC / OLS / jb_test) - `portfolio_decomp` — 组合分解(risk_parity / mean_variance / 因子归因) -- `risk` — 风险指标(边际 / 风险贡献) +- `risk` — ndarray 低层风险公式 + 标签安全、可分组的 Euler 成分风险分解 - `perf_stats` — 详细绩效(与 metrics 并存) - `logging` — 统一 logger(标准库 + 可选 loguru) @@ -123,6 +124,17 @@ print(factor_backtest.returns) print(factor_backtest.stats()) print(factor_backtest.execution.ledger_frame) print(factor_backtest.execution.trades_frame) +print(factor_backtest.position_weights) # 实际日末资产权重 +print(factor_backtest.cash_weights) + +# 所有分析都以实际成交后的 Ledger 为事实源,不直接使用目标权重伪造结果。 +attribution = factor_backtest.return_attribution() +print(attribution.asset_contributions) +print(attribution.transaction_cost) +print(attribution.residual) # 应接近 0;否则说明贡献未闭合到账本收益 + +# benchmark_returns 必须与成本后 factor_backtest.returns 使用完全相同的日期索引。 +print(factor_backtest.benchmark_stats(benchmark_returns)) # run_weight_backtest 是低层算子:只接受收益区间开始前已经生效的持仓权重。 # 不要把 signal-date 的 factor_scores/decision_weights 直接传给它。 diff --git a/docs/OPEN_SOURCE_REFERENCES.md b/docs/OPEN_SOURCE_REFERENCES.md new file mode 100644 index 0000000..fd2a6fa --- /dev/null +++ b/docs/OPEN_SOURCE_REFERENCES.md @@ -0,0 +1,24 @@ +# Open-source design references + +本项目采用“借鉴稳定语义、保留轻量实现”的策略。引入新量化能力前先检查成熟 +开源案例;除非维护成本和许可证收益明确优于本地小型实现,否则不增加框架级依赖。 + +## 2026-08-21:成交后归因与相对绩效 + +| 项目 | 借鉴内容 | 当前决策 | +|---|---|---| +| [Qlib](https://github.com/microsoft/qlib) | 信号时间与交易时间分离、成本前后超额收益分开报告 | 借鉴语义;不引入完整框架 | +| [Zipline](https://github.com/quantopian/zipline) | Ledger / transaction / portfolio value 状态模型 | 以现有 `ExecutionSimulationResult` 承担事实源 | +| [empyrical](https://github.com/quantopian/empyrical) | beta 协方差口径、alpha 几何年化、年化因子 | 移植小型公式;不增加老旧运行时依赖 | +| [Riskfolio-Lib](https://github.com/dcajasn/Riskfolio-Lib) | Euler component risk 与分组/因子风险贡献 | 只实现当前需要的 pandas/numpy 标签安全封装 | +| [PyPortfolioOpt](https://github.com/PyPortfolio/PyPortfolioOpt) | 协方差估计与优化器解耦 | 留作未来风险模型适配器参考 | + +当前核心不新增依赖。逐日收益归因必须从实际换仓前后持仓、成交记录、执行价和 +收盘估值推导;因子分数与目标权重只是意图,不能作为成交后归因事实源。 + +## hikyuu 的定位 + +[hikyuu](https://github.com/fasiondog/hikyuu) 的 SG / MM / CN / PG 部件化思想、 +A 股交易约束和系统组合方式仍有借鉴价值;但其完整 C++/Python 运行时、对象模型和 +数据体系不适合作为本项目核心依赖。当前原则是按真实研究链路吸收边界设计,不复制 +其框架层级,也不为了“架构完整”预先建设尚无端到端需求的抽象。 diff --git a/docs/handoff/2026-08-21-ledger-attribution.md b/docs/handoff/2026-08-21-ledger-attribution.md new file mode 100644 index 0000000..4b6ef8a --- /dev/null +++ b/docs/handoff/2026-08-21-ledger-attribution.md @@ -0,0 +1,42 @@ +# Ledger-backed attribution handoff + +## Goal + +在 `ExecutionSimulationResult` 日频 Ledger 之上增加轻量、可审计的成交后分析层: + +- 逐日隔夜 / 日内资产收益贡献; +- 佣金、印花税、滑点成本独立贡献; +- 贡献闭合到成本后日收益并显式暴露 residual; +- 严格日期对齐的 TE / IR / alpha / beta; +- 标签安全且可分组的 Euler component risk。 +- 从 Ledger 股数和收盘估值投影的实际资产 / 现金权重。 + +## Branch stack + +- 当前:`codex/ledger-attribution-20260821` +- 基线:`codex/post-execution-ledger-20260821` +- 再下层:`codex/core-contracts-20260821`(PR #2,尚待用户确认合并) + +本分支不得直接合并到 `main`。应按上述顺序逐层审阅;未经用户明确确认,不得合并 +L2 PR。 + +## Open-source decision + +调研结论记录在 `docs/OPEN_SOURCE_REFERENCES.md`。Qlib、Zipline、empyrical、 +Riskfolio-Lib 和 PyPortfolioOpt 只作为时间语义、Ledger、相对指标与 Euler 风险贡献 +的设计参考;本阶段没有新增运行时依赖。 + +## Verification + +- `pytest -q --cov=src --cov-report=term-missing`: 514 passed,9 个既有 SciPy warning,91% coverage; +- `mypy --strict src/`: 15 source files passed; +- 变更范围 `ruff check`: passed; +- 全仓 Ruff:仅 13 个既有 `tests/governance/*` PT009; +- workspace verify/status:passed,预期提示 quant_engine 非 main; +- global Gitea workflow check:passed,23 个无关仓库 warning。 + +## Next action + +先按堆叠顺序审阅 PR。基础 Ledger 分支完成后,再将本分支 rebase 到其最终提交, +运行唯一一次 `ship --ready`;随后将稳定输出适配到 `research_results` 与 +`research_platform`,不要在核心层直接写数据库。 diff --git a/src/quant_engine/attribution.py b/src/quant_engine/attribution.py new file mode 100644 index 0000000..cb0c2c8 --- /dev/null +++ b/src/quant_engine/attribution.py @@ -0,0 +1,141 @@ +"""Post-execution daily return attribution derived from the portfolio ledger. + +The ledger is the source of truth: previous-close holdings explain overnight +PnL, current-close holdings explain intraday PnL, and actual execution costs +remain a separate contribution. Target weights and factor scores are not +accepted here because they are intentions rather than realized positions. +""" + +from __future__ import annotations + +import math +from dataclasses import dataclass + +import pandas as pd + +from quant_engine.execution import ExecutionSimulationResult + +__all__ = ["DailyReturnAttribution", "compute_daily_return_attribution"] + + +@dataclass(frozen=True, slots=True, eq=False) +class DailyReturnAttribution: + """Auditable decomposition of each net portfolio return.""" + + overnight: pd.DataFrame + intraday: pd.DataFrame + transaction_cost: pd.Series + residual: pd.Series + total_return: pd.Series + + @property + def asset_contributions(self) -> pd.DataFrame: + """Return the combined overnight and intraday contribution by asset.""" + return self.overnight + self.intraday + + @property + def explained_return(self) -> pd.Series: + """Return asset contributions plus execution costs, before residual.""" + explained = self.asset_contributions.sum(axis=1) + self.transaction_cost + return explained.rename("explained_return") + + +def _validate_prices( + execution: ExecutionSimulationResult, + execution_prices: pd.DataFrame, + valuation_prices: pd.DataFrame, +) -> pd.DatetimeIndex: + if not isinstance(execution_prices, pd.DataFrame): + raise TypeError("execution_prices must be a pandas DataFrame") + if not isinstance(valuation_prices, pd.DataFrame): + raise TypeError("valuation_prices must be a pandas DataFrame") + if not isinstance(execution_prices.index, pd.DatetimeIndex): + raise TypeError("execution_prices must use a DatetimeIndex") + if not execution_prices.index.equals(valuation_prices.index): + raise ValueError("execution and valuation prices must use matching trading calendars") + if not execution_prices.columns.equals(valuation_prices.columns): + raise ValueError("execution and valuation prices must use matching asset labels") + + ledger_index = pd.DatetimeIndex(pd.Timestamp(position.date) for position in execution.positions) + if not ledger_index.equals(execution_prices.index): + raise ValueError("ledger and price histories must use matching trading calendars") + if len(execution.positions) != len(execution.daily_executions): + raise ValueError("ledger positions and executions must have matching lengths") + return execution_prices.index.copy() + + +def _price_for_held_asset( + prices: pd.DataFrame, + date: pd.Timestamp, + asset: str, + stage: str, +) -> float: + if asset not in prices.columns: + raise ValueError(f"missing {stage} price for held asset {asset} on {date}") + price = float(prices.at[date, asset]) + if not math.isfinite(price) or price <= 0: + raise ValueError(f"invalid {stage} price for held asset {asset} on {date}") + return price + + +def compute_daily_return_attribution( + execution: ExecutionSimulationResult, + execution_prices: pd.DataFrame, + valuation_prices: pd.DataFrame, +) -> DailyReturnAttribution: + """Decompose net daily returns using realized pre/post-execution holdings. + + For each session, previous-close shares earn the move from the previous + close to the current execution price; current-close shares earn the move + from execution price to current close. Actual commissions, stamp tax and + slippage are divided by the same previous NAV denominator. ``residual`` + exposes any failure of those components to close to the ledger return. + """ + index = _validate_prices(execution, execution_prices, valuation_prices) + columns = execution_prices.columns.copy() + overnight = pd.DataFrame(0.0, index=index.copy(), columns=columns) + intraday = pd.DataFrame(0.0, index=index.copy(), columns=columns) + cost = pd.Series(0.0, index=index.copy(), name="transaction_cost") + + previous_holdings: dict[str, float] = {} + previous_nav = execution.initial_cash + for row_number, (date, position, daily) in enumerate( + zip(index, execution.positions, execution.daily_executions, strict=True) + ): + if previous_nav <= 0 or not math.isfinite(previous_nav): + raise ValueError(f"previous portfolio value must be positive and finite on {date}") + + for asset, shares in previous_holdings.items(): + execution_price = _price_for_held_asset( + execution_prices, date, asset, "execution" + ) + previous_close = _price_for_held_asset( + valuation_prices, index[row_number - 1], asset, "previous valuation" + ) + overnight.at[date, asset] = shares * (execution_price - previous_close) / previous_nav + + for asset, shares in position.holdings.items(): + execution_price = _price_for_held_asset( + execution_prices, date, asset, "execution" + ) + close_price = _price_for_held_asset(valuation_prices, date, asset, "valuation") + intraday.at[date, asset] = shares * (close_price - execution_price) / previous_nav + + cost.at[date] = -sum(item.total_cost for item in daily.executions) / previous_nav + previous_holdings = position.holdings + previous_nav = position.portfolio_value + + total_return = pd.Series( + execution.daily_returns.to_numpy(copy=True), + index=index.copy(), + name="total_return", + ) + explained = (overnight + intraday).sum(axis=1) + cost + residual = (total_return - explained).rename("residual") + return DailyReturnAttribution( + overnight=overnight, + intraday=intraday, + transaction_cost=cost, + residual=residual, + total_return=total_return, + ) diff --git a/src/quant_engine/metrics.py b/src/quant_engine/metrics.py index 3f80b5e..2142500 100644 --- a/src/quant_engine/metrics.py +++ b/src/quant_engine/metrics.py @@ -130,6 +130,62 @@ def summary(r: pd.Series, rf: float = 0.0) -> Mapping[str, float]: } +def benchmark_summary( + portfolio_returns: pd.Series, + benchmark_returns: pd.Series, + *, + risk_free_daily: float = 0.0, + annualization: int = TRADING_DAYS_PER_YEAR, +) -> Mapping[str, float]: + """计算成本后组合相对基准的严格对齐绩效。 + + 与通用 ``summary`` 不同,本函数拒绝静默清洗或日期 inner join。alpha + 使用日频回归截距的几何年化;基准方差不足时 alpha/beta 为 NaN,明确 + 表示回归不可估计。 + """ + portfolio, benchmark = _validate_benchmark_inputs( + portfolio_returns, + benchmark_returns, + ) + if isinstance(annualization, bool) or not isinstance(annualization, int): + raise TypeError("annualization must be an integer") + if annualization <= 0: + raise ValueError("annualization must be positive") + if not np.isfinite(risk_free_daily): + raise ValueError("risk_free_daily must be finite") + + active = portfolio - benchmark + active_std = float(active.std()) + tracking_error = active_std * float(np.sqrt(annualization)) + information_ratio = ( + float(active.mean()) / active_std * float(np.sqrt(annualization)) + if active_std >= 1e-30 + else float("nan") + ) + + adjusted_portfolio = portfolio - risk_free_daily + adjusted_benchmark = benchmark - risk_free_daily + benchmark_variance = float(adjusted_benchmark.var()) + if benchmark_variance < 1e-30: + beta = float("nan") + alpha = float("nan") + else: + beta = float(adjusted_portfolio.cov(adjusted_benchmark) / benchmark_variance) + alpha_daily = float((adjusted_portfolio - beta * adjusted_benchmark).mean()) + alpha = ( + float((1.0 + alpha_daily) ** annualization - 1.0) + if alpha_daily > -1.0 + else float("nan") + ) + return { + "n_observations": len(portfolio), + "tracking_error": tracking_error, + "information_ratio": information_ratio, + "alpha": alpha, + "beta": beta, + } + + # ── 内部 ────────────────────────────────────── @@ -138,3 +194,29 @@ def _clean(r: pd.Series) -> pd.Series: if not isinstance(r, pd.Series): raise TypeError(f"expected pd.Series, got {type(r).__name__}") return r.replace([np.inf, -np.inf], np.nan).dropna() + + +def _validate_benchmark_inputs( + portfolio_returns: pd.Series, + benchmark_returns: pd.Series, +) -> tuple[pd.Series, pd.Series]: + if not isinstance(portfolio_returns, pd.Series): + raise TypeError("portfolio_returns must be a pandas Series") + if not isinstance(benchmark_returns, pd.Series): + raise TypeError("benchmark_returns must be a pandas Series") + if not portfolio_returns.index.equals(benchmark_returns.index): + raise ValueError("portfolio and benchmark returns must use matching indexes") + if not portfolio_returns.index.is_unique: + raise ValueError("portfolio and benchmark indexes must be unique") + if len(portfolio_returns) < 2: + raise ValueError("benchmark metrics require at least two observations") + + portfolio = portfolio_returns.astype(float, copy=True) + benchmark = benchmark_returns.astype(float, copy=True) + if not np.isfinite(portfolio.to_numpy()).all() or not np.isfinite( + benchmark.to_numpy() + ).all(): + raise ValueError("portfolio and benchmark returns must be finite") + if (portfolio < -1.0).any() or (benchmark < -1.0).any(): + raise ValueError("simple returns cannot be less than -1") + return portfolio, benchmark diff --git a/src/quant_engine/research_pipeline.py b/src/quant_engine/research_pipeline.py index 2b52e9d..bfdfb11 100644 --- a/src/quant_engine/research_pipeline.py +++ b/src/quant_engine/research_pipeline.py @@ -16,13 +16,14 @@ import numpy as np import pandas as pd from pandas.api.types import is_numeric_dtype +from quant_engine.attribution import DailyReturnAttribution, compute_daily_return_attribution from quant_engine.execution import ( ExecutionConfig, ExecutionSimulationResult, simulate_daily_ledger_with_audit, simulate_multi_day_with_audit, ) -from quant_engine.metrics import summary as metrics_summary +from quant_engine.metrics import benchmark_summary, summary as metrics_summary from quant_engine.portfolio_construction import scores_to_weight_table __all__ = [ @@ -86,10 +87,63 @@ class FactorBacktestResult: name="returns", ) + @property + def position_weights(self) -> pd.DataFrame: + """按日末实际股数、收盘估值和账本 NAV 投影资产权重。""" + weights = pd.DataFrame( + 0.0, + index=self.valuation_prices.index.copy(), + columns=self.valuation_prices.columns.copy(), + ) + for date, position in zip( + self.valuation_prices.index, + self.execution.positions, + strict=True, + ): + if position.portfolio_value <= 0: + raise ValueError(f"portfolio value must be positive on {date}") + for asset, shares in position.holdings.items(): + weights.at[date, asset] = ( + shares * float(self.valuation_prices.at[date, asset]) + / position.portfolio_value + ) + return weights + + @property + def cash_weights(self) -> pd.Series: + """返回与实际资产权重使用同一日末 NAV 分母的现金权重。""" + values = [] + for date, position in zip( + self.valuation_prices.index, + self.execution.positions, + strict=True, + ): + if position.portfolio_value <= 0: + raise ValueError(f"portfolio value must be positive on {date}") + values.append(position.cash / position.portfolio_value) + return pd.Series( + values, + index=self.valuation_prices.index.copy(), + dtype=float, + name="cash_weight", + ) + def stats(self, rf: float = 0.0) -> Mapping[str, float]: """复用标准绩效口径计算指标。""" return metrics_summary(self.returns, rf) + def return_attribution(self) -> DailyReturnAttribution: + """从实际成交后持仓与账本生成逐日净收益归因。""" + return compute_daily_return_attribution( + self.execution, + self.execution_prices, + self.valuation_prices, + ) + + def benchmark_stats(self, benchmark_returns: pd.Series) -> Mapping[str, float]: + """计算成本后日收益相对同日基准的 TE、IR、alpha 与 beta。""" + return benchmark_summary(self.returns, benchmark_returns) + def _validate_datetime_index(index: pd.Index, name: str) -> pd.DatetimeIndex: if not isinstance(index, pd.DatetimeIndex): diff --git a/src/quant_engine/risk.py b/src/quant_engine/risk.py index 95a1825..d2328fa 100644 --- a/src/quant_engine/risk.py +++ b/src/quant_engine/risk.py @@ -5,11 +5,48 @@ from __future__ import annotations +from dataclasses import dataclass from typing import Any import numpy as np +import pandas as pd from numpy.typing import NDArray +__all__ = [ + "ComponentRiskResult", + "component_var", + "labeled_component_risk", + "marginal_risk_contribution", + "risk_contribution", +] + + +@dataclass(frozen=True, slots=True, eq=False) +class ComponentRiskResult: + """Label-preserving Euler decomposition of portfolio volatility.""" + + portfolio_volatility: float + marginal: pd.Series + component: pd.Series + percentage: pd.Series + + def grouped_component(self, groups: pd.Series) -> pd.Series: + """Aggregate asset component risk by an explicitly aligned label series.""" + if not isinstance(groups, pd.Series): + raise TypeError("groups must be a pandas Series") + if not groups.index.is_unique: + raise ValueError("groups must contain unique asset labels") + if not self.component.index.difference(groups.index).empty or not groups.index.difference( + self.component.index + ).empty: + raise ValueError("groups and component risk must use the same asset labels") + aligned = groups.reindex(self.component.index) + if aligned.isna().any(): + raise ValueError("groups must contain a non-missing label for every asset") + grouped = self.component.groupby(aligned, sort=True).sum() + grouped.name = "component_risk" + return grouped + def _validate_inputs(weights: NDArray[Any], cov: NDArray[Any]) -> tuple[NDArray[Any], NDArray[Any]]: """Normalize a portfolio vector and its covariance matrix.""" @@ -59,3 +96,70 @@ def component_var(weights: NDArray[Any], cov: NDArray[Any]) -> NDArray[Any]: """成分方差: w_i · (Σw)_i; 与 RC 的关系 RC_i = CV_i / w'Σw。""" w, cov = _validate_inputs(weights, cov) return w * (cov @ w) # type: ignore[no-any-return] + + +def labeled_component_risk( + weights: pd.Series, + covariance: pd.DataFrame, +) -> ComponentRiskResult: + """Return a label-safe Euler decomposition that sums to portfolio volatility. + + The covariance matrix may use a different asset order, but its row and + column label sets must exactly match ``weights``. Invalid or indefinite + covariance input is rejected instead of silently producing misleading risk + percentages. + """ + if not isinstance(weights, pd.Series): + raise TypeError("weights must be a pandas Series") + if not isinstance(covariance, pd.DataFrame): + raise TypeError("covariance must be a pandas DataFrame") + if weights.empty: + raise ValueError("weights must contain at least one asset") + if not weights.index.is_unique: + raise ValueError("weights must contain unique asset labels") + if not covariance.index.is_unique or not covariance.columns.is_unique: + raise ValueError("covariance must contain unique asset labels") + if not weights.index.difference(covariance.index).empty or not covariance.index.difference( + weights.index + ).empty: + raise ValueError("weights and covariance must use the same asset labels") + if not weights.index.difference(covariance.columns).empty or not covariance.columns.difference( + weights.index + ).empty: + raise ValueError("weights and covariance must use the same asset labels") + + aligned_weights = weights.astype(float, copy=True) + aligned_covariance = covariance.reindex( + index=weights.index, + columns=weights.index, + ).astype(float, copy=True) + weight_values = aligned_weights.to_numpy() + covariance_values = aligned_covariance.to_numpy() + if not np.isfinite(weight_values).all(): + raise ValueError("weights must be finite") + if not np.isfinite(covariance_values).all(): + raise ValueError("covariance must be finite") + if not np.allclose(covariance_values, covariance_values.T, rtol=1e-10, atol=1e-12): + raise ValueError("covariance must be symmetric") + eigenvalues = np.linalg.eigvalsh(covariance_values) + scale = max(1.0, float(np.max(np.abs(eigenvalues)))) + if float(eigenvalues.min()) < -1e-10 * scale: + raise ValueError("covariance must be positive semidefinite") + + portfolio_variance = float(weight_values @ covariance_values @ weight_values) + if portfolio_variance <= 0 or not np.isfinite(portfolio_variance): + raise ValueError("weights and covariance must produce positive portfolio variance") + portfolio_volatility = float(np.sqrt(portfolio_variance)) + marginal_values = covariance_values @ weight_values / portfolio_volatility + component_values = weight_values * marginal_values + percentage_values = component_values / portfolio_volatility + return ComponentRiskResult( + portfolio_volatility=portfolio_volatility, + marginal=pd.Series(marginal_values, index=weights.index.copy(), name="marginal_risk"), + component=pd.Series(component_values, index=weights.index.copy(), name="component_risk"), + percentage=pd.Series( + percentage_values, + index=weights.index.copy(), + name="risk_contribution", + ), + ) diff --git a/tests/test_attribution.py b/tests/test_attribution.py new file mode 100644 index 0000000..99eccce --- /dev/null +++ b/tests/test_attribution.py @@ -0,0 +1,118 @@ +"""Post-execution return attribution contracts.""" + +from __future__ import annotations + +import pandas as pd +import pytest + +from quant_engine.attribution import DailyReturnAttribution +from quant_engine.execution import ExecutionConfig +from quant_engine.research_pipeline import run_factor_backtest_research + + +def _zero_cost_config() -> ExecutionConfig: + return ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ) + + +def test_daily_attribution_closes_across_rebalance_and_holding_days() -> None: + """开盘换仓时,隔夜和日内贡献必须来自实际换仓前后持仓。""" + dates = pd.date_range("2026-01-05", periods=4, freq="B") + scores = pd.DataFrame( + {"A": [2.0, 0.0], "B": [1.0, 3.0]}, + index=dates[:2], + ) + opens = pd.DataFrame( + {"A": [10.0, 10.0, 15.0, 15.0], "B": [20.0, 20.0, 20.0, 21.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [10.0, 12.0, 15.0, 15.0], "B": [20.0, 20.0, 18.0, 21.0]}, + index=dates, + ) + result = run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=_zero_cost_config(), + ) + + attribution = result.return_attribution() + + assert isinstance(attribution, DailyReturnAttribution) + assert attribution.overnight.loc[dates[2], "A"] == pytest.approx(0.25) + assert attribution.intraday.loc[dates[2], "B"] == pytest.approx(-0.125) + assert attribution.asset_contributions.loc[dates[3], "B"] == pytest.approx(1 / 6) + pd.testing.assert_series_equal( + attribution.total_return, + result.returns.rename("total_return"), + ) + pd.testing.assert_series_equal( + attribution.explained_return + attribution.residual, + attribution.total_return, + check_names=False, + ) + assert attribution.residual.abs().max() < 1e-12 + + +def test_daily_attribution_reports_execution_cost_separately() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + scores = pd.DataFrame({"A": [1.0]}, index=dates[:1]) + prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates) + config = ExecutionConfig( + commission_bps=10, + stamp_tax_bps=0, + slippage_bps=10, + min_trade_amount=0, + ) + result = run_factor_backtest_research( + scores, + prices, + prices, + top_k=1, + gross_exposure=0.5, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=config, + ) + + attribution = result.return_attribution() + execution = result.execution.daily_executions[1].executions[0] + + assert attribution.asset_contributions.loc[dates[1], "A"] == 0.0 + assert attribution.transaction_cost.loc[dates[1]] == pytest.approx( + -execution.total_cost / 1_000.0 + ) + assert attribution.total_return.loc[dates[1]] == pytest.approx( + attribution.transaction_cost.loc[dates[1]] + ) + assert attribution.residual.loc[dates[1]] == pytest.approx(0.0, abs=1e-12) + + +def test_return_attribution_is_empty_for_empty_research_result() -> None: + dates = pd.date_range("2026-01-05", periods=3, freq="B") + scores = pd.DataFrame(columns=["A"], index=pd.DatetimeIndex([]), dtype=float) + prices = pd.DataFrame({"A": [10.0, 10.0, 10.0]}, index=dates) + result = run_factor_backtest_research( + scores, + prices, + prices, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + ) + + attribution = result.return_attribution() + + assert attribution.overnight.empty + assert attribution.intraday.empty + assert attribution.total_return.empty diff --git a/tests/test_metrics.py b/tests/test_metrics.py index 959e6d8..44a296d 100644 --- a/tests/test_metrics.py +++ b/tests/test_metrics.py @@ -10,6 +10,7 @@ from quant_engine.metrics import ( TRADING_DAYS_PER_YEAR, annualized_return, annualized_volatility, + benchmark_summary, calmar_ratio, max_drawdown, sharpe_ratio, @@ -91,3 +92,50 @@ def test_short_and_empty_series_return_zero() -> None: assert annualized_volatility(pd.Series([0.01])) == 0.0 assert max_drawdown(pd.Series([0.01])) == 0.0 assert win_rate(pd.Series(dtype=float)) == 0.0 + + +def test_benchmark_summary_uses_aligned_active_returns_and_regression() -> None: + dates = pd.date_range("2026-01-05", periods=4, freq="B") + benchmark = pd.Series([-0.01, 0.0, 0.01, 0.02], index=dates) + portfolio = 0.001 + 1.5 * benchmark + active = portfolio - benchmark + + result = benchmark_summary(portfolio, benchmark) + + assert result["n_observations"] == 4 + assert result["tracking_error"] == pytest.approx( + active.std() * np.sqrt(TRADING_DAYS_PER_YEAR) + ) + assert result["information_ratio"] == pytest.approx( + active.mean() / active.std() * np.sqrt(TRADING_DAYS_PER_YEAR) + ) + assert result["beta"] == pytest.approx(1.5) + assert result["alpha"] == pytest.approx(1.001**TRADING_DAYS_PER_YEAR - 1.0) + + +def test_benchmark_summary_rejects_silent_calendar_alignment() -> None: + portfolio = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-05", periods=2)) + benchmark = pd.Series([0.01, 0.02], index=pd.date_range("2026-01-06", periods=2)) + + with pytest.raises(ValueError, match="matching indexes"): + benchmark_summary(portfolio, benchmark) + + +def test_benchmark_summary_rejects_missing_observations() -> None: + dates = pd.date_range("2026-01-05", periods=2) + portfolio = pd.Series([0.01, np.nan], index=dates) + benchmark = pd.Series([0.0, 0.01], index=dates) + + with pytest.raises(ValueError, match="finite"): + benchmark_summary(portfolio, benchmark) + + +def test_benchmark_summary_marks_constant_benchmark_regression_unestimable() -> None: + dates = pd.date_range("2026-01-05", periods=3) + portfolio = pd.Series([0.01, -0.01, 0.02], index=dates) + benchmark = pd.Series([0.0, 0.0, 0.0], index=dates) + + result = benchmark_summary(portfolio, benchmark) + + assert np.isnan(result["alpha"]) + assert np.isnan(result["beta"]) diff --git a/tests/test_research_pipeline.py b/tests/test_research_pipeline.py index b41ff88..310e245 100644 --- a/tests/test_research_pipeline.py +++ b/tests/test_research_pipeline.py @@ -265,3 +265,72 @@ def test_factor_backtest_starts_at_first_signal_instead_of_price_warmup() -> Non pd.Series([1.0, 1.1, 1.2], index=dates[2:], name="nav"), ) assert result.stats()["n_days"] == 3 + + +def test_factor_backtest_exposes_net_benchmark_metrics() -> None: + dates = _calendar() + scores = pd.DataFrame({"A": [1.0]}, index=dates[:1]) + prices = pd.DataFrame({"A": [10.0, 10.0, 11.0, 11.0]}, index=dates) + result = run_factor_backtest_research( + scores, + prices, + prices, + top_k=1, + execution_price_field="open", + valuation_price_field="close", + config=ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ), + ) + benchmark = pd.Series([0.0, 0.01, -0.01, 0.0], index=dates) + + relative = result.benchmark_stats(benchmark) + + assert relative["n_observations"] == len(result.returns) + assert relative["tracking_error"] > 0 + + +def test_factor_backtest_projects_actual_close_weights_from_ledger() -> None: + dates = _calendar() + scores = pd.DataFrame({"A": [1.0], "B": [0.0]}, index=dates[:1]) + opens = pd.DataFrame( + {"A": [10.0, 10.0, 10.0, 10.0], "B": [20.0, 20.0, 20.0, 20.0]}, + index=dates, + ) + closes = pd.DataFrame( + {"A": [10.0, 11.0, 12.0, 12.0], "B": [20.0, 20.0, 20.0, 20.0]}, + index=dates, + ) + result = run_factor_backtest_research( + scores, + opens, + closes, + top_k=1, + gross_exposure=0.5, + execution_price_field="open", + valuation_price_field="close", + initial_cash=1_000.0, + config=ExecutionConfig( + commission_bps=0, + stamp_tax_bps=0, + slippage_bps=0, + min_trade_amount=0, + ), + ) + + weights = result.position_weights + cash = result.cash_weights + + assert weights.index.equals(result.nav.index) + assert weights.columns.tolist() == ["A", "B"] + assert weights.loc[dates[0]].sum() == 0.0 + assert cash.loc[dates[0]] == 1.0 + assert weights.loc[dates[1], "A"] == pytest.approx(550.0 / 1_050.0) + pd.testing.assert_series_equal( + weights.sum(axis=1) + cash, + pd.Series(1.0, index=dates), + check_names=False, + ) diff --git a/tests/test_risk.py b/tests/test_risk.py index f1899be..2de4cab 100644 --- a/tests/test_risk.py +++ b/tests/test_risk.py @@ -3,9 +3,16 @@ from __future__ import annotations import numpy as np +import pandas as pd import pytest -from quant_engine.risk import component_var, marginal_risk_contribution, risk_contribution +from quant_engine.risk import ( + ComponentRiskResult, + component_var, + labeled_component_risk, + marginal_risk_contribution, + risk_contribution, +) def test_risk_contribution_sums_to_one_for_positive_portfolio_variance() -> None: @@ -52,3 +59,61 @@ def test_risk_functions_reject_covariance_shape_mismatch(function) -> None: def test_risk_functions_reject_empty_portfolio(function) -> None: with pytest.raises(ValueError, match="at least one asset"): function(np.array([]), np.empty((0, 0))) + + +def test_labeled_component_risk_aligns_covariance_and_closes_to_volatility() -> None: + weights = pd.Series({"A": 0.25, "B": 0.75}, name="weight") + covariance = pd.DataFrame( + [[0.09, 0.01], [0.01, 0.04]], + index=["B", "A"], + columns=["B", "A"], + ) + + result = labeled_component_risk(weights, covariance) + + aligned = covariance.reindex(index=weights.index, columns=weights.index) + expected_volatility = float(np.sqrt(weights @ aligned @ weights)) + assert isinstance(result, ComponentRiskResult) + assert result.component.index.tolist() == ["A", "B"] + assert result.portfolio_volatility == pytest.approx(expected_volatility) + assert result.component.sum() == pytest.approx(expected_volatility) + assert result.percentage.sum() == pytest.approx(1.0) + + +def test_component_risk_groups_actual_asset_contributions_by_label() -> None: + weights = pd.Series({"A": 0.2, "B": 0.3, "C": 0.5}) + covariance = pd.DataFrame(np.diag([0.04, 0.09, 0.16]), index=weights.index, columns=weights.index) + groups = pd.Series({"C": "growth", "A": "value", "B": "value"}) + + result = labeled_component_risk(weights, covariance) + grouped = result.grouped_component(groups) + + assert grouped.index.tolist() == ["growth", "value"] + assert grouped.loc["value"] == pytest.approx( + result.component.loc["A"] + result.component.loc["B"] + ) + assert grouped.sum() == pytest.approx(result.portfolio_volatility) + + +def test_labeled_component_risk_rejects_asset_label_mismatch() -> None: + weights = pd.Series({"A": 0.5, "B": 0.5}) + covariance = pd.DataFrame(np.eye(2), index=["A", "C"], columns=["A", "C"]) + + with pytest.raises(ValueError, match="same asset labels"): + labeled_component_risk(weights, covariance) + + +def test_labeled_component_risk_rejects_invalid_covariance() -> None: + weights = pd.Series({"A": 0.5, "B": 0.5}) + asymmetric = pd.DataFrame([[1.0, 0.2], [0.1, 1.0]], index=weights.index, columns=weights.index) + + with pytest.raises(ValueError, match="symmetric"): + labeled_component_risk(weights, asymmetric) + + +def test_labeled_component_risk_rejects_zero_variance_portfolio() -> None: + weights = pd.Series({"A": 0.5, "B": 0.5}) + covariance = pd.DataFrame(np.zeros((2, 2)), index=weights.index, columns=weights.index) + + with pytest.raises(ValueError, match="positive portfolio variance"): + labeled_component_risk(weights, covariance)