Compare commits

..
5 changed files with 1769 additions and 0 deletions
+21
View File
@@ -25,6 +25,7 @@
- `backtest` — weight-based 多日仿真(rebalance_table / compute_nav / compare_to_benchmark)
- `portfolio_construction` — 多期因子分数 → Top-K → 等权目标权重表
- `research_pipeline` — 因子日 → 下一真实交易日 → 显式执行价 → 日末估值 → 成本后绩效(防前视编排)
- `governed_pipeline` — 数据快照 → 因子版本 → 策略版本 → 回测运行 → 目标组合 → 风险决策 → Paper 订单意图;全链路带确定性 ID,风险拒绝时禁止生成订单意图
- `artifact` — 版本化、确定性、存储中立的完整 research run 事实表与 manifest
- `attribution` — 基于实际成交后持仓的隔夜 / 日内 / 交易成本逐日收益归因与闭合审计
- `metrics` — 绝对绩效 + 严格日期对齐的 TE / IR / alpha / beta 基准相对绩效
@@ -55,6 +56,9 @@ pytest # 单元测试
pytest --cov=src # 覆盖率
mypy --strict src/ # 类型检查
ruff check src/ tests/ # lint
# 无网络、无数据库、无券商的架构烟测
uv run python -m quant_engine.governed_pipeline
```
## 使用
@@ -191,6 +195,23 @@ print(backtest.stats())
print(backtest.benchmark_report())
```
## 治理垂直切片
`governed_pipeline` 不复制因子、回测、组合或执行算法,只编排现有能力并补充版本与风险契约。
调用方必须显式提供 `DatasetSnapshot`、`FactorVersion`、`StrategyVersion`、代码提交和
`RiskPolicy`。模块只会生成 `environment="paper"` 的订单意图,不连接数据库、数据供应商或
券商;风险决策为拒绝时,订单意图固定为空,直接调用创建函数也会失败关闭。
该切片对应 ResearchHub 架构的首个可执行验收链路:
```text
DatasetSnapshot → FactorVersion → StrategyVersion → BacktestRun
→ PortfolioTarget → RiskDecision → PaperOrderIntent
```
平台总架构、五仓职责和十二层能力映射仍以 `research_platform/docs/architecture/` 为权威;
本仓只拥有纯计算与离线模拟合同。
## 与 research_results 的关系
`research_results` 依赖 `quant_engine`(通过 re-export 保持向后兼容):
+372
View File
@@ -3271,6 +3271,363 @@ def evaluate_phase3_formula(name: str, **inputs: pd.Series) -> pd.Series:
return function(*(inputs[field] for field in required_inputs))
# ── Phase 4 formula contract: frozen alpha051-alpha100 surface ──────────────
# Phase 4 extends the versioned formula contract without mutating the Phase 3
# catalogue, digest, dispatch surface, or the existing formula functions.
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION = "1.0.0"
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 = (
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
)
_PHASE4_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"alpha_051": alpha_051,
"alpha_052": alpha_052,
"alpha_053": alpha_053,
"alpha_054": alpha_054,
"alpha_055": alpha_055,
"alpha_056": alpha_056,
"alpha_057": alpha_057,
"alpha_058": alpha_058,
"alpha_059": alpha_059,
"alpha_060": alpha_060,
"alpha_061": alpha_061,
"alpha_062": alpha_062,
"alpha_063": alpha_063,
"alpha_064": alpha_064,
"alpha_065": alpha_065,
"alpha_066": alpha_066,
"alpha_067": alpha_067,
"alpha_068": alpha_068,
"alpha_069": alpha_069,
"alpha_070": alpha_070,
"alpha_071": alpha_071,
"alpha_072": alpha_072,
"alpha_073": alpha_073,
"alpha_074": alpha_074,
"alpha_075": alpha_075,
"alpha_076": alpha_076,
"alpha_077": alpha_077,
"alpha_078": alpha_078,
"alpha_079": alpha_079,
"alpha_080": alpha_080,
"alpha_081": alpha_081,
"alpha_082": alpha_082,
"alpha_083": alpha_083,
"alpha_084": alpha_084,
"alpha_085": alpha_085,
"alpha_086": alpha_086,
"alpha_087": alpha_087,
"alpha_088": alpha_088,
"alpha_089": alpha_089,
"alpha_090": alpha_090,
"alpha_091": alpha_091,
"alpha_092": alpha_092,
"alpha_093": alpha_093,
"alpha_094": alpha_094,
"alpha_095": alpha_095,
"alpha_096": alpha_096,
"alpha_097": alpha_097,
"alpha_098": alpha_098,
"alpha_099": alpha_099,
"alpha_100": alpha_100,
}
_PHASE4_INPUT_CATEGORIES = {
1: "single",
2: "pair",
3: "triple",
4: "quadruple",
5: "quintuple",
}
def _build_phase4_formula_specs() -> dict[str, dict[str, Any]]:
specs: dict[str, dict[str, Any]] = {}
for alpha_id, function in _PHASE4_FORMULA_FUNCTIONS.items():
meta = ALPHA158_REGISTRY[alpha_id]
call_inputs = _phase3_call_inputs(function)
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
if input_category is None:
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
specs[alpha_id] = {
"name": alpha_id,
"contract_version": ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
"formula": meta["formula"],
"category": meta["category"],
"complexity": meta["complexity"],
"parameters": _phase3_string_list(meta, "params", alpha_id),
"description": meta["description"],
"references": _phase3_string_list(meta, "references", alpha_id),
"call_inputs": list(call_inputs),
"formula_inputs": formula_inputs,
"input_category": input_category,
}
return specs
ALPHA158_PHASE4_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
_freeze_phase3_formula_specs(_build_phase4_formula_specs())
)
def list_phase4_formulas() -> tuple[str, ...]:
"""Return the frozen alpha051-alpha100 formula IDs in stable order."""
return tuple(ALPHA158_PHASE4_FORMULA_SPECS)
def evaluate_phase4_formula(name: str, **inputs: pd.Series) -> pd.Series:
"""Evaluate a Phase 4 formula with an exact, alignment-safe input contract."""
if name not in ALPHA158_PHASE4_FORMULA_SPECS:
raise KeyError(f"formula {name!r} not registered")
spec = ALPHA158_PHASE4_FORMULA_SPECS[name]
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
missing_inputs = [field for field in required_inputs if field not in inputs]
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
if missing_inputs or unexpected_inputs:
details: list[str] = []
if missing_inputs:
details.append(f"missing inputs {missing_inputs}")
if unexpected_inputs:
details.append(f"unexpected inputs {unexpected_inputs}")
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
for field in required_inputs:
if not isinstance(inputs[field], pd.Series):
raise TypeError(f"{field} must be a pandas Series")
primary_field = required_inputs[0]
primary = inputs[primary_field]
for field in required_inputs[1:]:
if not primary.index.equals(inputs[field].index):
raise ValueError(f"{field} index must align with {primary_field}")
function = _PHASE4_FORMULA_FUNCTIONS[name]
return function(*(inputs[field] for field in required_inputs))
# ── Phase 5 formula contract: frozen alpha101-alpha150 surface ──────────────
# Phase 5 extends the versioned formula contract without mutating any earlier
# catalogue, digest, dispatch surface, or existing formula implementation.
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION = "1.0.0"
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 = (
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
)
_PHASE5_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"alpha_101": alpha_101,
"alpha_102": alpha_102,
"alpha_103": alpha_103,
"alpha_104": alpha_104,
"alpha_105": alpha_105,
"alpha_106": alpha_106,
"alpha_107": alpha_107,
"alpha_108": alpha_108,
"alpha_109": alpha_109,
"alpha_110": alpha_110,
"alpha_111": alpha_111,
"alpha_112": alpha_112,
"alpha_113": alpha_113,
"alpha_114": alpha_114,
"alpha_115": alpha_115,
"alpha_116": alpha_116,
"alpha_117": alpha_117,
"alpha_118": alpha_118,
"alpha_119": alpha_119,
"alpha_120": alpha_120,
"alpha_121": alpha_121,
"alpha_122": alpha_122,
"alpha_123": alpha_123,
"alpha_124": alpha_124,
"alpha_125": alpha_125,
"alpha_126": alpha_126,
"alpha_127": alpha_127,
"alpha_128": alpha_128,
"alpha_129": alpha_129,
"alpha_130": alpha_130,
"alpha_131": alpha_131,
"alpha_132": alpha_132,
"alpha_133": alpha_133,
"alpha_134": alpha_134,
"alpha_135": alpha_135,
"alpha_136": alpha_136,
"alpha_137": alpha_137,
"alpha_138": alpha_138,
"alpha_139": alpha_139,
"alpha_140": alpha_140,
"alpha_141": alpha_141,
"alpha_142": alpha_142,
"alpha_143": alpha_143,
"alpha_144": alpha_144,
"alpha_145": alpha_145,
"alpha_146": alpha_146,
"alpha_147": alpha_147,
"alpha_148": alpha_148,
"alpha_149": alpha_149,
"alpha_150": alpha_150,
}
def _build_phase5_formula_specs() -> dict[str, dict[str, Any]]:
specs: dict[str, dict[str, Any]] = {}
for alpha_id, function in _PHASE5_FORMULA_FUNCTIONS.items():
meta = ALPHA158_REGISTRY[alpha_id]
call_inputs = _phase3_call_inputs(function)
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
if input_category is None:
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
specs[alpha_id] = {
"name": alpha_id,
"contract_version": ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
"formula": meta["formula"],
"category": meta["category"],
"complexity": meta["complexity"],
"parameters": _phase3_string_list(meta, "params", alpha_id),
"description": meta["description"],
"references": _phase3_string_list(meta, "references", alpha_id),
"call_inputs": list(call_inputs),
"formula_inputs": formula_inputs,
"input_category": input_category,
}
return specs
ALPHA158_PHASE5_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
_freeze_phase3_formula_specs(_build_phase5_formula_specs())
)
def list_phase5_formulas() -> tuple[str, ...]:
"""Return the frozen alpha101-alpha150 formula IDs in stable order."""
return tuple(ALPHA158_PHASE5_FORMULA_SPECS)
def evaluate_phase5_formula(name: str, **inputs: pd.Series) -> pd.Series:
"""Evaluate a Phase 5 formula with an exact, alignment-safe input contract."""
if name not in ALPHA158_PHASE5_FORMULA_SPECS:
raise KeyError(f"formula {name!r} not registered")
spec = ALPHA158_PHASE5_FORMULA_SPECS[name]
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
missing_inputs = [field for field in required_inputs if field not in inputs]
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
if missing_inputs or unexpected_inputs:
details: list[str] = []
if missing_inputs:
details.append(f"missing inputs {missing_inputs}")
if unexpected_inputs:
details.append(f"unexpected inputs {unexpected_inputs}")
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
for field in required_inputs:
if not isinstance(inputs[field], pd.Series):
raise TypeError(f"{field} must be a pandas Series")
primary_field = required_inputs[0]
primary = inputs[primary_field]
for field in required_inputs[1:]:
if len(inputs[field]) != len(primary):
raise ValueError(f"{field} length must match {primary_field}")
if not primary.index.equals(inputs[field].index):
raise ValueError(f"{field} index must align with {primary_field}")
function = _PHASE5_FORMULA_FUNCTIONS[name]
return function(*(inputs[field] for field in required_inputs))
# ── Phase 6 formula contract: frozen alpha151-alpha158 surface ──────────────
# Phase 6 completes the versioned formula contract without mutating any
# earlier catalogue, digest, dispatch surface, or formula implementation.
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION = "1.0.0"
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 = (
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
)
_PHASE6_FORMULA_FUNCTIONS: dict[str, Callable[..., pd.Series]] = {
"alpha_151": alpha_151,
"alpha_152": alpha_152,
"alpha_153": alpha_153,
"alpha_154": alpha_154,
"alpha_155": alpha_155,
"alpha_156": alpha_156,
"alpha_157": alpha_157,
"alpha_158": alpha_158,
}
def _build_phase6_formula_specs() -> dict[str, dict[str, Any]]:
specs: dict[str, dict[str, Any]] = {}
for alpha_id, function in _PHASE6_FORMULA_FUNCTIONS.items():
meta = ALPHA158_REGISTRY[alpha_id]
call_inputs = _phase3_call_inputs(function)
formula_inputs = _phase3_string_list(meta, "inputs", alpha_id)
input_category = _PHASE4_INPUT_CATEGORIES.get(len(call_inputs))
if input_category is None:
raise RuntimeError(f"unsupported formula input count for {alpha_id}")
specs[alpha_id] = {
"name": alpha_id,
"contract_version": ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
"formula": meta["formula"],
"category": meta["category"],
"complexity": meta["complexity"],
"parameters": _phase3_string_list(meta, "params", alpha_id),
"description": meta["description"],
"references": _phase3_string_list(meta, "references", alpha_id),
"call_inputs": list(call_inputs),
"formula_inputs": formula_inputs,
"input_category": input_category,
}
return specs
ALPHA158_PHASE6_FORMULA_SPECS: Mapping[str, Mapping[str, Any]] = (
_freeze_phase3_formula_specs(_build_phase6_formula_specs())
)
def list_phase6_formulas() -> tuple[str, ...]:
"""Return the frozen alpha151-alpha158 formula IDs in stable order."""
return tuple(ALPHA158_PHASE6_FORMULA_SPECS)
def evaluate_phase6_formula(name: str, **inputs: pd.Series) -> pd.Series:
"""Evaluate a Phase 6 formula with an exact, alignment-safe input contract."""
if name not in ALPHA158_PHASE6_FORMULA_SPECS:
raise KeyError(f"formula {name!r} not registered")
spec = ALPHA158_PHASE6_FORMULA_SPECS[name]
required_inputs = cast(tuple[str, ...], spec["call_inputs"])
missing_inputs = [field for field in required_inputs if field not in inputs]
unexpected_inputs = sorted(field for field in inputs if field not in required_inputs)
if missing_inputs or unexpected_inputs:
details: list[str] = []
if missing_inputs:
details.append(f"missing inputs {missing_inputs}")
if unexpected_inputs:
details.append(f"unexpected inputs {unexpected_inputs}")
raise ValueError(f"invalid inputs for {name}: {'; '.join(details)}")
for field in required_inputs:
if not isinstance(inputs[field], pd.Series):
raise TypeError(f"{field} must be a pandas Series")
primary_field = required_inputs[0]
primary = inputs[primary_field]
for field in required_inputs[1:]:
if len(inputs[field]) != len(primary):
raise ValueError(f"{field} length must match {primary_field}")
if not primary.index.equals(inputs[field].index):
raise ValueError(f"{field} index must align with {primary_field}")
function = _PHASE6_FORMULA_FUNCTIONS[name]
return function(*(inputs[field] for field in required_inputs))
__all__ = [
"rank",
"delta",
@@ -3309,6 +3666,21 @@ __all__ = [
"ALPHA158_PHASE3_FORMULA_SPECS",
"list_phase3_formulas",
"evaluate_phase3_formula",
"ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE4_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE4_FORMULA_SPECS",
"list_phase4_formulas",
"evaluate_phase4_formula",
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE5_FORMULA_SPECS",
"list_phase5_formulas",
"evaluate_phase5_formula",
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE6_FORMULA_SPECS",
"list_phase6_formulas",
"evaluate_phase6_formula",
"alpha_001",
"alpha_002",
"alpha_003",
+540
View File
@@ -0,0 +1,540 @@
"""Versioned, risk-gated, paper-only quantitative research vertical slice.
The module composes existing ``quant_engine`` calculations into a small set of
storage-neutral governance contracts. It never reads a database, calls a data
vendor, loads credentials, or routes an order to a broker.
"""
from __future__ import annotations
import hashlib
import json
import math
import re
from collections.abc import Mapping
from dataclasses import asdict, dataclass
from datetime import UTC, datetime
from enum import StrEnum
from types import MappingProxyType
import pandas as pd
from quant_engine.execution import ExecutionConfig
from quant_engine.research_pipeline import FactorBacktestResult, run_factor_backtest_research
_SHA256 = re.compile(r"^[0-9a-f]{64}$")
_GIT_SHA = re.compile(r"^[0-9a-f]{40}$")
__all__ = [
"DatasetSnapshot",
"FactorVersion",
"StrategyStage",
"StrategyVersion",
"BacktestRun",
"PortfolioTarget",
"RiskPolicy",
"RiskDecisionStatus",
"RiskDecision",
"PaperOrderIntent",
"GovernedFactorSliceResult",
"evaluate_portfolio_risk",
"create_paper_order_intent",
"run_governed_factor_slice",
]
def _required_text(value: str, name: str) -> str:
normalized = value.strip()
if not normalized:
raise ValueError(f"{name} must be non-empty")
return normalized
def _aware_utc(value: datetime, name: str) -> datetime:
if not isinstance(value, datetime) or value.tzinfo is None or value.utcoffset() is None:
raise ValueError(f"{name} must be timezone-aware")
return value.astimezone(UTC)
def _canonical_value(value: object) -> object:
if value is None or isinstance(value, str | bool | int):
return value
if isinstance(value, float):
if math.isnan(value):
return "NaN"
if math.isinf(value):
return "Infinity" if value > 0 else "-Infinity"
return value
if isinstance(value, datetime):
return _aware_utc(value, "datetime").isoformat()
if isinstance(value, StrEnum):
return value.value
if isinstance(value, Mapping):
return {
str(key): _canonical_value(item)
for key, item in sorted(value.items(), key=lambda pair: str(pair[0]))
}
if isinstance(value, list | tuple):
return [_canonical_value(item) for item in value]
raise TypeError(f"unsupported canonical value: {type(value).__name__}")
def _stable_id(prefix: str, payload: Mapping[str, object]) -> str:
encoded = json.dumps(
_canonical_value(payload),
ensure_ascii=False,
sort_keys=True,
separators=(",", ":"),
allow_nan=False,
).encode("utf-8")
return f"{prefix}:{hashlib.sha256(encoded).hexdigest()}"
def _immutable_weights(values: Mapping[str, float]) -> Mapping[str, float]:
normalized: dict[str, float] = {}
for raw_asset, raw_weight in values.items():
asset = _required_text(str(raw_asset), "asset_id")
weight = float(raw_weight)
if not math.isfinite(weight) or weight < 0:
raise ValueError("portfolio weights must be finite and non-negative")
if asset in normalized:
raise ValueError(f"duplicate asset_id: {asset}")
normalized[asset] = weight
return MappingProxyType(dict(sorted(normalized.items())))
@dataclass(frozen=True, slots=True)
class DatasetSnapshot:
"""Point-in-time identity for caller-supplied research data."""
snapshot_id: str
schema_version: str
content_sha256: str
effective_at: datetime
available_at: datetime
ingested_at: datetime
def __post_init__(self) -> None:
object.__setattr__(self, "snapshot_id", _required_text(self.snapshot_id, "snapshot_id"))
object.__setattr__(
self, "schema_version", _required_text(self.schema_version, "schema_version")
)
digest = self.content_sha256.strip().lower()
if not _SHA256.fullmatch(digest):
raise ValueError("content_sha256 must be a lowercase SHA-256 digest")
object.__setattr__(self, "content_sha256", digest)
effective_at = _aware_utc(self.effective_at, "effective_at")
available_at = _aware_utc(self.available_at, "available_at")
ingested_at = _aware_utc(self.ingested_at, "ingested_at")
if not effective_at <= available_at <= ingested_at:
raise ValueError("timestamps must satisfy effective_at <= available_at <= ingested_at")
object.__setattr__(self, "effective_at", effective_at)
object.__setattr__(self, "available_at", available_at)
object.__setattr__(self, "ingested_at", ingested_at)
@dataclass(frozen=True, slots=True)
class FactorVersion:
"""Versioned factor definition identity without factor implementation duplication."""
factor_id: str
version: str
definition_sha256: str
dataset_schema_version: str
def __post_init__(self) -> None:
object.__setattr__(self, "factor_id", _required_text(self.factor_id, "factor_id"))
object.__setattr__(self, "version", _required_text(self.version, "version"))
object.__setattr__(
self,
"dataset_schema_version",
_required_text(self.dataset_schema_version, "dataset_schema_version"),
)
digest = self.definition_sha256.strip().lower()
if not _SHA256.fullmatch(digest):
raise ValueError("definition_sha256 must be a lowercase SHA-256 digest")
object.__setattr__(self, "definition_sha256", digest)
@property
def version_id(self) -> str:
return f"{self.factor_id}@{self.version}"
class StrategyStage(StrEnum):
DRAFT = "Draft"
RESEARCH = "Research"
VALIDATED = "Validated"
APPROVED = "Approved"
PAPER = "Paper"
LIVE = "Live"
PAUSED = "Paused"
RETIRED = "Retired"
@dataclass(frozen=True, slots=True)
class StrategyVersion:
"""Strategy lineage and lifecycle state used by the governed slice."""
strategy_id: str
version: str
factor_version_id: str
stage: StrategyStage
def __post_init__(self) -> None:
object.__setattr__(self, "strategy_id", _required_text(self.strategy_id, "strategy_id"))
object.__setattr__(self, "version", _required_text(self.version, "version"))
object.__setattr__(
self,
"factor_version_id",
_required_text(self.factor_version_id, "factor_version_id"),
)
if not isinstance(self.stage, StrategyStage):
raise TypeError("stage must be a StrategyStage")
@property
def version_id(self) -> str:
return f"{self.strategy_id}@{self.version}"
@dataclass(frozen=True, slots=True)
class BacktestRun:
run_id: str
dataset_snapshot_id: str
factor_version_id: str
strategy_version_id: str
code_revision: str
config_hash: str
created_at: datetime
@dataclass(frozen=True, slots=True)
class PortfolioTarget:
target_id: str
backtest_run_id: str
dataset_snapshot_id: str
weights: Mapping[str, float]
created_at: datetime
def __post_init__(self) -> None:
object.__setattr__(self, "weights", _immutable_weights(self.weights))
@property
def gross_exposure(self) -> float:
return float(sum(abs(weight) for weight in self.weights.values()))
@dataclass(frozen=True, slots=True)
class RiskPolicy:
policy_id: str
max_gross_exposure: float
max_single_asset_weight: float
max_positions: int
def __post_init__(self) -> None:
object.__setattr__(self, "policy_id", _required_text(self.policy_id, "policy_id"))
if not math.isfinite(self.max_gross_exposure) or self.max_gross_exposure <= 0:
raise ValueError("max_gross_exposure must be positive and finite")
if not math.isfinite(self.max_single_asset_weight) or not (
0 < self.max_single_asset_weight <= 1
):
raise ValueError("max_single_asset_weight must be in (0, 1]")
if (
isinstance(self.max_positions, bool)
or not isinstance(self.max_positions, int)
or self.max_positions <= 0
):
raise ValueError("max_positions must be a positive integer")
class RiskDecisionStatus(StrEnum):
APPROVED = "approved"
REJECTED = "rejected"
@dataclass(frozen=True, slots=True)
class RiskDecision:
decision_id: str
portfolio_target_id: str
policy_id: str
status: RiskDecisionStatus
reasons: tuple[str, ...]
created_at: datetime
@dataclass(frozen=True, slots=True, init=False)
class PaperOrderIntent:
intent_id: str
portfolio_target_id: str
risk_decision_id: str
environment: str
target_weights: Mapping[str, float]
created_at: datetime
def __init__(self, target: PortfolioTarget, decision: RiskDecision) -> None:
if decision.status is not RiskDecisionStatus.APPROVED:
raise ValueError("an approved risk decision is required")
if decision.portfolio_target_id != target.target_id:
raise ValueError("risk decision does not bind the supplied portfolio target")
target_weights = _immutable_weights(target.weights)
payload = {
"portfolio_target_id": target.target_id,
"risk_decision_id": decision.decision_id,
"environment": "paper",
"target_weights": target_weights,
"created_at": decision.created_at,
}
object.__setattr__(self, "intent_id", _stable_id("order-intent", payload))
object.__setattr__(self, "portfolio_target_id", target.target_id)
object.__setattr__(self, "risk_decision_id", decision.decision_id)
object.__setattr__(self, "environment", "paper")
object.__setattr__(self, "target_weights", target_weights)
object.__setattr__(self, "created_at", decision.created_at)
@dataclass(frozen=True, slots=True)
class GovernedFactorSliceResult:
dataset_snapshot: DatasetSnapshot
factor_version: FactorVersion
strategy_version: StrategyVersion
research_result: FactorBacktestResult
backtest_run: BacktestRun
portfolio_target: PortfolioTarget
risk_decision: RiskDecision
order_intent: PaperOrderIntent | None
def evaluate_portfolio_risk(
target: PortfolioTarget,
policy: RiskPolicy,
*,
created_at: datetime | None = None,
) -> RiskDecision:
"""Apply fail-closed paper risk limits to a versioned target portfolio."""
decision_time = target.created_at if created_at is None else _aware_utc(created_at, "created_at")
reasons: list[str] = []
if target.gross_exposure > policy.max_gross_exposure + 1e-12:
reasons.append(
f"gross exposure {target.gross_exposure:.12g} exceeds "
f"limit {policy.max_gross_exposure:.12g}"
)
positive_weights = [weight for weight in target.weights.values() if weight > 0]
largest_weight = max(positive_weights, default=0.0)
if largest_weight > policy.max_single_asset_weight + 1e-12:
reasons.append(
f"single-asset weight {largest_weight:.12g} exceeds "
f"limit {policy.max_single_asset_weight:.12g}"
)
if len(positive_weights) > policy.max_positions:
reasons.append(
f"position count {len(positive_weights)} exceeds limit {policy.max_positions}"
)
status = RiskDecisionStatus.REJECTED if reasons else RiskDecisionStatus.APPROVED
payload = {
"portfolio_target_id": target.target_id,
"policy_id": policy.policy_id,
"status": status,
"reasons": reasons,
"created_at": decision_time,
}
return RiskDecision(
decision_id=_stable_id("risk-decision", payload),
portfolio_target_id=target.target_id,
policy_id=policy.policy_id,
status=status,
reasons=tuple(reasons),
created_at=decision_time,
)
def create_paper_order_intent(
target: PortfolioTarget,
decision: RiskDecision,
) -> PaperOrderIntent:
"""Create a paper-only intent after an exact, approved risk decision."""
return PaperOrderIntent(target, decision)
def run_governed_factor_slice(
*,
factor_scores: pd.DataFrame,
execution_prices: pd.DataFrame,
valuation_prices: pd.DataFrame,
dataset_snapshot: DatasetSnapshot,
factor_version: FactorVersion,
strategy_version: StrategyVersion,
risk_policy: RiskPolicy,
code_revision: str,
created_at: datetime,
top_k: int,
execution_price_field: str,
valuation_price_field: str,
lag_sessions: int = 1,
gross_exposure: float = 1.0,
largest: bool = True,
initial_cash: float = 1_000_000.0,
execution_config: ExecutionConfig | None = None,
) -> GovernedFactorSliceResult:
"""Run snapshot → factor → strategy → backtest → target → risk → paper intent."""
if strategy_version.factor_version_id != factor_version.version_id:
raise ValueError("strategy factor lineage does not match factor_version")
if dataset_snapshot.schema_version != factor_version.dataset_schema_version:
raise ValueError("factor dataset schema does not match dataset snapshot")
if strategy_version.stage not in {StrategyStage.APPROVED, StrategyStage.PAPER}:
raise ValueError("strategy must be in Approved or Paper stage")
normalized_revision = code_revision.strip().lower()
if not _GIT_SHA.fullmatch(normalized_revision):
raise ValueError("code_revision must be a lowercase 40-character Git SHA")
run_time = _aware_utc(created_at, "created_at")
if dataset_snapshot.available_at > run_time or dataset_snapshot.ingested_at > run_time:
raise ValueError("dataset snapshot must be available before the research run")
if not factor_scores.empty:
latest_decision = pd.Timestamp(factor_scores.index.max())
latest_decision_date = latest_decision.date()
if latest_decision_date > run_time.date():
raise ValueError("factor_scores contain future decision dates")
config = ExecutionConfig() if execution_config is None else execution_config
if not isinstance(config, ExecutionConfig):
raise TypeError("execution_config must be an ExecutionConfig")
parameters: dict[str, object] = {
"top_k": top_k,
"execution_price_field": execution_price_field,
"valuation_price_field": valuation_price_field,
"lag_sessions": lag_sessions,
"gross_exposure": gross_exposure,
"largest": largest,
"initial_cash": initial_cash,
"execution_config": asdict(config),
}
config_hash = _stable_id("config", parameters).split(":", maxsplit=1)[1]
run_payload = {
"dataset_snapshot_id": dataset_snapshot.snapshot_id,
"dataset_content_sha256": dataset_snapshot.content_sha256,
"factor_version_id": factor_version.version_id,
"factor_definition_sha256": factor_version.definition_sha256,
"strategy_version_id": strategy_version.version_id,
"code_revision": normalized_revision,
"config_hash": config_hash,
"created_at": run_time,
}
backtest_run = BacktestRun(
run_id=_stable_id("backtest-run", run_payload),
dataset_snapshot_id=dataset_snapshot.snapshot_id,
factor_version_id=factor_version.version_id,
strategy_version_id=strategy_version.version_id,
code_revision=normalized_revision,
config_hash=config_hash,
created_at=run_time,
)
research_result = run_factor_backtest_research(
factor_scores,
execution_prices,
valuation_prices,
top_k=top_k,
execution_price_field=execution_price_field,
valuation_price_field=valuation_price_field,
lag_sessions=lag_sessions,
gross_exposure=gross_exposure,
largest=largest,
initial_cash=initial_cash,
config=config,
)
if research_result.schedule.decision_weights.empty:
raise ValueError("governed slice requires at least one target portfolio")
final_weights = {
str(asset): float(weight)
for asset, weight in research_result.schedule.decision_weights.iloc[-1].items()
}
target_payload = {
"backtest_run_id": backtest_run.run_id,
"dataset_snapshot_id": dataset_snapshot.snapshot_id,
"weights": final_weights,
"created_at": run_time,
}
portfolio_target = PortfolioTarget(
target_id=_stable_id("portfolio-target", target_payload),
backtest_run_id=backtest_run.run_id,
dataset_snapshot_id=dataset_snapshot.snapshot_id,
weights=final_weights,
created_at=run_time,
)
risk_decision = evaluate_portfolio_risk(
portfolio_target,
risk_policy,
created_at=run_time,
)
order_intent = (
create_paper_order_intent(portfolio_target, risk_decision)
if risk_decision.status is RiskDecisionStatus.APPROVED
else None
)
return GovernedFactorSliceResult(
dataset_snapshot=dataset_snapshot,
factor_version=factor_version,
strategy_version=strategy_version,
research_result=research_result,
backtest_run=backtest_run,
portfolio_target=portfolio_target,
risk_decision=risk_decision,
order_intent=order_intent,
)
def _demo() -> Mapping[str, object]:
dates = pd.date_range("2026-01-05", periods=4, freq="B")
scores = pd.DataFrame({"A": [2.0, 1.0], "B": [1.0, 2.0]}, index=dates[:2])
opens = pd.DataFrame({"A": [10.0, 10.0, 10.2, 10.3], "B": [20.0, 20.0, 20.5, 21.0]}, index=dates)
result = run_governed_factor_slice(
factor_scores=scores,
execution_prices=opens,
valuation_prices=opens * 1.01,
dataset_snapshot=DatasetSnapshot(
snapshot_id="dataset:architecture-smoke-v1",
schema_version="1.0.0",
content_sha256="a" * 64,
effective_at=datetime(2026, 1, 8, 7, tzinfo=UTC),
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
ingested_at=datetime(2026, 1, 8, 8, 5, tzinfo=UTC),
),
factor_version=FactorVersion(
factor_id="factor:architecture-smoke",
version="1.0.0",
definition_sha256="b" * 64,
dataset_schema_version="1.0.0",
),
strategy_version=StrategyVersion(
strategy_id="strategy:architecture-smoke",
version="1.0.0",
factor_version_id="factor:architecture-smoke@1.0.0",
stage=StrategyStage.APPROVED,
),
risk_policy=RiskPolicy(
policy_id="risk:architecture-smoke@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
),
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
),
)
return {
"backtest_run_id": result.backtest_run.run_id,
"portfolio_target_id": result.portfolio_target.target_id,
"risk_decision": result.risk_decision.status.value,
"order_intent_id": result.order_intent.intent_id if result.order_intent else None,
"environment": result.order_intent.environment if result.order_intent else None,
}
if __name__ == "__main__":
print(json.dumps(_demo(), ensure_ascii=False, sort_keys=True))
+516
View File
@@ -14,6 +14,15 @@ from quant_engine.alpha_factors import (
ALPHA158_PHASE3_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE3_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE3_FORMULA_SPECS,
ALPHA158_PHASE4_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE4_FORMULA_SPECS,
ALPHA158_PHASE5_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE5_FORMULA_SPECS,
ALPHA158_PHASE6_FORMULA_CATALOG_SHA256,
ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION,
ALPHA158_PHASE6_FORMULA_SPECS,
alpha_001,
alpha_002,
alpha_003,
@@ -175,9 +184,15 @@ from quant_engine.alpha_factors import (
evaluate_phase1_operator,
evaluate_phase2_operator,
evaluate_phase3_formula,
evaluate_phase4_formula,
evaluate_phase5_formula,
evaluate_phase6_formula,
list_phase1_operators,
list_phase2_operators,
list_phase3_formulas,
list_phase4_formulas,
list_phase5_formulas,
list_phase6_formulas,
correlation,
covariance,
decay_linear,
@@ -1641,3 +1656,504 @@ def test_phase3_dispatch_rejects_implicit_series_alignment():
close=inputs["close"],
volume=misaligned_volume,
)
# ── Alpha158 Phase 4: versioned alpha051-alpha100 formula contract ──────────
def test_phase4_formula_catalog_is_versioned_exact_and_content_addressed():
import hashlib
import json
from collections import Counter
expected_ids = tuple(f"alpha_{number:03d}" for number in range(51, 101))
expected_fields = {
"name",
"contract_version",
"formula",
"category",
"complexity",
"parameters",
"description",
"references",
"call_inputs",
"formula_inputs",
"input_category",
}
assert ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION == "1.0.0"
assert list_phase4_formulas() == expected_ids
assert tuple(ALPHA158_PHASE4_FORMULA_SPECS) == expected_ids
assert Counter(
spec["input_category"] for spec in ALPHA158_PHASE4_FORMULA_SPECS.values()
) == {"single": 1, "pair": 27, "triple": 11, "quadruple": 10, "quintuple": 1}
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
assert set(spec) == expected_fields
assert spec["name"] == alpha_id
assert spec["contract_version"] == ALPHA158_PHASE4_FORMULA_CONTRACT_VERSION
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
serializable_specs = {
alpha_id: {
field: list(value) if isinstance(value, tuple) else value
for field, value in spec.items()
}
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items()
}
encoded = json.dumps(
serializable_specs,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
).encode()
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE4_FORMULA_CATALOG_SHA256
assert ALPHA158_PHASE4_FORMULA_CATALOG_SHA256 == (
"858daf5e2abab5063fc28fcf7c79936096e2c76dde582fc7bab78b3458f17054"
)
def test_phase4_formula_catalog_is_recursively_immutable():
import operator
with pytest.raises(TypeError):
operator.setitem(ALPHA158_PHASE4_FORMULA_SPECS, "alpha_051", {})
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"],
"formula",
"changed",
)
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE4_FORMULA_SPECS["alpha_051"]["call_inputs"],
0,
"volume",
)
def test_phase4_catalog_freezes_callable_and_formula_inputs():
import inspect
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
function = getattr(alpha_factors_module, alpha_id)
signature_inputs = tuple(
"open" if name == "open_" else name
for name in inspect.signature(function).parameters
)
assert spec["call_inputs"] == signature_inputs
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
assert ALPHA158_PHASE4_FORMULA_SPECS["alpha_055"]["call_inputs"] == (
"open",
"high",
"low",
"volume",
"close",
)
def test_phase4_dispatch_matches_all_existing_alpha051_alpha100_functions():
inputs = _phase3_market_inputs()
for alpha_id, spec in ALPHA158_PHASE4_FORMULA_SPECS.items():
call_inputs = spec["call_inputs"]
function = getattr(alpha_factors_module, alpha_id)
expected = function(*(inputs[name] for name in call_inputs))
actual = evaluate_phase4_formula(
alpha_id,
**{name: inputs[name] for name in reversed(call_inputs)},
)
pd.testing.assert_series_equal(actual, expected)
def test_phase4_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
inputs = _phase3_market_inputs()
with pytest.raises(KeyError, match="not registered"):
evaluate_phase4_formula("alpha_050", close=inputs["close"])
with pytest.raises(ValueError, match=r"missing inputs.*low"):
evaluate_phase4_formula("alpha_051", high=inputs["high"])
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
evaluate_phase4_formula(
"alpha_051",
high=inputs["high"],
low=inputs["low"],
vwap=inputs["vwap"],
)
with pytest.raises(TypeError, match="high must be a pandas Series"):
evaluate_phase4_formula( # type: ignore[arg-type]
"alpha_051",
high=[1.0, 2.0],
low=inputs["low"],
)
def test_phase4_dispatch_rejects_implicit_series_alignment():
inputs = _phase3_market_inputs()
misaligned_low = inputs["low"].rename(index={79: 80})
with pytest.raises(ValueError, match="low index must align with high"):
evaluate_phase4_formula(
"alpha_051",
high=inputs["high"],
low=misaligned_low,
)
# ── Alpha158 Phase 5: versioned alpha101-alpha150 formula contract ──────────
def test_phase5_formula_catalog_is_versioned_exact_and_content_addressed():
import hashlib
import json
from collections import Counter
expected_ids = tuple(f"alpha_{number:03d}" for number in range(101, 151))
expected_fields = {
"name",
"contract_version",
"formula",
"category",
"complexity",
"parameters",
"description",
"references",
"call_inputs",
"formula_inputs",
"input_category",
}
assert ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION == "1.0.0"
assert list_phase5_formulas() == expected_ids
assert tuple(ALPHA158_PHASE5_FORMULA_SPECS) == expected_ids
assert Counter(
spec["input_category"] for spec in ALPHA158_PHASE5_FORMULA_SPECS.values()
) == {"pair": 33, "triple": 14, "quadruple": 3}
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
assert set(spec) == expected_fields
assert spec["name"] == alpha_id
assert spec["contract_version"] == ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
serializable_specs = {
alpha_id: {
field: list(value) if isinstance(value, tuple) else value
for field, value in spec.items()
}
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items()
}
encoded = json.dumps(
serializable_specs,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
).encode()
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE5_FORMULA_CATALOG_SHA256
assert ALPHA158_PHASE5_FORMULA_CATALOG_SHA256 == (
"3368796169c9fbd39c4a34ea137e569964b15882fbf4de25124790d548db6533"
)
def test_phase5_formula_catalog_is_recursively_immutable():
import operator
with pytest.raises(TypeError):
operator.setitem(ALPHA158_PHASE5_FORMULA_SPECS, "alpha_101", {})
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"],
"formula",
"changed",
)
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"],
0,
"volume",
)
def test_phase5_catalog_freezes_callable_and_formula_inputs():
import inspect
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
function = getattr(alpha_factors_module, alpha_id)
signature_inputs = tuple(
"open" if name == "open_" else name
for name in inspect.signature(function).parameters
)
assert spec["call_inputs"] == signature_inputs
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
assert ALPHA158_PHASE5_FORMULA_SPECS["alpha_101"]["call_inputs"] == (
"close",
"high",
"low",
)
def test_phase5_dispatch_matches_all_existing_alpha101_alpha150_functions():
inputs = _phase3_market_inputs()
for alpha_id, spec in ALPHA158_PHASE5_FORMULA_SPECS.items():
call_inputs = spec["call_inputs"]
function = getattr(alpha_factors_module, alpha_id)
expected = function(*(inputs[name] for name in call_inputs))
actual = evaluate_phase5_formula(
alpha_id,
**{name: inputs[name] for name in reversed(call_inputs)},
)
pd.testing.assert_series_equal(actual, expected)
def test_phase5_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
inputs = _phase3_market_inputs()
with pytest.raises(KeyError, match="not registered"):
evaluate_phase5_formula("alpha_100", close=inputs["close"])
with pytest.raises(KeyError, match="not registered"):
evaluate_phase5_formula("alpha_151", close=inputs["close"])
with pytest.raises(ValueError, match=r"missing inputs.*low"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=inputs["high"],
)
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=inputs["high"],
low=inputs["low"],
vwap=inputs["vwap"],
)
with pytest.raises(TypeError, match="high must be a pandas Series"):
evaluate_phase5_formula( # type: ignore[arg-type]
"alpha_101",
close=inputs["close"],
high=[1.0, 2.0],
low=inputs["low"],
)
def test_phase5_dispatch_rejects_length_and_index_alignment_errors():
inputs = _phase3_market_inputs()
shorter_low = inputs["low"].iloc[:-1]
misaligned_high = inputs["high"].rename(index={79: 80})
with pytest.raises(ValueError, match="low length must match close"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=inputs["high"],
low=shorter_low,
)
with pytest.raises(ValueError, match="high index must align with close"):
evaluate_phase5_formula(
"alpha_101",
close=inputs["close"],
high=misaligned_high,
low=inputs["low"],
)
def test_phase5_contract_is_publicly_exported():
assert {
"ALPHA158_PHASE5_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE5_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE5_FORMULA_SPECS",
"list_phase5_formulas",
"evaluate_phase5_formula",
} <= set(alpha_factors_module.__all__)
# ── Alpha158 Phase 6: versioned alpha151-alpha158 formula contract ──────────
def test_phase6_formula_catalog_is_versioned_exact_and_content_addressed():
import hashlib
import json
from collections import Counter
expected_ids = tuple(f"alpha_{number:03d}" for number in range(151, 159))
expected_fields = {
"name",
"contract_version",
"formula",
"category",
"complexity",
"parameters",
"description",
"references",
"call_inputs",
"formula_inputs",
"input_category",
}
assert ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION == "1.0.0"
assert list_phase6_formulas() == expected_ids
assert tuple(ALPHA158_PHASE6_FORMULA_SPECS) == expected_ids
assert Counter(
spec["input_category"] for spec in ALPHA158_PHASE6_FORMULA_SPECS.values()
) == {"pair": 6, "triple": 2}
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
assert set(spec) == expected_fields
assert spec["name"] == alpha_id
assert spec["contract_version"] == ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION
assert spec["formula"] == ALPHA158_REGISTRY[alpha_id]["formula"]
serializable_specs = {
alpha_id: {
field: list(value) if isinstance(value, tuple) else value
for field, value in spec.items()
}
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items()
}
encoded = json.dumps(
serializable_specs,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=False,
).encode()
assert hashlib.sha256(encoded).hexdigest() == ALPHA158_PHASE6_FORMULA_CATALOG_SHA256
assert ALPHA158_PHASE6_FORMULA_CATALOG_SHA256 == (
"70ffae16ca6cbb59a1e8644f5cdeec1d291fa75ff8b5647fb534af0083408219"
)
def test_phase6_formula_catalog_is_recursively_immutable():
import operator
with pytest.raises(TypeError):
operator.setitem(ALPHA158_PHASE6_FORMULA_SPECS, "alpha_151", {})
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"],
"formula",
"changed",
)
with pytest.raises(TypeError):
operator.setitem(
ALPHA158_PHASE6_FORMULA_SPECS["alpha_151"]["call_inputs"],
0,
"volume",
)
def test_phase6_catalog_freezes_callable_and_formula_inputs():
import inspect
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
function = getattr(alpha_factors_module, alpha_id)
signature_inputs = tuple(
"open" if name == "open_" else name
for name in inspect.signature(function).parameters
)
assert spec["call_inputs"] == signature_inputs
assert spec["formula_inputs"] == tuple(ALPHA158_REGISTRY[alpha_id]["inputs"])
assert ALPHA158_PHASE6_FORMULA_SPECS["alpha_158"]["call_inputs"] == (
"high",
"low",
"volume",
)
def test_phase6_dispatch_matches_all_existing_alpha151_alpha158_functions():
inputs = _phase3_market_inputs()
for alpha_id, spec in ALPHA158_PHASE6_FORMULA_SPECS.items():
call_inputs = spec["call_inputs"]
function = getattr(alpha_factors_module, alpha_id)
expected = function(*(inputs[name] for name in call_inputs))
actual = evaluate_phase6_formula(
alpha_id,
**{name: inputs[name] for name in reversed(call_inputs)},
)
pd.testing.assert_series_equal(actual, expected)
def test_phase6_dispatch_rejects_unknown_missing_extra_and_non_series_inputs():
inputs = _phase3_market_inputs()
with pytest.raises(KeyError, match="not registered"):
evaluate_phase6_formula("alpha_150", close=inputs["close"])
with pytest.raises(KeyError, match="not registered"):
evaluate_phase6_formula("alpha_159", close=inputs["close"])
with pytest.raises(ValueError, match=r"missing inputs.*volume"):
evaluate_phase6_formula("alpha_151", close=inputs["close"])
with pytest.raises(ValueError, match=r"unexpected inputs.*vwap"):
evaluate_phase6_formula(
"alpha_151",
close=inputs["close"],
volume=inputs["volume"],
vwap=inputs["vwap"],
)
with pytest.raises(TypeError, match="volume must be a pandas Series"):
evaluate_phase6_formula( # type: ignore[arg-type]
"alpha_151",
close=inputs["close"],
volume=[1.0, 2.0],
)
def test_phase6_dispatch_rejects_length_and_index_alignment_errors():
inputs = _phase3_market_inputs()
shorter_volume = inputs["volume"].iloc[:-1]
misaligned_low = inputs["low"].rename(index={79: 80})
with pytest.raises(ValueError, match="volume length must match close"):
evaluate_phase6_formula(
"alpha_151",
close=inputs["close"],
volume=shorter_volume,
)
with pytest.raises(ValueError, match="low index must align with high"):
evaluate_phase6_formula(
"alpha_158",
high=inputs["high"],
low=misaligned_low,
volume=inputs["volume"],
)
def test_phase6_preserves_complete_alpha001_alpha158_formula_body_fingerprint():
import ast
import hashlib
import inspect
import json
import textwrap
fingerprints = {}
for number in range(1, 159):
alpha_id = f"alpha_{number:03d}"
source = textwrap.dedent(inspect.getsource(getattr(alpha_factors_module, alpha_id)))
node = ast.parse(source).body[0]
assert isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef))
body = ast.dump(
ast.Module(body=node.body, type_ignores=[]),
include_attributes=False,
)
fingerprints[alpha_id] = hashlib.sha256(body.encode()).hexdigest()
encoded = json.dumps(
fingerprints,
sort_keys=True,
separators=(",", ":"),
).encode()
assert hashlib.sha256(encoded).hexdigest() == (
"f3ae807983cf8ca083d0b924c3807ffd84a62a5bd354f9efb1a1a16ca40c7da6"
)
def test_phase6_contract_is_publicly_exported():
assert {
"ALPHA158_PHASE6_FORMULA_CONTRACT_VERSION",
"ALPHA158_PHASE6_FORMULA_CATALOG_SHA256",
"ALPHA158_PHASE6_FORMULA_SPECS",
"list_phase6_formulas",
"evaluate_phase6_formula",
} <= set(alpha_factors_module.__all__)
+320
View File
@@ -0,0 +1,320 @@
"""Governed Personal Quant OS vertical-slice contracts."""
from __future__ import annotations
from datetime import UTC, datetime
import pandas as pd
import pytest
from quant_engine.execution import ExecutionConfig
from quant_engine.governed_pipeline import (
DatasetSnapshot,
FactorVersion,
PaperOrderIntent,
RiskDecisionStatus,
RiskPolicy,
StrategyStage,
StrategyVersion,
create_paper_order_intent,
run_governed_factor_slice,
)
def _calendar() -> pd.DatetimeIndex:
return pd.date_range("2026-01-05", periods=4, freq="B")
def _scores() -> pd.DataFrame:
dates = _calendar()
return pd.DataFrame(
{"A": [3.0, 1.0], "B": [2.0, 3.0], "C": [1.0, 2.0]},
index=dates[:2],
)
def _prices() -> tuple[pd.DataFrame, pd.DataFrame]:
dates = _calendar()
opens = pd.DataFrame(
{"A": [10.0, 10.0, 10.2, 10.4], "B": [20.0, 20.0, 20.5, 21.0], "C": [30.0, 30.0, 30.0, 30.0]},
index=dates,
)
closes = opens * 1.01
return opens, closes
def _snapshot() -> DatasetSnapshot:
return DatasetSnapshot(
snapshot_id="dataset:cn-a-daily-20260108-v1",
schema_version="1.0.0",
content_sha256="a" * 64,
effective_at=datetime(2026, 1, 8, 7, tzinfo=UTC),
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
ingested_at=datetime(2026, 1, 8, 8, 5, tzinfo=UTC),
)
def _factor() -> FactorVersion:
return FactorVersion(
factor_id="factor:demo-momentum",
version="1.0.0",
definition_sha256="b" * 64,
dataset_schema_version="1.0.0",
)
def _strategy() -> StrategyVersion:
return StrategyVersion(
strategy_id="strategy:demo-top2",
version="1.0.0",
factor_version_id="factor:demo-momentum@1.0.0",
stage=StrategyStage.APPROVED,
)
def _execution_config() -> ExecutionConfig:
return ExecutionConfig(
commission_bps=0,
stamp_tax_bps=0,
slippage_bps=0,
min_trade_amount=0,
)
def test_governed_slice_is_reproducible_and_creates_only_paper_intent() -> None:
opens, closes = _prices()
created_at = datetime(2026, 1, 9, 1, tzinfo=UTC)
policy = RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
)
result = run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=_strategy(),
risk_policy=policy,
code_revision="c" * 40,
created_at=created_at,
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
assert result.backtest_run.dataset_snapshot_id == _snapshot().snapshot_id
assert result.backtest_run.factor_version_id == _factor().version_id
assert result.backtest_run.strategy_version_id == _strategy().version_id
assert result.backtest_run.code_revision == "c" * 40
assert len(result.backtest_run.config_hash) == 64
assert result.portfolio_target.backtest_run_id == result.backtest_run.run_id
assert result.risk_decision.status is RiskDecisionStatus.APPROVED
assert result.risk_decision.portfolio_target_id == result.portfolio_target.target_id
assert result.order_intent is not None
assert result.order_intent.environment == "paper"
assert result.order_intent.risk_decision_id == result.risk_decision.decision_id
assert result.order_intent.portfolio_target_id == result.portfolio_target.target_id
repeated = run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=_strategy(),
risk_policy=policy,
code_revision="c" * 40,
created_at=created_at,
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
assert repeated.backtest_run.run_id == result.backtest_run.run_id
assert repeated.portfolio_target.target_id == result.portfolio_target.target_id
assert repeated.risk_decision.decision_id == result.risk_decision.decision_id
assert repeated.order_intent == result.order_intent
def test_risk_rejection_blocks_order_intent() -> None:
opens, closes = _prices()
result = run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=_strategy(),
risk_policy=RiskPolicy(
policy_id="risk:no-concentration@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.4,
max_positions=10,
),
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
assert result.risk_decision.status is RiskDecisionStatus.REJECTED
assert any("single-asset weight" in reason for reason in result.risk_decision.reasons)
assert result.order_intent is None
with pytest.raises(ValueError, match="approved risk decision"):
create_paper_order_intent(result.portfolio_target, result.risk_decision)
with pytest.raises(ValueError, match="approved risk decision"):
PaperOrderIntent(result.portfolio_target, result.risk_decision)
def test_dataset_snapshot_requires_point_in_time_ordering_and_aware_times() -> None:
with pytest.raises(ValueError, match="timezone-aware"):
DatasetSnapshot(
snapshot_id="dataset:invalid",
schema_version="1.0.0",
content_sha256="a" * 64,
effective_at=datetime(2026, 1, 8, 7),
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
ingested_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
)
with pytest.raises(ValueError, match="effective_at <= available_at <= ingested_at"):
DatasetSnapshot(
snapshot_id="dataset:invalid",
schema_version="1.0.0",
content_sha256="a" * 64,
effective_at=datetime(2026, 1, 8, 9, tzinfo=UTC),
available_at=datetime(2026, 1, 8, 8, tzinfo=UTC),
ingested_at=datetime(2026, 1, 8, 10, tzinfo=UTC),
)
def test_strategy_factor_lineage_must_match() -> None:
opens, closes = _prices()
mismatched = StrategyVersion(
strategy_id="strategy:demo-top2",
version="1.0.0",
factor_version_id="factor:other@1.0.0",
stage=StrategyStage.APPROVED,
)
with pytest.raises(ValueError, match="factor lineage"):
run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=mismatched,
risk_policy=RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
),
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
def test_governed_slice_requires_matching_schema_and_snapshot_available_by_run_time() -> None:
opens, closes = _prices()
common = {
"factor_scores": _scores(),
"execution_prices": opens,
"valuation_prices": closes,
"strategy_version": _strategy(),
"risk_policy": RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
),
"code_revision": "c" * 40,
"top_k": 2,
"execution_price_field": "open",
"valuation_price_field": "close",
"execution_config": _execution_config(),
}
with pytest.raises(ValueError, match="dataset schema"):
run_governed_factor_slice(
dataset_snapshot=_snapshot(),
factor_version=FactorVersion(
factor_id="factor:demo-momentum",
version="1.0.0",
definition_sha256="b" * 64,
dataset_schema_version="2.0.0",
),
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
**common,
)
with pytest.raises(ValueError, match="available before the research run"):
run_governed_factor_slice(
dataset_snapshot=_snapshot(),
factor_version=_factor(),
created_at=datetime(2026, 1, 8, 7, 30, tzinfo=UTC),
**common,
)
future_scores = _scores()
future_scores.index = pd.date_range("2026-01-12", periods=2, freq="B")
with pytest.raises(ValueError, match="future decision dates"):
run_governed_factor_slice(
dataset_snapshot=_snapshot(),
factor_version=_factor(),
factor_scores=future_scores,
execution_prices=opens,
valuation_prices=closes,
strategy_version=common["strategy_version"],
risk_policy=common["risk_policy"],
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)
def test_paper_intent_requires_approved_strategy_stage() -> None:
opens, closes = _prices()
validated = StrategyVersion(
strategy_id="strategy:demo-top2",
version="1.0.0",
factor_version_id=_factor().version_id,
stage=StrategyStage.VALIDATED,
)
with pytest.raises(ValueError, match="Approved or Paper"):
run_governed_factor_slice(
factor_scores=_scores(),
execution_prices=opens,
valuation_prices=closes,
dataset_snapshot=_snapshot(),
factor_version=_factor(),
strategy_version=validated,
risk_policy=RiskPolicy(
policy_id="risk:paper-default@1.0.0",
max_gross_exposure=1.0,
max_single_asset_weight=0.6,
max_positions=10,
),
code_revision="c" * 40,
created_at=datetime(2026, 1, 9, 1, tzinfo=UTC),
top_k=2,
execution_price_field="open",
valuation_price_field="close",
execution_config=_execution_config(),
)