"""Contracts for reusable factor diagnostics and transformations.""" from __future__ import annotations import numpy as np import pandas as pd import pytest from quant_engine.factor_library import ( annualized_sharpe, apply_factor_direction, cross_sectional_momentum, cross_sectional_pct_rank, cross_sectional_rank_with_direction, ic_summary, jb_test, kurtosis, ols_regress, rolling_annual_vol, rolling_zscore, skewness, spearman_ic, time_series_momentum, turnover, winsorize, ) def test_turnover_supports_one_way_and_round_trip_conventions() -> None: weights = pd.DataFrame({"A": [1.0, 0.0], "B": [0.0, 1.0]}) pd.testing.assert_series_equal(turnover(weights), pd.Series([1.0], index=[1])) pd.testing.assert_series_equal( turnover(weights, divide_by_two=False), pd.Series([2.0], index=[1]) ) assert turnover(weights.iloc[:1]).empty def test_ic_functions_measure_monotonic_relationship() -> None: factor = pd.Series([1.0, 2.0, 3.0, 4.0]) forward = pd.Series([10.0, 20.0, 30.0, 40.0]) assert spearman_ic(factor, forward) == pytest.approx(1.0) result = ic_summary(factor, forward, periods=(1,), method="pearson") assert result.loc[1, "ic_mean"] == pytest.approx(1.0) assert result.loc[1, "n"] == 4 def test_ic_summary_rejects_unknown_method() -> None: with pytest.raises(ValueError, match="not supported"): ic_summary(pd.Series([1, 2, 3]), pd.Series([1, 2, 3]), method="kendall") def test_winsorize_clips_tails_and_preserves_nan() -> None: values = pd.Series([0.0, 1.0, 2.0, 100.0, np.nan]) result = winsorize(values, lower=0.25, upper=0.75) assert result.iloc[0] == pytest.approx(0.75) assert result.iloc[3] == pytest.approx(26.5) assert pd.isna(result.iloc[4]) def test_distribution_diagnostics_handle_short_samples() -> None: assert np.isnan(skewness(pd.Series([1.0, 2.0]))) assert np.isnan(kurtosis(pd.Series([1.0, 2.0, 3.0]))) jb, p_value = jb_test(pd.Series(range(7), dtype=float)) assert np.isnan(jb) assert np.isnan(p_value) def test_distribution_diagnostics_return_finite_values() -> None: values = pd.Series([-2.0, -1.0, -0.5, 0.0, 0.25, 0.75, 1.0, 3.0]) assert np.isfinite(skewness(values)) assert np.isfinite(kurtosis(values)) jb, p_value = jb_test(values) assert jb >= 0 assert 0 <= p_value <= 1 def test_ols_recovers_linear_coefficients_and_residual_index() -> None: index = pd.date_range("2026-01-01", periods=8) factor = pd.Series(np.arange(8, dtype=float), index=index, name="factor") target = 1.5 + 2.0 * factor result = ols_regress(target, factor) assert result.alpha == pytest.approx(1.5) assert result.beta["factor"] == pytest.approx(2.0) assert result.r_squared == pytest.approx(1.0) assert result.n == 8 assert result.resid.index.equals(index) def test_ols_handles_collinear_factors_without_crashing() -> None: x = pd.DataFrame({"a": np.arange(8, dtype=float), "b": np.arange(8, dtype=float)}) y = pd.Series(1.0 + x["a"]) result = ols_regress(y, x) assert result.n == 8 assert np.isfinite(result.beta).all() np.testing.assert_allclose(result.resid, 0.0, atol=1e-12) def test_ols_short_sample_returns_empty_estimate() -> None: result = ols_regress(pd.Series([1.0, 2.0]), pd.Series([1.0, 2.0], name="x")) assert np.isnan(result.alpha) assert result.beta.empty assert result.n == 2 def test_momentum_and_rolling_transforms_match_manual_values() -> None: prices = pd.DataFrame({"A": [100.0, 110.0, 121.0, 133.1]}) momentum = cross_sectional_momentum(prices, lookback=2, skip=0) assert momentum.iloc[2, 0] == pytest.approx(0.21) returns = pd.Series([0.1, 0.1, -0.5, -0.5]) pd.testing.assert_series_equal( time_series_momentum(returns, lookback=2), pd.Series([0, 1, -1, -1]), ) values = pd.Series([1.0, 2.0, 3.0]) zscore = rolling_zscore(values, window=3) assert zscore.iloc[-1] == pytest.approx(1.0) annual_vol = rolling_annual_vol(returns, window=2, min_periods=2, trading_days=4) assert annual_vol.iloc[1] == pytest.approx(0.0) def test_rank_helpers_support_global_and_grouped_ranking() -> None: frame = pd.DataFrame( {"factor": [3.0, 1.0, 2.0, 4.0], "industry": ["x", "x", "y", "y"]} ) global_rank = cross_sectional_pct_rank(frame, "factor", ascending=True) grouped_rank = cross_sectional_pct_rank( frame, "factor", group_col="industry", ascending=True ) assert global_rank.tolist() == [0.75, 0.25, 0.5, 1.0] assert grouped_rank.tolist() == [1.0, 0.5, 0.5, 1.0] assert cross_sectional_pct_rank(frame, "missing").empty def test_factor_direction_and_directional_rank() -> None: pe = pd.Series([10.0, 20.0], name="pe_ttm") pd.testing.assert_series_equal(apply_factor_direction(pe), -pe) frame = pd.DataFrame({"pe_ttm": [10.0, 20.0], "roe": [0.1, 0.2]}) assert cross_sectional_rank_with_direction(frame, "pe_ttm").tolist() == [1.0, 0.5] assert cross_sectional_rank_with_direction(frame, "roe").tolist() == [0.5, 1.0] @pytest.mark.parametrize("direction", ["sideways", "", "REVERSE"]) def test_factor_direction_rejects_unknown_values(direction: str) -> None: factor = pd.Series([1.0, 2.0], name="roe") with pytest.raises(ValueError, match="direction"): apply_factor_direction(factor, direction=direction) with pytest.raises(ValueError, match="direction"): cross_sectional_rank_with_direction( pd.DataFrame({"roe": factor}), "roe", direction=direction ) def test_annualized_sharpe_handles_empty_and_nonzero_returns() -> None: assert annualized_sharpe(pd.Series(dtype=float)) == 0.0 returns = pd.Series([0.01, -0.01, 0.02, 0.0]) expected = returns.mean() * 252 / (returns.std() * np.sqrt(252)) assert annualized_sharpe(returns) == pytest.approx(expected)