ParallelLLC/algorithmic_trading
2732
1"""Statistical machinery.2 3These tests matter more than the engine's, because a validation suite that4always says "no edge" is as useless as one that always says "great edge". Each5class below checks both directions: it must reject noise *and* detect signal.6"""7 8from __future__ import annotations9 10import numpy as np11import pandas as pd12import pytest13 14from algotrader.data import simulate_ohlcv15from algotrader.strategies import get_strategy16from algotrader.validation.deflated_sharpe import (17 deflated_sharpe_ratio,18 expected_max_sharpe,19 min_track_record_length,20 probabilistic_sharpe_ratio,21)22from algotrader.validation.pbo import probability_of_backtest_overfitting23from algotrader.validation.permutation import permutation_test, permute_bars24from algotrader.validation.walkforward import walk_forward25 26 27def trending_market(n: int = 2200, phi: float = 0.35, seed: int = 5) -> pd.DataFrame:28 """A market with genuine, exploitable serial correlation."""29 rng = np.random.default_rng(seed)30 returns = np.zeros(n)31 noise = rng.normal(0, 0.01, n)32 for i in range(1, n):33 returns[i] = phi * returns[i - 1] + noise[i]34 close = 100 * np.exp(np.cumsum(returns))35 index = pd.date_range("2012-01-01", periods=n, freq="B")36 return pd.DataFrame(37 {"open": close, "high": close * 1.004, "low": close * 0.996, "close": close, "volume": 1e6},38 index=index,39 )40 41 42class TestPermutationMechanics:43 def test_shuffling_preserves_the_distribution_of_moves(self):44 df = simulate_ohlcv("SPY", "2018-01-01", "2023-01-01")45 shuffled = permute_bars(df, np.random.default_rng(0))46 47 assert len(shuffled) == len(df)48 assert shuffled.index.equals(df.index)49 original = np.sort(np.log(df["close"] / df["open"]).to_numpy()[1:])50 permuted = np.sort(np.log(shuffled["close"] / shuffled["open"]).to_numpy()[1:])51 np.testing.assert_allclose(original, permuted, rtol=1e-9)52 53 def test_shuffling_keeps_bars_internally_valid(self):54 df = simulate_ohlcv("AAPL", "2019-01-01", "2023-01-01")55 shuffled = permute_bars(df, np.random.default_rng(3))56 assert (shuffled["high"] >= shuffled["low"]).all()57 assert (shuffled["high"] >= shuffled["close"]).all()58 assert (shuffled["low"] <= shuffled["close"]).all()59 assert (shuffled["close"] > 0).all()60 61 def test_shuffling_destroys_serial_correlation(self):62 df = trending_market()63 real = df["close"].pct_change().dropna().autocorr(1)64 shuffled = permute_bars(df, np.random.default_rng(1))["close"].pct_change().dropna().autocorr(1)65 assert real > 0.266 assert abs(shuffled) < 0.167 68 def test_block_mode_retains_some_structure(self):69 df = trending_market()70 blocked = permute_bars(df, np.random.default_rng(2), method="block", block=40)71 assert blocked["close"].pct_change().dropna().autocorr(1) > 0.172 73 def test_p_value_can_never_be_zero(self):74 """+1 correction: the observed run is itself a draw from the null."""75 df = trending_market()76 strategy = get_strategy("momentum")77 result = permutation_test(78 df, lambda f: strategy.generate(f, {"lookback": 5}), n_permutations=30, seed=079 )80 assert result.p_value >= 1 / 3181 assert 0 < result.p_value <= 182 83 84class TestPermutationPower:85 def test_real_edge_is_detected(self):86 df = trending_market()87 strategy = get_strategy("momentum")88 result = permutation_test(89 df, lambda f: strategy.generate(f, {"lookback": 5}), n_permutations=200, seed=190 )91 assert result.observed > result.null.mean()92 assert result.p_value < 0.0593 94 def test_random_strategy_on_a_structureless_market_is_not_significant(self):95 df = simulate_ohlcv("SIM", "2010-01-01", "2023-01-01")96 strategy = get_strategy("coin_flip")97 result = permutation_test(98 df, lambda f: strategy.generate(f, {"hold": 5, "seed": 7}), n_permutations=200, seed=299 )100 assert result.p_value > 0.05101 102 103class TestDeflatedSharpe:104 def test_selection_bar_rises_with_the_number_of_trials(self):105 low = expected_max_sharpe(10, 0.01)106 high = expected_max_sharpe(1000, 0.01)107 assert 0 < low < high108 109 def test_a_single_trial_has_no_selection_bar(self):110 assert expected_max_sharpe(1, 0.01) == 0.0111 112 def test_more_trials_lowers_the_deflated_sharpe(self):113 rng = np.random.default_rng(7)114 returns = rng.normal(0.0006, 0.01, 2000)115 few = deflated_sharpe_ratio(returns, 1.0, 252, n_trials=2, variance_of_trials=0.01)116 many = deflated_sharpe_ratio(returns, 1.0, 252, n_trials=500, variance_of_trials=0.01)117 assert many["dsr"] < few["dsr"]118 assert many["psr"] == pytest.approx(few["psr"]) # PSR ignores selection119 120 def test_psr_rises_with_track_record_length(self):121 short = probabilistic_sharpe_ratio(0.05, 100)122 long = probabilistic_sharpe_ratio(0.05, 5000)123 assert 0.5 < short < long < 1.0124 125 def test_negative_skew_and_fat_tails_are_penalised(self):126 clean = probabilistic_sharpe_ratio(0.06, 1000, skew=0.0, kurtosis=3.0)127 nasty = probabilistic_sharpe_ratio(0.06, 1000, skew=-1.5, kurtosis=12.0)128 assert nasty < clean129 130 def test_track_record_requirement_is_infinite_below_the_bar(self):131 assert min_track_record_length(0.01, 500, benchmark=0.05) == float("inf")132 assert np.isfinite(min_track_record_length(0.10, 500, benchmark=0.02))133 134 135class TestPBO:136 def test_pure_noise_scores_near_one_half(self):137 rng = np.random.default_rng(11)138 matrix = rng.normal(0, 0.01, size=(1200, 30)) # 30 skill-free variants139 result = probability_of_backtest_overfitting(matrix, n_splits=8)140 assert 0.3 < result["pbo"] < 0.7141 142 def test_a_genuinely_better_variant_is_not_flagged(self):143 rng = np.random.default_rng(12)144 matrix = rng.normal(0, 0.01, size=(1200, 20))145 matrix[:, 3] += 0.004 # column 3 has a persistent, real edge146 result = probability_of_backtest_overfitting(matrix, n_splits=8)147 assert result["pbo"] < 0.15148 assert result["most_selected_index"] == 3149 assert result["selection_stability"] > 0.9150 151 def test_too_few_variants_returns_nan_not_a_crash(self):152 rng = np.random.default_rng(13)153 result = probability_of_backtest_overfitting(rng.normal(0, 0.01, size=(500, 1)))154 assert np.isnan(result["pbo"])155 assert result["note"]156 157 def test_odd_split_counts_are_made_even(self):158 rng = np.random.default_rng(14)159 result = probability_of_backtest_overfitting(rng.normal(0, 0.01, (800, 10)), n_splits=7)160 assert result["n_combinations"] > 0161 162 163class TestWalkForward:164 def test_a_real_edge_survives_out_of_sample(self):165 result = walk_forward(trending_market(), get_strategy("momentum"), n_folds=4)166 assert result["folds"]167 assert result["mean_oos_sharpe"] > 0168 assert result["efficiency"] > 0.3169 170 def test_folds_do_not_overlap_train_and_test(self):171 result = walk_forward(trending_market(), get_strategy("sma_cross"), n_folds=4)172 for fold in result["folds"]:173 assert fold["train_end"] <= fold["test_start"]174 175 def test_short_history_degrades_gracefully(self):176 df = trending_market(n=150)177 result = walk_forward(df, get_strategy("momentum"), n_folds=5)178 assert result["folds"] == []179 assert result["note"]180 