File size: 7,539 Bytes
3339913
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
"""Statistical machinery.

These tests matter more than the engine's, because a validation suite that
always says "no edge" is as useless as one that always says "great edge". Each
class below checks both directions: it must reject noise *and* detect signal.
"""

from __future__ import annotations

import numpy as np
import pandas as pd
import pytest

from algotrader.data import simulate_ohlcv
from algotrader.strategies import get_strategy
from algotrader.validation.deflated_sharpe import (
    deflated_sharpe_ratio,
    expected_max_sharpe,
    min_track_record_length,
    probabilistic_sharpe_ratio,
)
from algotrader.validation.pbo import probability_of_backtest_overfitting
from algotrader.validation.permutation import permutation_test, permute_bars
from algotrader.validation.walkforward import walk_forward


def trending_market(n: int = 2200, phi: float = 0.35, seed: int = 5) -> pd.DataFrame:
    """A market with genuine, exploitable serial correlation."""
    rng = np.random.default_rng(seed)
    returns = np.zeros(n)
    noise = rng.normal(0, 0.01, n)
    for i in range(1, n):
        returns[i] = phi * returns[i - 1] + noise[i]
    close = 100 * np.exp(np.cumsum(returns))
    index = pd.date_range("2012-01-01", periods=n, freq="B")
    return pd.DataFrame(
        {"open": close, "high": close * 1.004, "low": close * 0.996, "close": close, "volume": 1e6},
        index=index,
    )


class TestPermutationMechanics:
    def test_shuffling_preserves_the_distribution_of_moves(self):
        df = simulate_ohlcv("SPY", "2018-01-01", "2023-01-01")
        shuffled = permute_bars(df, np.random.default_rng(0))

        assert len(shuffled) == len(df)
        assert shuffled.index.equals(df.index)
        original = np.sort(np.log(df["close"] / df["open"]).to_numpy()[1:])
        permuted = np.sort(np.log(shuffled["close"] / shuffled["open"]).to_numpy()[1:])
        np.testing.assert_allclose(original, permuted, rtol=1e-9)

    def test_shuffling_keeps_bars_internally_valid(self):
        df = simulate_ohlcv("AAPL", "2019-01-01", "2023-01-01")
        shuffled = permute_bars(df, np.random.default_rng(3))
        assert (shuffled["high"] >= shuffled["low"]).all()
        assert (shuffled["high"] >= shuffled["close"]).all()
        assert (shuffled["low"] <= shuffled["close"]).all()
        assert (shuffled["close"] > 0).all()

    def test_shuffling_destroys_serial_correlation(self):
        df = trending_market()
        real = df["close"].pct_change().dropna().autocorr(1)
        shuffled = permute_bars(df, np.random.default_rng(1))["close"].pct_change().dropna().autocorr(1)
        assert real > 0.2
        assert abs(shuffled) < 0.1

    def test_block_mode_retains_some_structure(self):
        df = trending_market()
        blocked = permute_bars(df, np.random.default_rng(2), method="block", block=40)
        assert blocked["close"].pct_change().dropna().autocorr(1) > 0.1

    def test_p_value_can_never_be_zero(self):
        """+1 correction: the observed run is itself a draw from the null."""
        df = trending_market()
        strategy = get_strategy("momentum")
        result = permutation_test(
            df, lambda f: strategy.generate(f, {"lookback": 5}), n_permutations=30, seed=0
        )
        assert result.p_value >= 1 / 31
        assert 0 < result.p_value <= 1


class TestPermutationPower:
    def test_real_edge_is_detected(self):
        df = trending_market()
        strategy = get_strategy("momentum")
        result = permutation_test(
            df, lambda f: strategy.generate(f, {"lookback": 5}), n_permutations=200, seed=1
        )
        assert result.observed > result.null.mean()
        assert result.p_value < 0.05

    def test_random_strategy_on_a_structureless_market_is_not_significant(self):
        df = simulate_ohlcv("SIM", "2010-01-01", "2023-01-01")
        strategy = get_strategy("coin_flip")
        result = permutation_test(
            df, lambda f: strategy.generate(f, {"hold": 5, "seed": 7}), n_permutations=200, seed=2
        )
        assert result.p_value > 0.05


class TestDeflatedSharpe:
    def test_selection_bar_rises_with_the_number_of_trials(self):
        low = expected_max_sharpe(10, 0.01)
        high = expected_max_sharpe(1000, 0.01)
        assert 0 < low < high

    def test_a_single_trial_has_no_selection_bar(self):
        assert expected_max_sharpe(1, 0.01) == 0.0

    def test_more_trials_lowers_the_deflated_sharpe(self):
        rng = np.random.default_rng(7)
        returns = rng.normal(0.0006, 0.01, 2000)
        few = deflated_sharpe_ratio(returns, 1.0, 252, n_trials=2, variance_of_trials=0.01)
        many = deflated_sharpe_ratio(returns, 1.0, 252, n_trials=500, variance_of_trials=0.01)
        assert many["dsr"] < few["dsr"]
        assert many["psr"] == pytest.approx(few["psr"])  # PSR ignores selection

    def test_psr_rises_with_track_record_length(self):
        short = probabilistic_sharpe_ratio(0.05, 100)
        long = probabilistic_sharpe_ratio(0.05, 5000)
        assert 0.5 < short < long < 1.0

    def test_negative_skew_and_fat_tails_are_penalised(self):
        clean = probabilistic_sharpe_ratio(0.06, 1000, skew=0.0, kurtosis=3.0)
        nasty = probabilistic_sharpe_ratio(0.06, 1000, skew=-1.5, kurtosis=12.0)
        assert nasty < clean

    def test_track_record_requirement_is_infinite_below_the_bar(self):
        assert min_track_record_length(0.01, 500, benchmark=0.05) == float("inf")
        assert np.isfinite(min_track_record_length(0.10, 500, benchmark=0.02))


class TestPBO:
    def test_pure_noise_scores_near_one_half(self):
        rng = np.random.default_rng(11)
        matrix = rng.normal(0, 0.01, size=(1200, 30))  # 30 skill-free variants
        result = probability_of_backtest_overfitting(matrix, n_splits=8)
        assert 0.3 < result["pbo"] < 0.7

    def test_a_genuinely_better_variant_is_not_flagged(self):
        rng = np.random.default_rng(12)
        matrix = rng.normal(0, 0.01, size=(1200, 20))
        matrix[:, 3] += 0.004  # column 3 has a persistent, real edge
        result = probability_of_backtest_overfitting(matrix, n_splits=8)
        assert result["pbo"] < 0.15
        assert result["most_selected_index"] == 3
        assert result["selection_stability"] > 0.9

    def test_too_few_variants_returns_nan_not_a_crash(self):
        rng = np.random.default_rng(13)
        result = probability_of_backtest_overfitting(rng.normal(0, 0.01, size=(500, 1)))
        assert np.isnan(result["pbo"])
        assert result["note"]

    def test_odd_split_counts_are_made_even(self):
        rng = np.random.default_rng(14)
        result = probability_of_backtest_overfitting(rng.normal(0, 0.01, (800, 10)), n_splits=7)
        assert result["n_combinations"] > 0


class TestWalkForward:
    def test_a_real_edge_survives_out_of_sample(self):
        result = walk_forward(trending_market(), get_strategy("momentum"), n_folds=4)
        assert result["folds"]
        assert result["mean_oos_sharpe"] > 0
        assert result["efficiency"] > 0.3

    def test_folds_do_not_overlap_train_and_test(self):
        result = walk_forward(trending_market(), get_strategy("sma_cross"), n_folds=4)
        for fold in result["folds"]:
            assert fold["train_end"] <= fold["test_start"]

    def test_short_history_degrades_gracefully(self):
        df = trending_market(n=150)
        result = walk_forward(df, get_strategy("momentum"), n_folds=5)
        assert result["folds"] == []
        assert result["note"]