""" Tests for quant/significance.py — deflated Sharpe, probabilistic Sharpe, haircut. """ import numpy as np from quant.significance import ( deflated_sharpe_ratio, probabilistic_sharpe_ratio, sharpe_haircut, QuantVerdict, validate_strategy, ) class TestDeflatedSharpeRatio: def test_no_trials_same_as_observed(self): """With 1 trial, DSR should equal the standard N(0,1) probability.""" dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1, n_periods=100) assert 0.95 < dsr < 1.0 def test_many_trials_deflates(self): """With 1000 trials, a Sharpe of 2.0 should be heavily deflated.""" dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1000, n_periods=100) assert dsr < 0.5 def test_negative_sharpe_zero_dsr(self): """Negative Sharpe should produce near-zero DSR.""" dsr = deflated_sharpe_ratio(sharpe=-1.0, n_trials=100, n_periods=100) assert dsr < 0.05 def test_extreme_sharpe_approaches_one(self): """Extremely high Sharpe should survive deflation.""" dsr = deflated_sharpe_ratio(sharpe=10.0, n_trials=100, n_periods=100) assert dsr > 0.9 class TestProbabilisticSharpeRatio: def test_large_sample_high_sharpe(self): """With many periods, high Sharpe has high PSR.""" psr = probabilistic_sharpe_ratio(sharpe=1.5, n_periods=252, benchmark=0) assert psr > 0.90 def test_few_periods_uncertain(self): """With few periods, Sharpe must be modest to show uncertainty.""" psr = probabilistic_sharpe_ratio(sharpe=0.3, n_periods=5, benchmark=0) assert psr < 0.80 # Low Sharpe with few periods = uncertain def test_zero_sharpe_fifty_percent(self): """Zero Sharpe should give ~50% PSR (even odds).""" psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0) assert 0.40 < psr < 0.60 def test_zero_sharpe_fifty_percent(self): """Zero Sharpe should give ~50% PSR (even odds).""" psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0) assert 0.40 < psr < 0.60 def test_benchmark_above_observed_low_psr(self): """If benchmark > Sharpe, PSR should be low.""" psr = probabilistic_sharpe_ratio(sharpe=0.5, n_periods=100, benchmark=1.0) assert psr < 0.30 class TestSharpeHaircut: def test_single_trial_no_haircut(self): """With 1 trial, haircut is minimal.""" hc = sharpe_haircut(sharpe=1.0, n_trials=1, n_periods=100) assert hc > 0.5 def test_many_trials_heavy_haircut(self): """With many trials, haircut should be significant below observed.""" hc = sharpe_haircut(sharpe=0.5, n_trials=639, n_periods=100) assert hc < 0.5 # Deflated below starting value def test_haircut_range(self): """Haircut should be between -1 and 6 typically.""" for s in [0.5, 1.0, 2.0, 5.0]: hc = sharpe_haircut(sharpe=s, n_trials=100, n_periods=100) assert -1.0 < hc < 6.0 class TestQuantVerdict: def test_strong_strategy_deploy(self): verdict = QuantVerdict( observed_sharpe=2.0, wf_consistency=0.80, n_trials=10, n_periods=252, positive_regimes=3, ) result = verdict.evaluate() assert result["verdict"] in ("DEPLOY", "SIMULATE") def test_weak_strategy_discard(self): verdict = QuantVerdict( observed_sharpe=0.3, wf_consistency=0.20, n_trials=639, n_periods=10, positive_regimes=0, ) result = verdict.evaluate() assert result["verdict"] == "DISCARD" def test_grid_mm_simulated(self): """Simulate grid_mm 1h: S=8.56, 4 trades, 639 trials — should SIMULATE or DISCARD.""" verdict = QuantVerdict( observed_sharpe=8.56, wf_consistency=0.25, n_trials=639, n_periods=4, positive_regimes=1, ) result = verdict.evaluate() assert result["verdict"] in ("DISCARD", "SIMULATE") assert result["psr"] > 0 # PSR should be computed assert result["haircut_sharpe"] < result["observed_sharpe"] # Haircut reduces Sharpe def test_summary_includes_all_fields(self): v = QuantVerdict( observed_sharpe=1.5, wf_consistency=0.70, n_trials=50, n_periods=100, positive_regimes=2, ) r = v.evaluate() for key in ("verdict", "deflated_sharpe", "psr", "haircut_sharpe", "wf_consistency", "skewness", "recommendation"): assert key in r