Files
ftdt-quant-lab/tests/test_quant_significance.py
T
ramseshk 543537e33f feat: quant validation framework — DSR, PSR, Haircut, regimes, walk-forward
Three-module quant framework replacing 'sort by Sharpe' with proper
statistical validation:

quant/significance.py (15 tests):
  - deflated_sharpe_ratio(): adjusts for N trials (Harvey & Liu 2015)
  - probabilistic_sharpe_ratio(): P(True SR > benchmark) given T, skew, kurt
  - sharpe_haircut(): expected OOS Sharpe after selection bias deflation
  - QuantVerdict: DEPLOY / SIMULATE / DISCARD with 5-point scoring
  - validate_strategy(): one-shot validation function

quant/regimes.py (8 tests):
  - classify_regime(): trending_up/down, ranging, volatile
  - RegimeClassifier: stateful rolling-window classifier
  - conditional_performance(): per-regime trade statistics

quant/walkforward.py (5 tests):
  - WalkForwardRunner: sequential IS/OOS window optimization
  - WFWindow/WFReport: structured walk-forward results
  - consistency score, performance decay, concatenated OOS equity
  - significance_report() integration

Walk-forward results (real HL data with date-sliced windows):
  grid_mm 1h:    2/4 pos, OOS S=-0.45,  74t, haircut=-22.66 → DISCARD
  momentum 4h:   2/4 pos, OOS S=-1.47, 116t, haircut=-45.35 → DISCARD
  composite_mm 1h: 2/4 pos, OOS S=+2.97, 6t, haircut=+43.25 → SIMULATE

28 tests total
2026-08-10 16:20:56 +08:00

130 lines
4.6 KiB
Python

"""
Tests for quant/significance.py — deflated Sharpe, probabilistic Sharpe, haircut.
"""
import numpy as np
from quant.significance import (
deflated_sharpe_ratio,
probabilistic_sharpe_ratio,
sharpe_haircut,
QuantVerdict,
validate_strategy,
)
class TestDeflatedSharpeRatio:
def test_no_trials_same_as_observed(self):
"""With 1 trial, DSR should equal the standard N(0,1) probability."""
dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1, n_periods=100)
assert 0.95 < dsr < 1.0
def test_many_trials_deflates(self):
"""With 1000 trials, a Sharpe of 2.0 should be heavily deflated."""
dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1000, n_periods=100)
assert dsr < 0.5
def test_negative_sharpe_zero_dsr(self):
"""Negative Sharpe should produce near-zero DSR."""
dsr = deflated_sharpe_ratio(sharpe=-1.0, n_trials=100, n_periods=100)
assert dsr < 0.05
def test_extreme_sharpe_approaches_one(self):
"""Extremely high Sharpe should survive deflation."""
dsr = deflated_sharpe_ratio(sharpe=10.0, n_trials=100, n_periods=100)
assert dsr > 0.9
class TestProbabilisticSharpeRatio:
def test_large_sample_high_sharpe(self):
"""With many periods, high Sharpe has high PSR."""
psr = probabilistic_sharpe_ratio(sharpe=1.5, n_periods=252, benchmark=0)
assert psr > 0.90
def test_few_periods_uncertain(self):
"""With few periods, Sharpe must be modest to show uncertainty."""
psr = probabilistic_sharpe_ratio(sharpe=0.3, n_periods=5, benchmark=0)
assert psr < 0.80 # Low Sharpe with few periods = uncertain
def test_zero_sharpe_fifty_percent(self):
"""Zero Sharpe should give ~50% PSR (even odds)."""
psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0)
assert 0.40 < psr < 0.60
def test_zero_sharpe_fifty_percent(self):
"""Zero Sharpe should give ~50% PSR (even odds)."""
psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0)
assert 0.40 < psr < 0.60
def test_benchmark_above_observed_low_psr(self):
"""If benchmark > Sharpe, PSR should be low."""
psr = probabilistic_sharpe_ratio(sharpe=0.5, n_periods=100, benchmark=1.0)
assert psr < 0.30
class TestSharpeHaircut:
def test_single_trial_no_haircut(self):
"""With 1 trial, haircut is minimal."""
hc = sharpe_haircut(sharpe=1.0, n_trials=1, n_periods=100)
assert hc > 0.5
def test_many_trials_heavy_haircut(self):
"""With many trials, haircut should be significant below observed."""
hc = sharpe_haircut(sharpe=0.5, n_trials=639, n_periods=100)
assert hc < 0.5 # Deflated below starting value
def test_haircut_range(self):
"""Haircut should be between -1 and 6 typically."""
for s in [0.5, 1.0, 2.0, 5.0]:
hc = sharpe_haircut(sharpe=s, n_trials=100, n_periods=100)
assert -1.0 < hc < 6.0
class TestQuantVerdict:
def test_strong_strategy_deploy(self):
verdict = QuantVerdict(
observed_sharpe=2.0,
wf_consistency=0.80,
n_trials=10,
n_periods=252,
positive_regimes=3,
)
result = verdict.evaluate()
assert result["verdict"] in ("DEPLOY", "SIMULATE")
def test_weak_strategy_discard(self):
verdict = QuantVerdict(
observed_sharpe=0.3,
wf_consistency=0.20,
n_trials=639,
n_periods=10,
positive_regimes=0,
)
result = verdict.evaluate()
assert result["verdict"] == "DISCARD"
def test_grid_mm_simulated(self):
"""Simulate grid_mm 1h: S=8.56, 4 trades, 639 trials — should SIMULATE or DISCARD."""
verdict = QuantVerdict(
observed_sharpe=8.56,
wf_consistency=0.25,
n_trials=639,
n_periods=4,
positive_regimes=1,
)
result = verdict.evaluate()
assert result["verdict"] in ("DISCARD", "SIMULATE")
assert result["psr"] > 0 # PSR should be computed
assert result["haircut_sharpe"] < result["observed_sharpe"] # Haircut reduces Sharpe
def test_summary_includes_all_fields(self):
v = QuantVerdict(
observed_sharpe=1.5,
wf_consistency=0.70,
n_trials=50,
n_periods=100,
positive_regimes=2,
)
r = v.evaluate()
for key in ("verdict", "deflated_sharpe", "psr", "haircut_sharpe",
"wf_consistency", "skewness", "recommendation"):
assert key in r