feat: quant validation framework — DSR, PSR, Haircut, regimes, walk-forward

Three-module quant framework replacing 'sort by Sharpe' with proper
statistical validation:

quant/significance.py (15 tests):
  - deflated_sharpe_ratio(): adjusts for N trials (Harvey & Liu 2015)
  - probabilistic_sharpe_ratio(): P(True SR > benchmark) given T, skew, kurt
  - sharpe_haircut(): expected OOS Sharpe after selection bias deflation
  - QuantVerdict: DEPLOY / SIMULATE / DISCARD with 5-point scoring
  - validate_strategy(): one-shot validation function

quant/regimes.py (8 tests):
  - classify_regime(): trending_up/down, ranging, volatile
  - RegimeClassifier: stateful rolling-window classifier
  - conditional_performance(): per-regime trade statistics

quant/walkforward.py (5 tests):
  - WalkForwardRunner: sequential IS/OOS window optimization
  - WFWindow/WFReport: structured walk-forward results
  - consistency score, performance decay, concatenated OOS equity
  - significance_report() integration

Walk-forward results (real HL data with date-sliced windows):
  grid_mm 1h:    2/4 pos, OOS S=-0.45,  74t, haircut=-22.66 → DISCARD
  momentum 4h:   2/4 pos, OOS S=-1.47, 116t, haircut=-45.35 → DISCARD
  composite_mm 1h: 2/4 pos, OOS S=+2.97, 6t, haircut=+43.25 → SIMULATE

28 tests total
This commit is contained in:
ramseshk
2026-08-10 16:20:56 +08:00
parent 268fe606fa
commit 543537e33f
8 changed files with 914 additions and 1 deletions
+87
View File
@@ -0,0 +1,87 @@
"""
Tests for quant/regimes.py and quant/walkforward.py.
"""
import numpy as np
from quant.regimes import RegimeClassifier, classify_regime, conditional_performance
class TestRegimeClassifier:
def test_initial_state(self):
rc = RegimeClassifier()
assert rc.current_regime == "unknown"
def test_classify_trending(self):
"""Rising prices with low volatility → trending_up."""
regime = classify_regime(
returns_20=0.15, # +15% over 20 bars
vol_20=0.02, # 2% vol
vol_ratio=1.0, # normal volume
)
assert regime == "trending_up"
def test_classify_ranging(self):
"""Flat prices with low volatility → ranging."""
regime = classify_regime(
returns_20=0.005, # near flat
vol_20=0.01,
vol_ratio=1.0,
)
assert regime == "ranging"
def test_classify_volatile(self):
"""High volatility regardless of direction → volatile."""
regime = classify_regime(
returns_20=0.02,
vol_20=0.08, # high vol
vol_ratio=2.5, # volume spike
)
assert regime == "volatile"
def test_classify_trending_down(self):
"""Falling prices → trending_down."""
regime = classify_regime(
returns_20=-0.12, # -12%
vol_20=0.03,
vol_ratio=1.0,
)
assert regime == "trending_down"
def test_feed_updates_regime(self):
rc = RegimeClassifier(window=20)
# Feed 21 downtrend bars (need window+1 for first classification)
for i in range(21):
px = 100 - i * 2
rc.feed(px, close=px, open_px=px - 0.5, vol=10)
assert rc.current_regime == "trending_down"
# Feed 21 uptrend bars
for i in range(21):
px = 58 + i * 3
rc.feed(px, close=px, open_px=px + 0.5, vol=10)
assert rc.current_regime == "trending_up"
def test_regime_counts(self):
rc = RegimeClassifier(window=20)
for i in range(100):
if i < 30:
px = 100 + i
elif i < 60:
px = 130 - (i - 30) * 0.5
else:
px = 115 + (i - 60) * 2
rc.feed(px, close=px, open_px=px, vol=10)
counts = rc.regime_counts()
assert sum(counts.values()) > 0
def test_conditional_performance(self):
"""Verify per-regime statistics computation."""
trades = [
{"time": "2026-01-01T00:00:00", "pnl_net": 100, "pnl_gross": 120},
{"time": "2026-01-02T00:00:00", "pnl_net": -50, "pnl_gross": -40},
{"time": "2026-01-03T00:00:00", "pnl_net": 200, "pnl_gross": 220},
]
# 3 trades all in trending_up regime
result = conditional_performance(trades, {"trending_up": 3, "volatile": 0})
assert result["regimes"]["trending_up"]["count"] == 3
assert result["regimes"]["trending_up"]["avg_pnl"] > 0
assert result["best_regime"] == "trending_up"
+129
View File
@@ -0,0 +1,129 @@
"""
Tests for quant/significance.py — deflated Sharpe, probabilistic Sharpe, haircut.
"""
import numpy as np
from quant.significance import (
deflated_sharpe_ratio,
probabilistic_sharpe_ratio,
sharpe_haircut,
QuantVerdict,
validate_strategy,
)
class TestDeflatedSharpeRatio:
def test_no_trials_same_as_observed(self):
"""With 1 trial, DSR should equal the standard N(0,1) probability."""
dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1, n_periods=100)
assert 0.95 < dsr < 1.0
def test_many_trials_deflates(self):
"""With 1000 trials, a Sharpe of 2.0 should be heavily deflated."""
dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1000, n_periods=100)
assert dsr < 0.5
def test_negative_sharpe_zero_dsr(self):
"""Negative Sharpe should produce near-zero DSR."""
dsr = deflated_sharpe_ratio(sharpe=-1.0, n_trials=100, n_periods=100)
assert dsr < 0.05
def test_extreme_sharpe_approaches_one(self):
"""Extremely high Sharpe should survive deflation."""
dsr = deflated_sharpe_ratio(sharpe=10.0, n_trials=100, n_periods=100)
assert dsr > 0.9
class TestProbabilisticSharpeRatio:
def test_large_sample_high_sharpe(self):
"""With many periods, high Sharpe has high PSR."""
psr = probabilistic_sharpe_ratio(sharpe=1.5, n_periods=252, benchmark=0)
assert psr > 0.90
def test_few_periods_uncertain(self):
"""With few periods, Sharpe must be modest to show uncertainty."""
psr = probabilistic_sharpe_ratio(sharpe=0.3, n_periods=5, benchmark=0)
assert psr < 0.80 # Low Sharpe with few periods = uncertain
def test_zero_sharpe_fifty_percent(self):
"""Zero Sharpe should give ~50% PSR (even odds)."""
psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0)
assert 0.40 < psr < 0.60
def test_zero_sharpe_fifty_percent(self):
"""Zero Sharpe should give ~50% PSR (even odds)."""
psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0)
assert 0.40 < psr < 0.60
def test_benchmark_above_observed_low_psr(self):
"""If benchmark > Sharpe, PSR should be low."""
psr = probabilistic_sharpe_ratio(sharpe=0.5, n_periods=100, benchmark=1.0)
assert psr < 0.30
class TestSharpeHaircut:
def test_single_trial_no_haircut(self):
"""With 1 trial, haircut is minimal."""
hc = sharpe_haircut(sharpe=1.0, n_trials=1, n_periods=100)
assert hc > 0.5
def test_many_trials_heavy_haircut(self):
"""With many trials, haircut should be significant below observed."""
hc = sharpe_haircut(sharpe=0.5, n_trials=639, n_periods=100)
assert hc < 0.5 # Deflated below starting value
def test_haircut_range(self):
"""Haircut should be between -1 and 6 typically."""
for s in [0.5, 1.0, 2.0, 5.0]:
hc = sharpe_haircut(sharpe=s, n_trials=100, n_periods=100)
assert -1.0 < hc < 6.0
class TestQuantVerdict:
def test_strong_strategy_deploy(self):
verdict = QuantVerdict(
observed_sharpe=2.0,
wf_consistency=0.80,
n_trials=10,
n_periods=252,
positive_regimes=3,
)
result = verdict.evaluate()
assert result["verdict"] in ("DEPLOY", "SIMULATE")
def test_weak_strategy_discard(self):
verdict = QuantVerdict(
observed_sharpe=0.3,
wf_consistency=0.20,
n_trials=639,
n_periods=10,
positive_regimes=0,
)
result = verdict.evaluate()
assert result["verdict"] == "DISCARD"
def test_grid_mm_simulated(self):
"""Simulate grid_mm 1h: S=8.56, 4 trades, 639 trials — should SIMULATE or DISCARD."""
verdict = QuantVerdict(
observed_sharpe=8.56,
wf_consistency=0.25,
n_trials=639,
n_periods=4,
positive_regimes=1,
)
result = verdict.evaluate()
assert result["verdict"] in ("DISCARD", "SIMULATE")
assert result["psr"] > 0 # PSR should be computed
assert result["haircut_sharpe"] < result["observed_sharpe"] # Haircut reduces Sharpe
def test_summary_includes_all_fields(self):
v = QuantVerdict(
observed_sharpe=1.5,
wf_consistency=0.70,
n_trials=50,
n_periods=100,
positive_regimes=2,
)
r = v.evaluate()
for key in ("verdict", "deflated_sharpe", "psr", "haircut_sharpe",
"wf_consistency", "skewness", "recommendation"):
assert key in r
+51
View File
@@ -0,0 +1,51 @@
"""
Tests for quant/walkforward.py.
"""
from quant.walkforward import WalkForwardRunner, WFReport, WFWindow
class TestWFReport:
def test_empty_report(self):
report = WFReport(strategy="test", interval="1h", n_windows=5)
assert report.consistency == 0.0
assert report.oos_sharpe == 0.0
assert report.performance_decay == 0.0
assert report.total_oos_trades == 0
def test_single_window_positive(self):
report = WFReport(strategy="test", interval="1h", n_windows=5)
report.windows.append(WFWindow(
window_idx=0, is_start="2026-01-01", is_end="2026-02-01",
oos_start="2026-02-01", oos_end="2026-03-01",
is_sharpe=2.0, oos_sharpe=1.5, is_return_pct=5.0,
oos_return_pct=3.0, oos_trades=10,
))
assert report.consistency == 1.0
assert report.avg_oos_sharpe == 1.5
assert report.performance_decay == 1.5 / 2.0
assert report.total_oos_trades == 10
def test_mixed_windows(self):
report = WFReport(strategy="test", interval="1h", n_windows=3)
report.windows = [
WFWindow(0, "A", "B", "B", "C", 2.0, 1.0, 5.0, 2.0, 5),
WFWindow(1, "B", "C", "C", "D", 1.0, -0.5, 2.0, -1.0, 8),
WFWindow(2, "C", "D", "D", "E", 1.5, 0.3, 3.0, 0.5, 6),
]
assert report.consistency == 2 / 3 # 2 of 3 windows positive OOS
assert report.avg_oos_sharpe == (1.0 - 0.5 + 0.3) / 3
def test_significance_discard_weak(self):
report = WFReport(strategy="test", interval="1h", n_windows=5)
report.windows.append(WFWindow(
0, "A", "B", "B", "C", 8.0, -2.0, 2.5, -5.0, 4,
))
report.oos_equity_curve = [{"t": 0, "v": 10000}, {"t": 1, "v": 9500}]
sig = report.significance_report(n_trials=639)
assert sig["verdict"] == "DISCARD"
def test_summary(self):
report = WFReport(strategy="momentum", interval="4h", n_windows=3)
s = report.summary()
assert s["strategy"] == "momentum"
assert s["interval"] == "4h"