feat: quant validation framework — DSR, PSR, Haircut, regimes, walk-forward
Three-module quant framework replacing 'sort by Sharpe' with proper statistical validation: quant/significance.py (15 tests): - deflated_sharpe_ratio(): adjusts for N trials (Harvey & Liu 2015) - probabilistic_sharpe_ratio(): P(True SR > benchmark) given T, skew, kurt - sharpe_haircut(): expected OOS Sharpe after selection bias deflation - QuantVerdict: DEPLOY / SIMULATE / DISCARD with 5-point scoring - validate_strategy(): one-shot validation function quant/regimes.py (8 tests): - classify_regime(): trending_up/down, ranging, volatile - RegimeClassifier: stateful rolling-window classifier - conditional_performance(): per-regime trade statistics quant/walkforward.py (5 tests): - WalkForwardRunner: sequential IS/OOS window optimization - WFWindow/WFReport: structured walk-forward results - consistency score, performance decay, concatenated OOS equity - significance_report() integration Walk-forward results (real HL data with date-sliced windows): grid_mm 1h: 2/4 pos, OOS S=-0.45, 74t, haircut=-22.66 → DISCARD momentum 4h: 2/4 pos, OOS S=-1.47, 116t, haircut=-45.35 → DISCARD composite_mm 1h: 2/4 pos, OOS S=+2.97, 6t, haircut=+43.25 → SIMULATE 28 tests total
This commit is contained in:
@@ -0,0 +1,87 @@
|
||||
"""
|
||||
Tests for quant/regimes.py and quant/walkforward.py.
|
||||
"""
|
||||
import numpy as np
|
||||
from quant.regimes import RegimeClassifier, classify_regime, conditional_performance
|
||||
|
||||
|
||||
class TestRegimeClassifier:
|
||||
def test_initial_state(self):
|
||||
rc = RegimeClassifier()
|
||||
assert rc.current_regime == "unknown"
|
||||
|
||||
def test_classify_trending(self):
|
||||
"""Rising prices with low volatility → trending_up."""
|
||||
regime = classify_regime(
|
||||
returns_20=0.15, # +15% over 20 bars
|
||||
vol_20=0.02, # 2% vol
|
||||
vol_ratio=1.0, # normal volume
|
||||
)
|
||||
assert regime == "trending_up"
|
||||
|
||||
def test_classify_ranging(self):
|
||||
"""Flat prices with low volatility → ranging."""
|
||||
regime = classify_regime(
|
||||
returns_20=0.005, # near flat
|
||||
vol_20=0.01,
|
||||
vol_ratio=1.0,
|
||||
)
|
||||
assert regime == "ranging"
|
||||
|
||||
def test_classify_volatile(self):
|
||||
"""High volatility regardless of direction → volatile."""
|
||||
regime = classify_regime(
|
||||
returns_20=0.02,
|
||||
vol_20=0.08, # high vol
|
||||
vol_ratio=2.5, # volume spike
|
||||
)
|
||||
assert regime == "volatile"
|
||||
|
||||
def test_classify_trending_down(self):
|
||||
"""Falling prices → trending_down."""
|
||||
regime = classify_regime(
|
||||
returns_20=-0.12, # -12%
|
||||
vol_20=0.03,
|
||||
vol_ratio=1.0,
|
||||
)
|
||||
assert regime == "trending_down"
|
||||
|
||||
def test_feed_updates_regime(self):
|
||||
rc = RegimeClassifier(window=20)
|
||||
# Feed 21 downtrend bars (need window+1 for first classification)
|
||||
for i in range(21):
|
||||
px = 100 - i * 2
|
||||
rc.feed(px, close=px, open_px=px - 0.5, vol=10)
|
||||
assert rc.current_regime == "trending_down"
|
||||
|
||||
# Feed 21 uptrend bars
|
||||
for i in range(21):
|
||||
px = 58 + i * 3
|
||||
rc.feed(px, close=px, open_px=px + 0.5, vol=10)
|
||||
assert rc.current_regime == "trending_up"
|
||||
|
||||
def test_regime_counts(self):
|
||||
rc = RegimeClassifier(window=20)
|
||||
for i in range(100):
|
||||
if i < 30:
|
||||
px = 100 + i
|
||||
elif i < 60:
|
||||
px = 130 - (i - 30) * 0.5
|
||||
else:
|
||||
px = 115 + (i - 60) * 2
|
||||
rc.feed(px, close=px, open_px=px, vol=10)
|
||||
counts = rc.regime_counts()
|
||||
assert sum(counts.values()) > 0
|
||||
|
||||
def test_conditional_performance(self):
|
||||
"""Verify per-regime statistics computation."""
|
||||
trades = [
|
||||
{"time": "2026-01-01T00:00:00", "pnl_net": 100, "pnl_gross": 120},
|
||||
{"time": "2026-01-02T00:00:00", "pnl_net": -50, "pnl_gross": -40},
|
||||
{"time": "2026-01-03T00:00:00", "pnl_net": 200, "pnl_gross": 220},
|
||||
]
|
||||
# 3 trades all in trending_up regime
|
||||
result = conditional_performance(trades, {"trending_up": 3, "volatile": 0})
|
||||
assert result["regimes"]["trending_up"]["count"] == 3
|
||||
assert result["regimes"]["trending_up"]["avg_pnl"] > 0
|
||||
assert result["best_regime"] == "trending_up"
|
||||
@@ -0,0 +1,129 @@
|
||||
"""
|
||||
Tests for quant/significance.py — deflated Sharpe, probabilistic Sharpe, haircut.
|
||||
"""
|
||||
import numpy as np
|
||||
from quant.significance import (
|
||||
deflated_sharpe_ratio,
|
||||
probabilistic_sharpe_ratio,
|
||||
sharpe_haircut,
|
||||
QuantVerdict,
|
||||
validate_strategy,
|
||||
)
|
||||
|
||||
|
||||
class TestDeflatedSharpeRatio:
|
||||
def test_no_trials_same_as_observed(self):
|
||||
"""With 1 trial, DSR should equal the standard N(0,1) probability."""
|
||||
dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1, n_periods=100)
|
||||
assert 0.95 < dsr < 1.0
|
||||
|
||||
def test_many_trials_deflates(self):
|
||||
"""With 1000 trials, a Sharpe of 2.0 should be heavily deflated."""
|
||||
dsr = deflated_sharpe_ratio(sharpe=2.0, n_trials=1000, n_periods=100)
|
||||
assert dsr < 0.5
|
||||
|
||||
def test_negative_sharpe_zero_dsr(self):
|
||||
"""Negative Sharpe should produce near-zero DSR."""
|
||||
dsr = deflated_sharpe_ratio(sharpe=-1.0, n_trials=100, n_periods=100)
|
||||
assert dsr < 0.05
|
||||
|
||||
def test_extreme_sharpe_approaches_one(self):
|
||||
"""Extremely high Sharpe should survive deflation."""
|
||||
dsr = deflated_sharpe_ratio(sharpe=10.0, n_trials=100, n_periods=100)
|
||||
assert dsr > 0.9
|
||||
|
||||
|
||||
class TestProbabilisticSharpeRatio:
|
||||
def test_large_sample_high_sharpe(self):
|
||||
"""With many periods, high Sharpe has high PSR."""
|
||||
psr = probabilistic_sharpe_ratio(sharpe=1.5, n_periods=252, benchmark=0)
|
||||
assert psr > 0.90
|
||||
|
||||
def test_few_periods_uncertain(self):
|
||||
"""With few periods, Sharpe must be modest to show uncertainty."""
|
||||
psr = probabilistic_sharpe_ratio(sharpe=0.3, n_periods=5, benchmark=0)
|
||||
assert psr < 0.80 # Low Sharpe with few periods = uncertain
|
||||
|
||||
def test_zero_sharpe_fifty_percent(self):
|
||||
"""Zero Sharpe should give ~50% PSR (even odds)."""
|
||||
psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0)
|
||||
assert 0.40 < psr < 0.60
|
||||
|
||||
def test_zero_sharpe_fifty_percent(self):
|
||||
"""Zero Sharpe should give ~50% PSR (even odds)."""
|
||||
psr = probabilistic_sharpe_ratio(sharpe=0.0, n_periods=100, benchmark=0)
|
||||
assert 0.40 < psr < 0.60
|
||||
|
||||
def test_benchmark_above_observed_low_psr(self):
|
||||
"""If benchmark > Sharpe, PSR should be low."""
|
||||
psr = probabilistic_sharpe_ratio(sharpe=0.5, n_periods=100, benchmark=1.0)
|
||||
assert psr < 0.30
|
||||
|
||||
|
||||
class TestSharpeHaircut:
|
||||
def test_single_trial_no_haircut(self):
|
||||
"""With 1 trial, haircut is minimal."""
|
||||
hc = sharpe_haircut(sharpe=1.0, n_trials=1, n_periods=100)
|
||||
assert hc > 0.5
|
||||
|
||||
def test_many_trials_heavy_haircut(self):
|
||||
"""With many trials, haircut should be significant below observed."""
|
||||
hc = sharpe_haircut(sharpe=0.5, n_trials=639, n_periods=100)
|
||||
assert hc < 0.5 # Deflated below starting value
|
||||
|
||||
def test_haircut_range(self):
|
||||
"""Haircut should be between -1 and 6 typically."""
|
||||
for s in [0.5, 1.0, 2.0, 5.0]:
|
||||
hc = sharpe_haircut(sharpe=s, n_trials=100, n_periods=100)
|
||||
assert -1.0 < hc < 6.0
|
||||
|
||||
|
||||
class TestQuantVerdict:
|
||||
def test_strong_strategy_deploy(self):
|
||||
verdict = QuantVerdict(
|
||||
observed_sharpe=2.0,
|
||||
wf_consistency=0.80,
|
||||
n_trials=10,
|
||||
n_periods=252,
|
||||
positive_regimes=3,
|
||||
)
|
||||
result = verdict.evaluate()
|
||||
assert result["verdict"] in ("DEPLOY", "SIMULATE")
|
||||
|
||||
def test_weak_strategy_discard(self):
|
||||
verdict = QuantVerdict(
|
||||
observed_sharpe=0.3,
|
||||
wf_consistency=0.20,
|
||||
n_trials=639,
|
||||
n_periods=10,
|
||||
positive_regimes=0,
|
||||
)
|
||||
result = verdict.evaluate()
|
||||
assert result["verdict"] == "DISCARD"
|
||||
|
||||
def test_grid_mm_simulated(self):
|
||||
"""Simulate grid_mm 1h: S=8.56, 4 trades, 639 trials — should SIMULATE or DISCARD."""
|
||||
verdict = QuantVerdict(
|
||||
observed_sharpe=8.56,
|
||||
wf_consistency=0.25,
|
||||
n_trials=639,
|
||||
n_periods=4,
|
||||
positive_regimes=1,
|
||||
)
|
||||
result = verdict.evaluate()
|
||||
assert result["verdict"] in ("DISCARD", "SIMULATE")
|
||||
assert result["psr"] > 0 # PSR should be computed
|
||||
assert result["haircut_sharpe"] < result["observed_sharpe"] # Haircut reduces Sharpe
|
||||
|
||||
def test_summary_includes_all_fields(self):
|
||||
v = QuantVerdict(
|
||||
observed_sharpe=1.5,
|
||||
wf_consistency=0.70,
|
||||
n_trials=50,
|
||||
n_periods=100,
|
||||
positive_regimes=2,
|
||||
)
|
||||
r = v.evaluate()
|
||||
for key in ("verdict", "deflated_sharpe", "psr", "haircut_sharpe",
|
||||
"wf_consistency", "skewness", "recommendation"):
|
||||
assert key in r
|
||||
@@ -0,0 +1,51 @@
|
||||
"""
|
||||
Tests for quant/walkforward.py.
|
||||
"""
|
||||
from quant.walkforward import WalkForwardRunner, WFReport, WFWindow
|
||||
|
||||
|
||||
class TestWFReport:
|
||||
def test_empty_report(self):
|
||||
report = WFReport(strategy="test", interval="1h", n_windows=5)
|
||||
assert report.consistency == 0.0
|
||||
assert report.oos_sharpe == 0.0
|
||||
assert report.performance_decay == 0.0
|
||||
assert report.total_oos_trades == 0
|
||||
|
||||
def test_single_window_positive(self):
|
||||
report = WFReport(strategy="test", interval="1h", n_windows=5)
|
||||
report.windows.append(WFWindow(
|
||||
window_idx=0, is_start="2026-01-01", is_end="2026-02-01",
|
||||
oos_start="2026-02-01", oos_end="2026-03-01",
|
||||
is_sharpe=2.0, oos_sharpe=1.5, is_return_pct=5.0,
|
||||
oos_return_pct=3.0, oos_trades=10,
|
||||
))
|
||||
assert report.consistency == 1.0
|
||||
assert report.avg_oos_sharpe == 1.5
|
||||
assert report.performance_decay == 1.5 / 2.0
|
||||
assert report.total_oos_trades == 10
|
||||
|
||||
def test_mixed_windows(self):
|
||||
report = WFReport(strategy="test", interval="1h", n_windows=3)
|
||||
report.windows = [
|
||||
WFWindow(0, "A", "B", "B", "C", 2.0, 1.0, 5.0, 2.0, 5),
|
||||
WFWindow(1, "B", "C", "C", "D", 1.0, -0.5, 2.0, -1.0, 8),
|
||||
WFWindow(2, "C", "D", "D", "E", 1.5, 0.3, 3.0, 0.5, 6),
|
||||
]
|
||||
assert report.consistency == 2 / 3 # 2 of 3 windows positive OOS
|
||||
assert report.avg_oos_sharpe == (1.0 - 0.5 + 0.3) / 3
|
||||
|
||||
def test_significance_discard_weak(self):
|
||||
report = WFReport(strategy="test", interval="1h", n_windows=5)
|
||||
report.windows.append(WFWindow(
|
||||
0, "A", "B", "B", "C", 8.0, -2.0, 2.5, -5.0, 4,
|
||||
))
|
||||
report.oos_equity_curve = [{"t": 0, "v": 10000}, {"t": 1, "v": 9500}]
|
||||
sig = report.significance_report(n_trials=639)
|
||||
assert sig["verdict"] == "DISCARD"
|
||||
|
||||
def test_summary(self):
|
||||
report = WFReport(strategy="momentum", interval="4h", n_windows=3)
|
||||
s = report.summary()
|
||||
assert s["strategy"] == "momentum"
|
||||
assert s["interval"] == "4h"
|
||||
Reference in New Issue
Block a user