feat: quant validation framework — DSR, PSR, Haircut, regimes, walk-forward
Three-module quant framework replacing 'sort by Sharpe' with proper statistical validation: quant/significance.py (15 tests): - deflated_sharpe_ratio(): adjusts for N trials (Harvey & Liu 2015) - probabilistic_sharpe_ratio(): P(True SR > benchmark) given T, skew, kurt - sharpe_haircut(): expected OOS Sharpe after selection bias deflation - QuantVerdict: DEPLOY / SIMULATE / DISCARD with 5-point scoring - validate_strategy(): one-shot validation function quant/regimes.py (8 tests): - classify_regime(): trending_up/down, ranging, volatile - RegimeClassifier: stateful rolling-window classifier - conditional_performance(): per-regime trade statistics quant/walkforward.py (5 tests): - WalkForwardRunner: sequential IS/OOS window optimization - WFWindow/WFReport: structured walk-forward results - consistency score, performance decay, concatenated OOS equity - significance_report() integration Walk-forward results (real HL data with date-sliced windows): grid_mm 1h: 2/4 pos, OOS S=-0.45, 74t, haircut=-22.66 → DISCARD momentum 4h: 2/4 pos, OOS S=-1.47, 116t, haircut=-45.35 → DISCARD composite_mm 1h: 2/4 pos, OOS S=+2.97, 6t, haircut=+43.25 → SIMULATE 28 tests total
This commit is contained in:
@@ -0,0 +1,235 @@
|
||||
"""
|
||||
Statistical significance framework for backtest validation.
|
||||
|
||||
Deflated Sharpe Ratio (Harvey & Liu 2015): adjusts for multiple testing.
|
||||
Probabilistic Sharpe Ratio (Bailey & López de Prado 2012): probability
|
||||
that true Sharpe exceeds a benchmark, given sample size and skew/kurtosis.
|
||||
Haircut: expected OOS Sharpe given IS Sharpe and number of trials.
|
||||
|
||||
QuantVerdict: combines all three into a deploy/simulate/discard decision.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
|
||||
import numpy as np
|
||||
|
||||
# ── Deflated Sharpe Ratio ───────────────────────────────────
|
||||
|
||||
def deflated_sharpe_ratio(
|
||||
sharpe: float,
|
||||
n_trials: int = 1,
|
||||
n_periods: int = 100,
|
||||
) -> float:
|
||||
"""Probability that the true Sharpe ratio exceeds the observed value,
|
||||
adjusted for multiple testing (data snooping).
|
||||
|
||||
DSR = Φ⁻¹[(1 - p)^(1/N)] formulated as a probability.
|
||||
|
||||
Args:
|
||||
sharpe: observed annualized Sharpe ratio
|
||||
n_trials: number of independent strategy variations tested
|
||||
n_periods: number of return observations
|
||||
|
||||
Returns: probability (0-1) that true Sharpe > 0 after deflation.
|
||||
"""
|
||||
if n_trials <= 0 or n_periods <= 0 or math.isnan(sharpe):
|
||||
return 0.0
|
||||
|
||||
# Convert Sharpe to standard normal probability
|
||||
from scipy.stats import norm
|
||||
p_value = 1.0 - norm.cdf(sharpe)
|
||||
|
||||
# Bonferroni-type adjustment for multiple testing
|
||||
adjusted_p = min(1.0, p_value * n_trials)
|
||||
|
||||
# DSR = 1 - adjusted_p (probability true Sharpe > 0)
|
||||
dsr = max(0.0, 1.0 - adjusted_p)
|
||||
return round(dsr, 6)
|
||||
|
||||
|
||||
# ── Probabilistic Sharpe Ratio ──────────────────────────────
|
||||
|
||||
def probabilistic_sharpe_ratio(
|
||||
sharpe: float,
|
||||
n_periods: int,
|
||||
benchmark: float = 0.0,
|
||||
skewness: float = 0.0,
|
||||
kurtosis: float = 3.0,
|
||||
) -> float:
|
||||
"""Probability that the true Sharpe ratio exceeds the benchmark.
|
||||
|
||||
PSR = Φ((ŜR - SR*) * √T / √(1 - γ₃ * ŜR + (γ₄ - 1)/4 * ŜR²))
|
||||
|
||||
Args:
|
||||
sharpe: observed (annualized) Sharpe ratio
|
||||
n_periods: number of return observations
|
||||
benchmark: target Sharpe ratio (default 0)
|
||||
skewness: sample skewness of returns
|
||||
kurtosis: sample kurtosis of returns (normal = 3)
|
||||
|
||||
Returns: probability (0-1) that true SR > benchmark
|
||||
"""
|
||||
if n_periods <= 0 or math.isnan(sharpe):
|
||||
return 0.0
|
||||
|
||||
from scipy.stats import norm
|
||||
|
||||
numerator = (sharpe - benchmark) * math.sqrt(n_periods)
|
||||
denominator = math.sqrt(
|
||||
1.0 - skewness * sharpe + (kurtosis - 1.0) / 4.0 * sharpe * sharpe
|
||||
)
|
||||
|
||||
if denominator <= 0:
|
||||
return 0.5
|
||||
|
||||
z_score = numerator / denominator
|
||||
psr = float(norm.cdf(z_score))
|
||||
return round(psr, 6)
|
||||
|
||||
|
||||
# ── Sharpe Haircut ──────────────────────────────────────────
|
||||
|
||||
def sharpe_haircut(
|
||||
sharpe: float,
|
||||
n_trials: int = 1,
|
||||
n_periods: int = 100,
|
||||
) -> float:
|
||||
"""Expected out-of-sample Sharpe after adjusting for data snooping.
|
||||
|
||||
E[OOS Sharpe] ≈ IS Sharpe - E[max_i Z_i] / √T
|
||||
|
||||
Where E[max_i Z_i] is the expected maximum of N independent
|
||||
standard normals (the selection bias term).
|
||||
|
||||
Args:
|
||||
sharpe: observed in-sample Sharpe ratio
|
||||
n_trials: number of independent trials
|
||||
n_periods: number of observations
|
||||
|
||||
Returns: expected OOS Sharpe ratio (the "haircut" value)
|
||||
"""
|
||||
if n_periods <= 1:
|
||||
n_periods = 1
|
||||
|
||||
n_trials = max(1, n_trials)
|
||||
n_periods = max(2, n_periods)
|
||||
|
||||
if n_trials <= 1:
|
||||
return round(sharpe, 6)
|
||||
|
||||
euler_gamma = 0.5772156649
|
||||
|
||||
# Expected maximum of N independent standard normals
|
||||
log_n = math.log(n_trials)
|
||||
if log_n <= 0:
|
||||
expected_max = 0.0
|
||||
else:
|
||||
expected_max = math.sqrt(2.0 * log_n)
|
||||
log_log_n = math.log(log_n)
|
||||
expected_max -= (log_log_n + math.log(4 * math.pi)) / (2 * math.sqrt(2 * log_n))
|
||||
expected_max += euler_gamma / math.sqrt(2 * log_n)
|
||||
|
||||
expected_max = max(expected_max, 0.0)
|
||||
|
||||
# Deflation: observed - selection bias
|
||||
haircut = sharpe - expected_max / math.sqrt(n_periods)
|
||||
|
||||
return round(haircut, 6)
|
||||
|
||||
|
||||
# ── QuantVerdict ─────────────────────────────────────────────
|
||||
|
||||
class QuantVerdict:
|
||||
"""Combined validation report: DSR + PSR + Haircut + walk-forward.
|
||||
|
||||
Usage:
|
||||
verdict = QuantVerdict(observed_sharpe=2.0, wf_consistency=0.7, ...)
|
||||
report = verdict.evaluate()
|
||||
print(report['verdict'], report['recommendation'])
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
observed_sharpe: float,
|
||||
wf_consistency: float, # fraction of walk-forward windows with positive OOS Sharpe
|
||||
n_trials: int = 1, # total strategies/intervals tested
|
||||
n_periods: int = 100, # return observations
|
||||
positive_regimes: int = 0, # number of regimes with positive Sharpe
|
||||
benchmark_sharpe: float = 0.0,
|
||||
skewness: float = 0.0,
|
||||
kurtosis: float = 3.0,
|
||||
):
|
||||
self.observed_sharpe = observed_sharpe
|
||||
self.wf_consistency = wf_consistency
|
||||
self.n_trials = n_trials
|
||||
self.n_periods = n_periods
|
||||
self.positive_regimes = positive_regimes
|
||||
self.benchmark_sharpe = benchmark_sharpe
|
||||
self.skewness = skewness
|
||||
self.kurtosis = kurtosis
|
||||
|
||||
def evaluate(self) -> dict:
|
||||
dsr = deflated_sharpe_ratio(self.observed_sharpe, self.n_trials, self.n_periods)
|
||||
psr = probabilistic_sharpe_ratio(
|
||||
self.observed_sharpe, self.n_periods,
|
||||
self.benchmark_sharpe, self.skewness, self.kurtosis,
|
||||
)
|
||||
hc = sharpe_haircut(self.observed_sharpe, self.n_trials, self.n_periods)
|
||||
|
||||
# Decision logic
|
||||
score = 0
|
||||
if dsr > 0.80: score += 1
|
||||
if psr > 0.70: score += 1
|
||||
if self.wf_consistency > 0.50: score += 1
|
||||
if self.positive_regimes >= 2: score += 1
|
||||
if hc > 0.5: score += 1
|
||||
|
||||
if score >= 4:
|
||||
verdict = "DEPLOY"
|
||||
rec = "Strategy passes all significance tests. Deploy with 1/10th size + daily PnL stop."
|
||||
elif score >= 2:
|
||||
verdict = "SIMULATE"
|
||||
rec = "Marginal significance. Run through Phase 3 queue simulator before live."
|
||||
else:
|
||||
verdict = "DISCARD"
|
||||
rec = "Fails significance tests. Strategy is indistinguishable from noise."
|
||||
|
||||
return {
|
||||
"verdict": verdict,
|
||||
"deflated_sharpe": dsr,
|
||||
"psr": psr,
|
||||
"haircut_sharpe": hc,
|
||||
"wf_consistency": round(self.wf_consistency, 3),
|
||||
"observed_sharpe": round(self.observed_sharpe, 3),
|
||||
"n_trials": self.n_trials,
|
||||
"n_periods": self.n_periods,
|
||||
"positive_regimes": self.positive_regimes,
|
||||
"skewness": round(self.skewness, 4),
|
||||
"kurtosis": round(self.kurtosis, 4),
|
||||
"score": f"{score}/5",
|
||||
"recommendation": rec,
|
||||
}
|
||||
|
||||
|
||||
def validate_strategy(
|
||||
sharpe: float,
|
||||
n_trades: int,
|
||||
n_trials: int = 1,
|
||||
wf_consistency: float = 0.0,
|
||||
positive_regimes: int = 0,
|
||||
skewness: float = 0.0,
|
||||
kurtosis: float = 3.0,
|
||||
) -> dict:
|
||||
"""Quick one-shot validation of a strategy."""
|
||||
v = QuantVerdict(
|
||||
observed_sharpe=sharpe,
|
||||
n_periods=n_trades,
|
||||
n_trials=n_trials,
|
||||
wf_consistency=wf_consistency,
|
||||
positive_regimes=positive_regimes,
|
||||
skewness=skewness,
|
||||
kurtosis=kurtosis,
|
||||
)
|
||||
return v.evaluate()
|
||||
Reference in New Issue
Block a user