Files
ramseshk 543537e33f feat: quant validation framework — DSR, PSR, Haircut, regimes, walk-forward
Three-module quant framework replacing 'sort by Sharpe' with proper
statistical validation:

quant/significance.py (15 tests):
  - deflated_sharpe_ratio(): adjusts for N trials (Harvey & Liu 2015)
  - probabilistic_sharpe_ratio(): P(True SR > benchmark) given T, skew, kurt
  - sharpe_haircut(): expected OOS Sharpe after selection bias deflation
  - QuantVerdict: DEPLOY / SIMULATE / DISCARD with 5-point scoring
  - validate_strategy(): one-shot validation function

quant/regimes.py (8 tests):
  - classify_regime(): trending_up/down, ranging, volatile
  - RegimeClassifier: stateful rolling-window classifier
  - conditional_performance(): per-regime trade statistics

quant/walkforward.py (5 tests):
  - WalkForwardRunner: sequential IS/OOS window optimization
  - WFWindow/WFReport: structured walk-forward results
  - consistency score, performance decay, concatenated OOS equity
  - significance_report() integration

Walk-forward results (real HL data with date-sliced windows):
  grid_mm 1h:    2/4 pos, OOS S=-0.45,  74t, haircut=-22.66 → DISCARD
  momentum 4h:   2/4 pos, OOS S=-1.47, 116t, haircut=-45.35 → DISCARD
  composite_mm 1h: 2/4 pos, OOS S=+2.97, 6t, haircut=+43.25 → SIMULATE

28 tests total
2026-08-10 16:20:56 +08:00

236 lines
7.5 KiB
Python

"""
Statistical significance framework for backtest validation.
Deflated Sharpe Ratio (Harvey & Liu 2015): adjusts for multiple testing.
Probabilistic Sharpe Ratio (Bailey & López de Prado 2012): probability
that true Sharpe exceeds a benchmark, given sample size and skew/kurtosis.
Haircut: expected OOS Sharpe given IS Sharpe and number of trials.
QuantVerdict: combines all three into a deploy/simulate/discard decision.
"""
from __future__ import annotations
import math
import numpy as np
# ── Deflated Sharpe Ratio ───────────────────────────────────
def deflated_sharpe_ratio(
sharpe: float,
n_trials: int = 1,
n_periods: int = 100,
) -> float:
"""Probability that the true Sharpe ratio exceeds the observed value,
adjusted for multiple testing (data snooping).
DSR = Φ⁻¹[(1 - p)^(1/N)] formulated as a probability.
Args:
sharpe: observed annualized Sharpe ratio
n_trials: number of independent strategy variations tested
n_periods: number of return observations
Returns: probability (0-1) that true Sharpe > 0 after deflation.
"""
if n_trials <= 0 or n_periods <= 0 or math.isnan(sharpe):
return 0.0
# Convert Sharpe to standard normal probability
from scipy.stats import norm
p_value = 1.0 - norm.cdf(sharpe)
# Bonferroni-type adjustment for multiple testing
adjusted_p = min(1.0, p_value * n_trials)
# DSR = 1 - adjusted_p (probability true Sharpe > 0)
dsr = max(0.0, 1.0 - adjusted_p)
return round(dsr, 6)
# ── Probabilistic Sharpe Ratio ──────────────────────────────
def probabilistic_sharpe_ratio(
sharpe: float,
n_periods: int,
benchmark: float = 0.0,
skewness: float = 0.0,
kurtosis: float = 3.0,
) -> float:
"""Probability that the true Sharpe ratio exceeds the benchmark.
PSR = Φ((ŜR - SR*) * √T / √(1 - γ₃ * ŜR + (γ₄ - 1)/4 * ŜR²))
Args:
sharpe: observed (annualized) Sharpe ratio
n_periods: number of return observations
benchmark: target Sharpe ratio (default 0)
skewness: sample skewness of returns
kurtosis: sample kurtosis of returns (normal = 3)
Returns: probability (0-1) that true SR > benchmark
"""
if n_periods <= 0 or math.isnan(sharpe):
return 0.0
from scipy.stats import norm
numerator = (sharpe - benchmark) * math.sqrt(n_periods)
denominator = math.sqrt(
1.0 - skewness * sharpe + (kurtosis - 1.0) / 4.0 * sharpe * sharpe
)
if denominator <= 0:
return 0.5
z_score = numerator / denominator
psr = float(norm.cdf(z_score))
return round(psr, 6)
# ── Sharpe Haircut ──────────────────────────────────────────
def sharpe_haircut(
sharpe: float,
n_trials: int = 1,
n_periods: int = 100,
) -> float:
"""Expected out-of-sample Sharpe after adjusting for data snooping.
E[OOS Sharpe] ≈ IS Sharpe - E[max_i Z_i] / √T
Where E[max_i Z_i] is the expected maximum of N independent
standard normals (the selection bias term).
Args:
sharpe: observed in-sample Sharpe ratio
n_trials: number of independent trials
n_periods: number of observations
Returns: expected OOS Sharpe ratio (the "haircut" value)
"""
if n_periods <= 1:
n_periods = 1
n_trials = max(1, n_trials)
n_periods = max(2, n_periods)
if n_trials <= 1:
return round(sharpe, 6)
euler_gamma = 0.5772156649
# Expected maximum of N independent standard normals
log_n = math.log(n_trials)
if log_n <= 0:
expected_max = 0.0
else:
expected_max = math.sqrt(2.0 * log_n)
log_log_n = math.log(log_n)
expected_max -= (log_log_n + math.log(4 * math.pi)) / (2 * math.sqrt(2 * log_n))
expected_max += euler_gamma / math.sqrt(2 * log_n)
expected_max = max(expected_max, 0.0)
# Deflation: observed - selection bias
haircut = sharpe - expected_max / math.sqrt(n_periods)
return round(haircut, 6)
# ── QuantVerdict ─────────────────────────────────────────────
class QuantVerdict:
"""Combined validation report: DSR + PSR + Haircut + walk-forward.
Usage:
verdict = QuantVerdict(observed_sharpe=2.0, wf_consistency=0.7, ...)
report = verdict.evaluate()
print(report['verdict'], report['recommendation'])
"""
def __init__(
self,
observed_sharpe: float,
wf_consistency: float, # fraction of walk-forward windows with positive OOS Sharpe
n_trials: int = 1, # total strategies/intervals tested
n_periods: int = 100, # return observations
positive_regimes: int = 0, # number of regimes with positive Sharpe
benchmark_sharpe: float = 0.0,
skewness: float = 0.0,
kurtosis: float = 3.0,
):
self.observed_sharpe = observed_sharpe
self.wf_consistency = wf_consistency
self.n_trials = n_trials
self.n_periods = n_periods
self.positive_regimes = positive_regimes
self.benchmark_sharpe = benchmark_sharpe
self.skewness = skewness
self.kurtosis = kurtosis
def evaluate(self) -> dict:
dsr = deflated_sharpe_ratio(self.observed_sharpe, self.n_trials, self.n_periods)
psr = probabilistic_sharpe_ratio(
self.observed_sharpe, self.n_periods,
self.benchmark_sharpe, self.skewness, self.kurtosis,
)
hc = sharpe_haircut(self.observed_sharpe, self.n_trials, self.n_periods)
# Decision logic
score = 0
if dsr > 0.80: score += 1
if psr > 0.70: score += 1
if self.wf_consistency > 0.50: score += 1
if self.positive_regimes >= 2: score += 1
if hc > 0.5: score += 1
if score >= 4:
verdict = "DEPLOY"
rec = "Strategy passes all significance tests. Deploy with 1/10th size + daily PnL stop."
elif score >= 2:
verdict = "SIMULATE"
rec = "Marginal significance. Run through Phase 3 queue simulator before live."
else:
verdict = "DISCARD"
rec = "Fails significance tests. Strategy is indistinguishable from noise."
return {
"verdict": verdict,
"deflated_sharpe": dsr,
"psr": psr,
"haircut_sharpe": hc,
"wf_consistency": round(self.wf_consistency, 3),
"observed_sharpe": round(self.observed_sharpe, 3),
"n_trials": self.n_trials,
"n_periods": self.n_periods,
"positive_regimes": self.positive_regimes,
"skewness": round(self.skewness, 4),
"kurtosis": round(self.kurtosis, 4),
"score": f"{score}/5",
"recommendation": rec,
}
def validate_strategy(
sharpe: float,
n_trades: int,
n_trials: int = 1,
wf_consistency: float = 0.0,
positive_regimes: int = 0,
skewness: float = 0.0,
kurtosis: float = 3.0,
) -> dict:
"""Quick one-shot validation of a strategy."""
v = QuantVerdict(
observed_sharpe=sharpe,
n_periods=n_trades,
n_trials=n_trials,
wf_consistency=wf_consistency,
positive_regimes=positive_regimes,
skewness=skewness,
kurtosis=kurtosis,
)
return v.evaluate()