""" Statistical significance framework for backtest validation. Deflated Sharpe Ratio (Harvey & Liu 2015): adjusts for multiple testing. Probabilistic Sharpe Ratio (Bailey & López de Prado 2012): probability that true Sharpe exceeds a benchmark, given sample size and skew/kurtosis. Haircut: expected OOS Sharpe given IS Sharpe and number of trials. QuantVerdict: combines all three into a deploy/simulate/discard decision. """ from __future__ import annotations import math import numpy as np # ── Deflated Sharpe Ratio ─────────────────────────────────── def deflated_sharpe_ratio( sharpe: float, n_trials: int = 1, n_periods: int = 100, ) -> float: """Probability that the true Sharpe ratio exceeds the observed value, adjusted for multiple testing (data snooping). DSR = Φ⁻¹[(1 - p)^(1/N)] formulated as a probability. Args: sharpe: observed annualized Sharpe ratio n_trials: number of independent strategy variations tested n_periods: number of return observations Returns: probability (0-1) that true Sharpe > 0 after deflation. """ if n_trials <= 0 or n_periods <= 0 or math.isnan(sharpe): return 0.0 # Convert Sharpe to standard normal probability from scipy.stats import norm p_value = 1.0 - norm.cdf(sharpe) # Bonferroni-type adjustment for multiple testing adjusted_p = min(1.0, p_value * n_trials) # DSR = 1 - adjusted_p (probability true Sharpe > 0) dsr = max(0.0, 1.0 - adjusted_p) return round(dsr, 6) # ── Probabilistic Sharpe Ratio ────────────────────────────── def probabilistic_sharpe_ratio( sharpe: float, n_periods: int, benchmark: float = 0.0, skewness: float = 0.0, kurtosis: float = 3.0, ) -> float: """Probability that the true Sharpe ratio exceeds the benchmark. PSR = Φ((ŜR - SR*) * √T / √(1 - γ₃ * ŜR + (γ₄ - 1)/4 * ŜR²)) Args: sharpe: observed (annualized) Sharpe ratio n_periods: number of return observations benchmark: target Sharpe ratio (default 0) skewness: sample skewness of returns kurtosis: sample kurtosis of returns (normal = 3) Returns: probability (0-1) that true SR > benchmark """ if n_periods <= 0 or math.isnan(sharpe): return 0.0 from scipy.stats import norm numerator = (sharpe - benchmark) * math.sqrt(n_periods) denominator = math.sqrt( 1.0 - skewness * sharpe + (kurtosis - 1.0) / 4.0 * sharpe * sharpe ) if denominator <= 0: return 0.5 z_score = numerator / denominator psr = float(norm.cdf(z_score)) return round(psr, 6) # ── Sharpe Haircut ────────────────────────────────────────── def sharpe_haircut( sharpe: float, n_trials: int = 1, n_periods: int = 100, ) -> float: """Expected out-of-sample Sharpe after adjusting for data snooping. E[OOS Sharpe] ≈ IS Sharpe - E[max_i Z_i] / √T Where E[max_i Z_i] is the expected maximum of N independent standard normals (the selection bias term). Args: sharpe: observed in-sample Sharpe ratio n_trials: number of independent trials n_periods: number of observations Returns: expected OOS Sharpe ratio (the "haircut" value) """ if n_periods <= 1: n_periods = 1 n_trials = max(1, n_trials) n_periods = max(2, n_periods) if n_trials <= 1: return round(sharpe, 6) euler_gamma = 0.5772156649 # Expected maximum of N independent standard normals log_n = math.log(n_trials) if log_n <= 0: expected_max = 0.0 else: expected_max = math.sqrt(2.0 * log_n) log_log_n = math.log(log_n) expected_max -= (log_log_n + math.log(4 * math.pi)) / (2 * math.sqrt(2 * log_n)) expected_max += euler_gamma / math.sqrt(2 * log_n) expected_max = max(expected_max, 0.0) # Deflation: observed - selection bias haircut = sharpe - expected_max / math.sqrt(n_periods) return round(haircut, 6) # ── QuantVerdict ───────────────────────────────────────────── class QuantVerdict: """Combined validation report: DSR + PSR + Haircut + walk-forward. Usage: verdict = QuantVerdict(observed_sharpe=2.0, wf_consistency=0.7, ...) report = verdict.evaluate() print(report['verdict'], report['recommendation']) """ def __init__( self, observed_sharpe: float, wf_consistency: float, # fraction of walk-forward windows with positive OOS Sharpe n_trials: int = 1, # total strategies/intervals tested n_periods: int = 100, # return observations positive_regimes: int = 0, # number of regimes with positive Sharpe benchmark_sharpe: float = 0.0, skewness: float = 0.0, kurtosis: float = 3.0, ): self.observed_sharpe = observed_sharpe self.wf_consistency = wf_consistency self.n_trials = n_trials self.n_periods = n_periods self.positive_regimes = positive_regimes self.benchmark_sharpe = benchmark_sharpe self.skewness = skewness self.kurtosis = kurtosis def evaluate(self) -> dict: dsr = deflated_sharpe_ratio(self.observed_sharpe, self.n_trials, self.n_periods) psr = probabilistic_sharpe_ratio( self.observed_sharpe, self.n_periods, self.benchmark_sharpe, self.skewness, self.kurtosis, ) hc = sharpe_haircut(self.observed_sharpe, self.n_trials, self.n_periods) # Decision logic score = 0 if dsr > 0.80: score += 1 if psr > 0.70: score += 1 if self.wf_consistency > 0.50: score += 1 if self.positive_regimes >= 2: score += 1 if hc > 0.5: score += 1 if score >= 4: verdict = "DEPLOY" rec = "Strategy passes all significance tests. Deploy with 1/10th size + daily PnL stop." elif score >= 2: verdict = "SIMULATE" rec = "Marginal significance. Run through Phase 3 queue simulator before live." else: verdict = "DISCARD" rec = "Fails significance tests. Strategy is indistinguishable from noise." return { "verdict": verdict, "deflated_sharpe": dsr, "psr": psr, "haircut_sharpe": hc, "wf_consistency": round(self.wf_consistency, 3), "observed_sharpe": round(self.observed_sharpe, 3), "n_trials": self.n_trials, "n_periods": self.n_periods, "positive_regimes": self.positive_regimes, "skewness": round(self.skewness, 4), "kurtosis": round(self.kurtosis, 4), "score": f"{score}/5", "recommendation": rec, } def validate_strategy( sharpe: float, n_trades: int, n_trials: int = 1, wf_consistency: float = 0.0, positive_regimes: int = 0, skewness: float = 0.0, kurtosis: float = 3.0, ) -> dict: """Quick one-shot validation of a strategy.""" v = QuantVerdict( observed_sharpe=sharpe, n_periods=n_trades, n_trials=n_trials, wf_consistency=wf_consistency, positive_regimes=positive_regimes, skewness=skewness, kurtosis=kurtosis, ) return v.evaluate()