543537e33f
Three-module quant framework replacing 'sort by Sharpe' with proper statistical validation: quant/significance.py (15 tests): - deflated_sharpe_ratio(): adjusts for N trials (Harvey & Liu 2015) - probabilistic_sharpe_ratio(): P(True SR > benchmark) given T, skew, kurt - sharpe_haircut(): expected OOS Sharpe after selection bias deflation - QuantVerdict: DEPLOY / SIMULATE / DISCARD with 5-point scoring - validate_strategy(): one-shot validation function quant/regimes.py (8 tests): - classify_regime(): trending_up/down, ranging, volatile - RegimeClassifier: stateful rolling-window classifier - conditional_performance(): per-regime trade statistics quant/walkforward.py (5 tests): - WalkForwardRunner: sequential IS/OOS window optimization - WFWindow/WFReport: structured walk-forward results - consistency score, performance decay, concatenated OOS equity - significance_report() integration Walk-forward results (real HL data with date-sliced windows): grid_mm 1h: 2/4 pos, OOS S=-0.45, 74t, haircut=-22.66 → DISCARD momentum 4h: 2/4 pos, OOS S=-1.47, 116t, haircut=-45.35 → DISCARD composite_mm 1h: 2/4 pos, OOS S=+2.97, 6t, haircut=+43.25 → SIMULATE 28 tests total
254 lines
9.3 KiB
Python
254 lines
9.3 KiB
Python
"""
|
|
Walk-forward validation framework.
|
|
|
|
Splits market data into sequential IS/OOS windows, optimizes strategy
|
|
parameters on in-sample data, and tests on out-of-sample data. This is
|
|
the minimum bar for any strategy before live deployment.
|
|
|
|
Computes:
|
|
- OOS Sharpe per window
|
|
- Walk-forward consistency (% positive OOS windows)
|
|
- Performance decay (IS → OOS degradation)
|
|
- Concatenated OOS equity curve
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from dataclasses import dataclass, field
|
|
from datetime import datetime, timezone
|
|
from typing import Optional
|
|
|
|
import numpy as np
|
|
|
|
from quant.significance import QuantVerdict
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
@dataclass
|
|
class WFWindow:
|
|
"""Single walk-forward window result."""
|
|
window_idx: int
|
|
is_start: str
|
|
is_end: str
|
|
oos_start: str
|
|
oos_end: str
|
|
is_sharpe: float
|
|
oos_sharpe: float
|
|
is_return_pct: float
|
|
oos_return_pct: float
|
|
oos_trades: int
|
|
best_params: dict = field(default_factory=dict)
|
|
|
|
|
|
@dataclass
|
|
class WFReport:
|
|
"""Complete walk-forward analysis report."""
|
|
strategy: str
|
|
interval: str
|
|
n_windows: int
|
|
windows: list[WFWindow] = field(default_factory=list)
|
|
oos_equity_curve: list[dict] = field(default_factory=list)
|
|
|
|
@property
|
|
def oos_sharpe(self) -> float:
|
|
if not self.oos_equity_curve:
|
|
return 0.0
|
|
vals = [p["v"] for p in self.oos_equity_curve if p.get("v")]
|
|
if len(vals) < 2:
|
|
return 0.0
|
|
rets = [(vals[i] - vals[i - 1]) / vals[i - 1] for i in range(1, len(vals)) if vals[i - 1] > 0]
|
|
if not rets:
|
|
return 0.0
|
|
mean = sum(rets) / len(rets)
|
|
std = (sum((r - mean) ** 2 for r in rets) / max(len(rets) - 1, 1)) ** 0.5
|
|
return mean / std * np.sqrt(365 * 24) if std > 0 else 0.0
|
|
|
|
@property
|
|
def consistency(self) -> float:
|
|
"""Fraction of windows with positive OOS Sharpe."""
|
|
if not self.windows:
|
|
return 0.0
|
|
positive = sum(1 for w in self.windows if w.oos_sharpe > 0)
|
|
return positive / len(self.windows)
|
|
|
|
@property
|
|
def avg_oos_sharpe(self) -> float:
|
|
if not self.windows:
|
|
return 0.0
|
|
return sum(w.oos_sharpe for w in self.windows) / len(self.windows)
|
|
|
|
@property
|
|
def performance_decay(self) -> float:
|
|
"""IS Sharpe → OOS decay ratio. <1 = decay, >1 = improvement (rare)."""
|
|
avg_is = sum(w.is_sharpe for w in self.windows) / max(len(self.windows), 1)
|
|
avg_oos = self.avg_oos_sharpe
|
|
return avg_oos / avg_is if avg_is != 0 else 0.0
|
|
|
|
@property
|
|
def total_oos_trades(self) -> int:
|
|
return sum(w.oos_trades for w in self.windows)
|
|
|
|
def significance_report(self, n_trials: int = 639) -> dict:
|
|
return QuantVerdict(
|
|
observed_sharpe=self.oos_sharpe,
|
|
wf_consistency=self.consistency,
|
|
n_trials=n_trials,
|
|
n_periods=max(self.total_oos_trades, 1),
|
|
positive_regimes=0,
|
|
).evaluate()
|
|
|
|
def summary(self) -> dict:
|
|
return {
|
|
"strategy": self.strategy,
|
|
"interval": self.interval,
|
|
"n_windows": self.n_windows,
|
|
"consistency": round(self.consistency, 3),
|
|
"oos_sharpe": round(self.oos_sharpe, 3),
|
|
"avg_oos_sharpe": round(self.avg_oos_sharpe, 3),
|
|
"performance_decay": round(self.performance_decay, 3),
|
|
"total_oos_trades": self.total_oos_trades,
|
|
"verdict": self.significance_report()["verdict"],
|
|
}
|
|
|
|
|
|
class WalkForwardRunner:
|
|
"""Run walk-forward validation using VBT runner on historical data."""
|
|
|
|
def __init__(
|
|
self,
|
|
n_windows: int = 5,
|
|
bar_limits: list[int] | None = None,
|
|
fee_tier: int = 0,
|
|
staking_tier: str = "none",
|
|
):
|
|
self._n_windows = n_windows
|
|
self._bar_limits = bar_limits or [100, 200, 500, 1000, 2000]
|
|
self._fee_tier = fee_tier
|
|
self._staking_tier = staking_tier
|
|
|
|
def run(
|
|
self,
|
|
strategy: str,
|
|
interval: str = "1h",
|
|
coin: str = "BTC",
|
|
testnet: bool = False,
|
|
) -> WFReport:
|
|
"""Execute walk-forward validation on real Hyperliquid data.
|
|
|
|
Uses HyperliquidDataProvider to fetch candle data, then splits
|
|
into sequential IS/OOS windows. For each window:
|
|
1. Optimize bar limit on IS data (pick best by Sharpe)
|
|
2. Test the optimal bar limit on OOS data
|
|
3. Record IS/OOS Sharpe, returns, trades
|
|
"""
|
|
from framework.data import HyperliquidDataProvider
|
|
from backtests.vbt_runner import VBTBacktestRunner
|
|
|
|
provider = HyperliquidDataProvider(testnet=testnet)
|
|
|
|
# Fetch maximum data needed
|
|
max_bars = max(self._bar_limits) * (self._n_windows + 1)
|
|
df = provider.fetch_candles(coin, interval=interval, limit=max_bars)
|
|
|
|
if df.empty or len(df) < 100:
|
|
return WFReport(strategy=strategy, interval=interval, n_windows=self._n_windows)
|
|
|
|
total_bars = len(df)
|
|
window_size = total_bars // (self._n_windows + 1)
|
|
if window_size < 50:
|
|
return WFReport(strategy=strategy, interval=interval, n_windows=self._n_windows)
|
|
|
|
report = WFReport(strategy=strategy, interval=interval, n_windows=self._n_windows)
|
|
cumulative_oos_equity = 10000.0
|
|
report.oos_equity_curve.append({"t": 0, "v": cumulative_oos_equity})
|
|
|
|
for w in range(self._n_windows):
|
|
is_start_idx = w * window_size
|
|
is_end_idx = is_start_idx + window_size
|
|
oos_start_idx = is_end_idx
|
|
oos_end_idx = min(oos_start_idx + window_size, total_bars)
|
|
|
|
is_start_ts = str(df.index[is_start_idx])[:10]
|
|
is_end_ts = str(df.index[min(is_end_idx - 1, total_bars - 1)])[:10]
|
|
oos_start_ts = str(df.index[min(oos_start_idx, total_bars - 1)])[:10]
|
|
oos_end_ts = str(df.index[min(oos_end_idx - 1, total_bars - 1)])[:10]
|
|
|
|
# Convert dates to ms for HL API
|
|
is_start_ms = int(df.index[is_start_idx].timestamp() * 1000)
|
|
is_end_ms = int(df.index[min(is_end_idx - 1, total_bars - 1)].timestamp() * 1000)
|
|
oos_start_ms = int(df.index[min(oos_start_idx, total_bars - 1)].timestamp() * 1000)
|
|
oos_end_ms = int(df.index[min(oos_end_idx - 1, total_bars - 1)].timestamp() * 1000)
|
|
|
|
# IN-SAMPLE: optimize bar limit
|
|
best_limit = self._bar_limits[0]
|
|
best_is_sharpe = -999.0
|
|
best_is_return = 0.0
|
|
|
|
for limit in self._bar_limits:
|
|
is_bars = min(limit, window_size)
|
|
try:
|
|
runner = VBTBacktestRunner(
|
|
vip_tier=self._fee_tier, staking_tier=self._staking_tier
|
|
)
|
|
result = runner.run_strategy(
|
|
strategy=strategy, interval=interval, testnet=testnet,
|
|
limit=is_bars, start_ms=is_start_ms, end_ms=is_end_ms,
|
|
)
|
|
if result and result.get("sharpe", -999) > best_is_sharpe:
|
|
best_is_sharpe = result.get("sharpe", -999)
|
|
best_is_return = result.get("total_return_pct", 0)
|
|
best_limit = limit
|
|
except Exception:
|
|
pass
|
|
|
|
# OUT-OF-SAMPLE: test the best bar limit
|
|
if best_is_sharpe <= -998:
|
|
continue
|
|
|
|
oos_bars = min(best_limit, oos_end_idx - oos_start_idx)
|
|
try:
|
|
runner = VBTBacktestRunner(
|
|
vip_tier=self._fee_tier, staking_tier=self._staking_tier
|
|
)
|
|
oos_result = runner.run_strategy(
|
|
strategy=strategy, interval=interval, testnet=testnet,
|
|
limit=oos_bars, start_ms=oos_start_ms, end_ms=oos_end_ms,
|
|
)
|
|
if oos_result:
|
|
oos_sharpe = oos_result.get("sharpe", 0)
|
|
oos_return = oos_result.get("total_return_pct", 0)
|
|
oos_trades = len(oos_result.get("trades", []))
|
|
|
|
cumulative_oos_equity += cumulative_oos_equity * oos_return / 100.0
|
|
report.oos_equity_curve.append({
|
|
"t": w + 1,
|
|
"v": round(cumulative_oos_equity, 2),
|
|
})
|
|
|
|
report.windows.append(WFWindow(
|
|
window_idx=w,
|
|
is_start=is_start_ts, is_end=is_end_ts,
|
|
oos_start=oos_start_ts, oos_end=oos_end_ts,
|
|
is_sharpe=round(best_is_sharpe, 3),
|
|
oos_sharpe=round(oos_sharpe, 3),
|
|
is_return_pct=round(best_is_return, 2),
|
|
oos_return_pct=round(oos_return, 2),
|
|
oos_trades=oos_trades,
|
|
best_params={"limit": best_limit},
|
|
))
|
|
except Exception:
|
|
pass
|
|
|
|
return report
|
|
|
|
|
|
# ── Quick validation ────────────────────────────────────────
|
|
|
|
def quick_validate(strategy: str, interval: str = "1h", **kwargs) -> dict:
|
|
"""Run walk-forward and return significance report in one call."""
|
|
wfr = WalkForwardRunner(**kwargs)
|
|
report = wfr.run(strategy=strategy, interval=interval)
|
|
return report.summary()
|