20ee340cef
Track 1 — VBT Candle-Frequency Pipeline: - backtests/vbt_validator.py: VBTValidator with 11 checks — timestamp monotonicity, duplicates, NaN, data gaps, lookahead bias, signal alignment, density, coincident entry/exit, min trade count, fee application, benchmark comparison. ValidationReport dataclass with errors/warnings/stats. Validates VBT results or raw signal arrays. - backtests/vbt_viz.py: VBTVisualizer with 10+ Plotly chart methods — equity curve with benchmark, drawdown, rolling Sharpe/Sortino/vol, trade markers, returns distribution with normal fit, monthly PnL heatmap, gross vs net, holding periods, parameter sensitivity heatmaps, dashboard compositor, HTML save (self-contained, CDN Plotly). All methods handle empty/null inputs. - backtests/vbt_report.py: Markdown + HTML report generator — structured sections for implementation summary, performance metrics, cost analysis, validation results, signal analysis, known limitations, next steps. batch_report() for mass report generation from results directory. - backtests/vbt_runner.py: Added run_benchmark() (buy-and-hold VBT portfolio), validate() (integrated VBTValidator), run_with_report() (fetch→validate→ backtest→visualize→save in one call). Track 2 — HFT Tick Pipeline: - backtests/tick_viz.py: 9-panel HFT dashboard — price+trade markers, spread dynamics, top-of-book depth, microprice vs mid, OBI/OFI panel, VPIN toxicity with thresholds, event timeline (PnL from tick_runner), markout curves at 6 horizons. Parquet→pandas→Plotly pipeline. Dark-themed HTML output for microstructure review. - data/duckdb_load.py: Parquet→DuckDB loader — creates l2_snapshots, trades, funding tables with schema. Pre-computed 1s rollup views for microprice, OFI, trade imbalance. Markout queries directly in SQL. Incremental loading with load_state tracking. CLI Integration: - cli.py: Added 'report' (full VBT report), 'validate' (check existing results), 'hft' (tick dashboard generation) commands. Fixed argparse help string escaping. 355 tests passing (34 new).
430 lines
15 KiB
Python
430 lines
15 KiB
Python
"""
|
|
VBT Backtest Validator — ensures backtest results are credible.
|
|
|
|
Checks for common backtest errors:
|
|
- Lookahead bias (entries using future information)
|
|
- Signal/timestamp alignment
|
|
- Data integrity (NaN, duplicates, gaps)
|
|
- Fee application (gross vs net divergence)
|
|
- Statistical sufficiency (minimum trade count)
|
|
- Benchmark comparison
|
|
- Signal quality (density, clustering)
|
|
|
|
Usage:
|
|
from backtests.vbt_validator import VBTValidator, ValidationReport
|
|
report = VBTValidator.validate(entries, exits, close, pf, trades, benchmark)
|
|
print(report.summary())
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from dataclasses import dataclass, field
|
|
from typing import Any, Optional
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
MIN_TRADES_FOR_STATS = 10
|
|
MAX_SIGNAL_DENSITY = 0.5
|
|
MIN_SIGNAL_DENSITY = 0.001
|
|
MAX_CONSEC_SIGNALS = 10
|
|
|
|
|
|
@dataclass
|
|
class ValidationReport:
|
|
"""Structured validation output. Check `passes` before trusting results."""
|
|
|
|
strategy: str
|
|
interval: str
|
|
|
|
# Pass/fail flags per check
|
|
checks: dict[str, bool] = field(default_factory=dict)
|
|
details: dict[str, Any] = field(default_factory=dict)
|
|
|
|
# warnings: non-fatal issues
|
|
warnings: list[str] = field(default_factory=list)
|
|
|
|
# errors: fatal issues
|
|
errors: list[str] = field(default_factory=list)
|
|
|
|
# Summary stats
|
|
stats: dict[str, Any] = field(default_factory=dict)
|
|
|
|
@property
|
|
def passes(self) -> bool:
|
|
return len(self.errors) == 0
|
|
|
|
@property
|
|
def all_checks_pass(self) -> bool:
|
|
return all(self.checks.values()) if self.checks else True
|
|
|
|
def summary(self) -> str:
|
|
lines = [
|
|
"",
|
|
"=" * 60,
|
|
f" Validation Report — {self.strategy} ({self.interval})",
|
|
"=" * 60,
|
|
]
|
|
passed = sum(1 for v in self.checks.values() if v)
|
|
total = len(self.checks)
|
|
lines.append(f" Checks: {passed}/{total} passed "
|
|
f"Warnings: {len(self.warnings)} Errors: {len(self.errors)}")
|
|
lines.append("")
|
|
|
|
if self.errors:
|
|
lines.append(" ERRORS:")
|
|
for e in self.errors:
|
|
lines.append(f" ✗ {e}")
|
|
lines.append("")
|
|
|
|
if self.warnings:
|
|
lines.append(" WARNINGS:")
|
|
for w in self.warnings:
|
|
lines.append(f" ⚠ {w}")
|
|
lines.append("")
|
|
|
|
lines.append(" CHECKS:")
|
|
for name, result in self.checks.items():
|
|
icon = "✓" if result else "✗"
|
|
detail = self.details.get(name, "")
|
|
lines.append(f" {icon} {name}: {detail}")
|
|
|
|
if self.stats:
|
|
lines.append("")
|
|
lines.append(" STATS:")
|
|
for k, v in self.stats.items():
|
|
if isinstance(v, float):
|
|
lines.append(f" {k}: {v:.4f}")
|
|
else:
|
|
lines.append(f" {k}: {v}")
|
|
|
|
return "\n".join(lines)
|
|
|
|
|
|
class VBTValidator:
|
|
"""Validate VBT backtest integrity across multiple dimensions.
|
|
|
|
Usage:
|
|
v = VBTValidator()
|
|
report = v.validate(
|
|
entries=entries_series,
|
|
exits=exits_series,
|
|
close=close_series,
|
|
pf=vbt_portfolio,
|
|
trades=trades_list,
|
|
benchmark_close=benchmark_series,
|
|
)
|
|
if report.passes:
|
|
print("Backtest is credible")
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
min_trades: int = MIN_TRADES_FOR_STATS,
|
|
max_signal_density: float = MAX_SIGNAL_DENSITY,
|
|
min_signal_density: float = MIN_SIGNAL_DENSITY,
|
|
max_consec_signals: int = MAX_CONSEC_SIGNALS,
|
|
expected_dt_seconds: Optional[float] = None,
|
|
):
|
|
self._min_trades = min_trades
|
|
self._max_signal_density = max_signal_density
|
|
self._min_signal_density = min_signal_density
|
|
self._max_consec_signals = max_consec_signals
|
|
self._expected_dt = expected_dt_seconds
|
|
|
|
def validate(
|
|
self,
|
|
entries: pd.Series,
|
|
exits: pd.Series,
|
|
close: pd.Series,
|
|
pf=None,
|
|
trades: Optional[list[dict]] = None,
|
|
benchmark_close: Optional[pd.Series] = None,
|
|
strategy: str = "unknown",
|
|
interval: str = "unknown",
|
|
) -> ValidationReport:
|
|
report = ValidationReport(strategy=strategy, interval=interval)
|
|
|
|
self._check_timestamp_monotonic(close, report)
|
|
self._check_no_duplicates(close, report)
|
|
self._check_no_nan_close(close, report)
|
|
self._check_data_gaps(close, report)
|
|
self._check_signal_alignment(entries, exits, close, report)
|
|
self._check_no_lookahead(entries, exits, close, report)
|
|
self._check_signal_density(entries, report)
|
|
self._check_no_coincident_signals(entries, exits, report)
|
|
self._check_min_trades(trades, pf, report)
|
|
self._check_fee_application(pf, report)
|
|
self._check_benchmark(close, benchmark_close, report)
|
|
|
|
self._compute_stats(entries, exits, close, pf, trades, report)
|
|
return report
|
|
|
|
# ── Individual checks ─────────────────────────────────────
|
|
|
|
def _check_timestamp_monotonic(self, close: pd.Series, report: ValidationReport):
|
|
ok = bool(close.index.is_monotonic_increasing)
|
|
report.checks["timestamps_monotonic"] = ok
|
|
if not ok:
|
|
report.errors.append("Timestamps are not monotonically increasing — data must be sorted")
|
|
|
|
def _check_no_duplicates(self, close: pd.Series, report: ValidationReport):
|
|
dupes = close.index.duplicated().sum()
|
|
ok = dupes == 0
|
|
report.checks["no_duplicate_timestamps"] = ok
|
|
report.details["duplicate_timestamps"] = dupes
|
|
if not ok:
|
|
report.errors.append(f"{dupes} duplicate timestamps found in index")
|
|
|
|
def _check_no_nan_close(self, close: pd.Series, report: ValidationReport):
|
|
nans = close.isna().sum()
|
|
ok = nans == 0
|
|
report.checks["no_nan_close"] = ok
|
|
report.details["nan_close_count"] = nans
|
|
if not ok:
|
|
report.errors.append(f"{nans} NaN values in close prices")
|
|
|
|
def _check_data_gaps(self, close: pd.Series, report: ValidationReport):
|
|
if self._expected_dt is None:
|
|
diffs = close.index.to_series().diff().dropna()
|
|
if len(diffs) > 0:
|
|
median_dt = diffs.dt.total_seconds().median()
|
|
else:
|
|
median_dt = 0
|
|
else:
|
|
median_dt = self._expected_dt
|
|
|
|
if median_dt <= 0:
|
|
report.checks["no_large_gaps"] = True
|
|
report.details["expected_interval_seconds"] = 0
|
|
return
|
|
|
|
diffs = close.index.to_series().diff().dropna()
|
|
large_gaps = (diffs.dt.total_seconds() > median_dt * 5).sum()
|
|
ok = large_gaps == 0
|
|
report.checks["no_large_gaps"] = ok
|
|
report.details["expected_interval_seconds"] = round(median_dt, 1)
|
|
report.details["large_gaps"] = int(large_gaps)
|
|
if not ok:
|
|
report.warnings.append(f"{large_gaps} gaps > 5x expected interval ({median_dt:.0f}s)")
|
|
|
|
def _check_signal_alignment(
|
|
self, entries: pd.Series, exits: pd.Series, close: pd.Series, report: ValidationReport
|
|
):
|
|
entry_ok = len(entries) == len(close)
|
|
exit_ok = len(exits) == len(close)
|
|
align = entry_ok and exit_ok
|
|
report.checks["signal_index_aligned"] = align
|
|
report.details["entry_len"] = len(entries)
|
|
report.details["exit_len"] = len(exits)
|
|
report.details["close_len"] = len(close)
|
|
if not align:
|
|
report.errors.append(
|
|
f"Signal/close length mismatch: entries={len(entries)} "
|
|
f"exits={len(exits)} close={len(close)}"
|
|
)
|
|
|
|
def _check_no_lookahead(
|
|
self, entries: pd.Series, exits: pd.Series, close: pd.Series, report: ValidationReport
|
|
):
|
|
if len(entries) < 2 or len(close) < 2:
|
|
report.checks["no_lookahead"] = True
|
|
return
|
|
|
|
first_signal_idx = -1
|
|
for i, v in enumerate(entries):
|
|
if v:
|
|
first_signal_idx = i
|
|
break
|
|
|
|
ok = first_signal_idx > 0 or first_signal_idx < 0
|
|
report.checks["no_lookahead"] = ok
|
|
report.details["first_signal_at_bar"] = first_signal_idx
|
|
if not ok:
|
|
report.errors.append("Signal found at bar 0 — possible lookahead bias")
|
|
|
|
overlap_signals = entries.iloc[:3].any() or exits.iloc[:3].any()
|
|
if overlap_signals:
|
|
early_entries = int(entries.iloc[:3].sum())
|
|
early_exits = int(exits.iloc[:3].sum())
|
|
if early_entries > 0:
|
|
report.warnings.append(
|
|
f"{early_entries} entry signals in first 3 bars — "
|
|
f"rolling indicators may not be warmed up"
|
|
)
|
|
|
|
def _check_signal_density(self, entries: pd.Series, report: ValidationReport):
|
|
n = max(len(entries), 1)
|
|
n_signals = int(entries.sum())
|
|
density = n_signals / n
|
|
|
|
if density > self._max_signal_density:
|
|
ok = False
|
|
report.warnings.append(
|
|
f"Signal density {density:.1%} exceeds {self._max_signal_density:.0%} "
|
|
f"— strategy may be overtrading"
|
|
)
|
|
elif density < self._min_signal_density and n_signals > 0:
|
|
ok = True
|
|
report.warnings.append(
|
|
f"Signal density {density:.1%} is very low — insufficient statistical power"
|
|
)
|
|
else:
|
|
ok = True
|
|
|
|
report.checks["signal_density_reasonable"] = ok
|
|
report.details["signal_density"] = round(density, 4)
|
|
report.details["total_signals"] = n_signals
|
|
|
|
def _check_no_coincident_signals(
|
|
self, entries: pd.Series, exits: pd.Series, report: ValidationReport
|
|
):
|
|
both = (entries & exits).sum()
|
|
ok = both == 0
|
|
report.checks["no_coincident_entry_exit"] = ok
|
|
report.details["coincident_signals"] = int(both)
|
|
if not ok:
|
|
report.errors.append(f"{both} bars have both entry and exit signals simultaneously")
|
|
|
|
def _check_min_trades(
|
|
self, trades: Optional[list[dict]], pf, report: ValidationReport
|
|
):
|
|
n_trades = 0
|
|
if trades is not None:
|
|
n_trades = len(trades)
|
|
elif pf is not None:
|
|
try:
|
|
n_trades = int(pf.trades.count())
|
|
except Exception:
|
|
n_trades = 0
|
|
|
|
ok = n_trades >= self._min_trades
|
|
report.checks["min_trade_count"] = ok
|
|
report.details["trade_count"] = n_trades
|
|
if not ok:
|
|
report.warnings.append(
|
|
f"Only {n_trades} trades (minimum {self._min_trades} required). "
|
|
f"Metrics like Sharpe are unreliable with <{self._min_trades} trades."
|
|
)
|
|
|
|
def _check_fee_application(self, pf, report: ValidationReport):
|
|
if pf is None:
|
|
report.checks["fees_applied"] = True
|
|
report.details["fee_check"] = "no portfolio object"
|
|
return
|
|
|
|
try:
|
|
value = pf.value().dropna()
|
|
if len(value) < 2:
|
|
report.checks["fees_applied"] = True
|
|
return
|
|
|
|
if hasattr(pf, 'get_filled_orders'):
|
|
gross_value = pf.asset_value().dropna()
|
|
else:
|
|
gross_value = value
|
|
|
|
ok = True
|
|
detail = "fees_applied"
|
|
if hasattr(pf, '_fees') or hasattr(pf, 'fees'):
|
|
detail = "fees_tracked"
|
|
except Exception:
|
|
ok = True
|
|
detail = "fee_check_unavailable"
|
|
|
|
report.checks["fees_applied"] = ok
|
|
report.details["fee_check"] = detail
|
|
|
|
def _check_benchmark(
|
|
self,
|
|
close: pd.Series,
|
|
benchmark_close: Optional[pd.Series],
|
|
report: ValidationReport,
|
|
):
|
|
if benchmark_close is None:
|
|
report.checks["benchmark_available"] = True
|
|
report.details["benchmark"] = "none_provided"
|
|
return
|
|
|
|
try:
|
|
aligned = benchmark_close.reindex(close.index).dropna()
|
|
if len(aligned) < 2:
|
|
report.checks["benchmark_available"] = True
|
|
report.details["benchmark"] = "insufficient_data"
|
|
return
|
|
|
|
bm_return = (aligned.iloc[-1] / aligned.iloc[0] - 1) * 100
|
|
report.checks["benchmark_available"] = True
|
|
report.details["benchmark_return_pct"] = round(bm_return, 2)
|
|
|
|
bm_rets = aligned.pct_change().dropna()
|
|
if len(bm_rets) > 1 and bm_rets.std() > 0:
|
|
bm_sharpe = float(bm_rets.mean() / bm_rets.std() * np.sqrt(365 * 24))
|
|
else:
|
|
bm_sharpe = 0.0
|
|
report.stats["benchmark_sharpe"] = round(bm_sharpe, 3)
|
|
except Exception:
|
|
report.checks["benchmark_available"] = True
|
|
report.details["benchmark"] = "computation_error"
|
|
|
|
def _compute_stats(
|
|
self,
|
|
entries: pd.Series,
|
|
exits: pd.Series,
|
|
close: pd.Series,
|
|
pf,
|
|
trades: Optional[list[dict]],
|
|
report: ValidationReport,
|
|
):
|
|
report.stats["n_bars"] = len(close)
|
|
report.stats["n_signals"] = int(entries.sum())
|
|
|
|
if len(close) > 1:
|
|
report.stats["start_date"] = str(close.index[0])[:19]
|
|
report.stats["end_date"] = str(close.index[-1])[:19]
|
|
|
|
if len(entries) > 1:
|
|
signal_gaps = np.diff(np.where(entries)[0]) if entries.sum() > 1 else np.array([])
|
|
if len(signal_gaps) > 0:
|
|
report.stats["avg_signal_interval"] = round(float(np.mean(signal_gaps)), 1)
|
|
report.stats["max_consecutive_signals"] = self._max_consecutive(entries.values)
|
|
|
|
if trades:
|
|
pnls = [float(t.get("pnl_net", t.get("pnl", 0))) for t in trades]
|
|
if pnls:
|
|
report.stats["total_gross_pnl"] = round(sum(
|
|
float(t.get("pnl_gross", t.get("pnl", 0))) for t in trades
|
|
), 4)
|
|
report.stats["total_fees"] = round(sum(
|
|
float(t.get("fee", 0)) for t in trades
|
|
), 4)
|
|
wins = sum(1 for p in pnls if p > 0)
|
|
report.stats["win_rate"] = round(wins / len(pnls), 3) if pnls else 0
|
|
|
|
if pf is not None:
|
|
try:
|
|
value = pf.value().dropna()
|
|
if len(value) > 1:
|
|
report.stats["final_equity"] = round(float(value.iloc[-1]), 2)
|
|
report.stats["max_drawdown_pct"] = round(
|
|
float((value.cummax() - value) / value.cummax()).max() * 100, 2
|
|
)
|
|
except Exception:
|
|
pass
|
|
|
|
@staticmethod
|
|
def _max_consecutive(arr) -> int:
|
|
"""Max consecutive True values in a boolean array."""
|
|
max_run = 0
|
|
current = 0
|
|
for v in arr:
|
|
if v:
|
|
current += 1
|
|
max_run = max(max_run, current)
|
|
else:
|
|
current = 0
|
|
return max_run
|