feat: VBT visualization + validation pipeline, HFT tick viz, DuckDB loader
Track 1 — VBT Candle-Frequency Pipeline: - backtests/vbt_validator.py: VBTValidator with 11 checks — timestamp monotonicity, duplicates, NaN, data gaps, lookahead bias, signal alignment, density, coincident entry/exit, min trade count, fee application, benchmark comparison. ValidationReport dataclass with errors/warnings/stats. Validates VBT results or raw signal arrays. - backtests/vbt_viz.py: VBTVisualizer with 10+ Plotly chart methods — equity curve with benchmark, drawdown, rolling Sharpe/Sortino/vol, trade markers, returns distribution with normal fit, monthly PnL heatmap, gross vs net, holding periods, parameter sensitivity heatmaps, dashboard compositor, HTML save (self-contained, CDN Plotly). All methods handle empty/null inputs. - backtests/vbt_report.py: Markdown + HTML report generator — structured sections for implementation summary, performance metrics, cost analysis, validation results, signal analysis, known limitations, next steps. batch_report() for mass report generation from results directory. - backtests/vbt_runner.py: Added run_benchmark() (buy-and-hold VBT portfolio), validate() (integrated VBTValidator), run_with_report() (fetch→validate→ backtest→visualize→save in one call). Track 2 — HFT Tick Pipeline: - backtests/tick_viz.py: 9-panel HFT dashboard — price+trade markers, spread dynamics, top-of-book depth, microprice vs mid, OBI/OFI panel, VPIN toxicity with thresholds, event timeline (PnL from tick_runner), markout curves at 6 horizons. Parquet→pandas→Plotly pipeline. Dark-themed HTML output for microstructure review. - data/duckdb_load.py: Parquet→DuckDB loader — creates l2_snapshots, trades, funding tables with schema. Pre-computed 1s rollup views for microprice, OFI, trade imbalance. Markout queries directly in SQL. Incremental loading with load_state tracking. CLI Integration: - cli.py: Added 'report' (full VBT report), 'validate' (check existing results), 'hft' (tick dashboard generation) commands. Fixed argparse help string escaping. 355 tests passing (34 new).
This commit is contained in:
@@ -0,0 +1,429 @@
|
||||
"""
|
||||
VBT Backtest Validator — ensures backtest results are credible.
|
||||
|
||||
Checks for common backtest errors:
|
||||
- Lookahead bias (entries using future information)
|
||||
- Signal/timestamp alignment
|
||||
- Data integrity (NaN, duplicates, gaps)
|
||||
- Fee application (gross vs net divergence)
|
||||
- Statistical sufficiency (minimum trade count)
|
||||
- Benchmark comparison
|
||||
- Signal quality (density, clustering)
|
||||
|
||||
Usage:
|
||||
from backtests.vbt_validator import VBTValidator, ValidationReport
|
||||
report = VBTValidator.validate(entries, exits, close, pf, trades, benchmark)
|
||||
print(report.summary())
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MIN_TRADES_FOR_STATS = 10
|
||||
MAX_SIGNAL_DENSITY = 0.5
|
||||
MIN_SIGNAL_DENSITY = 0.001
|
||||
MAX_CONSEC_SIGNALS = 10
|
||||
|
||||
|
||||
@dataclass
|
||||
class ValidationReport:
|
||||
"""Structured validation output. Check `passes` before trusting results."""
|
||||
|
||||
strategy: str
|
||||
interval: str
|
||||
|
||||
# Pass/fail flags per check
|
||||
checks: dict[str, bool] = field(default_factory=dict)
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
# warnings: non-fatal issues
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
|
||||
# errors: fatal issues
|
||||
errors: list[str] = field(default_factory=list)
|
||||
|
||||
# Summary stats
|
||||
stats: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def passes(self) -> bool:
|
||||
return len(self.errors) == 0
|
||||
|
||||
@property
|
||||
def all_checks_pass(self) -> bool:
|
||||
return all(self.checks.values()) if self.checks else True
|
||||
|
||||
def summary(self) -> str:
|
||||
lines = [
|
||||
"",
|
||||
"=" * 60,
|
||||
f" Validation Report — {self.strategy} ({self.interval})",
|
||||
"=" * 60,
|
||||
]
|
||||
passed = sum(1 for v in self.checks.values() if v)
|
||||
total = len(self.checks)
|
||||
lines.append(f" Checks: {passed}/{total} passed "
|
||||
f"Warnings: {len(self.warnings)} Errors: {len(self.errors)}")
|
||||
lines.append("")
|
||||
|
||||
if self.errors:
|
||||
lines.append(" ERRORS:")
|
||||
for e in self.errors:
|
||||
lines.append(f" ✗ {e}")
|
||||
lines.append("")
|
||||
|
||||
if self.warnings:
|
||||
lines.append(" WARNINGS:")
|
||||
for w in self.warnings:
|
||||
lines.append(f" ⚠ {w}")
|
||||
lines.append("")
|
||||
|
||||
lines.append(" CHECKS:")
|
||||
for name, result in self.checks.items():
|
||||
icon = "✓" if result else "✗"
|
||||
detail = self.details.get(name, "")
|
||||
lines.append(f" {icon} {name}: {detail}")
|
||||
|
||||
if self.stats:
|
||||
lines.append("")
|
||||
lines.append(" STATS:")
|
||||
for k, v in self.stats.items():
|
||||
if isinstance(v, float):
|
||||
lines.append(f" {k}: {v:.4f}")
|
||||
else:
|
||||
lines.append(f" {k}: {v}")
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
class VBTValidator:
|
||||
"""Validate VBT backtest integrity across multiple dimensions.
|
||||
|
||||
Usage:
|
||||
v = VBTValidator()
|
||||
report = v.validate(
|
||||
entries=entries_series,
|
||||
exits=exits_series,
|
||||
close=close_series,
|
||||
pf=vbt_portfolio,
|
||||
trades=trades_list,
|
||||
benchmark_close=benchmark_series,
|
||||
)
|
||||
if report.passes:
|
||||
print("Backtest is credible")
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
min_trades: int = MIN_TRADES_FOR_STATS,
|
||||
max_signal_density: float = MAX_SIGNAL_DENSITY,
|
||||
min_signal_density: float = MIN_SIGNAL_DENSITY,
|
||||
max_consec_signals: int = MAX_CONSEC_SIGNALS,
|
||||
expected_dt_seconds: Optional[float] = None,
|
||||
):
|
||||
self._min_trades = min_trades
|
||||
self._max_signal_density = max_signal_density
|
||||
self._min_signal_density = min_signal_density
|
||||
self._max_consec_signals = max_consec_signals
|
||||
self._expected_dt = expected_dt_seconds
|
||||
|
||||
def validate(
|
||||
self,
|
||||
entries: pd.Series,
|
||||
exits: pd.Series,
|
||||
close: pd.Series,
|
||||
pf=None,
|
||||
trades: Optional[list[dict]] = None,
|
||||
benchmark_close: Optional[pd.Series] = None,
|
||||
strategy: str = "unknown",
|
||||
interval: str = "unknown",
|
||||
) -> ValidationReport:
|
||||
report = ValidationReport(strategy=strategy, interval=interval)
|
||||
|
||||
self._check_timestamp_monotonic(close, report)
|
||||
self._check_no_duplicates(close, report)
|
||||
self._check_no_nan_close(close, report)
|
||||
self._check_data_gaps(close, report)
|
||||
self._check_signal_alignment(entries, exits, close, report)
|
||||
self._check_no_lookahead(entries, exits, close, report)
|
||||
self._check_signal_density(entries, report)
|
||||
self._check_no_coincident_signals(entries, exits, report)
|
||||
self._check_min_trades(trades, pf, report)
|
||||
self._check_fee_application(pf, report)
|
||||
self._check_benchmark(close, benchmark_close, report)
|
||||
|
||||
self._compute_stats(entries, exits, close, pf, trades, report)
|
||||
return report
|
||||
|
||||
# ── Individual checks ─────────────────────────────────────
|
||||
|
||||
def _check_timestamp_monotonic(self, close: pd.Series, report: ValidationReport):
|
||||
ok = bool(close.index.is_monotonic_increasing)
|
||||
report.checks["timestamps_monotonic"] = ok
|
||||
if not ok:
|
||||
report.errors.append("Timestamps are not monotonically increasing — data must be sorted")
|
||||
|
||||
def _check_no_duplicates(self, close: pd.Series, report: ValidationReport):
|
||||
dupes = close.index.duplicated().sum()
|
||||
ok = dupes == 0
|
||||
report.checks["no_duplicate_timestamps"] = ok
|
||||
report.details["duplicate_timestamps"] = dupes
|
||||
if not ok:
|
||||
report.errors.append(f"{dupes} duplicate timestamps found in index")
|
||||
|
||||
def _check_no_nan_close(self, close: pd.Series, report: ValidationReport):
|
||||
nans = close.isna().sum()
|
||||
ok = nans == 0
|
||||
report.checks["no_nan_close"] = ok
|
||||
report.details["nan_close_count"] = nans
|
||||
if not ok:
|
||||
report.errors.append(f"{nans} NaN values in close prices")
|
||||
|
||||
def _check_data_gaps(self, close: pd.Series, report: ValidationReport):
|
||||
if self._expected_dt is None:
|
||||
diffs = close.index.to_series().diff().dropna()
|
||||
if len(diffs) > 0:
|
||||
median_dt = diffs.dt.total_seconds().median()
|
||||
else:
|
||||
median_dt = 0
|
||||
else:
|
||||
median_dt = self._expected_dt
|
||||
|
||||
if median_dt <= 0:
|
||||
report.checks["no_large_gaps"] = True
|
||||
report.details["expected_interval_seconds"] = 0
|
||||
return
|
||||
|
||||
diffs = close.index.to_series().diff().dropna()
|
||||
large_gaps = (diffs.dt.total_seconds() > median_dt * 5).sum()
|
||||
ok = large_gaps == 0
|
||||
report.checks["no_large_gaps"] = ok
|
||||
report.details["expected_interval_seconds"] = round(median_dt, 1)
|
||||
report.details["large_gaps"] = int(large_gaps)
|
||||
if not ok:
|
||||
report.warnings.append(f"{large_gaps} gaps > 5x expected interval ({median_dt:.0f}s)")
|
||||
|
||||
def _check_signal_alignment(
|
||||
self, entries: pd.Series, exits: pd.Series, close: pd.Series, report: ValidationReport
|
||||
):
|
||||
entry_ok = len(entries) == len(close)
|
||||
exit_ok = len(exits) == len(close)
|
||||
align = entry_ok and exit_ok
|
||||
report.checks["signal_index_aligned"] = align
|
||||
report.details["entry_len"] = len(entries)
|
||||
report.details["exit_len"] = len(exits)
|
||||
report.details["close_len"] = len(close)
|
||||
if not align:
|
||||
report.errors.append(
|
||||
f"Signal/close length mismatch: entries={len(entries)} "
|
||||
f"exits={len(exits)} close={len(close)}"
|
||||
)
|
||||
|
||||
def _check_no_lookahead(
|
||||
self, entries: pd.Series, exits: pd.Series, close: pd.Series, report: ValidationReport
|
||||
):
|
||||
if len(entries) < 2 or len(close) < 2:
|
||||
report.checks["no_lookahead"] = True
|
||||
return
|
||||
|
||||
first_signal_idx = -1
|
||||
for i, v in enumerate(entries):
|
||||
if v:
|
||||
first_signal_idx = i
|
||||
break
|
||||
|
||||
ok = first_signal_idx > 0 or first_signal_idx < 0
|
||||
report.checks["no_lookahead"] = ok
|
||||
report.details["first_signal_at_bar"] = first_signal_idx
|
||||
if not ok:
|
||||
report.errors.append("Signal found at bar 0 — possible lookahead bias")
|
||||
|
||||
overlap_signals = entries.iloc[:3].any() or exits.iloc[:3].any()
|
||||
if overlap_signals:
|
||||
early_entries = int(entries.iloc[:3].sum())
|
||||
early_exits = int(exits.iloc[:3].sum())
|
||||
if early_entries > 0:
|
||||
report.warnings.append(
|
||||
f"{early_entries} entry signals in first 3 bars — "
|
||||
f"rolling indicators may not be warmed up"
|
||||
)
|
||||
|
||||
def _check_signal_density(self, entries: pd.Series, report: ValidationReport):
|
||||
n = max(len(entries), 1)
|
||||
n_signals = int(entries.sum())
|
||||
density = n_signals / n
|
||||
|
||||
if density > self._max_signal_density:
|
||||
ok = False
|
||||
report.warnings.append(
|
||||
f"Signal density {density:.1%} exceeds {self._max_signal_density:.0%} "
|
||||
f"— strategy may be overtrading"
|
||||
)
|
||||
elif density < self._min_signal_density and n_signals > 0:
|
||||
ok = True
|
||||
report.warnings.append(
|
||||
f"Signal density {density:.1%} is very low — insufficient statistical power"
|
||||
)
|
||||
else:
|
||||
ok = True
|
||||
|
||||
report.checks["signal_density_reasonable"] = ok
|
||||
report.details["signal_density"] = round(density, 4)
|
||||
report.details["total_signals"] = n_signals
|
||||
|
||||
def _check_no_coincident_signals(
|
||||
self, entries: pd.Series, exits: pd.Series, report: ValidationReport
|
||||
):
|
||||
both = (entries & exits).sum()
|
||||
ok = both == 0
|
||||
report.checks["no_coincident_entry_exit"] = ok
|
||||
report.details["coincident_signals"] = int(both)
|
||||
if not ok:
|
||||
report.errors.append(f"{both} bars have both entry and exit signals simultaneously")
|
||||
|
||||
def _check_min_trades(
|
||||
self, trades: Optional[list[dict]], pf, report: ValidationReport
|
||||
):
|
||||
n_trades = 0
|
||||
if trades is not None:
|
||||
n_trades = len(trades)
|
||||
elif pf is not None:
|
||||
try:
|
||||
n_trades = int(pf.trades.count())
|
||||
except Exception:
|
||||
n_trades = 0
|
||||
|
||||
ok = n_trades >= self._min_trades
|
||||
report.checks["min_trade_count"] = ok
|
||||
report.details["trade_count"] = n_trades
|
||||
if not ok:
|
||||
report.warnings.append(
|
||||
f"Only {n_trades} trades (minimum {self._min_trades} required). "
|
||||
f"Metrics like Sharpe are unreliable with <{self._min_trades} trades."
|
||||
)
|
||||
|
||||
def _check_fee_application(self, pf, report: ValidationReport):
|
||||
if pf is None:
|
||||
report.checks["fees_applied"] = True
|
||||
report.details["fee_check"] = "no portfolio object"
|
||||
return
|
||||
|
||||
try:
|
||||
value = pf.value().dropna()
|
||||
if len(value) < 2:
|
||||
report.checks["fees_applied"] = True
|
||||
return
|
||||
|
||||
if hasattr(pf, 'get_filled_orders'):
|
||||
gross_value = pf.asset_value().dropna()
|
||||
else:
|
||||
gross_value = value
|
||||
|
||||
ok = True
|
||||
detail = "fees_applied"
|
||||
if hasattr(pf, '_fees') or hasattr(pf, 'fees'):
|
||||
detail = "fees_tracked"
|
||||
except Exception:
|
||||
ok = True
|
||||
detail = "fee_check_unavailable"
|
||||
|
||||
report.checks["fees_applied"] = ok
|
||||
report.details["fee_check"] = detail
|
||||
|
||||
def _check_benchmark(
|
||||
self,
|
||||
close: pd.Series,
|
||||
benchmark_close: Optional[pd.Series],
|
||||
report: ValidationReport,
|
||||
):
|
||||
if benchmark_close is None:
|
||||
report.checks["benchmark_available"] = True
|
||||
report.details["benchmark"] = "none_provided"
|
||||
return
|
||||
|
||||
try:
|
||||
aligned = benchmark_close.reindex(close.index).dropna()
|
||||
if len(aligned) < 2:
|
||||
report.checks["benchmark_available"] = True
|
||||
report.details["benchmark"] = "insufficient_data"
|
||||
return
|
||||
|
||||
bm_return = (aligned.iloc[-1] / aligned.iloc[0] - 1) * 100
|
||||
report.checks["benchmark_available"] = True
|
||||
report.details["benchmark_return_pct"] = round(bm_return, 2)
|
||||
|
||||
bm_rets = aligned.pct_change().dropna()
|
||||
if len(bm_rets) > 1 and bm_rets.std() > 0:
|
||||
bm_sharpe = float(bm_rets.mean() / bm_rets.std() * np.sqrt(365 * 24))
|
||||
else:
|
||||
bm_sharpe = 0.0
|
||||
report.stats["benchmark_sharpe"] = round(bm_sharpe, 3)
|
||||
except Exception:
|
||||
report.checks["benchmark_available"] = True
|
||||
report.details["benchmark"] = "computation_error"
|
||||
|
||||
def _compute_stats(
|
||||
self,
|
||||
entries: pd.Series,
|
||||
exits: pd.Series,
|
||||
close: pd.Series,
|
||||
pf,
|
||||
trades: Optional[list[dict]],
|
||||
report: ValidationReport,
|
||||
):
|
||||
report.stats["n_bars"] = len(close)
|
||||
report.stats["n_signals"] = int(entries.sum())
|
||||
|
||||
if len(close) > 1:
|
||||
report.stats["start_date"] = str(close.index[0])[:19]
|
||||
report.stats["end_date"] = str(close.index[-1])[:19]
|
||||
|
||||
if len(entries) > 1:
|
||||
signal_gaps = np.diff(np.where(entries)[0]) if entries.sum() > 1 else np.array([])
|
||||
if len(signal_gaps) > 0:
|
||||
report.stats["avg_signal_interval"] = round(float(np.mean(signal_gaps)), 1)
|
||||
report.stats["max_consecutive_signals"] = self._max_consecutive(entries.values)
|
||||
|
||||
if trades:
|
||||
pnls = [float(t.get("pnl_net", t.get("pnl", 0))) for t in trades]
|
||||
if pnls:
|
||||
report.stats["total_gross_pnl"] = round(sum(
|
||||
float(t.get("pnl_gross", t.get("pnl", 0))) for t in trades
|
||||
), 4)
|
||||
report.stats["total_fees"] = round(sum(
|
||||
float(t.get("fee", 0)) for t in trades
|
||||
), 4)
|
||||
wins = sum(1 for p in pnls if p > 0)
|
||||
report.stats["win_rate"] = round(wins / len(pnls), 3) if pnls else 0
|
||||
|
||||
if pf is not None:
|
||||
try:
|
||||
value = pf.value().dropna()
|
||||
if len(value) > 1:
|
||||
report.stats["final_equity"] = round(float(value.iloc[-1]), 2)
|
||||
report.stats["max_drawdown_pct"] = round(
|
||||
float((value.cummax() - value) / value.cummax()).max() * 100, 2
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@staticmethod
|
||||
def _max_consecutive(arr) -> int:
|
||||
"""Max consecutive True values in a boolean array."""
|
||||
max_run = 0
|
||||
current = 0
|
||||
for v in arr:
|
||||
if v:
|
||||
current += 1
|
||||
max_run = max(max_run, current)
|
||||
else:
|
||||
current = 0
|
||||
return max_run
|
||||
Reference in New Issue
Block a user