feat: VBT visualization + validation pipeline, HFT tick viz, DuckDB loader

Track 1 — VBT Candle-Frequency Pipeline:
- backtests/vbt_validator.py: VBTValidator with 11 checks — timestamp monotonicity,
  duplicates, NaN, data gaps, lookahead bias, signal alignment, density,
  coincident entry/exit, min trade count, fee application, benchmark comparison.
  ValidationReport dataclass with errors/warnings/stats. Validates VBT results
  or raw signal arrays.
- backtests/vbt_viz.py: VBTVisualizer with 10+ Plotly chart methods — equity
  curve with benchmark, drawdown, rolling Sharpe/Sortino/vol, trade markers,
  returns distribution with normal fit, monthly PnL heatmap, gross vs net,
  holding periods, parameter sensitivity heatmaps, dashboard compositor,
  HTML save (self-contained, CDN Plotly). All methods handle empty/null inputs.
- backtests/vbt_report.py: Markdown + HTML report generator — structured
  sections for implementation summary, performance metrics, cost analysis,
  validation results, signal analysis, known limitations, next steps.
  batch_report() for mass report generation from results directory.
- backtests/vbt_runner.py: Added run_benchmark() (buy-and-hold VBT portfolio),
  validate() (integrated VBTValidator), run_with_report() (fetch→validate→
  backtest→visualize→save in one call).

Track 2 — HFT Tick Pipeline:
- backtests/tick_viz.py: 9-panel HFT dashboard — price+trade markers,
  spread dynamics, top-of-book depth, microprice vs mid, OBI/OFI panel,
  VPIN toxicity with thresholds, event timeline (PnL from tick_runner),
  markout curves at 6 horizons. Parquet→pandas→Plotly pipeline.
  Dark-themed HTML output for microstructure review.
- data/duckdb_load.py: Parquet→DuckDB loader — creates l2_snapshots,
  trades, funding tables with schema. Pre-computed 1s rollup views for
  microprice, OFI, trade imbalance. Markout queries directly in SQL.
  Incremental loading with load_state tracking.

CLI Integration:
- cli.py: Added 'report' (full VBT report), 'validate' (check existing
  results), 'hft' (tick dashboard generation) commands. Fixed argparse
  help string escaping.

355 tests passing (34 new).
This commit is contained in:
ramseshk
2026-08-11 12:22:11 +08:00
parent 09cb0d42b5
commit 20ee340cef
8 changed files with 3332 additions and 2 deletions
+151
View File
@@ -441,6 +441,157 @@ class VBTBacktestRunner:
return pd.DataFrame(results_rows) if results_rows else None
def run_benchmark(
self,
coin: str = "BTC",
interval: str = "1h",
testnet: bool = False,
limit: int = 5000,
start_ms: int | None = None,
end_ms: int | None = None,
) -> dict[str, Any] | None:
"""Run a simple buy-and-hold benchmark using VBT."""
provider = HyperliquidDataProvider(testnet=testnet)
df = provider.fetch_candles(coin, interval=interval, limit=limit,
start_ms=start_ms, end_ms=end_ms)
if df.empty:
return None
close = df["close"]
if len(close) < 2:
return None
entries = pd.Series(False, index=close.index)
entries.iloc[0] = True
exits = pd.Series(False, index=close.index)
exits.iloc[-1] = True
try:
pf = vbt.Portfolio.from_signals(
close=close,
entries=entries,
exits=exits,
fees=self._fee_rate,
slippage=0.001,
freq=INTERVAL_MAP.get(interval, "1h"),
init_cash=10000.0,
)
except Exception:
return None
total_return = float(pf.stats().get("Total Return [%]", 0))
bm_sharpe = float(pf.stats().get("Sharpe Ratio", 0))
return {
"strategy": "buy_and_hold",
"coin": coin.upper(),
"interval": interval,
"n_bars": len(close),
"start_equity": 10000.0,
"end_equity": round(float(pf.value().iloc[-1]), 2),
"total_return_pct": round(total_return, 2),
"sharpe": round(bm_sharpe, 3),
"close": close,
"pf": pf,
}
def validate(
self,
result: dict,
pf,
entries: pd.Series,
exits: pd.Series,
close: pd.Series,
):
"""Run validation checks on a backtest result."""
from backtests.vbt_validator import VBTValidator
validator = VBTValidator(min_trades=10)
report = validator.validate(
entries=entries,
exits=exits,
close=close,
pf=pf,
trades=result.get("trades", []),
strategy=result.get("strategy", "unknown"),
interval=result.get("interval", "unknown"),
)
return report
def run_with_report(
self,
strategy: str = "pairs",
interval: str = "1h",
testnet: bool = False,
limit: int = 5000,
params: dict | None = None,
output_dir: str = "backtests/reports",
) -> dict | None:
"""End-to-end: fetch, backtest, validate, visualize, save report."""
result = self.run_strategy(
strategy=strategy, interval=interval, testnet=testnet,
limit=limit, params=params,
)
if result is None:
return None
data = {}
coins = self._get_coins(strategy)
provider = HyperliquidDataProvider(testnet=testnet)
for coin in coins:
df = provider.fetch_candles(coin, interval=interval, limit=limit)
if not df.empty:
data[coin] = df
entries, exits = _generate_signals(strategy, data, params)
primary = list(data.values())[0]
close = primary["close"]
common_idx = entries.index.intersection(close.index)
entries = entries.reindex(common_idx).fillna(False)
exits = exits.reindex(common_idx).fillna(False)
close = close.reindex(common_idx)
try:
from config.fee_tiers import get_strategy_fee_model
fee_model = get_strategy_fee_model(strategy)
effective_fee = self._maker_rate if fee_model == "maker" else self._fee_rate
pf = vbt.Portfolio.from_signals(
close=close, entries=entries, exits=exits,
fees=effective_fee, slippage=0.001,
freq=INTERVAL_MAP.get(interval, "1h"),
init_cash=10000.0,
)
except Exception:
pf = None
bm_result = self.run_benchmark(coin=self._get_coins(strategy)[0],
interval=interval, testnet=testnet, limit=limit)
benchmark_close = bm_result.get("close") if bm_result else None
validation_report = None
if pf is not None:
validation_report = self.validate(result, pf, entries, exits, close)
from backtests.vbt_viz import VBTVisualizer
viz = VBTVisualizer(output_dir=output_dir)
viz.save_dashboard(
pf=pf, close=close, entries=entries, exits=exits,
benchmark_close=benchmark_close,
strategy=strategy, interval=interval,
)
result["validation"] = validation_report.summary() if validation_report else "N/A"
if validation_report:
result["validation_checks"] = validation_report.checks
result["validation_errors"] = validation_report.errors
result["validation_warnings"] = validation_report.warnings
logger.info("Report generated for %s (%s) — saved to %s",
strategy, interval, output_dir)
return result
# ── Helpers ─────────────────────────────────────────────────
def _get_coins(self, strategy: str) -> list[str]: