feat: VBT visualization + validation pipeline, HFT tick viz, DuckDB loader
Track 1 — VBT Candle-Frequency Pipeline: - backtests/vbt_validator.py: VBTValidator with 11 checks — timestamp monotonicity, duplicates, NaN, data gaps, lookahead bias, signal alignment, density, coincident entry/exit, min trade count, fee application, benchmark comparison. ValidationReport dataclass with errors/warnings/stats. Validates VBT results or raw signal arrays. - backtests/vbt_viz.py: VBTVisualizer with 10+ Plotly chart methods — equity curve with benchmark, drawdown, rolling Sharpe/Sortino/vol, trade markers, returns distribution with normal fit, monthly PnL heatmap, gross vs net, holding periods, parameter sensitivity heatmaps, dashboard compositor, HTML save (self-contained, CDN Plotly). All methods handle empty/null inputs. - backtests/vbt_report.py: Markdown + HTML report generator — structured sections for implementation summary, performance metrics, cost analysis, validation results, signal analysis, known limitations, next steps. batch_report() for mass report generation from results directory. - backtests/vbt_runner.py: Added run_benchmark() (buy-and-hold VBT portfolio), validate() (integrated VBTValidator), run_with_report() (fetch→validate→ backtest→visualize→save in one call). Track 2 — HFT Tick Pipeline: - backtests/tick_viz.py: 9-panel HFT dashboard — price+trade markers, spread dynamics, top-of-book depth, microprice vs mid, OBI/OFI panel, VPIN toxicity with thresholds, event timeline (PnL from tick_runner), markout curves at 6 horizons. Parquet→pandas→Plotly pipeline. Dark-themed HTML output for microstructure review. - data/duckdb_load.py: Parquet→DuckDB loader — creates l2_snapshots, trades, funding tables with schema. Pre-computed 1s rollup views for microprice, OFI, trade imbalance. Markout queries directly in SQL. Incremental loading with load_state tracking. CLI Integration: - cli.py: Added 'report' (full VBT report), 'validate' (check existing results), 'hft' (tick dashboard generation) commands. Fixed argparse help string escaping. 355 tests passing (34 new).
This commit is contained in:
@@ -0,0 +1,347 @@
|
||||
"""
|
||||
VBT Report Generator — produces Markdown and HTML research reports.
|
||||
|
||||
Consolidates backtest results, validation reports, performance metrics,
|
||||
and visualizations into a single shareable document.
|
||||
|
||||
Usage:
|
||||
python -m backtests.vbt_report --strategy pairs --interval 1h
|
||||
python -m backtests.vbt_report --strategy all --output reports/weekly.md
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import numpy as np
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
REPORT_DIR = Path(__file__).resolve().parent / "reports"
|
||||
REPORT_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
|
||||
def format_metric(value: Any, decimals: int = 2) -> str:
|
||||
if isinstance(value, float):
|
||||
return f"{value:.{decimals}f}"
|
||||
return str(value)
|
||||
|
||||
|
||||
def generate_markdown_report(
|
||||
result: dict,
|
||||
validation_report=None,
|
||||
includes_viz: bool = False,
|
||||
viz_path: str = "",
|
||||
) -> str:
|
||||
"""Generate a Markdown research report from a backtest result."""
|
||||
|
||||
strategy = result.get("strategy", "unknown")
|
||||
interval = result.get("interval", "unknown")
|
||||
pnl = result.get("pnl", 0)
|
||||
total_return = result.get("total_return_pct", 0)
|
||||
sharpe = result.get("sharpe", 0)
|
||||
sortino = result.get("sortino", 0)
|
||||
max_dd = result.get("max_drawdown_pct", 0)
|
||||
win_rate = result.get("win_rate", 0)
|
||||
profit_factor = result.get("profit_factor", 0)
|
||||
expectancy = result.get("expectancy", 0)
|
||||
n_bars = result.get("n_bars", 0)
|
||||
trades = result.get("trades", [])
|
||||
n_trades = result.get("total_trades", len(trades))
|
||||
params = result.get("params", {})
|
||||
fee_info = result.get("fee_info", {})
|
||||
timing = result.get("generated_at", datetime.now(timezone.utc).isoformat())
|
||||
|
||||
lines = []
|
||||
|
||||
lines.append(f"# VBT Backtest Report — {strategy} ({interval})")
|
||||
lines.append("")
|
||||
lines.append(f"**Generated:** {timing}")
|
||||
lines.append(f"**Data:** Hyperliquid {'mainnet' if 'mainnet' in str(result.get('coin', '')) else 'testnet'}")
|
||||
lines.append("")
|
||||
|
||||
lines.append("---")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 1. Implementation Summary")
|
||||
lines.append("")
|
||||
lines.append(f"- **Strategy:** `{strategy}`")
|
||||
lines.append(f"- **Interval:** `{interval}`")
|
||||
lines.append(f"- **Bars:** {n_bars}")
|
||||
lines.append(f"- **Trades:** {n_trades}")
|
||||
lines.append("")
|
||||
|
||||
strategy_type = params.get("type", "Unknown")
|
||||
lines.append(f"- **Strategy Type:** {strategy_type}")
|
||||
if params:
|
||||
param_str = ", ".join(f"{k}={v}" for k, v in params.items() if k != "type")
|
||||
lines.append(f"- **Parameters:** {param_str}")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 2. Performance Metrics")
|
||||
lines.append("")
|
||||
lines.append("| Metric | Value |")
|
||||
lines.append("|--------|-------|")
|
||||
lines.append(f"| Start Equity | ${result.get('start_equity', 10000.0):,.2f} |")
|
||||
lines.append(f"| End Equity | ${result.get('end_equity', 10000.0):,.2f} |")
|
||||
lines.append(f"| Net PnL | ${pnl:,.2f} |")
|
||||
lines.append(f"| Total Return | {total_return:.2f}% |")
|
||||
lines.append(f"| Sharpe Ratio | {sharpe:.3f} |")
|
||||
lines.append(f"| Sortino Ratio | {sortino:.3f} |")
|
||||
lines.append(f"| Max Drawdown | {max_dd:.2f}% |")
|
||||
lines.append(f"| Win Rate | {win_rate:.1%} |")
|
||||
lines.append(f"| Profit Factor | {profit_factor:.3f} |")
|
||||
lines.append(f"| Expectancy | {expectancy:.3f} |")
|
||||
lines.append(f"| Total Trades | {n_trades} |")
|
||||
|
||||
if fee_info:
|
||||
lines.append(f"| Fee Rate | {fee_info.get('effective_rate_pct', 0):.4f}% |")
|
||||
lines.append(f"| Fee Tier | {fee_info.get('tier_name', 'N/A')} |")
|
||||
lines.append(f"| Staking Tier | {fee_info.get('staking_name', 'N/A')} |")
|
||||
lines.append(f"| Fee Model | {fee_info.get('fee_model', 'N/A')} |")
|
||||
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 3. Cost Analysis")
|
||||
lines.append("")
|
||||
|
||||
if trades:
|
||||
gross_pnls = [float(t.get("pnl_gross", t.get("pnl", 0))) for t in trades]
|
||||
net_pnls = [float(t.get("pnl_net", t.get("pnl", 0))) for t in trades]
|
||||
fees = [float(t.get("fee", 0)) for t in trades]
|
||||
total_gross = sum(gross_pnls)
|
||||
total_net = sum(net_pnls)
|
||||
total_fees = sum(fees)
|
||||
slippage_est = n_trades * 0.001 * 10000.0 * 0.001
|
||||
|
||||
lines.append("| Component | Amount |")
|
||||
lines.append("|-----------|--------|")
|
||||
lines.append(f"| Gross PnL | ${total_gross:,.4f} |")
|
||||
lines.append(f"| Total Fees | ${total_fees:,.4f} |")
|
||||
lines.append(f"| Est. Slippage | ${slippage_est:,.4f} |")
|
||||
lines.append(f"| Net PnL | ${total_net:,.4f} |")
|
||||
|
||||
cost_pct = (total_fees / abs(total_gross) * 100) if abs(total_gross) > 0 else 0
|
||||
lines.append(f"| Fee/Gross Ratio | {cost_pct:.1f}% |")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 4. Validation Results")
|
||||
lines.append("")
|
||||
|
||||
if validation_report:
|
||||
if hasattr(validation_report, 'summary'):
|
||||
lines.append("```")
|
||||
lines.append(validation_report.summary())
|
||||
lines.append("```")
|
||||
else:
|
||||
lines.append("```")
|
||||
lines.append(str(validation_report))
|
||||
lines.append("```")
|
||||
else:
|
||||
lines.append("⚠ No validation report available.")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 5. Signal Analysis")
|
||||
lines.append("")
|
||||
lines.append(f"- **Total signals:** {result.get('n_bars', 0)} bars processed")
|
||||
|
||||
if trades:
|
||||
holds = []
|
||||
for t in trades:
|
||||
dur = str(t.get("duration", ""))
|
||||
if dur:
|
||||
try:
|
||||
td = pd_from_timedelta(dur)
|
||||
if td:
|
||||
holds.append(td.total_seconds() / 3600)
|
||||
except Exception:
|
||||
pass
|
||||
if holds:
|
||||
lines.append(f"- **Avg holding period:** {np.mean(holds):.2f} hours")
|
||||
lines.append(f"- **Median holding period:** {np.median(holds):.2f} hours")
|
||||
lines.append(f"- **Max holding period:** {np.max(holds):.2f} hours")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 6. Known Limitations")
|
||||
lines.append("")
|
||||
lines.append("1. **VBT is candle-level backtesting only.** It cannot model:")
|
||||
lines.append(" - Queue position / price-time priority")
|
||||
lines.append(" - Realistic adverse selection at tick-level")
|
||||
lines.append(" - Latency-dependent fill probability")
|
||||
lines.append(" - VPIN-gated market making")
|
||||
lines.append("2. **Volume-based OBI is a proxy.** Real OBI requires L2 order book data.")
|
||||
lines.append("3. **A-S MM simulation is synthetic.** Uses candle high/low as virtual orderbook,")
|
||||
lines.append(" not real exchange order book queue position.")
|
||||
lines.append(f"4. **Signal frequency:** {n_trades} trades in {n_bars} bars — "
|
||||
f"this is a {'scalping' if n_bars > 0 and n_trades / n_bars > 0.01 else 'low-frequency'} strategy.")
|
||||
lines.append("5. **No walk-forward validation** performed in this report. "
|
||||
"Run `python -m cli walkforward --strategy {strategy}` for OOS testing.")
|
||||
lines.append("")
|
||||
|
||||
lines.append("## 7. Next Steps")
|
||||
lines.append("")
|
||||
lines.append(f"1. Run walk-forward validation: `python -m cli walkforward --strategy {strategy} --interval {interval}`")
|
||||
lines.append(f"2. Run tick-level backtest: `python -m cli tick --maker vpin_as_mm --coin BTC`")
|
||||
lines.append(f"3. Paper trade for 7+ days before live deployment")
|
||||
lines.append(f"4. Correlate strategy with other strategies to build diversified portfolio")
|
||||
|
||||
if includes_viz and viz_path:
|
||||
lines.append("")
|
||||
lines.append("## 8. Visualizations")
|
||||
lines.append("")
|
||||
lines.append(f"Interactive dashboard: [{viz_path}]({viz_path})")
|
||||
|
||||
lines.append("")
|
||||
lines.append("---")
|
||||
lines.append(f"*Generated by FTDT Quant Lab VBT Pipeline*")
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def pd_from_timedelta(dur_str: str):
|
||||
"""Safe Timedelta parsing."""
|
||||
try:
|
||||
import pandas as pd
|
||||
return pd.Timedelta(dur_str)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def generate_html_report(
|
||||
result: dict,
|
||||
validation_report=None,
|
||||
viz_path: str = "",
|
||||
) -> str:
|
||||
"""Wrap the Markdown report in HTML with styling."""
|
||||
md_body = generate_markdown_report(result, validation_report, bool(viz_path), viz_path)
|
||||
|
||||
try:
|
||||
import markdown
|
||||
body = markdown.markdown(md_body, extensions=["tables", "fenced_code"])
|
||||
except ImportError:
|
||||
body = "<pre>" + md_body.replace("<", "<") + "</pre>"
|
||||
|
||||
html = f"""<!DOCTYPE html>
|
||||
<html>
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>VBT Backtest Report</title>
|
||||
<style>
|
||||
body {{ font-family: system-ui, -apple-system, sans-serif; max-width: 900px;
|
||||
margin: 0 auto; padding: 40px 20px; color: #333; line-height: 1.6; }}
|
||||
h1, h2 {{ color: #1a1a2e; border-bottom: 1px solid #eee; padding-bottom: 8px; }}
|
||||
table {{ border-collapse: collapse; width: 100%; margin: 12px 0; }}
|
||||
th, td {{ border: 1px solid #ddd; padding: 8px 12px; text-align: left; }}
|
||||
th {{ background: #f0f0f0; }}
|
||||
code, pre {{ background: #f5f5f5; border-radius: 4px; }}
|
||||
pre {{ padding: 12px; overflow-x: auto; }}
|
||||
.warn {{ color: #c0392b; font-weight: bold; }}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
{body}
|
||||
</body>
|
||||
</html>"""
|
||||
|
||||
return html
|
||||
|
||||
|
||||
def save_report(
|
||||
result: dict,
|
||||
validation_report=None,
|
||||
viz_path: str = "",
|
||||
output_dir: str = "",
|
||||
fmt: str = "md",
|
||||
) -> str:
|
||||
"""Save a report to disk. Returns filepath."""
|
||||
out = Path(output_dir) if output_dir else REPORT_DIR
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
strategy = result.get("strategy", "unknown")
|
||||
interval = result.get("interval", "unknown")
|
||||
ts = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
|
||||
base = f"{strategy}_{interval}_{ts}"
|
||||
|
||||
if fmt == "html":
|
||||
content = generate_html_report(result, validation_report, viz_path)
|
||||
ext = ".html"
|
||||
else:
|
||||
content = generate_markdown_report(result, validation_report, bool(viz_path), viz_path)
|
||||
ext = ".md"
|
||||
|
||||
fpath = out / (base + ext)
|
||||
fpath.write_text(content)
|
||||
logger.info("Report saved to %s", fpath)
|
||||
return str(fpath)
|
||||
|
||||
|
||||
def batch_report(
|
||||
results_dir: str = "backtests/results",
|
||||
output_dir: str = "backtests/reports",
|
||||
) -> list[str]:
|
||||
"""Generate reports for all recent backtest results."""
|
||||
import json as _json
|
||||
rd = Path(results_dir)
|
||||
files = sorted(rd.glob("*.json"), key=os.path.getmtime, reverse=True)
|
||||
paths = []
|
||||
|
||||
by_strategy = {}
|
||||
for fp in files[:30]:
|
||||
try:
|
||||
data = _json.loads(fp.read_text())
|
||||
strat = data.get("strategy", "?")
|
||||
if strat not in by_strategy:
|
||||
by_strategy[strat] = data
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
for strat, result in by_strategy.items():
|
||||
p = save_report(result, output_dir=output_dir)
|
||||
paths.append(p)
|
||||
|
||||
return paths
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
p = argparse.ArgumentParser(description="VBT Report Generator")
|
||||
p.add_argument("--strategy", default="pairs")
|
||||
p.add_argument("--interval", default="1h")
|
||||
p.add_argument("--results-dir", default="backtests/results")
|
||||
p.add_argument("--output-dir", default="backtests/reports")
|
||||
p.add_argument("--format", default="md", choices=["md", "html"])
|
||||
p.add_argument("--batch", action="store_true", help="Generate reports for all recent backtests")
|
||||
args = p.parse_args()
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s", datefmt="%H:%M:%S")
|
||||
|
||||
if args.batch:
|
||||
paths = batch_report(args.results_dir, args.output_dir)
|
||||
print(f"Generated {len(paths)} reports")
|
||||
else:
|
||||
import json as _json
|
||||
rd = Path(args.results_dir)
|
||||
files = sorted(rd.glob("*.json"))
|
||||
found = None
|
||||
for fp in files:
|
||||
try:
|
||||
d = _json.loads(fp.read_text())
|
||||
if d.get("strategy") == args.strategy and d.get("interval") == args.interval:
|
||||
found = d
|
||||
break
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if found:
|
||||
p = save_report(found, output_dir=args.output_dir, fmt=args.format)
|
||||
print(f"Report: {p}")
|
||||
else:
|
||||
print(f"No result found for {args.strategy} ({args.interval})")
|
||||
Reference in New Issue
Block a user