20ee340cef
Track 1 — VBT Candle-Frequency Pipeline: - backtests/vbt_validator.py: VBTValidator with 11 checks — timestamp monotonicity, duplicates, NaN, data gaps, lookahead bias, signal alignment, density, coincident entry/exit, min trade count, fee application, benchmark comparison. ValidationReport dataclass with errors/warnings/stats. Validates VBT results or raw signal arrays. - backtests/vbt_viz.py: VBTVisualizer with 10+ Plotly chart methods — equity curve with benchmark, drawdown, rolling Sharpe/Sortino/vol, trade markers, returns distribution with normal fit, monthly PnL heatmap, gross vs net, holding periods, parameter sensitivity heatmaps, dashboard compositor, HTML save (self-contained, CDN Plotly). All methods handle empty/null inputs. - backtests/vbt_report.py: Markdown + HTML report generator — structured sections for implementation summary, performance metrics, cost analysis, validation results, signal analysis, known limitations, next steps. batch_report() for mass report generation from results directory. - backtests/vbt_runner.py: Added run_benchmark() (buy-and-hold VBT portfolio), validate() (integrated VBTValidator), run_with_report() (fetch→validate→ backtest→visualize→save in one call). Track 2 — HFT Tick Pipeline: - backtests/tick_viz.py: 9-panel HFT dashboard — price+trade markers, spread dynamics, top-of-book depth, microprice vs mid, OBI/OFI panel, VPIN toxicity with thresholds, event timeline (PnL from tick_runner), markout curves at 6 horizons. Parquet→pandas→Plotly pipeline. Dark-themed HTML output for microstructure review. - data/duckdb_load.py: Parquet→DuckDB loader — creates l2_snapshots, trades, funding tables with schema. Pre-computed 1s rollup views for microprice, OFI, trade imbalance. Markout queries directly in SQL. Incremental loading with load_state tracking. CLI Integration: - cli.py: Added 'report' (full VBT report), 'validate' (check existing results), 'hft' (tick dashboard generation) commands. Fixed argparse help string escaping. 355 tests passing (34 new).
781 lines
29 KiB
Python
781 lines
29 KiB
Python
"""
|
|
FTDT Quant Lab — unified CLI.
|
|
|
|
Subcommands:
|
|
collect — Run data collector (streams to Parquet)
|
|
analyze — Run analytics on stored data (Phase 2)
|
|
simulate — Run market-making simulator on stored data (Phase 3)
|
|
run — Start production trading node (Phase 4)
|
|
backtest — Run VectorBT backtest (existing)
|
|
|
|
Usage:
|
|
python -m cli collect --coins BTC,ETH --data-dir data/raw
|
|
python -m cli analyze --data-dir data/raw --start 2026-08-01 --end 2026-08-07
|
|
python -m cli simulate --data-dir data/raw --coin BTC --hours 24
|
|
python -m cli run --coins BTC,ETH --mode paper
|
|
python -m cli backtest --strategy pairs --interval 1h
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
import os
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
|
|
|
|
def cmd_collect(args):
|
|
"""Run the Hyperliquid data collector."""
|
|
from data.collectors.hyperliquid import HyperliquidCollector
|
|
from data.store import RawMessageStore
|
|
|
|
store = RawMessageStore(
|
|
data_dir=args.data_dir,
|
|
flush_interval_sec=args.flush_interval,
|
|
)
|
|
|
|
collector = HyperliquidCollector(
|
|
store=store,
|
|
coins=args.coins,
|
|
testnet=not args.mainnet,
|
|
poll_interval_sec=args.poll_interval,
|
|
)
|
|
|
|
asyncio.run(collector.run())
|
|
|
|
|
|
def cmd_analyze(args):
|
|
"""Run microstructure analytics on stored data."""
|
|
from data.store import read_range
|
|
|
|
print(f"Reading {args.channel}/{args.coin} from {args.start_date} to {args.end_date}...")
|
|
messages = read_range(
|
|
args.data_dir,
|
|
channel=args.channel,
|
|
coin=args.coin.upper(),
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
print(f"Loaded {len(messages)} messages")
|
|
|
|
if args.channel == "l2book":
|
|
from microstructure.book import batch_book_stats
|
|
snapshots = []
|
|
for msg in messages:
|
|
payload = msg["payload"]
|
|
levels = payload.get("levels", [])
|
|
if levels and isinstance(levels, list) and len(levels) >= 2:
|
|
bids = {}
|
|
asks = {}
|
|
for bid in levels[0]:
|
|
if float(bid.get("sz", 0)) > 0:
|
|
bids[float(bid["px"])] = float(bid["sz"])
|
|
for ask in levels[1]:
|
|
if float(ask.get("sz", 0)) > 0:
|
|
asks[float(ask["px"])] = float(ask["sz"])
|
|
snapshots.append({"bids": bids, "asks": asks})
|
|
|
|
stats = batch_book_stats(snapshots)
|
|
print(json.dumps(stats, indent=2, default=str))
|
|
|
|
elif args.channel == "trades":
|
|
from microstructure.trades import classify_bulk_lee_ready, trade_arrival_rate, trade_volume_profile
|
|
|
|
trades = [msg["payload"] for msg in messages]
|
|
mids = [float(msg["payload"].get("px", 0)) for msg in messages]
|
|
times = [msg["exchange_ts"] for msg in messages]
|
|
sides = classify_bulk_lee_ready(trades, mids)
|
|
buys = sum(1 for s in sides if s == "buy")
|
|
sells = sum(1 for s in sides if s == "sell")
|
|
arrival = trade_arrival_rate(times)
|
|
vol = trade_volume_profile(trades)
|
|
|
|
print(f"Trades: {len(trades)} total ({buys} buy, {sells} sell)")
|
|
print(f"Arrival rate: {json.dumps(arrival, indent=2, default=str)}")
|
|
print(f"Volume profile: {json.dumps(vol, indent=2, default=str)}")
|
|
|
|
elif args.channel == "funding":
|
|
from microstructure.funding import funding_regime, basis_spread
|
|
|
|
rates = [float(msg["payload"].get("funding", 0)) for msg in messages]
|
|
marks = [float(msg["payload"].get("mark_px", 0)) for msg in messages]
|
|
regime = funding_regime(rates, window_hours=24, n_samples_per_hour=1)
|
|
print(f"Funding regime: {json.dumps(regime, indent=2, default=str)}")
|
|
|
|
elif args.channel == "markouts":
|
|
from microstructure.trades import compute_markouts, markout_summary
|
|
|
|
l2_messages = read_range(
|
|
args.data_dir,
|
|
channel="l2book",
|
|
coin=args.coin.upper(),
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
trade_messages = read_range(
|
|
args.data_dir,
|
|
channel="trades",
|
|
coin=args.coin.upper(),
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
|
|
trades = []
|
|
mids = []
|
|
times = []
|
|
book_state = {}
|
|
for msg in sorted(l2_messages + trade_messages, key=lambda m: m.get("exchange_ts", 0) or 0):
|
|
ch = msg.get("channel", "")
|
|
payload = msg.get("payload", {})
|
|
ts = msg.get("exchange_ts", 0) or 0
|
|
if ch == "l2book":
|
|
levels = payload.get("levels", [])
|
|
if isinstance(levels, list) and len(levels) >= 2:
|
|
bids = [float(l["px"]) for l in levels[0] if float(l.get("sz", 0)) > 0]
|
|
asks = [float(l["px"]) for l in levels[1] if float(l.get("sz", 0)) > 0]
|
|
if bids and asks:
|
|
book_state["mid"] = (bids[0] + asks[0]) / 2
|
|
elif ch == "trades":
|
|
px = float(payload.get("px", 0))
|
|
if px > 0:
|
|
trades.append(payload)
|
|
mid = book_state.get("mid", px)
|
|
mids.append(mid)
|
|
times.append(ts)
|
|
|
|
if not trades:
|
|
print("No trade data with L2 context available for markout analysis.")
|
|
return
|
|
|
|
print(f"Analyzing {len(trades)} trades with L2 context...")
|
|
markouts = compute_markouts(trades, mids, times)
|
|
summary = markout_summary(markouts)
|
|
|
|
print(f"\n{'─' * 70}")
|
|
print(f"{'Horizon':>10s} {'Buy Mean':>10s} {'Buy T-Stat':>10s} {'Buy N':>7s} "
|
|
f"{'Sell Mean':>10s} {'Sell T-Stat':>10s} {'Sell N':>7s}")
|
|
print(f"{'─' * 70}")
|
|
horizons = [100, 500, 1000, 5000, 10000, 30000, 60000]
|
|
for h in horizons:
|
|
b = summary.get("buy", {}).get(h, {})
|
|
s = summary.get("sell", {}).get(h, {})
|
|
print(f"{f'{h}ms':>10s} "
|
|
f"{b.get('mean_bps', 0):>10.2f} {b.get('t_stat', 0):>10.3f} {b.get('count', 0):>7d} "
|
|
f"{s.get('mean_bps', 0):>10.2f} {s.get('t_stat', 0):>10.3f} {s.get('count', 0):>7d}")
|
|
|
|
print(f"\nBuy markout: + = price rises after buy (good for seller, bad for buyer)")
|
|
print(f"Sell markout: + = price falls after sell (good for buyer, bad for seller)")
|
|
print(f"t-stat > 2.0 = statistically significant predictive power")
|
|
|
|
else:
|
|
print(f"Channel '{args.channel}' — raw dump:")
|
|
for msg in messages[:5]:
|
|
print(json.dumps(msg, indent=2, default=str))
|
|
if len(messages) > 5:
|
|
print(f"... and {len(messages) - 5} more")
|
|
|
|
|
|
def cmd_simulate(args):
|
|
"""Run market-making simulator on stored data with L2 events."""
|
|
from data.store import read_range
|
|
from sim.engine import SimulationEngine, SimConfig
|
|
from sim.maker import MakerConfig
|
|
|
|
print(f"Loading L2 book data for {args.coin} from {args.start_date} to {args.end_date}...")
|
|
l2_messages = read_range(
|
|
args.data_dir,
|
|
channel="l2book",
|
|
coin=args.coin.upper(),
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
print(f"Loaded {len(l2_messages)} L2 updates")
|
|
|
|
trade_messages = read_range(
|
|
args.data_dir,
|
|
channel="trades",
|
|
coin=args.coin.upper(),
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
print(f"Loaded {len(trade_messages)} trades")
|
|
|
|
events = []
|
|
for msg in l2_messages:
|
|
payload = msg["payload"]
|
|
levels = payload.get("levels", [])
|
|
bids = {}
|
|
asks = {}
|
|
if levels and isinstance(levels, list) and len(levels) >= 2:
|
|
for bid in levels[0]:
|
|
if float(bid.get("sz", 0)) > 0:
|
|
bids[float(bid["px"])] = float(bid["sz"])
|
|
for ask in levels[1]:
|
|
if float(ask.get("sz", 0)) > 0:
|
|
asks[float(ask["px"])] = float(ask["sz"])
|
|
events.append({
|
|
"type": "l2",
|
|
"data": {"bids": bids, "asks": asks},
|
|
"time": msg["local_ts"],
|
|
"coin": args.coin.upper(),
|
|
})
|
|
|
|
for msg in trade_messages:
|
|
payload = msg["payload"]
|
|
events.append({
|
|
"type": "trade",
|
|
"data": payload,
|
|
"time": msg["local_ts"],
|
|
"coin": args.coin.upper(),
|
|
})
|
|
|
|
events.sort(key=lambda e: e["time"])
|
|
print(f"Total events: {len(events)}")
|
|
|
|
config = SimConfig(
|
|
maker=MakerConfig(
|
|
base_size=args.base_size,
|
|
max_inventory=args.max_inventory,
|
|
gamma=args.gamma,
|
|
),
|
|
max_inventory=args.max_inventory,
|
|
cancel_after_ms=args.cancel_after_ms,
|
|
quote_refresh_ms=args.quote_refresh_ms,
|
|
seed=args.seed,
|
|
)
|
|
|
|
engine = SimulationEngine(config=config, seed=args.seed)
|
|
engine.run(events)
|
|
|
|
stats = engine.stats()
|
|
breakdown = engine.breakdown()
|
|
|
|
print("\n=== Simulation Results ===")
|
|
print(f"Duration: {events[-1]['time'] - events[0]['time']:.0f}s" if events else "0s")
|
|
print(f"Trades: {stats.total_trades} ({stats.bid_fills} bid, {stats.ask_fills} ask)")
|
|
print(f"Toxic fills: {stats.toxic_fills} ({stats.adverse_rate:.1%})")
|
|
print(f"Cancels: {stats.cancels}")
|
|
print(f"Avg spread: {stats.avg_spread_bps} bps")
|
|
print(f"Max inventory: {stats.max_inventory}")
|
|
print(f"Max drawdown: {stats.max_drawdown}%")
|
|
print(f"Sharpe: {stats.sharpe} Sortino: {stats.sortino}")
|
|
print(f"Uptime: {stats.uptime_pct}%")
|
|
print(f"\nPnL Breakdown:")
|
|
print(f" Spread capture: ${breakdown.spread_capture:.4f}")
|
|
print(f" Inventory PnL: ${breakdown.inventory_pnl:.4f}")
|
|
print(f" Maker fees: ${breakdown.maker_fees:.4f}")
|
|
print(f" Taker fees: ${breakdown.taker_fees:.4f}")
|
|
print(f" Funding PnL: ${breakdown.funding_pnl:.4f}")
|
|
print(f" Adverse selection: ${breakdown.adverse_selection_cost:.4f}")
|
|
print(f" ─────────────────────────────")
|
|
print(f" Gross PnL: ${breakdown.gross_pnl:.4f}")
|
|
print(f" Net PnL: ${breakdown.net_pnl:.4f}")
|
|
|
|
|
|
def cmd_run(args):
|
|
"""Start the production trading node."""
|
|
import asyncio
|
|
from live.node_v2 import ProductionNode
|
|
|
|
coins = [c.strip().upper() for c in args.coins.split(",") if c.strip()]
|
|
|
|
node = ProductionNode(
|
|
coins=coins,
|
|
testnet=not args.mainnet,
|
|
mode=args.mode,
|
|
max_position_per_coin=args.max_position,
|
|
base_quote_size=args.base_size,
|
|
initial_equity=args.equity,
|
|
tick_interval_sec=args.tick_interval,
|
|
metrics_file=args.metrics_file,
|
|
)
|
|
|
|
asyncio.run(node.run())
|
|
|
|
|
|
def cmd_backtest(args):
|
|
"""Run a VBT backtest (existing functionality)."""
|
|
from backtests.vbt_runner import VBTBacktestRunner
|
|
runner = VBTBacktestRunner()
|
|
result = runner.run_strategy(strategy=args.strategy, interval=args.interval, limit=args.limit)
|
|
import json as _json
|
|
print(_json.dumps({k: v for k, v in (result or {}).items()
|
|
if k not in ("trades", "equity_curve")}, indent=2, default=str))
|
|
if result and result.get("trades"):
|
|
print(f"\n{len(result['trades'])} trades")
|
|
|
|
|
|
def cmd_discover(args):
|
|
"""Signal discovery — test microstructure signals against forward returns.
|
|
|
|
For each L2 snapshot, computes WQI, OBI, VPIN, depth imbalance, and microprice.
|
|
Then measures how well each signal predicts mid-price movement at multiple horizons.
|
|
"""
|
|
from data.store import read_range
|
|
from microstructure.book import (
|
|
mid_price, order_book_imbalance, depth_imbalance, microprice, spread_stats, depth_resiliency,
|
|
)
|
|
from microstructure.trades import classify_lee_ready
|
|
from microstructure.toxicity import compute_vpin
|
|
from strategies.queue_imbalance import QueueImbalance
|
|
import numpy as np
|
|
|
|
horizons = [int(h) for h in args.horizons.split(",")]
|
|
|
|
l2_msgs = read_range(
|
|
args.data_dir,
|
|
channel="l2book",
|
|
coin=args.coin.upper(),
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
trade_msgs = read_range(
|
|
args.data_dir,
|
|
channel="trades",
|
|
coin=args.coin.upper(),
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
|
|
if not l2_msgs or not trade_msgs:
|
|
print("No data available for signal discovery.")
|
|
return
|
|
|
|
print(f"Loading {len(l2_msgs)} L2 messages and {len(trade_msgs)} trades for {args.coin}...")
|
|
|
|
book_state = {"bids": {}, "asks": {}, "mid": 0.0, "ts": 0.0}
|
|
buy_vol = []
|
|
sell_vol = []
|
|
prev_bids = {}
|
|
prev_asks = {}
|
|
qi = QueueImbalance(depth_levels=10)
|
|
prev_mid = 0.0
|
|
|
|
events = []
|
|
for msg in sorted(l2_msgs + trade_msgs, key=lambda m: m.get("exchange_ts", 0) or 0):
|
|
ch = msg.get("channel", "")
|
|
payload = msg.get("payload", {})
|
|
ts = float(msg.get("exchange_ts", 0) or 0) / 1000.0
|
|
events.append((ts, ch, payload))
|
|
|
|
events.sort(key=lambda e: e[0])
|
|
|
|
signals = []
|
|
mids_series = []
|
|
times_series = []
|
|
|
|
for ts, ch, payload in events:
|
|
if ch == "l2book":
|
|
bids = {}
|
|
asks = {}
|
|
levels = payload.get("levels", [])
|
|
if isinstance(levels, list) and len(levels) >= 2:
|
|
bid_list = [(float(l["px"]), float(l["sz"])) for l in levels[0] if float(l.get("sz", 0)) > 0]
|
|
ask_list = [(float(l["px"]), float(l["sz"])) for l in levels[1] if float(l.get("sz", 0)) > 0]
|
|
bids = dict(bid_list)
|
|
asks = dict(ask_list)
|
|
|
|
if bids and asks:
|
|
book_state["bids"] = bids
|
|
book_state["asks"] = asks
|
|
mid = mid_price(bids, asks)
|
|
book_state["mid"] = mid
|
|
book_state["ts"] = ts
|
|
|
|
obi = order_book_imbalance(bids, asks)
|
|
di = depth_imbalance(bids, asks)
|
|
mp = microprice(bids, asks)
|
|
ss = spread_stats(bids, asks)
|
|
dr = depth_resiliency(bids, asks)
|
|
|
|
wqi = qi.compute_wqi(bid_list, ask_list)
|
|
|
|
vpin_val = 0.0
|
|
if buy_vol and sell_vol:
|
|
v = compute_vpin(buy_vol, sell_vol, n_buckets=50)
|
|
vpin_val = v.get("vpin_value", 0.0)
|
|
|
|
bid_depth = sum(sz for _, sz in bid_list[:10])
|
|
ask_depth = sum(sz for _, sz in ask_list[:10])
|
|
total_depth = bid_depth + ask_depth
|
|
|
|
signals.append({
|
|
"ts": ts,
|
|
"mid": mid,
|
|
"obi": obi,
|
|
"wqi": wqi,
|
|
"vpin": vpin_val,
|
|
"depth_imbalance": di,
|
|
"microprice_ratio": mp / mid if mid > 0 else 1.0,
|
|
"spread_bps": ss["spread_bps"],
|
|
"bid_depth": bid_depth,
|
|
"ask_depth": ask_depth,
|
|
"depth_total": total_depth,
|
|
"resiliency": dr.get("resiliency", 0),
|
|
})
|
|
mids_series.append(mid)
|
|
times_series.append(ts)
|
|
prev_bids = bid_list
|
|
prev_asks = ask_list
|
|
|
|
elif ch == "trades":
|
|
px = float(payload.get("px", 0))
|
|
sz = float(payload.get("sz", 0))
|
|
if px > 0 and sz > 0:
|
|
side = classify_lee_ready(px, book_state["mid"])
|
|
if side == "buy":
|
|
buy_vol.append(sz)
|
|
else:
|
|
sell_vol.append(sz)
|
|
|
|
n = len(signals)
|
|
if n < 50:
|
|
print("Too few L2 snapshots for signal discovery. Need more data.")
|
|
return
|
|
|
|
print(f"Signal discovery on {n} L2 snapshots across {horizons}ms horizons...")
|
|
|
|
signal_names = ["obi", "wqi", "vpin", "depth_imbalance", "spread_bps", "resiliency"]
|
|
horizons = sorted(horizons)
|
|
|
|
print(f"\n{'─' * 80}")
|
|
print(f"{'Signal':>18s} ", end="")
|
|
for h in horizons:
|
|
print(f" {'t-{h}ms':>10s}", end="")
|
|
print(f" {'r²':>8s}")
|
|
print(f"{'─' * 80}")
|
|
|
|
for sig_name in signal_names:
|
|
sig_vals = [s.get(sig_name, 0) for s in signals]
|
|
print(f"{sig_name:>18s} ", end="")
|
|
|
|
for horizon in horizons:
|
|
t_stats = []
|
|
for i in range(n - 1):
|
|
future_idx = i
|
|
for j in range(i + 1, min(n, len(times_series))):
|
|
if times_series[j] - times_series[i] >= horizon / 1000.0:
|
|
future_idx = j
|
|
break
|
|
if future_idx > i and mids_series[i] > 0:
|
|
forward_return = (mids_series[future_idx] - mids_series[i]) / mids_series[i] * 10000
|
|
if abs(forward_return) < 500:
|
|
zipped = list(zip(sig_vals, [forward_return] * len(sig_vals)))
|
|
t_tuple = zipped[i] if i < len(zipped) else None
|
|
|
|
t_stats = []
|
|
for i in range(n - 1):
|
|
future_idx = i
|
|
for j in range(i + 1, n):
|
|
if times_series[j] - times_series[i] >= horizon / 1000.0:
|
|
future_idx = j
|
|
break
|
|
if future_idx > i and mids_series[i] > 0:
|
|
forward_return = (mids_series[future_idx] - mids_series[i]) / mids_series[i] * 10000
|
|
signal_val = sig_vals[i]
|
|
if abs(forward_return) < 500 and abs(signal_val) < 100:
|
|
t_stats.append((signal_val, forward_return))
|
|
|
|
if len(t_stats) >= 10:
|
|
xs = np.array([t[0] for t in t_stats])
|
|
ys = np.array([t[1] for t in t_stats])
|
|
r = np.corrcoef(xs, ys)[0, 1] if len(xs) > 1 else 0
|
|
t_stat = r * np.sqrt(len(t_stats) - 2) / np.sqrt(1 - r * r) if abs(r) < 1 else 0
|
|
print(f" {t_stat:>10.3f}", end="")
|
|
else:
|
|
print(f" {'N/A':>10s}", end="")
|
|
|
|
if len(t_stats) >= 10:
|
|
xs = np.array([t[0] for t in t_stats])
|
|
ys = np.array([t[1] for t in t_stats])
|
|
r_sq = np.corrcoef(xs, ys)[0, 1] ** 2 if len(xs) > 1 else 0
|
|
print(f" {r_sq:>8.4f}")
|
|
else:
|
|
print()
|
|
|
|
print(f"\nPipeline ready. Run 'python -m cli tick' to backtest strategies on this data.")
|
|
|
|
|
|
def cmd_funding(args):
|
|
"""Funding rate arb discovery — analyze historical funding rates and run backtests."""
|
|
from strategies.funding_arb_strategy import run_funding_discovery
|
|
import json as _json
|
|
|
|
result = run_funding_discovery(
|
|
data_dir=args.data_dir,
|
|
coin=args.coin,
|
|
start_date=args.start_date,
|
|
end_date=args.end_date,
|
|
)
|
|
|
|
if "error" in result:
|
|
print(f"Error: {result['error']}")
|
|
return
|
|
|
|
print(f"\n{'═' * 60}")
|
|
print(f" Funding Rate Analysis — {args.coin}")
|
|
print(f" {result['n_observations']:,} observations")
|
|
print(f"{'═' * 60}")
|
|
|
|
dist = result["rate_distribution"]
|
|
print(f"\n Rate Distribution (annualized):")
|
|
print(f" Mean: {dist['mean_apr_pct']:>8.2f}%")
|
|
print(f" Std: {dist['std_apr_pct']:>8.2f}%")
|
|
print(f" Max: {dist['max_apr_pct']:>8.2f}%")
|
|
print(f" Min: {dist['min_apr_pct']:>8.2f}%")
|
|
print(f"\n Absolute Rate Percentiles:")
|
|
for k, v in dist["abs_percentiles"].items():
|
|
print(f" {k}: {v:>8.2f}%")
|
|
|
|
print(f"\n{'─' * 60}")
|
|
print(f" Backtest Results by Threshold:")
|
|
print(f" {'Threshold':>12s} {'Trades':>7s} {'Win Rate':>9s} "
|
|
f"{'Net PnL':>10s} {'Avg PnL':>10s} {'Avg Hold':>9s}")
|
|
print(f" {'─' * 60}")
|
|
for label, bt in result.get("backtests", {}).items():
|
|
print(f" {label:>12s} {bt['total_trades']:>7d} "
|
|
f"{bt['win_rate']:>8.1%} "
|
|
f"${bt['total_net_pnl']:>9.4f} ${bt['avg_net_pnl']:>9.4f} "
|
|
f"{bt['avg_hold_hours']:>8.1f}h")
|
|
|
|
print(f"\n Run 'python -m cli collect --mainnet' to gather data.")
|
|
print(f" Then 'python -m cli funding --coin BTC' to re-run.")
|
|
|
|
|
|
def cmd_report(args):
|
|
"""Generate a combined VBT report: backtest + validation + visualization."""
|
|
import json as _json
|
|
from backtests.vbt_runner import VBTBacktestRunner
|
|
|
|
runner = VBTBacktestRunner()
|
|
result = runner.run_with_report(
|
|
strategy=args.strategy,
|
|
interval=args.interval,
|
|
testnet=False,
|
|
limit=args.limit,
|
|
output_dir=args.output_dir,
|
|
)
|
|
|
|
if result:
|
|
print(f"Report generated for {args.strategy} ({args.interval})")
|
|
print(f" Strategy: {result['strategy']}")
|
|
print(f" Sharpe: {result.get('sharpe', 0):.3f}")
|
|
print(f" Net PnL: ${result.get('pnl', 0):.2f}")
|
|
print(f" Trades: {result.get('total_trades', 0)}")
|
|
print(f" Validation: {len(result.get('validation_errors', []))} errors, "
|
|
f"{len(result.get('validation_warnings', []))} warnings")
|
|
print(f" Output: {args.output_dir}/")
|
|
else:
|
|
print(f"No data available for {args.strategy}. "
|
|
f"Try: python -m cli backtest --strategy {args.strategy} --interval {args.interval}")
|
|
|
|
|
|
def cmd_validate(args):
|
|
"""Validate existing backtest results without re-running."""
|
|
import json as _json
|
|
from pathlib import Path
|
|
from backtests.vbt_validator import VBTValidator, ValidationReport
|
|
|
|
rd = Path(args.results_dir)
|
|
files = sorted(rd.glob("*.json"))
|
|
if not files:
|
|
print(f"No backtest results found in {args.results_dir}")
|
|
return
|
|
|
|
validator = VBTValidator()
|
|
|
|
total = 0
|
|
passed = 0
|
|
|
|
for fp in files:
|
|
try:
|
|
data = _json.loads(fp.read_text())
|
|
except Exception:
|
|
continue
|
|
|
|
strat = data.get("strategy", "?")
|
|
if args.strategy != "all" and args.strategy.lower() != strat.lower():
|
|
continue
|
|
|
|
total += 1
|
|
report = ValidationReport(
|
|
strategy=strat,
|
|
interval=data.get("interval", "?"),
|
|
)
|
|
|
|
trades = data.get("trades", [])
|
|
n_trades = data.get("total_trades", len(trades))
|
|
if n_trades < 10:
|
|
report.warnings.append(
|
|
f"{fp.name}: only {n_trades} trades — insufficient for stats"
|
|
)
|
|
|
|
sharpe = data.get("sharpe", 0)
|
|
if n_trades > 0 and abs(sharpe) > 5:
|
|
report.warnings.append(
|
|
f"{fp.name}: extreme Sharpe {sharpe:.2f} with {n_trades} trades"
|
|
)
|
|
|
|
if report.errors or report.warnings:
|
|
print(report.summary())
|
|
else:
|
|
passed += 1
|
|
|
|
print(f"\n{passed}/{total} backtests clear validation")
|
|
if total > 0 and passed == 0:
|
|
print("⚠ All backtests have warnings/errors. Review needed.")
|
|
print(f"\nFull validation requires re-running with VBTValidator.validate().")
|
|
print(f"Use: python -m cli report --strategy <name> for full validation.")
|
|
|
|
|
|
def cmd_hft_viz(args):
|
|
"""Generate HFT tick visualization dashboard."""
|
|
from backtests.tick_viz import cmd_tick_viz
|
|
cmd_tick_viz(args)
|
|
|
|
|
|
def main():
|
|
import argparse
|
|
p = argparse.ArgumentParser(description="FTDT Quant Lab CLI")
|
|
sp = p.add_subparsers(dest="command", required=True)
|
|
|
|
# collect
|
|
pc = sp.add_parser("collect", help="Run data collector")
|
|
pc.add_argument("--coins", nargs="+", default=["BTC", "ETH"])
|
|
pc.add_argument("--mainnet", action="store_true")
|
|
pc.add_argument("--data-dir", default="data/raw")
|
|
pc.add_argument("--poll-interval", type=float, default=60.0)
|
|
pc.add_argument("--flush-interval", type=float, default=5.0)
|
|
|
|
# analyze
|
|
pa = sp.add_parser("analyze", help="Run microstructure analytics")
|
|
pa.add_argument("--data-dir", default="data/raw")
|
|
pa.add_argument("--channel", default="l2book", choices=["l2book", "trades", "funding", "mark", "open_interest", "liquidation", "markouts"])
|
|
pa.add_argument("--coin", default="BTC")
|
|
pa.add_argument("--start-date", default="2026-08-01")
|
|
pa.add_argument("--end-date", default="2026-08-07")
|
|
|
|
# simulate
|
|
ps = sp.add_parser("simulate", help="Run market-making simulator")
|
|
ps.add_argument("--data-dir", default="data/raw")
|
|
ps.add_argument("--coin", default="BTC")
|
|
ps.add_argument("--start-date", default="2026-08-01")
|
|
ps.add_argument("--end-date", default="2026-08-07")
|
|
ps.add_argument("--gamma", type=float, default=0.1)
|
|
ps.add_argument("--base-size", type=float, default=0.001)
|
|
ps.add_argument("--max-inventory", type=float, default=0.005)
|
|
ps.add_argument("--cancel-after-ms", type=float, default=5000.0)
|
|
ps.add_argument("--quote-refresh-ms", type=float, default=2000.0)
|
|
ps.add_argument("--seed", type=int, default=42)
|
|
|
|
# run
|
|
pr = sp.add_parser("run", help="Start production node")
|
|
pr.add_argument("--coins", default="BTC,ETH", help="Comma-separated coin list")
|
|
pr.add_argument("--mainnet", action="store_true")
|
|
pr.add_argument("--mode", default="paper", choices=["paper", "live"])
|
|
pr.add_argument("--max-position", type=float, default=0.003)
|
|
pr.add_argument("--base-size", type=float, default=0.0002)
|
|
pr.add_argument("--equity", type=float, default=10000.0)
|
|
pr.add_argument("--tick-interval", type=float, default=2.0)
|
|
pr.add_argument("--metrics-file", default="/tmp/ftdt-metrics-v2.json")
|
|
|
|
# backtest
|
|
pb = sp.add_parser("backtest", help="Run VBT backtest")
|
|
pb.add_argument("--strategy", default="pairs")
|
|
pb.add_argument("--interval", default="1h")
|
|
pb.add_argument("--limit", type=int, default=500)
|
|
|
|
# tick
|
|
pt = sp.add_parser("tick", help="Tick-level backtest (Parquet L2+trade replay)")
|
|
pt.add_argument("--coin", default="BTC")
|
|
pt.add_argument("--data-dir", default="data/raw")
|
|
pt.add_argument("--start-date", default="2026-08-01")
|
|
pt.add_argument("--end-date", default="2026-08-07")
|
|
pt.add_argument("--maker", default="as_mm", choices=["as_mm", "vpin_as_mm"])
|
|
pt.add_argument("--gamma", type=float, default=0.1)
|
|
pt.add_argument("--base-size", type=float, default=0.001)
|
|
pt.add_argument("--max-inventory", type=float, default=0.005)
|
|
pt.add_argument("--skew-factor", type=float, default=0.5)
|
|
pt.add_argument("--vpin-threshold", type=float, default=0.30)
|
|
pt.add_argument("--vpin-alarm", type=float, default=0.50)
|
|
pt.add_argument("--maker-fee", type=float, default=0.02, help="Maker fee (e.g. 0.02 = 2bps)")
|
|
pt.add_argument("--taker-fee", type=float, default=0.05, help="Taker fee (e.g. 0.05 = 5bps)")
|
|
pt.add_argument("--adverse-prob", type=float, default=0.15)
|
|
pt.add_argument("--cancel-after-ms", type=float, default=5000.0)
|
|
pt.add_argument("--quote-refresh-ms", type=float, default=2000.0)
|
|
pt.add_argument("--seed", type=int, default=42)
|
|
|
|
# discover
|
|
pd = sp.add_parser("discover", help="Signal discovery — test microstructure signals against forward returns")
|
|
pd.add_argument("--data-dir", default="data/raw")
|
|
pd.add_argument("--coin", default="BTC")
|
|
pd.add_argument("--start-date", default="2026-08-01")
|
|
pd.add_argument("--end-date", default="2026-08-07")
|
|
pd.add_argument("--horizons", default="100,500,1000,5000,10000", help="Comma-separated ms horizons")
|
|
|
|
# funding
|
|
pf = sp.add_parser("funding", help="Funding rate arb discovery — analyze historical funding rates")
|
|
pf.add_argument("--data-dir", default="data/raw")
|
|
pf.add_argument("--coin", default="BTC")
|
|
pf.add_argument("--start-date", default="2026-01-01")
|
|
pf.add_argument("--end-date", default="2030-01-01")
|
|
|
|
# report
|
|
prp = sp.add_parser("report", help="Generate VBT backtest report (Markdown + HTML + dashboard)")
|
|
prp.add_argument("--strategy", default="pairs")
|
|
prp.add_argument("--interval", default="1h")
|
|
prp.add_argument("--limit", type=int, default=5000)
|
|
prp.add_argument("--output-dir", default="backtests/reports")
|
|
prp.add_argument("--format", default="html", choices=["md", "html"])
|
|
|
|
# validate
|
|
pv = sp.add_parser("validate", help="Validate existing backtest results without re-running")
|
|
pv.add_argument("--strategy", default="all", help="Strategy name or 'all'")
|
|
pv.add_argument("--results-dir", default="backtests/results")
|
|
|
|
# hft
|
|
ph = sp.add_parser("hft", help="Generate HFT tick visualization dashboard")
|
|
ph.add_argument("--data-dir", default="data/raw")
|
|
ph.add_argument("--coin", default="BTC")
|
|
ph.add_argument("--start-date", default="2026-08-01")
|
|
ph.add_argument("--end-date", default="2026-08-02")
|
|
ph.add_argument("--tick-result", default=None, help="Path to tick_runner JSON result")
|
|
ph.add_argument("--output-dir", default="backtests/reports")
|
|
|
|
args = p.parse_args()
|
|
|
|
import json as _json
|
|
import json
|
|
|
|
if args.command == "collect":
|
|
cmd_collect(args)
|
|
elif args.command == "analyze":
|
|
cmd_analyze(args)
|
|
elif args.command == "simulate":
|
|
cmd_simulate(args)
|
|
elif args.command == "run":
|
|
cmd_run(args)
|
|
elif args.command == "backtest":
|
|
cmd_backtest(args)
|
|
elif args.command == "tick":
|
|
from backtests.tick_runner import cmd_tick_backtest
|
|
cmd_tick_backtest(args)
|
|
elif args.command == "discover":
|
|
cmd_discover(args)
|
|
elif args.command == "funding":
|
|
cmd_funding(args)
|
|
elif args.command == "report":
|
|
cmd_report(args)
|
|
elif args.command == "validate":
|
|
cmd_validate(args)
|
|
elif args.command == "hft":
|
|
cmd_hft_viz(args)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s", datefmt="%H:%M:%S")
|
|
main()
|