feat: Phase 2 — microstructure analytics + 81 tests

New microstructure/ module with pure-function analytics:

microstructure/book.py:
  microprice() — depth-weighted mid price
  mid_price() — simple bid/ask midpoint
  order_book_imbalance() — ranged [-1, 1] volume skew
  depth_imbalance() — imbalance at fixed price distance
  spread_stats() — spread, spread_bps, mid, bid, ask
  depth_resiliency() — bid/ask volume within impact radius
  queue_depletion_prob() — Poisson fill probability at level
  batch_book_stats() — aggregate stats across snapshots

microstructure/trades.py:
  classify_lee_ready() — Lee-Ready aggressor classification
  classify_bulk_lee_ready() — batch classification with mids/bids/asks
  compute_markouts() — forward mid-price change at configurable horizons
  markout_summary() — mean/std/t-stat per side per horizon
  trade_volume_profile() — size bucket distribution
  trade_arrival_rate() — rolling trades/sec with burst detection

microstructure/toxicity.py:
  compute_vpin() — volume-synchronized informed trading probability
  compute_vpin_time_series() — rolling VPIN with alarm threshold
  fill_toxicity() — adverse price movement post-trade
  adverse_selection_ratio() — per-side adverse selection
  liquidation_clustering() — cluster detection in liquidation events

microstructure/funding.py:
  funding_regime() — classify regime (neutral/positive/negative/high)
  funding_predictability() — AR(1) autocorrelation analysis
  funding_carry_pnl() — cumulative carry PnL estimation
  basis_spread() — perp premium over spot (bps)
  basis_convergence_speed() — mean-reversion half-life via AR(1)

microstructure/signals.py:
  composite_signal() — weighted OBI + trade + VPIN + funding signal
  SignalPipeline — stateful pipeline accumulating book/trade updates
  detect_hft_regime() — regime classifier for HFT strategy selection

Bug fixes in Phase 1:
  - data/latency.py: proper linear-interpolation percentiles
  - data/normalizer.py: UTC timezone for naive datetimes
  - data/normalizer.py: detect_sequence_gap returns gap-1 (missing count)
  - microstructure/toxicity.py: consistent vpin_value key in compute_vpin

81 tests across 4 test files (store, normalizer, latency, microstructure)
This commit is contained in:
ramseshk
2026-08-07 14:34:18 +08:00
parent a7f811eb81
commit fcfc136384
13 changed files with 1817 additions and 5 deletions
+179
View File
@@ -0,0 +1,179 @@
"""
Funding and basis behavior analytics.
Analyses funding rate regimes, basis dynamics (spot vs perp),
and carry trade profitability.
"""
from __future__ import annotations
import math
import numpy as np
# ── Funding rate analytics ──────────────────────────────────
def funding_regime(
funding_rates: list[float],
window_hours: int = 24,
n_samples_per_hour: int = 60, # e.g., 1 sample/min → 60/hr
) -> dict:
"""Classify current funding regime.
Returns regime classification and rolling stats.
"""
window = window_hours * n_samples_per_hour
if len(funding_rates) < window:
return {"regime": "insufficient_data", "mean_annual": 0, "volatility": 0}
recent = funding_rates[-window:]
mean_rate = float(np.mean(recent))
std_rate = float(np.std(recent))
# Funding is per-hour rate. Annualize: compounded 3x daily.
# Hyperliquid funding: 8h rate * 3 = daily, * 365 = annual (approx)
ann_rate = mean_rate * 3 * 365 * 100 # *100 to convert from fraction to %
if ann_rate > 15:
regime = "high_positive"
elif ann_rate > 5:
regime = "positive"
elif ann_rate < -15:
regime = "high_negative"
elif ann_rate < -5:
regime = "negative"
else:
regime = "neutral"
return {
"regime": regime,
"mean_hourly": round(float(mean_rate), 8),
"mean_annual_pct": round(ann_rate, 2),
"volatility": round(float(std_rate), 8),
"window_hours": window_hours,
}
def funding_predictability(
funding_history: list[float],
n_lags: int = 3,
) -> dict:
"""Measure funding rate autocorrelation — is funding momentum persistent?"""
if len(funding_history) < n_lags + 2:
return {"autocorr": [], "momentum_strength": 0}
autocorr = []
for lag in range(1, n_lags + 1):
x = funding_history[:-lag]
y = funding_history[lag:]
if len(x) < 2:
autocorr.append(0)
continue
corr = np.corrcoef(x, y)[0, 1]
autocorr.append(round(float(corr) if not np.isnan(corr) else 0, 4))
momentum = float(np.mean([abs(a) for a in autocorr]))
return {
"autocorr": autocorr,
"momentum_strength": round(momentum, 4),
"is_momentum": momentum > 0.3,
}
def funding_carry_pnl(
funding_rates: list[float],
position_size: float = 1.0,
mark_prices: list[float] | None = None,
n_samples_per_hour: int = 60,
) -> dict:
"""Estimate carry PnL from holding a position given funding rates.
For positive funding → shorts earn, longs pay.
"""
if not funding_rates:
return {"cumulative_pnl": 0, "hourly_pnl": []}
hourly = []
cum = 0.0
for i, rate in enumerate(funding_rates):
if i % n_samples_per_hour == 0:
notional = position_size * (mark_prices[i] if mark_prices and i < len(mark_prices) else 1.0)
pnl = rate * notional # funding rate * position notional
cum += pnl
hourly.append(round(pnl, 8))
return {
"cumulative_pnl": round(cum, 6),
"hourly_pnl": hourly[-72:], # last 72 hours
"n_hours": len(hourly),
}
# ── Basis analytics ──────────────────────────────────────────
def basis_spread(
perp_prices: list[float],
spot_prices: list[float],
) -> dict:
"""Compute basis (perp premium over spot) and its statistics.
basis_bps = (perp - spot) / spot * 10000
"""
min_len = min(len(perp_prices), len(spot_prices))
if min_len < 2:
return {"current_basis_bps": 0, "mean_basis_bps": 0, "max_basis_bps": 0}
perp = perp_prices[-min_len:]
spot = spot_prices[-min_len:]
basis_arr = []
for p, s in zip(perp, spot):
if s > 0:
basis_arr.append((p - s) / s * 10000)
a = np.array(basis_arr) if basis_arr else np.array([0.0])
return {
"current_basis_bps": round(float(a[-1]), 2) if len(a) > 0 else 0,
"mean_basis_bps": round(float(np.mean(a)), 2),
"std_basis_bps": round(float(np.std(a)), 2),
"max_basis_bps": round(float(np.max(a)), 2),
"min_basis_bps": round(float(np.min(a)), 2),
"n_samples": len(a),
}
def basis_convergence_speed(
basis_history: list[float],
half_life_lookback: int = 1440, # 24h at 1min samples
) -> dict:
"""Estimate basis mean-reversion half-life via AR(1).
Half-life = -log(2) / log(|rho|)
"""
if len(basis_history) < 10:
return {"half_life_minutes": 0, "ar1_coef": 0, "mean_reverting": False}
x = basis_history[-half_life_lookback:]
if len(x) < 10:
return {"half_life_minutes": 0, "ar1_coef": 0, "mean_reverting": False}
x_t = x[:-1]
x_t1 = x[1:]
rho = np.corrcoef(x_t, x_t1)[0, 1]
rho = max(min(rho, 0.999), -0.999)
if abs(rho) < 0.01:
half_life = 0
else:
half_life = -math.log(2) / math.log(abs(rho))
return {
"half_life_minutes": round(half_life, 1),
"ar1_coef": round(float(rho), 4),
"mean_reverting": abs(rho) > 0.1 and rho < 0.95,
}