feat: Phase 2 — microstructure analytics + 81 tests

New microstructure/ module with pure-function analytics:

microstructure/book.py:
  microprice() — depth-weighted mid price
  mid_price() — simple bid/ask midpoint
  order_book_imbalance() — ranged [-1, 1] volume skew
  depth_imbalance() — imbalance at fixed price distance
  spread_stats() — spread, spread_bps, mid, bid, ask
  depth_resiliency() — bid/ask volume within impact radius
  queue_depletion_prob() — Poisson fill probability at level
  batch_book_stats() — aggregate stats across snapshots

microstructure/trades.py:
  classify_lee_ready() — Lee-Ready aggressor classification
  classify_bulk_lee_ready() — batch classification with mids/bids/asks
  compute_markouts() — forward mid-price change at configurable horizons
  markout_summary() — mean/std/t-stat per side per horizon
  trade_volume_profile() — size bucket distribution
  trade_arrival_rate() — rolling trades/sec with burst detection

microstructure/toxicity.py:
  compute_vpin() — volume-synchronized informed trading probability
  compute_vpin_time_series() — rolling VPIN with alarm threshold
  fill_toxicity() — adverse price movement post-trade
  adverse_selection_ratio() — per-side adverse selection
  liquidation_clustering() — cluster detection in liquidation events

microstructure/funding.py:
  funding_regime() — classify regime (neutral/positive/negative/high)
  funding_predictability() — AR(1) autocorrelation analysis
  funding_carry_pnl() — cumulative carry PnL estimation
  basis_spread() — perp premium over spot (bps)
  basis_convergence_speed() — mean-reversion half-life via AR(1)

microstructure/signals.py:
  composite_signal() — weighted OBI + trade + VPIN + funding signal
  SignalPipeline — stateful pipeline accumulating book/trade updates
  detect_hft_regime() — regime classifier for HFT strategy selection

Bug fixes in Phase 1:
  - data/latency.py: proper linear-interpolation percentiles
  - data/normalizer.py: UTC timezone for naive datetimes
  - data/normalizer.py: detect_sequence_gap returns gap-1 (missing count)
  - microstructure/toxicity.py: consistent vpin_value key in compute_vpin

81 tests across 4 test files (store, normalizer, latency, microstructure)
This commit is contained in:
ramseshk
2026-08-07 14:34:18 +08:00
parent a7f811eb81
commit fcfc136384
13 changed files with 1817 additions and 5 deletions
+59
View File
@@ -0,0 +1,59 @@
"""
Microstructure analytics — book, trade, toxicity, funding, and signal composition.
Pure functions that take market data arrays and return computed metrics.
"""
from microstructure.book import (
microprice,
mid_price,
order_book_imbalance,
depth_imbalance,
spread_stats,
depth_resiliency,
queue_depletion_prob,
batch_book_stats,
)
from microstructure.trades import (
classify_lee_ready,
classify_bulk_lee_ready,
compute_markouts,
markout_summary,
trade_volume_profile,
trade_arrival_rate,
)
from microstructure.toxicity import (
compute_vpin,
compute_vpin_time_series,
fill_toxicity,
adverse_selection_ratio,
liquidation_clustering,
)
from microstructure.funding import (
funding_regime,
funding_predictability,
funding_carry_pnl,
basis_spread,
basis_convergence_speed,
)
from microstructure.signals import (
composite_signal,
SignalPipeline,
detect_hft_regime,
)
__all__ = [
"microprice", "mid_price", "order_book_imbalance", "depth_imbalance",
"spread_stats", "depth_resiliency", "queue_depletion_prob", "batch_book_stats",
"classify_lee_ready", "classify_bulk_lee_ready", "compute_markouts",
"markout_summary", "trade_volume_profile", "trade_arrival_rate",
"compute_vpin", "compute_vpin_time_series", "fill_toxicity",
"adverse_selection_ratio", "liquidation_clustering",
"funding_regime", "funding_predictability", "funding_carry_pnl",
"basis_spread", "basis_convergence_speed",
"composite_signal", "SignalPipeline", "detect_hft_regime",
]
+235
View File
@@ -0,0 +1,235 @@
"""
Order book microstructure analytics.
Functions operate on book snapshots (bids/asks dicts or DataFrames)
and return time-series or summary stats.
"""
from __future__ import annotations
import math
from collections import deque
from typing import Optional
import numpy as np
# ── Microprice ───────────────────────────────────────────────
def microprice(
bids: dict[float, float],
asks: dict[float, float],
weight_bid: float = 0.5,
) -> float:
"""Weighted mid based on depth imbalance.
microprice = weight * bid_side + (1-weight) * ask_side
where weight = bid_depth / (bid_depth + ask_depth)
Falls back to simple mid if no depth.
"""
bid_prices = sorted(bids.keys(), reverse=True)
ask_prices = sorted(asks.keys())
if not bid_prices or not ask_prices:
return 0.0
best_bid = bid_prices[0]
best_ask = ask_prices[0]
bid_vol = sum(bids[px] for px in bid_prices[:10])
ask_vol = sum(asks[px] for px in ask_prices[:10])
total = bid_vol + ask_vol
if total == 0:
return (best_bid + best_ask) / 2.0
w = bid_vol / total
return w * best_bid + (1 - w) * best_ask
def mid_price(bids: dict[float, float], asks: dict[float, float]) -> float:
bid_prices = sorted(bids.keys(), reverse=True)
ask_prices = sorted(asks.keys())
if not bid_prices or not ask_prices:
return 0.0
return (bid_prices[0] + ask_prices[0]) / 2.0
# ── Order-book imbalance ─────────────────────────────────────
def order_book_imbalance(
bids: dict[float, float],
asks: dict[float, float],
levels: int = 10,
) -> float:
"""OBI = (bid_vol - ask_vol) / (bid_vol + ask_vol). Range [-1, 1]."""
bid_prices = sorted(bids.keys(), reverse=True)[:levels]
ask_prices = sorted(asks.keys())[:levels]
bid_vol = sum(bids[px] for px in bid_prices)
ask_vol = sum(asks[px] for px in ask_prices)
total = bid_vol + ask_vol
if total == 0:
return 0.0
return (bid_vol - ask_vol) / total
def depth_imbalance(
bids: dict[float, float],
asks: dict[float, float],
price_distance_pct: float = 0.01,
) -> float:
"""Imbalance at a fixed price distance from mid (percentage-based)."""
mid = mid_price(bids, asks)
if mid <= 0:
return 0.0
lo = mid * (1 - price_distance_pct)
hi = mid * (1 + price_distance_pct)
bid_vol = sum(sz for px, sz in bids.items() if px >= lo)
ask_vol = sum(sz for px, sz in asks.items() if px <= hi)
total = bid_vol + ask_vol
if total == 0:
return 0.0
return (bid_vol - ask_vol) / total
# ── Spread statistics ────────────────────────────────────────
def spread_stats(
bids: dict[float, float],
asks: dict[float, float],
) -> dict:
bid_prices = sorted(bids.keys(), reverse=True)
ask_prices = sorted(asks.keys())
if not bid_prices or not ask_prices:
return {"spread": 0, "spread_bps": 0, "mid": 0, "bid": 0, "ask": 0}
best_bid = bid_prices[0]
best_ask = ask_prices[0]
mid = (best_bid + best_ask) / 2.0
spread = best_ask - best_bid
spread_bps = (spread / mid * 10000) if mid > 0 else 0
return {
"spread": round(spread, 2),
"spread_bps": round(spread_bps, 2),
"mid": round(mid, 2),
"best_bid": best_bid,
"best_ask": best_ask,
}
# ── Depth resiliency ──────────────────────────────────────────
def depth_resiliency(
bids: dict[float, float],
asks: dict[float, float],
impact_bps: float = 10.0,
) -> dict:
"""How much size sits within N bps of mid on each side.
Returns volume and level count within the impact radius, plus
a resiliency score: bid_depth / ask_depth (ratio).
"""
mid = mid_price(bids, asks)
if mid <= 0:
return {"bid_vol": 0, "ask_vol": 0, "bid_levels": 0, "ask_levels": 0, "resiliency": 0}
radius = mid * impact_bps / 10000
bid_lo = mid - radius
ask_hi = mid + radius
bid_vol = sum(sz for px, sz in bids.items() if px >= bid_lo)
ask_vol = sum(sz for px, sz in asks.items() if px <= ask_hi)
bid_levels = sum(1 for px in bids if px >= bid_lo)
ask_levels = sum(1 for px in asks if px <= ask_hi)
resiliency = bid_vol / ask_vol if ask_vol > 0 else float("inf")
return {
"bid_vol": round(bid_vol, 8),
"ask_vol": round(ask_vol, 8),
"bid_levels": bid_levels,
"ask_levels": ask_levels,
"resiliency": round(resiliency, 4),
}
def queue_depletion_prob(
bids: dict[float, float],
asks: dict[float, float],
level_distance: int = 0,
trade_rate_per_sec: float = 1.0,
avg_trade_size: float = 0.01,
) -> float:
"""Probability queued order at best level(s) gets filled within 1 second.
Simple Poisson model: P(fill) = 1 - exp(-λ)
where λ = trade_rate * avg_trade_size / depth_at_level
level_distance=0 means best bid/ask, 1 = one level behind, etc.
"""
bid_prices = sorted(bids.keys(), reverse=True)
ask_prices = sorted(asks.keys())
if not bid_prices or not ask_prices:
return 0.0
px = bid_prices[min(level_distance, len(bid_prices) - 1)]
depth = bids.get(px, 0)
if depth <= 0:
return 0.0
lam = trade_rate_per_sec * avg_trade_size / depth
prob = 1.0 - math.exp(-lam)
return round(min(prob, 0.9999), 6)
# ── Batch processing ─────────────────────────────────────────
def batch_book_stats(
snapshots: list[dict],
levels: int = 10,
) -> dict:
"""Process a list of book snapshots and return aggregate stats.
Each snapshot: {"bids": {price: size, ...}, "asks": {price: size, ...}}
"""
obis = []
spreads_bps = []
micros = []
depths = []
resiliencies = []
for snap in snapshots:
bids = snap.get("bids", {})
asks = snap.get("asks", {})
if not bids or not asks:
continue
obis.append(order_book_imbalance(bids, asks, levels))
ss = spread_stats(bids, asks)
spreads_bps.append(ss["spread_bps"])
micros.append(microprice(bids, asks))
dr = depth_resiliency(bids, asks)
depths.append(dr["bid_vol"] + dr["ask_vol"])
resiliencies.append(dr["resiliency"])
def _summarize(vals):
if not vals:
return {"mean": 0, "median": 0, "std": 0, "min": 0, "max": 0, "count": 0}
a = np.array(vals, dtype=float)
a = a[np.isfinite(a)]
return {
"mean": round(float(np.mean(a)), 4),
"median": round(float(np.median(a)), 4),
"std": round(float(np.std(a)), 4),
"min": round(float(np.min(a)), 4),
"max": round(float(np.max(a)), 4),
"count": len(a),
}
return {
"obi": _summarize(obis),
"spread_bps": _summarize(spreads_bps),
"microprice_ratio": _summarize([m / mid_price(s["bids"], s["asks"])
if mid_price(s["bids"], s["asks"]) > 0 else 1.0
for s, m in zip(snapshots, micros)]),
"depth_total": _summarize(depths),
"resiliency": _summarize([r for r in resiliencies if r < 1e6]),
}
+179
View File
@@ -0,0 +1,179 @@
"""
Funding and basis behavior analytics.
Analyses funding rate regimes, basis dynamics (spot vs perp),
and carry trade profitability.
"""
from __future__ import annotations
import math
import numpy as np
# ── Funding rate analytics ──────────────────────────────────
def funding_regime(
funding_rates: list[float],
window_hours: int = 24,
n_samples_per_hour: int = 60, # e.g., 1 sample/min → 60/hr
) -> dict:
"""Classify current funding regime.
Returns regime classification and rolling stats.
"""
window = window_hours * n_samples_per_hour
if len(funding_rates) < window:
return {"regime": "insufficient_data", "mean_annual": 0, "volatility": 0}
recent = funding_rates[-window:]
mean_rate = float(np.mean(recent))
std_rate = float(np.std(recent))
# Funding is per-hour rate. Annualize: compounded 3x daily.
# Hyperliquid funding: 8h rate * 3 = daily, * 365 = annual (approx)
ann_rate = mean_rate * 3 * 365 * 100 # *100 to convert from fraction to %
if ann_rate > 15:
regime = "high_positive"
elif ann_rate > 5:
regime = "positive"
elif ann_rate < -15:
regime = "high_negative"
elif ann_rate < -5:
regime = "negative"
else:
regime = "neutral"
return {
"regime": regime,
"mean_hourly": round(float(mean_rate), 8),
"mean_annual_pct": round(ann_rate, 2),
"volatility": round(float(std_rate), 8),
"window_hours": window_hours,
}
def funding_predictability(
funding_history: list[float],
n_lags: int = 3,
) -> dict:
"""Measure funding rate autocorrelation — is funding momentum persistent?"""
if len(funding_history) < n_lags + 2:
return {"autocorr": [], "momentum_strength": 0}
autocorr = []
for lag in range(1, n_lags + 1):
x = funding_history[:-lag]
y = funding_history[lag:]
if len(x) < 2:
autocorr.append(0)
continue
corr = np.corrcoef(x, y)[0, 1]
autocorr.append(round(float(corr) if not np.isnan(corr) else 0, 4))
momentum = float(np.mean([abs(a) for a in autocorr]))
return {
"autocorr": autocorr,
"momentum_strength": round(momentum, 4),
"is_momentum": momentum > 0.3,
}
def funding_carry_pnl(
funding_rates: list[float],
position_size: float = 1.0,
mark_prices: list[float] | None = None,
n_samples_per_hour: int = 60,
) -> dict:
"""Estimate carry PnL from holding a position given funding rates.
For positive funding → shorts earn, longs pay.
"""
if not funding_rates:
return {"cumulative_pnl": 0, "hourly_pnl": []}
hourly = []
cum = 0.0
for i, rate in enumerate(funding_rates):
if i % n_samples_per_hour == 0:
notional = position_size * (mark_prices[i] if mark_prices and i < len(mark_prices) else 1.0)
pnl = rate * notional # funding rate * position notional
cum += pnl
hourly.append(round(pnl, 8))
return {
"cumulative_pnl": round(cum, 6),
"hourly_pnl": hourly[-72:], # last 72 hours
"n_hours": len(hourly),
}
# ── Basis analytics ──────────────────────────────────────────
def basis_spread(
perp_prices: list[float],
spot_prices: list[float],
) -> dict:
"""Compute basis (perp premium over spot) and its statistics.
basis_bps = (perp - spot) / spot * 10000
"""
min_len = min(len(perp_prices), len(spot_prices))
if min_len < 2:
return {"current_basis_bps": 0, "mean_basis_bps": 0, "max_basis_bps": 0}
perp = perp_prices[-min_len:]
spot = spot_prices[-min_len:]
basis_arr = []
for p, s in zip(perp, spot):
if s > 0:
basis_arr.append((p - s) / s * 10000)
a = np.array(basis_arr) if basis_arr else np.array([0.0])
return {
"current_basis_bps": round(float(a[-1]), 2) if len(a) > 0 else 0,
"mean_basis_bps": round(float(np.mean(a)), 2),
"std_basis_bps": round(float(np.std(a)), 2),
"max_basis_bps": round(float(np.max(a)), 2),
"min_basis_bps": round(float(np.min(a)), 2),
"n_samples": len(a),
}
def basis_convergence_speed(
basis_history: list[float],
half_life_lookback: int = 1440, # 24h at 1min samples
) -> dict:
"""Estimate basis mean-reversion half-life via AR(1).
Half-life = -log(2) / log(|rho|)
"""
if len(basis_history) < 10:
return {"half_life_minutes": 0, "ar1_coef": 0, "mean_reverting": False}
x = basis_history[-half_life_lookback:]
if len(x) < 10:
return {"half_life_minutes": 0, "ar1_coef": 0, "mean_reverting": False}
x_t = x[:-1]
x_t1 = x[1:]
rho = np.corrcoef(x_t, x_t1)[0, 1]
rho = max(min(rho, 0.999), -0.999)
if abs(rho) < 0.01:
half_life = 0
else:
half_life = -math.log(2) / math.log(abs(rho))
return {
"half_life_minutes": round(half_life, 1),
"ar1_coef": round(float(rho), 4),
"mean_reverting": abs(rho) > 0.1 and rho < 0.95,
}
+178
View File
@@ -0,0 +1,178 @@
"""
Composite signal construction from microstructure features.
Combines book imbalance, trade flow, toxicity, and funding signals
into a single directional signal with confidence score.
"""
from __future__ import annotations
from typing import Optional
import numpy as np
# ── Composite signal ────────────────────────────────────────
def composite_signal(
obi: float, # [-1, 1] order book imbalance
trade_imbalance: float = 0.0, # [-1, 1] recent trade aggressor skew
vpin: float = 0.0, # [0, 1] flow toxicity (higher = toxic)
funding_regime: str = "neutral", # "high_positive", "negative", etc.
spread_bps: float = 1.0, # current spread
weight_obi: float = 0.35,
weight_trade: float = 0.25,
weight_vpin: float = -0.20, # negative: high VPIN → reduce confidence
weight_funding: float = 0.20,
) -> dict:
"""Combine microstructure features into a single directional signal.
Returns:
signal: "buy", "sell", or "neutral"
score: [-1, 1] raw composite (positive = buy pressure)
confidence: [0, 1] confidence in the signal
breakdown: per-component contributions
"""
obi_score = np.clip(obi, -1.0, 1.0)
trade_score = np.clip(trade_imbalance, -1.0, 1.0)
vpin_score = np.clip(vpin, 0.0, 1.0)
# Funding: positive funding → short gets paid → sell bias; negative → buy bias
funding_score_map = {
"high_positive": -0.8,
"positive": -0.4,
"neutral": 0.0,
"negative": 0.4,
"high_negative": 0.8,
}
funding_score = funding_score_map.get(funding_regime, 0.0)
raw = (
weight_obi * obi_score +
weight_trade * trade_score +
weight_vpin * vpin_score +
weight_funding * funding_score
)
# Confidence: base from signal magnitude, reduced by VPIN and spread
signal_magnitude = abs(raw)
vpin_penalty = np.clip(vpin * 0.5, 0.0, 0.3) if raw != 0 else 0
spread_penalty = min(spread_bps / 50.0, 0.3) # wide spread → lower confidence
confidence = max(0.0, min(1.0, signal_magnitude * 1.5 - vpin_penalty - spread_penalty))
if raw > 0.1:
signal = "buy"
elif raw < -0.1:
signal = "sell"
else:
signal = "neutral"
return {
"signal": signal,
"score": round(raw, 4),
"confidence": round(confidence, 4),
"breakdown": {
"obi": round(obi_score * weight_obi, 4),
"trade": round(trade_score * weight_trade, 4),
"vpin": round(vpin_score * weight_vpin, 4),
"funding": round(funding_score * weight_funding, 4),
},
}
# ── Signal pipeline ─────────────────────────────────────────
class SignalPipeline:
"""Stateful pipeline that accumulates microstructure data and emits signals.
Usage:
pipeline = SignalPipeline()
pipeline.update_book(bids, asks)
pipeline.update_trade(px, sz, mid)
signal = pipeline.emit()
"""
def __init__(
self,
obi_window: int = 100,
trade_window: int = 500,
vpin_volume_size: float = 100.0,
vpin_buckets: int = 50,
):
self._obi_window = obi_window
self._trade_window = trade_window
self._vpin_volume_size = vpin_volume_size
self._vpin_buckets = vpin_buckets
self._buy_vol: list[float] = []
self._sell_vol: list[float] = []
self._obuys: int = 0
self._osells: int = 0
self._obuy: int = 1
def update_book(self, bids: dict[float, float], asks: dict[float, float]):
from microstructure.book import order_book_imbalance
self._obuys = order_book_imbalance(bids, asks)
def update_trade(self, px: float, sz: float, mid: float):
if px >= mid:
self._buy_vol.append(sz)
else:
self._sell_vol.append(sz)
if len(self._buy_vol) > self._trade_window:
self._buy_vol = self._buy_vol[-self._trade_window:]
if len(self._sell_vol) > self._trade_window:
self._sell_vol = self._sell_vol[-self._trade_window:]
def recent_trade_imbalance(self) -> float:
bv = sum(self._buy_vol[-100:])
sv = sum(self._sell_vol[-100:])
total = bv + sv
return (bv - sv) / total if total > 0 else 0.0
def current_vpin(self) -> float:
from microstructure.toxicity import compute_vpin
result = compute_vpin(
self._buy_vol, self._sell_vol,
volume_bucket_size=self._vpin_volume_size,
n_buckets=self._vpin_buckets,
)
return result.get("vpin_value", 0.0)
def emit(self) -> dict:
return composite_signal(
obi=self._obuys,
trade_imbalance=self.recent_trade_imbalance(),
vpin=self.current_vpin(),
)
# ── Regime detection ─────────────────────────────────────────
def detect_hft_regime(
obi_std: float,
spread_mean_bps: float,
trade_rate_per_sec: float,
vpin: float,
) -> str:
"""Classify current market regime for HFT strategy selection.
Returns one of:
- "trending" — directional, high OBI variance, low VPIN
- "ranging" — low OBI variance, tight spread, active
- "toxic" — high VPIN, wide spread → don't quote
- "quiet" — low activity, avoid
"""
if vpin > 0.4:
return "toxic"
if trade_rate_per_sec < 0.1:
return "quiet"
if obi_std > 0.3 and spread_mean_bps < 5:
return "trending"
if obi_std < 0.15 and spread_mean_bps < 3:
return "ranging"
if spread_mean_bps > 10:
return "toxic"
return "quiet"
+248
View File
@@ -0,0 +1,248 @@
"""
Flow toxicity and adverse selection analytics.
VPIN (Volume-synchronized Probability of Informed Trading),
fill toxicity metrics, and adverse selection indicators based
on order book and trade data.
"""
from __future__ import annotations
import math
from collections import deque
import numpy as np
# ── VPIN ─────────────────────────────────────────────────────
def compute_vpin(
buy_volume: list[float],
sell_volume: list[float],
volume_bucket_size: float | None = None,
n_buckets: int = 50,
) -> dict:
"""Volume-synchronized Probability of Informed Trading.
Args:
buy_volume: volume classified as buyer-initiated per bar/period
sell_volume: volume classified as seller-initiated per bar/period
volume_bucket_size: target volume per bucket (auto if None)
n_buckets: number of buckets for rolling VPIN
Returns dict with vpin values and summary.
"""
if not buy_volume or len(buy_volume) != len(sell_volume):
return {"vpin_value": 0, "n_buckets": 0, "bucket_size": 0}
if volume_bucket_size is None:
total_vol = sum(buy_volume) + sum(sell_volume)
volume_bucket_size = total_vol / max(len(buy_volume), 1) * 5
buy_sell = [(b, s) for b, s in zip(buy_volume, sell_volume)]
buckets: list[dict] = []
current_buy = 0.0
current_sell = 0.0
for b, s in buy_sell:
current_buy += b
current_sell += s
if current_buy + current_sell >= volume_bucket_size:
total = current_buy + current_sell
buckets.append({
"buy": current_buy,
"sell": current_sell,
"total": total,
"imbalance": abs(current_buy - current_sell),
})
# Carry over excess
excess = total - volume_bucket_size
current_buy = excess * (current_buy / total) if total > 0 else 0
current_sell = excess * (current_sell / total) if total > 0 else 0
vpin_value = 0.0
if len(buckets) >= n_buckets:
recent = buckets[-n_buckets:]
total_imb = sum(b["imbalance"] for b in recent)
total_vol = sum(b["total"] for b in recent)
vpin_value = total_imb / total_vol if total_vol > 0 else 0.0
return {
"vpin_value": round(vpin_value, 4),
"n_buckets": len(buckets),
"bucket_size": round(volume_bucket_size, 2),
}
def compute_vpin_time_series(
buy_volume: list[float],
sell_volume: list[float],
volume_bucket_size: float | None = None,
n_buckets: int = 50,
) -> dict:
"""Compute rolling VPIN time series."""
if not buy_volume or len(buy_volume) != len(sell_volume):
return {"vpin_values": [], "mean": 0, "std": 0, "max": 0, "threshold_alarm": 0}
if volume_bucket_size is None:
total_vol = sum(buy_volume) + sum(sell_volume)
volume_bucket_size = total_vol / max(len(buy_volume), 1) * 5
buy_sell = [(b, s) for b, s in zip(buy_volume, sell_volume)]
buckets: list[float] = []
current_buy = 0.0
current_sell = 0.0
vpin_series = []
for b, s in buy_sell:
current_buy += b
current_sell += s
if current_buy + current_sell >= volume_bucket_size:
total = current_buy + current_sell
imb = abs(current_buy - current_sell)
buckets.append(imb / total if total > 0 else 0.5)
excess = total - volume_bucket_size
ratio = current_buy / total if total > 0 else 0.5
current_buy = ratio * excess
current_sell = (1 - ratio) * excess
if len(buckets) >= n_buckets:
vpin_series.append(sum(buckets[-n_buckets:]) / n_buckets)
else:
vpin_series.append(sum(buckets) / len(buckets))
a = np.array(vpin_series) if vpin_series else np.array([0.0])
return {
"vpin_values": [round(v, 4) for v in vpin_series],
"mean": round(float(np.mean(a)), 4),
"std": round(float(np.std(a)), 4),
"max": round(float(np.max(a)), 4),
"threshold_alarm": round(float(np.mean(a) + 2 * np.std(a)), 4),
}
# ── Fill toxicity ────────────────────────────────────────────
def fill_toxicity(
trade_prices: list[float],
mids: list[float],
trade_sides: list[str],
horizon_ticks: int = 10,
) -> dict:
"""Compute toxicity per trade: did price move against you after fill?
For each buy trade: toxicity = (mid before - mid after) / mid
For each sell: toxicity = (mid after - mid before) / mid
Positive toxicity = adverse price movement post-trade.
"""
if len(trade_prices) < horizon_ticks + 1:
return {"buy_toxicity_mean": 0, "sell_toxicity_mean": 0, "overall": 0}
buy_tox = []
sell_tox = []
n = len(trade_prices)
for i in range(n - horizon_ticks):
side = trade_sides[i] if i < len(trade_sides) else "unknown"
mid_before = mids[min(i + 1, n - 1)]
mid_after = mids[min(i + horizon_ticks, n - 1)]
if mid_before <= 0 or mid_after <= 0:
continue
change = (mid_before - mid_after) / mid_before
if side == "buy":
buy_tox.append(change)
elif side == "sell":
sell_tox.append(-change)
return {
"buy_toxicity_mean_bps": round(float(np.mean(buy_tox)) * 10000, 2) if buy_tox else 0,
"sell_toxicity_mean_bps": round(float(np.mean(sell_tox)) * 10000, 2) if sell_tox else 0,
"buy_count": len(buy_tox),
"sell_count": len(sell_tox),
"overall_bps": round(float(np.mean(buy_tox + sell_tox)) * 10000, 2) if buy_tox or sell_tox else 0,
}
# ── Adverse selection ───────────────────────────────────────
def adverse_selection_ratio(
mid_after_trades: list[float],
mid_before_trades: list[float],
trade_sides: list[str],
) -> dict:
"""Adverse selection ratio per side (mid after / mid before - 1).
Higher values = more adverse selection (price moves against you).
"""
buys = []
sells = []
for i, side in enumerate(trade_sides):
if i >= len(mid_after_trades) or i >= len(mid_before_trades):
break
before = mid_before_trades[i]
after = mid_after_trades[i]
if before <= 0:
continue
sel = (after - before) / before
if side == "buy":
buys.append(-sel)
elif side == "sell":
sells.append(sel)
ba = np.array(buys) if buys else np.array([0.0])
sa = np.array(sells) if sells else np.array([0.0])
return {
"buy_adverse_bps": round(float(np.mean(ba)) * 10000, 2),
"sell_adverse_bps": round(float(np.mean(sa)) * 10000, 2),
"buy_win_pct": round(np.sum(ba <= 0) / len(ba), 4) if len(ba) > 0 else 0,
"sell_win_pct": round(np.sum(sa <= 0) / len(sa), 4) if len(sa) > 0 else 0,
}
# ── Liquidation clustering ──────────────────────────────────
def liquidation_clustering(
liquidation_times_ms: list[int],
window_sec: int = 300,
) -> dict:
"""Detect liquidation clusters — unusual concentration of liquidations.
Returns cluster periods and intensity.
"""
if len(liquidation_times_ms) < 2:
return {"clusters": [], "mean_interval_s": 0, "clustered_pct": 0}
intervals = [
(liquidation_times_ms[i + 1] - liquidation_times_ms[i]) / 1000
for i in range(len(liquidation_times_ms) - 1)
]
mean_interval = float(np.mean(intervals))
std_interval = float(np.std(intervals))
clusters = []
cluster_start = None
for i, interval in enumerate(intervals):
if interval < mean_interval * 0.3: # Tight clustering threshold
if cluster_start is None:
cluster_start = liquidation_times_ms[i]
else:
if cluster_start is not None and liquidation_times_ms[i] - cluster_start < window_sec * 1000:
clusters.append({
"start_ms": cluster_start,
"end_ms": liquidation_times_ms[i],
"count": i - liquidation_times_ms.index(cluster_start) + 1 if cluster_start in liquidation_times_ms else 0,
})
cluster_start = None
clustered_count = sum(c.get("count", 0) for c in clusters)
return {
"clusters": clusters,
"n_clusters": len(clusters),
"mean_interval_s": round(mean_interval, 2),
"clustered_pct": round(clustered_count / len(liquidation_times_ms), 4) if liquidation_times_ms else 0,
}
+202
View File
@@ -0,0 +1,202 @@
"""
Trade microstructure analytics — aggressor classification and markout curves.
Lee-Ready algorithm for trade direction classification, plus
forward markout analysis: what happens to mid price N seconds after
a trade of a given type.
"""
from __future__ import annotations
import numpy as np
# ── Aggressor classification ─────────────────────────────────
def classify_lee_ready(
trade_px: float,
mid_at_trade: float,
bid_at_trade: float | None = None,
ask_at_trade: float | None = None,
) -> str:
"""Lee-Ready: trade above mid = buy, below mid = sell.
At mid: compare to previous tick (quote rule) — if unavailable,
compare to bid/ask (trade at bid = sell, at ask = buy).
"""
if trade_px > mid_at_trade:
return "buy"
elif trade_px < mid_at_trade:
return "sell"
else:
if ask_at_trade is not None and trade_px >= ask_at_trade:
return "buy"
if bid_at_trade is not None and trade_px <= bid_at_trade:
return "sell"
return "unknown"
def classify_bulk_lee_ready(
trades: list[dict],
mids: list[float] | None = None,
bids: list[float] | None = None,
asks: list[float] | None = None,
) -> list[str]:
"""Classify a list of trades using Lee-Ready.
trades: [{"px": float, ...}, ...]
mids: optional list of mid prices at each trade time
bids/asks: optional best bid/ask at each trade time
"""
results = []
for i, trade in enumerate(trades):
px = float(trade.get("px", 0))
mid = float(mids[i]) if mids and i < len(mids) else px
bid = float(bids[i]) if bids and i < len(bids) else None
ask = float(asks[i]) if asks and i < len(asks) else None
results.append(classify_lee_ready(px, mid, bid, ask))
return results
# ── Markout curves ───────────────────────────────────────────
def compute_markouts(
trades: list[dict],
mid_prices: list[float],
trade_times: list[int], # ms since epoch
horizons_ms: list[int] | None = None,
) -> dict:
"""For each trade, compute mid-price change at specified horizons.
Returns:
{"buys": {horizon_ms: [markout_values...]}, "sells": {...}, ...}
"""
if horizons_ms is None:
horizons_ms = [100, 500, 1000, 5000, 10000, 30000, 60000]
results: dict[str, dict[int, list[float]]] = {
"buy": {h: [] for h in horizons_ms},
"sell": {h: [] for h in horizons_ms},
}
bids_at_trade = []
asks_at_trade = []
mids_at_trade = []
for i, (trade, mid) in enumerate(zip(trades, mid_prices)):
mids_at_trade.append(mid)
bids_at_trade.append(mid * 0.9995 if mid > 0 else 0)
asks_at_trade.append(mid * 1.0005 if mid > 0 else 0)
sides = classify_bulk_lee_ready(trades, mids_at_trade, bids_at_trade, asks_at_trade)
for i, (trade, side, t0) in enumerate(zip(trades, sides, trade_times)):
base_mid = mid_prices[i] if i < len(mid_prices) else 0
if base_mid <= 0:
continue
for horizon in horizons_ms:
target_ts = t0 + horizon
future_mid = base_mid
for j in range(i + 1, len(mid_prices)):
if trade_times[j] >= target_ts:
future_mid = mid_prices[j]
break
else:
if len(mid_prices) > i + 1:
future_mid = mid_prices[-1]
markout = (future_mid - base_mid) / base_mid * 10000 # bps
if side in ("buy", "sell"):
results[side][horizon].append(markout)
return results
def markout_summary(
markouts: dict[str, dict[int, list[float]]],
) -> dict:
"""Summarize markout curves with mean, std, t-stat."""
summary = {}
for side in ("buy", "sell"):
summary[side] = {}
for horizon, vals in markouts.get(side, {}).items():
if not vals:
summary[side][horizon] = {"mean": 0, "std": 0, "t_stat": 0, "count": 0}
continue
a = np.array(vals, dtype=float)
a = a[np.isfinite(a)]
mean = float(np.mean(a))
std = float(np.std(a, ddof=1))
t_stat = mean / std * np.sqrt(len(a)) if std > 0 else 0
summary[side][horizon] = {
"mean_bps": round(mean, 2),
"std_bps": round(std, 2),
"t_stat": round(t_stat, 3),
"count": len(a),
}
return summary
# ── Trade metrics ────────────────────────────────────────────
def trade_volume_profile(
trades: list[dict],
n_buckets: int = 20,
) -> dict:
"""Volume profile: trade count and volume by size bucket."""
sizes = [float(t.get("sz", 0)) for t in trades if float(t.get("sz", 0)) > 0]
if not sizes:
return {"buckets": [], "counts": [], "volumes": []}
min_sz, max_sz = min(sizes), max(sizes)
if min_sz == max_sz:
buckets = [min_sz]
else:
buckets = np.linspace(min_sz, max_sz, n_buckets + 1).tolist()
counts = [0] * n_buckets
volumes = [0.0] * n_buckets
for sz in sizes:
for b in range(n_buckets):
if buckets[b] <= sz < buckets[b + 1] or (b == n_buckets - 1 and sz == buckets[b + 1]):
counts[b] += 1
volumes[b] += sz
break
return {
"buckets": [round((buckets[i] + buckets[i + 1]) / 2, 6) for i in range(n_buckets)],
"counts": counts,
"volumes": [round(v, 6) for v in volumes],
}
def trade_arrival_rate(
trade_times_ms: list[int],
window_sec: int = 60,
) -> dict:
"""Trade arrival intensity (trades per second) over rolling windows."""
if not trade_times_ms:
return {"mean_rate": 0, "max_rate": 0, "burst_count": 0, "rates": []}
t0 = trade_times_ms[0]
rates = []
burst_count = 0
window_ms = window_sec * 1000
for start in range(t0, trade_times_ms[-1], window_ms):
end = start + window_ms
count = sum(1 for t in trade_times_ms if start <= t < end)
rate = count / window_sec
rates.append(rate)
if rate > rates[-2] * 3 if len(rates) > 1 else rate > 10:
burst_count += 1
a = np.array(rates, dtype=float) if rates else np.array([0.0])
return {
"mean_rate": round(float(np.mean(a)), 3),
"max_rate": round(float(np.max(a)), 3),
"std_rate": round(float(np.std(a)), 3),
"burst_count": burst_count,
"rates": [round(r, 3) for r in rates[-100:]],
}