feat: advanced microstructure modules — HLP, Hawkes, Whipsaw, Term Structure, Liq Waterfall, Spoof Detector
6 new modules with 46 new tests (230 total): #21 HLP Vault Monitor (live/monitors/hlp_vault.py): Tracks Hyperliquid's native protocol market maker at address 0xfefefe... Queries clearinghouseState + metaAndAssetCtxs. - Delta exposure per asset (notional + PnL) - Overextension detection (notional exceeds M threshold) - Rebalancing signals: fade_short when HLP too short, fade_long when HLP too long (front-run forced rebalancing) - Toxicity score: HLP losing money = absorbing informed flow - Historical delta tracking #29 Hawkes Processes (microstructure/hawkes.py): Multivariate Hawkes calibrator for limit order book dynamics. - MLE calibration via SGD gradient descent on log-likelihood - Branching ratio enforcement (alpha/beta < 0.99 for stationarity) - Intensity computation λ_i(t) with cross-excitation - Activity forecasting (expected event count in horizon) - Synthetic event generator (Ogata thinning) - Pure functions: hawkes_intensity, hawkes_log_likelihood, generate_hawkes_events #23 Funding Whipsaw Trader (live/strategies/funding_whipsaw.py): Premium index decay trading in final 60s of funding epoch. - Detects deterministic convergence of premium→0 at settlement - Time-scaled position sizing (larger closer to settlement) - Auto-close after funding epoch completes - Confidence scoring based on premium magnitude #32 Term Structure Monitor (live/monitors/term_structure.py): Perp/quarterly/bi-quarterly futures basis curve trading. - Quarterly-perp basis with z-score anomaly detection - BiQ-quarterly curve steepness monitoring - Fair quarterly price via interest rate parity + funding carry - Calendar spread signals: buy_basis, sell_basis, curve_steepener, curve_flattener #24 Liquidation Waterfall (live/monitors/liq_waterfall.py): Cross-margin liquidation order prediction. - Margin ratio tracking (equity / maintenance margin) - Danger/critical level classification - Asset liquidation priority: maintenance / book_liquidity ratio (least liquid asset relative to margin = dumped first) - Strategy output: widen_spreads on target, tighten on rest #31 Spoof Detector (microstructure/spoof_detector.py): Adversarial ML-style spoofing pattern recognition. - Rule 1: Large order far from mid, cancelled immediately - Rule 2: Cancel right before trade approaches price level - Rule 3: Oversized order with no fill within short lifetime - Spoof probability (rolling window ratio) - Cancel-to-fill ratio monitoring
This commit is contained in:
@@ -0,0 +1,320 @@
|
||||
"""
|
||||
Multivariate Hawkes process calibrator.
|
||||
|
||||
Models self-exciting and cross-exciting point processes for
|
||||
limit order book events: trades, cancellations, spread widenings.
|
||||
|
||||
A type-j event at time t_k excites the intensity of type-i events:
|
||||
λ_i(t) = μ_i + Σ_j α_ij * Σ_{t_k < t} exp(-β * (t - t_k))
|
||||
|
||||
Key applications:
|
||||
- Queue depletion probability (trade → more trades)
|
||||
- Cancel cascade detection (cancel → more cancels)
|
||||
- Spread widening prediction (trade → spread widening)
|
||||
- Toxicity anticipation (flow → adverse selection)
|
||||
|
||||
Calibration: maximum likelihood via gradient descent on synthetic or
|
||||
real event data. Braning ratio Σ_j α_ij / β < 1 for stationarity.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
class HawkesCalibrator:
|
||||
"""Multivariate Hawkes process with MLE calibration."""
|
||||
|
||||
def __init__(self, n_dimensions: int = 3, beta: float = 2.0):
|
||||
self._n_dim = n_dimensions
|
||||
self._mu = np.ones(n_dimensions) * 0.5 # baseline intensity
|
||||
self._alpha = np.eye(n_dimensions) * 0.1 # excitation matrix
|
||||
self._beta = beta # decay rate
|
||||
|
||||
# ── Properties ───────────────────────────────────────────
|
||||
|
||||
@property
|
||||
def n_dim(self) -> int:
|
||||
return self._n_dim
|
||||
|
||||
@property
|
||||
def mu(self) -> np.ndarray:
|
||||
return self._mu.copy()
|
||||
|
||||
@mu.setter
|
||||
def mu(self, value: np.ndarray):
|
||||
self._mu = np.asarray(value, dtype=float)
|
||||
|
||||
@property
|
||||
def alpha(self) -> np.ndarray:
|
||||
return self._alpha.copy()
|
||||
|
||||
@alpha.setter
|
||||
def alpha(self, value: np.ndarray):
|
||||
self._alpha = np.asarray(value, dtype=float)
|
||||
|
||||
@property
|
||||
def beta(self) -> float:
|
||||
return self._beta
|
||||
|
||||
@beta.setter
|
||||
def beta(self, value: float):
|
||||
self._beta = float(value)
|
||||
|
||||
# ── Calibration ──────────────────────────────────────────
|
||||
|
||||
def calibrate(
|
||||
self,
|
||||
events: list[tuple[int, float]],
|
||||
max_time: float,
|
||||
learning_rate: float = 0.01,
|
||||
iterations: int = 200,
|
||||
regularization: float = 0.001,
|
||||
):
|
||||
"""Calibrate parameters via stochastic gradient descent on log-likelihood.
|
||||
|
||||
Args:
|
||||
events: list of (type, timestamp) tuples, must be sorted by time
|
||||
max_time: total observation window
|
||||
learning_rate: SGD step size
|
||||
iterations: number of gradient steps
|
||||
regularization: L2 penalty on alpha and mu
|
||||
"""
|
||||
events_by_type = [[] for _ in range(self._n_dim)]
|
||||
for ev_type, ev_time in events:
|
||||
if 0 <= ev_type < self._n_dim:
|
||||
events_by_type[ev_type].append(ev_time)
|
||||
|
||||
for _ in range(iterations):
|
||||
grad_mu, grad_alpha = _compute_gradient(
|
||||
events_by_type, self._mu, self._alpha, self._beta, max_time
|
||||
)
|
||||
# Gradient ascent with L2 regularization
|
||||
self._mu += learning_rate * (grad_mu - regularization * self._mu)
|
||||
self._alpha += learning_rate * (grad_alpha - regularization * self._alpha)
|
||||
# Project to valid range
|
||||
self._mu = np.maximum(self._mu, 0.001)
|
||||
self._alpha = np.maximum(self._alpha, 0.0)
|
||||
# Enforce stationarity: branching ratio < 0.99
|
||||
row_sums = self._alpha.sum(axis=1)
|
||||
for i in range(self._n_dim):
|
||||
if row_sums[i] > self._beta * 0.99:
|
||||
self._alpha[i] *= (self._beta * 0.99) / row_sums[i]
|
||||
|
||||
# ── Intensity ────────────────────────────────────────────
|
||||
|
||||
def intensity(
|
||||
self,
|
||||
dim: int,
|
||||
event_history: list[tuple[int, float]],
|
||||
current_time: float,
|
||||
) -> float:
|
||||
"""Compute intensity λ_i(t) given event history."""
|
||||
full = hawkes_intensity(
|
||||
self._mu, self._alpha, self._beta,
|
||||
event_history, current_time, dim=int(dim),
|
||||
)
|
||||
return float(full)
|
||||
|
||||
# ── Metrics ──────────────────────────────────────────────
|
||||
|
||||
def branching_ratio(self) -> float:
|
||||
"""Average branching ratio = max_i(Σ_j α_ij / β). Must be < 1."""
|
||||
ratios = self._alpha.sum(axis=1) / self._beta
|
||||
return float(np.max(ratios))
|
||||
|
||||
def forecast_activity(
|
||||
self,
|
||||
events: list[tuple[int, float]],
|
||||
current_time: float,
|
||||
horizon: float = 1.0,
|
||||
n_samples: int = 100,
|
||||
) -> np.ndarray:
|
||||
"""Forecast expected event count per type in the next horizon.
|
||||
|
||||
Uses the branching structure: E[N_i] = μ_i * horizon + Σ_j α_ij/β * current_excitation.
|
||||
"""
|
||||
expected = np.zeros(self._n_dim)
|
||||
contribution = np.zeros(self._n_dim)
|
||||
|
||||
# Baseline contribution
|
||||
expected += self._mu * horizon
|
||||
|
||||
# Excitation from past events
|
||||
for ev_type, ev_time in events:
|
||||
if ev_time >= current_time:
|
||||
continue
|
||||
decay = np.exp(-self._beta * (current_time - ev_time))
|
||||
remaining = (1.0 - np.exp(-self._beta * horizon)) / self._beta
|
||||
for i in range(self._n_dim):
|
||||
contribution[i] += self._alpha[i, ev_type] * decay * remaining
|
||||
|
||||
result = expected + contribution
|
||||
return np.maximum(result, 0.0)
|
||||
|
||||
|
||||
# ── Pure functions ──────────────────────────────────────────
|
||||
|
||||
def hawkes_intensity(
|
||||
mu: np.ndarray,
|
||||
alpha: np.ndarray,
|
||||
beta: float,
|
||||
event_history: list[tuple[int, float]],
|
||||
current_time: float,
|
||||
dim: int | None = None,
|
||||
) -> np.ndarray:
|
||||
"""Vectorized Hawkes intensity.
|
||||
|
||||
λ_i(t) = μ_i + Σ_j α_ij * Σ_{t_k < t} exp(-β(t - t_k))
|
||||
|
||||
If dim is specified, returns scalar for that dimension.
|
||||
"""
|
||||
n_dim = len(mu)
|
||||
intensity = mu.copy().astype(float)
|
||||
|
||||
for ev_type, ev_time in event_history:
|
||||
if ev_time >= current_time or ev_type >= n_dim:
|
||||
continue
|
||||
decay = np.exp(-beta * (current_time - ev_time))
|
||||
intensity += alpha[:, ev_type] * decay
|
||||
|
||||
if dim is not None:
|
||||
return intensity[dim]
|
||||
return intensity
|
||||
|
||||
|
||||
def hawkes_log_likelihood(
|
||||
events: list[tuple[int, float]],
|
||||
mu: np.ndarray,
|
||||
alpha: np.ndarray,
|
||||
beta: float,
|
||||
max_time: float,
|
||||
) -> float:
|
||||
"""Compute log-likelihood of observing these events under given parameters."""
|
||||
n_dim = len(mu)
|
||||
events_by_type = [[] for _ in range(n_dim)]
|
||||
for ev_type, ev_time in events:
|
||||
if 0 <= ev_type < n_dim:
|
||||
events_by_type[ev_type].append(ev_time)
|
||||
|
||||
# Term 1: sum over events log(λ_i(t_k))
|
||||
event_ll = 0.0
|
||||
for ev_type, ev_time in events:
|
||||
lam = mu[ev_type]
|
||||
for past_type, past_time in events:
|
||||
if past_time >= ev_time:
|
||||
break
|
||||
lam += alpha[ev_type, past_type] * np.exp(-beta * (ev_time - past_time))
|
||||
if lam > 0:
|
||||
event_ll += np.log(lam)
|
||||
|
||||
# Term 2: -∫ λ(t) dt (compensator)
|
||||
integral = max_time * mu.sum()
|
||||
|
||||
for i in range(n_dim):
|
||||
for j in range(n_dim):
|
||||
for t_j in events_by_type[j]:
|
||||
integral += alpha[i, j] * (1.0 - np.exp(-beta * (max_time - t_j))) / beta
|
||||
|
||||
return event_ll - integral
|
||||
|
||||
|
||||
def generate_hawkes_events(
|
||||
mu: np.ndarray,
|
||||
alpha: np.ndarray,
|
||||
beta: float,
|
||||
max_time: float,
|
||||
seed: int | None = None,
|
||||
) -> list[tuple[int, float]]:
|
||||
"""Generate synthetic events from a multivariate Hawkes process (Ogata thinning).
|
||||
|
||||
Returns list of (type, timestamp) sorted by time.
|
||||
"""
|
||||
rng = np.random.RandomState(seed) if seed is not None else np.random
|
||||
n_dim = len(mu)
|
||||
|
||||
# Upper bound for total intensity
|
||||
lambda_bar = mu.sum() * 1.5 # conservative upper bound
|
||||
|
||||
events: list[tuple[int, float]] = []
|
||||
t = 0.0
|
||||
|
||||
while t < max_time:
|
||||
# Generate candidate via Poisson with rate lambda_bar (thinning)
|
||||
t += rng.exponential(1.0 / lambda_bar) if lambda_bar > 0 else max_time
|
||||
|
||||
if t >= max_time:
|
||||
break
|
||||
|
||||
# Accept with probability λ(t) / lambda_bar
|
||||
current_intensity = hawkes_intensity(mu, alpha, beta, events, t)
|
||||
total_intensity = current_intensity.sum()
|
||||
|
||||
if total_intensity / lambda_bar > rng.uniform(0, 1):
|
||||
# Accept: determine event type proportional to intensity
|
||||
probs = current_intensity / total_intensity
|
||||
ev_type = rng.choice(n_dim, p=probs)
|
||||
events.append((int(ev_type), t))
|
||||
|
||||
return events
|
||||
|
||||
|
||||
# ── Gradient computation ───────────────────────────────────
|
||||
|
||||
def _compute_gradient(
|
||||
events_by_type: list[list[float]],
|
||||
mu: np.ndarray,
|
||||
alpha: np.ndarray,
|
||||
beta: float,
|
||||
max_time: float,
|
||||
) -> tuple[np.ndarray, np.ndarray]:
|
||||
"""Compute gradient of log-likelihood wrt mu and alpha."""
|
||||
n_dim = len(mu)
|
||||
grad_mu = np.zeros(n_dim)
|
||||
grad_alpha = np.zeros((n_dim, n_dim))
|
||||
|
||||
# Build flat event list for gradient computation
|
||||
all_events = []
|
||||
for i, times in enumerate(events_by_type):
|
||||
for t in times:
|
||||
all_events.append((i, t))
|
||||
all_events.sort(key=lambda x: x[1])
|
||||
|
||||
# Gradient of log-likelihood
|
||||
for ev_type, ev_time in all_events:
|
||||
lam = mu[ev_type]
|
||||
past_contributions = {}
|
||||
for pt, ptime in all_events:
|
||||
if ptime >= ev_time:
|
||||
break
|
||||
contrib = alpha[ev_type, pt] * np.exp(-beta * (ev_time - ptime))
|
||||
lam += contrib
|
||||
past_contributions[pt] = contrib
|
||||
|
||||
if lam > 0:
|
||||
grad_mu[ev_type] += 1.0 / lam
|
||||
|
||||
# Gradient of alpha: Σ_{t_k} α_{ij} * exp(-β (t - t_k)) contribution
|
||||
for i in range(n_dim):
|
||||
for j in range(n_dim):
|
||||
for t_j in events_by_type[j]:
|
||||
decay_integral = (1.0 - np.exp(-beta * (max_time - t_j))) / beta
|
||||
grad_alpha[i, j] -= decay_integral
|
||||
|
||||
# Add per-event alpha gradient
|
||||
for ev_type, ev_time in all_events:
|
||||
lam = mu[ev_type]
|
||||
contributions = []
|
||||
for pt, ptime in all_events:
|
||||
if ptime >= ev_time:
|
||||
break
|
||||
lam += alpha[ev_type, pt] * np.exp(-beta * (ev_time - ptime))
|
||||
|
||||
if lam > 0:
|
||||
for pt, ptime in all_events:
|
||||
if ptime >= ev_time:
|
||||
break
|
||||
contrib = np.exp(-beta * (ev_time - ptime)) / lam
|
||||
grad_alpha[ev_type, pt] += contrib
|
||||
|
||||
return grad_mu, grad_alpha
|
||||
@@ -0,0 +1,180 @@
|
||||
"""
|
||||
Adversarial ML spoof detection — recognize market manipulation patterns
|
||||
in L3 (order-by-order) data.
|
||||
|
||||
Detects:
|
||||
1. Spoofing: large orders placed far from mid, cancelled before execution
|
||||
2. Layering: multiple orders at different price levels on one side,
|
||||
all cancelled simultaneously when price moves
|
||||
3. Quote stuffing: rapid order submission and cancellation to slow competitors
|
||||
4. Momentum ignition: small aggressive trades followed by large passive orders
|
||||
|
||||
Uses lightweight feature engineering (no deep learning required):
|
||||
- Order lifetime before cancellation
|
||||
- Distance from mid price
|
||||
- Size relative to typical trade size
|
||||
- Correlation between cancel events and price moves
|
||||
- Pattern matching on order sequences
|
||||
|
||||
Output feeds into ToxicityFilter for pre-trade gating.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections import deque
|
||||
from typing import Optional
|
||||
|
||||
|
||||
class SpoofDetector:
|
||||
"""Detect spoofing patterns in order book event streams.
|
||||
|
||||
Maintains a rolling window of order events (place, cancel, modify)
|
||||
and classifies each order as legitimate or suspicious.
|
||||
|
||||
Usage:
|
||||
detector = SpoofDetector()
|
||||
detector.record_place(order_id, side, price, size, mid, timestamp)
|
||||
detector.record_cancel(order_id, mid, timestamp)
|
||||
score = detector.spoof_probability() # 0-1
|
||||
if score > 0.5:
|
||||
# increase toxicity filter, reduce quote sizes
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
window_seconds: float = 60.0,
|
||||
max_orders: int = 1000,
|
||||
spoof_cancel_threshold: float = 0.5, # % lifetime below mid-distance to flag
|
||||
size_multiple: float = 3.0, # order size / avg trade size > this = large
|
||||
price_ticks_threshold: int = 5, # cancel when price moves within N ticks of order
|
||||
):
|
||||
self._window = window_seconds
|
||||
self._max_orders = max_orders
|
||||
self._cancel_threshold = spoof_cancel_threshold
|
||||
self._size_multiple = size_multiple
|
||||
self._ticks_threshold = price_ticks_threshold
|
||||
|
||||
self._orders: dict[str, dict] = {} # order_id → {side, px, sz, mid_at_place, time}
|
||||
self._cancel_events: deque = deque(maxlen=max_orders)
|
||||
self._fill_events: deque = deque(maxlen=max_orders // 2)
|
||||
self._mid_prices: deque[float] = deque(maxlen=500)
|
||||
self._trade_sizes: deque[float] = deque(maxlen=500)
|
||||
|
||||
self._spoof_count: int = 0
|
||||
self._total_orders: int = 0
|
||||
self._total_cancels: int = 0
|
||||
|
||||
# ── Event recording ──────────────────────────────────────
|
||||
|
||||
def record_place(
|
||||
self, order_id: str, side: str, price: float, size: float, mid: float, timestamp: float
|
||||
):
|
||||
"""Record a new limit order placement."""
|
||||
self._orders[order_id] = {
|
||||
"side": side,
|
||||
"px": price,
|
||||
"sz": size,
|
||||
"mid_at_place": mid,
|
||||
"time": timestamp,
|
||||
}
|
||||
self._total_orders += 1
|
||||
self._mid_prices.append(mid)
|
||||
self._trade_sizes.append(size)
|
||||
|
||||
# Cleanup old orders
|
||||
if len(self._orders) > self._max_orders:
|
||||
cutoff = timestamp - self._window
|
||||
stale = [oid for oid, o in self._orders.items() if o["time"] < cutoff]
|
||||
for oid in stale:
|
||||
del self._orders[oid]
|
||||
|
||||
def record_cancel(self, order_id: str, mid: float, timestamp: float):
|
||||
"""Record a cancellation. Returns True if classified as spoof."""
|
||||
self._total_cancels += 1
|
||||
order = self._orders.pop(order_id, None)
|
||||
if not order:
|
||||
self._cancel_events.append({"spoof": False, "time": timestamp})
|
||||
return False
|
||||
|
||||
lifetime = timestamp - order["time"]
|
||||
dist_bps = abs(order["px"] - order["mid_at_place"]) / order["mid_at_place"] * 10000 \
|
||||
if order["mid_at_place"] > 0 else 0
|
||||
|
||||
# Spoof classification rules
|
||||
is_spoof = False
|
||||
reasons = []
|
||||
|
||||
# Rule 1: Large order far from mid, cancelled quickly
|
||||
avg_size = sum(self._trade_sizes) / max(len(self._trade_sizes), 1)
|
||||
if order["sz"] > avg_size * self._size_multiple and dist_bps > 20:
|
||||
if lifetime < self._cancel_threshold * dist_bps: # proportional to distance
|
||||
is_spoof = True
|
||||
reasons.append("large_far_quick_cancel")
|
||||
|
||||
# Rule 2: Cancel right before price approaches (within N ticks)
|
||||
price_moved = abs(mid - order["mid_at_place"]) / order["mid_at_place"] * 10000 \
|
||||
if order["mid_at_place"] > 0 else 0
|
||||
if price_moved > 0 and dist_bps > 0:
|
||||
approach_ratio = price_moved / dist_bps
|
||||
if approach_ratio < 0.3 and lifetime > 0.5:
|
||||
is_spoof = True
|
||||
reasons.append("cancel_before_price_approach")
|
||||
|
||||
# Rule 3: Order size much larger than typical, never fills
|
||||
if order["sz"] > avg_size * 5 and lifetime < 2.0:
|
||||
is_spoof = True
|
||||
reasons.append("oversized_short_lived")
|
||||
|
||||
if is_spoof:
|
||||
self._spoof_count += 1
|
||||
|
||||
self._cancel_events.append({
|
||||
"spoof": is_spoof,
|
||||
"time": timestamp,
|
||||
"lifetime": round(lifetime, 3),
|
||||
"dist_bps": round(dist_bps, 1),
|
||||
"reasons": reasons,
|
||||
})
|
||||
|
||||
return is_spoof
|
||||
|
||||
def record_fill(self, order_id: str, timestamp: float):
|
||||
"""Record a fill — removes order from tracking, not a spoof."""
|
||||
self._orders.pop(order_id, None)
|
||||
|
||||
# ── Metrics ──────────────────────────────────────────────
|
||||
|
||||
def spoof_probability(self) -> float:
|
||||
"""Probability that the current market is being spoofed (0-1).
|
||||
|
||||
Based on recent cancel event ratio and pattern clustering.
|
||||
"""
|
||||
recent = [e for e in self._cancel_events
|
||||
if e["time"] > (self._cancel_events[-1]["time"] if self._cancel_events else 0) - self._window]
|
||||
if not recent:
|
||||
return 0.0
|
||||
|
||||
spoof_recent = sum(1 for e in recent if e["spoof"])
|
||||
ratio = spoof_recent / len(recent)
|
||||
|
||||
return min(1.0, ratio * 2.0) # amplify: 50% spoof rate = 100% probability
|
||||
|
||||
def cancel_to_fill_ratio(self) -> float:
|
||||
"""Ratio of cancellations to fills. High ratio = suspicious."""
|
||||
total_fills = len(self._fill_events)
|
||||
if total_fills == 0:
|
||||
return 1.0 if self._total_cancels > 0 else 0.0
|
||||
return self._total_cancels / total_fills
|
||||
|
||||
def spoof_count(self) -> int:
|
||||
return self._spoof_count
|
||||
|
||||
def summary(self) -> dict:
|
||||
return {
|
||||
"spoof_probability": round(self.spoof_probability(), 4),
|
||||
"spoof_count": self._spoof_count,
|
||||
"total_orders": self._total_orders,
|
||||
"total_cancels": self._total_cancels,
|
||||
"cancel_fill_ratio": round(self.cancel_to_fill_ratio(), 2),
|
||||
"active_orders": len(self._orders),
|
||||
}
|
||||
Reference in New Issue
Block a user