feat: Phase 1 — real-time & historical data system
New data/ module with: - data/store.py: Parquet-based raw message storage with background writer thread. Messages partitioned by channel/coin/date. Thread-safe queue. Supports pyarrow Parquet with zstd compression. Includes read_range() helper for replay. - data/collectors/hyperliquid.py: HL WebSocket + REST collector - WebSocket: l2Book (full book reconstruction), trades, allMids (mark prices) - REST pollers: funding rates, predicted funding, open interest, liquidations - Per-coin OrderBook class with snapshot/update reconstruction - Sequence gap detection with per-coin re-snapshot on gap - Latency tracking (exchange transport, signal, order, roundtrip) - Periodic stats reporter (book stats + latency summary every 60s) - CLI entrypoint: python -m data.collectors.hyperliquid --coins BTC ETH - data/normalizer.py: Timestamp normalization (ms, s, ISO strings from HL/Binance/Bybit/OKX/Coinbase/Deribit) + SequenceTracker with gap detection - data/latency.py: Rolling-window latency metrics (p50/p90/p95/p99) for transport, signal computation, order submission, and roundtrip - Added pyarrow + aiohttp to requirements.txt
This commit is contained in:
@@ -0,0 +1,90 @@
|
||||
"""
|
||||
Exchange vs signal latency tracking.
|
||||
|
||||
Measures:
|
||||
1. Exchange transport latency: exchange_ts → local receipt time
|
||||
2. Signal computation latency: data receipt → signal generated
|
||||
3. Order latency: signal → order accepted on exchange
|
||||
4. Round-trip latency: signal → fill confirmation
|
||||
|
||||
Each metric is tracked as a rolling window with percentiles.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from collections import deque
|
||||
from typing import Optional
|
||||
|
||||
|
||||
class LatencyTracker:
|
||||
"""Track exchange and signal latencies with rolling percentiles."""
|
||||
|
||||
def __init__(self, window_seconds: float = 300.0, max_samples: int = 10000):
|
||||
self._window = window_seconds
|
||||
self._transport: deque[tuple[float, float]] = deque(maxlen=max_samples) # (time, ms)
|
||||
self._signal: deque[tuple[float, float]] = deque(maxlen=max_samples)
|
||||
self._order: deque[tuple[float, float]] = deque(maxlen=max_samples)
|
||||
self._roundtrip: deque[tuple[float, float]] = deque(maxlen=max_samples)
|
||||
|
||||
def record_transport(self, exchange_ts_ms: int, local_ts: float | None = None):
|
||||
"""Exchange timestamp → local receipt (ms)."""
|
||||
local = local_ts or time.time()
|
||||
lat = (local * 1000) - exchange_ts_ms
|
||||
if 0 <= lat < 300_000: # Ignore clock skew > 5 min
|
||||
self._transport.append((time.time(), lat))
|
||||
|
||||
def record_signal(self, duration_ms: float):
|
||||
"""Time from data receipt to signal generation (ms)."""
|
||||
if duration_ms >= 0:
|
||||
self._signal.append((time.time(), duration_ms))
|
||||
|
||||
def record_order(self, duration_ms: float):
|
||||
"""Signal generation → order accepted on exchange (ms)."""
|
||||
if duration_ms >= 0:
|
||||
self._order.append((time.time(), duration_ms))
|
||||
|
||||
def record_roundtrip(self, duration_ms: float):
|
||||
"""Signal generation → fill confirmed (ms)."""
|
||||
if duration_ms >= 0:
|
||||
self._roundtrip.append((time.time(), duration_ms))
|
||||
|
||||
# ── Stats ──────────────────────────────────────────────────
|
||||
|
||||
def stats(self) -> dict:
|
||||
return {
|
||||
"transport_ms": self._percentiles(self._transport),
|
||||
"signal_ms": self._percentiles(self._signal),
|
||||
"order_ms": self._percentiles(self._order),
|
||||
"roundtrip_ms": self._percentiles(self._roundtrip),
|
||||
}
|
||||
|
||||
def summary(self) -> dict:
|
||||
"""Compact summary: just p50/p99 for each metric."""
|
||||
s = self.stats()
|
||||
out = {}
|
||||
for key, pct in s.items():
|
||||
out[key] = {"p50": pct.get("p50", 0), "p99": pct.get("p99", 0)}
|
||||
return out
|
||||
|
||||
# ── Internals ──────────────────────────────────────────────
|
||||
|
||||
def _prune(self, buffer: deque):
|
||||
cutoff = time.time() - self._window
|
||||
while buffer and buffer[0][0] < cutoff:
|
||||
buffer.popleft()
|
||||
|
||||
def _percentiles(self, buffer: deque) -> dict:
|
||||
self._prune(buffer)
|
||||
if not buffer:
|
||||
return {"p50": 0, "p90": 0, "p95": 0, "p99": 0, "count": 0}
|
||||
vals = sorted(v for _, v in buffer)
|
||||
n = len(vals)
|
||||
return {
|
||||
"p50": round(vals[int(n * 0.50)], 2),
|
||||
"p90": round(vals[int(n * 0.90)], 2),
|
||||
"p95": round(vals[int(n * 0.95)], 2),
|
||||
"p99": round(vals[int(n * 0.99)], 2),
|
||||
"max": round(vals[-1], 2),
|
||||
"count": n,
|
||||
}
|
||||
Reference in New Issue
Block a user