HK Weather Prediction Market Pipeline: WeatherNext + HKO + Polymarket
- Open-Meteo WeatherNext API client for HK forecasts - HKO public data client (current conditions, 9-day forecast, typhoon warnings) - HK-specific weather extraction and calibration - Polymarket market scanning, price discovery, and market creation proposals - Trading strategy engine: edge detection, Kelly criterion sizing, probability calibration - End-to-end pipeline with dry-run mode and scheduled runner - Interactive dashboard with live HK weather + forecasts + trading signals Dependencies: Python 3.10+, openmeteo-requests, pandas No API keys needed for dry-run mode. Polymarket trading requires private key in .env.
This commit is contained in:
@@ -0,0 +1,7 @@
|
||||
"""Trading strategy components."""
|
||||
|
||||
from .calibrator import ProbabilityCalibrator
|
||||
from .kelly import KellyCriterion
|
||||
from .signals import SignalGenerator
|
||||
|
||||
__all__ = ["ProbabilityCalibrator", "KellyCriterion", "SignalGenerator"]
|
||||
@@ -0,0 +1,156 @@
|
||||
"""Probability calibration for WeatherNext forecasts.
|
||||
|
||||
Converts raw model outputs into well-calibrated probabilities
|
||||
suitable for prediction market trading.
|
||||
"""
|
||||
|
||||
import json
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional, List, Tuple
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
class ProbabilityCalibrator:
|
||||
"""
|
||||
Calibrate raw model probabilities using historical performance.
|
||||
|
||||
Methods:
|
||||
- Platt scaling (logistic regression on historical outcomes)
|
||||
- Isotonic regression (non-parametric)
|
||||
- Ensemble (combine multiple calibration methods)
|
||||
"""
|
||||
|
||||
def __init__(self, calibration_file: str = "data/calibration_history.json"):
|
||||
self.calibration_file = Path(__file__).parent.parent / calibration_file
|
||||
self.history: List[Dict] = self._load_history()
|
||||
self.platt_params: Dict[str, Tuple[float, float]] = {}
|
||||
self._fit()
|
||||
|
||||
def _load_history(self) -> List[Dict]:
|
||||
"""Load historical forecast vs outcome data."""
|
||||
if self.calibration_file.exists():
|
||||
try:
|
||||
with open(self.calibration_file) as f:
|
||||
return json.load(f)
|
||||
except Exception:
|
||||
return []
|
||||
return []
|
||||
|
||||
def save_history(self):
|
||||
"""Save calibration history."""
|
||||
self.calibration_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(self.calibration_file, "w") as f:
|
||||
json.dump(self.history, f, indent=2)
|
||||
|
||||
def record_outcome(
|
||||
self,
|
||||
date: str,
|
||||
variable: str,
|
||||
predicted_probability: float,
|
||||
actual_outcome: bool,
|
||||
):
|
||||
"""Record a prediction-outcome pair for future calibration."""
|
||||
self.history.append({
|
||||
"date": date,
|
||||
"variable": variable,
|
||||
"predicted_probability": predicted_probability,
|
||||
"actual_outcome": actual_outcome,
|
||||
"recorded_at": datetime.now().isoformat(),
|
||||
})
|
||||
self.save_history()
|
||||
self._fit() # Re-fit on new data
|
||||
|
||||
def _fit(self):
|
||||
"""Fit Platt scaling parameters from history."""
|
||||
by_variable: Dict[str, List[Tuple[float, int]]] = {}
|
||||
for record in self.history:
|
||||
var = record["variable"]
|
||||
if var not in by_variable:
|
||||
by_variable[var] = []
|
||||
by_variable[var].append((
|
||||
record["predicted_probability"] / 100.0,
|
||||
1 if record["actual_outcome"] else 0,
|
||||
))
|
||||
|
||||
for var, data in by_variable.items():
|
||||
if len(data) >= 5:
|
||||
try:
|
||||
from sklearn.linear_model import LogisticRegression
|
||||
X = np.array([[d[0]] for d in data])
|
||||
y = np.array([d[1] for d in data])
|
||||
lr = LogisticRegression()
|
||||
lr.fit(X, y)
|
||||
self.platt_params[var] = (lr.coef_[0][0], lr.intercept_[0])
|
||||
except ImportError:
|
||||
# Fallback: simple linear correction
|
||||
self._simple_fit(var, data)
|
||||
except Exception:
|
||||
self._simple_fit(var, data)
|
||||
elif len(data) >= 2:
|
||||
self._simple_fit(var, data)
|
||||
|
||||
def _simple_fit(self, var: str, data: List[Tuple[float, int]]):
|
||||
"""Simple linear calibration for small datasets."""
|
||||
probs = np.array([d[0] for d in data])
|
||||
outcomes = np.array([d[1] for d in data])
|
||||
mean_prob = probs.mean()
|
||||
mean_outcome = outcomes.mean()
|
||||
|
||||
slope = 1.0
|
||||
intercept = mean_outcome - mean_prob
|
||||
|
||||
self.platt_params[var] = (slope, intercept)
|
||||
|
||||
def calibrate(self, variable: str, raw_probability: float) -> float:
|
||||
"""Calibrate a raw probability (0-100) to a calibrated one."""
|
||||
x = raw_probability / 100.0
|
||||
|
||||
if variable in self.platt_params:
|
||||
a, b = self.platt_params[variable]
|
||||
calibrated = 1.0 / (1.0 + np.exp(-(a * x + b)))
|
||||
return float(np.clip(calibrated * 100.0, 0.5, 99.5))
|
||||
|
||||
return float(np.clip(raw_probability, 0.5, 99.5))
|
||||
|
||||
def ensemble_calibrate(
|
||||
self, variable: str, raw_probability: float
|
||||
) -> Tuple[float, float]:
|
||||
"""
|
||||
Return (calibrated_probability, confidence_interval_width).
|
||||
Confidence width shrinks with more historical data.
|
||||
"""
|
||||
cal_prob = self.calibrate(variable, raw_probability)
|
||||
|
||||
n_obs = sum(1 for h in self.history if h["variable"] == variable)
|
||||
if n_obs < 5:
|
||||
ci_width = 15.0
|
||||
elif n_obs < 20:
|
||||
ci_width = 10.0
|
||||
elif n_obs < 50:
|
||||
ci_width = 5.0
|
||||
else:
|
||||
ci_width = 3.0
|
||||
|
||||
return cal_prob, ci_width
|
||||
|
||||
def get_calibration_stats(self, variable: str) -> Dict:
|
||||
"""Get calibration statistics for a variable."""
|
||||
relevant = [h for h in self.history if h["variable"] == variable]
|
||||
if not relevant:
|
||||
return {"n_observations": 0, "brier_score": None, "calibration_error": None}
|
||||
|
||||
preds = np.array([h["predicted_probability"] / 100.0 for h in relevant])
|
||||
outcomes = np.array([1 if h["actual_outcome"] else 0 for h in relevant])
|
||||
|
||||
brier = float(np.mean((preds - outcomes) ** 2))
|
||||
cal_error = float(np.abs(preds.mean() - outcomes.mean()))
|
||||
|
||||
return {
|
||||
"n_observations": len(relevant),
|
||||
"brier_score": brier,
|
||||
"calibration_error": cal_error,
|
||||
"mean_prediction": float(preds.mean() * 100),
|
||||
"mean_outcome": float(outcomes.mean() * 100),
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
"""Kelly Criterion position sizing for prediction market betting.
|
||||
|
||||
Implements fractional Kelly to control risk while maximizing
|
||||
log-wealth growth based on model edge vs market implied probability.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
from dataclasses import dataclass
|
||||
|
||||
from config import KELLY_FRACTION, MAX_POSITION_USDC
|
||||
|
||||
|
||||
@dataclass
|
||||
class KellyResult:
|
||||
"""Result of Kelly sizing calculation."""
|
||||
full_kelly_fraction: float # Fraction of bankroll to bet (full Kelly)
|
||||
fractional_kelly: float # Fraction after applying Kelly fraction
|
||||
size_usdc: float # Absolute bet size in USDC
|
||||
kelly_active: bool # Whether full Kelly recommends a bet
|
||||
edge: float # Edge in decimal (not bps)
|
||||
log_utility: float # Expected log-utility gain
|
||||
|
||||
|
||||
class KellyCriterion:
|
||||
"""
|
||||
Kelly Criterion for binary prediction markets.
|
||||
|
||||
For binary markets:
|
||||
f* = p - q / (b)
|
||||
where:
|
||||
p = our estimated probability of winning
|
||||
q = 1 - p
|
||||
b = net odds received (payout / bet - 1)
|
||||
|
||||
In prediction markets:
|
||||
If we buy YES at price P, and it resolves YES, we get 1.
|
||||
So b = (1-P)/P if buying YES, or P/(1-P) if buying NO.
|
||||
"""
|
||||
|
||||
def __init__(self, bankroll_usdc: float = 1000.0, fraction: float = KELLY_FRACTION):
|
||||
self.bankroll = bankroll_usdc
|
||||
self.fraction = fraction
|
||||
|
||||
def size_bet(
|
||||
self,
|
||||
our_probability: float, # Our probability (0-100 or 0-1)
|
||||
market_probability: float, # Market probability (0-100 or 0-1)
|
||||
side: str = "buy_yes", # "buy_yes" or "buy_no"
|
||||
max_size: float = MAX_POSITION_USDC,
|
||||
) -> KellyResult:
|
||||
"""
|
||||
Calculate Kelly-optimal bet size.
|
||||
|
||||
our_probability: Our model's probability of YES outcome (0-1 or 0-100)
|
||||
market_probability: Market-implied probability of YES outcome (0-1 or 0-100)
|
||||
"""
|
||||
# Normalize to 0-1 range
|
||||
if our_probability > 1:
|
||||
our_probability /= 100.0
|
||||
if market_probability > 1:
|
||||
market_probability /= 100.0
|
||||
|
||||
# Clamp to avoid division by zero or log(0)
|
||||
our_probability = np.clip(our_probability, 0.001, 0.999)
|
||||
market_probability = np.clip(market_probability, 0.001, 0.999)
|
||||
|
||||
if side == "buy_yes":
|
||||
# Buy YES: we win 1-P per share at cost P
|
||||
b = (1.0 - market_probability) / market_probability # Net odds
|
||||
p = our_probability
|
||||
q = 1.0 - our_probability
|
||||
else:
|
||||
# Buy NO: symmetric
|
||||
b = market_probability / (1.0 - market_probability)
|
||||
p = 1.0 - our_probability # We win if NO
|
||||
q = our_probability
|
||||
|
||||
# Kelly formula: f* = (p * b - q) / b = p - q/b
|
||||
if b > 0:
|
||||
full_kelly = p - q / b
|
||||
else:
|
||||
full_kelly = 0.0
|
||||
|
||||
# Edge in decimal
|
||||
if side == "buy_yes":
|
||||
edge = our_probability - market_probability
|
||||
else:
|
||||
edge = (1.0 - our_probability) - (1.0 - market_probability)
|
||||
edge = market_probability - our_probability # Same thing
|
||||
|
||||
# Only bet when we have positive edge
|
||||
kelly_active = full_kelly > 0.001
|
||||
|
||||
if not kelly_active:
|
||||
return KellyResult(
|
||||
full_kelly_fraction=0.0,
|
||||
fractional_kelly=0.0,
|
||||
size_usdc=0.0,
|
||||
kelly_active=False,
|
||||
edge=edge,
|
||||
log_utility=0.0,
|
||||
)
|
||||
|
||||
# Apply fraction for safety
|
||||
fractional_kelly = full_kelly * self.fraction
|
||||
size_usdc = min(fractional_kelly * self.bankroll, max_size)
|
||||
|
||||
# Log utility uses fractions of bankroll
|
||||
log_utility = self._expected_log_utility(
|
||||
p_win=our_probability,
|
||||
market_price=market_probability,
|
||||
side=side,
|
||||
bet_fraction=min(fractional_kelly, 0.99) if kelly_active else 0.0,
|
||||
)
|
||||
|
||||
return KellyResult(
|
||||
full_kelly_fraction=full_kelly,
|
||||
fractional_kelly=fractional_kelly,
|
||||
size_usdc=size_usdc,
|
||||
kelly_active=kelly_active,
|
||||
edge=edge,
|
||||
log_utility=log_utility,
|
||||
)
|
||||
|
||||
def compare_sides(
|
||||
self,
|
||||
our_probability: float,
|
||||
market_probability: float,
|
||||
) -> dict:
|
||||
"""Compare betting YES vs NO and return the better side."""
|
||||
yes_result = self.size_bet(our_probability, market_probability, "buy_yes")
|
||||
no_result = self.size_bet(our_probability, market_probability, "buy_no")
|
||||
|
||||
if yes_result.size_usdc > no_result.size_usdc:
|
||||
return {
|
||||
"recommended_side": "buy_yes",
|
||||
"size_usdc": yes_result.size_usdc,
|
||||
"edge": yes_result.edge,
|
||||
"log_utility": yes_result.log_utility,
|
||||
}
|
||||
else:
|
||||
return {
|
||||
"recommended_side": "buy_no",
|
||||
"size_usdc": no_result.size_usdc,
|
||||
"edge": no_result.edge,
|
||||
"log_utility": no_result.log_utility,
|
||||
}
|
||||
|
||||
def update_bankroll(self, new_bankroll: float):
|
||||
"""Update bankroll after wins/losses."""
|
||||
self.bankroll = new_bankroll
|
||||
|
||||
@staticmethod
|
||||
def _expected_log_utility(
|
||||
p_win: float,
|
||||
market_price: float,
|
||||
side: str,
|
||||
bet_fraction: float,
|
||||
) -> float:
|
||||
"""Calculate expected log-utility (Kelly criterion) of a fractional bet."""
|
||||
if bet_fraction <= 0:
|
||||
return 0.0
|
||||
|
||||
if side == "buy_yes":
|
||||
win_mult = (1.0 - market_price) / market_price
|
||||
else:
|
||||
win_mult = market_price / (1.0 - market_price)
|
||||
|
||||
# bet_fraction is fraction of bankroll
|
||||
# Win: bankroll becomes bankroll * (1 + bet_fraction * win_mult)
|
||||
# Lose: bankroll becomes bankroll * (1 - bet_fraction)
|
||||
total_after_win = 1.0 + bet_fraction * win_mult
|
||||
total_after_loss = 1.0 - bet_fraction
|
||||
|
||||
if total_after_loss <= 0:
|
||||
return -999.0
|
||||
|
||||
if side == "buy_yes":
|
||||
return p_win * np.log(max(1e-10, total_after_win)) + (1.0 - p_win) * np.log(max(1e-10, total_after_loss))
|
||||
else:
|
||||
return (1.0 - p_win) * np.log(max(1e-10, total_after_win)) + p_win * np.log(max(1e-10, total_after_loss))
|
||||
@@ -0,0 +1,206 @@
|
||||
"""Signal generator for HK weather prediction market trading.
|
||||
|
||||
Combines model forecasts, probability calibration, and Kelly sizing
|
||||
to generate trading signals for Polymarket execution.
|
||||
"""
|
||||
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Optional, Dict, List
|
||||
|
||||
from weather.hk_extractor import HKExtractor
|
||||
from strategy.calibrator import ProbabilityCalibrator
|
||||
from strategy.kelly import KellyCriterion, KellyResult
|
||||
from markets.polymarket_client import PolymarketClient
|
||||
from markets.trader import TradeSignal
|
||||
from config import MIN_EDGE_BPS
|
||||
|
||||
|
||||
class SignalGenerator:
|
||||
"""Generate trading signals from weather forecasts and market prices."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
bankroll_usdc: float = 1000.0,
|
||||
min_edge_bps: float = MIN_EDGE_BPS,
|
||||
):
|
||||
self.weather = HKExtractor()
|
||||
self.polymarket = PolymarketClient()
|
||||
self.calibrator = ProbabilityCalibrator()
|
||||
self.kelly = KellyCriterion(bankroll_usdc=bankroll_usdc)
|
||||
self.min_edge_bps = min_edge_bps
|
||||
self.signals: List[TradeSignal] = []
|
||||
|
||||
def generate_signals(self) -> List[TradeSignal]:
|
||||
"""Generate all trading signals for available markets."""
|
||||
self.signals = []
|
||||
|
||||
markets = self.polymarket.find_relevant_weather_markets()
|
||||
|
||||
if not markets:
|
||||
print("No relevant markets found on Polymarket")
|
||||
self._generate_standalone_signals()
|
||||
return self.signals
|
||||
|
||||
for market in markets:
|
||||
signal = self._analyze_market(market)
|
||||
if signal:
|
||||
self.signals.append(signal)
|
||||
|
||||
self.signals.sort(key=lambda s: abs(s.edge_bps), reverse=True)
|
||||
return self.signals
|
||||
|
||||
def _analyze_market(self, market: Dict) -> Optional[TradeSignal]:
|
||||
"""Analyze a single market and generate a signal."""
|
||||
condition_id = market["condition_id"]
|
||||
question = market["question"].lower()
|
||||
|
||||
if not market.get("active") or market.get("closed"):
|
||||
return None
|
||||
|
||||
if market.get("liquidity", 0) < 50:
|
||||
return None # Too illiquid
|
||||
|
||||
# Get market-implied probability
|
||||
market_prob = self.polymarket.get_market_implied_probability(condition_id, 0)
|
||||
if market_prob is None:
|
||||
return None
|
||||
|
||||
# Determine what we're predicting
|
||||
model_prob, variable = self._get_model_probability(question)
|
||||
|
||||
if model_prob is None:
|
||||
return None
|
||||
|
||||
# Calibrate our probability
|
||||
cal_prob = self.calibrator.calibrate(variable, model_prob)
|
||||
|
||||
# Calculate edge
|
||||
edge_bps = (cal_prob - market_prob) * 100 # Convert to basis points
|
||||
|
||||
if abs(edge_bps) < self.min_edge_bps:
|
||||
return TradeSignal(
|
||||
market_id=market["id"],
|
||||
condition_id=condition_id,
|
||||
question=question,
|
||||
outcome_index=0,
|
||||
outcome_label=market["outcomes"][0] if market.get("outcomes") else "Yes",
|
||||
model_probability=cal_prob,
|
||||
market_probability=market_prob,
|
||||
edge_bps=edge_bps,
|
||||
recommended_size_usdc=0,
|
||||
max_size_usdc=0,
|
||||
signal_type="pass",
|
||||
)
|
||||
|
||||
# Determine side
|
||||
side = "buy_yes" if edge_bps > 0 else "buy_no"
|
||||
side_prob = cal_prob if side == "buy_yes" else 100 - cal_prob
|
||||
|
||||
# Kelly sizing
|
||||
kelly_result = self.kelly.size_bet(
|
||||
our_probability=cal_prob,
|
||||
market_probability=market_prob,
|
||||
side=side,
|
||||
)
|
||||
|
||||
return TradeSignal(
|
||||
market_id=market["id"],
|
||||
condition_id=condition_id,
|
||||
question=question,
|
||||
outcome_index=0,
|
||||
outcome_label=market["outcomes"][0] if market.get("outcomes") else "Yes",
|
||||
model_probability=cal_prob,
|
||||
market_probability=market_prob,
|
||||
edge_bps=edge_bps,
|
||||
recommended_size_usdc=kelly_result.size_usdc,
|
||||
max_size_usdc=kelly_result.size_usdc,
|
||||
signal_type=side,
|
||||
)
|
||||
|
||||
def _get_model_probability(self, question: str) -> tuple:
|
||||
"""Get our model's probability for a given market question."""
|
||||
question = question.lower()
|
||||
|
||||
if "rain" in question or "precipitation" in question:
|
||||
prob = self.weather.should_bet_rain_tomorrow()
|
||||
return (prob, "rain") if prob is not None else (None, "")
|
||||
|
||||
if "temperature" in question and ("above" in question or "exceed" in question):
|
||||
if "30" in question or "thirty" in question:
|
||||
prob = self.weather.should_bet_temp_above(30.0)
|
||||
elif "35" in question or "thirty five" in question:
|
||||
prob = self.weather.should_bet_temp_above(35.0)
|
||||
else:
|
||||
prob = self.weather.should_bet_temp_above(33.0)
|
||||
return (prob, "temperature") if prob is not None else (None, "")
|
||||
|
||||
if "typhoon" in question or "t8" in question or "tropical cyclone" in question:
|
||||
forecast = self.weather.get_hk_forecast()
|
||||
typhoon = forecast.get("typhoon_info", {})
|
||||
prob = 30.0 if typhoon else 5.0
|
||||
return (prob, "typhoon")
|
||||
|
||||
if "weather" in question or "storm" in question:
|
||||
forecast = self.weather.get_combined_tomorrow_forecast()
|
||||
tomorrow = forecast.get("tomorrow", {})
|
||||
if tomorrow:
|
||||
rain_prob = tomorrow.get("precipitation_probability_calibrated", 50)
|
||||
return (rain_prob, "rain")
|
||||
|
||||
return (None, "")
|
||||
|
||||
def _generate_standalone_signals(self):
|
||||
"""Generate signals even when no Polymarket markets exist.
|
||||
Useful for tracking model predictions and for creating new markets.
|
||||
"""
|
||||
tomorrow = datetime.now() + timedelta(days=1)
|
||||
forecast = self.weather.get_hk_forecast()
|
||||
consensus = forecast.get("consensus", {})
|
||||
tmrw = consensus.get("tomorrow", {})
|
||||
|
||||
if tmrw:
|
||||
self.signals.append(TradeSignal(
|
||||
market_id="standalone",
|
||||
condition_id="standalone",
|
||||
question=f"Will it rain in Hong Kong on {tomorrow:%Y-%m-%d}?",
|
||||
outcome_index=0,
|
||||
outcome_label="Yes",
|
||||
model_probability=tmrw.get("precipitation_probability_calibrated", 50),
|
||||
market_probability=50.0,
|
||||
edge_bps=0,
|
||||
recommended_size_usdc=0,
|
||||
max_size_usdc=0,
|
||||
signal_type="pass",
|
||||
))
|
||||
|
||||
print(f"\nGenerated {len(self.signals)} standalone signals")
|
||||
for s in self.signals:
|
||||
print(f" {s.question} -> P={s.model_probability:.1f}%")
|
||||
|
||||
def get_signal_summary(self) -> str:
|
||||
"""Get a human-readable summary of current signals."""
|
||||
if not self.signals:
|
||||
return "No signals generated."
|
||||
|
||||
lines = []
|
||||
active = [s for s in self.signals if s.signal_type != "pass"]
|
||||
passed = [s for s in self.signals if s.signal_type == "pass"]
|
||||
|
||||
lines.append(f"\n=== Signal Summary ({datetime.now():%Y-%m-%d %H:%M}) ===")
|
||||
lines.append(f"Active signals: {len(active)}")
|
||||
lines.append(f"Passed (no edge): {len(passed)}")
|
||||
lines.append("")
|
||||
|
||||
if active:
|
||||
lines.append("TRADE SIGNALS:")
|
||||
for s in active:
|
||||
lines.append(f" [{s.signal_type.upper()}] {s.question}")
|
||||
lines.append(f" Model: {s.model_probability:.1f}% | Market: {s.market_probability:.1f}%")
|
||||
lines.append(f" Edge: {s.edge_bps:.0f}bps | Size: ${s.recommended_size_usdc:.2f}")
|
||||
|
||||
if passed:
|
||||
lines.append("PASSED (edge < threshold):")
|
||||
for s in passed[:5]: # Limit to 5
|
||||
lines.append(f" {s.question} (edge: {s.edge_bps:.0f}bps)")
|
||||
|
||||
return "\n".join(lines)
|
||||
Reference in New Issue
Block a user