HK Weather Prediction Market Pipeline: WeatherNext + HKO + Polymarket

- Open-Meteo WeatherNext API client for HK forecasts
- HKO public data client (current conditions, 9-day forecast, typhoon warnings)
- HK-specific weather extraction and calibration
- Polymarket market scanning, price discovery, and market creation proposals
- Trading strategy engine: edge detection, Kelly criterion sizing, probability calibration
- End-to-end pipeline with dry-run mode and scheduled runner
- Interactive dashboard with live HK weather + forecasts + trading signals

Dependencies: Python 3.10+, openmeteo-requests, pandas
No API keys needed for dry-run mode.
Polymarket trading requires private key in .env.
This commit is contained in:
ramseshk
2026-08-10 12:48:05 +08:00
commit c93af97059
21 changed files with 2497 additions and 0 deletions
+7
View File
@@ -0,0 +1,7 @@
"""Trading strategy components."""
from .calibrator import ProbabilityCalibrator
from .kelly import KellyCriterion
from .signals import SignalGenerator
__all__ = ["ProbabilityCalibrator", "KellyCriterion", "SignalGenerator"]
+156
View File
@@ -0,0 +1,156 @@
"""Probability calibration for WeatherNext forecasts.
Converts raw model outputs into well-calibrated probabilities
suitable for prediction market trading.
"""
import json
from datetime import datetime
from pathlib import Path
from typing import Dict, Optional, List, Tuple
import numpy as np
class ProbabilityCalibrator:
"""
Calibrate raw model probabilities using historical performance.
Methods:
- Platt scaling (logistic regression on historical outcomes)
- Isotonic regression (non-parametric)
- Ensemble (combine multiple calibration methods)
"""
def __init__(self, calibration_file: str = "data/calibration_history.json"):
self.calibration_file = Path(__file__).parent.parent / calibration_file
self.history: List[Dict] = self._load_history()
self.platt_params: Dict[str, Tuple[float, float]] = {}
self._fit()
def _load_history(self) -> List[Dict]:
"""Load historical forecast vs outcome data."""
if self.calibration_file.exists():
try:
with open(self.calibration_file) as f:
return json.load(f)
except Exception:
return []
return []
def save_history(self):
"""Save calibration history."""
self.calibration_file.parent.mkdir(parents=True, exist_ok=True)
with open(self.calibration_file, "w") as f:
json.dump(self.history, f, indent=2)
def record_outcome(
self,
date: str,
variable: str,
predicted_probability: float,
actual_outcome: bool,
):
"""Record a prediction-outcome pair for future calibration."""
self.history.append({
"date": date,
"variable": variable,
"predicted_probability": predicted_probability,
"actual_outcome": actual_outcome,
"recorded_at": datetime.now().isoformat(),
})
self.save_history()
self._fit() # Re-fit on new data
def _fit(self):
"""Fit Platt scaling parameters from history."""
by_variable: Dict[str, List[Tuple[float, int]]] = {}
for record in self.history:
var = record["variable"]
if var not in by_variable:
by_variable[var] = []
by_variable[var].append((
record["predicted_probability"] / 100.0,
1 if record["actual_outcome"] else 0,
))
for var, data in by_variable.items():
if len(data) >= 5:
try:
from sklearn.linear_model import LogisticRegression
X = np.array([[d[0]] for d in data])
y = np.array([d[1] for d in data])
lr = LogisticRegression()
lr.fit(X, y)
self.platt_params[var] = (lr.coef_[0][0], lr.intercept_[0])
except ImportError:
# Fallback: simple linear correction
self._simple_fit(var, data)
except Exception:
self._simple_fit(var, data)
elif len(data) >= 2:
self._simple_fit(var, data)
def _simple_fit(self, var: str, data: List[Tuple[float, int]]):
"""Simple linear calibration for small datasets."""
probs = np.array([d[0] for d in data])
outcomes = np.array([d[1] for d in data])
mean_prob = probs.mean()
mean_outcome = outcomes.mean()
slope = 1.0
intercept = mean_outcome - mean_prob
self.platt_params[var] = (slope, intercept)
def calibrate(self, variable: str, raw_probability: float) -> float:
"""Calibrate a raw probability (0-100) to a calibrated one."""
x = raw_probability / 100.0
if variable in self.platt_params:
a, b = self.platt_params[variable]
calibrated = 1.0 / (1.0 + np.exp(-(a * x + b)))
return float(np.clip(calibrated * 100.0, 0.5, 99.5))
return float(np.clip(raw_probability, 0.5, 99.5))
def ensemble_calibrate(
self, variable: str, raw_probability: float
) -> Tuple[float, float]:
"""
Return (calibrated_probability, confidence_interval_width).
Confidence width shrinks with more historical data.
"""
cal_prob = self.calibrate(variable, raw_probability)
n_obs = sum(1 for h in self.history if h["variable"] == variable)
if n_obs < 5:
ci_width = 15.0
elif n_obs < 20:
ci_width = 10.0
elif n_obs < 50:
ci_width = 5.0
else:
ci_width = 3.0
return cal_prob, ci_width
def get_calibration_stats(self, variable: str) -> Dict:
"""Get calibration statistics for a variable."""
relevant = [h for h in self.history if h["variable"] == variable]
if not relevant:
return {"n_observations": 0, "brier_score": None, "calibration_error": None}
preds = np.array([h["predicted_probability"] / 100.0 for h in relevant])
outcomes = np.array([1 if h["actual_outcome"] else 0 for h in relevant])
brier = float(np.mean((preds - outcomes) ** 2))
cal_error = float(np.abs(preds.mean() - outcomes.mean()))
return {
"n_observations": len(relevant),
"brier_score": brier,
"calibration_error": cal_error,
"mean_prediction": float(preds.mean() * 100),
"mean_outcome": float(outcomes.mean() * 100),
}
+181
View File
@@ -0,0 +1,181 @@
"""Kelly Criterion position sizing for prediction market betting.
Implements fractional Kelly to control risk while maximizing
log-wealth growth based on model edge vs market implied probability.
"""
import numpy as np
from dataclasses import dataclass
from config import KELLY_FRACTION, MAX_POSITION_USDC
@dataclass
class KellyResult:
"""Result of Kelly sizing calculation."""
full_kelly_fraction: float # Fraction of bankroll to bet (full Kelly)
fractional_kelly: float # Fraction after applying Kelly fraction
size_usdc: float # Absolute bet size in USDC
kelly_active: bool # Whether full Kelly recommends a bet
edge: float # Edge in decimal (not bps)
log_utility: float # Expected log-utility gain
class KellyCriterion:
"""
Kelly Criterion for binary prediction markets.
For binary markets:
f* = p - q / (b)
where:
p = our estimated probability of winning
q = 1 - p
b = net odds received (payout / bet - 1)
In prediction markets:
If we buy YES at price P, and it resolves YES, we get 1.
So b = (1-P)/P if buying YES, or P/(1-P) if buying NO.
"""
def __init__(self, bankroll_usdc: float = 1000.0, fraction: float = KELLY_FRACTION):
self.bankroll = bankroll_usdc
self.fraction = fraction
def size_bet(
self,
our_probability: float, # Our probability (0-100 or 0-1)
market_probability: float, # Market probability (0-100 or 0-1)
side: str = "buy_yes", # "buy_yes" or "buy_no"
max_size: float = MAX_POSITION_USDC,
) -> KellyResult:
"""
Calculate Kelly-optimal bet size.
our_probability: Our model's probability of YES outcome (0-1 or 0-100)
market_probability: Market-implied probability of YES outcome (0-1 or 0-100)
"""
# Normalize to 0-1 range
if our_probability > 1:
our_probability /= 100.0
if market_probability > 1:
market_probability /= 100.0
# Clamp to avoid division by zero or log(0)
our_probability = np.clip(our_probability, 0.001, 0.999)
market_probability = np.clip(market_probability, 0.001, 0.999)
if side == "buy_yes":
# Buy YES: we win 1-P per share at cost P
b = (1.0 - market_probability) / market_probability # Net odds
p = our_probability
q = 1.0 - our_probability
else:
# Buy NO: symmetric
b = market_probability / (1.0 - market_probability)
p = 1.0 - our_probability # We win if NO
q = our_probability
# Kelly formula: f* = (p * b - q) / b = p - q/b
if b > 0:
full_kelly = p - q / b
else:
full_kelly = 0.0
# Edge in decimal
if side == "buy_yes":
edge = our_probability - market_probability
else:
edge = (1.0 - our_probability) - (1.0 - market_probability)
edge = market_probability - our_probability # Same thing
# Only bet when we have positive edge
kelly_active = full_kelly > 0.001
if not kelly_active:
return KellyResult(
full_kelly_fraction=0.0,
fractional_kelly=0.0,
size_usdc=0.0,
kelly_active=False,
edge=edge,
log_utility=0.0,
)
# Apply fraction for safety
fractional_kelly = full_kelly * self.fraction
size_usdc = min(fractional_kelly * self.bankroll, max_size)
# Log utility uses fractions of bankroll
log_utility = self._expected_log_utility(
p_win=our_probability,
market_price=market_probability,
side=side,
bet_fraction=min(fractional_kelly, 0.99) if kelly_active else 0.0,
)
return KellyResult(
full_kelly_fraction=full_kelly,
fractional_kelly=fractional_kelly,
size_usdc=size_usdc,
kelly_active=kelly_active,
edge=edge,
log_utility=log_utility,
)
def compare_sides(
self,
our_probability: float,
market_probability: float,
) -> dict:
"""Compare betting YES vs NO and return the better side."""
yes_result = self.size_bet(our_probability, market_probability, "buy_yes")
no_result = self.size_bet(our_probability, market_probability, "buy_no")
if yes_result.size_usdc > no_result.size_usdc:
return {
"recommended_side": "buy_yes",
"size_usdc": yes_result.size_usdc,
"edge": yes_result.edge,
"log_utility": yes_result.log_utility,
}
else:
return {
"recommended_side": "buy_no",
"size_usdc": no_result.size_usdc,
"edge": no_result.edge,
"log_utility": no_result.log_utility,
}
def update_bankroll(self, new_bankroll: float):
"""Update bankroll after wins/losses."""
self.bankroll = new_bankroll
@staticmethod
def _expected_log_utility(
p_win: float,
market_price: float,
side: str,
bet_fraction: float,
) -> float:
"""Calculate expected log-utility (Kelly criterion) of a fractional bet."""
if bet_fraction <= 0:
return 0.0
if side == "buy_yes":
win_mult = (1.0 - market_price) / market_price
else:
win_mult = market_price / (1.0 - market_price)
# bet_fraction is fraction of bankroll
# Win: bankroll becomes bankroll * (1 + bet_fraction * win_mult)
# Lose: bankroll becomes bankroll * (1 - bet_fraction)
total_after_win = 1.0 + bet_fraction * win_mult
total_after_loss = 1.0 - bet_fraction
if total_after_loss <= 0:
return -999.0
if side == "buy_yes":
return p_win * np.log(max(1e-10, total_after_win)) + (1.0 - p_win) * np.log(max(1e-10, total_after_loss))
else:
return (1.0 - p_win) * np.log(max(1e-10, total_after_win)) + p_win * np.log(max(1e-10, total_after_loss))
+206
View File
@@ -0,0 +1,206 @@
"""Signal generator for HK weather prediction market trading.
Combines model forecasts, probability calibration, and Kelly sizing
to generate trading signals for Polymarket execution.
"""
from datetime import datetime, timedelta
from typing import Optional, Dict, List
from weather.hk_extractor import HKExtractor
from strategy.calibrator import ProbabilityCalibrator
from strategy.kelly import KellyCriterion, KellyResult
from markets.polymarket_client import PolymarketClient
from markets.trader import TradeSignal
from config import MIN_EDGE_BPS
class SignalGenerator:
"""Generate trading signals from weather forecasts and market prices."""
def __init__(
self,
bankroll_usdc: float = 1000.0,
min_edge_bps: float = MIN_EDGE_BPS,
):
self.weather = HKExtractor()
self.polymarket = PolymarketClient()
self.calibrator = ProbabilityCalibrator()
self.kelly = KellyCriterion(bankroll_usdc=bankroll_usdc)
self.min_edge_bps = min_edge_bps
self.signals: List[TradeSignal] = []
def generate_signals(self) -> List[TradeSignal]:
"""Generate all trading signals for available markets."""
self.signals = []
markets = self.polymarket.find_relevant_weather_markets()
if not markets:
print("No relevant markets found on Polymarket")
self._generate_standalone_signals()
return self.signals
for market in markets:
signal = self._analyze_market(market)
if signal:
self.signals.append(signal)
self.signals.sort(key=lambda s: abs(s.edge_bps), reverse=True)
return self.signals
def _analyze_market(self, market: Dict) -> Optional[TradeSignal]:
"""Analyze a single market and generate a signal."""
condition_id = market["condition_id"]
question = market["question"].lower()
if not market.get("active") or market.get("closed"):
return None
if market.get("liquidity", 0) < 50:
return None # Too illiquid
# Get market-implied probability
market_prob = self.polymarket.get_market_implied_probability(condition_id, 0)
if market_prob is None:
return None
# Determine what we're predicting
model_prob, variable = self._get_model_probability(question)
if model_prob is None:
return None
# Calibrate our probability
cal_prob = self.calibrator.calibrate(variable, model_prob)
# Calculate edge
edge_bps = (cal_prob - market_prob) * 100 # Convert to basis points
if abs(edge_bps) < self.min_edge_bps:
return TradeSignal(
market_id=market["id"],
condition_id=condition_id,
question=question,
outcome_index=0,
outcome_label=market["outcomes"][0] if market.get("outcomes") else "Yes",
model_probability=cal_prob,
market_probability=market_prob,
edge_bps=edge_bps,
recommended_size_usdc=0,
max_size_usdc=0,
signal_type="pass",
)
# Determine side
side = "buy_yes" if edge_bps > 0 else "buy_no"
side_prob = cal_prob if side == "buy_yes" else 100 - cal_prob
# Kelly sizing
kelly_result = self.kelly.size_bet(
our_probability=cal_prob,
market_probability=market_prob,
side=side,
)
return TradeSignal(
market_id=market["id"],
condition_id=condition_id,
question=question,
outcome_index=0,
outcome_label=market["outcomes"][0] if market.get("outcomes") else "Yes",
model_probability=cal_prob,
market_probability=market_prob,
edge_bps=edge_bps,
recommended_size_usdc=kelly_result.size_usdc,
max_size_usdc=kelly_result.size_usdc,
signal_type=side,
)
def _get_model_probability(self, question: str) -> tuple:
"""Get our model's probability for a given market question."""
question = question.lower()
if "rain" in question or "precipitation" in question:
prob = self.weather.should_bet_rain_tomorrow()
return (prob, "rain") if prob is not None else (None, "")
if "temperature" in question and ("above" in question or "exceed" in question):
if "30" in question or "thirty" in question:
prob = self.weather.should_bet_temp_above(30.0)
elif "35" in question or "thirty five" in question:
prob = self.weather.should_bet_temp_above(35.0)
else:
prob = self.weather.should_bet_temp_above(33.0)
return (prob, "temperature") if prob is not None else (None, "")
if "typhoon" in question or "t8" in question or "tropical cyclone" in question:
forecast = self.weather.get_hk_forecast()
typhoon = forecast.get("typhoon_info", {})
prob = 30.0 if typhoon else 5.0
return (prob, "typhoon")
if "weather" in question or "storm" in question:
forecast = self.weather.get_combined_tomorrow_forecast()
tomorrow = forecast.get("tomorrow", {})
if tomorrow:
rain_prob = tomorrow.get("precipitation_probability_calibrated", 50)
return (rain_prob, "rain")
return (None, "")
def _generate_standalone_signals(self):
"""Generate signals even when no Polymarket markets exist.
Useful for tracking model predictions and for creating new markets.
"""
tomorrow = datetime.now() + timedelta(days=1)
forecast = self.weather.get_hk_forecast()
consensus = forecast.get("consensus", {})
tmrw = consensus.get("tomorrow", {})
if tmrw:
self.signals.append(TradeSignal(
market_id="standalone",
condition_id="standalone",
question=f"Will it rain in Hong Kong on {tomorrow:%Y-%m-%d}?",
outcome_index=0,
outcome_label="Yes",
model_probability=tmrw.get("precipitation_probability_calibrated", 50),
market_probability=50.0,
edge_bps=0,
recommended_size_usdc=0,
max_size_usdc=0,
signal_type="pass",
))
print(f"\nGenerated {len(self.signals)} standalone signals")
for s in self.signals:
print(f" {s.question} -> P={s.model_probability:.1f}%")
def get_signal_summary(self) -> str:
"""Get a human-readable summary of current signals."""
if not self.signals:
return "No signals generated."
lines = []
active = [s for s in self.signals if s.signal_type != "pass"]
passed = [s for s in self.signals if s.signal_type == "pass"]
lines.append(f"\n=== Signal Summary ({datetime.now():%Y-%m-%d %H:%M}) ===")
lines.append(f"Active signals: {len(active)}")
lines.append(f"Passed (no edge): {len(passed)}")
lines.append("")
if active:
lines.append("TRADE SIGNALS:")
for s in active:
lines.append(f" [{s.signal_type.upper()}] {s.question}")
lines.append(f" Model: {s.model_probability:.1f}% | Market: {s.market_probability:.1f}%")
lines.append(f" Edge: {s.edge_bps:.0f}bps | Size: ${s.recommended_size_usdc:.2f}")
if passed:
lines.append("PASSED (edge < threshold):")
for s in passed[:5]: # Limit to 5
lines.append(f" {s.question} (edge: {s.edge_bps:.0f}bps)")
return "\n".join(lines)