Add ML prediction pipeline — LightGBM, calibration fix, ensemble disagreement

Tier 1 ML enhancements:
- Feature engineering (37 features across 5 groups: thermal, dynamic,
  moisture, temporal, interaction) from NWP model output
- 7 LightGBM probability models for rain/temp/wind thresholds
- Temperature-scaled probabilities to prevent overconfidence on bootstrap data
- MLPredictor: unified inference pipeline replacing heuristic sigmoids
- Ensemble disagreement signals (composite spread → edge amplification)
- Fixed calibration loop: update_calibration() now functional (EMA of errors)
- record_outcome() wired for post-resolution feedback
- Nautilus strategy updated: ML predictions take priority, heuristics as fallback
- Historical backtest engine with Sharpe/ROI/max-DD simulation
- Bootstrap training data generator from HK climate normals

Run: python ml/train.py && python ml/backtest.py --edge 50
This commit is contained in:
ramseshk
2026-08-10 17:50:07 +08:00
parent 533939d178
commit 7d7a67bd20
9 changed files with 1713 additions and 16 deletions
+332
View File
@@ -0,0 +1,332 @@
"""LightGBM probability models for HK weather prediction targets.
One model per (target, lead_time_hours) pair:
- rain_gt_0mm_24h: P(precipitation > 0mm at t+24h)
- rain_gt_10mm_24h: P(precipitation > 10mm at t+24h)
- temp_gt_30c_24h: P(Tmax > 30°C at t+24h)
- temp_gt_33c_24h: P(Tmax > 33°C at t+24h)
- temp_gt_35c_24h: P(Tmax > 35°C at t+24h)
- typhoon_t3_72h: P(T3+ signal at t+72h)
- typhoon_t8_72h: P(T8+ signal at t+72h)
Each model is a LightGBM classifier with binary logloss objective,
trained to output calibrated probabilities directly.
"""
import os
import json
from pathlib import Path
from typing import Dict, Optional, Tuple, List
import numpy as np
import pandas as pd
try:
import lightgbm as lgb
except ImportError:
lgb = None
from config import DATA_DIR, PROJECT_ROOT
MODEL_DIR = Path(DATA_DIR) / "models"
# Target definitions: (target_name, feature_to_compare, threshold, operation, description)
TARGET_DEFINITIONS = {
"rain_gt_0mm_24h": {
"variable": "precipitation_sum",
"threshold": 0.0,
"op": "gt",
"description": "Precipitation > 0mm at t+24h",
},
"rain_gt_5mm_24h": {
"variable": "precipitation_sum",
"threshold": 5.0,
"op": "gt",
"description": "Precipitation > 5mm at t+24h",
},
"rain_gt_10mm_24h": {
"variable": "precipitation_sum",
"threshold": 10.0,
"op": "gt",
"description": "Precipitation > 10mm at t+24h",
},
"temp_gt_30c_24h": {
"variable": "temperature_2m_max",
"threshold": 30.0,
"op": "gt",
"description": "Tmax > 30°C at t+24h",
},
"temp_gt_33c_24h": {
"variable": "temperature_2m_max",
"threshold": 33.0,
"op": "gt",
"description": "Tmax > 33°C at t+24h",
},
"temp_gt_35c_24h": {
"variable": "temperature_2m_max",
"threshold": 35.0,
"op": "gt",
"description": "Tmax > 35°C at t+24h",
},
"wind_gt_30kmh_24h": {
"variable": "wind_speed_10m_max",
"threshold": 30.0,
"op": "gt",
"description": "Wind gust > 30 km/h at t+24h",
},
}
LGBM_PARAMS = {
"objective": "binary",
"metric": "binary_logloss",
"boosting_type": "gbdt",
"num_leaves": 15, # Reduced from 31 — less leaf complexity
"learning_rate": 0.03, # Reduced from 0.05 — slower learning
"feature_fraction": 0.7, # Reduced from 0.8 — more regularization
"bagging_fraction": 0.7,
"bagging_freq": 5,
"min_data_in_leaf": 50, # Increased from 20 — prevents tiny leaf nodes
"min_gain_to_split": 0.05, # Increased from 0.01 — stronger split criterion
"lambda_l1": 0.5, # Increased from 0.1 — L1 regularization
"lambda_l2": 1.0, # Increased from 0.1 — L2 regularization
"max_depth": 4, # Reduced from 6 — shallower trees
"verbose": -1,
"random_state": 42,
}
class WeatherModel:
"""
LightGBM-backed probability model for a single weather target.
Usage:
model = WeatherModel("temp_gt_30c_24h")
model.train(X_train, y_train, X_val, y_val) # y is binary
prob = model.predict_proba(X_single) # returns 0-100
model.save()
"""
def __init__(self, target_name: str):
if target_name not in TARGET_DEFINITIONS:
raise ValueError(f"Unknown target: {target_name}. Available: {list(TARGET_DEFINITIONS.keys())}")
self.target_name = target_name
self.target_def = TARGET_DEFINITIONS[target_name]
self.model: Optional[lgb.Booster] = None
self.feature_importance: Dict[str, float] = {}
self.calibration_curve: Optional[Tuple[np.ndarray, np.ndarray]] = None
self._trained = False
def train(
self,
X_train: np.ndarray,
y_train: np.ndarray,
X_val: Optional[np.ndarray] = None,
y_val: Optional[np.ndarray] = None,
params: Optional[Dict] = None,
early_stopping_rounds: int = 50,
verbose: bool = True,
):
"""Train the LightGBM model."""
if lgb is None:
raise ImportError("lightgbm not installed")
train_params = {**LGBM_PARAMS, **(params or {})}
n_classes = len(np.unique(y_train))
train_params["num_class"] = n_classes if n_classes > 2 else 1
dtrain = lgb.Dataset(X_train, label=y_train)
if X_val is not None and y_val is not None:
dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)
valid_sets = [dtrain, dval]
valid_names = ["train", "valid"]
else:
valid_sets = None
valid_names = None
self.model = lgb.train(
train_params,
dtrain,
num_boost_round=500,
valid_sets=valid_sets,
valid_names=valid_names,
callbacks=[
lgb.early_stopping(early_stopping_rounds),
lgb.log_evaluation(period=50 if verbose else 0),
] if X_val is not None else None,
)
self._trained = True
self._compute_feature_importance()
def predict_proba(self, X: np.ndarray) -> np.ndarray:
"""Predict probability (0-100) for binary outcome YES.
Applies temperature scaling to prevent extreme probabilities
when models are too confident on synthetic/bootstrap data.
"""
if not self._trained or self.model is None:
raise RuntimeError("Model not trained or loaded")
raw = self.model.predict(X)
# Temperature scaling: push extremes toward 0.5
# T=0.5 sharpens, T=2.0 flattens. Using T=2.0 for cautious predictions
temperature = 2.0
scaled = 1.0 / (1.0 + np.exp(-np.log(np.maximum(raw, 1e-9) / np.maximum(1 - raw, 1e-9)) / temperature))
return np.clip(scaled * 100.0, 1.0, 99.0)
def predict(self, X: np.ndarray, threshold: float = 50.0) -> np.ndarray:
"""Binary prediction at given probability threshold."""
proba = self.predict_proba(X)
return (proba >= threshold).astype(int)
def evaluate(self, X: np.ndarray, y: np.ndarray) -> Dict[str, float]:
"""Evaluate model performance on test set."""
proba = self.predict_proba(X) / 100.0
pred = (proba >= 0.5).astype(int)
from sklearn.metrics import (
accuracy_score, brier_score_loss, roc_auc_score, log_loss
)
return {
"accuracy": float(accuracy_score(y, pred)),
"brier_score": float(brier_score_loss(y, proba)),
"roc_auc": float(roc_auc_score(y, proba)) if len(np.unique(y)) > 1 else 0.5,
"log_loss": float(log_loss(y, proba)),
"n_samples": len(y),
"p_yes_actual": float(y.mean() * 100),
"p_yes_predicted": float(proba.mean() * 100),
}
def _compute_feature_importance(self):
"""Extract feature importance from trained model."""
if self.model is None:
return
gain = self.model.feature_importance(importance_type="gain")
names = self.model.feature_name()
self.feature_importance = dict(sorted(
zip(names, gain), key=lambda x: x[1], reverse=True
))
def top_features(self, n: int = 15) -> Dict[str, float]:
"""Return top N most important features."""
items = sorted(
self.feature_importance.items(), key=lambda x: x[1], reverse=True
)
return dict(items[:n])
def save(self, path: Optional[str] = None):
"""Save model to disk."""
MODEL_DIR.mkdir(parents=True, exist_ok=True)
p = path or (MODEL_DIR / f"{self.target_name}.lgb")
if self.model:
self.model.save_model(str(p))
meta = {
"target_name": self.target_name,
"target_definition": self.target_def,
"feature_importance": self.feature_importance,
"trained": self._trained,
}
meta_path = str(p).replace(".lgb", "_meta.json")
with open(meta_path, "w") as f:
json.dump(meta, f, indent=2)
def load(self, path: Optional[str] = None):
"""Load model from disk."""
if lgb is None:
raise ImportError("lightgbm not installed")
p = path or (MODEL_DIR / f"{self.target_name}.lgb")
if not os.path.exists(p):
raise FileNotFoundError(f"Model not found: {p}")
self.model = lgb.Booster(model_file=str(p))
self._trained = True
meta_path = str(p).replace(".lgb", "_meta.json")
if os.path.exists(meta_path):
with open(meta_path) as f:
meta = json.load(f)
self.feature_importance = meta.get("feature_importance", {})
@staticmethod
def build_target(df: pd.DataFrame, variable: str, threshold: float, op: str = "gt") -> np.ndarray:
"""Build binary target array from a DataFrame."""
if variable not in df.columns:
raise ValueError(f"Variable '{variable}' not in DataFrame columns: {list(df.columns)}")
values = df[variable].values
if op == "gt":
return (values > threshold).astype(int)
elif op == "ge":
return (values >= threshold).astype(int)
elif op == "lt":
return (values < threshold).astype(int)
elif op == "le":
return (values <= threshold).astype(int)
else:
raise ValueError(f"Unknown operator: {op}")
class ModelEnsemble:
"""
Manage multiple WeatherModel instances for all targets.
Usage:
ensemble = ModelEnsemble()
ensemble.load_all() # Load all trained models
probs = ensemble.predict_all(X) # Dict of {target: probability}
"""
def __init__(self):
self.models: Dict[str, WeatherModel] = {}
def load_all(self):
"""Load all available trained models from disk."""
MODEL_DIR.mkdir(parents=True, exist_ok=True)
for target in TARGET_DEFINITIONS:
model_path = MODEL_DIR / f"{target}.lgb"
if model_path.exists():
model = WeatherModel(target)
model.load(str(model_path))
self.models[target] = model
if not self.models:
print(f"No trained models found in {MODEL_DIR}. Run ml/train.py first.")
return self.models
def load(self, target: str):
"""Load a specific model."""
model = WeatherModel(target)
model.load()
self.models[target] = model
return model
def predict_all(self, X: np.ndarray) -> Dict[str, float]:
"""Predict all targets for a feature vector."""
if X.ndim == 1:
X = X.reshape(1, -1)
return {name: float(model.predict_proba(X)[0]) for name, model in self.models.items()}
def predict(self, target: str, X: np.ndarray) -> float:
"""Predict a single target."""
if target not in self.models:
raise KeyError(f"Model '{target}' not loaded. Available: {list(self.models.keys())}")
return float(self.models[target].predict_proba(X)[0])
def has(self, target: str) -> bool:
return target in self.models
@property
def available_targets(self) -> List[str]:
return list(self.models.keys())
def print_feature_importance(self, top_n: int = 10):
"""Print top features for each model."""
for name, model in self.models.items():
print(f"\n--- {name} ({model.target_def['description']}) ---")
for feat, imp in list(model.top_features(top_n).items()):
print(f" {feat:30s} {imp:>10.1f}")