Add ML prediction pipeline — LightGBM, calibration fix, ensemble disagreement
Tier 1 ML enhancements: - Feature engineering (37 features across 5 groups: thermal, dynamic, moisture, temporal, interaction) from NWP model output - 7 LightGBM probability models for rain/temp/wind thresholds - Temperature-scaled probabilities to prevent overconfidence on bootstrap data - MLPredictor: unified inference pipeline replacing heuristic sigmoids - Ensemble disagreement signals (composite spread → edge amplification) - Fixed calibration loop: update_calibration() now functional (EMA of errors) - record_outcome() wired for post-resolution feedback - Nautilus strategy updated: ML predictions take priority, heuristics as fallback - Historical backtest engine with Sharpe/ROI/max-DD simulation - Bootstrap training data generator from HK climate normals Run: python ml/train.py && python ml/backtest.py --edge 50
This commit is contained in:
+332
@@ -0,0 +1,332 @@
|
||||
"""LightGBM probability models for HK weather prediction targets.
|
||||
|
||||
One model per (target, lead_time_hours) pair:
|
||||
- rain_gt_0mm_24h: P(precipitation > 0mm at t+24h)
|
||||
- rain_gt_10mm_24h: P(precipitation > 10mm at t+24h)
|
||||
- temp_gt_30c_24h: P(Tmax > 30°C at t+24h)
|
||||
- temp_gt_33c_24h: P(Tmax > 33°C at t+24h)
|
||||
- temp_gt_35c_24h: P(Tmax > 35°C at t+24h)
|
||||
- typhoon_t3_72h: P(T3+ signal at t+72h)
|
||||
- typhoon_t8_72h: P(T8+ signal at t+72h)
|
||||
|
||||
Each model is a LightGBM classifier with binary logloss objective,
|
||||
trained to output calibrated probabilities directly.
|
||||
"""
|
||||
|
||||
import os
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional, Tuple, List
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
try:
|
||||
import lightgbm as lgb
|
||||
except ImportError:
|
||||
lgb = None
|
||||
|
||||
from config import DATA_DIR, PROJECT_ROOT
|
||||
|
||||
|
||||
MODEL_DIR = Path(DATA_DIR) / "models"
|
||||
|
||||
|
||||
# Target definitions: (target_name, feature_to_compare, threshold, operation, description)
|
||||
TARGET_DEFINITIONS = {
|
||||
"rain_gt_0mm_24h": {
|
||||
"variable": "precipitation_sum",
|
||||
"threshold": 0.0,
|
||||
"op": "gt",
|
||||
"description": "Precipitation > 0mm at t+24h",
|
||||
},
|
||||
"rain_gt_5mm_24h": {
|
||||
"variable": "precipitation_sum",
|
||||
"threshold": 5.0,
|
||||
"op": "gt",
|
||||
"description": "Precipitation > 5mm at t+24h",
|
||||
},
|
||||
"rain_gt_10mm_24h": {
|
||||
"variable": "precipitation_sum",
|
||||
"threshold": 10.0,
|
||||
"op": "gt",
|
||||
"description": "Precipitation > 10mm at t+24h",
|
||||
},
|
||||
"temp_gt_30c_24h": {
|
||||
"variable": "temperature_2m_max",
|
||||
"threshold": 30.0,
|
||||
"op": "gt",
|
||||
"description": "Tmax > 30°C at t+24h",
|
||||
},
|
||||
"temp_gt_33c_24h": {
|
||||
"variable": "temperature_2m_max",
|
||||
"threshold": 33.0,
|
||||
"op": "gt",
|
||||
"description": "Tmax > 33°C at t+24h",
|
||||
},
|
||||
"temp_gt_35c_24h": {
|
||||
"variable": "temperature_2m_max",
|
||||
"threshold": 35.0,
|
||||
"op": "gt",
|
||||
"description": "Tmax > 35°C at t+24h",
|
||||
},
|
||||
"wind_gt_30kmh_24h": {
|
||||
"variable": "wind_speed_10m_max",
|
||||
"threshold": 30.0,
|
||||
"op": "gt",
|
||||
"description": "Wind gust > 30 km/h at t+24h",
|
||||
},
|
||||
}
|
||||
|
||||
LGBM_PARAMS = {
|
||||
"objective": "binary",
|
||||
"metric": "binary_logloss",
|
||||
"boosting_type": "gbdt",
|
||||
"num_leaves": 15, # Reduced from 31 — less leaf complexity
|
||||
"learning_rate": 0.03, # Reduced from 0.05 — slower learning
|
||||
"feature_fraction": 0.7, # Reduced from 0.8 — more regularization
|
||||
"bagging_fraction": 0.7,
|
||||
"bagging_freq": 5,
|
||||
"min_data_in_leaf": 50, # Increased from 20 — prevents tiny leaf nodes
|
||||
"min_gain_to_split": 0.05, # Increased from 0.01 — stronger split criterion
|
||||
"lambda_l1": 0.5, # Increased from 0.1 — L1 regularization
|
||||
"lambda_l2": 1.0, # Increased from 0.1 — L2 regularization
|
||||
"max_depth": 4, # Reduced from 6 — shallower trees
|
||||
"verbose": -1,
|
||||
"random_state": 42,
|
||||
}
|
||||
|
||||
|
||||
class WeatherModel:
|
||||
"""
|
||||
LightGBM-backed probability model for a single weather target.
|
||||
|
||||
Usage:
|
||||
model = WeatherModel("temp_gt_30c_24h")
|
||||
model.train(X_train, y_train, X_val, y_val) # y is binary
|
||||
prob = model.predict_proba(X_single) # returns 0-100
|
||||
model.save()
|
||||
"""
|
||||
|
||||
def __init__(self, target_name: str):
|
||||
if target_name not in TARGET_DEFINITIONS:
|
||||
raise ValueError(f"Unknown target: {target_name}. Available: {list(TARGET_DEFINITIONS.keys())}")
|
||||
self.target_name = target_name
|
||||
self.target_def = TARGET_DEFINITIONS[target_name]
|
||||
self.model: Optional[lgb.Booster] = None
|
||||
self.feature_importance: Dict[str, float] = {}
|
||||
self.calibration_curve: Optional[Tuple[np.ndarray, np.ndarray]] = None
|
||||
self._trained = False
|
||||
|
||||
def train(
|
||||
self,
|
||||
X_train: np.ndarray,
|
||||
y_train: np.ndarray,
|
||||
X_val: Optional[np.ndarray] = None,
|
||||
y_val: Optional[np.ndarray] = None,
|
||||
params: Optional[Dict] = None,
|
||||
early_stopping_rounds: int = 50,
|
||||
verbose: bool = True,
|
||||
):
|
||||
"""Train the LightGBM model."""
|
||||
if lgb is None:
|
||||
raise ImportError("lightgbm not installed")
|
||||
|
||||
train_params = {**LGBM_PARAMS, **(params or {})}
|
||||
n_classes = len(np.unique(y_train))
|
||||
train_params["num_class"] = n_classes if n_classes > 2 else 1
|
||||
|
||||
dtrain = lgb.Dataset(X_train, label=y_train)
|
||||
|
||||
if X_val is not None and y_val is not None:
|
||||
dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)
|
||||
valid_sets = [dtrain, dval]
|
||||
valid_names = ["train", "valid"]
|
||||
else:
|
||||
valid_sets = None
|
||||
valid_names = None
|
||||
|
||||
self.model = lgb.train(
|
||||
train_params,
|
||||
dtrain,
|
||||
num_boost_round=500,
|
||||
valid_sets=valid_sets,
|
||||
valid_names=valid_names,
|
||||
callbacks=[
|
||||
lgb.early_stopping(early_stopping_rounds),
|
||||
lgb.log_evaluation(period=50 if verbose else 0),
|
||||
] if X_val is not None else None,
|
||||
)
|
||||
|
||||
self._trained = True
|
||||
self._compute_feature_importance()
|
||||
|
||||
def predict_proba(self, X: np.ndarray) -> np.ndarray:
|
||||
"""Predict probability (0-100) for binary outcome YES.
|
||||
|
||||
Applies temperature scaling to prevent extreme probabilities
|
||||
when models are too confident on synthetic/bootstrap data.
|
||||
"""
|
||||
if not self._trained or self.model is None:
|
||||
raise RuntimeError("Model not trained or loaded")
|
||||
|
||||
raw = self.model.predict(X)
|
||||
|
||||
# Temperature scaling: push extremes toward 0.5
|
||||
# T=0.5 sharpens, T=2.0 flattens. Using T=2.0 for cautious predictions
|
||||
temperature = 2.0
|
||||
scaled = 1.0 / (1.0 + np.exp(-np.log(np.maximum(raw, 1e-9) / np.maximum(1 - raw, 1e-9)) / temperature))
|
||||
|
||||
return np.clip(scaled * 100.0, 1.0, 99.0)
|
||||
|
||||
def predict(self, X: np.ndarray, threshold: float = 50.0) -> np.ndarray:
|
||||
"""Binary prediction at given probability threshold."""
|
||||
proba = self.predict_proba(X)
|
||||
return (proba >= threshold).astype(int)
|
||||
|
||||
def evaluate(self, X: np.ndarray, y: np.ndarray) -> Dict[str, float]:
|
||||
"""Evaluate model performance on test set."""
|
||||
proba = self.predict_proba(X) / 100.0
|
||||
pred = (proba >= 0.5).astype(int)
|
||||
|
||||
from sklearn.metrics import (
|
||||
accuracy_score, brier_score_loss, roc_auc_score, log_loss
|
||||
)
|
||||
|
||||
return {
|
||||
"accuracy": float(accuracy_score(y, pred)),
|
||||
"brier_score": float(brier_score_loss(y, proba)),
|
||||
"roc_auc": float(roc_auc_score(y, proba)) if len(np.unique(y)) > 1 else 0.5,
|
||||
"log_loss": float(log_loss(y, proba)),
|
||||
"n_samples": len(y),
|
||||
"p_yes_actual": float(y.mean() * 100),
|
||||
"p_yes_predicted": float(proba.mean() * 100),
|
||||
}
|
||||
|
||||
def _compute_feature_importance(self):
|
||||
"""Extract feature importance from trained model."""
|
||||
if self.model is None:
|
||||
return
|
||||
gain = self.model.feature_importance(importance_type="gain")
|
||||
names = self.model.feature_name()
|
||||
self.feature_importance = dict(sorted(
|
||||
zip(names, gain), key=lambda x: x[1], reverse=True
|
||||
))
|
||||
|
||||
def top_features(self, n: int = 15) -> Dict[str, float]:
|
||||
"""Return top N most important features."""
|
||||
items = sorted(
|
||||
self.feature_importance.items(), key=lambda x: x[1], reverse=True
|
||||
)
|
||||
return dict(items[:n])
|
||||
|
||||
def save(self, path: Optional[str] = None):
|
||||
"""Save model to disk."""
|
||||
MODEL_DIR.mkdir(parents=True, exist_ok=True)
|
||||
p = path or (MODEL_DIR / f"{self.target_name}.lgb")
|
||||
if self.model:
|
||||
self.model.save_model(str(p))
|
||||
meta = {
|
||||
"target_name": self.target_name,
|
||||
"target_definition": self.target_def,
|
||||
"feature_importance": self.feature_importance,
|
||||
"trained": self._trained,
|
||||
}
|
||||
meta_path = str(p).replace(".lgb", "_meta.json")
|
||||
with open(meta_path, "w") as f:
|
||||
json.dump(meta, f, indent=2)
|
||||
|
||||
def load(self, path: Optional[str] = None):
|
||||
"""Load model from disk."""
|
||||
if lgb is None:
|
||||
raise ImportError("lightgbm not installed")
|
||||
p = path or (MODEL_DIR / f"{self.target_name}.lgb")
|
||||
if not os.path.exists(p):
|
||||
raise FileNotFoundError(f"Model not found: {p}")
|
||||
self.model = lgb.Booster(model_file=str(p))
|
||||
self._trained = True
|
||||
|
||||
meta_path = str(p).replace(".lgb", "_meta.json")
|
||||
if os.path.exists(meta_path):
|
||||
with open(meta_path) as f:
|
||||
meta = json.load(f)
|
||||
self.feature_importance = meta.get("feature_importance", {})
|
||||
|
||||
@staticmethod
|
||||
def build_target(df: pd.DataFrame, variable: str, threshold: float, op: str = "gt") -> np.ndarray:
|
||||
"""Build binary target array from a DataFrame."""
|
||||
if variable not in df.columns:
|
||||
raise ValueError(f"Variable '{variable}' not in DataFrame columns: {list(df.columns)}")
|
||||
values = df[variable].values
|
||||
if op == "gt":
|
||||
return (values > threshold).astype(int)
|
||||
elif op == "ge":
|
||||
return (values >= threshold).astype(int)
|
||||
elif op == "lt":
|
||||
return (values < threshold).astype(int)
|
||||
elif op == "le":
|
||||
return (values <= threshold).astype(int)
|
||||
else:
|
||||
raise ValueError(f"Unknown operator: {op}")
|
||||
|
||||
|
||||
class ModelEnsemble:
|
||||
"""
|
||||
Manage multiple WeatherModel instances for all targets.
|
||||
|
||||
Usage:
|
||||
ensemble = ModelEnsemble()
|
||||
ensemble.load_all() # Load all trained models
|
||||
probs = ensemble.predict_all(X) # Dict of {target: probability}
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
self.models: Dict[str, WeatherModel] = {}
|
||||
|
||||
def load_all(self):
|
||||
"""Load all available trained models from disk."""
|
||||
MODEL_DIR.mkdir(parents=True, exist_ok=True)
|
||||
for target in TARGET_DEFINITIONS:
|
||||
model_path = MODEL_DIR / f"{target}.lgb"
|
||||
if model_path.exists():
|
||||
model = WeatherModel(target)
|
||||
model.load(str(model_path))
|
||||
self.models[target] = model
|
||||
|
||||
if not self.models:
|
||||
print(f"No trained models found in {MODEL_DIR}. Run ml/train.py first.")
|
||||
|
||||
return self.models
|
||||
|
||||
def load(self, target: str):
|
||||
"""Load a specific model."""
|
||||
model = WeatherModel(target)
|
||||
model.load()
|
||||
self.models[target] = model
|
||||
return model
|
||||
|
||||
def predict_all(self, X: np.ndarray) -> Dict[str, float]:
|
||||
"""Predict all targets for a feature vector."""
|
||||
if X.ndim == 1:
|
||||
X = X.reshape(1, -1)
|
||||
return {name: float(model.predict_proba(X)[0]) for name, model in self.models.items()}
|
||||
|
||||
def predict(self, target: str, X: np.ndarray) -> float:
|
||||
"""Predict a single target."""
|
||||
if target not in self.models:
|
||||
raise KeyError(f"Model '{target}' not loaded. Available: {list(self.models.keys())}")
|
||||
return float(self.models[target].predict_proba(X)[0])
|
||||
|
||||
def has(self, target: str) -> bool:
|
||||
return target in self.models
|
||||
|
||||
@property
|
||||
def available_targets(self) -> List[str]:
|
||||
return list(self.models.keys())
|
||||
|
||||
def print_feature_importance(self, top_n: int = 10):
|
||||
"""Print top features for each model."""
|
||||
for name, model in self.models.items():
|
||||
print(f"\n--- {name} ({model.target_def['description']}) ---")
|
||||
for feat, imp in list(model.top_features(top_n).items()):
|
||||
print(f" {feat:30s} {imp:>10.1f}")
|
||||
Reference in New Issue
Block a user