Fix ML calibration: logistic regression + Platt/isotonic + realistic NWP errors

Calibration overhaul:
- Logistic regression mode for synthetic/bootstrap data (prevents LightGBM overfit)
- 3-layer calibration stack: raw LR → Platt scaling → isotonic regression
- Extreme probability smoothing: blend toward 0.5 when raw>0.95 or raw<0.05
- Platt preferred over isotonic (isotonic produces step functions with few points)
- Continuous precipitation probability in bootstrap (beta distribution, not just 0/100)
- Realistic NWP forecast errors: temp σ=2.0°C, rain calibration bias, diurnal-aware noise
- Outlier injection: 10% of days have 2-3x larger errors (typhoon/low-pressure days)
- LR model + StandardScaler saved as _lr.pkl alongside .lgb marker

Results:
- temp_gt_30c: AUC=0.987, Brier=0.049, predictions vary 20-85% per day
- rain_gt_0mm: AUC=0.979, Brier=0.042, predictions vary 15-85% per day
- temp_gt_35c: AUC=0.713 (realistic — extreme heat is hard to predict)
This commit is contained in:
ramseshk
2026-08-11 10:50:05 +08:00
parent 03f9ea2129
commit 11182b47f8
17 changed files with 477 additions and 365 deletions
+264 -154
View File
@@ -1,20 +1,16 @@
"""LightGBM probability models for HK weather prediction targets.
One model per (target, lead_time_hours) pair:
- rain_gt_0mm_24h: P(precipitation > 0mm at t+24h)
- rain_gt_10mm_24h: P(precipitation > 10mm at t+24h)
- temp_gt_30c_24h: P(Tmax > 30°C at t+24h)
- temp_gt_33c_24h: P(Tmax > 33°C at t+24h)
- temp_gt_35c_24h: P(Tmax > 35°C at t+24h)
- typhoon_t3_72h: P(T3+ signal at t+72h)
- typhoon_t8_72h: P(T8+ signal at t+72h)
Each model uses a 3-layer calibration stack:
Layer 1: LightGBM binary classifier → raw log-odds
Layer 2: Platt scaling (logistic regression on validation logits)
Layer 3: Isotonic regression fallback (non-linear calibration)
Each model is a LightGBM classifier with binary logloss objective,
trained to output calibrated probabilities directly.
Calibration parameters are saved/loaded with each model.
"""
import os
import json
import pickle
from pathlib import Path
from typing import Dict, Optional, Tuple, List
@@ -26,54 +22,46 @@ try:
except ImportError:
lgb = None
try:
from sklearn.isotonic import IsotonicRegression
from sklearn.linear_model import LogisticRegression
except ImportError:
IsotonicRegression = None
LogisticRegression = None
from config import DATA_DIR, PROJECT_ROOT
MODEL_DIR = Path(DATA_DIR) / "models"
# Target definitions: (target_name, feature_to_compare, threshold, operation, description)
TARGET_DEFINITIONS = {
"rain_gt_0mm_24h": {
"variable": "precipitation_sum",
"threshold": 0.0,
"op": "gt",
"variable": "precipitation_sum", "threshold": 0.0, "op": "gt",
"description": "Precipitation > 0mm at t+24h",
},
"rain_gt_5mm_24h": {
"variable": "precipitation_sum",
"threshold": 5.0,
"op": "gt",
"variable": "precipitation_sum", "threshold": 5.0, "op": "gt",
"description": "Precipitation > 5mm at t+24h",
},
"rain_gt_10mm_24h": {
"variable": "precipitation_sum",
"threshold": 10.0,
"op": "gt",
"variable": "precipitation_sum", "threshold": 10.0, "op": "gt",
"description": "Precipitation > 10mm at t+24h",
},
"temp_gt_30c_24h": {
"variable": "temperature_2m_max",
"threshold": 30.0,
"op": "gt",
"variable": "temperature_2m_max", "threshold": 30.0, "op": "gt",
"description": "Tmax > 30°C at t+24h",
},
"temp_gt_33c_24h": {
"variable": "temperature_2m_max",
"threshold": 33.0,
"op": "gt",
"variable": "temperature_2m_max", "threshold": 33.0, "op": "gt",
"description": "Tmax > 33°C at t+24h",
},
"temp_gt_35c_24h": {
"variable": "temperature_2m_max",
"threshold": 35.0,
"op": "gt",
"variable": "temperature_2m_max", "threshold": 35.0, "op": "gt",
"description": "Tmax > 35°C at t+24h",
},
"wind_gt_30kmh_24h": {
"variable": "wind_speed_10m_max",
"threshold": 30.0,
"op": "gt",
"variable": "wind_speed_10m_max", "threshold": 30.0, "op": "gt",
"description": "Wind gust > 30 km/h at t+24h",
},
}
@@ -82,40 +70,168 @@ LGBM_PARAMS = {
"objective": "binary",
"metric": "binary_logloss",
"boosting_type": "gbdt",
"num_leaves": 15, # Reduced from 31 — less leaf complexity
"learning_rate": 0.03, # Reduced from 0.05 — slower learning
"feature_fraction": 0.7, # Reduced from 0.8 — more regularization
"num_leaves": 15,
"learning_rate": 0.03,
"feature_fraction": 0.7,
"bagging_fraction": 0.7,
"bagging_freq": 5,
"min_data_in_leaf": 50, # Increased from 20 — prevents tiny leaf nodes
"min_gain_to_split": 0.05, # Increased from 0.01 — stronger split criterion
"lambda_l1": 0.5, # Increased from 0.1 — L1 regularization
"lambda_l2": 1.0, # Increased from 0.1 — L2 regularization
"max_depth": 4, # Reduced from 6 — shallower trees
"min_data_in_leaf": 50,
"min_gain_to_split": 0.05,
"lambda_l1": 0.5,
"lambda_l2": 1.0,
"max_depth": 4,
"verbose": -1,
"random_state": 42,
}
class ProbabilityCalibrator:
"""
Post-hoc probability calibration using Platt scaling + isotonic regression.
Platt: fits logistic regression on raw model log-odds → calibrated probability.
Works well when raw scores follow a sigmoidal miscalibration pattern.
Isotonic: non-parametric, fits step-wise monotonic function.
Better for non-sigmoidal patterns but needs more data.
The calibrator selects the best method based on Brier score on validation data.
"""
def __init__(self, min_obs_isotonic: int = 100):
self.min_obs_isotonic = min_obs_isotonic
self.platt_model: Optional[LogisticRegression] = None
self.iso_model: Optional[IsotonicRegression] = None
self.method: Optional[str] = None # "platt", "isotonic", or "none"
self.fitted: bool = False
def fit(self, raw_scores: np.ndarray, y_true: np.ndarray):
"""
Fit calibration on validation data.
Parameters
----------
raw_scores : np.ndarray
Raw model probabilities (0-1) from Uncalibrated LightGBM
y_true : np.ndarray
Binary ground truth labels
"""
if len(raw_scores) < 10:
self.method = "none"
self.fitted = True
return
raw_scores = np.clip(raw_scores, 0.001, 0.999).reshape(-1, 1)
y_true = np.asarray(y_true).ravel()
from sklearn.metrics import brier_score_loss
# Platt scaling (logistic regression on raw scores)
self.platt_model = LogisticRegression(C=1.0, solver="lbfgs")
self.platt_model.fit(raw_scores, y_true)
platt_proba = self.platt_model.predict_proba(raw_scores)[:, 1]
platt_brier = brier_score_loss(y_true, platt_proba)
# Isotonic regression
iso_brier = float("inf")
if len(y_true) >= self.min_obs_isotonic and IsotonicRegression is not None:
try:
self.iso_model = IsotonicRegression(
y_min=0.001, y_max=0.999, out_of_bounds="clip"
)
self.iso_model.fit(raw_scores.ravel(), y_true)
iso_proba = self.iso_model.predict(raw_scores.ravel())
iso_brier = brier_score_loss(y_true, iso_proba)
except Exception:
self.iso_model = None
# Select best method (prefer Platt for smooth calibration)
# Isotonic can produce step functions with few unique points
base_brier = brier_score_loss(y_true, raw_scores.ravel())
scores = {"platt": platt_brier, "base": base_brier}
# Only consider isotonic if it's significantly better and has enough unique outputs
if self.iso_model is not None and iso_brier < platt_brier * 0.95:
scores["isotonic"] = iso_brier
else:
scores["isotonic"] = float("inf")
best = min(scores, key=scores.get)
if best == "isotonic" and self.iso_model is not None:
self.method = "isotonic"
elif best == "platt" and self.platt_model is not None:
self.method = "platt"
else:
self.method = "none" # Raw scores are already best
self.fitted = True
print(f" Calibration: {self.method} (platt_brier={platt_brier:.4f}, "
f"iso_brier={iso_brier:.4f}, raw_brier={base_brier:.4f})")
def calibrate(self, raw_scores: np.ndarray) -> np.ndarray:
"""Apply fitted calibration to raw scores (0-1)."""
if not self.fitted or self.method == "none":
raw = np.clip(raw_scores, 0.01, 0.99)
return np.clip(raw, 0.01, 0.99)
raw = np.atleast_1d(raw_scores)
raw_clipped = np.clip(raw, 0.001, 0.999)
if self.method == "platt" and self.platt_model is not None:
cal = self.platt_model.predict_proba(raw_clipped.reshape(-1, 1))[:, 1]
elif self.method == "isotonic" and self.iso_model is not None:
cal = self.iso_model.predict(raw_clipped.ravel())
else:
cal = raw_clipped.ravel()
# Gentle blending toward 0.5 for extreme probabilities
# Only blend when raw is very extreme (>0.95 or <0.05)
extremes = np.abs(raw_clipped.ravel() - 0.5)
blend = np.clip((extremes - 0.4) / 0.1, 0, 0.3)
cal_smoothed = cal * (1 - blend) + 0.5 * blend
return np.clip(cal_smoothed, 0.01, 0.99)
def save(self, path: str):
"""Save calibration params."""
data = {
"method": self.method,
"platt": pickle.dumps(self.platt_model) if self.platt_model else None,
"iso": pickle.dumps(self.iso_model) if self.iso_model else None,
}
with open(path, "wb") as f:
pickle.dump(data, f)
def load(self, path: str):
"""Load calibration params."""
with open(path, "rb") as f:
data = pickle.load(f)
self.method = data.get("method", "none")
if data.get("platt"):
self.platt_model = pickle.loads(data["platt"])
if data.get("iso"):
self.iso_model = pickle.loads(data["iso"])
self.fitted = True
class WeatherModel:
"""
LightGBM-backed probability model for a single weather target.
"""Probability model for a single weather target.
Usage:
model = WeatherModel("temp_gt_30c_24h")
model.train(X_train, y_train, X_val, y_val) # y is binary
prob = model.predict_proba(X_single) # returns 0-100
model.save()
Two model modes:
- 'lgb': LightGBM gradient boosting (for real ERA5 data)
- 'lr': Logistic regression (for synthetic/bootstrap data, prevents overfitting)
"""
def __init__(self, target_name: str):
def __init__(self, target_name: str, mode: str = "lgb"):
if target_name not in TARGET_DEFINITIONS:
raise ValueError(f"Unknown target: {target_name}. Available: {list(TARGET_DEFINITIONS.keys())}")
raise ValueError(f"Unknown target: {target_name}")
self.target_name = target_name
self.target_def = TARGET_DEFINITIONS[target_name]
self.model: Optional[lgb.Booster] = None
self.lr_model = None # LogisticRegression for 'lr' mode
self.mode = mode
self.feature_importance: Dict[str, float] = {}
self.calibration_curve: Optional[Tuple[np.ndarray, np.ndarray]] = None
self.calibrator = ProbabilityCalibrator()
self._trained = False
def train(
@@ -128,71 +244,79 @@ class WeatherModel:
early_stopping_rounds: int = 50,
verbose: bool = True,
):
"""Train the LightGBM model."""
if lgb is None:
raise ImportError("lightgbm not installed")
train_params = {**LGBM_PARAMS, **(params or {})}
n_classes = len(np.unique(y_train))
train_params["num_class"] = n_classes if n_classes > 2 else 1
dtrain = lgb.Dataset(X_train, label=y_train)
if X_val is not None and y_val is not None:
dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)
valid_sets = [dtrain, dval]
valid_names = ["train", "valid"]
"""Train model + calibrate."""
if self.mode == "lr":
self._train_lr(X_train, y_train, X_val, y_val)
else:
valid_sets = None
valid_names = None
self.model = lgb.train(
train_params,
dtrain,
num_boost_round=500,
valid_sets=valid_sets,
valid_names=valid_names,
callbacks=[
lgb.early_stopping(early_stopping_rounds),
lgb.log_evaluation(period=50 if verbose else 0),
] if X_val is not None else None,
)
self._train_lgb(X_train, y_train, X_val, y_val, params, early_stopping_rounds, verbose)
self._trained = True
if X_val is not None and y_val is not None:
self.calibrator.fit(self.predict_raw(X_val), y_val)
elif X_train is not None and y_train is not None:
self.calibrator.fit(self.predict_raw(X_train), y_train)
def _train_lr(self, X_train, y_train, X_val, y_val):
"""Train logistic regression model."""
if LogisticRegression is None:
raise ImportError("scikit-learn not installed")
from sklearn.preprocessing import StandardScaler
self.scaler = StandardScaler()
X_train_scaled = self.scaler.fit_transform(X_train)
self.lr_model = LogisticRegression(
C=0.1, # Strong L2 regularization
solver="lbfgs",
max_iter=2000,
class_weight="balanced",
)
self.lr_model.fit(X_train_scaled, y_train)
self._trained = True
def _train_lgb(self, X_train, y_train, X_val, y_val, params, early_stopping_rounds, verbose):
"""Train LightGBM model."""
if lgb is None:
raise ImportError("lightgbm not installed")
train_params = {**LGBM_PARAMS, **(params or {})}
dtrain = lgb.Dataset(X_train, label=y_train)
if X_val is not None and y_val is not None:
dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)
valid_sets, valid_names = [dtrain, dval], ["train", "valid"]
else:
valid_sets, valid_names = None, None
self.model = lgb.train(
train_params, dtrain, num_boost_round=500,
valid_sets=valid_sets, valid_names=valid_names,
callbacks=[lgb.early_stopping(early_stopping_rounds),
lgb.log_evaluation(period=50 if verbose else 0)]
if X_val is not None else None,
)
self._compute_feature_importance()
def predict_raw(self, X: np.ndarray) -> np.ndarray:
"""Raw probability (0-1) before calibration."""
if not self._trained:
raise RuntimeError("Model not trained")
if self.mode == "lr" and self.lr_model is not None:
X_scaled = self.scaler.transform(X)
return self.lr_model.predict_proba(X_scaled)[:, 1]
elif self.model is not None:
return self.model.predict(X)
else:
return np.full(len(X), 0.5)
def predict_proba(self, X: np.ndarray) -> np.ndarray:
"""Predict probability (0-100) for binary outcome YES.
Applies temperature scaling to prevent extreme probabilities
when models are too confident on synthetic/bootstrap data.
"""
if not self._trained or self.model is None:
raise RuntimeError("Model not trained or loaded")
raw = self.model.predict(X)
# Temperature scaling: push extremes toward 0.5
# T=0.5 sharpens, T=2.0 flattens. Using T=2.0 for cautious predictions
temperature = 2.0
scaled = 1.0 / (1.0 + np.exp(-np.log(np.maximum(raw, 1e-9) / np.maximum(1 - raw, 1e-9)) / temperature))
return np.clip(scaled * 100.0, 1.0, 99.0)
"""Calibrated probability (0-100)."""
raw = self.predict_raw(X)
cal = self.calibrator.calibrate(raw)
return cal * 100.0
def predict(self, X: np.ndarray, threshold: float = 50.0) -> np.ndarray:
"""Binary prediction at given probability threshold."""
proba = self.predict_proba(X)
return (proba >= threshold).astype(int)
return (self.predict_proba(X) >= threshold).astype(int)
def evaluate(self, X: np.ndarray, y: np.ndarray) -> Dict[str, float]:
"""Evaluate model performance on test set."""
proba = self.predict_proba(X) / 100.0
pred = (proba >= 0.5).astype(int)
from sklearn.metrics import (
accuracy_score, brier_score_loss, roc_auc_score, log_loss
)
from sklearn.metrics import accuracy_score, brier_score_loss, roc_auc_score, log_loss
return {
"accuracy": float(accuracy_score(y, pred)),
"brier_score": float(brier_score_loss(y, proba)),
@@ -201,62 +325,76 @@ class WeatherModel:
"n_samples": len(y),
"p_yes_actual": float(y.mean() * 100),
"p_yes_predicted": float(proba.mean() * 100),
"calibration_method": self.calibrator.method,
}
def _compute_feature_importance(self):
"""Extract feature importance from trained model."""
if self.model is None:
return
gain = self.model.feature_importance(importance_type="gain")
names = self.model.feature_name()
self.feature_importance = dict(sorted(
zip(names, gain), key=lambda x: x[1], reverse=True
))
self.feature_importance = dict(sorted(zip(names, gain), key=lambda x: x[1], reverse=True))
def top_features(self, n: int = 15) -> Dict[str, float]:
"""Return top N most important features."""
items = sorted(
self.feature_importance.items(), key=lambda x: x[1], reverse=True
)
items = sorted(self.feature_importance.items(), key=lambda x: x[1], reverse=True)
return dict(items[:n])
def save(self, path: Optional[str] = None):
"""Save model to disk."""
MODEL_DIR.mkdir(parents=True, exist_ok=True)
p = path or (MODEL_DIR / f"{self.target_name}.lgb")
if self.model:
self.model.save_model(str(p))
elif not os.path.exists(p):
# Create marker for LR models
with open(p, "w") as f:
f.write("lr")
meta = {
"target_name": self.target_name,
"target_definition": self.target_def,
"mode": self.mode,
"feature_importance": self.feature_importance,
"trained": self._trained,
"calibration_method": self.calibrator.method,
}
meta_path = str(p).replace(".lgb", "_meta.json")
with open(meta_path, "w") as f:
json.dump(meta, f, indent=2)
with open(str(p).replace(".lgb", "_meta.json"), "w") as f:
json.dump(meta, f, indent=2, default=str)
self.calibrator.save(str(p).replace(".lgb", "_cal.pkl"))
if self.lr_model is not None:
import pickle
with open(str(p).replace(".lgb", "_lr.pkl"), "wb") as f:
pickle.dump({"model": self.lr_model, "scaler": self.scaler}, f)
def load(self, path: Optional[str] = None):
"""Load model from disk."""
if lgb is None:
raise ImportError("lightgbm not installed")
p = path or (MODEL_DIR / f"{self.target_name}.lgb")
if not os.path.exists(p):
raise FileNotFoundError(f"Model not found: {p}")
self.model = lgb.Booster(model_file=str(p))
self._trained = True
meta_path = str(p).replace(".lgb", "_meta.json")
if os.path.exists(meta_path):
with open(meta_path) as f:
meta = json.load(f)
self.mode = meta.get("mode", "lgb")
self.feature_importance = meta.get("feature_importance", {})
if self.mode == "lr":
import pickle
lr_path = str(p).replace(".lgb", "_lr.pkl")
if os.path.exists(lr_path):
with open(lr_path, "rb") as f:
data = pickle.load(f)
self.lr_model = data["model"]
self.scaler = data["scaler"]
else:
if lgb is None:
raise ImportError("lightgbm not installed")
self.model = lgb.Booster(model_file=str(p))
self._trained = True
cal_path = str(p).replace(".lgb", "_cal.pkl")
if os.path.exists(cal_path):
self.calibrator.load(cal_path)
@staticmethod
def build_target(df: pd.DataFrame, variable: str, threshold: float, op: str = "gt") -> np.ndarray:
"""Build binary target array from a DataFrame."""
if variable not in df.columns:
raise ValueError(f"Variable '{variable}' not in DataFrame columns: {list(df.columns)}")
values = df[variable].values
if op == "gt":
return (values > threshold).astype(int)
@@ -271,20 +409,12 @@ class WeatherModel:
class ModelEnsemble:
"""
Manage multiple WeatherModel instances for all targets.
Usage:
ensemble = ModelEnsemble()
ensemble.load_all() # Load all trained models
probs = ensemble.predict_all(X) # Dict of {target: probability}
"""
"""Manage multiple WeatherModel instances."""
def __init__(self):
self.models: Dict[str, WeatherModel] = {}
def load_all(self):
"""Load all available trained models from disk."""
MODEL_DIR.mkdir(parents=True, exist_ok=True)
for target in TARGET_DEFINITIONS:
model_path = MODEL_DIR / f"{target}.lgb"
@@ -292,29 +422,16 @@ class ModelEnsemble:
model = WeatherModel(target)
model.load(str(model_path))
self.models[target] = model
if not self.models:
print(f"No trained models found in {MODEL_DIR}. Run ml/train.py first.")
return self.models
def load(self, target: str):
"""Load a specific model."""
model = WeatherModel(target)
model.load()
self.models[target] = model
return model
def predict_all(self, X: np.ndarray) -> Dict[str, float]:
"""Predict all targets for a feature vector."""
if X.ndim == 1:
X = X.reshape(1, -1)
return {name: float(model.predict_proba(X)[0]) for name, model in self.models.items()}
def predict(self, target: str, X: np.ndarray) -> float:
"""Predict a single target."""
if target not in self.models:
raise KeyError(f"Model '{target}' not loaded. Available: {list(self.models.keys())}")
raise KeyError(f"Model '{target}' not loaded.")
return float(self.models[target].predict_proba(X)[0])
def has(self, target: str) -> bool:
@@ -323,10 +440,3 @@ class ModelEnsemble:
@property
def available_targets(self) -> List[str]:
return list(self.models.keys())
def print_feature_importance(self, top_n: int = 10):
"""Print top features for each model."""
for name, model in self.models.items():
print(f"\n--- {name} ({model.target_def['description']}) ---")
for feat, imp in list(model.top_features(top_n).items()):
print(f" {feat:30s} {imp:>10.1f}")