"""LightGBM probability models for HK weather prediction targets. One model per (target, lead_time_hours) pair: - rain_gt_0mm_24h: P(precipitation > 0mm at t+24h) - rain_gt_10mm_24h: P(precipitation > 10mm at t+24h) - temp_gt_30c_24h: P(Tmax > 30°C at t+24h) - temp_gt_33c_24h: P(Tmax > 33°C at t+24h) - temp_gt_35c_24h: P(Tmax > 35°C at t+24h) - typhoon_t3_72h: P(T3+ signal at t+72h) - typhoon_t8_72h: P(T8+ signal at t+72h) Each model is a LightGBM classifier with binary logloss objective, trained to output calibrated probabilities directly. """ import os import json from pathlib import Path from typing import Dict, Optional, Tuple, List import numpy as np import pandas as pd try: import lightgbm as lgb except ImportError: lgb = None from config import DATA_DIR, PROJECT_ROOT MODEL_DIR = Path(DATA_DIR) / "models" # Target definitions: (target_name, feature_to_compare, threshold, operation, description) TARGET_DEFINITIONS = { "rain_gt_0mm_24h": { "variable": "precipitation_sum", "threshold": 0.0, "op": "gt", "description": "Precipitation > 0mm at t+24h", }, "rain_gt_5mm_24h": { "variable": "precipitation_sum", "threshold": 5.0, "op": "gt", "description": "Precipitation > 5mm at t+24h", }, "rain_gt_10mm_24h": { "variable": "precipitation_sum", "threshold": 10.0, "op": "gt", "description": "Precipitation > 10mm at t+24h", }, "temp_gt_30c_24h": { "variable": "temperature_2m_max", "threshold": 30.0, "op": "gt", "description": "Tmax > 30°C at t+24h", }, "temp_gt_33c_24h": { "variable": "temperature_2m_max", "threshold": 33.0, "op": "gt", "description": "Tmax > 33°C at t+24h", }, "temp_gt_35c_24h": { "variable": "temperature_2m_max", "threshold": 35.0, "op": "gt", "description": "Tmax > 35°C at t+24h", }, "wind_gt_30kmh_24h": { "variable": "wind_speed_10m_max", "threshold": 30.0, "op": "gt", "description": "Wind gust > 30 km/h at t+24h", }, } LGBM_PARAMS = { "objective": "binary", "metric": "binary_logloss", "boosting_type": "gbdt", "num_leaves": 15, # Reduced from 31 — less leaf complexity "learning_rate": 0.03, # Reduced from 0.05 — slower learning "feature_fraction": 0.7, # Reduced from 0.8 — more regularization "bagging_fraction": 0.7, "bagging_freq": 5, "min_data_in_leaf": 50, # Increased from 20 — prevents tiny leaf nodes "min_gain_to_split": 0.05, # Increased from 0.01 — stronger split criterion "lambda_l1": 0.5, # Increased from 0.1 — L1 regularization "lambda_l2": 1.0, # Increased from 0.1 — L2 regularization "max_depth": 4, # Reduced from 6 — shallower trees "verbose": -1, "random_state": 42, } class WeatherModel: """ LightGBM-backed probability model for a single weather target. Usage: model = WeatherModel("temp_gt_30c_24h") model.train(X_train, y_train, X_val, y_val) # y is binary prob = model.predict_proba(X_single) # returns 0-100 model.save() """ def __init__(self, target_name: str): if target_name not in TARGET_DEFINITIONS: raise ValueError(f"Unknown target: {target_name}. Available: {list(TARGET_DEFINITIONS.keys())}") self.target_name = target_name self.target_def = TARGET_DEFINITIONS[target_name] self.model: Optional[lgb.Booster] = None self.feature_importance: Dict[str, float] = {} self.calibration_curve: Optional[Tuple[np.ndarray, np.ndarray]] = None self._trained = False def train( self, X_train: np.ndarray, y_train: np.ndarray, X_val: Optional[np.ndarray] = None, y_val: Optional[np.ndarray] = None, params: Optional[Dict] = None, early_stopping_rounds: int = 50, verbose: bool = True, ): """Train the LightGBM model.""" if lgb is None: raise ImportError("lightgbm not installed") train_params = {**LGBM_PARAMS, **(params or {})} n_classes = len(np.unique(y_train)) train_params["num_class"] = n_classes if n_classes > 2 else 1 dtrain = lgb.Dataset(X_train, label=y_train) if X_val is not None and y_val is not None: dval = lgb.Dataset(X_val, label=y_val, reference=dtrain) valid_sets = [dtrain, dval] valid_names = ["train", "valid"] else: valid_sets = None valid_names = None self.model = lgb.train( train_params, dtrain, num_boost_round=500, valid_sets=valid_sets, valid_names=valid_names, callbacks=[ lgb.early_stopping(early_stopping_rounds), lgb.log_evaluation(period=50 if verbose else 0), ] if X_val is not None else None, ) self._trained = True self._compute_feature_importance() def predict_proba(self, X: np.ndarray) -> np.ndarray: """Predict probability (0-100) for binary outcome YES. Applies temperature scaling to prevent extreme probabilities when models are too confident on synthetic/bootstrap data. """ if not self._trained or self.model is None: raise RuntimeError("Model not trained or loaded") raw = self.model.predict(X) # Temperature scaling: push extremes toward 0.5 # T=0.5 sharpens, T=2.0 flattens. Using T=2.0 for cautious predictions temperature = 2.0 scaled = 1.0 / (1.0 + np.exp(-np.log(np.maximum(raw, 1e-9) / np.maximum(1 - raw, 1e-9)) / temperature)) return np.clip(scaled * 100.0, 1.0, 99.0) def predict(self, X: np.ndarray, threshold: float = 50.0) -> np.ndarray: """Binary prediction at given probability threshold.""" proba = self.predict_proba(X) return (proba >= threshold).astype(int) def evaluate(self, X: np.ndarray, y: np.ndarray) -> Dict[str, float]: """Evaluate model performance on test set.""" proba = self.predict_proba(X) / 100.0 pred = (proba >= 0.5).astype(int) from sklearn.metrics import ( accuracy_score, brier_score_loss, roc_auc_score, log_loss ) return { "accuracy": float(accuracy_score(y, pred)), "brier_score": float(brier_score_loss(y, proba)), "roc_auc": float(roc_auc_score(y, proba)) if len(np.unique(y)) > 1 else 0.5, "log_loss": float(log_loss(y, proba)), "n_samples": len(y), "p_yes_actual": float(y.mean() * 100), "p_yes_predicted": float(proba.mean() * 100), } def _compute_feature_importance(self): """Extract feature importance from trained model.""" if self.model is None: return gain = self.model.feature_importance(importance_type="gain") names = self.model.feature_name() self.feature_importance = dict(sorted( zip(names, gain), key=lambda x: x[1], reverse=True )) def top_features(self, n: int = 15) -> Dict[str, float]: """Return top N most important features.""" items = sorted( self.feature_importance.items(), key=lambda x: x[1], reverse=True ) return dict(items[:n]) def save(self, path: Optional[str] = None): """Save model to disk.""" MODEL_DIR.mkdir(parents=True, exist_ok=True) p = path or (MODEL_DIR / f"{self.target_name}.lgb") if self.model: self.model.save_model(str(p)) meta = { "target_name": self.target_name, "target_definition": self.target_def, "feature_importance": self.feature_importance, "trained": self._trained, } meta_path = str(p).replace(".lgb", "_meta.json") with open(meta_path, "w") as f: json.dump(meta, f, indent=2) def load(self, path: Optional[str] = None): """Load model from disk.""" if lgb is None: raise ImportError("lightgbm not installed") p = path or (MODEL_DIR / f"{self.target_name}.lgb") if not os.path.exists(p): raise FileNotFoundError(f"Model not found: {p}") self.model = lgb.Booster(model_file=str(p)) self._trained = True meta_path = str(p).replace(".lgb", "_meta.json") if os.path.exists(meta_path): with open(meta_path) as f: meta = json.load(f) self.feature_importance = meta.get("feature_importance", {}) @staticmethod def build_target(df: pd.DataFrame, variable: str, threshold: float, op: str = "gt") -> np.ndarray: """Build binary target array from a DataFrame.""" if variable not in df.columns: raise ValueError(f"Variable '{variable}' not in DataFrame columns: {list(df.columns)}") values = df[variable].values if op == "gt": return (values > threshold).astype(int) elif op == "ge": return (values >= threshold).astype(int) elif op == "lt": return (values < threshold).astype(int) elif op == "le": return (values <= threshold).astype(int) else: raise ValueError(f"Unknown operator: {op}") class ModelEnsemble: """ Manage multiple WeatherModel instances for all targets. Usage: ensemble = ModelEnsemble() ensemble.load_all() # Load all trained models probs = ensemble.predict_all(X) # Dict of {target: probability} """ def __init__(self): self.models: Dict[str, WeatherModel] = {} def load_all(self): """Load all available trained models from disk.""" MODEL_DIR.mkdir(parents=True, exist_ok=True) for target in TARGET_DEFINITIONS: model_path = MODEL_DIR / f"{target}.lgb" if model_path.exists(): model = WeatherModel(target) model.load(str(model_path)) self.models[target] = model if not self.models: print(f"No trained models found in {MODEL_DIR}. Run ml/train.py first.") return self.models def load(self, target: str): """Load a specific model.""" model = WeatherModel(target) model.load() self.models[target] = model return model def predict_all(self, X: np.ndarray) -> Dict[str, float]: """Predict all targets for a feature vector.""" if X.ndim == 1: X = X.reshape(1, -1) return {name: float(model.predict_proba(X)[0]) for name, model in self.models.items()} def predict(self, target: str, X: np.ndarray) -> float: """Predict a single target.""" if target not in self.models: raise KeyError(f"Model '{target}' not loaded. Available: {list(self.models.keys())}") return float(self.models[target].predict_proba(X)[0]) def has(self, target: str) -> bool: return target in self.models @property def available_targets(self) -> List[str]: return list(self.models.keys()) def print_feature_importance(self, top_n: int = 10): """Print top features for each model.""" for name, model in self.models.items(): print(f"\n--- {name} ({model.target_def['description']}) ---") for feat, imp in list(model.top_features(top_n).items()): print(f" {feat:30s} {imp:>10.1f}")