Files
fantabeto/src/models/survival_model.py
T
ramseshk b0fab62a87 feat: add 10 new ML models for auction optimization (Phases 1-6)
Phase 1 - Quick Wins:
- QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding
- MinutesSurvivalModel: Weibull AFT for minutes distribution modeling

Phase 2 - Adaptive Auction:
- BanditAuctionSolver: Thompson Sampling for live auction bids
- OpponentBidModel: Predict competitor bids via LightGBM
- BudgetOptimizer: Bayesian optimization for role-level allocation

Phase 3 - Deep Learning:
- RLAuctionPolicy: Double DQN agent for auction strategy
- SetTransformer: Team composition valuation via set-based ML

Phase 4 - Probabilistic:
- BayesianPlayerModel: Hierarchical pooling for rookie uncertainty
- ConformalPredictor: Calibrated prediction intervals

Phase 5 - Chemistry & Form:
- PlayerChemistryGAT: Graph attention network for player synergies
- PlayerFormModel: Hawkes process for form momentum

Phase 6 - Causal:
- TransferCausalModel: Causal forest for transfer effects
- AuctionEffectAnalyzer: Bid adjustment from causal analysis

81 tests passing
2026-08-11 17:56:03 +08:00

355 lines
13 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Minutes-played survival model via Weibull AFT.
Models the distribution of minutes played per matchweek using Weibull
Accelerated Failure Time. Supports both lifelines (preferred) and a pure-scipy
MLE fallback so the module works in minimal environments.
Provides:
- Expected minutes / confidence intervals
- Starter probability (≥60 min)
- Full-match probability (90 min)
"""
import logging
import math
from typing import Optional, Tuple
import numpy as np
import pandas as pd
from sklearn.linear_model import LinearRegression
from sklearn.preprocessing import StandardScaler
from .base_model import BaseModel
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Feature-selection keyword list
# ---------------------------------------------------------------------------
_MINUTE_KEYWORDS = [
"minute", "game", "rest", "fatigue", "age", "injury",
"played", "starter", "bench", "appearance",
"recovery", "rotation", "squad", "season",
"match", "form", "fitness",
]
def _select_survival_features(X: pd.DataFrame) -> list:
"""Pick columns whose name contains any survival-relevant keyword."""
lower_cols = {c: str(c).lower() for c in X.columns}
selected = [
c for c, cl in lower_cols.items()
if any(kw in cl for kw in _MINUTE_KEYWORDS)
]
if not selected:
selected = list(X.select_dtypes(include=[np.number]).columns[:20])
logger.info("No keyword-matched survival features; using first 20 numeric columns")
else:
logger.info(f"Selected {len(selected)} survival features via keyword matching")
return selected
# ===================================================================
# Weibull helper functions for the scipy fallback
# ===================================================================
def _weibull_log_likelihood(params, X, t, event, eps=1e-10):
"""Negative log-likelihood for Weibull AFT model.
Parameters
----------
params : ndarray (p_features + 1,)
First p entries: beta (coefficients for X).
Last entry: log_k (log shape parameter ensures k > 0).
X : ndarray (n, p)
Scaled feature matrix.
t : ndarray (n,)
Observed durations (minutes played).
event : ndarray (n,)
0 → exact failure (subbed off), 1 → right-censored (completed 90).
eps : float
Small epsilon for numerical stability.
Returns
-------
neg_ll : float
Negative log-likelihood (to be minimized).
"""
p = X.shape[1]
beta = params[:p]
log_k = params[p]
k = np.exp(log_k) + eps
log_lambda = X.dot(beta) # log(λ_i) = X_i * beta
lambda_ = np.exp(log_lambda) + eps
log_t = np.log(np.maximum(t, eps))
z = t / lambda_
# Log-PDF for uncensored (event == 0)
log_pdf = np.log(k) - log_lambda + (k - 1.0) * (log_t - log_lambda) - z ** k
# Log-SF for censored (event == 1)
log_sf = -(z ** k)
# event==1 → censored → use SF; event==0 → observed → use PDF
ll = np.where(event == 1, log_sf, log_pdf)
return -ll.sum()
def _fit_weibull_mle(X, t, event):
"""Fit Weibull AFT via scipy MLE.
Returns
-------
beta : ndarray (p,)
Feature coefficients (scaled to original duration range).
k : float
Shape parameter.
t_scale : float
Scale factor to convert normalized predictions back to minutes.
"""
from scipy.optimize import minimize
n, p = X.shape
t_scale = max(t.max(), 1.0)
t_norm = np.clip(t / t_scale, 1e-6, 1.0)
log_t_norm = np.log(np.maximum(t_norm, 1e-9))
lr = LinearRegression(fit_intercept=False)
lr.fit(X, log_t_norm)
beta0 = np.clip(lr.coef_.copy(), -5, 5)
bounds = [(-10, 10)] * p + [(-5, 3)]
init = np.concatenate([beta0, [0.0]])
result = minimize(
_weibull_log_likelihood,
init,
args=(X, t_norm, event),
method="L-BFGS-B",
bounds=bounds,
options={"maxiter": 2000, "ftol": 1e-10},
)
if not result.success:
logger.warning(f"Weibull MLE did not converge: {result.message}")
beta = result.x[:p]
k = max(np.exp(result.x[p]), 1e-4)
return beta, k, t_scale
# ===================================================================
# MinutesSurvivalModel
# ===================================================================
class MinutesSurvivalModel(BaseModel):
"""Weibull AFT model for minutes-played distribution.
Parameters
----------
model_dir : str
Directory for persisting trained models.
force_scipy : bool
If True, use the pure-scipy MLE fallback even when lifelines
is installed.
"""
def __init__(
self,
model_dir: str = "models_trained",
force_scipy: bool = False,
):
super().__init__(model_dir)
self.force_scipy = force_scipy
self.scaler = StandardScaler()
self.feature_names = None
# Weibull parameters
self._beta = None # feature coefficients → log(λ)
self._k = None # shape parameter
self._afitter = None # lifelines WeibullAFTFitter instance (if used)
self._t_scale = 90.0
self._use_lifelines = False
# ------------------------------------------------------------------
# Fit
# ------------------------------------------------------------------
def fit(
self,
X: pd.DataFrame,
durations: np.ndarray,
events: np.ndarray,
**kwargs,
):
"""Fit the Weibull AFT model.
Args:
X: Feature matrix (one row per player-match).
durations: Minutes played (0–90); `y` alias for BaseModel compat.
events:
0 → exact duration observed (subbed off before 90).
1 → right-censored (player completed the full 90 minutes).
"""
self.feature_names = _select_survival_features(X)
X_clean = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
X_scaled = self.scaler.fit_transform(X_clean)
durations = np.asarray(durations, dtype=float)
events = np.asarray(events, dtype=int)
# Try lifelines first --------------------------------------------------
if not self.force_scipy:
try:
import lifelines # noqa: F401
from lifelines import WeibullAFTFitter
df = pd.DataFrame(X_scaled, columns=self.feature_names)
df["duration"] = durations
df["event"] = events
aft = WeibullAFTFitter()
aft.fit(df, duration_col="duration", event_col="event")
self._afitter = aft
self._use_lifelines = True
self._t_scale = 1.0 # lifelines works in original duration scale
logger.info(
"MinutesSurvivalModel fitted via lifelines "
f"(n={len(durations)}, features={len(self.feature_names)})"
)
return self
except ImportError:
logger.info("lifelines not installed; falling back to scipy MLE")
except Exception as exc:
logger.warning(f"lifelines failed ({exc}); falling back to scipy MLE")
# Scipy fallback -------------------------------------------------------
self._use_lifelines = False
self._beta, self._k, self._t_scale = _fit_weibull_mle(X_scaled, durations, events)
logger.info(
f"MinutesSurvivalModel fitted via scipy MLE "
f"(n={len(durations)}, features={len(self.feature_names)}, "
f"k={self._k:.3f})"
)
return self
# ------------------------------------------------------------------
# Predict (BaseModel interface — returns expected minutes)
# ------------------------------------------------------------------
def predict(self, X: pd.DataFrame) -> np.ndarray:
"""Return expected minutes (E[T]) — BaseModel interface."""
return self.predict_expected_minutes(X)
# ------------------------------------------------------------------
# Preprocessing
# ------------------------------------------------------------------
def _preprocess(self, X: pd.DataFrame) -> np.ndarray:
if self.feature_names is None:
raise RuntimeError("Model not trained. Call fit() first.")
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
return self.scaler.transform(X_c)
# ------------------------------------------------------------------
# Core distribution
# ------------------------------------------------------------------
def predict_distribution(
self, X: pd.DataFrame
) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
"""Return (expected_minutes, lower_bound, upper_bound).
``lower_bound`` and ``upper_bound`` are approximate 95 % confidence
intervals derived from the Weibull variance.
"""
X_scaled = self._preprocess(X)
if self._use_lifelines and self._afitter is not None:
df = pd.DataFrame(X_scaled, columns=self.feature_names)
# Lifelines returns median survival in its summary; we approximate
# expected minutes using the median and estimated shape.
median = self._afitter.predict_median(df).values.flatten()
# Heuristic: for Weibull, E[T] ≈ median / (ln 2)^(1/k).
# Derive k from the lifelines summary if possible, else guess ~1.
try:
summary = self._afitter.summary
log_k = summary.loc["lambda_", "coef"]
k = 1.0 / np.exp(log_k) if abs(log_k) > 1e-8 else 1.0
except Exception:
k = 1.0
expected = median * np.exp(np.log(np.log(2)) / k)
# Std via coefficient of variation
coef_var = np.sqrt(np.exp(
np.log(math.gamma(1 + 2 / k)) - 2 * np.log(math.gamma(1 + 1 / k))
))
std = expected * coef_var
lower = np.maximum(0, expected - 1.96 * std)
upper = np.minimum(90, expected + 1.96 * std)
return expected, lower, upper
# Scipy / stored parameters (normalized scale, convert to minutes)
log_lambda = X_scaled.dot(self._beta)
lambda_ = np.exp(log_lambda) * self._t_scale
k = self._k
# Expected value: λ * Γ(1 + 1/k)
gamma_1 = math.gamma(1.0 + 1.0 / k)
expected = lambda_ * gamma_1
# Variance = λ² * (Γ(1+2/k) - Γ²(1+1/k))
gamma_2 = math.gamma(1.0 + 2.0 / k)
var = (lambda_ ** 2) * (gamma_2 - gamma_1 ** 2)
std = np.sqrt(np.maximum(var, 0.01))
lower = np.maximum(0, expected - 1.96 * std)
upper = np.minimum(90, expected + 1.96 * std)
expected = np.clip(expected, 0, 90)
return expected, lower, upper
def predict_expected_minutes(self, X: pd.DataFrame) -> np.ndarray:
"""Return E[minutes] for each row."""
expected, _, _ = self.predict_distribution(X)
return expected
def predict_full_match_probability(self, X: pd.DataFrame) -> np.ndarray:
"""Probability the player completes 90 minutes: P(T ≥ 90)."""
X_scaled = self._preprocess(X)
if self._use_lifelines and self._afitter is not None:
try:
surv = self._afitter.predict_survival_function(
pd.DataFrame(X_scaled, columns=self.feature_names),
times=[90.0],
)
return 1.0 - surv.values.flatten()
except Exception:
pass
log_lambda = X_scaled.dot(self._beta)
lambda_ = np.exp(log_lambda) * self._t_scale
surv = np.exp(-((90.0 / lambda_) ** self._k))
prob = 1.0 - surv
return np.clip(prob, 0.0, 1.0)
def predict_starter_probability(
self, X: pd.DataFrame, min_minutes: float = 60.0
) -> np.ndarray:
"""Probability player plays at least *min_minutes* (default 60).
Useful as a "likely starter" proxy.
"""
X_scaled = self._preprocess(X)
if self._use_lifelines and self._afitter is not None:
try:
surv = self._afitter.predict_survival_function(
pd.DataFrame(X_scaled, columns=self.feature_names),
times=[min_minutes],
)
return surv.values.flatten()
except Exception:
pass
log_lambda = X_scaled.dot(self._beta)
lambda_ = np.exp(log_lambda) * self._t_scale
surv = np.exp(-((min_minutes / lambda_) ** self._k))
return np.clip(surv, 0.0, 1.0)