b0fab62a87
Phase 1 - Quick Wins: - QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding - MinutesSurvivalModel: Weibull AFT for minutes distribution modeling Phase 2 - Adaptive Auction: - BanditAuctionSolver: Thompson Sampling for live auction bids - OpponentBidModel: Predict competitor bids via LightGBM - BudgetOptimizer: Bayesian optimization for role-level allocation Phase 3 - Deep Learning: - RLAuctionPolicy: Double DQN agent for auction strategy - SetTransformer: Team composition valuation via set-based ML Phase 4 - Probabilistic: - BayesianPlayerModel: Hierarchical pooling for rookie uncertainty - ConformalPredictor: Calibrated prediction intervals Phase 5 - Chemistry & Form: - PlayerChemistryGAT: Graph attention network for player synergies - PlayerFormModel: Hawkes process for form momentum Phase 6 - Causal: - TransferCausalModel: Causal forest for transfer effects - AuctionEffectAnalyzer: Bid adjustment from causal analysis 81 tests passing
355 lines
13 KiB
Python
355 lines
13 KiB
Python
"""Minutes-played survival model via Weibull AFT.
|
||
|
||
Models the distribution of minutes played per matchweek using Weibull
|
||
Accelerated Failure Time. Supports both lifelines (preferred) and a pure-scipy
|
||
MLE fallback so the module works in minimal environments.
|
||
|
||
Provides:
|
||
- Expected minutes / confidence intervals
|
||
- Starter probability (≥60 min)
|
||
- Full-match probability (90 min)
|
||
"""
|
||
|
||
import logging
|
||
import math
|
||
from typing import Optional, Tuple
|
||
|
||
import numpy as np
|
||
import pandas as pd
|
||
from sklearn.linear_model import LinearRegression
|
||
from sklearn.preprocessing import StandardScaler
|
||
|
||
from .base_model import BaseModel
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Feature-selection keyword list
|
||
# ---------------------------------------------------------------------------
|
||
_MINUTE_KEYWORDS = [
|
||
"minute", "game", "rest", "fatigue", "age", "injury",
|
||
"played", "starter", "bench", "appearance",
|
||
"recovery", "rotation", "squad", "season",
|
||
"match", "form", "fitness",
|
||
]
|
||
|
||
|
||
def _select_survival_features(X: pd.DataFrame) -> list:
|
||
"""Pick columns whose name contains any survival-relevant keyword."""
|
||
lower_cols = {c: str(c).lower() for c in X.columns}
|
||
selected = [
|
||
c for c, cl in lower_cols.items()
|
||
if any(kw in cl for kw in _MINUTE_KEYWORDS)
|
||
]
|
||
if not selected:
|
||
selected = list(X.select_dtypes(include=[np.number]).columns[:20])
|
||
logger.info("No keyword-matched survival features; using first 20 numeric columns")
|
||
else:
|
||
logger.info(f"Selected {len(selected)} survival features via keyword matching")
|
||
return selected
|
||
|
||
|
||
# ===================================================================
|
||
# Weibull helper functions for the scipy fallback
|
||
# ===================================================================
|
||
|
||
def _weibull_log_likelihood(params, X, t, event, eps=1e-10):
|
||
"""Negative log-likelihood for Weibull AFT model.
|
||
|
||
Parameters
|
||
----------
|
||
params : ndarray (p_features + 1,)
|
||
First p entries: beta (coefficients for X).
|
||
Last entry: log_k (log shape parameter ensures k > 0).
|
||
X : ndarray (n, p)
|
||
Scaled feature matrix.
|
||
t : ndarray (n,)
|
||
Observed durations (minutes played).
|
||
event : ndarray (n,)
|
||
0 → exact failure (subbed off), 1 → right-censored (completed 90).
|
||
eps : float
|
||
Small epsilon for numerical stability.
|
||
|
||
Returns
|
||
-------
|
||
neg_ll : float
|
||
Negative log-likelihood (to be minimized).
|
||
"""
|
||
p = X.shape[1]
|
||
beta = params[:p]
|
||
log_k = params[p]
|
||
k = np.exp(log_k) + eps
|
||
|
||
log_lambda = X.dot(beta) # log(λ_i) = X_i * beta
|
||
lambda_ = np.exp(log_lambda) + eps
|
||
log_t = np.log(np.maximum(t, eps))
|
||
z = t / lambda_
|
||
|
||
# Log-PDF for uncensored (event == 0)
|
||
log_pdf = np.log(k) - log_lambda + (k - 1.0) * (log_t - log_lambda) - z ** k
|
||
|
||
# Log-SF for censored (event == 1)
|
||
log_sf = -(z ** k)
|
||
|
||
# event==1 → censored → use SF; event==0 → observed → use PDF
|
||
ll = np.where(event == 1, log_sf, log_pdf)
|
||
return -ll.sum()
|
||
|
||
|
||
def _fit_weibull_mle(X, t, event):
|
||
"""Fit Weibull AFT via scipy MLE.
|
||
|
||
Returns
|
||
-------
|
||
beta : ndarray (p,)
|
||
Feature coefficients (scaled to original duration range).
|
||
k : float
|
||
Shape parameter.
|
||
t_scale : float
|
||
Scale factor to convert normalized predictions back to minutes.
|
||
"""
|
||
from scipy.optimize import minimize
|
||
|
||
n, p = X.shape
|
||
t_scale = max(t.max(), 1.0)
|
||
t_norm = np.clip(t / t_scale, 1e-6, 1.0)
|
||
log_t_norm = np.log(np.maximum(t_norm, 1e-9))
|
||
|
||
lr = LinearRegression(fit_intercept=False)
|
||
lr.fit(X, log_t_norm)
|
||
beta0 = np.clip(lr.coef_.copy(), -5, 5)
|
||
|
||
bounds = [(-10, 10)] * p + [(-5, 3)]
|
||
init = np.concatenate([beta0, [0.0]])
|
||
|
||
result = minimize(
|
||
_weibull_log_likelihood,
|
||
init,
|
||
args=(X, t_norm, event),
|
||
method="L-BFGS-B",
|
||
bounds=bounds,
|
||
options={"maxiter": 2000, "ftol": 1e-10},
|
||
)
|
||
if not result.success:
|
||
logger.warning(f"Weibull MLE did not converge: {result.message}")
|
||
|
||
beta = result.x[:p]
|
||
k = max(np.exp(result.x[p]), 1e-4)
|
||
return beta, k, t_scale
|
||
|
||
|
||
# ===================================================================
|
||
# MinutesSurvivalModel
|
||
# ===================================================================
|
||
|
||
class MinutesSurvivalModel(BaseModel):
|
||
"""Weibull AFT model for minutes-played distribution.
|
||
|
||
Parameters
|
||
----------
|
||
model_dir : str
|
||
Directory for persisting trained models.
|
||
force_scipy : bool
|
||
If True, use the pure-scipy MLE fallback even when lifelines
|
||
is installed.
|
||
"""
|
||
|
||
def __init__(
|
||
self,
|
||
model_dir: str = "models_trained",
|
||
force_scipy: bool = False,
|
||
):
|
||
super().__init__(model_dir)
|
||
self.force_scipy = force_scipy
|
||
|
||
self.scaler = StandardScaler()
|
||
self.feature_names = None
|
||
|
||
# Weibull parameters
|
||
self._beta = None # feature coefficients → log(λ)
|
||
self._k = None # shape parameter
|
||
self._afitter = None # lifelines WeibullAFTFitter instance (if used)
|
||
self._t_scale = 90.0
|
||
self._use_lifelines = False
|
||
|
||
# ------------------------------------------------------------------
|
||
# Fit
|
||
# ------------------------------------------------------------------
|
||
def fit(
|
||
self,
|
||
X: pd.DataFrame,
|
||
durations: np.ndarray,
|
||
events: np.ndarray,
|
||
**kwargs,
|
||
):
|
||
"""Fit the Weibull AFT model.
|
||
|
||
Args:
|
||
X: Feature matrix (one row per player-match).
|
||
durations: Minutes played (0–90); `y` alias for BaseModel compat.
|
||
events:
|
||
0 → exact duration observed (subbed off before 90).
|
||
1 → right-censored (player completed the full 90 minutes).
|
||
"""
|
||
self.feature_names = _select_survival_features(X)
|
||
X_clean = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||
X_scaled = self.scaler.fit_transform(X_clean)
|
||
|
||
durations = np.asarray(durations, dtype=float)
|
||
events = np.asarray(events, dtype=int)
|
||
|
||
# Try lifelines first --------------------------------------------------
|
||
if not self.force_scipy:
|
||
try:
|
||
import lifelines # noqa: F401
|
||
from lifelines import WeibullAFTFitter
|
||
|
||
df = pd.DataFrame(X_scaled, columns=self.feature_names)
|
||
df["duration"] = durations
|
||
df["event"] = events
|
||
|
||
aft = WeibullAFTFitter()
|
||
aft.fit(df, duration_col="duration", event_col="event")
|
||
self._afitter = aft
|
||
self._use_lifelines = True
|
||
self._t_scale = 1.0 # lifelines works in original duration scale
|
||
logger.info(
|
||
"MinutesSurvivalModel fitted via lifelines "
|
||
f"(n={len(durations)}, features={len(self.feature_names)})"
|
||
)
|
||
return self
|
||
except ImportError:
|
||
logger.info("lifelines not installed; falling back to scipy MLE")
|
||
except Exception as exc:
|
||
logger.warning(f"lifelines failed ({exc}); falling back to scipy MLE")
|
||
|
||
# Scipy fallback -------------------------------------------------------
|
||
self._use_lifelines = False
|
||
self._beta, self._k, self._t_scale = _fit_weibull_mle(X_scaled, durations, events)
|
||
logger.info(
|
||
f"MinutesSurvivalModel fitted via scipy MLE "
|
||
f"(n={len(durations)}, features={len(self.feature_names)}, "
|
||
f"k={self._k:.3f})"
|
||
)
|
||
return self
|
||
|
||
# ------------------------------------------------------------------
|
||
# Predict (BaseModel interface — returns expected minutes)
|
||
# ------------------------------------------------------------------
|
||
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
||
"""Return expected minutes (E[T]) — BaseModel interface."""
|
||
return self.predict_expected_minutes(X)
|
||
|
||
# ------------------------------------------------------------------
|
||
# Preprocessing
|
||
# ------------------------------------------------------------------
|
||
def _preprocess(self, X: pd.DataFrame) -> np.ndarray:
|
||
if self.feature_names is None:
|
||
raise RuntimeError("Model not trained. Call fit() first.")
|
||
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||
return self.scaler.transform(X_c)
|
||
|
||
# ------------------------------------------------------------------
|
||
# Core distribution
|
||
# ------------------------------------------------------------------
|
||
def predict_distribution(
|
||
self, X: pd.DataFrame
|
||
) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
|
||
"""Return (expected_minutes, lower_bound, upper_bound).
|
||
|
||
``lower_bound`` and ``upper_bound`` are approximate 95 % confidence
|
||
intervals derived from the Weibull variance.
|
||
"""
|
||
X_scaled = self._preprocess(X)
|
||
|
||
if self._use_lifelines and self._afitter is not None:
|
||
df = pd.DataFrame(X_scaled, columns=self.feature_names)
|
||
# Lifelines returns median survival in its summary; we approximate
|
||
# expected minutes using the median and estimated shape.
|
||
median = self._afitter.predict_median(df).values.flatten()
|
||
# Heuristic: for Weibull, E[T] ≈ median / (ln 2)^(1/k).
|
||
# Derive k from the lifelines summary if possible, else guess ~1.
|
||
try:
|
||
summary = self._afitter.summary
|
||
log_k = summary.loc["lambda_", "coef"]
|
||
k = 1.0 / np.exp(log_k) if abs(log_k) > 1e-8 else 1.0
|
||
except Exception:
|
||
k = 1.0
|
||
expected = median * np.exp(np.log(np.log(2)) / k)
|
||
# Std via coefficient of variation
|
||
coef_var = np.sqrt(np.exp(
|
||
np.log(math.gamma(1 + 2 / k)) - 2 * np.log(math.gamma(1 + 1 / k))
|
||
))
|
||
std = expected * coef_var
|
||
lower = np.maximum(0, expected - 1.96 * std)
|
||
upper = np.minimum(90, expected + 1.96 * std)
|
||
return expected, lower, upper
|
||
|
||
# Scipy / stored parameters (normalized scale, convert to minutes)
|
||
log_lambda = X_scaled.dot(self._beta)
|
||
lambda_ = np.exp(log_lambda) * self._t_scale
|
||
k = self._k
|
||
|
||
# Expected value: λ * Γ(1 + 1/k)
|
||
gamma_1 = math.gamma(1.0 + 1.0 / k)
|
||
expected = lambda_ * gamma_1
|
||
|
||
# Variance = λ² * (Γ(1+2/k) - Γ²(1+1/k))
|
||
gamma_2 = math.gamma(1.0 + 2.0 / k)
|
||
var = (lambda_ ** 2) * (gamma_2 - gamma_1 ** 2)
|
||
std = np.sqrt(np.maximum(var, 0.01))
|
||
|
||
lower = np.maximum(0, expected - 1.96 * std)
|
||
upper = np.minimum(90, expected + 1.96 * std)
|
||
expected = np.clip(expected, 0, 90)
|
||
return expected, lower, upper
|
||
|
||
def predict_expected_minutes(self, X: pd.DataFrame) -> np.ndarray:
|
||
"""Return E[minutes] for each row."""
|
||
expected, _, _ = self.predict_distribution(X)
|
||
return expected
|
||
|
||
def predict_full_match_probability(self, X: pd.DataFrame) -> np.ndarray:
|
||
"""Probability the player completes 90 minutes: P(T ≥ 90)."""
|
||
X_scaled = self._preprocess(X)
|
||
|
||
if self._use_lifelines and self._afitter is not None:
|
||
try:
|
||
surv = self._afitter.predict_survival_function(
|
||
pd.DataFrame(X_scaled, columns=self.feature_names),
|
||
times=[90.0],
|
||
)
|
||
return 1.0 - surv.values.flatten()
|
||
except Exception:
|
||
pass
|
||
|
||
log_lambda = X_scaled.dot(self._beta)
|
||
lambda_ = np.exp(log_lambda) * self._t_scale
|
||
surv = np.exp(-((90.0 / lambda_) ** self._k))
|
||
prob = 1.0 - surv
|
||
return np.clip(prob, 0.0, 1.0)
|
||
|
||
def predict_starter_probability(
|
||
self, X: pd.DataFrame, min_minutes: float = 60.0
|
||
) -> np.ndarray:
|
||
"""Probability player plays at least *min_minutes* (default 60).
|
||
|
||
Useful as a "likely starter" proxy.
|
||
"""
|
||
X_scaled = self._preprocess(X)
|
||
|
||
if self._use_lifelines and self._afitter is not None:
|
||
try:
|
||
surv = self._afitter.predict_survival_function(
|
||
pd.DataFrame(X_scaled, columns=self.feature_names),
|
||
times=[min_minutes],
|
||
)
|
||
return surv.values.flatten()
|
||
except Exception:
|
||
pass
|
||
|
||
log_lambda = X_scaled.dot(self._beta)
|
||
lambda_ = np.exp(log_lambda) * self._t_scale
|
||
surv = np.exp(-((min_minutes / lambda_) ** self._k))
|
||
return np.clip(surv, 0.0, 1.0)
|