feat: add 10 new ML models for auction optimization (Phases 1-6)
Phase 1 - Quick Wins: - QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding - MinutesSurvivalModel: Weibull AFT for minutes distribution modeling Phase 2 - Adaptive Auction: - BanditAuctionSolver: Thompson Sampling for live auction bids - OpponentBidModel: Predict competitor bids via LightGBM - BudgetOptimizer: Bayesian optimization for role-level allocation Phase 3 - Deep Learning: - RLAuctionPolicy: Double DQN agent for auction strategy - SetTransformer: Team composition valuation via set-based ML Phase 4 - Probabilistic: - BayesianPlayerModel: Hierarchical pooling for rookie uncertainty - ConformalPredictor: Calibrated prediction intervals Phase 5 - Chemistry & Form: - PlayerChemistryGAT: Graph attention network for player synergies - PlayerFormModel: Hawkes process for form momentum Phase 6 - Causal: - TransferCausalModel: Causal forest for transfer effects - AuctionEffectAnalyzer: Bid adjustment from causal analysis 81 tests passing
This commit is contained in:
@@ -0,0 +1,354 @@
|
||||
"""Minutes-played survival model via Weibull AFT.
|
||||
|
||||
Models the distribution of minutes played per matchweek using Weibull
|
||||
Accelerated Failure Time. Supports both lifelines (preferred) and a pure-scipy
|
||||
MLE fallback so the module works in minimal environments.
|
||||
|
||||
Provides:
|
||||
- Expected minutes / confidence intervals
|
||||
- Starter probability (≥60 min)
|
||||
- Full-match probability (90 min)
|
||||
"""
|
||||
|
||||
import logging
|
||||
import math
|
||||
from typing import Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.linear_model import LinearRegression
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
from .base_model import BaseModel
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Feature-selection keyword list
|
||||
# ---------------------------------------------------------------------------
|
||||
_MINUTE_KEYWORDS = [
|
||||
"minute", "game", "rest", "fatigue", "age", "injury",
|
||||
"played", "starter", "bench", "appearance",
|
||||
"recovery", "rotation", "squad", "season",
|
||||
"match", "form", "fitness",
|
||||
]
|
||||
|
||||
|
||||
def _select_survival_features(X: pd.DataFrame) -> list:
|
||||
"""Pick columns whose name contains any survival-relevant keyword."""
|
||||
lower_cols = {c: str(c).lower() for c in X.columns}
|
||||
selected = [
|
||||
c for c, cl in lower_cols.items()
|
||||
if any(kw in cl for kw in _MINUTE_KEYWORDS)
|
||||
]
|
||||
if not selected:
|
||||
selected = list(X.select_dtypes(include=[np.number]).columns[:20])
|
||||
logger.info("No keyword-matched survival features; using first 20 numeric columns")
|
||||
else:
|
||||
logger.info(f"Selected {len(selected)} survival features via keyword matching")
|
||||
return selected
|
||||
|
||||
|
||||
# ===================================================================
|
||||
# Weibull helper functions for the scipy fallback
|
||||
# ===================================================================
|
||||
|
||||
def _weibull_log_likelihood(params, X, t, event, eps=1e-10):
|
||||
"""Negative log-likelihood for Weibull AFT model.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
params : ndarray (p_features + 1,)
|
||||
First p entries: beta (coefficients for X).
|
||||
Last entry: log_k (log shape parameter ensures k > 0).
|
||||
X : ndarray (n, p)
|
||||
Scaled feature matrix.
|
||||
t : ndarray (n,)
|
||||
Observed durations (minutes played).
|
||||
event : ndarray (n,)
|
||||
0 → exact failure (subbed off), 1 → right-censored (completed 90).
|
||||
eps : float
|
||||
Small epsilon for numerical stability.
|
||||
|
||||
Returns
|
||||
-------
|
||||
neg_ll : float
|
||||
Negative log-likelihood (to be minimized).
|
||||
"""
|
||||
p = X.shape[1]
|
||||
beta = params[:p]
|
||||
log_k = params[p]
|
||||
k = np.exp(log_k) + eps
|
||||
|
||||
log_lambda = X.dot(beta) # log(λ_i) = X_i * beta
|
||||
lambda_ = np.exp(log_lambda) + eps
|
||||
log_t = np.log(np.maximum(t, eps))
|
||||
z = t / lambda_
|
||||
|
||||
# Log-PDF for uncensored (event == 0)
|
||||
log_pdf = np.log(k) - log_lambda + (k - 1.0) * (log_t - log_lambda) - z ** k
|
||||
|
||||
# Log-SF for censored (event == 1)
|
||||
log_sf = -(z ** k)
|
||||
|
||||
# event==1 → censored → use SF; event==0 → observed → use PDF
|
||||
ll = np.where(event == 1, log_sf, log_pdf)
|
||||
return -ll.sum()
|
||||
|
||||
|
||||
def _fit_weibull_mle(X, t, event):
|
||||
"""Fit Weibull AFT via scipy MLE.
|
||||
|
||||
Returns
|
||||
-------
|
||||
beta : ndarray (p,)
|
||||
Feature coefficients (scaled to original duration range).
|
||||
k : float
|
||||
Shape parameter.
|
||||
t_scale : float
|
||||
Scale factor to convert normalized predictions back to minutes.
|
||||
"""
|
||||
from scipy.optimize import minimize
|
||||
|
||||
n, p = X.shape
|
||||
t_scale = max(t.max(), 1.0)
|
||||
t_norm = np.clip(t / t_scale, 1e-6, 1.0)
|
||||
log_t_norm = np.log(np.maximum(t_norm, 1e-9))
|
||||
|
||||
lr = LinearRegression(fit_intercept=False)
|
||||
lr.fit(X, log_t_norm)
|
||||
beta0 = np.clip(lr.coef_.copy(), -5, 5)
|
||||
|
||||
bounds = [(-10, 10)] * p + [(-5, 3)]
|
||||
init = np.concatenate([beta0, [0.0]])
|
||||
|
||||
result = minimize(
|
||||
_weibull_log_likelihood,
|
||||
init,
|
||||
args=(X, t_norm, event),
|
||||
method="L-BFGS-B",
|
||||
bounds=bounds,
|
||||
options={"maxiter": 2000, "ftol": 1e-10},
|
||||
)
|
||||
if not result.success:
|
||||
logger.warning(f"Weibull MLE did not converge: {result.message}")
|
||||
|
||||
beta = result.x[:p]
|
||||
k = max(np.exp(result.x[p]), 1e-4)
|
||||
return beta, k, t_scale
|
||||
|
||||
|
||||
# ===================================================================
|
||||
# MinutesSurvivalModel
|
||||
# ===================================================================
|
||||
|
||||
class MinutesSurvivalModel(BaseModel):
|
||||
"""Weibull AFT model for minutes-played distribution.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
model_dir : str
|
||||
Directory for persisting trained models.
|
||||
force_scipy : bool
|
||||
If True, use the pure-scipy MLE fallback even when lifelines
|
||||
is installed.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model_dir: str = "models_trained",
|
||||
force_scipy: bool = False,
|
||||
):
|
||||
super().__init__(model_dir)
|
||||
self.force_scipy = force_scipy
|
||||
|
||||
self.scaler = StandardScaler()
|
||||
self.feature_names = None
|
||||
|
||||
# Weibull parameters
|
||||
self._beta = None # feature coefficients → log(λ)
|
||||
self._k = None # shape parameter
|
||||
self._afitter = None # lifelines WeibullAFTFitter instance (if used)
|
||||
self._t_scale = 90.0
|
||||
self._use_lifelines = False
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Fit
|
||||
# ------------------------------------------------------------------
|
||||
def fit(
|
||||
self,
|
||||
X: pd.DataFrame,
|
||||
durations: np.ndarray,
|
||||
events: np.ndarray,
|
||||
**kwargs,
|
||||
):
|
||||
"""Fit the Weibull AFT model.
|
||||
|
||||
Args:
|
||||
X: Feature matrix (one row per player-match).
|
||||
durations: Minutes played (0–90); `y` alias for BaseModel compat.
|
||||
events:
|
||||
0 → exact duration observed (subbed off before 90).
|
||||
1 → right-censored (player completed the full 90 minutes).
|
||||
"""
|
||||
self.feature_names = _select_survival_features(X)
|
||||
X_clean = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.fit_transform(X_clean)
|
||||
|
||||
durations = np.asarray(durations, dtype=float)
|
||||
events = np.asarray(events, dtype=int)
|
||||
|
||||
# Try lifelines first --------------------------------------------------
|
||||
if not self.force_scipy:
|
||||
try:
|
||||
import lifelines # noqa: F401
|
||||
from lifelines import WeibullAFTFitter
|
||||
|
||||
df = pd.DataFrame(X_scaled, columns=self.feature_names)
|
||||
df["duration"] = durations
|
||||
df["event"] = events
|
||||
|
||||
aft = WeibullAFTFitter()
|
||||
aft.fit(df, duration_col="duration", event_col="event")
|
||||
self._afitter = aft
|
||||
self._use_lifelines = True
|
||||
self._t_scale = 1.0 # lifelines works in original duration scale
|
||||
logger.info(
|
||||
"MinutesSurvivalModel fitted via lifelines "
|
||||
f"(n={len(durations)}, features={len(self.feature_names)})"
|
||||
)
|
||||
return self
|
||||
except ImportError:
|
||||
logger.info("lifelines not installed; falling back to scipy MLE")
|
||||
except Exception as exc:
|
||||
logger.warning(f"lifelines failed ({exc}); falling back to scipy MLE")
|
||||
|
||||
# Scipy fallback -------------------------------------------------------
|
||||
self._use_lifelines = False
|
||||
self._beta, self._k, self._t_scale = _fit_weibull_mle(X_scaled, durations, events)
|
||||
logger.info(
|
||||
f"MinutesSurvivalModel fitted via scipy MLE "
|
||||
f"(n={len(durations)}, features={len(self.feature_names)}, "
|
||||
f"k={self._k:.3f})"
|
||||
)
|
||||
return self
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Predict (BaseModel interface — returns expected minutes)
|
||||
# ------------------------------------------------------------------
|
||||
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
||||
"""Return expected minutes (E[T]) — BaseModel interface."""
|
||||
return self.predict_expected_minutes(X)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Preprocessing
|
||||
# ------------------------------------------------------------------
|
||||
def _preprocess(self, X: pd.DataFrame) -> np.ndarray:
|
||||
if self.feature_names is None:
|
||||
raise RuntimeError("Model not trained. Call fit() first.")
|
||||
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
return self.scaler.transform(X_c)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Core distribution
|
||||
# ------------------------------------------------------------------
|
||||
def predict_distribution(
|
||||
self, X: pd.DataFrame
|
||||
) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
|
||||
"""Return (expected_minutes, lower_bound, upper_bound).
|
||||
|
||||
``lower_bound`` and ``upper_bound`` are approximate 95 % confidence
|
||||
intervals derived from the Weibull variance.
|
||||
"""
|
||||
X_scaled = self._preprocess(X)
|
||||
|
||||
if self._use_lifelines and self._afitter is not None:
|
||||
df = pd.DataFrame(X_scaled, columns=self.feature_names)
|
||||
# Lifelines returns median survival in its summary; we approximate
|
||||
# expected minutes using the median and estimated shape.
|
||||
median = self._afitter.predict_median(df).values.flatten()
|
||||
# Heuristic: for Weibull, E[T] ≈ median / (ln 2)^(1/k).
|
||||
# Derive k from the lifelines summary if possible, else guess ~1.
|
||||
try:
|
||||
summary = self._afitter.summary
|
||||
log_k = summary.loc["lambda_", "coef"]
|
||||
k = 1.0 / np.exp(log_k) if abs(log_k) > 1e-8 else 1.0
|
||||
except Exception:
|
||||
k = 1.0
|
||||
expected = median * np.exp(np.log(np.log(2)) / k)
|
||||
# Std via coefficient of variation
|
||||
coef_var = np.sqrt(np.exp(
|
||||
np.log(math.gamma(1 + 2 / k)) - 2 * np.log(math.gamma(1 + 1 / k))
|
||||
))
|
||||
std = expected * coef_var
|
||||
lower = np.maximum(0, expected - 1.96 * std)
|
||||
upper = np.minimum(90, expected + 1.96 * std)
|
||||
return expected, lower, upper
|
||||
|
||||
# Scipy / stored parameters (normalized scale, convert to minutes)
|
||||
log_lambda = X_scaled.dot(self._beta)
|
||||
lambda_ = np.exp(log_lambda) * self._t_scale
|
||||
k = self._k
|
||||
|
||||
# Expected value: λ * Γ(1 + 1/k)
|
||||
gamma_1 = math.gamma(1.0 + 1.0 / k)
|
||||
expected = lambda_ * gamma_1
|
||||
|
||||
# Variance = λ² * (Γ(1+2/k) - Γ²(1+1/k))
|
||||
gamma_2 = math.gamma(1.0 + 2.0 / k)
|
||||
var = (lambda_ ** 2) * (gamma_2 - gamma_1 ** 2)
|
||||
std = np.sqrt(np.maximum(var, 0.01))
|
||||
|
||||
lower = np.maximum(0, expected - 1.96 * std)
|
||||
upper = np.minimum(90, expected + 1.96 * std)
|
||||
expected = np.clip(expected, 0, 90)
|
||||
return expected, lower, upper
|
||||
|
||||
def predict_expected_minutes(self, X: pd.DataFrame) -> np.ndarray:
|
||||
"""Return E[minutes] for each row."""
|
||||
expected, _, _ = self.predict_distribution(X)
|
||||
return expected
|
||||
|
||||
def predict_full_match_probability(self, X: pd.DataFrame) -> np.ndarray:
|
||||
"""Probability the player completes 90 minutes: P(T ≥ 90)."""
|
||||
X_scaled = self._preprocess(X)
|
||||
|
||||
if self._use_lifelines and self._afitter is not None:
|
||||
try:
|
||||
surv = self._afitter.predict_survival_function(
|
||||
pd.DataFrame(X_scaled, columns=self.feature_names),
|
||||
times=[90.0],
|
||||
)
|
||||
return 1.0 - surv.values.flatten()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
log_lambda = X_scaled.dot(self._beta)
|
||||
lambda_ = np.exp(log_lambda) * self._t_scale
|
||||
surv = np.exp(-((90.0 / lambda_) ** self._k))
|
||||
prob = 1.0 - surv
|
||||
return np.clip(prob, 0.0, 1.0)
|
||||
|
||||
def predict_starter_probability(
|
||||
self, X: pd.DataFrame, min_minutes: float = 60.0
|
||||
) -> np.ndarray:
|
||||
"""Probability player plays at least *min_minutes* (default 60).
|
||||
|
||||
Useful as a "likely starter" proxy.
|
||||
"""
|
||||
X_scaled = self._preprocess(X)
|
||||
|
||||
if self._use_lifelines and self._afitter is not None:
|
||||
try:
|
||||
surv = self._afitter.predict_survival_function(
|
||||
pd.DataFrame(X_scaled, columns=self.feature_names),
|
||||
times=[min_minutes],
|
||||
)
|
||||
return surv.values.flatten()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
log_lambda = X_scaled.dot(self._beta)
|
||||
lambda_ = np.exp(log_lambda) * self._t_scale
|
||||
surv = np.exp(-((min_minutes / lambda_) ** self._k))
|
||||
return np.clip(surv, 0.0, 1.0)
|
||||
Reference in New Issue
Block a user