feat: add 10 new ML models for auction optimization (Phases 1-6)

Phase 1 - Quick Wins:
- QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding
- MinutesSurvivalModel: Weibull AFT for minutes distribution modeling

Phase 2 - Adaptive Auction:
- BanditAuctionSolver: Thompson Sampling for live auction bids
- OpponentBidModel: Predict competitor bids via LightGBM
- BudgetOptimizer: Bayesian optimization for role-level allocation

Phase 3 - Deep Learning:
- RLAuctionPolicy: Double DQN agent for auction strategy
- SetTransformer: Team composition valuation via set-based ML

Phase 4 - Probabilistic:
- BayesianPlayerModel: Hierarchical pooling for rookie uncertainty
- ConformalPredictor: Calibrated prediction intervals

Phase 5 - Chemistry & Form:
- PlayerChemistryGAT: Graph attention network for player synergies
- PlayerFormModel: Hawkes process for form momentum

Phase 6 - Causal:
- TransferCausalModel: Causal forest for transfer effects
- AuctionEffectAnalyzer: Bid adjustment from causal analysis

81 tests passing
This commit is contained in:
ramseshk
2026-08-11 17:56:03 +08:00
parent 916278a640
commit b0fab62a87
16 changed files with 4955 additions and 21 deletions
+354
View File
@@ -0,0 +1,354 @@
"""Minutes-played survival model via Weibull AFT.
Models the distribution of minutes played per matchweek using Weibull
Accelerated Failure Time. Supports both lifelines (preferred) and a pure-scipy
MLE fallback so the module works in minimal environments.
Provides:
- Expected minutes / confidence intervals
- Starter probability (≥60 min)
- Full-match probability (90 min)
"""
import logging
import math
from typing import Optional, Tuple
import numpy as np
import pandas as pd
from sklearn.linear_model import LinearRegression
from sklearn.preprocessing import StandardScaler
from .base_model import BaseModel
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Feature-selection keyword list
# ---------------------------------------------------------------------------
_MINUTE_KEYWORDS = [
"minute", "game", "rest", "fatigue", "age", "injury",
"played", "starter", "bench", "appearance",
"recovery", "rotation", "squad", "season",
"match", "form", "fitness",
]
def _select_survival_features(X: pd.DataFrame) -> list:
"""Pick columns whose name contains any survival-relevant keyword."""
lower_cols = {c: str(c).lower() for c in X.columns}
selected = [
c for c, cl in lower_cols.items()
if any(kw in cl for kw in _MINUTE_KEYWORDS)
]
if not selected:
selected = list(X.select_dtypes(include=[np.number]).columns[:20])
logger.info("No keyword-matched survival features; using first 20 numeric columns")
else:
logger.info(f"Selected {len(selected)} survival features via keyword matching")
return selected
# ===================================================================
# Weibull helper functions for the scipy fallback
# ===================================================================
def _weibull_log_likelihood(params, X, t, event, eps=1e-10):
"""Negative log-likelihood for Weibull AFT model.
Parameters
----------
params : ndarray (p_features + 1,)
First p entries: beta (coefficients for X).
Last entry: log_k (log shape parameter ensures k > 0).
X : ndarray (n, p)
Scaled feature matrix.
t : ndarray (n,)
Observed durations (minutes played).
event : ndarray (n,)
0 → exact failure (subbed off), 1 → right-censored (completed 90).
eps : float
Small epsilon for numerical stability.
Returns
-------
neg_ll : float
Negative log-likelihood (to be minimized).
"""
p = X.shape[1]
beta = params[:p]
log_k = params[p]
k = np.exp(log_k) + eps
log_lambda = X.dot(beta) # log(λ_i) = X_i * beta
lambda_ = np.exp(log_lambda) + eps
log_t = np.log(np.maximum(t, eps))
z = t / lambda_
# Log-PDF for uncensored (event == 0)
log_pdf = np.log(k) - log_lambda + (k - 1.0) * (log_t - log_lambda) - z ** k
# Log-SF for censored (event == 1)
log_sf = -(z ** k)
# event==1 → censored → use SF; event==0 → observed → use PDF
ll = np.where(event == 1, log_sf, log_pdf)
return -ll.sum()
def _fit_weibull_mle(X, t, event):
"""Fit Weibull AFT via scipy MLE.
Returns
-------
beta : ndarray (p,)
Feature coefficients (scaled to original duration range).
k : float
Shape parameter.
t_scale : float
Scale factor to convert normalized predictions back to minutes.
"""
from scipy.optimize import minimize
n, p = X.shape
t_scale = max(t.max(), 1.0)
t_norm = np.clip(t / t_scale, 1e-6, 1.0)
log_t_norm = np.log(np.maximum(t_norm, 1e-9))
lr = LinearRegression(fit_intercept=False)
lr.fit(X, log_t_norm)
beta0 = np.clip(lr.coef_.copy(), -5, 5)
bounds = [(-10, 10)] * p + [(-5, 3)]
init = np.concatenate([beta0, [0.0]])
result = minimize(
_weibull_log_likelihood,
init,
args=(X, t_norm, event),
method="L-BFGS-B",
bounds=bounds,
options={"maxiter": 2000, "ftol": 1e-10},
)
if not result.success:
logger.warning(f"Weibull MLE did not converge: {result.message}")
beta = result.x[:p]
k = max(np.exp(result.x[p]), 1e-4)
return beta, k, t_scale
# ===================================================================
# MinutesSurvivalModel
# ===================================================================
class MinutesSurvivalModel(BaseModel):
"""Weibull AFT model for minutes-played distribution.
Parameters
----------
model_dir : str
Directory for persisting trained models.
force_scipy : bool
If True, use the pure-scipy MLE fallback even when lifelines
is installed.
"""
def __init__(
self,
model_dir: str = "models_trained",
force_scipy: bool = False,
):
super().__init__(model_dir)
self.force_scipy = force_scipy
self.scaler = StandardScaler()
self.feature_names = None
# Weibull parameters
self._beta = None # feature coefficients → log(λ)
self._k = None # shape parameter
self._afitter = None # lifelines WeibullAFTFitter instance (if used)
self._t_scale = 90.0
self._use_lifelines = False
# ------------------------------------------------------------------
# Fit
# ------------------------------------------------------------------
def fit(
self,
X: pd.DataFrame,
durations: np.ndarray,
events: np.ndarray,
**kwargs,
):
"""Fit the Weibull AFT model.
Args:
X: Feature matrix (one row per player-match).
durations: Minutes played (0–90); `y` alias for BaseModel compat.
events:
0 → exact duration observed (subbed off before 90).
1 → right-censored (player completed the full 90 minutes).
"""
self.feature_names = _select_survival_features(X)
X_clean = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
X_scaled = self.scaler.fit_transform(X_clean)
durations = np.asarray(durations, dtype=float)
events = np.asarray(events, dtype=int)
# Try lifelines first --------------------------------------------------
if not self.force_scipy:
try:
import lifelines # noqa: F401
from lifelines import WeibullAFTFitter
df = pd.DataFrame(X_scaled, columns=self.feature_names)
df["duration"] = durations
df["event"] = events
aft = WeibullAFTFitter()
aft.fit(df, duration_col="duration", event_col="event")
self._afitter = aft
self._use_lifelines = True
self._t_scale = 1.0 # lifelines works in original duration scale
logger.info(
"MinutesSurvivalModel fitted via lifelines "
f"(n={len(durations)}, features={len(self.feature_names)})"
)
return self
except ImportError:
logger.info("lifelines not installed; falling back to scipy MLE")
except Exception as exc:
logger.warning(f"lifelines failed ({exc}); falling back to scipy MLE")
# Scipy fallback -------------------------------------------------------
self._use_lifelines = False
self._beta, self._k, self._t_scale = _fit_weibull_mle(X_scaled, durations, events)
logger.info(
f"MinutesSurvivalModel fitted via scipy MLE "
f"(n={len(durations)}, features={len(self.feature_names)}, "
f"k={self._k:.3f})"
)
return self
# ------------------------------------------------------------------
# Predict (BaseModel interface — returns expected minutes)
# ------------------------------------------------------------------
def predict(self, X: pd.DataFrame) -> np.ndarray:
"""Return expected minutes (E[T]) — BaseModel interface."""
return self.predict_expected_minutes(X)
# ------------------------------------------------------------------
# Preprocessing
# ------------------------------------------------------------------
def _preprocess(self, X: pd.DataFrame) -> np.ndarray:
if self.feature_names is None:
raise RuntimeError("Model not trained. Call fit() first.")
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
return self.scaler.transform(X_c)
# ------------------------------------------------------------------
# Core distribution
# ------------------------------------------------------------------
def predict_distribution(
self, X: pd.DataFrame
) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
"""Return (expected_minutes, lower_bound, upper_bound).
``lower_bound`` and ``upper_bound`` are approximate 95 % confidence
intervals derived from the Weibull variance.
"""
X_scaled = self._preprocess(X)
if self._use_lifelines and self._afitter is not None:
df = pd.DataFrame(X_scaled, columns=self.feature_names)
# Lifelines returns median survival in its summary; we approximate
# expected minutes using the median and estimated shape.
median = self._afitter.predict_median(df).values.flatten()
# Heuristic: for Weibull, E[T] ≈ median / (ln 2)^(1/k).
# Derive k from the lifelines summary if possible, else guess ~1.
try:
summary = self._afitter.summary
log_k = summary.loc["lambda_", "coef"]
k = 1.0 / np.exp(log_k) if abs(log_k) > 1e-8 else 1.0
except Exception:
k = 1.0
expected = median * np.exp(np.log(np.log(2)) / k)
# Std via coefficient of variation
coef_var = np.sqrt(np.exp(
np.log(math.gamma(1 + 2 / k)) - 2 * np.log(math.gamma(1 + 1 / k))
))
std = expected * coef_var
lower = np.maximum(0, expected - 1.96 * std)
upper = np.minimum(90, expected + 1.96 * std)
return expected, lower, upper
# Scipy / stored parameters (normalized scale, convert to minutes)
log_lambda = X_scaled.dot(self._beta)
lambda_ = np.exp(log_lambda) * self._t_scale
k = self._k
# Expected value: λ * Γ(1 + 1/k)
gamma_1 = math.gamma(1.0 + 1.0 / k)
expected = lambda_ * gamma_1
# Variance = λ² * (Γ(1+2/k) - Γ²(1+1/k))
gamma_2 = math.gamma(1.0 + 2.0 / k)
var = (lambda_ ** 2) * (gamma_2 - gamma_1 ** 2)
std = np.sqrt(np.maximum(var, 0.01))
lower = np.maximum(0, expected - 1.96 * std)
upper = np.minimum(90, expected + 1.96 * std)
expected = np.clip(expected, 0, 90)
return expected, lower, upper
def predict_expected_minutes(self, X: pd.DataFrame) -> np.ndarray:
"""Return E[minutes] for each row."""
expected, _, _ = self.predict_distribution(X)
return expected
def predict_full_match_probability(self, X: pd.DataFrame) -> np.ndarray:
"""Probability the player completes 90 minutes: P(T ≥ 90)."""
X_scaled = self._preprocess(X)
if self._use_lifelines and self._afitter is not None:
try:
surv = self._afitter.predict_survival_function(
pd.DataFrame(X_scaled, columns=self.feature_names),
times=[90.0],
)
return 1.0 - surv.values.flatten()
except Exception:
pass
log_lambda = X_scaled.dot(self._beta)
lambda_ = np.exp(log_lambda) * self._t_scale
surv = np.exp(-((90.0 / lambda_) ** self._k))
prob = 1.0 - surv
return np.clip(prob, 0.0, 1.0)
def predict_starter_probability(
self, X: pd.DataFrame, min_minutes: float = 60.0
) -> np.ndarray:
"""Probability player plays at least *min_minutes* (default 60).
Useful as a "likely starter" proxy.
"""
X_scaled = self._preprocess(X)
if self._use_lifelines and self._afitter is not None:
try:
surv = self._afitter.predict_survival_function(
pd.DataFrame(X_scaled, columns=self.feature_names),
times=[min_minutes],
)
return surv.values.flatten()
except Exception:
pass
log_lambda = X_scaled.dot(self._beta)
lambda_ = np.exp(log_lambda) * self._t_scale
surv = np.exp(-((min_minutes / lambda_) ** self._k))
return np.clip(surv, 0.0, 1.0)