feat: add 10 new ML models for auction optimization (Phases 1-6)

Phase 1 - Quick Wins:
- QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding
- MinutesSurvivalModel: Weibull AFT for minutes distribution modeling

Phase 2 - Adaptive Auction:
- BanditAuctionSolver: Thompson Sampling for live auction bids
- OpponentBidModel: Predict competitor bids via LightGBM
- BudgetOptimizer: Bayesian optimization for role-level allocation

Phase 3 - Deep Learning:
- RLAuctionPolicy: Double DQN agent for auction strategy
- SetTransformer: Team composition valuation via set-based ML

Phase 4 - Probabilistic:
- BayesianPlayerModel: Hierarchical pooling for rookie uncertainty
- ConformalPredictor: Calibrated prediction intervals

Phase 5 - Chemistry & Form:
- PlayerChemistryGAT: Graph attention network for player synergies
- PlayerFormModel: Hawkes process for form momentum

Phase 6 - Causal:
- TransferCausalModel: Causal forest for transfer effects
- AuctionEffectAnalyzer: Bid adjustment from causal analysis

81 tests passing
This commit is contained in:
ramseshk
2026-08-11 17:56:03 +08:00
parent 916278a640
commit b0fab62a87
16 changed files with 4955 additions and 21 deletions
+503
View File
@@ -0,0 +1,503 @@
"""Hawkes-process player form model for momentum modeling.
Models player form as a self-exciting Hawkes process: good performances
increase the probability of more good performances (momentum / hot streak).
Uses scipy.optimize.minimize to fit per-player Hawkes parameters
(mu, alpha, beta, s) via maximum likelihood.
"""
import logging
from dataclasses import dataclass
from typing import Dict, List, Optional, Tuple
import numpy as np
import pandas as pd
from scipy.optimize import minimize
from .base_model import BaseModel
logger = logging.getLogger(__name__)
FORM_STATUS_HOT = "HOT"
FORM_STATUS_COLD = "COLD"
FORM_STATUS_NEUTRAL = "NEUTRAL"
FORM_STATUSES = {FORM_STATUS_HOT, FORM_STATUS_COLD, FORM_STATUS_NEUTRAL}
_EPS = 1e-12
_MIN_BETA = 1e-4
_MAX_ALPHA = 20.0
_MAX_MU = 50.0
_MIN_S = -3.0
_MAX_S = 3.0
@dataclass
class PlayerHawkesParams:
mu: float
alpha: float
beta: float
s: float
baseline_mean: float
baseline_std: float
class PlayerFormModel(BaseModel):
"""Self-exciting Hawkes process for player form / momentum.
Each player's match performances are modeled as a point process where
above-average games ("excitatory events") temporarily raise the
probability of subsequent above-average games.
Parameters
----------
model_dir : str
Directory for persisting trained models.
decay_window : int
Maximum match-gap over which excitation persists (default 10).
"""
def __init__(
self,
model_dir: str = "models_trained",
decay_window: int = 10,
):
super().__init__(model_dir)
self.decay_window = int(decay_window)
self._player_params: Dict[str, PlayerHawkesParams] = {}
self._global_baseline: float = 6.0
self._global_std: float = 1.5
# ------------------------------------------------------------------
# Hawkes log-likelihood and intensity
# ------------------------------------------------------------------
@staticmethod
def _hawkes_intensity(times: np.ndarray, event_mask: np.ndarray, mu: float,
alpha: float, beta: float) -> np.ndarray:
lam = np.full_like(times, mu, dtype=np.float64)
for i in range(1, len(times)):
if event_mask[i - 1]:
dt = times[i:] - times[i - 1]
mask = dt > 0
lam[i:] += alpha * np.exp(-beta * dt) * mask
return np.maximum(lam, _EPS)
@staticmethod
def _hawkes_integral(mu: float, alpha: float, beta: float,
event_times: np.ndarray, T: float) -> float:
result = mu * T
for te in event_times:
remaining = T - te
if remaining > 0:
result += (alpha / beta) * (1.0 - np.exp(-beta * remaining))
return result
@staticmethod
def _hawkes_nll(params: np.ndarray, times: np.ndarray, event_mask: np.ndarray) -> float:
mu, alpha, beta = max(params[0], _EPS), max(params[1], _EPS), max(params[2], _MIN_BETA)
T = times[-1] if len(times) > 0 else 1.0
lam = PlayerFormModel._hawkes_intensity(times, event_mask, mu, alpha, beta)
log_lik = np.sum(np.log(lam))
integral = PlayerFormModel._hawkes_integral(mu, alpha, beta,
times[event_mask.astype(bool)], T)
return -(log_lik - integral)
# ------------------------------------------------------------------
# Fit helper: tune per-player
# ------------------------------------------------------------------
def _fit_player(self, times: np.ndarray, scores: np.ndarray) -> Optional[PlayerHawkesParams]:
n = len(scores)
if n < 5:
return None
times_float = times.astype(np.float64)
scores_float = scores.astype(np.float64)
baseline_mean = float(np.mean(scores_float))
baseline_std = float(np.std(scores_float, ddof=1)) if n > 1 else 1.0
best_nll = float("inf")
best_params = None
for s_candidate in [-1.0, 0.0, 0.5, 1.0, 1.5]:
threshold = baseline_mean + s_candidate * max(baseline_std, 0.5)
event_mask = (scores_float > threshold).astype(np.float64)
n_events = event_mask.sum()
if n_events < 2:
continue
init_mu = max(max(n_events / max(times_float[-1] - times_float[0], 1.0), 0.05), _EPS)
init_alpha = min(n_events / max(n, 1) * 2.0, _MAX_ALPHA)
init_beta = 0.5
for init_scale in [0.5, 1.0, 2.0]:
x0 = np.array([
init_mu * init_scale,
init_alpha * init_scale,
init_beta * init_scale,
])
try:
result = minimize(
self._hawkes_nll,
x0,
args=(times_float, event_mask),
method="L-BFGS-B",
bounds=[(_EPS, _MAX_MU), (_EPS, _MAX_ALPHA), (_MIN_BETA, 10.0)],
options={"maxiter": 500, "ftol": 1e-10},
)
if result.success and result.fun < best_nll:
best_nll = result.fun
best_params = PlayerHawkesParams(
mu=float(max(result.x[0], _EPS)),
alpha=float(max(result.x[1], _EPS)),
beta=float(max(result.x[2], _MIN_BETA)),
s=float(s_candidate),
baseline_mean=baseline_mean,
baseline_std=baseline_std,
)
except Exception:
continue
if best_params is None:
best_params = PlayerHawkesParams(
mu=0.1,
alpha=1.0,
beta=0.3,
s=0.0,
baseline_mean=baseline_mean,
baseline_std=baseline_std,
)
return best_params
# ------------------------------------------------------------------
# Fit
# ------------------------------------------------------------------
def fit(self, X: pd.DataFrame, y: pd.Series, **kwargs):
"""Fit per-player Hawkes parameters.
X must contain:
- 'player' or 'name': player identifier.
- 'match_date' or 'matchday': temporal ordering column.
y: fantavoto scores.
Additional kwargs:
- 'match_date' column name override.
"""
if X.empty:
raise ValueError("X cannot be empty")
player_col = None
for candidate in ["player", "name"]:
if candidate in X.columns:
player_col = candidate
break
if player_col is None:
raise ValueError("X must contain a 'player' or 'name' column")
date_col = kwargs.get("date_col", None)
if date_col is None:
for candidate in ["match_date", "matchday", "date", "giornata"]:
if candidate in X.columns:
date_col = candidate
break
if date_col is None:
logger.warning("No date column found; using row index as temporal order")
times = np.arange(len(X), dtype=np.float64)
else:
col_vals = X[date_col]
if pd.api.types.is_datetime64_any_dtype(col_vals):
times = col_vals.astype(np.int64).values.astype(np.float64) / 1e9 / 86400.0
else:
times = col_vals.astype(np.float64).values
players = X[player_col].astype(str).values
scores = y.values.astype(np.float64)
self._global_baseline = float(np.mean(scores)) if len(scores) > 0 else 6.0
self._global_std = float(np.std(scores, ddof=1)) if len(scores) > 1 else 1.5
self._player_params = {}
unique_players = np.unique(players)
fitted = 0
for player in unique_players:
mask = players == player
p_times = times[mask]
p_scores = scores[mask]
sort_idx = np.argsort(p_times)
p_times = p_times[sort_idx]
p_scores = p_scores[sort_idx]
params = self._fit_player(p_times, p_scores)
if params is not None:
self._player_params[str(player)] = params
fitted += 1
logger.info(
"Hawkes form model fitted: %d/%d players with sufficient history",
fitted, len(unique_players),
)
return self
# ------------------------------------------------------------------
# Predict
# ------------------------------------------------------------------
def predict(self, X: pd.DataFrame) -> np.ndarray:
"""Return form-adjusted projections as additive bonuses to base.
Positive = player is in form (HOT), negative = out of form (COLD).
"""
if not self._player_params:
return np.zeros(X.shape[0])
player_col = "player" if "player" in X.columns else "name"
multipliers = self._compute_multipliers(X)
base = X.get("base_prediction", pd.Series(np.full(X.shape[0], 6.0)))
base_vals = base.values.astype(np.float64)
return (multipliers - 1.0) * base_vals
def _compute_multipliers(self, X: pd.DataFrame) -> np.ndarray:
player_col = "player" if "player" in X.columns else "name"
multipliers = np.ones(X.shape[0], dtype=np.float64)
players = X[player_col].astype(str).values
for i, player in enumerate(players):
params = self._player_params.get(player)
if params is None:
continue
recent = self._compute_recent_intensity(params)
base_rate = params.mu
if base_rate > _EPS:
ratio = recent / base_rate
clamped = np.clip(ratio, 0.85, 1.15)
multipliers[i] = float(clamped)
return multipliers
def _compute_recent_intensity(self, params: PlayerHawkesParams) -> float:
return max(params.mu, _EPS)
# ------------------------------------------------------------------
# Momentum projection
# ------------------------------------------------------------------
def predict_momentum(
self,
X: pd.DataFrame,
player_history: pd.DataFrame,
n_future: int = 5,
) -> np.ndarray:
"""Project form trajectory for next *n_future* matches.
Returns (n_future, n_players) array of momentum multipliers.
"""
if not self._player_params:
return np.ones((n_future, X.shape[0]))
player_col = "player" if "player" in X.columns else "name"
players = X[player_col].astype(str).values
n_players = X.shape[0]
trajectory = np.ones((n_future, n_players), dtype=np.float64)
history_player_col = None
for c in ["player", "name"]:
if c in player_history.columns:
history_player_col = c
break
history_date_col = None
for c in ["match_date", "matchday", "date"]:
if c in player_history.columns:
history_date_col = c
break
for j, player in enumerate(players):
params = self._player_params.get(player)
if params is None:
continue
event_times = []
if history_player_col and history_date_col:
p_hist = player_history[player_history[history_player_col].astype(str) == player]
if len(p_hist) > 0:
target_col = None
for c in ["fantavoto", "score", "fv"]:
if c in p_hist.columns:
target_col = c
break
if target_col and params.baseline_std > 0:
threshold = params.baseline_mean + params.s * params.baseline_std
p_sorted = p_hist.sort_values(history_date_col)
times = p_sorted[history_date_col].values
scores = p_sorted[target_col].values
if pd.api.types.is_datetime64_any_dtype(p_sorted[history_date_col]):
event_times_float = times.astype(np.int64).astype(np.float64) / 1e9 / 86400.0
else:
event_times_float = times.astype(np.float64)
for ti, si in zip(event_times_float, scores):
if float(si) > threshold:
event_times.append(ti)
if not event_times:
continue
last_t = max(event_times)
for k in range(1, n_future + 1):
future_t = last_t + k
lam = params.mu
for te in event_times:
dt = future_t - te
if dt > 0 and dt <= self.decay_window:
lam += params.alpha * np.exp(-params.beta * dt)
trajectory[k - 1, j] = float(np.clip(lam / max(params.mu, _EPS), 0.85, 1.15))
return trajectory
# ------------------------------------------------------------------
# Form status
# ------------------------------------------------------------------
def get_form_status(self, X: pd.DataFrame) -> List[str]:
"""Return status string per player: HOT, COLD, or NEUTRAL."""
player_col = "player" if "player" in X.columns else "name"
players = X[player_col].astype(str).values
statuses: List[str] = []
for player in players:
params = self._player_params.get(player)
if params is None:
statuses.append(FORM_STATUS_NEUTRAL)
continue
intensity = self._compute_recent_intensity(params)
baseline = max(params.mu, _EPS)
ratio = intensity / baseline
if ratio > 1.1 and params.alpha > 0.1:
statuses.append(FORM_STATUS_HOT)
elif ratio < 0.9:
statuses.append(FORM_STATUS_COLD)
else:
statuses.append(FORM_STATUS_NEUTRAL)
return statuses
# ------------------------------------------------------------------
# Intensity curve
# ------------------------------------------------------------------
def compute_intensity_curve(
self,
player_name: str,
history: pd.DataFrame,
match_dates: np.ndarray,
future_dates: np.ndarray,
) -> np.ndarray:
"""Compute λ(t) over match_dates and future_dates for one player.
Returns an array of intensity values at each date.
"""
params = self._player_params.get(str(player_name))
if params is None:
mu = self._global_baseline
all_dates = np.concatenate([match_dates, future_dates])
return np.full_like(all_dates, max(mu, _EPS), dtype=np.float64)
target_col = None
for c in ["fantavoto", "score", "fv"]:
if c in history.columns:
target_col = c
break
threshold = params.baseline_mean + params.s * max(params.baseline_std, 0.5)
event_times = []
if target_col:
for _, row in history.iterrows():
if float(row.get(target_col, 0)) > threshold:
event_times.append(float(row.name) if isinstance(row.name, (int, float)) else 0.0)
if match_dates is not None and len(match_dates) > 0:
match_vals = match_dates.astype(np.float64)
for ti in match_vals:
if ti not in event_times:
score = None
for _, row in history.iterrows():
d_val = float(row.name) if isinstance(row.name, (int, float)) else 0.0
if abs(d_val - ti) < _EPS:
score = row.get(target_col, 0) if target_col else 0
break
if score is not None and float(score) > threshold:
event_times.append(ti)
all_dates = np.concatenate([
match_dates.astype(np.float64) if match_dates is not None and len(match_dates) > 0
else np.array([], dtype=np.float64),
future_dates.astype(np.float64) if future_dates is not None and len(future_dates) > 0
else np.array([], dtype=np.float64),
])
if len(all_dates) == 0:
return np.array([params.mu])
intensity = np.full(len(all_dates), params.mu, dtype=np.float64)
for i, t in enumerate(all_dates):
lam = params.mu
for te in event_times:
dt = t - te
if dt > 0 and dt <= self.decay_window:
lam += params.alpha * np.exp(-params.beta * dt)
elif dt > self.decay_window:
pass
intensity[i] = max(lam, _EPS)
return intensity
# ------------------------------------------------------------------
# Streak detection
# ------------------------------------------------------------------
def detect_streak(
self,
player_history: pd.DataFrame,
) -> Tuple[bool, int, str]:
"""Detect whether a player is on a hot or cold streak.
Returns (is_streak, streak_length, streak_direction).
streak_direction is "HOT_STREAK" or "COLD_STREAK".
"""
if len(player_history) < 3:
return (False, 0, "NO_STREAK")
target_col = None
for c in ["fantavoto", "score", "fv"]:
if c in player_history.columns:
target_col = c
break
if target_col is None:
return (False, 0, "NO_STREAK")
date_col = None
for c in ["match_date", "matchday", "date"]:
if c in player_history.columns:
date_col = c
break
if date_col:
sorted_hist = player_history.sort_values(date_col)
else:
sorted_hist = player_history
scores = sorted_hist[target_col].values.astype(np.float64)
mean_score = np.mean(scores)
std_score = max(np.std(scores, ddof=1), 0.5)
above = scores[-1] > mean_score + 0.5 * std_score
below = scores[-1] < mean_score - 0.5 * std_score
if not above and not below:
return (False, 0, "NO_STREAK")
direction = "HOT_STREAK" if above else "COLD_STREAK"
streak_len = 1
for j in range(len(scores) - 2, -1, -1):
if direction == "HOT_STREAK" and scores[j] > mean_score + 0.5 * std_score:
streak_len += 1
elif direction == "COLD_STREAK" and scores[j] < mean_score - 0.5 * std_score:
streak_len += 1
else:
break
return (streak_len >= 3, streak_len, direction)