b0fab62a87
Phase 1 - Quick Wins: - QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding - MinutesSurvivalModel: Weibull AFT for minutes distribution modeling Phase 2 - Adaptive Auction: - BanditAuctionSolver: Thompson Sampling for live auction bids - OpponentBidModel: Predict competitor bids via LightGBM - BudgetOptimizer: Bayesian optimization for role-level allocation Phase 3 - Deep Learning: - RLAuctionPolicy: Double DQN agent for auction strategy - SetTransformer: Team composition valuation via set-based ML Phase 4 - Probabilistic: - BayesianPlayerModel: Hierarchical pooling for rookie uncertainty - ConformalPredictor: Calibrated prediction intervals Phase 5 - Chemistry & Form: - PlayerChemistryGAT: Graph attention network for player synergies - PlayerFormModel: Hawkes process for form momentum Phase 6 - Causal: - TransferCausalModel: Causal forest for transfer effects - AuctionEffectAnalyzer: Bid adjustment from causal analysis 81 tests passing
504 lines
19 KiB
Python
504 lines
19 KiB
Python
"""Hawkes-process player form model for momentum modeling.
|
|
|
|
Models player form as a self-exciting Hawkes process: good performances
|
|
increase the probability of more good performances (momentum / hot streak).
|
|
|
|
Uses scipy.optimize.minimize to fit per-player Hawkes parameters
|
|
(mu, alpha, beta, s) via maximum likelihood.
|
|
"""
|
|
|
|
import logging
|
|
from dataclasses import dataclass
|
|
from typing import Dict, List, Optional, Tuple
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
from scipy.optimize import minimize
|
|
|
|
from .base_model import BaseModel
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
FORM_STATUS_HOT = "HOT"
|
|
FORM_STATUS_COLD = "COLD"
|
|
FORM_STATUS_NEUTRAL = "NEUTRAL"
|
|
FORM_STATUSES = {FORM_STATUS_HOT, FORM_STATUS_COLD, FORM_STATUS_NEUTRAL}
|
|
|
|
_EPS = 1e-12
|
|
_MIN_BETA = 1e-4
|
|
_MAX_ALPHA = 20.0
|
|
_MAX_MU = 50.0
|
|
_MIN_S = -3.0
|
|
_MAX_S = 3.0
|
|
|
|
|
|
@dataclass
|
|
class PlayerHawkesParams:
|
|
mu: float
|
|
alpha: float
|
|
beta: float
|
|
s: float
|
|
baseline_mean: float
|
|
baseline_std: float
|
|
|
|
|
|
class PlayerFormModel(BaseModel):
|
|
"""Self-exciting Hawkes process for player form / momentum.
|
|
|
|
Each player's match performances are modeled as a point process where
|
|
above-average games ("excitatory events") temporarily raise the
|
|
probability of subsequent above-average games.
|
|
|
|
Parameters
|
|
----------
|
|
model_dir : str
|
|
Directory for persisting trained models.
|
|
decay_window : int
|
|
Maximum match-gap over which excitation persists (default 10).
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
model_dir: str = "models_trained",
|
|
decay_window: int = 10,
|
|
):
|
|
super().__init__(model_dir)
|
|
self.decay_window = int(decay_window)
|
|
self._player_params: Dict[str, PlayerHawkesParams] = {}
|
|
self._global_baseline: float = 6.0
|
|
self._global_std: float = 1.5
|
|
|
|
# ------------------------------------------------------------------
|
|
# Hawkes log-likelihood and intensity
|
|
# ------------------------------------------------------------------
|
|
@staticmethod
|
|
def _hawkes_intensity(times: np.ndarray, event_mask: np.ndarray, mu: float,
|
|
alpha: float, beta: float) -> np.ndarray:
|
|
lam = np.full_like(times, mu, dtype=np.float64)
|
|
for i in range(1, len(times)):
|
|
if event_mask[i - 1]:
|
|
dt = times[i:] - times[i - 1]
|
|
mask = dt > 0
|
|
lam[i:] += alpha * np.exp(-beta * dt) * mask
|
|
return np.maximum(lam, _EPS)
|
|
|
|
@staticmethod
|
|
def _hawkes_integral(mu: float, alpha: float, beta: float,
|
|
event_times: np.ndarray, T: float) -> float:
|
|
result = mu * T
|
|
for te in event_times:
|
|
remaining = T - te
|
|
if remaining > 0:
|
|
result += (alpha / beta) * (1.0 - np.exp(-beta * remaining))
|
|
return result
|
|
|
|
@staticmethod
|
|
def _hawkes_nll(params: np.ndarray, times: np.ndarray, event_mask: np.ndarray) -> float:
|
|
mu, alpha, beta = max(params[0], _EPS), max(params[1], _EPS), max(params[2], _MIN_BETA)
|
|
T = times[-1] if len(times) > 0 else 1.0
|
|
lam = PlayerFormModel._hawkes_intensity(times, event_mask, mu, alpha, beta)
|
|
log_lik = np.sum(np.log(lam))
|
|
integral = PlayerFormModel._hawkes_integral(mu, alpha, beta,
|
|
times[event_mask.astype(bool)], T)
|
|
return -(log_lik - integral)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Fit helper: tune per-player
|
|
# ------------------------------------------------------------------
|
|
def _fit_player(self, times: np.ndarray, scores: np.ndarray) -> Optional[PlayerHawkesParams]:
|
|
n = len(scores)
|
|
if n < 5:
|
|
return None
|
|
|
|
times_float = times.astype(np.float64)
|
|
scores_float = scores.astype(np.float64)
|
|
baseline_mean = float(np.mean(scores_float))
|
|
baseline_std = float(np.std(scores_float, ddof=1)) if n > 1 else 1.0
|
|
|
|
best_nll = float("inf")
|
|
best_params = None
|
|
|
|
for s_candidate in [-1.0, 0.0, 0.5, 1.0, 1.5]:
|
|
threshold = baseline_mean + s_candidate * max(baseline_std, 0.5)
|
|
event_mask = (scores_float > threshold).astype(np.float64)
|
|
n_events = event_mask.sum()
|
|
if n_events < 2:
|
|
continue
|
|
|
|
init_mu = max(max(n_events / max(times_float[-1] - times_float[0], 1.0), 0.05), _EPS)
|
|
init_alpha = min(n_events / max(n, 1) * 2.0, _MAX_ALPHA)
|
|
init_beta = 0.5
|
|
|
|
for init_scale in [0.5, 1.0, 2.0]:
|
|
x0 = np.array([
|
|
init_mu * init_scale,
|
|
init_alpha * init_scale,
|
|
init_beta * init_scale,
|
|
])
|
|
|
|
try:
|
|
result = minimize(
|
|
self._hawkes_nll,
|
|
x0,
|
|
args=(times_float, event_mask),
|
|
method="L-BFGS-B",
|
|
bounds=[(_EPS, _MAX_MU), (_EPS, _MAX_ALPHA), (_MIN_BETA, 10.0)],
|
|
options={"maxiter": 500, "ftol": 1e-10},
|
|
)
|
|
if result.success and result.fun < best_nll:
|
|
best_nll = result.fun
|
|
best_params = PlayerHawkesParams(
|
|
mu=float(max(result.x[0], _EPS)),
|
|
alpha=float(max(result.x[1], _EPS)),
|
|
beta=float(max(result.x[2], _MIN_BETA)),
|
|
s=float(s_candidate),
|
|
baseline_mean=baseline_mean,
|
|
baseline_std=baseline_std,
|
|
)
|
|
except Exception:
|
|
continue
|
|
|
|
if best_params is None:
|
|
best_params = PlayerHawkesParams(
|
|
mu=0.1,
|
|
alpha=1.0,
|
|
beta=0.3,
|
|
s=0.0,
|
|
baseline_mean=baseline_mean,
|
|
baseline_std=baseline_std,
|
|
)
|
|
|
|
return best_params
|
|
|
|
# ------------------------------------------------------------------
|
|
# Fit
|
|
# ------------------------------------------------------------------
|
|
def fit(self, X: pd.DataFrame, y: pd.Series, **kwargs):
|
|
"""Fit per-player Hawkes parameters.
|
|
|
|
X must contain:
|
|
- 'player' or 'name': player identifier.
|
|
- 'match_date' or 'matchday': temporal ordering column.
|
|
|
|
y: fantavoto scores.
|
|
|
|
Additional kwargs:
|
|
- 'match_date' column name override.
|
|
"""
|
|
if X.empty:
|
|
raise ValueError("X cannot be empty")
|
|
|
|
player_col = None
|
|
for candidate in ["player", "name"]:
|
|
if candidate in X.columns:
|
|
player_col = candidate
|
|
break
|
|
if player_col is None:
|
|
raise ValueError("X must contain a 'player' or 'name' column")
|
|
|
|
date_col = kwargs.get("date_col", None)
|
|
if date_col is None:
|
|
for candidate in ["match_date", "matchday", "date", "giornata"]:
|
|
if candidate in X.columns:
|
|
date_col = candidate
|
|
break
|
|
if date_col is None:
|
|
logger.warning("No date column found; using row index as temporal order")
|
|
times = np.arange(len(X), dtype=np.float64)
|
|
else:
|
|
col_vals = X[date_col]
|
|
if pd.api.types.is_datetime64_any_dtype(col_vals):
|
|
times = col_vals.astype(np.int64).values.astype(np.float64) / 1e9 / 86400.0
|
|
else:
|
|
times = col_vals.astype(np.float64).values
|
|
|
|
players = X[player_col].astype(str).values
|
|
scores = y.values.astype(np.float64)
|
|
|
|
self._global_baseline = float(np.mean(scores)) if len(scores) > 0 else 6.0
|
|
self._global_std = float(np.std(scores, ddof=1)) if len(scores) > 1 else 1.5
|
|
|
|
self._player_params = {}
|
|
unique_players = np.unique(players)
|
|
fitted = 0
|
|
for player in unique_players:
|
|
mask = players == player
|
|
p_times = times[mask]
|
|
p_scores = scores[mask]
|
|
sort_idx = np.argsort(p_times)
|
|
p_times = p_times[sort_idx]
|
|
p_scores = p_scores[sort_idx]
|
|
params = self._fit_player(p_times, p_scores)
|
|
if params is not None:
|
|
self._player_params[str(player)] = params
|
|
fitted += 1
|
|
|
|
logger.info(
|
|
"Hawkes form model fitted: %d/%d players with sufficient history",
|
|
fitted, len(unique_players),
|
|
)
|
|
return self
|
|
|
|
# ------------------------------------------------------------------
|
|
# Predict
|
|
# ------------------------------------------------------------------
|
|
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
|
"""Return form-adjusted projections as additive bonuses to base.
|
|
|
|
Positive = player is in form (HOT), negative = out of form (COLD).
|
|
"""
|
|
if not self._player_params:
|
|
return np.zeros(X.shape[0])
|
|
|
|
player_col = "player" if "player" in X.columns else "name"
|
|
multipliers = self._compute_multipliers(X)
|
|
base = X.get("base_prediction", pd.Series(np.full(X.shape[0], 6.0)))
|
|
base_vals = base.values.astype(np.float64)
|
|
return (multipliers - 1.0) * base_vals
|
|
|
|
def _compute_multipliers(self, X: pd.DataFrame) -> np.ndarray:
|
|
player_col = "player" if "player" in X.columns else "name"
|
|
multipliers = np.ones(X.shape[0], dtype=np.float64)
|
|
players = X[player_col].astype(str).values
|
|
|
|
for i, player in enumerate(players):
|
|
params = self._player_params.get(player)
|
|
if params is None:
|
|
continue
|
|
recent = self._compute_recent_intensity(params)
|
|
base_rate = params.mu
|
|
if base_rate > _EPS:
|
|
ratio = recent / base_rate
|
|
clamped = np.clip(ratio, 0.85, 1.15)
|
|
multipliers[i] = float(clamped)
|
|
|
|
return multipliers
|
|
|
|
def _compute_recent_intensity(self, params: PlayerHawkesParams) -> float:
|
|
return max(params.mu, _EPS)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Momentum projection
|
|
# ------------------------------------------------------------------
|
|
def predict_momentum(
|
|
self,
|
|
X: pd.DataFrame,
|
|
player_history: pd.DataFrame,
|
|
n_future: int = 5,
|
|
) -> np.ndarray:
|
|
"""Project form trajectory for next *n_future* matches.
|
|
|
|
Returns (n_future, n_players) array of momentum multipliers.
|
|
"""
|
|
if not self._player_params:
|
|
return np.ones((n_future, X.shape[0]))
|
|
|
|
player_col = "player" if "player" in X.columns else "name"
|
|
players = X[player_col].astype(str).values
|
|
n_players = X.shape[0]
|
|
trajectory = np.ones((n_future, n_players), dtype=np.float64)
|
|
|
|
history_player_col = None
|
|
for c in ["player", "name"]:
|
|
if c in player_history.columns:
|
|
history_player_col = c
|
|
break
|
|
|
|
history_date_col = None
|
|
for c in ["match_date", "matchday", "date"]:
|
|
if c in player_history.columns:
|
|
history_date_col = c
|
|
break
|
|
|
|
for j, player in enumerate(players):
|
|
params = self._player_params.get(player)
|
|
if params is None:
|
|
continue
|
|
|
|
event_times = []
|
|
if history_player_col and history_date_col:
|
|
p_hist = player_history[player_history[history_player_col].astype(str) == player]
|
|
if len(p_hist) > 0:
|
|
target_col = None
|
|
for c in ["fantavoto", "score", "fv"]:
|
|
if c in p_hist.columns:
|
|
target_col = c
|
|
break
|
|
if target_col and params.baseline_std > 0:
|
|
threshold = params.baseline_mean + params.s * params.baseline_std
|
|
p_sorted = p_hist.sort_values(history_date_col)
|
|
times = p_sorted[history_date_col].values
|
|
scores = p_sorted[target_col].values
|
|
if pd.api.types.is_datetime64_any_dtype(p_sorted[history_date_col]):
|
|
event_times_float = times.astype(np.int64).astype(np.float64) / 1e9 / 86400.0
|
|
else:
|
|
event_times_float = times.astype(np.float64)
|
|
for ti, si in zip(event_times_float, scores):
|
|
if float(si) > threshold:
|
|
event_times.append(ti)
|
|
|
|
if not event_times:
|
|
continue
|
|
|
|
last_t = max(event_times)
|
|
for k in range(1, n_future + 1):
|
|
future_t = last_t + k
|
|
lam = params.mu
|
|
for te in event_times:
|
|
dt = future_t - te
|
|
if dt > 0 and dt <= self.decay_window:
|
|
lam += params.alpha * np.exp(-params.beta * dt)
|
|
trajectory[k - 1, j] = float(np.clip(lam / max(params.mu, _EPS), 0.85, 1.15))
|
|
|
|
return trajectory
|
|
|
|
# ------------------------------------------------------------------
|
|
# Form status
|
|
# ------------------------------------------------------------------
|
|
def get_form_status(self, X: pd.DataFrame) -> List[str]:
|
|
"""Return status string per player: HOT, COLD, or NEUTRAL."""
|
|
player_col = "player" if "player" in X.columns else "name"
|
|
players = X[player_col].astype(str).values
|
|
statuses: List[str] = []
|
|
|
|
for player in players:
|
|
params = self._player_params.get(player)
|
|
if params is None:
|
|
statuses.append(FORM_STATUS_NEUTRAL)
|
|
continue
|
|
intensity = self._compute_recent_intensity(params)
|
|
baseline = max(params.mu, _EPS)
|
|
ratio = intensity / baseline
|
|
if ratio > 1.1 and params.alpha > 0.1:
|
|
statuses.append(FORM_STATUS_HOT)
|
|
elif ratio < 0.9:
|
|
statuses.append(FORM_STATUS_COLD)
|
|
else:
|
|
statuses.append(FORM_STATUS_NEUTRAL)
|
|
|
|
return statuses
|
|
|
|
# ------------------------------------------------------------------
|
|
# Intensity curve
|
|
# ------------------------------------------------------------------
|
|
def compute_intensity_curve(
|
|
self,
|
|
player_name: str,
|
|
history: pd.DataFrame,
|
|
match_dates: np.ndarray,
|
|
future_dates: np.ndarray,
|
|
) -> np.ndarray:
|
|
"""Compute λ(t) over match_dates and future_dates for one player.
|
|
|
|
Returns an array of intensity values at each date.
|
|
"""
|
|
params = self._player_params.get(str(player_name))
|
|
if params is None:
|
|
mu = self._global_baseline
|
|
all_dates = np.concatenate([match_dates, future_dates])
|
|
return np.full_like(all_dates, max(mu, _EPS), dtype=np.float64)
|
|
|
|
target_col = None
|
|
for c in ["fantavoto", "score", "fv"]:
|
|
if c in history.columns:
|
|
target_col = c
|
|
break
|
|
|
|
threshold = params.baseline_mean + params.s * max(params.baseline_std, 0.5)
|
|
event_times = []
|
|
if target_col:
|
|
for _, row in history.iterrows():
|
|
if float(row.get(target_col, 0)) > threshold:
|
|
event_times.append(float(row.name) if isinstance(row.name, (int, float)) else 0.0)
|
|
|
|
if match_dates is not None and len(match_dates) > 0:
|
|
match_vals = match_dates.astype(np.float64)
|
|
for ti in match_vals:
|
|
if ti not in event_times:
|
|
score = None
|
|
for _, row in history.iterrows():
|
|
d_val = float(row.name) if isinstance(row.name, (int, float)) else 0.0
|
|
if abs(d_val - ti) < _EPS:
|
|
score = row.get(target_col, 0) if target_col else 0
|
|
break
|
|
if score is not None and float(score) > threshold:
|
|
event_times.append(ti)
|
|
|
|
all_dates = np.concatenate([
|
|
match_dates.astype(np.float64) if match_dates is not None and len(match_dates) > 0
|
|
else np.array([], dtype=np.float64),
|
|
future_dates.astype(np.float64) if future_dates is not None and len(future_dates) > 0
|
|
else np.array([], dtype=np.float64),
|
|
])
|
|
|
|
if len(all_dates) == 0:
|
|
return np.array([params.mu])
|
|
|
|
intensity = np.full(len(all_dates), params.mu, dtype=np.float64)
|
|
for i, t in enumerate(all_dates):
|
|
lam = params.mu
|
|
for te in event_times:
|
|
dt = t - te
|
|
if dt > 0 and dt <= self.decay_window:
|
|
lam += params.alpha * np.exp(-params.beta * dt)
|
|
elif dt > self.decay_window:
|
|
pass
|
|
intensity[i] = max(lam, _EPS)
|
|
|
|
return intensity
|
|
|
|
# ------------------------------------------------------------------
|
|
# Streak detection
|
|
# ------------------------------------------------------------------
|
|
def detect_streak(
|
|
self,
|
|
player_history: pd.DataFrame,
|
|
) -> Tuple[bool, int, str]:
|
|
"""Detect whether a player is on a hot or cold streak.
|
|
|
|
Returns (is_streak, streak_length, streak_direction).
|
|
streak_direction is "HOT_STREAK" or "COLD_STREAK".
|
|
"""
|
|
if len(player_history) < 3:
|
|
return (False, 0, "NO_STREAK")
|
|
|
|
target_col = None
|
|
for c in ["fantavoto", "score", "fv"]:
|
|
if c in player_history.columns:
|
|
target_col = c
|
|
break
|
|
if target_col is None:
|
|
return (False, 0, "NO_STREAK")
|
|
|
|
date_col = None
|
|
for c in ["match_date", "matchday", "date"]:
|
|
if c in player_history.columns:
|
|
date_col = c
|
|
break
|
|
if date_col:
|
|
sorted_hist = player_history.sort_values(date_col)
|
|
else:
|
|
sorted_hist = player_history
|
|
|
|
scores = sorted_hist[target_col].values.astype(np.float64)
|
|
mean_score = np.mean(scores)
|
|
std_score = max(np.std(scores, ddof=1), 0.5)
|
|
|
|
above = scores[-1] > mean_score + 0.5 * std_score
|
|
below = scores[-1] < mean_score - 0.5 * std_score
|
|
|
|
if not above and not below:
|
|
return (False, 0, "NO_STREAK")
|
|
|
|
direction = "HOT_STREAK" if above else "COLD_STREAK"
|
|
streak_len = 1
|
|
for j in range(len(scores) - 2, -1, -1):
|
|
if direction == "HOT_STREAK" and scores[j] > mean_score + 0.5 * std_score:
|
|
streak_len += 1
|
|
elif direction == "COLD_STREAK" and scores[j] < mean_score - 0.5 * std_score:
|
|
streak_len += 1
|
|
else:
|
|
break
|
|
|
|
return (streak_len >= 3, streak_len, direction)
|