feat: add 10 new ML models for auction optimization (Phases 1-6)
Phase 1 - Quick Wins: - QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding - MinutesSurvivalModel: Weibull AFT for minutes distribution modeling Phase 2 - Adaptive Auction: - BanditAuctionSolver: Thompson Sampling for live auction bids - OpponentBidModel: Predict competitor bids via LightGBM - BudgetOptimizer: Bayesian optimization for role-level allocation Phase 3 - Deep Learning: - RLAuctionPolicy: Double DQN agent for auction strategy - SetTransformer: Team composition valuation via set-based ML Phase 4 - Probabilistic: - BayesianPlayerModel: Hierarchical pooling for rookie uncertainty - ConformalPredictor: Calibrated prediction intervals Phase 5 - Chemistry & Form: - PlayerChemistryGAT: Graph attention network for player synergies - PlayerFormModel: Hawkes process for form momentum Phase 6 - Causal: - TransferCausalModel: Causal forest for transfer effects - AuctionEffectAnalyzer: Bid adjustment from causal analysis 81 tests passing
This commit is contained in:
@@ -0,0 +1,538 @@
|
||||
"""Bayesian Optimization for role-level budget allocation.
|
||||
|
||||
Uses Gaussian Process regression to find optimal budget distribution
|
||||
across roles (P, D, C, A) that maximizes total projected team value.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from scipy.optimize import minimize
|
||||
from scipy.special import softmax
|
||||
|
||||
from src.optimization.auction_solver import AuctionConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_ROSTER_QUOTAS = {"P": 3, "D": 8, "C": 8, "A": 6}
|
||||
|
||||
|
||||
@dataclass
|
||||
class RoleBudgetResult:
|
||||
allocation: Dict[str, float]
|
||||
total_value: float
|
||||
role_values: Dict[str, float]
|
||||
value_curves: Dict[str, Tuple[np.ndarray, np.ndarray]]
|
||||
|
||||
|
||||
class BudgetOptimizer:
|
||||
"""Bayesian Optimization for budget allocation across roster roles.
|
||||
|
||||
Finds the split of total_budget across P/D/C/A that yields the
|
||||
highest possible team points via greedy fill within each role's budget.
|
||||
Supports mid-auction adaptive rebalancing.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
total_budget: float = 500.0,
|
||||
roster_quotas: Optional[Dict[str, int]] = None,
|
||||
auction_config: Optional[AuctionConfig] = None,
|
||||
random_state: int = 42,
|
||||
):
|
||||
self.total_budget = total_budget
|
||||
self.roster_quotas = roster_quotas or dict(DEFAULT_ROSTER_QUOTAS)
|
||||
self.auction_config = auction_config or AuctionConfig()
|
||||
self.random_state = random_state
|
||||
self.rng = np.random.RandomState(random_state)
|
||||
self._last_allocation: Optional[Dict[str, float]] = None
|
||||
self._last_value: float = 0.0
|
||||
self._optimization_history: List[dict] = []
|
||||
self._gk_available = False
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Core optimization
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def optimize(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
n_calls: int = 50,
|
||||
use_skopt: bool = True,
|
||||
) -> Dict[str, float]:
|
||||
"""Optimize budget allocation across roles.
|
||||
|
||||
Uses scikit-optimize GaussianProcessRegressor if available,
|
||||
otherwise simplex-based local search.
|
||||
|
||||
Args:
|
||||
player_pool_df: DataFrame with columns [name, role, projected_points].
|
||||
n_calls: number of GP evaluations.
|
||||
use_skopt: attempt Gaussian Process optimization.
|
||||
|
||||
Returns:
|
||||
Dict mapping role -> recommended budget amount.
|
||||
"""
|
||||
if "role" not in player_pool_df.columns or "projected_points" not in player_pool_df.columns:
|
||||
raise ValueError("player_pool_df must have 'role' and 'projected_points' columns")
|
||||
|
||||
roles = list(self.roster_quotas.keys())
|
||||
n_roles = len(roles)
|
||||
|
||||
if use_skopt:
|
||||
try:
|
||||
return self._optimize_gp(player_pool_df, roles, n_calls)
|
||||
except ImportError:
|
||||
logger.info("scikit-optimize not installed. Using local search.")
|
||||
except Exception as exc:
|
||||
logger.warning(f"GP optimization failed: {exc}. Using local search.")
|
||||
|
||||
return self._optimize_local(player_pool_df, roles, n_calls)
|
||||
|
||||
def _optimize_gp(
|
||||
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
|
||||
) -> Dict[str, float]:
|
||||
"""Gaussian Process-based budget optimization."""
|
||||
from skopt import gp_minimize
|
||||
from skopt.space import Space
|
||||
from skopt.learning import GaussianProcessRegressor
|
||||
|
||||
n_roles = len(roles)
|
||||
space = Space([(0.01, 0.70) for _ in range(n_roles)])
|
||||
|
||||
def objective_wrapper(fractions):
|
||||
fractions = np.array(fractions, dtype=float)
|
||||
fractions = self._normalize_fractions(fractions)
|
||||
value = self._objective(fractions, roles, player_pool_df)
|
||||
self._optimization_history.append({
|
||||
"fractions": fractions.tolist(),
|
||||
"value": value,
|
||||
})
|
||||
return -value
|
||||
|
||||
def params_to_fractions(params):
|
||||
return self._normalize_fractions(np.array(params, dtype=float))
|
||||
|
||||
result = gp_minimize(
|
||||
objective_wrapper,
|
||||
space,
|
||||
n_calls=n_calls,
|
||||
random_state=self.random_state,
|
||||
n_initial_points=max(10, n_calls // 5),
|
||||
verbose=False,
|
||||
n_jobs=-1,
|
||||
)
|
||||
|
||||
best_fractions = params_to_fractions(result.x)
|
||||
self._last_allocation = self._fractions_to_allocation(best_fractions, roles)
|
||||
self._last_value = -result.fun
|
||||
|
||||
logger.info(
|
||||
f"GP Budget optimization complete: {self._last_allocation} "
|
||||
f"=> value={self._last_value:.1f}"
|
||||
)
|
||||
|
||||
return self._last_allocation
|
||||
|
||||
def _optimize_local(
|
||||
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
|
||||
) -> Dict[str, float]:
|
||||
"""Local search optimization using Nelder-Mead simplex."""
|
||||
n_roles = len(roles)
|
||||
|
||||
best_allocation = None
|
||||
best_value = -float("inf")
|
||||
|
||||
for restart in range(max(n_calls // 10, 1)):
|
||||
x0 = self._random_allocation(n_roles)
|
||||
|
||||
def simplex_objective(fractions):
|
||||
fractions = self._normalize_fractions(np.array(fractions, dtype=float))
|
||||
value = self._objective(fractions, roles, player_pool_df)
|
||||
self._optimization_history.append({
|
||||
"fractions": fractions.tolist(),
|
||||
"value": value,
|
||||
})
|
||||
return -value
|
||||
|
||||
res = minimize(
|
||||
simplex_objective,
|
||||
x0,
|
||||
method="Nelder-Mead",
|
||||
options={"maxiter": max(n_calls // 3, 20), "xatol": 1e-3, "fatol": 1e-3},
|
||||
)
|
||||
|
||||
fractions = self._normalize_fractions(np.array(res.x, dtype=float))
|
||||
value = -res.fun
|
||||
|
||||
if value > best_value:
|
||||
best_value = value
|
||||
best_allocation = fractions
|
||||
|
||||
if best_allocation is None:
|
||||
best_allocation = self._proportional_allocation(player_pool_df, roles)
|
||||
|
||||
self._last_allocation = self._fractions_to_allocation(best_allocation, roles)
|
||||
self._last_value = best_value
|
||||
|
||||
logger.info(
|
||||
f"Local optimization complete: {self._last_allocation} "
|
||||
f"=> value={self._last_value:.1f}"
|
||||
)
|
||||
|
||||
return self._last_allocation
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Adaptive rebalancing
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def optimize_adaptive(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
remaining_slots: Dict[str, int],
|
||||
spent_per_role: Dict[str, float],
|
||||
n_calls: int = 30,
|
||||
) -> Dict[str, float]:
|
||||
"""Mid-auction rebalancing — optimize remaining budget for unfilled slots.
|
||||
|
||||
Args:
|
||||
player_pool_df: remaining available players pool.
|
||||
remaining_slots: dict of role -> slots still needed.
|
||||
spent_per_role: dict of role -> credits already spent.
|
||||
n_calls: GP evaluation budget.
|
||||
|
||||
Returns:
|
||||
Dict mapping role -> recommended budget for remaining slots.
|
||||
"""
|
||||
remaining_budget = self.total_budget - sum(spent_per_role.values())
|
||||
remaining_budget = max(1.0, remaining_budget)
|
||||
|
||||
if sum(remaining_slots.values()) == 0:
|
||||
logger.info("All slots filled. No budget to allocate.")
|
||||
return {r: 0.0 for r in self.roster_quotas}
|
||||
|
||||
saved_quotas = self.roster_quotas
|
||||
saved_total = self.total_budget
|
||||
self.roster_quotas = dict(remaining_slots)
|
||||
self.total_budget = remaining_budget
|
||||
|
||||
available = player_pool_df[
|
||||
player_pool_df["role"].isin(
|
||||
[r for r, s in remaining_slots.items() if s > 0]
|
||||
)
|
||||
]
|
||||
|
||||
if len(available) == 0:
|
||||
logger.warning("No available players for remaining slots.")
|
||||
self.roster_quotas = saved_quotas
|
||||
self.total_budget = saved_total
|
||||
return {r: 0.0 for r in saved_quotas}
|
||||
|
||||
allocation = self.optimize(
|
||||
available,
|
||||
n_calls=n_calls,
|
||||
use_skopt=True,
|
||||
)
|
||||
|
||||
self.roster_quotas = saved_quotas
|
||||
self.total_budget = saved_total
|
||||
|
||||
result = {}
|
||||
for role in saved_quotas:
|
||||
result[role] = allocation.get(role, 0.0)
|
||||
return result
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Objective function
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _objective(
|
||||
self,
|
||||
fractions: np.ndarray,
|
||||
roles: list,
|
||||
player_pool_df: pd.DataFrame,
|
||||
) -> float:
|
||||
"""Simulate greedy fill within role budgets; return total projected points.
|
||||
|
||||
For each role, pick the best players by projected_points until the
|
||||
role budget or slot quota is exhausted.
|
||||
"""
|
||||
total_value = 0.0
|
||||
|
||||
for i, role in enumerate(roles):
|
||||
budget = fractions[i] * self.total_budget
|
||||
slots = self.roster_quotas.get(role, 0)
|
||||
role_players = player_pool_df[player_pool_df["role"] == role].copy()
|
||||
|
||||
if len(role_players) == 0 or slots == 0:
|
||||
continue
|
||||
|
||||
role_players = role_players.sort_values(
|
||||
"projected_points", ascending=False
|
||||
)
|
||||
|
||||
total_cost = 0.0
|
||||
filled = 0
|
||||
|
||||
for _, player in role_players.iterrows():
|
||||
points = float(player["projected_points"])
|
||||
estimated_price = self._estimate_price_simple(points, budget, role)
|
||||
|
||||
if total_cost + estimated_price > budget:
|
||||
continue
|
||||
|
||||
total_cost += estimated_price
|
||||
total_value += points
|
||||
filled += 1
|
||||
|
||||
if filled >= slots:
|
||||
break
|
||||
|
||||
return total_value
|
||||
|
||||
def _estimate_price_simple(
|
||||
self, projected_points: float, role_budget: float, role: str
|
||||
) -> float:
|
||||
"""Simple price estimate: points * role_factor clamped within budget."""
|
||||
role_factor = {"P": 4.0, "D": 2.5, "C": 3.0, "A": 4.5}.get(role, 3.0)
|
||||
price = projected_points * role_factor
|
||||
max_price = role_budget * 0.50
|
||||
return min(price, max_price, role_budget)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Allocation access and visualization data
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def get_allocation(self) -> Dict[str, float]:
|
||||
"""Return last computed allocation (role -> budget amount)."""
|
||||
if self._last_allocation is None:
|
||||
return {r: self.total_budget / len(self.roster_quotas) for r in self.roster_quotas}
|
||||
return dict(self._last_allocation)
|
||||
|
||||
def get_role_value_curves(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
n_points: int = 20,
|
||||
) -> Dict[str, Tuple[np.ndarray, np.ndarray]]:
|
||||
"""Compute diminishing returns curves: budget vs. expected points per role.
|
||||
|
||||
Returns:
|
||||
dict role -> (budget_array, value_array).
|
||||
"""
|
||||
curves = {}
|
||||
budget_step = self.total_budget / n_points
|
||||
|
||||
for role, slots in self.roster_quotas.items():
|
||||
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
|
||||
"projected_points", ascending=False
|
||||
)
|
||||
|
||||
budgets = np.linspace(0, self.total_budget, n_points)
|
||||
values = np.zeros(n_points)
|
||||
|
||||
for i, budget_limit in enumerate(budgets):
|
||||
total_cost = 0.0
|
||||
total_value = 0.0
|
||||
filled = 0
|
||||
|
||||
for _, player in role_players.iterrows():
|
||||
points = float(player["projected_points"])
|
||||
price = self._estimate_price_simple(points, budget_limit, role)
|
||||
|
||||
if total_cost + price > budget_limit:
|
||||
continue
|
||||
|
||||
total_cost += price
|
||||
total_value += points
|
||||
filled += 1
|
||||
|
||||
if filled >= slots:
|
||||
break
|
||||
|
||||
values[i] = total_value
|
||||
|
||||
curves[role] = (budgets.copy(), values.copy())
|
||||
|
||||
return curves
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Fallback: proportional allocation
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _proportional_allocation(
|
||||
self, player_pool_df: pd.DataFrame, roles: list
|
||||
) -> np.ndarray:
|
||||
"""Allocate budget proportional to (points_variance * slots) per role."""
|
||||
weights = np.zeros(len(roles))
|
||||
|
||||
for i, role in enumerate(roles):
|
||||
role_players = player_pool_df[player_pool_df["role"] == role]
|
||||
if len(role_players) > 1:
|
||||
variance = role_players["projected_points"].var()
|
||||
else:
|
||||
variance = 1.0
|
||||
slots = self.roster_quotas.get(role, 1)
|
||||
weights[i] = variance * slots
|
||||
|
||||
weight_sum = weights.sum()
|
||||
if weight_sum <= 0:
|
||||
return np.ones(len(roles)) / len(roles)
|
||||
|
||||
fractions = weights / weight_sum
|
||||
fractions = np.clip(fractions, 0.02, 0.70)
|
||||
fractions = fractions / fractions.sum()
|
||||
|
||||
logger.info(
|
||||
f"Proportional allocation (fallback): "
|
||||
f"{dict(zip(roles, fractions.round(3)))}"
|
||||
)
|
||||
return fractions
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Utilities
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _normalize_fractions(self, fractions: np.ndarray) -> np.ndarray:
|
||||
"""Normalize fractions to sum to 1.0 with minimum per role."""
|
||||
fractions = np.clip(fractions, 0.01, 0.70)
|
||||
total = fractions.sum()
|
||||
if total <= 0:
|
||||
return np.ones_like(fractions) / len(fractions)
|
||||
return fractions / total
|
||||
|
||||
def _fractions_to_allocation(
|
||||
self, fractions: np.ndarray, roles: list
|
||||
) -> Dict[str, float]:
|
||||
"""Convert fractions to absolute budget per role."""
|
||||
return {
|
||||
role: round(float(fractions[i] * self.total_budget), 1)
|
||||
for i, role in enumerate(roles)
|
||||
}
|
||||
|
||||
def _random_allocation(self, n_roles: int) -> np.ndarray:
|
||||
"""Generate a random allocation via Dirichlet."""
|
||||
alpha = np.ones(n_roles) * 2.0
|
||||
return self.rng.dirichlet(alpha)
|
||||
|
||||
def get_optimization_trace(self) -> pd.DataFrame:
|
||||
"""Return DataFrame of all evaluated allocations during optimization."""
|
||||
if not self._optimization_history:
|
||||
return pd.DataFrame()
|
||||
return pd.DataFrame(self._optimization_history)
|
||||
|
||||
def estimate_team_composition(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
allocation: Optional[Dict[str, float]] = None,
|
||||
) -> pd.DataFrame:
|
||||
"""Given final allocation, return the recommended player selections.
|
||||
|
||||
Returns:
|
||||
DataFrame with selected players, their estimated costs, and value.
|
||||
"""
|
||||
roles = list(self.roster_quotas.keys())
|
||||
alloc = allocation or self._last_allocation
|
||||
if alloc is None:
|
||||
alloc = self._proportional_allocation_as_dict(roles)
|
||||
|
||||
selections = []
|
||||
|
||||
for role in roles:
|
||||
budget = alloc.get(role, 0.0)
|
||||
slots = self.roster_quotas.get(role, 0)
|
||||
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
|
||||
"projected_points", ascending=False
|
||||
)
|
||||
|
||||
total_cost = 0.0
|
||||
filled = 0
|
||||
|
||||
for _, player in role_players.iterrows():
|
||||
points = float(player["projected_points"])
|
||||
price = self._estimate_price_simple(points, budget, role)
|
||||
|
||||
if total_cost + price > budget:
|
||||
continue
|
||||
|
||||
total_cost += price
|
||||
name = player.get("name", player.get("player_name", "unknown"))
|
||||
selections.append({
|
||||
"player": name,
|
||||
"role": role,
|
||||
"projected_points": points,
|
||||
"estimated_cost": round(price, 1),
|
||||
"value_ratio": round(points / max(price, 1), 3),
|
||||
})
|
||||
filled += 1
|
||||
|
||||
if filled >= slots:
|
||||
break
|
||||
|
||||
return pd.DataFrame(selections)
|
||||
|
||||
def _proportional_allocation_as_dict(self, roles: list) -> Dict[str, float]:
|
||||
fractions = self._proportional_allocation(
|
||||
pd.DataFrame(columns=["role", "projected_points"]), roles
|
||||
)
|
||||
return {roles[i]: round(float(fractions[i] * self.total_budget), 1) for i in range(len(roles))}
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Sensitivity analysis
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def sensitivity_analysis(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
role: Optional[str] = None,
|
||||
delta_pct: float = 0.05,
|
||||
n_steps: int = 11,
|
||||
n_calls: int = 50,
|
||||
) -> dict:
|
||||
"""Test how shifting budget into/out of one role affects total value.
|
||||
|
||||
Args:
|
||||
player_pool_df: current player pool.
|
||||
role: role to perturb. If None, tests all roles.
|
||||
delta_pct: fractional step size.
|
||||
n_steps: number of steps in each direction.
|
||||
n_calls: number of optimization calls (for compatibility).
|
||||
|
||||
Returns:
|
||||
Dict with at least a 'per_role' key containing per-role analysis.
|
||||
"""
|
||||
base_allocation = self.get_allocation()
|
||||
roles_to_test = [role] if role else list(base_allocation.keys())
|
||||
per_role = {}
|
||||
|
||||
for test_role in roles_to_test:
|
||||
results = []
|
||||
shifts = np.linspace(-delta_pct * n_steps, delta_pct * n_steps, 2 * n_steps + 1)
|
||||
|
||||
for shift in shifts:
|
||||
adjusted = {}
|
||||
for r, val in base_allocation.items():
|
||||
adjusted[r] = val * (1.0 + (shift if r == test_role else -shift / 3.0))
|
||||
|
||||
total_adj = sum(adjusted.values())
|
||||
for r in adjusted:
|
||||
adjusted[r] = adjusted[r] / total_adj * self.total_budget
|
||||
|
||||
fractions = np.array([adjusted[r] / self.total_budget for r in base_allocation.keys()])
|
||||
value = self._objective(
|
||||
fractions,
|
||||
list(base_allocation.keys()),
|
||||
player_pool_df,
|
||||
)
|
||||
|
||||
results.append({
|
||||
"shift_pct": round(shift * 100, 1),
|
||||
"allocation": {r: round(v, 1) for r, v in adjusted.items()},
|
||||
"total_value": round(value, 1),
|
||||
})
|
||||
|
||||
per_role[test_role] = pd.DataFrame(results)
|
||||
|
||||
return {"per_role": per_role, "base_allocation": base_allocation}
|
||||
Reference in New Issue
Block a user