b0fab62a87
Phase 1 - Quick Wins: - QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding - MinutesSurvivalModel: Weibull AFT for minutes distribution modeling Phase 2 - Adaptive Auction: - BanditAuctionSolver: Thompson Sampling for live auction bids - OpponentBidModel: Predict competitor bids via LightGBM - BudgetOptimizer: Bayesian optimization for role-level allocation Phase 3 - Deep Learning: - RLAuctionPolicy: Double DQN agent for auction strategy - SetTransformer: Team composition valuation via set-based ML Phase 4 - Probabilistic: - BayesianPlayerModel: Hierarchical pooling for rookie uncertainty - ConformalPredictor: Calibrated prediction intervals Phase 5 - Chemistry & Form: - PlayerChemistryGAT: Graph attention network for player synergies - PlayerFormModel: Hawkes process for form momentum Phase 6 - Causal: - TransferCausalModel: Causal forest for transfer effects - AuctionEffectAnalyzer: Bid adjustment from causal analysis 81 tests passing
539 lines
19 KiB
Python
539 lines
19 KiB
Python
"""Bayesian Optimization for role-level budget allocation.
|
|
|
|
Uses Gaussian Process regression to find optimal budget distribution
|
|
across roles (P, D, C, A) that maximizes total projected team value.
|
|
"""
|
|
|
|
import logging
|
|
from dataclasses import dataclass, field
|
|
from typing import Dict, List, Optional, Tuple
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
from scipy.optimize import minimize
|
|
from scipy.special import softmax
|
|
|
|
from src.optimization.auction_solver import AuctionConfig
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
DEFAULT_ROSTER_QUOTAS = {"P": 3, "D": 8, "C": 8, "A": 6}
|
|
|
|
|
|
@dataclass
|
|
class RoleBudgetResult:
|
|
allocation: Dict[str, float]
|
|
total_value: float
|
|
role_values: Dict[str, float]
|
|
value_curves: Dict[str, Tuple[np.ndarray, np.ndarray]]
|
|
|
|
|
|
class BudgetOptimizer:
|
|
"""Bayesian Optimization for budget allocation across roster roles.
|
|
|
|
Finds the split of total_budget across P/D/C/A that yields the
|
|
highest possible team points via greedy fill within each role's budget.
|
|
Supports mid-auction adaptive rebalancing.
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
total_budget: float = 500.0,
|
|
roster_quotas: Optional[Dict[str, int]] = None,
|
|
auction_config: Optional[AuctionConfig] = None,
|
|
random_state: int = 42,
|
|
):
|
|
self.total_budget = total_budget
|
|
self.roster_quotas = roster_quotas or dict(DEFAULT_ROSTER_QUOTAS)
|
|
self.auction_config = auction_config or AuctionConfig()
|
|
self.random_state = random_state
|
|
self.rng = np.random.RandomState(random_state)
|
|
self._last_allocation: Optional[Dict[str, float]] = None
|
|
self._last_value: float = 0.0
|
|
self._optimization_history: List[dict] = []
|
|
self._gk_available = False
|
|
|
|
# ------------------------------------------------------------------
|
|
# Core optimization
|
|
# ------------------------------------------------------------------
|
|
|
|
def optimize(
|
|
self,
|
|
player_pool_df: pd.DataFrame,
|
|
n_calls: int = 50,
|
|
use_skopt: bool = True,
|
|
) -> Dict[str, float]:
|
|
"""Optimize budget allocation across roles.
|
|
|
|
Uses scikit-optimize GaussianProcessRegressor if available,
|
|
otherwise simplex-based local search.
|
|
|
|
Args:
|
|
player_pool_df: DataFrame with columns [name, role, projected_points].
|
|
n_calls: number of GP evaluations.
|
|
use_skopt: attempt Gaussian Process optimization.
|
|
|
|
Returns:
|
|
Dict mapping role -> recommended budget amount.
|
|
"""
|
|
if "role" not in player_pool_df.columns or "projected_points" not in player_pool_df.columns:
|
|
raise ValueError("player_pool_df must have 'role' and 'projected_points' columns")
|
|
|
|
roles = list(self.roster_quotas.keys())
|
|
n_roles = len(roles)
|
|
|
|
if use_skopt:
|
|
try:
|
|
return self._optimize_gp(player_pool_df, roles, n_calls)
|
|
except ImportError:
|
|
logger.info("scikit-optimize not installed. Using local search.")
|
|
except Exception as exc:
|
|
logger.warning(f"GP optimization failed: {exc}. Using local search.")
|
|
|
|
return self._optimize_local(player_pool_df, roles, n_calls)
|
|
|
|
def _optimize_gp(
|
|
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
|
|
) -> Dict[str, float]:
|
|
"""Gaussian Process-based budget optimization."""
|
|
from skopt import gp_minimize
|
|
from skopt.space import Space
|
|
from skopt.learning import GaussianProcessRegressor
|
|
|
|
n_roles = len(roles)
|
|
space = Space([(0.01, 0.70) for _ in range(n_roles)])
|
|
|
|
def objective_wrapper(fractions):
|
|
fractions = np.array(fractions, dtype=float)
|
|
fractions = self._normalize_fractions(fractions)
|
|
value = self._objective(fractions, roles, player_pool_df)
|
|
self._optimization_history.append({
|
|
"fractions": fractions.tolist(),
|
|
"value": value,
|
|
})
|
|
return -value
|
|
|
|
def params_to_fractions(params):
|
|
return self._normalize_fractions(np.array(params, dtype=float))
|
|
|
|
result = gp_minimize(
|
|
objective_wrapper,
|
|
space,
|
|
n_calls=n_calls,
|
|
random_state=self.random_state,
|
|
n_initial_points=max(10, n_calls // 5),
|
|
verbose=False,
|
|
n_jobs=-1,
|
|
)
|
|
|
|
best_fractions = params_to_fractions(result.x)
|
|
self._last_allocation = self._fractions_to_allocation(best_fractions, roles)
|
|
self._last_value = -result.fun
|
|
|
|
logger.info(
|
|
f"GP Budget optimization complete: {self._last_allocation} "
|
|
f"=> value={self._last_value:.1f}"
|
|
)
|
|
|
|
return self._last_allocation
|
|
|
|
def _optimize_local(
|
|
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
|
|
) -> Dict[str, float]:
|
|
"""Local search optimization using Nelder-Mead simplex."""
|
|
n_roles = len(roles)
|
|
|
|
best_allocation = None
|
|
best_value = -float("inf")
|
|
|
|
for restart in range(max(n_calls // 10, 1)):
|
|
x0 = self._random_allocation(n_roles)
|
|
|
|
def simplex_objective(fractions):
|
|
fractions = self._normalize_fractions(np.array(fractions, dtype=float))
|
|
value = self._objective(fractions, roles, player_pool_df)
|
|
self._optimization_history.append({
|
|
"fractions": fractions.tolist(),
|
|
"value": value,
|
|
})
|
|
return -value
|
|
|
|
res = minimize(
|
|
simplex_objective,
|
|
x0,
|
|
method="Nelder-Mead",
|
|
options={"maxiter": max(n_calls // 3, 20), "xatol": 1e-3, "fatol": 1e-3},
|
|
)
|
|
|
|
fractions = self._normalize_fractions(np.array(res.x, dtype=float))
|
|
value = -res.fun
|
|
|
|
if value > best_value:
|
|
best_value = value
|
|
best_allocation = fractions
|
|
|
|
if best_allocation is None:
|
|
best_allocation = self._proportional_allocation(player_pool_df, roles)
|
|
|
|
self._last_allocation = self._fractions_to_allocation(best_allocation, roles)
|
|
self._last_value = best_value
|
|
|
|
logger.info(
|
|
f"Local optimization complete: {self._last_allocation} "
|
|
f"=> value={self._last_value:.1f}"
|
|
)
|
|
|
|
return self._last_allocation
|
|
|
|
# ------------------------------------------------------------------
|
|
# Adaptive rebalancing
|
|
# ------------------------------------------------------------------
|
|
|
|
def optimize_adaptive(
|
|
self,
|
|
player_pool_df: pd.DataFrame,
|
|
remaining_slots: Dict[str, int],
|
|
spent_per_role: Dict[str, float],
|
|
n_calls: int = 30,
|
|
) -> Dict[str, float]:
|
|
"""Mid-auction rebalancing — optimize remaining budget for unfilled slots.
|
|
|
|
Args:
|
|
player_pool_df: remaining available players pool.
|
|
remaining_slots: dict of role -> slots still needed.
|
|
spent_per_role: dict of role -> credits already spent.
|
|
n_calls: GP evaluation budget.
|
|
|
|
Returns:
|
|
Dict mapping role -> recommended budget for remaining slots.
|
|
"""
|
|
remaining_budget = self.total_budget - sum(spent_per_role.values())
|
|
remaining_budget = max(1.0, remaining_budget)
|
|
|
|
if sum(remaining_slots.values()) == 0:
|
|
logger.info("All slots filled. No budget to allocate.")
|
|
return {r: 0.0 for r in self.roster_quotas}
|
|
|
|
saved_quotas = self.roster_quotas
|
|
saved_total = self.total_budget
|
|
self.roster_quotas = dict(remaining_slots)
|
|
self.total_budget = remaining_budget
|
|
|
|
available = player_pool_df[
|
|
player_pool_df["role"].isin(
|
|
[r for r, s in remaining_slots.items() if s > 0]
|
|
)
|
|
]
|
|
|
|
if len(available) == 0:
|
|
logger.warning("No available players for remaining slots.")
|
|
self.roster_quotas = saved_quotas
|
|
self.total_budget = saved_total
|
|
return {r: 0.0 for r in saved_quotas}
|
|
|
|
allocation = self.optimize(
|
|
available,
|
|
n_calls=n_calls,
|
|
use_skopt=True,
|
|
)
|
|
|
|
self.roster_quotas = saved_quotas
|
|
self.total_budget = saved_total
|
|
|
|
result = {}
|
|
for role in saved_quotas:
|
|
result[role] = allocation.get(role, 0.0)
|
|
return result
|
|
|
|
# ------------------------------------------------------------------
|
|
# Objective function
|
|
# ------------------------------------------------------------------
|
|
|
|
def _objective(
|
|
self,
|
|
fractions: np.ndarray,
|
|
roles: list,
|
|
player_pool_df: pd.DataFrame,
|
|
) -> float:
|
|
"""Simulate greedy fill within role budgets; return total projected points.
|
|
|
|
For each role, pick the best players by projected_points until the
|
|
role budget or slot quota is exhausted.
|
|
"""
|
|
total_value = 0.0
|
|
|
|
for i, role in enumerate(roles):
|
|
budget = fractions[i] * self.total_budget
|
|
slots = self.roster_quotas.get(role, 0)
|
|
role_players = player_pool_df[player_pool_df["role"] == role].copy()
|
|
|
|
if len(role_players) == 0 or slots == 0:
|
|
continue
|
|
|
|
role_players = role_players.sort_values(
|
|
"projected_points", ascending=False
|
|
)
|
|
|
|
total_cost = 0.0
|
|
filled = 0
|
|
|
|
for _, player in role_players.iterrows():
|
|
points = float(player["projected_points"])
|
|
estimated_price = self._estimate_price_simple(points, budget, role)
|
|
|
|
if total_cost + estimated_price > budget:
|
|
continue
|
|
|
|
total_cost += estimated_price
|
|
total_value += points
|
|
filled += 1
|
|
|
|
if filled >= slots:
|
|
break
|
|
|
|
return total_value
|
|
|
|
def _estimate_price_simple(
|
|
self, projected_points: float, role_budget: float, role: str
|
|
) -> float:
|
|
"""Simple price estimate: points * role_factor clamped within budget."""
|
|
role_factor = {"P": 4.0, "D": 2.5, "C": 3.0, "A": 4.5}.get(role, 3.0)
|
|
price = projected_points * role_factor
|
|
max_price = role_budget * 0.50
|
|
return min(price, max_price, role_budget)
|
|
|
|
# ------------------------------------------------------------------
|
|
# Allocation access and visualization data
|
|
# ------------------------------------------------------------------
|
|
|
|
def get_allocation(self) -> Dict[str, float]:
|
|
"""Return last computed allocation (role -> budget amount)."""
|
|
if self._last_allocation is None:
|
|
return {r: self.total_budget / len(self.roster_quotas) for r in self.roster_quotas}
|
|
return dict(self._last_allocation)
|
|
|
|
def get_role_value_curves(
|
|
self,
|
|
player_pool_df: pd.DataFrame,
|
|
n_points: int = 20,
|
|
) -> Dict[str, Tuple[np.ndarray, np.ndarray]]:
|
|
"""Compute diminishing returns curves: budget vs. expected points per role.
|
|
|
|
Returns:
|
|
dict role -> (budget_array, value_array).
|
|
"""
|
|
curves = {}
|
|
budget_step = self.total_budget / n_points
|
|
|
|
for role, slots in self.roster_quotas.items():
|
|
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
|
|
"projected_points", ascending=False
|
|
)
|
|
|
|
budgets = np.linspace(0, self.total_budget, n_points)
|
|
values = np.zeros(n_points)
|
|
|
|
for i, budget_limit in enumerate(budgets):
|
|
total_cost = 0.0
|
|
total_value = 0.0
|
|
filled = 0
|
|
|
|
for _, player in role_players.iterrows():
|
|
points = float(player["projected_points"])
|
|
price = self._estimate_price_simple(points, budget_limit, role)
|
|
|
|
if total_cost + price > budget_limit:
|
|
continue
|
|
|
|
total_cost += price
|
|
total_value += points
|
|
filled += 1
|
|
|
|
if filled >= slots:
|
|
break
|
|
|
|
values[i] = total_value
|
|
|
|
curves[role] = (budgets.copy(), values.copy())
|
|
|
|
return curves
|
|
|
|
# ------------------------------------------------------------------
|
|
# Fallback: proportional allocation
|
|
# ------------------------------------------------------------------
|
|
|
|
def _proportional_allocation(
|
|
self, player_pool_df: pd.DataFrame, roles: list
|
|
) -> np.ndarray:
|
|
"""Allocate budget proportional to (points_variance * slots) per role."""
|
|
weights = np.zeros(len(roles))
|
|
|
|
for i, role in enumerate(roles):
|
|
role_players = player_pool_df[player_pool_df["role"] == role]
|
|
if len(role_players) > 1:
|
|
variance = role_players["projected_points"].var()
|
|
else:
|
|
variance = 1.0
|
|
slots = self.roster_quotas.get(role, 1)
|
|
weights[i] = variance * slots
|
|
|
|
weight_sum = weights.sum()
|
|
if weight_sum <= 0:
|
|
return np.ones(len(roles)) / len(roles)
|
|
|
|
fractions = weights / weight_sum
|
|
fractions = np.clip(fractions, 0.02, 0.70)
|
|
fractions = fractions / fractions.sum()
|
|
|
|
logger.info(
|
|
f"Proportional allocation (fallback): "
|
|
f"{dict(zip(roles, fractions.round(3)))}"
|
|
)
|
|
return fractions
|
|
|
|
# ------------------------------------------------------------------
|
|
# Utilities
|
|
# ------------------------------------------------------------------
|
|
|
|
def _normalize_fractions(self, fractions: np.ndarray) -> np.ndarray:
|
|
"""Normalize fractions to sum to 1.0 with minimum per role."""
|
|
fractions = np.clip(fractions, 0.01, 0.70)
|
|
total = fractions.sum()
|
|
if total <= 0:
|
|
return np.ones_like(fractions) / len(fractions)
|
|
return fractions / total
|
|
|
|
def _fractions_to_allocation(
|
|
self, fractions: np.ndarray, roles: list
|
|
) -> Dict[str, float]:
|
|
"""Convert fractions to absolute budget per role."""
|
|
return {
|
|
role: round(float(fractions[i] * self.total_budget), 1)
|
|
for i, role in enumerate(roles)
|
|
}
|
|
|
|
def _random_allocation(self, n_roles: int) -> np.ndarray:
|
|
"""Generate a random allocation via Dirichlet."""
|
|
alpha = np.ones(n_roles) * 2.0
|
|
return self.rng.dirichlet(alpha)
|
|
|
|
def get_optimization_trace(self) -> pd.DataFrame:
|
|
"""Return DataFrame of all evaluated allocations during optimization."""
|
|
if not self._optimization_history:
|
|
return pd.DataFrame()
|
|
return pd.DataFrame(self._optimization_history)
|
|
|
|
def estimate_team_composition(
|
|
self,
|
|
player_pool_df: pd.DataFrame,
|
|
allocation: Optional[Dict[str, float]] = None,
|
|
) -> pd.DataFrame:
|
|
"""Given final allocation, return the recommended player selections.
|
|
|
|
Returns:
|
|
DataFrame with selected players, their estimated costs, and value.
|
|
"""
|
|
roles = list(self.roster_quotas.keys())
|
|
alloc = allocation or self._last_allocation
|
|
if alloc is None:
|
|
alloc = self._proportional_allocation_as_dict(roles)
|
|
|
|
selections = []
|
|
|
|
for role in roles:
|
|
budget = alloc.get(role, 0.0)
|
|
slots = self.roster_quotas.get(role, 0)
|
|
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
|
|
"projected_points", ascending=False
|
|
)
|
|
|
|
total_cost = 0.0
|
|
filled = 0
|
|
|
|
for _, player in role_players.iterrows():
|
|
points = float(player["projected_points"])
|
|
price = self._estimate_price_simple(points, budget, role)
|
|
|
|
if total_cost + price > budget:
|
|
continue
|
|
|
|
total_cost += price
|
|
name = player.get("name", player.get("player_name", "unknown"))
|
|
selections.append({
|
|
"player": name,
|
|
"role": role,
|
|
"projected_points": points,
|
|
"estimated_cost": round(price, 1),
|
|
"value_ratio": round(points / max(price, 1), 3),
|
|
})
|
|
filled += 1
|
|
|
|
if filled >= slots:
|
|
break
|
|
|
|
return pd.DataFrame(selections)
|
|
|
|
def _proportional_allocation_as_dict(self, roles: list) -> Dict[str, float]:
|
|
fractions = self._proportional_allocation(
|
|
pd.DataFrame(columns=["role", "projected_points"]), roles
|
|
)
|
|
return {roles[i]: round(float(fractions[i] * self.total_budget), 1) for i in range(len(roles))}
|
|
|
|
# ------------------------------------------------------------------
|
|
# Sensitivity analysis
|
|
# ------------------------------------------------------------------
|
|
|
|
def sensitivity_analysis(
|
|
self,
|
|
player_pool_df: pd.DataFrame,
|
|
role: Optional[str] = None,
|
|
delta_pct: float = 0.05,
|
|
n_steps: int = 11,
|
|
n_calls: int = 50,
|
|
) -> dict:
|
|
"""Test how shifting budget into/out of one role affects total value.
|
|
|
|
Args:
|
|
player_pool_df: current player pool.
|
|
role: role to perturb. If None, tests all roles.
|
|
delta_pct: fractional step size.
|
|
n_steps: number of steps in each direction.
|
|
n_calls: number of optimization calls (for compatibility).
|
|
|
|
Returns:
|
|
Dict with at least a 'per_role' key containing per-role analysis.
|
|
"""
|
|
base_allocation = self.get_allocation()
|
|
roles_to_test = [role] if role else list(base_allocation.keys())
|
|
per_role = {}
|
|
|
|
for test_role in roles_to_test:
|
|
results = []
|
|
shifts = np.linspace(-delta_pct * n_steps, delta_pct * n_steps, 2 * n_steps + 1)
|
|
|
|
for shift in shifts:
|
|
adjusted = {}
|
|
for r, val in base_allocation.items():
|
|
adjusted[r] = val * (1.0 + (shift if r == test_role else -shift / 3.0))
|
|
|
|
total_adj = sum(adjusted.values())
|
|
for r in adjusted:
|
|
adjusted[r] = adjusted[r] / total_adj * self.total_budget
|
|
|
|
fractions = np.array([adjusted[r] / self.total_budget for r in base_allocation.keys()])
|
|
value = self._objective(
|
|
fractions,
|
|
list(base_allocation.keys()),
|
|
player_pool_df,
|
|
)
|
|
|
|
results.append({
|
|
"shift_pct": round(shift * 100, 1),
|
|
"allocation": {r: round(v, 1) for r, v in adjusted.items()},
|
|
"total_value": round(value, 1),
|
|
})
|
|
|
|
per_role[test_role] = pd.DataFrame(results)
|
|
|
|
return {"per_role": per_role, "base_allocation": base_allocation}
|