Files
fantabeto/src/optimization/budget_optimizer.py
T
ramseshk b0fab62a87 feat: add 10 new ML models for auction optimization (Phases 1-6)
Phase 1 - Quick Wins:
- QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding
- MinutesSurvivalModel: Weibull AFT for minutes distribution modeling

Phase 2 - Adaptive Auction:
- BanditAuctionSolver: Thompson Sampling for live auction bids
- OpponentBidModel: Predict competitor bids via LightGBM
- BudgetOptimizer: Bayesian optimization for role-level allocation

Phase 3 - Deep Learning:
- RLAuctionPolicy: Double DQN agent for auction strategy
- SetTransformer: Team composition valuation via set-based ML

Phase 4 - Probabilistic:
- BayesianPlayerModel: Hierarchical pooling for rookie uncertainty
- ConformalPredictor: Calibrated prediction intervals

Phase 5 - Chemistry & Form:
- PlayerChemistryGAT: Graph attention network for player synergies
- PlayerFormModel: Hawkes process for form momentum

Phase 6 - Causal:
- TransferCausalModel: Causal forest for transfer effects
- AuctionEffectAnalyzer: Bid adjustment from causal analysis

81 tests passing
2026-08-11 17:56:03 +08:00

539 lines
19 KiB
Python

"""Bayesian Optimization for role-level budget allocation.
Uses Gaussian Process regression to find optimal budget distribution
across roles (P, D, C, A) that maximizes total projected team value.
"""
import logging
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Tuple
import numpy as np
import pandas as pd
from scipy.optimize import minimize
from scipy.special import softmax
from src.optimization.auction_solver import AuctionConfig
logger = logging.getLogger(__name__)
DEFAULT_ROSTER_QUOTAS = {"P": 3, "D": 8, "C": 8, "A": 6}
@dataclass
class RoleBudgetResult:
allocation: Dict[str, float]
total_value: float
role_values: Dict[str, float]
value_curves: Dict[str, Tuple[np.ndarray, np.ndarray]]
class BudgetOptimizer:
"""Bayesian Optimization for budget allocation across roster roles.
Finds the split of total_budget across P/D/C/A that yields the
highest possible team points via greedy fill within each role's budget.
Supports mid-auction adaptive rebalancing.
"""
def __init__(
self,
total_budget: float = 500.0,
roster_quotas: Optional[Dict[str, int]] = None,
auction_config: Optional[AuctionConfig] = None,
random_state: int = 42,
):
self.total_budget = total_budget
self.roster_quotas = roster_quotas or dict(DEFAULT_ROSTER_QUOTAS)
self.auction_config = auction_config or AuctionConfig()
self.random_state = random_state
self.rng = np.random.RandomState(random_state)
self._last_allocation: Optional[Dict[str, float]] = None
self._last_value: float = 0.0
self._optimization_history: List[dict] = []
self._gk_available = False
# ------------------------------------------------------------------
# Core optimization
# ------------------------------------------------------------------
def optimize(
self,
player_pool_df: pd.DataFrame,
n_calls: int = 50,
use_skopt: bool = True,
) -> Dict[str, float]:
"""Optimize budget allocation across roles.
Uses scikit-optimize GaussianProcessRegressor if available,
otherwise simplex-based local search.
Args:
player_pool_df: DataFrame with columns [name, role, projected_points].
n_calls: number of GP evaluations.
use_skopt: attempt Gaussian Process optimization.
Returns:
Dict mapping role -> recommended budget amount.
"""
if "role" not in player_pool_df.columns or "projected_points" not in player_pool_df.columns:
raise ValueError("player_pool_df must have 'role' and 'projected_points' columns")
roles = list(self.roster_quotas.keys())
n_roles = len(roles)
if use_skopt:
try:
return self._optimize_gp(player_pool_df, roles, n_calls)
except ImportError:
logger.info("scikit-optimize not installed. Using local search.")
except Exception as exc:
logger.warning(f"GP optimization failed: {exc}. Using local search.")
return self._optimize_local(player_pool_df, roles, n_calls)
def _optimize_gp(
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
) -> Dict[str, float]:
"""Gaussian Process-based budget optimization."""
from skopt import gp_minimize
from skopt.space import Space
from skopt.learning import GaussianProcessRegressor
n_roles = len(roles)
space = Space([(0.01, 0.70) for _ in range(n_roles)])
def objective_wrapper(fractions):
fractions = np.array(fractions, dtype=float)
fractions = self._normalize_fractions(fractions)
value = self._objective(fractions, roles, player_pool_df)
self._optimization_history.append({
"fractions": fractions.tolist(),
"value": value,
})
return -value
def params_to_fractions(params):
return self._normalize_fractions(np.array(params, dtype=float))
result = gp_minimize(
objective_wrapper,
space,
n_calls=n_calls,
random_state=self.random_state,
n_initial_points=max(10, n_calls // 5),
verbose=False,
n_jobs=-1,
)
best_fractions = params_to_fractions(result.x)
self._last_allocation = self._fractions_to_allocation(best_fractions, roles)
self._last_value = -result.fun
logger.info(
f"GP Budget optimization complete: {self._last_allocation} "
f"=> value={self._last_value:.1f}"
)
return self._last_allocation
def _optimize_local(
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
) -> Dict[str, float]:
"""Local search optimization using Nelder-Mead simplex."""
n_roles = len(roles)
best_allocation = None
best_value = -float("inf")
for restart in range(max(n_calls // 10, 1)):
x0 = self._random_allocation(n_roles)
def simplex_objective(fractions):
fractions = self._normalize_fractions(np.array(fractions, dtype=float))
value = self._objective(fractions, roles, player_pool_df)
self._optimization_history.append({
"fractions": fractions.tolist(),
"value": value,
})
return -value
res = minimize(
simplex_objective,
x0,
method="Nelder-Mead",
options={"maxiter": max(n_calls // 3, 20), "xatol": 1e-3, "fatol": 1e-3},
)
fractions = self._normalize_fractions(np.array(res.x, dtype=float))
value = -res.fun
if value > best_value:
best_value = value
best_allocation = fractions
if best_allocation is None:
best_allocation = self._proportional_allocation(player_pool_df, roles)
self._last_allocation = self._fractions_to_allocation(best_allocation, roles)
self._last_value = best_value
logger.info(
f"Local optimization complete: {self._last_allocation} "
f"=> value={self._last_value:.1f}"
)
return self._last_allocation
# ------------------------------------------------------------------
# Adaptive rebalancing
# ------------------------------------------------------------------
def optimize_adaptive(
self,
player_pool_df: pd.DataFrame,
remaining_slots: Dict[str, int],
spent_per_role: Dict[str, float],
n_calls: int = 30,
) -> Dict[str, float]:
"""Mid-auction rebalancing — optimize remaining budget for unfilled slots.
Args:
player_pool_df: remaining available players pool.
remaining_slots: dict of role -> slots still needed.
spent_per_role: dict of role -> credits already spent.
n_calls: GP evaluation budget.
Returns:
Dict mapping role -> recommended budget for remaining slots.
"""
remaining_budget = self.total_budget - sum(spent_per_role.values())
remaining_budget = max(1.0, remaining_budget)
if sum(remaining_slots.values()) == 0:
logger.info("All slots filled. No budget to allocate.")
return {r: 0.0 for r in self.roster_quotas}
saved_quotas = self.roster_quotas
saved_total = self.total_budget
self.roster_quotas = dict(remaining_slots)
self.total_budget = remaining_budget
available = player_pool_df[
player_pool_df["role"].isin(
[r for r, s in remaining_slots.items() if s > 0]
)
]
if len(available) == 0:
logger.warning("No available players for remaining slots.")
self.roster_quotas = saved_quotas
self.total_budget = saved_total
return {r: 0.0 for r in saved_quotas}
allocation = self.optimize(
available,
n_calls=n_calls,
use_skopt=True,
)
self.roster_quotas = saved_quotas
self.total_budget = saved_total
result = {}
for role in saved_quotas:
result[role] = allocation.get(role, 0.0)
return result
# ------------------------------------------------------------------
# Objective function
# ------------------------------------------------------------------
def _objective(
self,
fractions: np.ndarray,
roles: list,
player_pool_df: pd.DataFrame,
) -> float:
"""Simulate greedy fill within role budgets; return total projected points.
For each role, pick the best players by projected_points until the
role budget or slot quota is exhausted.
"""
total_value = 0.0
for i, role in enumerate(roles):
budget = fractions[i] * self.total_budget
slots = self.roster_quotas.get(role, 0)
role_players = player_pool_df[player_pool_df["role"] == role].copy()
if len(role_players) == 0 or slots == 0:
continue
role_players = role_players.sort_values(
"projected_points", ascending=False
)
total_cost = 0.0
filled = 0
for _, player in role_players.iterrows():
points = float(player["projected_points"])
estimated_price = self._estimate_price_simple(points, budget, role)
if total_cost + estimated_price > budget:
continue
total_cost += estimated_price
total_value += points
filled += 1
if filled >= slots:
break
return total_value
def _estimate_price_simple(
self, projected_points: float, role_budget: float, role: str
) -> float:
"""Simple price estimate: points * role_factor clamped within budget."""
role_factor = {"P": 4.0, "D": 2.5, "C": 3.0, "A": 4.5}.get(role, 3.0)
price = projected_points * role_factor
max_price = role_budget * 0.50
return min(price, max_price, role_budget)
# ------------------------------------------------------------------
# Allocation access and visualization data
# ------------------------------------------------------------------
def get_allocation(self) -> Dict[str, float]:
"""Return last computed allocation (role -> budget amount)."""
if self._last_allocation is None:
return {r: self.total_budget / len(self.roster_quotas) for r in self.roster_quotas}
return dict(self._last_allocation)
def get_role_value_curves(
self,
player_pool_df: pd.DataFrame,
n_points: int = 20,
) -> Dict[str, Tuple[np.ndarray, np.ndarray]]:
"""Compute diminishing returns curves: budget vs. expected points per role.
Returns:
dict role -> (budget_array, value_array).
"""
curves = {}
budget_step = self.total_budget / n_points
for role, slots in self.roster_quotas.items():
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
"projected_points", ascending=False
)
budgets = np.linspace(0, self.total_budget, n_points)
values = np.zeros(n_points)
for i, budget_limit in enumerate(budgets):
total_cost = 0.0
total_value = 0.0
filled = 0
for _, player in role_players.iterrows():
points = float(player["projected_points"])
price = self._estimate_price_simple(points, budget_limit, role)
if total_cost + price > budget_limit:
continue
total_cost += price
total_value += points
filled += 1
if filled >= slots:
break
values[i] = total_value
curves[role] = (budgets.copy(), values.copy())
return curves
# ------------------------------------------------------------------
# Fallback: proportional allocation
# ------------------------------------------------------------------
def _proportional_allocation(
self, player_pool_df: pd.DataFrame, roles: list
) -> np.ndarray:
"""Allocate budget proportional to (points_variance * slots) per role."""
weights = np.zeros(len(roles))
for i, role in enumerate(roles):
role_players = player_pool_df[player_pool_df["role"] == role]
if len(role_players) > 1:
variance = role_players["projected_points"].var()
else:
variance = 1.0
slots = self.roster_quotas.get(role, 1)
weights[i] = variance * slots
weight_sum = weights.sum()
if weight_sum <= 0:
return np.ones(len(roles)) / len(roles)
fractions = weights / weight_sum
fractions = np.clip(fractions, 0.02, 0.70)
fractions = fractions / fractions.sum()
logger.info(
f"Proportional allocation (fallback): "
f"{dict(zip(roles, fractions.round(3)))}"
)
return fractions
# ------------------------------------------------------------------
# Utilities
# ------------------------------------------------------------------
def _normalize_fractions(self, fractions: np.ndarray) -> np.ndarray:
"""Normalize fractions to sum to 1.0 with minimum per role."""
fractions = np.clip(fractions, 0.01, 0.70)
total = fractions.sum()
if total <= 0:
return np.ones_like(fractions) / len(fractions)
return fractions / total
def _fractions_to_allocation(
self, fractions: np.ndarray, roles: list
) -> Dict[str, float]:
"""Convert fractions to absolute budget per role."""
return {
role: round(float(fractions[i] * self.total_budget), 1)
for i, role in enumerate(roles)
}
def _random_allocation(self, n_roles: int) -> np.ndarray:
"""Generate a random allocation via Dirichlet."""
alpha = np.ones(n_roles) * 2.0
return self.rng.dirichlet(alpha)
def get_optimization_trace(self) -> pd.DataFrame:
"""Return DataFrame of all evaluated allocations during optimization."""
if not self._optimization_history:
return pd.DataFrame()
return pd.DataFrame(self._optimization_history)
def estimate_team_composition(
self,
player_pool_df: pd.DataFrame,
allocation: Optional[Dict[str, float]] = None,
) -> pd.DataFrame:
"""Given final allocation, return the recommended player selections.
Returns:
DataFrame with selected players, their estimated costs, and value.
"""
roles = list(self.roster_quotas.keys())
alloc = allocation or self._last_allocation
if alloc is None:
alloc = self._proportional_allocation_as_dict(roles)
selections = []
for role in roles:
budget = alloc.get(role, 0.0)
slots = self.roster_quotas.get(role, 0)
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
"projected_points", ascending=False
)
total_cost = 0.0
filled = 0
for _, player in role_players.iterrows():
points = float(player["projected_points"])
price = self._estimate_price_simple(points, budget, role)
if total_cost + price > budget:
continue
total_cost += price
name = player.get("name", player.get("player_name", "unknown"))
selections.append({
"player": name,
"role": role,
"projected_points": points,
"estimated_cost": round(price, 1),
"value_ratio": round(points / max(price, 1), 3),
})
filled += 1
if filled >= slots:
break
return pd.DataFrame(selections)
def _proportional_allocation_as_dict(self, roles: list) -> Dict[str, float]:
fractions = self._proportional_allocation(
pd.DataFrame(columns=["role", "projected_points"]), roles
)
return {roles[i]: round(float(fractions[i] * self.total_budget), 1) for i in range(len(roles))}
# ------------------------------------------------------------------
# Sensitivity analysis
# ------------------------------------------------------------------
def sensitivity_analysis(
self,
player_pool_df: pd.DataFrame,
role: Optional[str] = None,
delta_pct: float = 0.05,
n_steps: int = 11,
n_calls: int = 50,
) -> dict:
"""Test how shifting budget into/out of one role affects total value.
Args:
player_pool_df: current player pool.
role: role to perturb. If None, tests all roles.
delta_pct: fractional step size.
n_steps: number of steps in each direction.
n_calls: number of optimization calls (for compatibility).
Returns:
Dict with at least a 'per_role' key containing per-role analysis.
"""
base_allocation = self.get_allocation()
roles_to_test = [role] if role else list(base_allocation.keys())
per_role = {}
for test_role in roles_to_test:
results = []
shifts = np.linspace(-delta_pct * n_steps, delta_pct * n_steps, 2 * n_steps + 1)
for shift in shifts:
adjusted = {}
for r, val in base_allocation.items():
adjusted[r] = val * (1.0 + (shift if r == test_role else -shift / 3.0))
total_adj = sum(adjusted.values())
for r in adjusted:
adjusted[r] = adjusted[r] / total_adj * self.total_budget
fractions = np.array([adjusted[r] / self.total_budget for r in base_allocation.keys()])
value = self._objective(
fractions,
list(base_allocation.keys()),
player_pool_df,
)
results.append({
"shift_pct": round(shift * 100, 1),
"allocation": {r: round(v, 1) for r, v in adjusted.items()},
"total_value": round(value, 1),
})
per_role[test_role] = pd.DataFrame(results)
return {"per_role": per_role, "base_allocation": base_allocation}