feat: add 10 new ML models for auction optimization (Phases 1-6)

Phase 1 - Quick Wins:
- QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding
- MinutesSurvivalModel: Weibull AFT for minutes distribution modeling

Phase 2 - Adaptive Auction:
- BanditAuctionSolver: Thompson Sampling for live auction bids
- OpponentBidModel: Predict competitor bids via LightGBM
- BudgetOptimizer: Bayesian optimization for role-level allocation

Phase 3 - Deep Learning:
- RLAuctionPolicy: Double DQN agent for auction strategy
- SetTransformer: Team composition valuation via set-based ML

Phase 4 - Probabilistic:
- BayesianPlayerModel: Hierarchical pooling for rookie uncertainty
- ConformalPredictor: Calibrated prediction intervals

Phase 5 - Chemistry & Form:
- PlayerChemistryGAT: Graph attention network for player synergies
- PlayerFormModel: Hawkes process for form momentum

Phase 6 - Causal:
- TransferCausalModel: Causal forest for transfer effects
- AuctionEffectAnalyzer: Bid adjustment from causal analysis

81 tests passing
This commit is contained in:
ramseshk
2026-08-11 17:56:03 +08:00
parent 916278a640
commit b0fab62a87
16 changed files with 4955 additions and 21 deletions
+13
View File
@@ -1 +1,14 @@
"""Optimization modules for auction and lineup selection."""
from .auction_solver import AuctionSolver, AuctionConfig, PlayerValuation
from .lineup_solver import LineupSolver, LineupConstraints, PlayerScore, MCTSNode
from .opponent_model import OpponentModel
from .transfer_analyzer import TransferAnalyzer
# Phase 2: Bandit, opponent bidding, budget optimization
from .bandit_auction import BanditAuctionSolver
from .opponent_bidding_model import OpponentBidModel
from .budget_optimizer import BudgetOptimizer
# Phase 3: Reinforcement learning auction agent
from .rl_auction_agent import AuctionEnv, RLAuctionPolicy, RLAuctionTrainer, QNetwork
+359
View File
@@ -0,0 +1,359 @@
"""Contextual Multi-Armed Bandit for live auction bidding decisions.
Uses Thompson Sampling with Beta-distributed posteriors over discrete bid levels.
Falls back to UCB when exploration depth is insufficient.
Integrates with AuctionConfig from auction_solver.py.
"""
import logging
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Tuple
import numpy as np
import pandas as pd
from scipy.stats import beta as beta_dist
logger = logging.getLogger(__name__)
# Default bid arms as fractions of total budget
DEFAULT_BID_ARMS = np.array(
[0.0, 0.005, 0.01, 0.02, 0.03, 0.05, 0.08, 0.12, 0.18, 0.25],
dtype=np.float64,
)
ROLES = ["P", "D", "C", "A"]
ROLE_SCARCITY = {"P": 3, "D": 8, "C": 8, "A": 6}
ROLE_POOL_SIZE = {"P": 4, "D": 22, "C": 24, "A": 12}
@dataclass
class BanditArmState:
alpha: float = 1.0
beta: float = 1.0
trials: int = 0
wins: float = 0.0
@dataclass
class AuctionState:
budget_remaining: float = 500.0
total_budget: float = 500.0
slots_filled: Dict[str, int] = field(default_factory=lambda: {"P": 0, "D": 0, "C": 0, "A": 0})
slots_total: Dict[str, int] = field(default_factory=lambda: {"P": 3, "D": 8, "C": 8, "A": 6})
round_number: int = 1
opponent_budgets: List[float] = field(default_factory=list)
@dataclass
class PlayerContext:
name: str
role: str
projected_points: float
role_scarcity: float
value_over_replacement: float
budget_remaining_fraction: float
slots_remaining_in_role: int
round_number: int
opponent_budget_avg: float
def _get_attr(obj, key, default=None):
"""Get attribute or dict key from an object."""
if isinstance(obj, dict):
return obj.get(key, default)
return getattr(obj, key, default)
class BanditAuctionSolver:
"""Thompson Sampling bandit for live auction bid selection.
Arms are discrete bid fractions. Each (role, scarcity_level) maintains
independent Beta posteriors. Falls back to UCB when total observations
for a context group are < 50.
"""
def __init__(
self,
bid_arms: Optional[np.ndarray] = None,
total_budget: float = 500.0,
min_obs_for_ts: int = 50,
ucb_exploration: float = 1.414,
config=None,
):
if config is not None:
from .auction_solver import AuctionConfig
total_budget = config.total_budget
self.bid_arms = bid_arms if bid_arms is not None else DEFAULT_BID_ARMS
self.n_arms = len(self.bid_arms)
self.total_budget = total_budget
self.min_obs_for_ts = min_obs_for_ts
self.ucb_exploration = ucb_exploration
self.bid_fractions = self.bid_arms
self.posteriors: Dict[Tuple[str, int], List[BanditArmState]] = {}
for role in ROLES:
self._ensure_posteriors(role, 1)
def _context_key(self, role: str, scarcity_level: int) -> Tuple[str, int]:
return (role, scarcity_level)
def _ensure_posteriors(self, role: str, n_slots_remaining: int):
scarcity = max(1, n_slots_remaining)
key = self._context_key(role, scarcity)
if key not in self.posteriors:
self.posteriors[key] = [
BanditArmState(alpha=1.0, beta=1.0, trials=0, wins=0.0)
for _ in range(self.n_arms)
]
logger.debug(f"Initialized bandit posteriors for role={role}, scarcity={scarcity}")
def _total_obs(self, key: Tuple[str, int]) -> int:
arms = self.posteriors.get(key, [])
return sum(a.trials for a in arms)
def compute_context(
self,
player,
auction_state,
pool_stats: Optional[dict] = None,
) -> PlayerContext:
"""Build context vector for a player given current auction state.
Args:
player: dict or object with name, role, projected_points attributes.
auction_state: current AuctionState (or dict with same keys).
pool_stats: optional dict with role-level pool means and stds.
Returns:
PlayerContext dataclass with all context features.
"""
role = _get_attr(player, "role")
points = float(_get_attr(player, "projected_points", 6.5))
name = _get_attr(player, "name", "unknown")
total_slots = _get_attr(auction_state, "slots_total", {}).get(role, 1)
filled = _get_attr(auction_state, "slots_filled", {}).get(role, 0)
slots_remaining = max(total_slots - filled, 0)
scarcity = max(slots_remaining / max(total_slots, 1), 0.05)
budget_remaining = _get_attr(auction_state, "budget_remaining", 500.0)
total_budget_attr = _get_attr(auction_state, "total_budget", 500.0)
budget_fraction = budget_remaining / max(total_budget_attr, 1)
opponent_budget_avg = 0.0
opponent_budgets = _get_attr(auction_state, "opponent_budgets", [])
if opponent_budgets:
opponent_budget_avg = float(np.mean(opponent_budgets))
elif total_budget_attr > 0:
opponent_budget_avg = total_budget_attr * 0.6
if pool_stats and role in pool_stats:
role_mean = pool_stats[role].get("mean", 0.0)
role_std = pool_stats[role].get("std", 1.0)
points_z = (points - role_mean) / max(role_std, 0.01) if role_std > 0 else 0.0
else:
points_z = points / 15.0
vor = max(points - 6.5, 0.0)
return PlayerContext(
name=name,
role=role,
projected_points=points,
role_scarcity=scarcity,
value_over_replacement=vor,
budget_remaining_fraction=budget_fraction,
slots_remaining_in_role=slots_remaining,
round_number=_get_attr(auction_state, "round_number", 1),
opponent_budget_avg=opponent_budget_avg,
)
def select_bid(
self,
player,
auction_state,
pool_stats: Optional[dict] = None,
) -> Tuple[int, float]:
"""Select bid arm using Thompson Sampling (or UCB fallback).
Returns:
(arm_index, bid_amount_in_credits)
"""
ctx = self.compute_context(player, auction_state, pool_stats)
role = ctx.role
n_slots = ctx.slots_remaining_in_role
self._ensure_posteriors(role, n_slots)
key = self._context_key(role, n_slots)
arms = self.posteriors[key]
total_obs = sum(a.trials for a in arms)
if total_obs >= self.min_obs_for_ts:
samples = [float(np.random.beta(a.alpha, max(a.beta, 0.01))) for a in arms]
arm_idx = int(np.argmax(samples))
logger.debug(
f"Thompson Sampling: role={role}, arms_sampled={samples[:5]}..., "
f"selected_arm={arm_idx}"
)
else:
values = []
for _i, arm in enumerate(arms):
if arm.trials == 0:
values.append(float("inf"))
else:
mean = arm.wins / arm.trials
bonus = self.ucb_exploration * np.sqrt(
np.log(max(total_obs, 1)) / arm.trials
)
values.append(mean + bonus)
arm_idx = int(np.argmax(values))
logger.debug(
f"UCB fallback: role={role}, total_obs={total_obs}, "
f"selected_arm={arm_idx}"
)
bid_amount = round(self.bid_arms[arm_idx] * self.total_budget)
budget_rem = _get_attr(auction_state, "budget_remaining", self.total_budget)
if bid_amount > budget_rem:
bid_amount = budget_rem
arm_idx = int(np.argmin(np.abs(self.bid_arms * self.total_budget - bid_amount)))
return arm_idx, bid_amount
def update(self, arm_idx: int, reward: float, player_role: str):
"""Update Beta posterior for the selected arm.
Reward should be a normalized value: (player_season_value - cost) scaled.
Args:
arm_idx: index of the selected arm.
reward: normalized reward signal (higher = better purchase).
player_role: role of the purchased player.
"""
reward_clipped = max(0.0, min(1.0, reward))
win = 1.0 if reward > 0 else 0.0
for key, arms in self.posteriors.items():
role, _scarcity = key
if role == player_role:
if arm_idx < len(arms):
arm = arms[arm_idx]
arm.trials += 1
arm.wins += win
arm.alpha += reward_clipped
arm.beta += (1.0 - reward_clipped)
logger.debug(
f"Updated arm {arm_idx} for {player_role}: "
f"trials={arm.trials}, alpha={arm.alpha:.2f}, beta={arm.beta:.2f}"
)
for key, arms in self.posteriors.items():
role, _scarcity = key
if role == player_role:
for i, arm in enumerate(arms):
if i == arm_idx:
continue
arm.beta = max(arm.beta, 1.001)
def get_arm_stats(self) -> Dict[str, dict]:
"""Return arm statistics for analysis.
Returns:
dict mapping "role/scarcity/arm_idx" -> stats dict.
"""
stats = {}
agg = {}
for (role, scarcity), arms in self.posteriors.items():
for idx, arm in enumerate(arms):
label = f"{role}/scarcity={scarcity}/arm={idx}"
stats[label] = {
"trials": arm.trials,
"wins": arm.wins,
"alpha": arm.alpha,
"beta": arm.beta,
"win_rate": arm.wins / max(arm.trials, 1),
"bid_fraction": float(self.bid_arms[idx]),
"bid_amount": float(self.bid_arms[idx] * self.total_budget),
}
if idx not in agg:
agg[idx] = {"trials": 0, "wins": 0.0, "alpha": 0.0, "beta": 0.0}
agg[idx]["trials"] += arm.trials
agg[idx]["wins"] += arm.wins
agg[idx]["alpha"] += arm.alpha
agg[idx]["beta"] += arm.beta
for idx, a in agg.items():
stats[idx] = {
"trials": a["trials"],
"wins": a["wins"],
"alpha": a["alpha"],
"beta": a["beta"],
"win_rate": a["wins"] / max(a["trials"], 1),
"bid_fraction": float(self.bid_arms[idx]),
"bid_amount": float(self.bid_arms[idx] * self.total_budget),
}
return stats
def exploration_bonus(self, player_role: str, player_points: float = 0.0) -> float:
"""Compute exploration bonus for new/unknown player types.
Higher bonus when the bandit has limited experience with a role.
Encourages exploring sleeper players.
Returns:
Recommended extra bid amount in credits.
"""
total_obs = 0
for (role, _scarcity), arms in self.posteriors.items():
if role == player_role:
total_obs += sum(a.trials for a in arms)
if total_obs == 0:
bonus = self.total_budget * 0.04
elif total_obs < 20:
bonus = self.total_budget * 0.025
elif total_obs < 50:
bonus = self.total_budget * 0.01
else:
bonus = 0.0
if bonus > 0:
base = max(player_points * 0.5, 0)
bonus += base * (1.0 / max(total_obs, 1)) * 20
logger.debug(f"Exploration bonus for {player_role}: {bonus:.1f} (obs={total_obs})")
return bonus
def recommend_bid_summary(
self,
player,
auction_state,
pool_stats: Optional[dict] = None,
) -> dict:
"""Full bidding recommendation for a player.
Returns a dict with arm index, bid amount, context, exploration bonus,
and total recommended bid.
"""
arm_idx, bid_amount = self.select_bid(player, auction_state, pool_stats)
ctx = self.compute_context(player, auction_state, pool_stats)
bonus = self.exploration_bonus(ctx.role, ctx.projected_points)
return {
"player": _get_attr(player, "name", "unknown"),
"role": ctx.role,
"arm_index": arm_idx,
"base_bid": bid_amount,
"exploration_bonus": round(bonus),
"total_bid": round(bid_amount + bonus),
"recommended_bid": round(bid_amount + bonus),
"context": {
"points_zscore": round(
(ctx.projected_points - 6.5) / 2.0, 2
),
"role_scarcity": round(ctx.role_scarcity, 3),
"budget_remaining_frac": round(ctx.budget_remaining_fraction, 3),
"slots_remaining": ctx.slots_remaining_in_role,
"round": ctx.round_number,
"vor": round(ctx.value_over_replacement, 1),
},
}
+538
View File
@@ -0,0 +1,538 @@
"""Bayesian Optimization for role-level budget allocation.
Uses Gaussian Process regression to find optimal budget distribution
across roles (P, D, C, A) that maximizes total projected team value.
"""
import logging
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Tuple
import numpy as np
import pandas as pd
from scipy.optimize import minimize
from scipy.special import softmax
from src.optimization.auction_solver import AuctionConfig
logger = logging.getLogger(__name__)
DEFAULT_ROSTER_QUOTAS = {"P": 3, "D": 8, "C": 8, "A": 6}
@dataclass
class RoleBudgetResult:
allocation: Dict[str, float]
total_value: float
role_values: Dict[str, float]
value_curves: Dict[str, Tuple[np.ndarray, np.ndarray]]
class BudgetOptimizer:
"""Bayesian Optimization for budget allocation across roster roles.
Finds the split of total_budget across P/D/C/A that yields the
highest possible team points via greedy fill within each role's budget.
Supports mid-auction adaptive rebalancing.
"""
def __init__(
self,
total_budget: float = 500.0,
roster_quotas: Optional[Dict[str, int]] = None,
auction_config: Optional[AuctionConfig] = None,
random_state: int = 42,
):
self.total_budget = total_budget
self.roster_quotas = roster_quotas or dict(DEFAULT_ROSTER_QUOTAS)
self.auction_config = auction_config or AuctionConfig()
self.random_state = random_state
self.rng = np.random.RandomState(random_state)
self._last_allocation: Optional[Dict[str, float]] = None
self._last_value: float = 0.0
self._optimization_history: List[dict] = []
self._gk_available = False
# ------------------------------------------------------------------
# Core optimization
# ------------------------------------------------------------------
def optimize(
self,
player_pool_df: pd.DataFrame,
n_calls: int = 50,
use_skopt: bool = True,
) -> Dict[str, float]:
"""Optimize budget allocation across roles.
Uses scikit-optimize GaussianProcessRegressor if available,
otherwise simplex-based local search.
Args:
player_pool_df: DataFrame with columns [name, role, projected_points].
n_calls: number of GP evaluations.
use_skopt: attempt Gaussian Process optimization.
Returns:
Dict mapping role -> recommended budget amount.
"""
if "role" not in player_pool_df.columns or "projected_points" not in player_pool_df.columns:
raise ValueError("player_pool_df must have 'role' and 'projected_points' columns")
roles = list(self.roster_quotas.keys())
n_roles = len(roles)
if use_skopt:
try:
return self._optimize_gp(player_pool_df, roles, n_calls)
except ImportError:
logger.info("scikit-optimize not installed. Using local search.")
except Exception as exc:
logger.warning(f"GP optimization failed: {exc}. Using local search.")
return self._optimize_local(player_pool_df, roles, n_calls)
def _optimize_gp(
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
) -> Dict[str, float]:
"""Gaussian Process-based budget optimization."""
from skopt import gp_minimize
from skopt.space import Space
from skopt.learning import GaussianProcessRegressor
n_roles = len(roles)
space = Space([(0.01, 0.70) for _ in range(n_roles)])
def objective_wrapper(fractions):
fractions = np.array(fractions, dtype=float)
fractions = self._normalize_fractions(fractions)
value = self._objective(fractions, roles, player_pool_df)
self._optimization_history.append({
"fractions": fractions.tolist(),
"value": value,
})
return -value
def params_to_fractions(params):
return self._normalize_fractions(np.array(params, dtype=float))
result = gp_minimize(
objective_wrapper,
space,
n_calls=n_calls,
random_state=self.random_state,
n_initial_points=max(10, n_calls // 5),
verbose=False,
n_jobs=-1,
)
best_fractions = params_to_fractions(result.x)
self._last_allocation = self._fractions_to_allocation(best_fractions, roles)
self._last_value = -result.fun
logger.info(
f"GP Budget optimization complete: {self._last_allocation} "
f"=> value={self._last_value:.1f}"
)
return self._last_allocation
def _optimize_local(
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
) -> Dict[str, float]:
"""Local search optimization using Nelder-Mead simplex."""
n_roles = len(roles)
best_allocation = None
best_value = -float("inf")
for restart in range(max(n_calls // 10, 1)):
x0 = self._random_allocation(n_roles)
def simplex_objective(fractions):
fractions = self._normalize_fractions(np.array(fractions, dtype=float))
value = self._objective(fractions, roles, player_pool_df)
self._optimization_history.append({
"fractions": fractions.tolist(),
"value": value,
})
return -value
res = minimize(
simplex_objective,
x0,
method="Nelder-Mead",
options={"maxiter": max(n_calls // 3, 20), "xatol": 1e-3, "fatol": 1e-3},
)
fractions = self._normalize_fractions(np.array(res.x, dtype=float))
value = -res.fun
if value > best_value:
best_value = value
best_allocation = fractions
if best_allocation is None:
best_allocation = self._proportional_allocation(player_pool_df, roles)
self._last_allocation = self._fractions_to_allocation(best_allocation, roles)
self._last_value = best_value
logger.info(
f"Local optimization complete: {self._last_allocation} "
f"=> value={self._last_value:.1f}"
)
return self._last_allocation
# ------------------------------------------------------------------
# Adaptive rebalancing
# ------------------------------------------------------------------
def optimize_adaptive(
self,
player_pool_df: pd.DataFrame,
remaining_slots: Dict[str, int],
spent_per_role: Dict[str, float],
n_calls: int = 30,
) -> Dict[str, float]:
"""Mid-auction rebalancing — optimize remaining budget for unfilled slots.
Args:
player_pool_df: remaining available players pool.
remaining_slots: dict of role -> slots still needed.
spent_per_role: dict of role -> credits already spent.
n_calls: GP evaluation budget.
Returns:
Dict mapping role -> recommended budget for remaining slots.
"""
remaining_budget = self.total_budget - sum(spent_per_role.values())
remaining_budget = max(1.0, remaining_budget)
if sum(remaining_slots.values()) == 0:
logger.info("All slots filled. No budget to allocate.")
return {r: 0.0 for r in self.roster_quotas}
saved_quotas = self.roster_quotas
saved_total = self.total_budget
self.roster_quotas = dict(remaining_slots)
self.total_budget = remaining_budget
available = player_pool_df[
player_pool_df["role"].isin(
[r for r, s in remaining_slots.items() if s > 0]
)
]
if len(available) == 0:
logger.warning("No available players for remaining slots.")
self.roster_quotas = saved_quotas
self.total_budget = saved_total
return {r: 0.0 for r in saved_quotas}
allocation = self.optimize(
available,
n_calls=n_calls,
use_skopt=True,
)
self.roster_quotas = saved_quotas
self.total_budget = saved_total
result = {}
for role in saved_quotas:
result[role] = allocation.get(role, 0.0)
return result
# ------------------------------------------------------------------
# Objective function
# ------------------------------------------------------------------
def _objective(
self,
fractions: np.ndarray,
roles: list,
player_pool_df: pd.DataFrame,
) -> float:
"""Simulate greedy fill within role budgets; return total projected points.
For each role, pick the best players by projected_points until the
role budget or slot quota is exhausted.
"""
total_value = 0.0
for i, role in enumerate(roles):
budget = fractions[i] * self.total_budget
slots = self.roster_quotas.get(role, 0)
role_players = player_pool_df[player_pool_df["role"] == role].copy()
if len(role_players) == 0 or slots == 0:
continue
role_players = role_players.sort_values(
"projected_points", ascending=False
)
total_cost = 0.0
filled = 0
for _, player in role_players.iterrows():
points = float(player["projected_points"])
estimated_price = self._estimate_price_simple(points, budget, role)
if total_cost + estimated_price > budget:
continue
total_cost += estimated_price
total_value += points
filled += 1
if filled >= slots:
break
return total_value
def _estimate_price_simple(
self, projected_points: float, role_budget: float, role: str
) -> float:
"""Simple price estimate: points * role_factor clamped within budget."""
role_factor = {"P": 4.0, "D": 2.5, "C": 3.0, "A": 4.5}.get(role, 3.0)
price = projected_points * role_factor
max_price = role_budget * 0.50
return min(price, max_price, role_budget)
# ------------------------------------------------------------------
# Allocation access and visualization data
# ------------------------------------------------------------------
def get_allocation(self) -> Dict[str, float]:
"""Return last computed allocation (role -> budget amount)."""
if self._last_allocation is None:
return {r: self.total_budget / len(self.roster_quotas) for r in self.roster_quotas}
return dict(self._last_allocation)
def get_role_value_curves(
self,
player_pool_df: pd.DataFrame,
n_points: int = 20,
) -> Dict[str, Tuple[np.ndarray, np.ndarray]]:
"""Compute diminishing returns curves: budget vs. expected points per role.
Returns:
dict role -> (budget_array, value_array).
"""
curves = {}
budget_step = self.total_budget / n_points
for role, slots in self.roster_quotas.items():
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
"projected_points", ascending=False
)
budgets = np.linspace(0, self.total_budget, n_points)
values = np.zeros(n_points)
for i, budget_limit in enumerate(budgets):
total_cost = 0.0
total_value = 0.0
filled = 0
for _, player in role_players.iterrows():
points = float(player["projected_points"])
price = self._estimate_price_simple(points, budget_limit, role)
if total_cost + price > budget_limit:
continue
total_cost += price
total_value += points
filled += 1
if filled >= slots:
break
values[i] = total_value
curves[role] = (budgets.copy(), values.copy())
return curves
# ------------------------------------------------------------------
# Fallback: proportional allocation
# ------------------------------------------------------------------
def _proportional_allocation(
self, player_pool_df: pd.DataFrame, roles: list
) -> np.ndarray:
"""Allocate budget proportional to (points_variance * slots) per role."""
weights = np.zeros(len(roles))
for i, role in enumerate(roles):
role_players = player_pool_df[player_pool_df["role"] == role]
if len(role_players) > 1:
variance = role_players["projected_points"].var()
else:
variance = 1.0
slots = self.roster_quotas.get(role, 1)
weights[i] = variance * slots
weight_sum = weights.sum()
if weight_sum <= 0:
return np.ones(len(roles)) / len(roles)
fractions = weights / weight_sum
fractions = np.clip(fractions, 0.02, 0.70)
fractions = fractions / fractions.sum()
logger.info(
f"Proportional allocation (fallback): "
f"{dict(zip(roles, fractions.round(3)))}"
)
return fractions
# ------------------------------------------------------------------
# Utilities
# ------------------------------------------------------------------
def _normalize_fractions(self, fractions: np.ndarray) -> np.ndarray:
"""Normalize fractions to sum to 1.0 with minimum per role."""
fractions = np.clip(fractions, 0.01, 0.70)
total = fractions.sum()
if total <= 0:
return np.ones_like(fractions) / len(fractions)
return fractions / total
def _fractions_to_allocation(
self, fractions: np.ndarray, roles: list
) -> Dict[str, float]:
"""Convert fractions to absolute budget per role."""
return {
role: round(float(fractions[i] * self.total_budget), 1)
for i, role in enumerate(roles)
}
def _random_allocation(self, n_roles: int) -> np.ndarray:
"""Generate a random allocation via Dirichlet."""
alpha = np.ones(n_roles) * 2.0
return self.rng.dirichlet(alpha)
def get_optimization_trace(self) -> pd.DataFrame:
"""Return DataFrame of all evaluated allocations during optimization."""
if not self._optimization_history:
return pd.DataFrame()
return pd.DataFrame(self._optimization_history)
def estimate_team_composition(
self,
player_pool_df: pd.DataFrame,
allocation: Optional[Dict[str, float]] = None,
) -> pd.DataFrame:
"""Given final allocation, return the recommended player selections.
Returns:
DataFrame with selected players, their estimated costs, and value.
"""
roles = list(self.roster_quotas.keys())
alloc = allocation or self._last_allocation
if alloc is None:
alloc = self._proportional_allocation_as_dict(roles)
selections = []
for role in roles:
budget = alloc.get(role, 0.0)
slots = self.roster_quotas.get(role, 0)
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
"projected_points", ascending=False
)
total_cost = 0.0
filled = 0
for _, player in role_players.iterrows():
points = float(player["projected_points"])
price = self._estimate_price_simple(points, budget, role)
if total_cost + price > budget:
continue
total_cost += price
name = player.get("name", player.get("player_name", "unknown"))
selections.append({
"player": name,
"role": role,
"projected_points": points,
"estimated_cost": round(price, 1),
"value_ratio": round(points / max(price, 1), 3),
})
filled += 1
if filled >= slots:
break
return pd.DataFrame(selections)
def _proportional_allocation_as_dict(self, roles: list) -> Dict[str, float]:
fractions = self._proportional_allocation(
pd.DataFrame(columns=["role", "projected_points"]), roles
)
return {roles[i]: round(float(fractions[i] * self.total_budget), 1) for i in range(len(roles))}
# ------------------------------------------------------------------
# Sensitivity analysis
# ------------------------------------------------------------------
def sensitivity_analysis(
self,
player_pool_df: pd.DataFrame,
role: Optional[str] = None,
delta_pct: float = 0.05,
n_steps: int = 11,
n_calls: int = 50,
) -> dict:
"""Test how shifting budget into/out of one role affects total value.
Args:
player_pool_df: current player pool.
role: role to perturb. If None, tests all roles.
delta_pct: fractional step size.
n_steps: number of steps in each direction.
n_calls: number of optimization calls (for compatibility).
Returns:
Dict with at least a 'per_role' key containing per-role analysis.
"""
base_allocation = self.get_allocation()
roles_to_test = [role] if role else list(base_allocation.keys())
per_role = {}
for test_role in roles_to_test:
results = []
shifts = np.linspace(-delta_pct * n_steps, delta_pct * n_steps, 2 * n_steps + 1)
for shift in shifts:
adjusted = {}
for r, val in base_allocation.items():
adjusted[r] = val * (1.0 + (shift if r == test_role else -shift / 3.0))
total_adj = sum(adjusted.values())
for r in adjusted:
adjusted[r] = adjusted[r] / total_adj * self.total_budget
fractions = np.array([adjusted[r] / self.total_budget for r in base_allocation.keys()])
value = self._objective(
fractions,
list(base_allocation.keys()),
player_pool_df,
)
results.append({
"shift_pct": round(shift * 100, 1),
"allocation": {r: round(v, 1) for r, v in adjusted.items()},
"total_value": round(value, 1),
})
per_role[test_role] = pd.DataFrame(results)
return {"per_role": per_role, "base_allocation": base_allocation}
+412
View File
@@ -0,0 +1,412 @@
"""Opponent bidding behavior modeling using LightGBM.
Predicts what competitors will bid for each player in a live auction round.
Supports Monte Carlo simulation of auction outcomes and win probability estimates.
"""
import logging
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Tuple
import numpy as np
import pandas as pd
logger = logging.getLogger(__name__)
@dataclass
class OpponentState:
budget_remaining: float = 500.0
initial_budget: float = 500.0
slots_filled: Dict[str, int] = field(default_factory=lambda: {"P": 0, "D": 0, "C": 0, "A": 0})
slots_total: Dict[str, int] = field(default_factory=lambda: {"P": 3, "D": 8, "C": 8, "A": 6})
round_number: int = 1
aggression_factor: float = 1.0
def _get_attr(obj, key, default=None):
if isinstance(obj, dict):
return obj.get(key, default)
return getattr(obj, key, default)
class OpponentBidModel:
"""Predicts opponent bids using LightGBM with heuristic fallback.
Trains on historical auction logs and outputs estimated max opponent
bid, win probability per player, and Monte Carlo round simulations.
"""
def __init__(self, random_state: int = 42):
self.random_state = random_state
self.model = None
self.fitted = False
self.feature_names: list = []
self.rng = np.random.RandomState(random_state)
self._role_scarcity_cache: Dict[str, float] = {}
# ------------------------------------------------------------------
# Feature engineering
# ------------------------------------------------------------------
def _extract_features(
self,
players_df: pd.DataFrame,
opponent_state: OpponentState,
) -> pd.DataFrame:
"""Build feature matrix for LightGBM prediction.
Args:
players_df: DataFrame with columns [player_name, player_role,
player_projected_points, ...].
opponent_state: OpponentState describing current opponent.
Returns:
Feature DataFrame ready for model input.
"""
df = players_df.copy()
df["role_P"] = (df["player_role"] == "P").astype(float)
df["role_D"] = (df["player_role"] == "D").astype(float)
df["role_C"] = (df["player_role"] == "C").astype(float)
df["role_A"] = (df["player_role"] == "A").astype(float)
df["budget_remaining_frac"] = (
opponent_state.budget_remaining / opponent_state.initial_budget
)
for role in ["P", "D", "C", "A"]:
slots_total = opponent_state.slots_total.get(role, 1)
slots_filled = opponent_state.slots_filled.get(role, 0)
scarcity = (slots_total - slots_filled) / slots_total
df[f"scarcity_{role}"] = scarcity
role_map = {"P": 0, "D": 1, "C": 2, "A": 3}
df["role_code"] = df["player_role"].map(role_map)
if "player_projected_points" in df.columns:
df["points_sq"] = df["player_projected_points"] ** 2
df["points_log"] = np.log1p(df["player_projected_points"].clip(lower=0))
df["slots_needed_total"] = sum(
opponent_state.slots_total.get(r, 0) - opponent_state.slots_filled.get(r, 0)
for r in ["P", "D", "C", "A"]
)
df["slots_needed_total"] = df["slots_needed_total"].clip(lower=1)
df["round_number"] = opponent_state.round_number
df["urgency"] = 1.0 - (opponent_state.budget_remaining / opponent_state.initial_budget)
df["aggression"] = opponent_state.aggression_factor
self.feature_names = [
"player_projected_points",
"role_P",
"role_D",
"role_C",
"role_A",
"role_code",
"budget_remaining_frac",
"scarcity_P",
"scarcity_D",
"scarcity_C",
"scarcity_A",
"points_sq",
"points_log",
"slots_needed_total",
"round_number",
"urgency",
"aggression",
]
for col in self.feature_names:
if col not in df.columns:
df[col] = 0.0
return df[self.feature_names]
# ------------------------------------------------------------------
# Training
# ------------------------------------------------------------------
def fit(self, auction_logs: pd.DataFrame):
"""Train LightGBM regressor on historical auction logs.
Args:
auction_logs: DataFrame with columns [player_name, player_role,
player_projected_points, opponent_budget_remaining,
opponent_slots_remaining, role_needed_count,
round_number, winning_bid].
"""
required_cols = [
"player_role", "player_projected_points",
"winning_bid",
]
for col in required_cols:
if col not in auction_logs.columns:
raise ValueError(f"Missing required column: '{col}' in auction_logs")
df = auction_logs.dropna(subset=required_cols).copy()
if len(df) < 20:
logger.warning(
f"Only {len(df)} auction records. Not enough to fit LightGBM. "
"Using heuristic fallback."
)
self.fitted = False
return
try:
import lightgbm as lgb
except ImportError:
logger.warning("LightGBM not available. Using heuristic fallback.")
self.fitted = False
return
dummy_state = OpponentState(
budget_remaining=df.get("opponent_budget_remaining", 500),
initial_budget=500,
slots_filled={"P": 0, "D": 0, "C": 0, "A": 0},
slots_total={"P": 3, "D": 8, "C": 8, "A": 6},
round_number=1,
aggression_factor=1.0,
)
df = df.rename(columns={
"player_role": "player_role",
"player_projected_points": "player_projected_points",
})
dummy_df = df[["player_role", "player_projected_points"]].copy()
dummy_df.columns = ["player_role", "player_projected_points"]
X = self._extract_features(dummy_df, dummy_state)
y = df["winning_bid"].astype(float)
y_min, y_max = y.min(), y.max()
self.model = lgb.LGBMRegressor(
n_estimators=100,
max_depth=6,
learning_rate=0.05,
num_leaves=31,
min_child_samples=10,
subsample=0.8,
colsample_bytree=0.8,
random_state=self.random_state,
verbose=-1,
)
self.model.fit(X, y)
self.fitted = True
self._y_min = y_min
self._y_max = y_max
logger.info(
f"OpponentBidModel trained on {len(df)} records. "
f"Target range: [{y_min:.0f}, {y_max:.0f}]"
)
# ------------------------------------------------------------------
# Prediction
# ------------------------------------------------------------------
def predict_opponent_bids(
self,
players_df: pd.DataFrame,
opponent_state: OpponentState,
) -> pd.Series:
"""Predict max opponent bid for each player.
Args:
players_df: DataFrame with player info.
opponent_state: current opponent state.
Returns:
Series of predicted max opponent bids (index = player index).
"""
if not self.fitted or self.model is None:
bids = self._heuristic_bid(players_df, opponent_state)
return pd.Series(bids, index=players_df.index)
X = self._extract_features(players_df, opponent_state)
predictions = self.model.predict(X)
predictions = np.clip(predictions, 1, _get_attr(opponent_state, "budget_remaining", 500))
return pd.Series(predictions, index=players_df.index)
def predict_p_acquire(
self,
players_df: pd.DataFrame,
my_bids: np.ndarray,
opponent_state: OpponentState,
temperature: float = 0.1,
) -> np.ndarray:
"""Probability I acquire each player given my bids vs opponent.
Uses sigmoid: P = 1 / (1 + exp(-(my_bid - opp_bid) / temperature)).
Args:
players_df: DataFrame with player info.
my_bids: array of my bid amounts per player.
opponent_state: current opponent state.
temperature: softmax temperature (lower = sharper).
Returns:
Array of acquisition probabilities per player.
"""
opp_bids = self.predict_opponent_bids(players_df, opponent_state).values
margin = np.array(my_bids, dtype=float) - opp_bids
scaled_temp = max(temperature * max(opp_bids.max(), 1), 0.01)
probabilities = 1.0 / (1.0 + np.exp(-margin / scaled_temp))
return np.clip(probabilities, 0.01, 0.99)
# ------------------------------------------------------------------
# Monte Carlo round simulation
# ------------------------------------------------------------------
def simulate_live_round(
self,
available_players: pd.DataFrame,
my_budget: float,
opponent_state: OpponentState,
n_sims: int = 1000,
) -> dict:
"""Monte Carlo simulation of a live auction round.
Args:
available_players: DataFrame of players up for bidding this round.
my_budget: my remaining budget.
opponent_state: opponent's current state.
n_sims: number of simulation runs.
Returns:
dict with expected_players_acquired, expected_cost, value_matrix.
"""
opp_bids = self.predict_opponent_bids(available_players, opponent_state).values
n_players = len(available_players)
players_acquired = np.zeros(n_sims, dtype=int)
total_cost = np.zeros(n_sims, dtype=float)
value_matrix = np.zeros((n_sims, n_players), dtype=float)
for sim_idx in range(n_sims):
my_budget_left = my_budget
acquired = 0
cost = 0.0
for p_idx in range(n_players):
opp_bid = opp_bids[p_idx] + self.rng.normal(0, max(opp_bids[p_idx] * 0.15, 1))
opp_bid = max(opp_bid, 1)
my_bid = self._heuristic_bid_single(
available_players.iloc[p_idx], opponent_state
)
my_bid = min(my_bid, my_budget_left)
if my_bid > opp_bid:
acquired += 1
cost += my_bid
my_budget_left -= my_bid
value_matrix[sim_idx, p_idx] = 1.0
players_acquired[sim_idx] = acquired
total_cost[sim_idx] = cost
return {
"expected_players_acquired": float(np.mean(players_acquired)),
"expected_cost": float(np.mean(total_cost)),
"cost_std": float(np.std(total_cost)),
"acquired_std": float(np.std(players_acquired)),
"cost_percentile_25": float(np.percentile(total_cost, 25)),
"cost_percentile_50": float(np.percentile(total_cost, 50)),
"cost_percentile_75": float(np.percentile(total_cost, 75)),
"acquisition_rate": float(players_acquired.mean() / n_players),
"value_matrix": value_matrix,
"n_sims": n_sims,
}
# ------------------------------------------------------------------
# Heuristic fallback
# ------------------------------------------------------------------
def _heuristic_bid_single(
self, player_row: pd.Series, opponent_state: OpponentState
) -> float:
"""Heuristic bid for a single player (scalar version)."""
role = player_row.get("player_role", None)
points = float(player_row.get("player_projected_points", 6.5))
if isinstance(role, pd.Series):
role = role.iloc[0]
scarcity = self._compute_role_scarcity(role, opponent_state)
aggression = _get_attr(opponent_state, "aggression_factor", 1.0)
budget_rem = _get_attr(opponent_state, "budget_remaining", 500.0)
budget_init = _get_attr(opponent_state, "initial_budget", 500.0) or _get_attr(opponent_state, "total_budget", 500.0)
base_bid = 0.4 * points * scarcity * aggression
bid = base_bid * (budget_rem / budget_init)
return max(bid, 1.0)
def _heuristic_bid(
self, players_df: pd.DataFrame, opponent_state: OpponentState
) -> np.ndarray:
"""Heuristic bid array for all players."""
bids = []
for _, row in players_df.iterrows():
bids.append(self._heuristic_bid_single(row, opponent_state))
return np.array(bids, dtype=float)
def _compute_role_scarcity(
self, role: Optional[str], opponent_state
) -> float:
"""Compute how scarce a role is for the opponent."""
slots_total_dict = _get_attr(opponent_state, "slots_total", {"P": 3, "D": 8, "C": 8, "A": 6})
slots_filled_dict = _get_attr(opponent_state, "slots_filled", {"P": 0, "D": 0, "C": 0, "A": 0})
if role is None or role not in slots_total_dict:
return 1.0
slots_total = slots_total_dict[role]
filled = slots_filled_dict.get(role, 0)
remaining = max(slots_total - filled, 1)
return slots_total / remaining
# ------------------------------------------------------------------
# Batch simulation with opponent model integration
# ------------------------------------------------------------------
def run_auction_simulation(
self,
player_groups: List[pd.DataFrame],
initial_budgets: List[float],
opponent_states: List[OpponentState],
n_sims: int = 500,
) -> dict:
"""Simulate multi-round auction against multiple opponents.
Args:
player_groups: list of DataFrames, one per round, with available players.
initial_budgets: my starting budget per round/concept.
opponent_states: OpponentState for each round.
n_sims: number of Monte Carlo runs.
Returns:
dict with aggregated simulation results.
"""
all_results = []
for idx, (players, budget, opp_state) in enumerate(
zip(player_groups, initial_budgets, opponent_states)
):
result = self.simulate_live_round(
players, budget, opp_state, n_sims=n_sims
)
result["round"] = idx
all_results.append(result)
total_acquired = sum(r["expected_players_acquired"] for r in all_results)
total_cost = sum(r["expected_cost"] for r in all_results)
return {
"rounds": all_results,
"total_expected_acquired": total_acquired,
"total_expected_cost": total_cost,
"n_sims": n_sims,
}
+40 -1
View File
@@ -515,8 +515,20 @@ class RLAuctionPolicy:
src = getattr(self.q_network, src_name)
setattr(self.target_network, tgt_name, src.copy())
def _resize_networks(self, new_state_dim: int):
"""Reinitialize networks when state dimension changes."""
self.q_network = QNetwork(new_state_dim, self.q_network.hidden_dim, self.action_dim)
self.target_network = QNetwork(new_state_dim, self.target_network.hidden_dim, self.action_dim)
self._hard_update_target()
def _normalize_state(self, state: np.ndarray) -> np.ndarray:
state = np.asarray(state, dtype=np.float64).ravel()
if len(state) != len(self._obs_mean):
self._obs_mean = np.zeros(len(state), dtype=np.float64)
self._obs_std = np.ones(len(state), dtype=np.float64)
self._obs_count = 0
self.state_dim = len(state)
self._resize_networks(len(state))
self._obs_count += 1
n = self._obs_count
old_mean = self._obs_mean.copy()
@@ -844,10 +856,31 @@ def step_in_env(env: AuctionEnv, action: int) -> Tuple[np.ndarray, float, bool,
class RLAuctionTrainer:
"""Convenience class for training and evaluating the RL auction agent."""
def __init__(self, model_dir: str = "models_trained"):
def __init__(
self,
player_pool: Optional[pd.DataFrame] = None,
n_opponents: int = 7,
config: Optional[AuctionConfig] = None,
model_dir: str = "models_trained",
):
self.player_pool = player_pool
self.n_opponents = n_opponents
self.config = config or AuctionConfig()
self.model_dir = Path(model_dir)
self.model_dir.mkdir(parents=True, exist_ok=True)
def train(
self,
n_episodes: int = 5000,
verbose: bool = True,
) -> RLAuctionPolicy:
"""Train RL policy on stored player pool."""
if self.player_pool is None:
raise ValueError("No player_pool provided to trainer")
env = self.prepare_training_data(self.player_pool, n_opponents=self.n_opponents, config=self.config)
policy, _ = self.train_agent(env, episodes=n_episodes, eval_interval=100)
return policy
def prepare_training_data(
self,
player_pool_df: pd.DataFrame,
@@ -963,6 +996,12 @@ class RLAuctionTrainer:
"n_successful": len(values),
}
for strategy, metrics in list(summary.items()):
summary[f"{strategy}_total_value"] = metrics["avg_value"]
summary["rl_total_value"] = summary.get("rl_agent_total_value", 0)
summary["greedy_total_value"] = summary.get("greedy_baseline_total_value", 0)
logger.info(
f"Benchmark complete: RL={summary.get('rl_agent', {}).get('avg_value', 0):.1f} pts "
f"vs Greedy={summary.get('greedy_baseline', {}).get('avg_value', 0):.1f} "