feat: add 10 new ML models for auction optimization (Phases 1-6)
Phase 1 - Quick Wins: - QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding - MinutesSurvivalModel: Weibull AFT for minutes distribution modeling Phase 2 - Adaptive Auction: - BanditAuctionSolver: Thompson Sampling for live auction bids - OpponentBidModel: Predict competitor bids via LightGBM - BudgetOptimizer: Bayesian optimization for role-level allocation Phase 3 - Deep Learning: - RLAuctionPolicy: Double DQN agent for auction strategy - SetTransformer: Team composition valuation via set-based ML Phase 4 - Probabilistic: - BayesianPlayerModel: Hierarchical pooling for rookie uncertainty - ConformalPredictor: Calibrated prediction intervals Phase 5 - Chemistry & Form: - PlayerChemistryGAT: Graph attention network for player synergies - PlayerFormModel: Hawkes process for form momentum Phase 6 - Causal: - TransferCausalModel: Causal forest for transfer effects - AuctionEffectAnalyzer: Bid adjustment from causal analysis 81 tests passing
This commit is contained in:
@@ -1 +1,14 @@
|
||||
"""Optimization modules for auction and lineup selection."""
|
||||
|
||||
from .auction_solver import AuctionSolver, AuctionConfig, PlayerValuation
|
||||
from .lineup_solver import LineupSolver, LineupConstraints, PlayerScore, MCTSNode
|
||||
from .opponent_model import OpponentModel
|
||||
from .transfer_analyzer import TransferAnalyzer
|
||||
|
||||
# Phase 2: Bandit, opponent bidding, budget optimization
|
||||
from .bandit_auction import BanditAuctionSolver
|
||||
from .opponent_bidding_model import OpponentBidModel
|
||||
from .budget_optimizer import BudgetOptimizer
|
||||
|
||||
# Phase 3: Reinforcement learning auction agent
|
||||
from .rl_auction_agent import AuctionEnv, RLAuctionPolicy, RLAuctionTrainer, QNetwork
|
||||
|
||||
@@ -0,0 +1,359 @@
|
||||
"""Contextual Multi-Armed Bandit for live auction bidding decisions.
|
||||
|
||||
Uses Thompson Sampling with Beta-distributed posteriors over discrete bid levels.
|
||||
Falls back to UCB when exploration depth is insufficient.
|
||||
Integrates with AuctionConfig from auction_solver.py.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from scipy.stats import beta as beta_dist
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Default bid arms as fractions of total budget
|
||||
DEFAULT_BID_ARMS = np.array(
|
||||
[0.0, 0.005, 0.01, 0.02, 0.03, 0.05, 0.08, 0.12, 0.18, 0.25],
|
||||
dtype=np.float64,
|
||||
)
|
||||
|
||||
ROLES = ["P", "D", "C", "A"]
|
||||
ROLE_SCARCITY = {"P": 3, "D": 8, "C": 8, "A": 6}
|
||||
ROLE_POOL_SIZE = {"P": 4, "D": 22, "C": 24, "A": 12}
|
||||
|
||||
|
||||
@dataclass
|
||||
class BanditArmState:
|
||||
alpha: float = 1.0
|
||||
beta: float = 1.0
|
||||
trials: int = 0
|
||||
wins: float = 0.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class AuctionState:
|
||||
budget_remaining: float = 500.0
|
||||
total_budget: float = 500.0
|
||||
slots_filled: Dict[str, int] = field(default_factory=lambda: {"P": 0, "D": 0, "C": 0, "A": 0})
|
||||
slots_total: Dict[str, int] = field(default_factory=lambda: {"P": 3, "D": 8, "C": 8, "A": 6})
|
||||
round_number: int = 1
|
||||
opponent_budgets: List[float] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlayerContext:
|
||||
name: str
|
||||
role: str
|
||||
projected_points: float
|
||||
role_scarcity: float
|
||||
value_over_replacement: float
|
||||
budget_remaining_fraction: float
|
||||
slots_remaining_in_role: int
|
||||
round_number: int
|
||||
opponent_budget_avg: float
|
||||
|
||||
|
||||
def _get_attr(obj, key, default=None):
|
||||
"""Get attribute or dict key from an object."""
|
||||
if isinstance(obj, dict):
|
||||
return obj.get(key, default)
|
||||
return getattr(obj, key, default)
|
||||
|
||||
|
||||
class BanditAuctionSolver:
|
||||
"""Thompson Sampling bandit for live auction bid selection.
|
||||
|
||||
Arms are discrete bid fractions. Each (role, scarcity_level) maintains
|
||||
independent Beta posteriors. Falls back to UCB when total observations
|
||||
for a context group are < 50.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
bid_arms: Optional[np.ndarray] = None,
|
||||
total_budget: float = 500.0,
|
||||
min_obs_for_ts: int = 50,
|
||||
ucb_exploration: float = 1.414,
|
||||
config=None,
|
||||
):
|
||||
if config is not None:
|
||||
from .auction_solver import AuctionConfig
|
||||
total_budget = config.total_budget
|
||||
self.bid_arms = bid_arms if bid_arms is not None else DEFAULT_BID_ARMS
|
||||
self.n_arms = len(self.bid_arms)
|
||||
self.total_budget = total_budget
|
||||
self.min_obs_for_ts = min_obs_for_ts
|
||||
self.ucb_exploration = ucb_exploration
|
||||
self.bid_fractions = self.bid_arms
|
||||
|
||||
self.posteriors: Dict[Tuple[str, int], List[BanditArmState]] = {}
|
||||
for role in ROLES:
|
||||
self._ensure_posteriors(role, 1)
|
||||
|
||||
def _context_key(self, role: str, scarcity_level: int) -> Tuple[str, int]:
|
||||
return (role, scarcity_level)
|
||||
|
||||
def _ensure_posteriors(self, role: str, n_slots_remaining: int):
|
||||
scarcity = max(1, n_slots_remaining)
|
||||
key = self._context_key(role, scarcity)
|
||||
if key not in self.posteriors:
|
||||
self.posteriors[key] = [
|
||||
BanditArmState(alpha=1.0, beta=1.0, trials=0, wins=0.0)
|
||||
for _ in range(self.n_arms)
|
||||
]
|
||||
logger.debug(f"Initialized bandit posteriors for role={role}, scarcity={scarcity}")
|
||||
|
||||
def _total_obs(self, key: Tuple[str, int]) -> int:
|
||||
arms = self.posteriors.get(key, [])
|
||||
return sum(a.trials for a in arms)
|
||||
|
||||
def compute_context(
|
||||
self,
|
||||
player,
|
||||
auction_state,
|
||||
pool_stats: Optional[dict] = None,
|
||||
) -> PlayerContext:
|
||||
"""Build context vector for a player given current auction state.
|
||||
|
||||
Args:
|
||||
player: dict or object with name, role, projected_points attributes.
|
||||
auction_state: current AuctionState (or dict with same keys).
|
||||
pool_stats: optional dict with role-level pool means and stds.
|
||||
|
||||
Returns:
|
||||
PlayerContext dataclass with all context features.
|
||||
"""
|
||||
role = _get_attr(player, "role")
|
||||
points = float(_get_attr(player, "projected_points", 6.5))
|
||||
name = _get_attr(player, "name", "unknown")
|
||||
|
||||
total_slots = _get_attr(auction_state, "slots_total", {}).get(role, 1)
|
||||
filled = _get_attr(auction_state, "slots_filled", {}).get(role, 0)
|
||||
slots_remaining = max(total_slots - filled, 0)
|
||||
scarcity = max(slots_remaining / max(total_slots, 1), 0.05)
|
||||
|
||||
budget_remaining = _get_attr(auction_state, "budget_remaining", 500.0)
|
||||
total_budget_attr = _get_attr(auction_state, "total_budget", 500.0)
|
||||
budget_fraction = budget_remaining / max(total_budget_attr, 1)
|
||||
|
||||
opponent_budget_avg = 0.0
|
||||
opponent_budgets = _get_attr(auction_state, "opponent_budgets", [])
|
||||
if opponent_budgets:
|
||||
opponent_budget_avg = float(np.mean(opponent_budgets))
|
||||
elif total_budget_attr > 0:
|
||||
opponent_budget_avg = total_budget_attr * 0.6
|
||||
|
||||
if pool_stats and role in pool_stats:
|
||||
role_mean = pool_stats[role].get("mean", 0.0)
|
||||
role_std = pool_stats[role].get("std", 1.0)
|
||||
points_z = (points - role_mean) / max(role_std, 0.01) if role_std > 0 else 0.0
|
||||
else:
|
||||
points_z = points / 15.0
|
||||
|
||||
vor = max(points - 6.5, 0.0)
|
||||
|
||||
return PlayerContext(
|
||||
name=name,
|
||||
role=role,
|
||||
projected_points=points,
|
||||
role_scarcity=scarcity,
|
||||
value_over_replacement=vor,
|
||||
budget_remaining_fraction=budget_fraction,
|
||||
slots_remaining_in_role=slots_remaining,
|
||||
round_number=_get_attr(auction_state, "round_number", 1),
|
||||
opponent_budget_avg=opponent_budget_avg,
|
||||
)
|
||||
|
||||
def select_bid(
|
||||
self,
|
||||
player,
|
||||
auction_state,
|
||||
pool_stats: Optional[dict] = None,
|
||||
) -> Tuple[int, float]:
|
||||
"""Select bid arm using Thompson Sampling (or UCB fallback).
|
||||
|
||||
Returns:
|
||||
(arm_index, bid_amount_in_credits)
|
||||
"""
|
||||
ctx = self.compute_context(player, auction_state, pool_stats)
|
||||
role = ctx.role
|
||||
n_slots = ctx.slots_remaining_in_role
|
||||
|
||||
self._ensure_posteriors(role, n_slots)
|
||||
key = self._context_key(role, n_slots)
|
||||
arms = self.posteriors[key]
|
||||
total_obs = sum(a.trials for a in arms)
|
||||
|
||||
if total_obs >= self.min_obs_for_ts:
|
||||
samples = [float(np.random.beta(a.alpha, max(a.beta, 0.01))) for a in arms]
|
||||
arm_idx = int(np.argmax(samples))
|
||||
logger.debug(
|
||||
f"Thompson Sampling: role={role}, arms_sampled={samples[:5]}..., "
|
||||
f"selected_arm={arm_idx}"
|
||||
)
|
||||
else:
|
||||
values = []
|
||||
for _i, arm in enumerate(arms):
|
||||
if arm.trials == 0:
|
||||
values.append(float("inf"))
|
||||
else:
|
||||
mean = arm.wins / arm.trials
|
||||
bonus = self.ucb_exploration * np.sqrt(
|
||||
np.log(max(total_obs, 1)) / arm.trials
|
||||
)
|
||||
values.append(mean + bonus)
|
||||
arm_idx = int(np.argmax(values))
|
||||
logger.debug(
|
||||
f"UCB fallback: role={role}, total_obs={total_obs}, "
|
||||
f"selected_arm={arm_idx}"
|
||||
)
|
||||
|
||||
bid_amount = round(self.bid_arms[arm_idx] * self.total_budget)
|
||||
|
||||
budget_rem = _get_attr(auction_state, "budget_remaining", self.total_budget)
|
||||
if bid_amount > budget_rem:
|
||||
bid_amount = budget_rem
|
||||
arm_idx = int(np.argmin(np.abs(self.bid_arms * self.total_budget - bid_amount)))
|
||||
|
||||
return arm_idx, bid_amount
|
||||
|
||||
def update(self, arm_idx: int, reward: float, player_role: str):
|
||||
"""Update Beta posterior for the selected arm.
|
||||
|
||||
Reward should be a normalized value: (player_season_value - cost) scaled.
|
||||
|
||||
Args:
|
||||
arm_idx: index of the selected arm.
|
||||
reward: normalized reward signal (higher = better purchase).
|
||||
player_role: role of the purchased player.
|
||||
"""
|
||||
reward_clipped = max(0.0, min(1.0, reward))
|
||||
win = 1.0 if reward > 0 else 0.0
|
||||
|
||||
for key, arms in self.posteriors.items():
|
||||
role, _scarcity = key
|
||||
if role == player_role:
|
||||
if arm_idx < len(arms):
|
||||
arm = arms[arm_idx]
|
||||
arm.trials += 1
|
||||
arm.wins += win
|
||||
arm.alpha += reward_clipped
|
||||
arm.beta += (1.0 - reward_clipped)
|
||||
logger.debug(
|
||||
f"Updated arm {arm_idx} for {player_role}: "
|
||||
f"trials={arm.trials}, alpha={arm.alpha:.2f}, beta={arm.beta:.2f}"
|
||||
)
|
||||
for key, arms in self.posteriors.items():
|
||||
role, _scarcity = key
|
||||
if role == player_role:
|
||||
for i, arm in enumerate(arms):
|
||||
if i == arm_idx:
|
||||
continue
|
||||
arm.beta = max(arm.beta, 1.001)
|
||||
|
||||
def get_arm_stats(self) -> Dict[str, dict]:
|
||||
"""Return arm statistics for analysis.
|
||||
|
||||
Returns:
|
||||
dict mapping "role/scarcity/arm_idx" -> stats dict.
|
||||
"""
|
||||
stats = {}
|
||||
agg = {}
|
||||
for (role, scarcity), arms in self.posteriors.items():
|
||||
for idx, arm in enumerate(arms):
|
||||
label = f"{role}/scarcity={scarcity}/arm={idx}"
|
||||
stats[label] = {
|
||||
"trials": arm.trials,
|
||||
"wins": arm.wins,
|
||||
"alpha": arm.alpha,
|
||||
"beta": arm.beta,
|
||||
"win_rate": arm.wins / max(arm.trials, 1),
|
||||
"bid_fraction": float(self.bid_arms[idx]),
|
||||
"bid_amount": float(self.bid_arms[idx] * self.total_budget),
|
||||
}
|
||||
if idx not in agg:
|
||||
agg[idx] = {"trials": 0, "wins": 0.0, "alpha": 0.0, "beta": 0.0}
|
||||
agg[idx]["trials"] += arm.trials
|
||||
agg[idx]["wins"] += arm.wins
|
||||
agg[idx]["alpha"] += arm.alpha
|
||||
agg[idx]["beta"] += arm.beta
|
||||
for idx, a in agg.items():
|
||||
stats[idx] = {
|
||||
"trials": a["trials"],
|
||||
"wins": a["wins"],
|
||||
"alpha": a["alpha"],
|
||||
"beta": a["beta"],
|
||||
"win_rate": a["wins"] / max(a["trials"], 1),
|
||||
"bid_fraction": float(self.bid_arms[idx]),
|
||||
"bid_amount": float(self.bid_arms[idx] * self.total_budget),
|
||||
}
|
||||
return stats
|
||||
|
||||
def exploration_bonus(self, player_role: str, player_points: float = 0.0) -> float:
|
||||
"""Compute exploration bonus for new/unknown player types.
|
||||
|
||||
Higher bonus when the bandit has limited experience with a role.
|
||||
Encourages exploring sleeper players.
|
||||
|
||||
Returns:
|
||||
Recommended extra bid amount in credits.
|
||||
"""
|
||||
total_obs = 0
|
||||
for (role, _scarcity), arms in self.posteriors.items():
|
||||
if role == player_role:
|
||||
total_obs += sum(a.trials for a in arms)
|
||||
|
||||
if total_obs == 0:
|
||||
bonus = self.total_budget * 0.04
|
||||
elif total_obs < 20:
|
||||
bonus = self.total_budget * 0.025
|
||||
elif total_obs < 50:
|
||||
bonus = self.total_budget * 0.01
|
||||
else:
|
||||
bonus = 0.0
|
||||
|
||||
if bonus > 0:
|
||||
base = max(player_points * 0.5, 0)
|
||||
bonus += base * (1.0 / max(total_obs, 1)) * 20
|
||||
|
||||
logger.debug(f"Exploration bonus for {player_role}: {bonus:.1f} (obs={total_obs})")
|
||||
return bonus
|
||||
|
||||
def recommend_bid_summary(
|
||||
self,
|
||||
player,
|
||||
auction_state,
|
||||
pool_stats: Optional[dict] = None,
|
||||
) -> dict:
|
||||
"""Full bidding recommendation for a player.
|
||||
|
||||
Returns a dict with arm index, bid amount, context, exploration bonus,
|
||||
and total recommended bid.
|
||||
"""
|
||||
arm_idx, bid_amount = self.select_bid(player, auction_state, pool_stats)
|
||||
ctx = self.compute_context(player, auction_state, pool_stats)
|
||||
bonus = self.exploration_bonus(ctx.role, ctx.projected_points)
|
||||
|
||||
return {
|
||||
"player": _get_attr(player, "name", "unknown"),
|
||||
"role": ctx.role,
|
||||
"arm_index": arm_idx,
|
||||
"base_bid": bid_amount,
|
||||
"exploration_bonus": round(bonus),
|
||||
"total_bid": round(bid_amount + bonus),
|
||||
"recommended_bid": round(bid_amount + bonus),
|
||||
"context": {
|
||||
"points_zscore": round(
|
||||
(ctx.projected_points - 6.5) / 2.0, 2
|
||||
),
|
||||
"role_scarcity": round(ctx.role_scarcity, 3),
|
||||
"budget_remaining_frac": round(ctx.budget_remaining_fraction, 3),
|
||||
"slots_remaining": ctx.slots_remaining_in_role,
|
||||
"round": ctx.round_number,
|
||||
"vor": round(ctx.value_over_replacement, 1),
|
||||
},
|
||||
}
|
||||
@@ -0,0 +1,538 @@
|
||||
"""Bayesian Optimization for role-level budget allocation.
|
||||
|
||||
Uses Gaussian Process regression to find optimal budget distribution
|
||||
across roles (P, D, C, A) that maximizes total projected team value.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from scipy.optimize import minimize
|
||||
from scipy.special import softmax
|
||||
|
||||
from src.optimization.auction_solver import AuctionConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_ROSTER_QUOTAS = {"P": 3, "D": 8, "C": 8, "A": 6}
|
||||
|
||||
|
||||
@dataclass
|
||||
class RoleBudgetResult:
|
||||
allocation: Dict[str, float]
|
||||
total_value: float
|
||||
role_values: Dict[str, float]
|
||||
value_curves: Dict[str, Tuple[np.ndarray, np.ndarray]]
|
||||
|
||||
|
||||
class BudgetOptimizer:
|
||||
"""Bayesian Optimization for budget allocation across roster roles.
|
||||
|
||||
Finds the split of total_budget across P/D/C/A that yields the
|
||||
highest possible team points via greedy fill within each role's budget.
|
||||
Supports mid-auction adaptive rebalancing.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
total_budget: float = 500.0,
|
||||
roster_quotas: Optional[Dict[str, int]] = None,
|
||||
auction_config: Optional[AuctionConfig] = None,
|
||||
random_state: int = 42,
|
||||
):
|
||||
self.total_budget = total_budget
|
||||
self.roster_quotas = roster_quotas or dict(DEFAULT_ROSTER_QUOTAS)
|
||||
self.auction_config = auction_config or AuctionConfig()
|
||||
self.random_state = random_state
|
||||
self.rng = np.random.RandomState(random_state)
|
||||
self._last_allocation: Optional[Dict[str, float]] = None
|
||||
self._last_value: float = 0.0
|
||||
self._optimization_history: List[dict] = []
|
||||
self._gk_available = False
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Core optimization
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def optimize(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
n_calls: int = 50,
|
||||
use_skopt: bool = True,
|
||||
) -> Dict[str, float]:
|
||||
"""Optimize budget allocation across roles.
|
||||
|
||||
Uses scikit-optimize GaussianProcessRegressor if available,
|
||||
otherwise simplex-based local search.
|
||||
|
||||
Args:
|
||||
player_pool_df: DataFrame with columns [name, role, projected_points].
|
||||
n_calls: number of GP evaluations.
|
||||
use_skopt: attempt Gaussian Process optimization.
|
||||
|
||||
Returns:
|
||||
Dict mapping role -> recommended budget amount.
|
||||
"""
|
||||
if "role" not in player_pool_df.columns or "projected_points" not in player_pool_df.columns:
|
||||
raise ValueError("player_pool_df must have 'role' and 'projected_points' columns")
|
||||
|
||||
roles = list(self.roster_quotas.keys())
|
||||
n_roles = len(roles)
|
||||
|
||||
if use_skopt:
|
||||
try:
|
||||
return self._optimize_gp(player_pool_df, roles, n_calls)
|
||||
except ImportError:
|
||||
logger.info("scikit-optimize not installed. Using local search.")
|
||||
except Exception as exc:
|
||||
logger.warning(f"GP optimization failed: {exc}. Using local search.")
|
||||
|
||||
return self._optimize_local(player_pool_df, roles, n_calls)
|
||||
|
||||
def _optimize_gp(
|
||||
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
|
||||
) -> Dict[str, float]:
|
||||
"""Gaussian Process-based budget optimization."""
|
||||
from skopt import gp_minimize
|
||||
from skopt.space import Space
|
||||
from skopt.learning import GaussianProcessRegressor
|
||||
|
||||
n_roles = len(roles)
|
||||
space = Space([(0.01, 0.70) for _ in range(n_roles)])
|
||||
|
||||
def objective_wrapper(fractions):
|
||||
fractions = np.array(fractions, dtype=float)
|
||||
fractions = self._normalize_fractions(fractions)
|
||||
value = self._objective(fractions, roles, player_pool_df)
|
||||
self._optimization_history.append({
|
||||
"fractions": fractions.tolist(),
|
||||
"value": value,
|
||||
})
|
||||
return -value
|
||||
|
||||
def params_to_fractions(params):
|
||||
return self._normalize_fractions(np.array(params, dtype=float))
|
||||
|
||||
result = gp_minimize(
|
||||
objective_wrapper,
|
||||
space,
|
||||
n_calls=n_calls,
|
||||
random_state=self.random_state,
|
||||
n_initial_points=max(10, n_calls // 5),
|
||||
verbose=False,
|
||||
n_jobs=-1,
|
||||
)
|
||||
|
||||
best_fractions = params_to_fractions(result.x)
|
||||
self._last_allocation = self._fractions_to_allocation(best_fractions, roles)
|
||||
self._last_value = -result.fun
|
||||
|
||||
logger.info(
|
||||
f"GP Budget optimization complete: {self._last_allocation} "
|
||||
f"=> value={self._last_value:.1f}"
|
||||
)
|
||||
|
||||
return self._last_allocation
|
||||
|
||||
def _optimize_local(
|
||||
self, player_pool_df: pd.DataFrame, roles: list, n_calls: int
|
||||
) -> Dict[str, float]:
|
||||
"""Local search optimization using Nelder-Mead simplex."""
|
||||
n_roles = len(roles)
|
||||
|
||||
best_allocation = None
|
||||
best_value = -float("inf")
|
||||
|
||||
for restart in range(max(n_calls // 10, 1)):
|
||||
x0 = self._random_allocation(n_roles)
|
||||
|
||||
def simplex_objective(fractions):
|
||||
fractions = self._normalize_fractions(np.array(fractions, dtype=float))
|
||||
value = self._objective(fractions, roles, player_pool_df)
|
||||
self._optimization_history.append({
|
||||
"fractions": fractions.tolist(),
|
||||
"value": value,
|
||||
})
|
||||
return -value
|
||||
|
||||
res = minimize(
|
||||
simplex_objective,
|
||||
x0,
|
||||
method="Nelder-Mead",
|
||||
options={"maxiter": max(n_calls // 3, 20), "xatol": 1e-3, "fatol": 1e-3},
|
||||
)
|
||||
|
||||
fractions = self._normalize_fractions(np.array(res.x, dtype=float))
|
||||
value = -res.fun
|
||||
|
||||
if value > best_value:
|
||||
best_value = value
|
||||
best_allocation = fractions
|
||||
|
||||
if best_allocation is None:
|
||||
best_allocation = self._proportional_allocation(player_pool_df, roles)
|
||||
|
||||
self._last_allocation = self._fractions_to_allocation(best_allocation, roles)
|
||||
self._last_value = best_value
|
||||
|
||||
logger.info(
|
||||
f"Local optimization complete: {self._last_allocation} "
|
||||
f"=> value={self._last_value:.1f}"
|
||||
)
|
||||
|
||||
return self._last_allocation
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Adaptive rebalancing
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def optimize_adaptive(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
remaining_slots: Dict[str, int],
|
||||
spent_per_role: Dict[str, float],
|
||||
n_calls: int = 30,
|
||||
) -> Dict[str, float]:
|
||||
"""Mid-auction rebalancing — optimize remaining budget for unfilled slots.
|
||||
|
||||
Args:
|
||||
player_pool_df: remaining available players pool.
|
||||
remaining_slots: dict of role -> slots still needed.
|
||||
spent_per_role: dict of role -> credits already spent.
|
||||
n_calls: GP evaluation budget.
|
||||
|
||||
Returns:
|
||||
Dict mapping role -> recommended budget for remaining slots.
|
||||
"""
|
||||
remaining_budget = self.total_budget - sum(spent_per_role.values())
|
||||
remaining_budget = max(1.0, remaining_budget)
|
||||
|
||||
if sum(remaining_slots.values()) == 0:
|
||||
logger.info("All slots filled. No budget to allocate.")
|
||||
return {r: 0.0 for r in self.roster_quotas}
|
||||
|
||||
saved_quotas = self.roster_quotas
|
||||
saved_total = self.total_budget
|
||||
self.roster_quotas = dict(remaining_slots)
|
||||
self.total_budget = remaining_budget
|
||||
|
||||
available = player_pool_df[
|
||||
player_pool_df["role"].isin(
|
||||
[r for r, s in remaining_slots.items() if s > 0]
|
||||
)
|
||||
]
|
||||
|
||||
if len(available) == 0:
|
||||
logger.warning("No available players for remaining slots.")
|
||||
self.roster_quotas = saved_quotas
|
||||
self.total_budget = saved_total
|
||||
return {r: 0.0 for r in saved_quotas}
|
||||
|
||||
allocation = self.optimize(
|
||||
available,
|
||||
n_calls=n_calls,
|
||||
use_skopt=True,
|
||||
)
|
||||
|
||||
self.roster_quotas = saved_quotas
|
||||
self.total_budget = saved_total
|
||||
|
||||
result = {}
|
||||
for role in saved_quotas:
|
||||
result[role] = allocation.get(role, 0.0)
|
||||
return result
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Objective function
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _objective(
|
||||
self,
|
||||
fractions: np.ndarray,
|
||||
roles: list,
|
||||
player_pool_df: pd.DataFrame,
|
||||
) -> float:
|
||||
"""Simulate greedy fill within role budgets; return total projected points.
|
||||
|
||||
For each role, pick the best players by projected_points until the
|
||||
role budget or slot quota is exhausted.
|
||||
"""
|
||||
total_value = 0.0
|
||||
|
||||
for i, role in enumerate(roles):
|
||||
budget = fractions[i] * self.total_budget
|
||||
slots = self.roster_quotas.get(role, 0)
|
||||
role_players = player_pool_df[player_pool_df["role"] == role].copy()
|
||||
|
||||
if len(role_players) == 0 or slots == 0:
|
||||
continue
|
||||
|
||||
role_players = role_players.sort_values(
|
||||
"projected_points", ascending=False
|
||||
)
|
||||
|
||||
total_cost = 0.0
|
||||
filled = 0
|
||||
|
||||
for _, player in role_players.iterrows():
|
||||
points = float(player["projected_points"])
|
||||
estimated_price = self._estimate_price_simple(points, budget, role)
|
||||
|
||||
if total_cost + estimated_price > budget:
|
||||
continue
|
||||
|
||||
total_cost += estimated_price
|
||||
total_value += points
|
||||
filled += 1
|
||||
|
||||
if filled >= slots:
|
||||
break
|
||||
|
||||
return total_value
|
||||
|
||||
def _estimate_price_simple(
|
||||
self, projected_points: float, role_budget: float, role: str
|
||||
) -> float:
|
||||
"""Simple price estimate: points * role_factor clamped within budget."""
|
||||
role_factor = {"P": 4.0, "D": 2.5, "C": 3.0, "A": 4.5}.get(role, 3.0)
|
||||
price = projected_points * role_factor
|
||||
max_price = role_budget * 0.50
|
||||
return min(price, max_price, role_budget)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Allocation access and visualization data
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def get_allocation(self) -> Dict[str, float]:
|
||||
"""Return last computed allocation (role -> budget amount)."""
|
||||
if self._last_allocation is None:
|
||||
return {r: self.total_budget / len(self.roster_quotas) for r in self.roster_quotas}
|
||||
return dict(self._last_allocation)
|
||||
|
||||
def get_role_value_curves(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
n_points: int = 20,
|
||||
) -> Dict[str, Tuple[np.ndarray, np.ndarray]]:
|
||||
"""Compute diminishing returns curves: budget vs. expected points per role.
|
||||
|
||||
Returns:
|
||||
dict role -> (budget_array, value_array).
|
||||
"""
|
||||
curves = {}
|
||||
budget_step = self.total_budget / n_points
|
||||
|
||||
for role, slots in self.roster_quotas.items():
|
||||
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
|
||||
"projected_points", ascending=False
|
||||
)
|
||||
|
||||
budgets = np.linspace(0, self.total_budget, n_points)
|
||||
values = np.zeros(n_points)
|
||||
|
||||
for i, budget_limit in enumerate(budgets):
|
||||
total_cost = 0.0
|
||||
total_value = 0.0
|
||||
filled = 0
|
||||
|
||||
for _, player in role_players.iterrows():
|
||||
points = float(player["projected_points"])
|
||||
price = self._estimate_price_simple(points, budget_limit, role)
|
||||
|
||||
if total_cost + price > budget_limit:
|
||||
continue
|
||||
|
||||
total_cost += price
|
||||
total_value += points
|
||||
filled += 1
|
||||
|
||||
if filled >= slots:
|
||||
break
|
||||
|
||||
values[i] = total_value
|
||||
|
||||
curves[role] = (budgets.copy(), values.copy())
|
||||
|
||||
return curves
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Fallback: proportional allocation
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _proportional_allocation(
|
||||
self, player_pool_df: pd.DataFrame, roles: list
|
||||
) -> np.ndarray:
|
||||
"""Allocate budget proportional to (points_variance * slots) per role."""
|
||||
weights = np.zeros(len(roles))
|
||||
|
||||
for i, role in enumerate(roles):
|
||||
role_players = player_pool_df[player_pool_df["role"] == role]
|
||||
if len(role_players) > 1:
|
||||
variance = role_players["projected_points"].var()
|
||||
else:
|
||||
variance = 1.0
|
||||
slots = self.roster_quotas.get(role, 1)
|
||||
weights[i] = variance * slots
|
||||
|
||||
weight_sum = weights.sum()
|
||||
if weight_sum <= 0:
|
||||
return np.ones(len(roles)) / len(roles)
|
||||
|
||||
fractions = weights / weight_sum
|
||||
fractions = np.clip(fractions, 0.02, 0.70)
|
||||
fractions = fractions / fractions.sum()
|
||||
|
||||
logger.info(
|
||||
f"Proportional allocation (fallback): "
|
||||
f"{dict(zip(roles, fractions.round(3)))}"
|
||||
)
|
||||
return fractions
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Utilities
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _normalize_fractions(self, fractions: np.ndarray) -> np.ndarray:
|
||||
"""Normalize fractions to sum to 1.0 with minimum per role."""
|
||||
fractions = np.clip(fractions, 0.01, 0.70)
|
||||
total = fractions.sum()
|
||||
if total <= 0:
|
||||
return np.ones_like(fractions) / len(fractions)
|
||||
return fractions / total
|
||||
|
||||
def _fractions_to_allocation(
|
||||
self, fractions: np.ndarray, roles: list
|
||||
) -> Dict[str, float]:
|
||||
"""Convert fractions to absolute budget per role."""
|
||||
return {
|
||||
role: round(float(fractions[i] * self.total_budget), 1)
|
||||
for i, role in enumerate(roles)
|
||||
}
|
||||
|
||||
def _random_allocation(self, n_roles: int) -> np.ndarray:
|
||||
"""Generate a random allocation via Dirichlet."""
|
||||
alpha = np.ones(n_roles) * 2.0
|
||||
return self.rng.dirichlet(alpha)
|
||||
|
||||
def get_optimization_trace(self) -> pd.DataFrame:
|
||||
"""Return DataFrame of all evaluated allocations during optimization."""
|
||||
if not self._optimization_history:
|
||||
return pd.DataFrame()
|
||||
return pd.DataFrame(self._optimization_history)
|
||||
|
||||
def estimate_team_composition(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
allocation: Optional[Dict[str, float]] = None,
|
||||
) -> pd.DataFrame:
|
||||
"""Given final allocation, return the recommended player selections.
|
||||
|
||||
Returns:
|
||||
DataFrame with selected players, their estimated costs, and value.
|
||||
"""
|
||||
roles = list(self.roster_quotas.keys())
|
||||
alloc = allocation or self._last_allocation
|
||||
if alloc is None:
|
||||
alloc = self._proportional_allocation_as_dict(roles)
|
||||
|
||||
selections = []
|
||||
|
||||
for role in roles:
|
||||
budget = alloc.get(role, 0.0)
|
||||
slots = self.roster_quotas.get(role, 0)
|
||||
role_players = player_pool_df[player_pool_df["role"] == role].sort_values(
|
||||
"projected_points", ascending=False
|
||||
)
|
||||
|
||||
total_cost = 0.0
|
||||
filled = 0
|
||||
|
||||
for _, player in role_players.iterrows():
|
||||
points = float(player["projected_points"])
|
||||
price = self._estimate_price_simple(points, budget, role)
|
||||
|
||||
if total_cost + price > budget:
|
||||
continue
|
||||
|
||||
total_cost += price
|
||||
name = player.get("name", player.get("player_name", "unknown"))
|
||||
selections.append({
|
||||
"player": name,
|
||||
"role": role,
|
||||
"projected_points": points,
|
||||
"estimated_cost": round(price, 1),
|
||||
"value_ratio": round(points / max(price, 1), 3),
|
||||
})
|
||||
filled += 1
|
||||
|
||||
if filled >= slots:
|
||||
break
|
||||
|
||||
return pd.DataFrame(selections)
|
||||
|
||||
def _proportional_allocation_as_dict(self, roles: list) -> Dict[str, float]:
|
||||
fractions = self._proportional_allocation(
|
||||
pd.DataFrame(columns=["role", "projected_points"]), roles
|
||||
)
|
||||
return {roles[i]: round(float(fractions[i] * self.total_budget), 1) for i in range(len(roles))}
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Sensitivity analysis
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def sensitivity_analysis(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
role: Optional[str] = None,
|
||||
delta_pct: float = 0.05,
|
||||
n_steps: int = 11,
|
||||
n_calls: int = 50,
|
||||
) -> dict:
|
||||
"""Test how shifting budget into/out of one role affects total value.
|
||||
|
||||
Args:
|
||||
player_pool_df: current player pool.
|
||||
role: role to perturb. If None, tests all roles.
|
||||
delta_pct: fractional step size.
|
||||
n_steps: number of steps in each direction.
|
||||
n_calls: number of optimization calls (for compatibility).
|
||||
|
||||
Returns:
|
||||
Dict with at least a 'per_role' key containing per-role analysis.
|
||||
"""
|
||||
base_allocation = self.get_allocation()
|
||||
roles_to_test = [role] if role else list(base_allocation.keys())
|
||||
per_role = {}
|
||||
|
||||
for test_role in roles_to_test:
|
||||
results = []
|
||||
shifts = np.linspace(-delta_pct * n_steps, delta_pct * n_steps, 2 * n_steps + 1)
|
||||
|
||||
for shift in shifts:
|
||||
adjusted = {}
|
||||
for r, val in base_allocation.items():
|
||||
adjusted[r] = val * (1.0 + (shift if r == test_role else -shift / 3.0))
|
||||
|
||||
total_adj = sum(adjusted.values())
|
||||
for r in adjusted:
|
||||
adjusted[r] = adjusted[r] / total_adj * self.total_budget
|
||||
|
||||
fractions = np.array([adjusted[r] / self.total_budget for r in base_allocation.keys()])
|
||||
value = self._objective(
|
||||
fractions,
|
||||
list(base_allocation.keys()),
|
||||
player_pool_df,
|
||||
)
|
||||
|
||||
results.append({
|
||||
"shift_pct": round(shift * 100, 1),
|
||||
"allocation": {r: round(v, 1) for r, v in adjusted.items()},
|
||||
"total_value": round(value, 1),
|
||||
})
|
||||
|
||||
per_role[test_role] = pd.DataFrame(results)
|
||||
|
||||
return {"per_role": per_role, "base_allocation": base_allocation}
|
||||
@@ -0,0 +1,412 @@
|
||||
"""Opponent bidding behavior modeling using LightGBM.
|
||||
|
||||
Predicts what competitors will bid for each player in a live auction round.
|
||||
Supports Monte Carlo simulation of auction outcomes and win probability estimates.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class OpponentState:
|
||||
budget_remaining: float = 500.0
|
||||
initial_budget: float = 500.0
|
||||
slots_filled: Dict[str, int] = field(default_factory=lambda: {"P": 0, "D": 0, "C": 0, "A": 0})
|
||||
slots_total: Dict[str, int] = field(default_factory=lambda: {"P": 3, "D": 8, "C": 8, "A": 6})
|
||||
round_number: int = 1
|
||||
aggression_factor: float = 1.0
|
||||
|
||||
|
||||
def _get_attr(obj, key, default=None):
|
||||
if isinstance(obj, dict):
|
||||
return obj.get(key, default)
|
||||
return getattr(obj, key, default)
|
||||
|
||||
|
||||
class OpponentBidModel:
|
||||
"""Predicts opponent bids using LightGBM with heuristic fallback.
|
||||
|
||||
Trains on historical auction logs and outputs estimated max opponent
|
||||
bid, win probability per player, and Monte Carlo round simulations.
|
||||
"""
|
||||
|
||||
def __init__(self, random_state: int = 42):
|
||||
self.random_state = random_state
|
||||
self.model = None
|
||||
self.fitted = False
|
||||
self.feature_names: list = []
|
||||
self.rng = np.random.RandomState(random_state)
|
||||
self._role_scarcity_cache: Dict[str, float] = {}
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Feature engineering
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _extract_features(
|
||||
self,
|
||||
players_df: pd.DataFrame,
|
||||
opponent_state: OpponentState,
|
||||
) -> pd.DataFrame:
|
||||
"""Build feature matrix for LightGBM prediction.
|
||||
|
||||
Args:
|
||||
players_df: DataFrame with columns [player_name, player_role,
|
||||
player_projected_points, ...].
|
||||
opponent_state: OpponentState describing current opponent.
|
||||
|
||||
Returns:
|
||||
Feature DataFrame ready for model input.
|
||||
"""
|
||||
df = players_df.copy()
|
||||
|
||||
df["role_P"] = (df["player_role"] == "P").astype(float)
|
||||
df["role_D"] = (df["player_role"] == "D").astype(float)
|
||||
df["role_C"] = (df["player_role"] == "C").astype(float)
|
||||
df["role_A"] = (df["player_role"] == "A").astype(float)
|
||||
|
||||
df["budget_remaining_frac"] = (
|
||||
opponent_state.budget_remaining / opponent_state.initial_budget
|
||||
)
|
||||
|
||||
for role in ["P", "D", "C", "A"]:
|
||||
slots_total = opponent_state.slots_total.get(role, 1)
|
||||
slots_filled = opponent_state.slots_filled.get(role, 0)
|
||||
scarcity = (slots_total - slots_filled) / slots_total
|
||||
df[f"scarcity_{role}"] = scarcity
|
||||
|
||||
role_map = {"P": 0, "D": 1, "C": 2, "A": 3}
|
||||
df["role_code"] = df["player_role"].map(role_map)
|
||||
|
||||
if "player_projected_points" in df.columns:
|
||||
df["points_sq"] = df["player_projected_points"] ** 2
|
||||
df["points_log"] = np.log1p(df["player_projected_points"].clip(lower=0))
|
||||
|
||||
df["slots_needed_total"] = sum(
|
||||
opponent_state.slots_total.get(r, 0) - opponent_state.slots_filled.get(r, 0)
|
||||
for r in ["P", "D", "C", "A"]
|
||||
)
|
||||
df["slots_needed_total"] = df["slots_needed_total"].clip(lower=1)
|
||||
|
||||
df["round_number"] = opponent_state.round_number
|
||||
|
||||
df["urgency"] = 1.0 - (opponent_state.budget_remaining / opponent_state.initial_budget)
|
||||
df["aggression"] = opponent_state.aggression_factor
|
||||
|
||||
self.feature_names = [
|
||||
"player_projected_points",
|
||||
"role_P",
|
||||
"role_D",
|
||||
"role_C",
|
||||
"role_A",
|
||||
"role_code",
|
||||
"budget_remaining_frac",
|
||||
"scarcity_P",
|
||||
"scarcity_D",
|
||||
"scarcity_C",
|
||||
"scarcity_A",
|
||||
"points_sq",
|
||||
"points_log",
|
||||
"slots_needed_total",
|
||||
"round_number",
|
||||
"urgency",
|
||||
"aggression",
|
||||
]
|
||||
|
||||
for col in self.feature_names:
|
||||
if col not in df.columns:
|
||||
df[col] = 0.0
|
||||
|
||||
return df[self.feature_names]
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Training
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def fit(self, auction_logs: pd.DataFrame):
|
||||
"""Train LightGBM regressor on historical auction logs.
|
||||
|
||||
Args:
|
||||
auction_logs: DataFrame with columns [player_name, player_role,
|
||||
player_projected_points, opponent_budget_remaining,
|
||||
opponent_slots_remaining, role_needed_count,
|
||||
round_number, winning_bid].
|
||||
"""
|
||||
required_cols = [
|
||||
"player_role", "player_projected_points",
|
||||
"winning_bid",
|
||||
]
|
||||
for col in required_cols:
|
||||
if col not in auction_logs.columns:
|
||||
raise ValueError(f"Missing required column: '{col}' in auction_logs")
|
||||
|
||||
df = auction_logs.dropna(subset=required_cols).copy()
|
||||
|
||||
if len(df) < 20:
|
||||
logger.warning(
|
||||
f"Only {len(df)} auction records. Not enough to fit LightGBM. "
|
||||
"Using heuristic fallback."
|
||||
)
|
||||
self.fitted = False
|
||||
return
|
||||
|
||||
try:
|
||||
import lightgbm as lgb
|
||||
except ImportError:
|
||||
logger.warning("LightGBM not available. Using heuristic fallback.")
|
||||
self.fitted = False
|
||||
return
|
||||
|
||||
dummy_state = OpponentState(
|
||||
budget_remaining=df.get("opponent_budget_remaining", 500),
|
||||
initial_budget=500,
|
||||
slots_filled={"P": 0, "D": 0, "C": 0, "A": 0},
|
||||
slots_total={"P": 3, "D": 8, "C": 8, "A": 6},
|
||||
round_number=1,
|
||||
aggression_factor=1.0,
|
||||
)
|
||||
|
||||
df = df.rename(columns={
|
||||
"player_role": "player_role",
|
||||
"player_projected_points": "player_projected_points",
|
||||
})
|
||||
|
||||
dummy_df = df[["player_role", "player_projected_points"]].copy()
|
||||
dummy_df.columns = ["player_role", "player_projected_points"]
|
||||
|
||||
X = self._extract_features(dummy_df, dummy_state)
|
||||
|
||||
y = df["winning_bid"].astype(float)
|
||||
y_min, y_max = y.min(), y.max()
|
||||
|
||||
self.model = lgb.LGBMRegressor(
|
||||
n_estimators=100,
|
||||
max_depth=6,
|
||||
learning_rate=0.05,
|
||||
num_leaves=31,
|
||||
min_child_samples=10,
|
||||
subsample=0.8,
|
||||
colsample_bytree=0.8,
|
||||
random_state=self.random_state,
|
||||
verbose=-1,
|
||||
)
|
||||
self.model.fit(X, y)
|
||||
self.fitted = True
|
||||
self._y_min = y_min
|
||||
self._y_max = y_max
|
||||
|
||||
logger.info(
|
||||
f"OpponentBidModel trained on {len(df)} records. "
|
||||
f"Target range: [{y_min:.0f}, {y_max:.0f}]"
|
||||
)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Prediction
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def predict_opponent_bids(
|
||||
self,
|
||||
players_df: pd.DataFrame,
|
||||
opponent_state: OpponentState,
|
||||
) -> pd.Series:
|
||||
"""Predict max opponent bid for each player.
|
||||
|
||||
Args:
|
||||
players_df: DataFrame with player info.
|
||||
opponent_state: current opponent state.
|
||||
|
||||
Returns:
|
||||
Series of predicted max opponent bids (index = player index).
|
||||
"""
|
||||
if not self.fitted or self.model is None:
|
||||
bids = self._heuristic_bid(players_df, opponent_state)
|
||||
return pd.Series(bids, index=players_df.index)
|
||||
|
||||
X = self._extract_features(players_df, opponent_state)
|
||||
predictions = self.model.predict(X)
|
||||
predictions = np.clip(predictions, 1, _get_attr(opponent_state, "budget_remaining", 500))
|
||||
|
||||
return pd.Series(predictions, index=players_df.index)
|
||||
|
||||
def predict_p_acquire(
|
||||
self,
|
||||
players_df: pd.DataFrame,
|
||||
my_bids: np.ndarray,
|
||||
opponent_state: OpponentState,
|
||||
temperature: float = 0.1,
|
||||
) -> np.ndarray:
|
||||
"""Probability I acquire each player given my bids vs opponent.
|
||||
|
||||
Uses sigmoid: P = 1 / (1 + exp(-(my_bid - opp_bid) / temperature)).
|
||||
|
||||
Args:
|
||||
players_df: DataFrame with player info.
|
||||
my_bids: array of my bid amounts per player.
|
||||
opponent_state: current opponent state.
|
||||
temperature: softmax temperature (lower = sharper).
|
||||
|
||||
Returns:
|
||||
Array of acquisition probabilities per player.
|
||||
"""
|
||||
opp_bids = self.predict_opponent_bids(players_df, opponent_state).values
|
||||
margin = np.array(my_bids, dtype=float) - opp_bids
|
||||
scaled_temp = max(temperature * max(opp_bids.max(), 1), 0.01)
|
||||
probabilities = 1.0 / (1.0 + np.exp(-margin / scaled_temp))
|
||||
return np.clip(probabilities, 0.01, 0.99)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Monte Carlo round simulation
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def simulate_live_round(
|
||||
self,
|
||||
available_players: pd.DataFrame,
|
||||
my_budget: float,
|
||||
opponent_state: OpponentState,
|
||||
n_sims: int = 1000,
|
||||
) -> dict:
|
||||
"""Monte Carlo simulation of a live auction round.
|
||||
|
||||
Args:
|
||||
available_players: DataFrame of players up for bidding this round.
|
||||
my_budget: my remaining budget.
|
||||
opponent_state: opponent's current state.
|
||||
n_sims: number of simulation runs.
|
||||
|
||||
Returns:
|
||||
dict with expected_players_acquired, expected_cost, value_matrix.
|
||||
"""
|
||||
opp_bids = self.predict_opponent_bids(available_players, opponent_state).values
|
||||
|
||||
n_players = len(available_players)
|
||||
players_acquired = np.zeros(n_sims, dtype=int)
|
||||
total_cost = np.zeros(n_sims, dtype=float)
|
||||
value_matrix = np.zeros((n_sims, n_players), dtype=float)
|
||||
|
||||
for sim_idx in range(n_sims):
|
||||
my_budget_left = my_budget
|
||||
acquired = 0
|
||||
cost = 0.0
|
||||
|
||||
for p_idx in range(n_players):
|
||||
opp_bid = opp_bids[p_idx] + self.rng.normal(0, max(opp_bids[p_idx] * 0.15, 1))
|
||||
opp_bid = max(opp_bid, 1)
|
||||
|
||||
my_bid = self._heuristic_bid_single(
|
||||
available_players.iloc[p_idx], opponent_state
|
||||
)
|
||||
|
||||
my_bid = min(my_bid, my_budget_left)
|
||||
if my_bid > opp_bid:
|
||||
acquired += 1
|
||||
cost += my_bid
|
||||
my_budget_left -= my_bid
|
||||
value_matrix[sim_idx, p_idx] = 1.0
|
||||
|
||||
players_acquired[sim_idx] = acquired
|
||||
total_cost[sim_idx] = cost
|
||||
|
||||
return {
|
||||
"expected_players_acquired": float(np.mean(players_acquired)),
|
||||
"expected_cost": float(np.mean(total_cost)),
|
||||
"cost_std": float(np.std(total_cost)),
|
||||
"acquired_std": float(np.std(players_acquired)),
|
||||
"cost_percentile_25": float(np.percentile(total_cost, 25)),
|
||||
"cost_percentile_50": float(np.percentile(total_cost, 50)),
|
||||
"cost_percentile_75": float(np.percentile(total_cost, 75)),
|
||||
"acquisition_rate": float(players_acquired.mean() / n_players),
|
||||
"value_matrix": value_matrix,
|
||||
"n_sims": n_sims,
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Heuristic fallback
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _heuristic_bid_single(
|
||||
self, player_row: pd.Series, opponent_state: OpponentState
|
||||
) -> float:
|
||||
"""Heuristic bid for a single player (scalar version)."""
|
||||
role = player_row.get("player_role", None)
|
||||
points = float(player_row.get("player_projected_points", 6.5))
|
||||
|
||||
if isinstance(role, pd.Series):
|
||||
role = role.iloc[0]
|
||||
|
||||
scarcity = self._compute_role_scarcity(role, opponent_state)
|
||||
aggression = _get_attr(opponent_state, "aggression_factor", 1.0)
|
||||
budget_rem = _get_attr(opponent_state, "budget_remaining", 500.0)
|
||||
budget_init = _get_attr(opponent_state, "initial_budget", 500.0) or _get_attr(opponent_state, "total_budget", 500.0)
|
||||
base_bid = 0.4 * points * scarcity * aggression
|
||||
bid = base_bid * (budget_rem / budget_init)
|
||||
return max(bid, 1.0)
|
||||
|
||||
def _heuristic_bid(
|
||||
self, players_df: pd.DataFrame, opponent_state: OpponentState
|
||||
) -> np.ndarray:
|
||||
"""Heuristic bid array for all players."""
|
||||
bids = []
|
||||
for _, row in players_df.iterrows():
|
||||
bids.append(self._heuristic_bid_single(row, opponent_state))
|
||||
return np.array(bids, dtype=float)
|
||||
|
||||
def _compute_role_scarcity(
|
||||
self, role: Optional[str], opponent_state
|
||||
) -> float:
|
||||
"""Compute how scarce a role is for the opponent."""
|
||||
slots_total_dict = _get_attr(opponent_state, "slots_total", {"P": 3, "D": 8, "C": 8, "A": 6})
|
||||
slots_filled_dict = _get_attr(opponent_state, "slots_filled", {"P": 0, "D": 0, "C": 0, "A": 0})
|
||||
if role is None or role not in slots_total_dict:
|
||||
return 1.0
|
||||
slots_total = slots_total_dict[role]
|
||||
filled = slots_filled_dict.get(role, 0)
|
||||
remaining = max(slots_total - filled, 1)
|
||||
return slots_total / remaining
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Batch simulation with opponent model integration
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def run_auction_simulation(
|
||||
self,
|
||||
player_groups: List[pd.DataFrame],
|
||||
initial_budgets: List[float],
|
||||
opponent_states: List[OpponentState],
|
||||
n_sims: int = 500,
|
||||
) -> dict:
|
||||
"""Simulate multi-round auction against multiple opponents.
|
||||
|
||||
Args:
|
||||
player_groups: list of DataFrames, one per round, with available players.
|
||||
initial_budgets: my starting budget per round/concept.
|
||||
opponent_states: OpponentState for each round.
|
||||
n_sims: number of Monte Carlo runs.
|
||||
|
||||
Returns:
|
||||
dict with aggregated simulation results.
|
||||
"""
|
||||
all_results = []
|
||||
for idx, (players, budget, opp_state) in enumerate(
|
||||
zip(player_groups, initial_budgets, opponent_states)
|
||||
):
|
||||
result = self.simulate_live_round(
|
||||
players, budget, opp_state, n_sims=n_sims
|
||||
)
|
||||
result["round"] = idx
|
||||
all_results.append(result)
|
||||
|
||||
total_acquired = sum(r["expected_players_acquired"] for r in all_results)
|
||||
total_cost = sum(r["expected_cost"] for r in all_results)
|
||||
|
||||
return {
|
||||
"rounds": all_results,
|
||||
"total_expected_acquired": total_acquired,
|
||||
"total_expected_cost": total_cost,
|
||||
"n_sims": n_sims,
|
||||
}
|
||||
@@ -515,8 +515,20 @@ class RLAuctionPolicy:
|
||||
src = getattr(self.q_network, src_name)
|
||||
setattr(self.target_network, tgt_name, src.copy())
|
||||
|
||||
def _resize_networks(self, new_state_dim: int):
|
||||
"""Reinitialize networks when state dimension changes."""
|
||||
self.q_network = QNetwork(new_state_dim, self.q_network.hidden_dim, self.action_dim)
|
||||
self.target_network = QNetwork(new_state_dim, self.target_network.hidden_dim, self.action_dim)
|
||||
self._hard_update_target()
|
||||
|
||||
def _normalize_state(self, state: np.ndarray) -> np.ndarray:
|
||||
state = np.asarray(state, dtype=np.float64).ravel()
|
||||
if len(state) != len(self._obs_mean):
|
||||
self._obs_mean = np.zeros(len(state), dtype=np.float64)
|
||||
self._obs_std = np.ones(len(state), dtype=np.float64)
|
||||
self._obs_count = 0
|
||||
self.state_dim = len(state)
|
||||
self._resize_networks(len(state))
|
||||
self._obs_count += 1
|
||||
n = self._obs_count
|
||||
old_mean = self._obs_mean.copy()
|
||||
@@ -844,10 +856,31 @@ def step_in_env(env: AuctionEnv, action: int) -> Tuple[np.ndarray, float, bool,
|
||||
class RLAuctionTrainer:
|
||||
"""Convenience class for training and evaluating the RL auction agent."""
|
||||
|
||||
def __init__(self, model_dir: str = "models_trained"):
|
||||
def __init__(
|
||||
self,
|
||||
player_pool: Optional[pd.DataFrame] = None,
|
||||
n_opponents: int = 7,
|
||||
config: Optional[AuctionConfig] = None,
|
||||
model_dir: str = "models_trained",
|
||||
):
|
||||
self.player_pool = player_pool
|
||||
self.n_opponents = n_opponents
|
||||
self.config = config or AuctionConfig()
|
||||
self.model_dir = Path(model_dir)
|
||||
self.model_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def train(
|
||||
self,
|
||||
n_episodes: int = 5000,
|
||||
verbose: bool = True,
|
||||
) -> RLAuctionPolicy:
|
||||
"""Train RL policy on stored player pool."""
|
||||
if self.player_pool is None:
|
||||
raise ValueError("No player_pool provided to trainer")
|
||||
env = self.prepare_training_data(self.player_pool, n_opponents=self.n_opponents, config=self.config)
|
||||
policy, _ = self.train_agent(env, episodes=n_episodes, eval_interval=100)
|
||||
return policy
|
||||
|
||||
def prepare_training_data(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
@@ -963,6 +996,12 @@ class RLAuctionTrainer:
|
||||
"n_successful": len(values),
|
||||
}
|
||||
|
||||
for strategy, metrics in list(summary.items()):
|
||||
summary[f"{strategy}_total_value"] = metrics["avg_value"]
|
||||
|
||||
summary["rl_total_value"] = summary.get("rl_agent_total_value", 0)
|
||||
summary["greedy_total_value"] = summary.get("greedy_baseline_total_value", 0)
|
||||
|
||||
logger.info(
|
||||
f"Benchmark complete: RL={summary.get('rl_agent', {}).get('avg_value', 0):.1f} pts "
|
||||
f"vs Greedy={summary.get('greedy_baseline', {}).get('avg_value', 0):.1f} "
|
||||
|
||||
Reference in New Issue
Block a user