feat: add 10 new ML models for auction optimization (Phases 1-6)
Phase 1 - Quick Wins: - QuantileEnsemble: P10/P50/P90 predictions for risk-aware bidding - MinutesSurvivalModel: Weibull AFT for minutes distribution modeling Phase 2 - Adaptive Auction: - BanditAuctionSolver: Thompson Sampling for live auction bids - OpponentBidModel: Predict competitor bids via LightGBM - BudgetOptimizer: Bayesian optimization for role-level allocation Phase 3 - Deep Learning: - RLAuctionPolicy: Double DQN agent for auction strategy - SetTransformer: Team composition valuation via set-based ML Phase 4 - Probabilistic: - BayesianPlayerModel: Hierarchical pooling for rookie uncertainty - ConformalPredictor: Calibrated prediction intervals Phase 5 - Chemistry & Form: - PlayerChemistryGAT: Graph attention network for player synergies - PlayerFormModel: Hawkes process for form momentum Phase 6 - Causal: - TransferCausalModel: Causal forest for transfer effects - AuctionEffectAnalyzer: Bid adjustment from causal analysis 81 tests passing
This commit is contained in:
@@ -515,8 +515,20 @@ class RLAuctionPolicy:
|
||||
src = getattr(self.q_network, src_name)
|
||||
setattr(self.target_network, tgt_name, src.copy())
|
||||
|
||||
def _resize_networks(self, new_state_dim: int):
|
||||
"""Reinitialize networks when state dimension changes."""
|
||||
self.q_network = QNetwork(new_state_dim, self.q_network.hidden_dim, self.action_dim)
|
||||
self.target_network = QNetwork(new_state_dim, self.target_network.hidden_dim, self.action_dim)
|
||||
self._hard_update_target()
|
||||
|
||||
def _normalize_state(self, state: np.ndarray) -> np.ndarray:
|
||||
state = np.asarray(state, dtype=np.float64).ravel()
|
||||
if len(state) != len(self._obs_mean):
|
||||
self._obs_mean = np.zeros(len(state), dtype=np.float64)
|
||||
self._obs_std = np.ones(len(state), dtype=np.float64)
|
||||
self._obs_count = 0
|
||||
self.state_dim = len(state)
|
||||
self._resize_networks(len(state))
|
||||
self._obs_count += 1
|
||||
n = self._obs_count
|
||||
old_mean = self._obs_mean.copy()
|
||||
@@ -844,10 +856,31 @@ def step_in_env(env: AuctionEnv, action: int) -> Tuple[np.ndarray, float, bool,
|
||||
class RLAuctionTrainer:
|
||||
"""Convenience class for training and evaluating the RL auction agent."""
|
||||
|
||||
def __init__(self, model_dir: str = "models_trained"):
|
||||
def __init__(
|
||||
self,
|
||||
player_pool: Optional[pd.DataFrame] = None,
|
||||
n_opponents: int = 7,
|
||||
config: Optional[AuctionConfig] = None,
|
||||
model_dir: str = "models_trained",
|
||||
):
|
||||
self.player_pool = player_pool
|
||||
self.n_opponents = n_opponents
|
||||
self.config = config or AuctionConfig()
|
||||
self.model_dir = Path(model_dir)
|
||||
self.model_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def train(
|
||||
self,
|
||||
n_episodes: int = 5000,
|
||||
verbose: bool = True,
|
||||
) -> RLAuctionPolicy:
|
||||
"""Train RL policy on stored player pool."""
|
||||
if self.player_pool is None:
|
||||
raise ValueError("No player_pool provided to trainer")
|
||||
env = self.prepare_training_data(self.player_pool, n_opponents=self.n_opponents, config=self.config)
|
||||
policy, _ = self.train_agent(env, episodes=n_episodes, eval_interval=100)
|
||||
return policy
|
||||
|
||||
def prepare_training_data(
|
||||
self,
|
||||
player_pool_df: pd.DataFrame,
|
||||
@@ -963,6 +996,12 @@ class RLAuctionTrainer:
|
||||
"n_successful": len(values),
|
||||
}
|
||||
|
||||
for strategy, metrics in list(summary.items()):
|
||||
summary[f"{strategy}_total_value"] = metrics["avg_value"]
|
||||
|
||||
summary["rl_total_value"] = summary.get("rl_agent_total_value", 0)
|
||||
summary["greedy_total_value"] = summary.get("greedy_baseline_total_value", 0)
|
||||
|
||||
logger.info(
|
||||
f"Benchmark complete: RL={summary.get('rl_agent', {}).get('avg_value', 0):.1f} pts "
|
||||
f"vs Greedy={summary.get('greedy_baseline', {}).get('avg_value', 0):.1f} "
|
||||
|
||||
Reference in New Issue
Block a user