Add RL auction agent (Double DQN) and Set Transformer for team valuation

- rl_auction_agent.py: Custom Gym-free auction environment with N opponents,
  numpy-only Q-network with manual backprop, Double DQN policy with target
  network, full training loop with epsilon decay and periodic evaluation,
  and baseline comparison vs greedy and MILP strategies.
- set_transformer.py: Team-level valuation model treating roster as an
  unordered set. Two modes: full PyTorch Set Transformer with ISAB/PMA
  when torch is available, or sklearn Bag-of-Players fallback using
  per-role aggregates, pairwise cosine similarities, and position entropy.
This commit is contained in:
ramseshk
2026-08-11 17:27:26 +08:00
parent f83a521649
commit 916278a640
2 changed files with 1723 additions and 0 deletions
+714
View File
@@ -0,0 +1,714 @@
"""Set Transformer for team-level valuation.
Models a Fantacalcio roster as an unordered set of players rather than
a simple sum of individual projections. Captures non-linear team composition
effects: synergies between players, role balance, and diminishing returns
from overlapping skill sets.
Two implementation modes:
- PyTorch: Full Set Transformer with Induced Set Attention Blocks (ISAB)
and Pooling by Multihead Attention (PMA).
- sklearn: "Bag-of-Players" approximation using per-role aggregates,
pairwise interactions, and non-linear regression.
"""
import logging
import math
from collections import Counter
from pathlib import Path
from typing import Dict, List, Optional, Tuple
import numpy as np
import pandas as pd
from .base_model import BaseModel
logger = logging.getLogger(__name__)
ROLES = ["P", "D", "C", "A"]
ROLE_INDEX = {"P": 0, "D": 1, "C": 2, "A": 3}
_EMBEDDING_FEATURES = [
"projected_points",
"market_value",
"ceiling_price",
"vor", # value over replacement
"goals_per_game",
"assists_per_game",
"cs_prob", # clean sheet probability
"minutes_avg",
"form_recent",
"injury_risk",
"starter_prob",
]
def _has_torch() -> bool:
try:
import torch # noqa: F401
return True
except ImportError:
return False
# ---------------------------------------------------------------------------
# Torch Set Transformer modules (lazy-built to handle missing torch)
# ---------------------------------------------------------------------------
def _build_torch_classes():
"""Factory that returns torch nn.Module classes for the Set Transformer.
Only called when torch is available. All returned classes are proper
nn.Module subclasses so they work with nn.ModuleList, nn.Sequential, etc.
"""
import torch
import torch.nn as nn
import torch.nn.functional as F
class MAB(nn.Module):
"""Multihead Attention Block.
H = LayerNorm(X + Multihead(Q=X, K=Y, V=Y))
Out = LayerNorm(H + FFN(H))
"""
def __init__(self, dim, n_heads, rff_dim, dropout=0.1):
super().__init__()
self.dim = dim
self.n_heads = n_heads
self.head_dim = dim // n_heads
self.scale = self.head_dim ** 0.5
self.ln1 = nn.LayerNorm(dim)
self.ln2 = nn.LayerNorm(dim)
self.W_q = nn.Linear(dim, dim, bias=False)
self.W_k = nn.Linear(dim, dim, bias=False)
self.W_v = nn.Linear(dim, dim, bias=False)
self.W_o = nn.Linear(dim, dim, bias=False)
self.ffn = nn.Sequential(
nn.Linear(dim, rff_dim),
nn.ReLU(),
nn.Dropout(dropout),
nn.Linear(rff_dim, dim),
nn.Dropout(dropout),
)
def forward(self, X, Y):
H = self.ln1(X + self._mh_attention(X, Y, Y))
return self.ln2(H + self.ffn(H))
def _mh_attention(self, Q, K, V):
B, N_Q, _ = Q.shape
N_K = K.shape[1]
q = self.W_q(Q).view(B, N_Q, self.n_heads, self.head_dim).transpose(1, 2)
k = self.W_k(K).view(B, N_K, self.n_heads, self.head_dim).transpose(1, 2)
v = self.W_v(V).view(B, N_K, self.n_heads, self.head_dim).transpose(1, 2)
attn = torch.matmul(q, k.transpose(-2, -1)) / self.scale
attn = torch.softmax(attn, dim=-1)
out = torch.matmul(attn, v)
out = out.transpose(1, 2).contiguous().view(B, N_Q, self.dim)
return self.W_o(out)
class ISAB(nn.Module):
"""Induced Set Attention Block.
X -> MAB(X, MAB(I, X))
with learnable inducing points I.
"""
def __init__(self, dim, n_heads, n_inducing, rff_dim, dropout=0.1):
super().__init__()
self.mab1 = MAB(dim, n_heads, rff_dim, dropout)
self.mab2 = MAB(dim, n_heads, rff_dim, dropout)
self.I = nn.Parameter(torch.randn(1, n_inducing, dim) * 0.1)
self.n_inducing = n_inducing
def forward(self, X):
B = X.shape[0]
I_expanded = self.I.expand(B, -1, -1)
H = self.mab1(I_expanded, X)
return self.mab2(X, H)
class PMA(nn.Module):
"""Pooling by Multihead Attention.
SEMA = MAB(S, X) where S is a learnable seed vector.
"""
def __init__(self, dim, n_heads, n_seeds, rff_dim, dropout=0.1):
super().__init__()
self.mab = MAB(dim, n_heads, rff_dim, dropout)
self.S = nn.Parameter(torch.randn(1, n_seeds, dim) * 0.1)
self.n_seeds = n_seeds
def forward(self, X):
B = X.shape[0]
S_expanded = self.S.expand(B, -1, -1)
return self.mab(S_expanded, X)
class SetTransformerTorch(nn.Module):
"""Full Set Transformer with ISAB layers and PMA pooling."""
def __init__(self, d_input, d_model=128, n_heads=4, n_layers=3,
n_inducing=32, dropout=0.1):
super().__init__()
self.d_model = d_model
self.input_proj = nn.Linear(d_input, d_model)
self.isabs = nn.ModuleList([
ISAB(d_model, n_heads, n_inducing, d_model * 2, dropout)
for _ in range(n_layers)
])
self.pma = PMA(d_model, n_heads, 1, d_model * 2, dropout)
self.output_head = nn.Sequential(
nn.Linear(d_model, d_model // 2),
nn.ReLU(),
nn.Dropout(dropout),
nn.Linear(d_model // 2, 1),
)
def forward(self, x):
h = self.input_proj(x)
for isab in self.isabs:
h = isab(h)
pooled = self.pma(h)
pooled = pooled.squeeze(1)
return self.output_head(pooled).view(-1)
return MAB, ISAB, PMA, SetTransformerTorch
# ---------------------------------------------------------------------------
# sklearn fallback: Bag-of-Players aggregator
# ---------------------------------------------------------------------------
class _BagOfPlayers:
"""Sklearn-based team valuation using set-level aggregate features.
Instead of learning over the entire permutation-invariant set structure,
we compute fixed aggregate statistics per team:
- Per-role: mean, max, min, std of each numeric feature
- Pairwise cosine similarities between player embeddings
- Position entropy
- Budget allocation fractions
"""
def __init__(self, random_state: int = 42):
self.random_state = random_state
self.model = None
self.fitted = False
self._feature_names: List[str] = []
self._scaler = None
self._n_agg_features = 0
def _extract_features(self, team_df: pd.DataFrame) -> np.ndarray:
df = team_df.copy()
present_cols = [c for c in _EMBEDDING_FEATURES if c in df.columns]
if not present_cols:
present_cols = list(df.select_dtypes(include=[np.number]).columns)
if "role" not in df.columns:
df["role"] = "C"
df = df.fillna(0.0)
features: List[float] = []
# Per-role aggregates
for role in ROLES:
subset = df[df["role"] == role][present_cols]
if len(subset) == 0:
for col in present_cols:
features.extend([0.0, 0.0, 0.0, 0.0])
continue
values = subset.values.astype(np.float64)
for col_idx, _col in enumerate(present_cols):
col_vals = values[:, col_idx]
features.append(float(np.mean(col_vals)))
features.append(float(np.max(col_vals)))
features.append(float(np.min(col_vals)))
features.append(float(np.std(col_vals)) if len(col_vals) > 1 else 0.0)
# Role count features
n_players = len(df)
for role in ROLES:
features.append(float((df["role"] == role).sum()) / max(n_players, 1))
# Position entropy
role_counts = Counter(df["role"].tolist())
entropy = 0.0
for role in ROLES:
count = role_counts.get(role, 0)
if count > 0:
p = count / max(n_players, 1)
entropy -= p * math.log(p + 1e-10)
features.append(entropy)
features.append(float(n_players))
# Pairwise cosine similarities
if "projected_points" in df.columns and "market_value" in df.columns:
emb_cols = [c for c in ["projected_points", "market_value", "vor", "starter_prob"]
if c in df.columns]
if emb_cols:
emb = df[emb_cols].fillna(0.0).values.astype(np.float64)
emb_norm = emb / (np.linalg.norm(emb, axis=1, keepdims=True) + 1e-8)
sim_matrix = emb_norm @ emb_norm.T
n = sim_matrix.shape[0]
if n > 1:
triu_idx = np.triu_indices(n, k=1)
triu_vals = sim_matrix[triu_idx]
features.append(float(np.mean(triu_vals)))
features.append(float(np.max(triu_vals)))
features.append(float(np.min(triu_vals)))
features.append(float(np.std(triu_vals)))
else:
features.extend([0.0, 0.0, 0.0, 0.0])
else:
features.extend([0.0, 0.0, 0.0, 0.0])
# Cross-role interaction: dot products between role-group means
role_means = {}
for role in ROLES:
subset = df[df["role"] == role]
if len(subset) == 0 or "projected_points" not in subset.columns:
role_means[role] = np.zeros(len(present_cols))
else:
role_means[role] = subset[present_cols].fillna(0.0).mean().values
for i, r1 in enumerate(ROLES):
for r2 in ROLES[i + 1:]:
if np.linalg.norm(role_means[r1]) > 1e-8 and np.linalg.norm(role_means[r2]) > 1e-8:
dot = np.dot(role_means[r1], role_means[r2]) / max(
np.linalg.norm(role_means[r1]) * np.linalg.norm(role_means[r2]), 1e-8
)
else:
dot = 0.0
features.append(float(dot))
# Budget allocation features
if "market_value" in df.columns:
total_team_value = df["market_value"].sum()
for role in ROLES:
subset = df[df["role"] == role]
role_value = subset["market_value"].sum() if len(subset) > 0 else 0.0
features.append(role_value / max(total_team_value, 1))
else:
for role in ROLES:
features.append(0.0)
return np.array(features, dtype=np.float64)
def fit(self, teams_data: List[pd.DataFrame], team_values: List[float]):
from sklearn.ensemble import RandomForestRegressor
from sklearn.preprocessing import StandardScaler
if len(teams_data) == 0:
raise ValueError("teams_data cannot be empty")
X_list = [self._extract_features(df) for df in teams_data]
X = np.vstack(X_list)
y = np.array(team_values, dtype=np.float64).ravel()
if len(X) != len(y):
raise ValueError(f"Mismatch: {len(X)} teams vs {len(y)} values")
self._scaler = StandardScaler()
X_scaled = self._scaler.fit_transform(X)
self.model = RandomForestRegressor(
n_estimators=200,
max_depth=12,
min_samples_leaf=5,
random_state=self.random_state,
n_jobs=-1,
)
self.model.fit(X_scaled, y)
self.fitted = True
self._n_agg_features = X.shape[1]
# Log feature importance
if hasattr(self.model, "feature_importances_"):
top_n = min(10, len(self.model.feature_importances_))
top_idx = np.argsort(self.model.feature_importances_)[::-1][:top_n]
logger.info(
f"BagOfPlayers fitted: {len(teams_data)} teams, "
f"{X.shape[1]} features. Top features: {top_idx.tolist()}"
)
return self
def predict(self, team_df: pd.DataFrame) -> float:
if not self.fitted:
raise RuntimeError("Model not fitted. Call fit() first.")
feats = self._extract_features(team_df).reshape(1, -1)
feats_scaled = self._scaler.transform(feats)
pred = self.model.predict(feats_scaled)
return float(pred[0])
# ---------------------------------------------------------------------------
# Set Transformer main model
# ---------------------------------------------------------------------------
class SetTransformer(BaseModel):
"""Set-based team valuation model.
Values an entire Fantacalcio roster as a permutation-invariant set,
capturing non-additive synergies between players.
Args:
model_dir: directory for model persistence.
d_model: hidden dimension.
n_heads: attention heads.
n_layers: ISAB layers.
use_torch: force PyTorch mode (auto-detect by default).
"""
def __init__(
self,
model_dir: str = "models_trained",
d_model: int = 128,
n_heads: int = 4,
n_layers: int = 3,
use_torch: Optional[bool] = None,
):
super().__init__(model_dir)
self.d_model = d_model
self.n_heads = n_heads
self.n_layers = n_layers
self.use_torch = use_torch if use_torch is not None else _has_torch()
self._torch_model = None
self._sklearn_model: Optional[_BagOfPlayers] = None
self._d_input: Optional[int] = None
self._optimal_tau: float = 0.02
self._device: Optional[str] = None
self._trained = False
self._using_torch = False
def _init_torch_model(self, d_input: int):
import torch
_, _, _, SetTransformerTorch = _build_torch_classes()
self._torch_model = SetTransformerTorch(
d_input=d_input,
d_model=self.d_model,
n_heads=self.n_heads,
n_layers=self.n_layers,
n_inducing=32,
dropout=0.1,
)
self._device = "cuda" if torch.cuda.is_available() else "cpu"
self._torch_model.to(self._device)
self._d_input = d_input
self._using_torch = True
logger.info(f"SetTransformer (torch) initialized on {self._device}")
def _init_sklearn_model(self):
self._sklearn_model = _BagOfPlayers(random_state=42)
self._using_torch = False
logger.info("SetTransformer (sklearn fallback) initialized")
# ------------------------------------------------------------------
# Data preparation
# ------------------------------------------------------------------
def _prepare_set(self, team_df: pd.DataFrame) -> np.ndarray:
"""Extract player feature vectors from a team DataFrame.
Returns array of shape (n_players, d_input).
"""
df = team_df.copy()
present_cols = [c for c in _EMBEDDING_FEATURES if c in df.columns]
if not present_cols:
present_cols = list(df.select_dtypes(include=[np.number]).columns)
df = df.fillna(0.0)
if "projected_points" not in df.columns and len(present_cols) > 0:
df["projected_points"] = df[present_cols[0]]
for col in _EMBEDDING_FEATURES:
if col not in df.columns:
df[col] = 0.0
features = df[_EMBEDDING_FEATURES].values.astype(np.float64)
return features
def _collate_sets(self, teams_data: List[pd.DataFrame]) -> List[np.ndarray]:
return [self._prepare_set(df) for df in teams_data]
# ------------------------------------------------------------------
# Fit
# ------------------------------------------------------------------
def fit(self, teams_data: List[pd.DataFrame], team_values: List[float], **kwargs):
"""Fit the Set Transformer on team-level data.
Args:
teams_data: list of DataFrames, one per team, each containing
player-level features.
team_values: list of total season points for each team.
"""
if len(teams_data) == 0:
raise ValueError("teams_data cannot be empty")
if len(teams_data) != len(team_values):
raise ValueError(
f"Length mismatch: {len(teams_data)} teams, {len(team_values)} values"
)
if self.use_torch and _has_torch():
return self._fit_torch(teams_data, team_values, **kwargs)
else:
if self.use_torch and not _has_torch():
logger.warning("torch requested but not installed. Falling back to sklearn.")
return self._fit_sklearn(teams_data, team_values)
def _fit_torch(
self,
teams_data: List[pd.DataFrame],
team_values: List[float],
epochs: int = 200,
batch_size: int = 16,
lr: float = 1e-3,
weight_decay: float = 1e-5,
) -> "SetTransformer":
import torch
import torch.nn as nn
import torch.optim as optim
sets = self._collate_sets(teams_data)
d_input = sets[0].shape[1]
self._init_torch_model(d_input)
y = np.array(team_values, dtype=np.float64)
y_mean = y.mean()
y_std = max(y.std(), 1e-8)
y_norm = (y - y_mean) / y_std
self._y_mean = y_mean
self._y_std = y_std
model = self._torch_model
optimizer = optim.Adam(model.parameters(), lr=lr, weight_decay=weight_decay)
loss_fn = nn.MSELoss()
n_samples = len(sets)
best_loss = float("inf")
best_state = model.state_dict()
for epoch in range(epochs):
perm = torch.randperm(n_samples)
epoch_loss = 0.0
for start in range(0, n_samples, batch_size):
idx = perm[start:start + batch_size]
batch_loss = torch.tensor(0.0, device=self._device)
for i in idx:
x_np = sets[i.item()]
x_t = torch.tensor(
x_np, dtype=torch.float32, device=self._device
).unsqueeze(0)
y_t = torch.tensor(
[float(y_norm[i.item()])], dtype=torch.float32, device=self._device
)
pred = model(x_t)
loss = loss_fn(pred, y_t)
batch_loss = batch_loss + loss
batch_loss = batch_loss / len(idx)
optimizer.zero_grad()
batch_loss.backward()
torch.nn.utils.clip_grad_norm_(model.parameters(), 5.0)
optimizer.step()
epoch_loss += batch_loss.item() * len(idx)
epoch_loss /= n_samples
if epoch_loss < best_loss:
best_loss = epoch_loss
best_state = {k: v.cpu().clone() for k, v in model.state_dict().items()}
if (epoch + 1) % max(epochs // 10, 1) == 0:
logger.info(f"Epoch {epoch + 1}/{epochs} | loss={epoch_loss:.6f}")
if best_state is not None:
model.load_state_dict(best_state)
self._trained = True
logger.info(
f"SetTransformer (torch) trained: {n_samples} teams, "
f"best_loss={best_loss:.6f}, d_model={self.d_model}"
)
return self
def _fit_sklearn(
self, teams_data: List[pd.DataFrame], team_values: List[float]
) -> "SetTransformer":
self._init_sklearn_model()
self._sklearn_model.fit(teams_data, team_values)
self._trained = True
self._using_torch = False
return self
# ------------------------------------------------------------------
# Predict
# ------------------------------------------------------------------
def predict(self, team_roster_df: pd.DataFrame) -> np.ndarray:
"""Predict total team value (season-long points)."""
if not self._trained:
raise RuntimeError("Model not fitted. Call fit() first.")
val = self._predict_single(team_roster_df)
return np.array([val])
def _predict_single(self, team_roster_df: pd.DataFrame) -> float:
if self._using_torch and self._torch_model is not None:
import torch
x_np = self._prepare_set(team_roster_df)
if len(x_np) == 0:
return 0.0
x_t = torch.tensor(x_np, dtype=torch.float32, device=self._device).unsqueeze(0)
with torch.no_grad():
raw = self._torch_model(x_t)
val = raw.item() * getattr(self, "_y_std", 1.0) + getattr(self, "_y_mean", 0.0)
return float(val)
elif self._sklearn_model is not None:
return self._sklearn_model.predict(team_roster_df)
return 0.0
def predict_batch(self, teams: List[pd.DataFrame]) -> np.ndarray:
if not self._trained:
raise RuntimeError("Model not fitted. Call fit() first.")
return np.array([self._predict_single(df) for df in teams])
# ------------------------------------------------------------------
# Marginal value analysis
# ------------------------------------------------------------------
def value_added(self, team_roster_df: pd.DataFrame, new_player: dict) -> float:
"""Marginal value: delta when adding new_player to the team."""
baseline = self._predict_single(team_roster_df)
augmented = pd.concat(
[team_roster_df, pd.DataFrame([new_player])], ignore_index=True
)
augmented_val = self._predict_single(augmented)
return augmented_val - baseline
def value_removed(self, team_roster_df: pd.DataFrame, removed_player_idx: int) -> float:
"""Marginal loss: delta when removing a player."""
baseline = self._predict_single(team_roster_df)
reduced = team_roster_df.drop(team_roster_df.index[removed_player_idx])
reduced_val = self._predict_single(reduced)
return baseline - reduced_val
def optimal_replacement(
self,
team_roster_df: pd.DataFrame,
candidate_pool: pd.DataFrame,
to_replace: List[int],
) -> Dict[int, pd.DataFrame]:
"""For each player to replace, rank candidates by predicted team value delta.
Args:
team_roster_df: current team roster.
candidate_pool: DataFrame of free-agent candidates.
to_replace: list of indices in team_roster_df to consider replacing.
Returns:
dict mapping replace_idx -> DataFrame of candidates ranked by delta.
"""
results = {}
for rp_idx in to_replace:
base_team = team_roster_df.drop(team_roster_df.index[rp_idx])
deltas = []
for _, cand in candidate_pool.iterrows():
cand_dict = cand.to_dict()
new_team = pd.concat(
[base_team, pd.DataFrame([cand_dict])], ignore_index=True
)
new_val = self._predict_single(new_team)
current_val = self._predict_single(team_roster_df)
deltas.append({
"candidate": cand.get("name", str(cand.name)),
"role": cand.get("role", ""),
"projected_points": cand.get("projected_points", 0),
"team_value_delta": new_val - current_val,
})
results[rp_idx] = pd.DataFrame(deltas).sort_values(
"team_value_delta", ascending=False
)
return results
# ------------------------------------------------------------------
# Roster analysis
# ------------------------------------------------------------------
def get_role_synergies(self, team_roster_df: pd.DataFrame) -> pd.DataFrame:
"""Analyze which role combinations maximize team value.
Systematically tests removing one role at a time and measures
the value contribution per role. Returns a DataFrame with
per-role synergy scores.
"""
baseline = self._predict_single(team_roster_df)
results = []
for role in ROLES:
role_players = team_roster_df[team_roster_df["role"] == role]
if len(role_players) == 0:
synergy = 0.0
per_player = 0.0
else:
without_role = team_roster_df[team_roster_df["role"] != role]
without_val = self._predict_single(without_role) if len(without_role) > 0 else 0.0
synergy = baseline - without_val
per_player = synergy / len(role_players)
results.append({
"role": role,
"n_players": len(role_players),
"absolute_synergy": synergy,
"synergy_per_player": per_player,
"synergy_pct": (synergy / max(baseline, 1)) * 100,
})
return pd.DataFrame(results).sort_values("absolute_synergy", ascending=False)
def get_redundancy_score(self, team_roster_df: pd.DataFrame) -> float:
"""Compute 0-1 redundancy score for the roster.
High redundancy = lots of overlapping skill sets = diminishing
returns beyond sum of individual projections.
Uses the ratio of (sum of individual values) / (team value) as
a proxy: if team value << sum of parts, players are redundant.
Returns:
float between 0 (perfect synergy) and 1 (maximum redundancy).
"""
team_val = self._predict_single(team_roster_df)
if team_val <= 0:
return 0.0
individual_sum = 0.0
for _, player in team_roster_df.iterrows():
solo = pd.DataFrame([player.to_dict()])
individual_sum += self._predict_single(solo)
if individual_sum <= 0:
return 0.0
ratio = team_val / individual_sum
# Map: ratio ~1 means additive (no synergy, no redundancy)
# ratio >1 means synergy
# ratio <<1 means redundancy
redundancy = np.clip(1.0 - ratio, 0.0, 1.0)
return float(redundancy)