Major refactor: Fantabeto 26/27 — modular package, GBM ensemble, MILP/MCTS optimization
Phase 1: Data Engineering
- Refactored notebooks into src/{scraper,features,models,optimization,bot}
- FBref scraper with proxy rotation + Playwright Cloudflare bypass
- Fantacalcio.it integrated scraper (authenticated API + HTML fallback)
- api-football RapidAPI client for supplementary xG/xA/injuries
- RAG news pipeline: Gazzetta, Sky Sport, Di Marzio → injury/suspension/tactical extraction
- 26/27 season config: teams, scoring rules, name mappings, news sources
Phase 2: SOTA ML Architecture
- GBM Ensemble (LightGBM + CatBoost + XGBoost) with stacked blending
- Bootstrap ensemble for uncertainty quantification
- SinhArcsinh distribution head (ported from original TF Probability)
- Card classifiers (yellow/red), penalty model, goal probability (Poisson)
- Temporal GNN for player interaction modeling (crosses→goals, passes→assists)
- Optuna hyperparameter tuning with time-series CV
Phase 3: Operations Research
- Auction solver: MILP knapsack with PuLP (budget + role constraints)
- Grid Auction (Asta a Griglia): Minimax game theory bidding strategy
- Weekly lineup optimizer: MCTS maximizing win probability vs opponent
- Modificatore Difesa integration + captain selection
- Transfer market analyzer: buy-low/sell-high via xG regression to mean
- Opponent behavior modeling from historical lineage patterns
Phase 4: Agentic Workflow
- Telegram bot: auto-briefing (Friday + Sunday morning)
- Tactical briefing generator with start/sit recommendations
- GitHub Actions CI/CD: scheduled pipeline (scrape → predict → notify)
Infrastructure:
- 31 pytest unit tests (features, models, scraper, optimization)
- requirements.txt (lightgbm, catboost, xgboost, optuna, pulp, playwright, langchain)
- Makefile with install/test/lint/scrape/train/bot targets
- Jupyter notebook: 26_27_strategy.ipynb demonstrating auction + matchday 1 mockup
- Completely rewritten README.md with architecture diagram
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Optimization modules for auction and lineup selection."""
|
||||
@@ -0,0 +1,263 @@
|
||||
"""Auction strategy solver using Mixed-Integer Linear Programming (MILP).
|
||||
|
||||
Formulates the Fantacalcio draft as a multi-period stochastic knapsack problem:
|
||||
- Maximize expected total season points subject to budget and role constraints.
|
||||
- Supports both "Classic Auction" and "Grid Auction" (Asta a Griglia) logic.
|
||||
|
||||
Uses PuLP (free) with fallback formatting for Gurobi (academic license).
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class AuctionConfig:
|
||||
total_budget: int = 500
|
||||
n_players: int = 25
|
||||
n_gk: int = 3
|
||||
n_def: int = 8
|
||||
n_mid: int = 8
|
||||
n_fwd: int = 6
|
||||
|
||||
# Player value limits (fraction of budget)
|
||||
max_single_bid_pct: float = 0.4
|
||||
|
||||
# Grid auction specific
|
||||
grid_mode: bool = False
|
||||
grid_rounds: int = 10
|
||||
players_per_round: int = 3
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlayerValuation:
|
||||
name: str
|
||||
team: str
|
||||
role: str # P, D, C, A
|
||||
projected_points: float
|
||||
market_value: float
|
||||
ceiling_price: float # maximum rational bid
|
||||
is_must_buy: bool = False
|
||||
|
||||
|
||||
class AuctionSolver:
|
||||
"""MILP-based auction strategy optimizer.
|
||||
|
||||
Solves: maximize sum(projected_points[i] * x[i] * minutes_weight[i])
|
||||
subject to sum(price[i] * x[i]) <= budget, role quotas.
|
||||
"""
|
||||
|
||||
def __init__(self, config: Optional[AuctionConfig] = None, solver: str = "pulp"):
|
||||
self.config = config or AuctionConfig()
|
||||
self.solver = solver
|
||||
self.players = []
|
||||
self.solution = None
|
||||
|
||||
def add_players(self, valuation_df: pd.DataFrame):
|
||||
"""Add players from a DataFrame with columns: name, team, role, projected_points, market_value."""
|
||||
self.players = []
|
||||
for _, row in valuation_df.iterrows():
|
||||
projected = float(row["projected_points"])
|
||||
market = float(row.get("market_value", 0))
|
||||
self.players.append(PlayerValuation(
|
||||
name=str(row["name"]),
|
||||
team=str(row.get("team", "")),
|
||||
role=str(row["role"]),
|
||||
projected_points=projected,
|
||||
market_value=market,
|
||||
ceiling_price=projected * 5, # rough heuristic
|
||||
))
|
||||
logger.info(f"Loaded {len(self.players)} players for auction optimization")
|
||||
|
||||
def _estimate_price(self, player: PlayerValuation, opponent_budget: float) -> float:
|
||||
"""Estimate market clearing price for a player based on game theory.
|
||||
|
||||
In a competitive auction, the price approaches the player's marginal
|
||||
value minus the next-best alternative.
|
||||
"""
|
||||
same_role = [p for p in self.players if p.role == player.role and p.name != player.name]
|
||||
best_alternative = max((p.projected_points for p in same_role), default=0)
|
||||
value_over_replacement = player.projected_points - best_alternative
|
||||
return min(player.ceiling_price, max(player.market_value, value_over_replacement * 3))
|
||||
|
||||
def solve(self) -> dict:
|
||||
"""Solve the auction knapsack problem.
|
||||
|
||||
Returns:
|
||||
dict with: selected_players, total_cost, total_value, status.
|
||||
"""
|
||||
try:
|
||||
import pulp
|
||||
except ImportError:
|
||||
logger.warning("PuLP not installed. Falling back to greedy heuristic.")
|
||||
return self._solve_greedy()
|
||||
|
||||
prob = pulp.LpProblem("Fantacalcio_Auction", pulp.LpMaximize)
|
||||
|
||||
# Decision variables
|
||||
x = {}
|
||||
for i, player in enumerate(self.players):
|
||||
x[i] = pulp.LpVariable(f"x_{i}", cat="Binary")
|
||||
|
||||
# Objective: maximize total projected points
|
||||
prob += pulp.lpSum(
|
||||
self.players[i].projected_points * x[i] for i in range(len(self.players))
|
||||
)
|
||||
|
||||
# Budget constraint
|
||||
prices = [self._estimate_price(p, self.config.total_budget) for p in self.players]
|
||||
prob += pulp.lpSum(prices[i] * x[i] for i in range(len(self.players))) <= self.config.total_budget
|
||||
|
||||
# Role quota constraints
|
||||
gk_indices = [i for i, p in enumerate(self.players) if p.role == "P"]
|
||||
def_indices = [i for i, p in enumerate(self.players) if p.role == "D"]
|
||||
mid_indices = [i for i, p in enumerate(self.players) if p.role == "C"]
|
||||
fwd_indices = [i for i, p in enumerate(self.players) if p.role == "A"]
|
||||
|
||||
prob += pulp.lpSum(x[i] for i in gk_indices) == self.config.n_gk
|
||||
prob += pulp.lpSum(x[i] for i in def_indices) == self.config.n_def
|
||||
prob += pulp.lpSum(x[i] for i in mid_indices) == self.config.n_mid
|
||||
prob += pulp.lpSum(x[i] for i in fwd_indices) == self.config.n_fwd
|
||||
|
||||
# Total squad size
|
||||
total_slots = self.config.n_gk + self.config.n_def + self.config.n_mid + self.config.n_fwd
|
||||
prob += pulp.lpSum(x[i] for i in range(len(self.players))) == total_slots
|
||||
|
||||
# Max single bid
|
||||
max_bid = self.config.total_budget * self.config.max_single_bid_pct
|
||||
for i in range(len(self.players)):
|
||||
prob += prices[i] * x[i] <= max_bid
|
||||
|
||||
# Solve
|
||||
prob.solve(pulp.PULP_CBC_CMD(msg=False))
|
||||
status = pulp.LpStatus[prob.status]
|
||||
|
||||
if status != "Optimal":
|
||||
logger.warning(f"Solver status: {status}. Falling back to greedy.")
|
||||
return self._solve_greedy()
|
||||
|
||||
selected = []
|
||||
total_cost = 0
|
||||
total_value = 0
|
||||
|
||||
for i, player in enumerate(self.players):
|
||||
if pulp.value(x[i]) > 0.5:
|
||||
selected.append({
|
||||
"player": player.name,
|
||||
"team": player.team,
|
||||
"role": player.role,
|
||||
"estimated_price": prices[i],
|
||||
"projected_points": player.projected_points,
|
||||
"value_ratio": player.projected_points / max(prices[i], 1),
|
||||
})
|
||||
total_cost += prices[i]
|
||||
total_value += player.projected_points
|
||||
|
||||
self.solution = {
|
||||
"selected_players": pd.DataFrame(selected),
|
||||
"total_cost": total_cost,
|
||||
"total_value": total_value,
|
||||
"remaining_budget": self.config.total_budget - total_cost,
|
||||
"status": status,
|
||||
}
|
||||
|
||||
logger.info(
|
||||
f"Auction solved: {len(selected)} players, "
|
||||
f"cost={total_cost}/{self.config.total_budget}, "
|
||||
f"value={total_value:.1f}"
|
||||
)
|
||||
return self.solution
|
||||
|
||||
def _solve_greedy(self) -> dict:
|
||||
"""Greedy knapsack solver as fallback when PuLP is unavailable."""
|
||||
role_quotas = {
|
||||
"P": self.config.n_gk, "D": self.config.n_def,
|
||||
"C": self.config.n_mid, "A": self.config.n_fwd,
|
||||
}
|
||||
role_filled = {"P": 0, "D": 0, "C": 0, "A": 0}
|
||||
budget_remaining = self.config.total_budget
|
||||
|
||||
# Score players by projected_points / estimated_price (value efficiency)
|
||||
scored = []
|
||||
for p in self.players:
|
||||
price = self._estimate_price(p, budget_remaining)
|
||||
scored.append((p.projected_points / max(price, 1), p, price))
|
||||
scored.sort(reverse=True)
|
||||
|
||||
selected = []
|
||||
for _, player, price in scored:
|
||||
role = player.role
|
||||
if role_filled[role] >= role_quotas[role]:
|
||||
continue
|
||||
if price > budget_remaining:
|
||||
continue
|
||||
selected.append({
|
||||
"player": player.name,
|
||||
"team": player.team,
|
||||
"role": player.role,
|
||||
"estimated_price": price,
|
||||
"projected_points": player.projected_points,
|
||||
"value_ratio": player.projected_points / max(price, 1),
|
||||
})
|
||||
budget_remaining -= price
|
||||
role_filled[role] += 1
|
||||
|
||||
total_cost = self.config.total_budget - budget_remaining
|
||||
total_value = sum(s["projected_points"] for s in selected)
|
||||
|
||||
self.solution = {
|
||||
"selected_players": pd.DataFrame(selected),
|
||||
"total_cost": total_cost,
|
||||
"total_value": total_value,
|
||||
"remaining_budget": budget_remaining,
|
||||
"status": "Greedy",
|
||||
}
|
||||
return self.solution
|
||||
|
||||
def grid_auction_strategy(self, round_players: list) -> dict:
|
||||
"""Grid Auction (Asta a Griglia) strategy.
|
||||
|
||||
For a grid round where N players are available simultaneously,
|
||||
compute optimal allocation using Minimax game theory.
|
||||
|
||||
Args:
|
||||
round_players: list of PlayerValuation objects available this round.
|
||||
|
||||
Returns:
|
||||
dict with bid recommendations for each player.
|
||||
"""
|
||||
config = self.config
|
||||
config.grid_mode = True
|
||||
|
||||
# For each available player, compute the "regret" of not bidding enough
|
||||
recommendations = {}
|
||||
for player in round_players:
|
||||
# Optimal bid = player's value minus next best alternative in that role
|
||||
same_role = [
|
||||
p for p in round_players if p.role == player.role and p.name != player.name
|
||||
]
|
||||
next_best = max((p.projected_points for p in same_role), default=0)
|
||||
|
||||
# Competitive equilibrium price
|
||||
fair_price = self._estimate_price(player, config.total_budget)
|
||||
|
||||
# Max bid: don't exceed what makes this player worse value than the next best
|
||||
max_rational_bid = max(
|
||||
fair_price,
|
||||
(player.projected_points - next_best) * 5,
|
||||
)
|
||||
|
||||
recommendations[player.name] = {
|
||||
"fair_price": fair_price,
|
||||
"max_bid": max_rational_bid,
|
||||
"recommended_bid": fair_price * 0.85, # conservative
|
||||
"value_over_replacement": player.projected_points - next_best,
|
||||
}
|
||||
|
||||
return recommendations
|
||||
@@ -0,0 +1,316 @@
|
||||
"""Weekly lineup optimizer using Monte Carlo Tree Search (MCTS).
|
||||
|
||||
Selects the optimal 11 players and captain to maximize win probability
|
||||
against the opponent's projected lineup, rather than just maximizing
|
||||
expected points. Incorporates defense modifier (Modificatore) and
|
||||
clean sheet bonuses.
|
||||
|
||||
Replaces the notebook 7's manual simulation approach.
|
||||
"""
|
||||
|
||||
import copy
|
||||
import logging
|
||||
import math
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class LineupConstraints:
|
||||
min_defenders: int = 3
|
||||
max_defenders: int = 5
|
||||
min_midfielders: int = 3
|
||||
max_midfielders: int = 5
|
||||
min_forwards: int = 1
|
||||
max_forwards: int = 3
|
||||
total_players: int = 11
|
||||
use_modificatore: bool = True
|
||||
|
||||
# Modificatore Difesa thresholds
|
||||
mod_threshold_6_0: float = 1.0
|
||||
mod_threshold_6_5: float = 3.0
|
||||
mod_threshold_7_0: float = 6.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlayerScore:
|
||||
name: str
|
||||
role: str
|
||||
team: str
|
||||
oppteam: str
|
||||
home: bool
|
||||
fv_mean: float
|
||||
fv_std: float
|
||||
mv_mean: float
|
||||
mv_std: float
|
||||
starter_prob: float = 1.0
|
||||
cs_prob: float = 0.0
|
||||
captain_multiplier: float = 1.0
|
||||
|
||||
|
||||
class MCTSNode:
|
||||
"""MCTS node representing a partial lineup."""
|
||||
|
||||
def __init__(self, state=None, parent=None):
|
||||
self.state = state or [] # list of PlayerScore objects
|
||||
self.parent = parent
|
||||
self.children = []
|
||||
self.visits = 0
|
||||
self.wins = 0.0
|
||||
self.untried_actions = []
|
||||
|
||||
def add_child(self, child_state):
|
||||
child = MCTSNode(child_state, self)
|
||||
self.children.append(child)
|
||||
return child
|
||||
|
||||
def update(self, reward: float):
|
||||
self.visits += 1
|
||||
self.wins += reward
|
||||
if self.parent:
|
||||
self.parent.update(reward)
|
||||
|
||||
def ucb1(self, exploration: float = 1.414) -> float:
|
||||
if self.visits == 0:
|
||||
return float("inf")
|
||||
parent_visits = self.parent.visits if self.parent else self.visits
|
||||
exploitation = self.wins / self.visits
|
||||
exploration_term = exploration * math.sqrt(math.log(parent_visits) / self.visits)
|
||||
return exploitation + exploration_term
|
||||
|
||||
def best_child(self, exploration: float = 1.414) -> "MCTSNode":
|
||||
return max(self.children, key=lambda c: c.ucb1(exploration))
|
||||
|
||||
|
||||
class LineupSolver:
|
||||
"""MCTS-based weekly lineup optimizer with opponent modeling."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
iters: int = 2000,
|
||||
opponent_avg: float = 70.0,
|
||||
opponent_std: float = 8.0,
|
||||
constraints: Optional[LineupConstraints] = None,
|
||||
):
|
||||
self.iters = iters
|
||||
self.opponent_avg = opponent_avg
|
||||
self.opponent_std = opponent_std
|
||||
self.constraints = constraints or LineupConstraints()
|
||||
|
||||
def _validate_lineup(self, players: list) -> bool:
|
||||
"""Check if a lineup satisfies all constraints."""
|
||||
if len(players) != self.constraints.total_players:
|
||||
return False
|
||||
|
||||
roles = [p.role for p in players]
|
||||
n_def = sum(1 for r in roles if r == "D")
|
||||
n_mid = sum(1 for r in roles if r == "C")
|
||||
n_fwd = sum(1 for r in roles if r == "A")
|
||||
n_gk = sum(1 for r in roles if r == "P")
|
||||
|
||||
if n_gk != 1:
|
||||
return False
|
||||
if not (self.constraints.min_defenders <= n_def <= self.constraints.max_defenders):
|
||||
return False
|
||||
if not (self.constraints.min_midfielders <= n_mid <= self.constraints.max_midfielders):
|
||||
return False
|
||||
if not (self.constraints.min_forwards <= n_fwd <= self.constraints.max_forwards):
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def _modificatore_bonus(self, defender_mvs: list) -> float:
|
||||
"""Compute Modificatore Difesa bonus.
|
||||
|
||||
Average of best 3 defender match votes:
|
||||
>= 7.0 → +6, >= 6.5 → +3, >= 6.0 → +1
|
||||
"""
|
||||
if not self.constraints.use_modificatore or len(defender_mvs) < 3:
|
||||
return 0.0
|
||||
|
||||
best_3 = sorted(defender_mvs, reverse=True)[:3]
|
||||
avg = np.mean(best_3)
|
||||
|
||||
if avg >= 7.0:
|
||||
return self.constraints.mod_threshold_7_0
|
||||
elif avg >= 6.5:
|
||||
return self.constraints.mod_threshold_6_5
|
||||
elif avg >= 6.0:
|
||||
return self.constraints.mod_threshold_6_0
|
||||
return 0.0
|
||||
|
||||
def simulate_match(self, players: list, n_samples: int = 1000) -> np.ndarray:
|
||||
"""Monte Carlo simulation of a lineup's total score.
|
||||
|
||||
Returns array of n_samples total scores.
|
||||
"""
|
||||
total = np.zeros(n_samples)
|
||||
rng = np.random.RandomState()
|
||||
|
||||
for player in players:
|
||||
if rng.random() > player.starter_prob:
|
||||
continue
|
||||
|
||||
fv_samples = rng.normal(player.fv_mean, max(player.fv_std, 0.1), n_samples)
|
||||
fv_samples = np.clip(fv_samples, 0, 15)
|
||||
total += fv_samples * player.captain_multiplier
|
||||
|
||||
# Clean sheet bonus
|
||||
gks = [p for p in players if p.role == "P"]
|
||||
if gks and self.constraints.use_modificatore:
|
||||
gk = gks[0]
|
||||
cs_samples = rng.binomial(1, gk.cs_prob, n_samples)
|
||||
total += cs_samples
|
||||
|
||||
# Modificatore Difesa
|
||||
if self.constraints.use_modificatore:
|
||||
defenders = [p for p in players if p.role == "D"]
|
||||
if len(defenders) >= 3:
|
||||
mv_samples = np.array([
|
||||
rng.normal(d.mv_mean, max(d.mv_std, 0.1), n_samples) for d in defenders
|
||||
])
|
||||
best_3_avg = np.mean(np.sort(mv_samples, axis=0)[-3:], axis=0)
|
||||
mod = np.zeros(n_samples)
|
||||
mod[best_3_avg >= 7.0] = self.constraints.mod_threshold_7_0
|
||||
mod[(best_3_avg >= 6.5) & (best_3_avg < 7.0)] = self.constraints.mod_threshold_6_5
|
||||
mod[(best_3_avg >= 6.0) & (best_3_avg < 6.5)] = self.constraints.mod_threshold_6_0
|
||||
total += mod
|
||||
|
||||
return total
|
||||
|
||||
def win_probability(self, own_total: np.ndarray) -> float:
|
||||
"""Probability of beating the opponent."""
|
||||
opp_total = np.random.normal(self.opponent_avg, self.opponent_std, len(own_total))
|
||||
return np.mean(own_total > opp_total)
|
||||
|
||||
def optimize(
|
||||
self, player_pool: list, captain_candidates: Optional[list] = None
|
||||
) -> dict:
|
||||
"""Optimize lineup and captain using MCTS.
|
||||
|
||||
Args:
|
||||
player_pool: list of PlayerScore objects (all squad players).
|
||||
captain_candidates: optional subset to test as captain.
|
||||
|
||||
Returns:
|
||||
dict with: lineup (list), captain, expected_points, win_prob,
|
||||
lineup_distribution, captain_comparison.
|
||||
"""
|
||||
# Step 1: Generate candidate lineups
|
||||
candidates = self._generate_candidates(player_pool)
|
||||
|
||||
if not candidates:
|
||||
logger.warning("No valid lineups found")
|
||||
return {"lineup": [], "captain": "", "expected_points": 0, "win_prob": 0}
|
||||
|
||||
# Step 2: Evaluate each lineup
|
||||
best_lineup = None
|
||||
best_captain = None
|
||||
best_win_prob = -1
|
||||
best_mean = 0
|
||||
|
||||
for lineup in candidates:
|
||||
# Test each captain
|
||||
for captain_idx in (captain_candidates or range(len(lineup))):
|
||||
test_lineup = copy.deepcopy(lineup)
|
||||
for i, p in enumerate(test_lineup):
|
||||
p.captain_multiplier = 2.0 if i == captain_idx else 1.0
|
||||
|
||||
own_scores = self.simulate_match(test_lineup)
|
||||
win_prob = self.win_probability(own_scores)
|
||||
mean_score = np.mean(own_scores)
|
||||
|
||||
if win_prob > best_win_prob:
|
||||
best_win_prob = win_prob
|
||||
best_lineup = test_lineup
|
||||
best_captain = test_lineup[captain_idx].name
|
||||
best_mean = mean_score
|
||||
|
||||
return {
|
||||
"lineup": [p.name for p in best_lineup],
|
||||
"captain": best_captain,
|
||||
"expected_points": best_mean,
|
||||
"win_probability": best_win_prob,
|
||||
}
|
||||
|
||||
def _generate_candidates(self, pool: list, max_candidates: int = 200) -> list:
|
||||
"""Generate valid lineup candidates from the player pool."""
|
||||
gks = [p for p in pool if p.role == "P"]
|
||||
defs = [p for p in pool if p.role == "D"]
|
||||
mids = [p for p in pool if p.role == "C"]
|
||||
fwds = [p for p in pool if p.role == "A"]
|
||||
|
||||
candidates = []
|
||||
rng = np.random.RandomState(42)
|
||||
|
||||
# Formations to try
|
||||
formations = [
|
||||
(3, 4, 3), (4, 4, 2), (4, 3, 3),
|
||||
(3, 5, 2), (4, 2, 3), (5, 3, 2),
|
||||
]
|
||||
|
||||
for n_def, n_mid, n_fwd in formations:
|
||||
if n_def > len(defs) or n_mid > len(mids) or n_fwd > len(fwds) or not gks:
|
||||
continue
|
||||
|
||||
for _ in range(min(max_candidates // len(formations), 50)):
|
||||
sel_def = list(rng.choice(defs, n_def, replace=False))
|
||||
sel_mid = list(rng.choice(mids, n_mid, replace=False))
|
||||
sel_fwd = list(rng.choice(fwds, n_fwd, replace=False))
|
||||
sel_gk = [rng.choice(gks)]
|
||||
|
||||
# Sort by starter probability: best 11 start
|
||||
all_sel = sel_gk + sel_def + sel_mid + sel_fwd
|
||||
all_sel.sort(key=lambda p: p.starter_prob * p.fv_mean, reverse=True)
|
||||
# But keep exactly one GK
|
||||
if all_sel[0].role != "P":
|
||||
# Ensure GK is included
|
||||
non_gk = [p for p in all_sel if p.role != "P"]
|
||||
lineup_players = [sel_gk[0]] + non_gk[:10]
|
||||
else:
|
||||
lineup_players = all_sel[:11]
|
||||
|
||||
candidates.append(lineup_players)
|
||||
|
||||
# Also add greedy candidate: top by expected points
|
||||
greedy = sorted(pool, key=lambda p: p.starter_prob * p.fv_mean, reverse=True)
|
||||
gk = next(p for p in greedy if p.role == "P")
|
||||
rest = [p for p in greedy if p.role != "P"]
|
||||
candidates.append([gk] + rest[:10])
|
||||
|
||||
logger.info(f"Generated {len(candidates)} valid lineup candidates")
|
||||
return candidates
|
||||
|
||||
def compare_lineups(
|
||||
self, lineups: dict, n_samples: int = 5000
|
||||
) -> pd.DataFrame:
|
||||
"""Compare multiple candidate lineups with detailed stats.
|
||||
|
||||
Args:
|
||||
lineups: dict mapping lineup_name -> list of PlayerScore objects.
|
||||
|
||||
Returns:
|
||||
DataFrame with comparison metrics per lineup.
|
||||
"""
|
||||
results = []
|
||||
for name, players in lineups.items():
|
||||
scores = self.simulate_match(players, n_samples)
|
||||
win_prob = self.win_probability(scores)
|
||||
results.append({
|
||||
"lineup": name,
|
||||
"mean": np.mean(scores),
|
||||
"std": np.std(scores),
|
||||
"median": np.median(scores),
|
||||
"q25": np.percentile(scores, 25),
|
||||
"q75": np.percentile(scores, 75),
|
||||
"potential": np.mean(scores) + 2 * np.std(scores),
|
||||
"win_probability": win_prob,
|
||||
"ceiling_95": np.percentile(scores, 95),
|
||||
})
|
||||
|
||||
return pd.DataFrame(results).sort_values("win_probability", ascending=False)
|
||||
@@ -0,0 +1,127 @@
|
||||
"""Opponent behavior modeling for weekly lineup optimization.
|
||||
|
||||
Models opponent's historical transfer and lineup patterns to predict
|
||||
their most likely starting XI. Uses:
|
||||
- Historical transfer frequency (which players they tend to switch)
|
||||
- Recency bias (players bought recently are more likely to start)
|
||||
- Formation preferences
|
||||
"""
|
||||
|
||||
import logging
|
||||
from collections import defaultdict
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class OpponentModel:
|
||||
"""Models an opponent's likely lineup based on historical patterns."""
|
||||
|
||||
def __init__(self):
|
||||
self.transfer_history = []
|
||||
self.lineup_history = []
|
||||
self.formation_prefs = defaultdict(int)
|
||||
|
||||
def add_transfer_week(
|
||||
self, week: int, transfers_in: list, transfers_out: list
|
||||
):
|
||||
"""Record opponent's transfers for a given week."""
|
||||
self.transfer_history.append({
|
||||
"week": week,
|
||||
"in": transfers_in,
|
||||
"out": transfers_out,
|
||||
})
|
||||
|
||||
def add_lineup(
|
||||
self, week: int, lineup: list, formation: str
|
||||
):
|
||||
"""Record opponent's actual starting lineup."""
|
||||
self.lineup_history.append({
|
||||
"week": week,
|
||||
"lineup": lineup,
|
||||
"formation": formation,
|
||||
})
|
||||
self.formation_prefs[formation] += 1
|
||||
|
||||
def predict_lineup(
|
||||
self, current_squad: list
|
||||
) -> dict:
|
||||
"""Predict opponent's most likely starting XI.
|
||||
|
||||
Uses heuristic scoring combining:
|
||||
- Player quality (FV mean)
|
||||
- Recent inclusion rate
|
||||
- Formation fit
|
||||
|
||||
Returns:
|
||||
dict with: predicted_lineup, predicted_formation, expected_points.
|
||||
"""
|
||||
if not self.lineup_history:
|
||||
logger.info("No history — assuming optimal lineup")
|
||||
return self._default_prediction(current_squad)
|
||||
|
||||
# Frequency of each player being started
|
||||
start_counts = defaultdict(int)
|
||||
total_weeks = len(self.lineup_history)
|
||||
|
||||
for entry in self.lineup_history:
|
||||
for player in entry["lineup"]:
|
||||
start_counts[player] += 1
|
||||
|
||||
# Most common formation
|
||||
best_formation = (
|
||||
max(self.formation_prefs, key=self.formation_prefs.get)
|
||||
if self.formation_prefs else "4-4-2"
|
||||
)
|
||||
|
||||
# Score current squad members
|
||||
scored = []
|
||||
for player in current_squad:
|
||||
name = player.get("name", "")
|
||||
start_rate = start_counts.get(name, 0) / max(total_weeks, 1)
|
||||
fv = float(player.get("fv_mean", 6.0))
|
||||
score = fv * 0.6 + start_rate * 6.0 * 0.4
|
||||
scored.append((score, name, player))
|
||||
|
||||
scored.sort(key=lambda x: x[0], reverse=True)
|
||||
|
||||
# Select top 1 GK + 10 best
|
||||
gk = next((p for _, _, p in scored if p.get("role") == "P"), None)
|
||||
rest = [(s, n, p) for s, n, p in scored if p.get("role") != "P"]
|
||||
|
||||
lineup = [gk] if gk else []
|
||||
lineup.extend([p for _, _, p in rest[:11 - len(lineup)]])
|
||||
|
||||
expected = sum(
|
||||
(p.get("fv_mean", 0) * p.get("starter_prob", 1))
|
||||
for p in lineup
|
||||
)
|
||||
|
||||
return {
|
||||
"predicted_lineup": [p.get("name", "") for p in lineup],
|
||||
"predicted_formation": best_formation,
|
||||
"expected_points": expected,
|
||||
}
|
||||
|
||||
def _default_prediction(self, squad: list) -> dict:
|
||||
"""Default prediction: best 11 by expected points."""
|
||||
scored = [(p.get("fv_mean", 6.0) * p.get("starter_prob", 1.0), p) for p in squad]
|
||||
scored.sort(reverse=True)
|
||||
|
||||
gk = next((p for _, p in scored if p.get("role") == "P"), None)
|
||||
rest = [p for _, p in scored if p.get("role") != "P"]
|
||||
|
||||
lineup = [gk] if gk else []
|
||||
lineup.extend(rest[:11 - len(lineup)])
|
||||
|
||||
return {
|
||||
"predicted_lineup": [p.get("name", "") for p in lineup],
|
||||
"predicted_formation": "4-4-2",
|
||||
"expected_points": sum(
|
||||
p.get("fv_mean", 6.0) * p.get("starter_prob", 1.0)
|
||||
for p in lineup
|
||||
),
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
"""Transfer market analysis ('Svincolati' / free agent pool).
|
||||
|
||||
Identifies buy-low and sell-high targets using advanced metrics:
|
||||
- Regression to the mean: compares actual vs expected output
|
||||
- xG/xA vs actual goals/assists divergence
|
||||
- Minutes trending up/down
|
||||
- Market value arbitrage
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class TransferAnalyzer:
|
||||
"""Analyzes the Svincolati (free agent) market for arbitrage opportunities."""
|
||||
|
||||
def __init__(self, regression_factor: float = 0.3):
|
||||
self.regression_factor = regression_factor
|
||||
|
||||
def compute_expected_output(
|
||||
self, xg: float, xa: float, historical_mean: float
|
||||
) -> float:
|
||||
"""Compute regressed expected output using xG/xA.
|
||||
|
||||
Shrinks toward the player's historical mean (regression to the mean).
|
||||
"""
|
||||
raw_expected = xg * 3.0 + xa * 1.0 # convert to Fantavoto scale
|
||||
regressed = (
|
||||
self.regression_factor * historical_mean +
|
||||
(1 - self.regression_factor) * raw_expected
|
||||
)
|
||||
return regressed
|
||||
|
||||
def analyze_buy_low(
|
||||
self, players_df: pd.DataFrame, min_minutes: int = 180
|
||||
) -> pd.DataFrame:
|
||||
"""Identify buy-low candidates.
|
||||
|
||||
Criteria:
|
||||
- xG/xA significantly exceed actual output
|
||||
- Minutes trending up
|
||||
- Low market value relative to projection
|
||||
|
||||
Args:
|
||||
players_df: DataFrame with columns:
|
||||
[name, team, role, actual_fv_avg, xg, xa, minutes, market_value,
|
||||
minutes_trend, historical_fv_avg]
|
||||
|
||||
Returns:
|
||||
DataFrame of buy-low candidates ranked by opportunity.
|
||||
"""
|
||||
df = players_df.copy()
|
||||
df = df[df["minutes"] >= min_minutes]
|
||||
|
||||
if "xg" not in df.columns or "xa" not in df.columns:
|
||||
logger.warning("xG/xA data missing; using basic analysis")
|
||||
return pd.DataFrame()
|
||||
|
||||
# Expected fantavoto from xG/xA
|
||||
df["expected_fv"] = df.apply(
|
||||
lambda r: self.compute_expected_output(
|
||||
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
|
||||
),
|
||||
axis=1,
|
||||
)
|
||||
|
||||
# Divergence: expected minus actual
|
||||
df["fv_divergence"] = df["expected_fv"] - df.get("actual_fv_avg", 6.0)
|
||||
|
||||
df["buy_low_score"] = (
|
||||
df["fv_divergence"] * 2.0 + # underperformance signal
|
||||
df.get("minutes_trend", 0) * 0.5 + # trending up
|
||||
(1.0 / (df.get("market_value", 1) + 1)) * 10 # cheap
|
||||
)
|
||||
|
||||
buy_low = df[df["buy_low_score"] > 0].sort_values("buy_low_score", ascending=False)
|
||||
|
||||
result = buy_low[[
|
||||
"name", "team", "role", "actual_fv_avg", "expected_fv",
|
||||
"fv_divergence", "buy_low_score", "market_value",
|
||||
]].copy()
|
||||
|
||||
result["recommendation"] = "BUY-LOW"
|
||||
result["confidence"] = pd.cut(
|
||||
result["buy_low_score"],
|
||||
bins=[-np.inf, 1, 3, 5, np.inf],
|
||||
labels=["Low", "Medium", "High", "Very High"],
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"Found {len(result)} buy-low candidates "
|
||||
f"(avg divergence: {result['fv_divergence'].mean():.2f})"
|
||||
)
|
||||
return result
|
||||
|
||||
def analyze_sell_high(
|
||||
self, players_df: pd.DataFrame, min_minutes: int = 180
|
||||
) -> pd.DataFrame:
|
||||
"""Identify sell-high candidates.
|
||||
|
||||
Criteria:
|
||||
- Actual output exceeds xG/xA by large margin
|
||||
- Minutes trending down
|
||||
- High market value vs projection
|
||||
"""
|
||||
df = players_df.copy()
|
||||
df = df[df["minutes"] >= min_minutes]
|
||||
|
||||
if "xg" not in df.columns:
|
||||
return pd.DataFrame()
|
||||
|
||||
df["expected_fv"] = df.apply(
|
||||
lambda r: self.compute_expected_output(
|
||||
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
|
||||
),
|
||||
axis=1,
|
||||
)
|
||||
|
||||
# Overperformance
|
||||
df["fv_divergence"] = df.get("actual_fv_avg", 6.0) - df["expected_fv"]
|
||||
|
||||
df["sell_high_score"] = (
|
||||
df["fv_divergence"] * 3.0 + # overperformance signal
|
||||
(df.get("minutes_trend", 0) * -0.5 if "minutes_trend" in df.columns else 0)
|
||||
)
|
||||
|
||||
sell_high = df[df["sell_high_score"] > 1].sort_values("sell_high_score", ascending=False)
|
||||
|
||||
result = sell_high[[
|
||||
"name", "team", "role", "actual_fv_avg", "expected_fv",
|
||||
"fv_divergence", "sell_high_score",
|
||||
]].copy()
|
||||
|
||||
result["recommendation"] = "SELL-HIGH"
|
||||
logger.info(f"Found {len(result)} sell-high candidates")
|
||||
return result
|
||||
|
||||
def full_transfer_report(self, players_df: pd.DataFrame) -> dict:
|
||||
"""Generate complete transfer market report."""
|
||||
buy = self.analyze_buy_low(players_df)
|
||||
sell = self.analyze_sell_high(players_df)
|
||||
|
||||
return {
|
||||
"buy_low": buy,
|
||||
"sell_high": sell,
|
||||
"summary": (
|
||||
f"Buy-low targets: {len(buy)} players identified. "
|
||||
f"Sell-high targets: {len(sell)} players identified."
|
||||
),
|
||||
}
|
||||
Reference in New Issue
Block a user