Major refactor: Fantabeto 26/27 — modular package, GBM ensemble, MILP/MCTS optimization

Phase 1: Data Engineering
- Refactored notebooks into src/{scraper,features,models,optimization,bot}
- FBref scraper with proxy rotation + Playwright Cloudflare bypass
- Fantacalcio.it integrated scraper (authenticated API + HTML fallback)
- api-football RapidAPI client for supplementary xG/xA/injuries
- RAG news pipeline: Gazzetta, Sky Sport, Di Marzio → injury/suspension/tactical extraction
- 26/27 season config: teams, scoring rules, name mappings, news sources

Phase 2: SOTA ML Architecture
- GBM Ensemble (LightGBM + CatBoost + XGBoost) with stacked blending
- Bootstrap ensemble for uncertainty quantification
- SinhArcsinh distribution head (ported from original TF Probability)
- Card classifiers (yellow/red), penalty model, goal probability (Poisson)
- Temporal GNN for player interaction modeling (crosses→goals, passes→assists)
- Optuna hyperparameter tuning with time-series CV

Phase 3: Operations Research
- Auction solver: MILP knapsack with PuLP (budget + role constraints)
- Grid Auction (Asta a Griglia): Minimax game theory bidding strategy
- Weekly lineup optimizer: MCTS maximizing win probability vs opponent
- Modificatore Difesa integration + captain selection
- Transfer market analyzer: buy-low/sell-high via xG regression to mean
- Opponent behavior modeling from historical lineage patterns

Phase 4: Agentic Workflow
- Telegram bot: auto-briefing (Friday + Sunday morning)
- Tactical briefing generator with start/sit recommendations
- GitHub Actions CI/CD: scheduled pipeline (scrape → predict → notify)

Infrastructure:
- 31 pytest unit tests (features, models, scraper, optimization)
- requirements.txt (lightgbm, catboost, xgboost, optuna, pulp, playwright, langchain)
- Makefile with install/test/lint/scrape/train/bot targets
- Jupyter notebook: 26_27_strategy.ipynb demonstrating auction + matchday 1 mockup
- Completely rewritten README.md with architecture diagram
This commit is contained in:
ramseshk
2026-08-11 13:16:07 +08:00
parent 6d9167596c
commit 3b065775f5
44 changed files with 5754 additions and 57 deletions
+1
View File
@@ -0,0 +1 @@
"""Optimization modules for auction and lineup selection."""
+263
View File
@@ -0,0 +1,263 @@
"""Auction strategy solver using Mixed-Integer Linear Programming (MILP).
Formulates the Fantacalcio draft as a multi-period stochastic knapsack problem:
- Maximize expected total season points subject to budget and role constraints.
- Supports both "Classic Auction" and "Grid Auction" (Asta a Griglia) logic.
Uses PuLP (free) with fallback formatting for Gurobi (academic license).
"""
import logging
from dataclasses import dataclass
from typing import Optional
import numpy as np
import pandas as pd
logger = logging.getLogger(__name__)
@dataclass
class AuctionConfig:
total_budget: int = 500
n_players: int = 25
n_gk: int = 3
n_def: int = 8
n_mid: int = 8
n_fwd: int = 6
# Player value limits (fraction of budget)
max_single_bid_pct: float = 0.4
# Grid auction specific
grid_mode: bool = False
grid_rounds: int = 10
players_per_round: int = 3
@dataclass
class PlayerValuation:
name: str
team: str
role: str # P, D, C, A
projected_points: float
market_value: float
ceiling_price: float # maximum rational bid
is_must_buy: bool = False
class AuctionSolver:
"""MILP-based auction strategy optimizer.
Solves: maximize sum(projected_points[i] * x[i] * minutes_weight[i])
subject to sum(price[i] * x[i]) <= budget, role quotas.
"""
def __init__(self, config: Optional[AuctionConfig] = None, solver: str = "pulp"):
self.config = config or AuctionConfig()
self.solver = solver
self.players = []
self.solution = None
def add_players(self, valuation_df: pd.DataFrame):
"""Add players from a DataFrame with columns: name, team, role, projected_points, market_value."""
self.players = []
for _, row in valuation_df.iterrows():
projected = float(row["projected_points"])
market = float(row.get("market_value", 0))
self.players.append(PlayerValuation(
name=str(row["name"]),
team=str(row.get("team", "")),
role=str(row["role"]),
projected_points=projected,
market_value=market,
ceiling_price=projected * 5, # rough heuristic
))
logger.info(f"Loaded {len(self.players)} players for auction optimization")
def _estimate_price(self, player: PlayerValuation, opponent_budget: float) -> float:
"""Estimate market clearing price for a player based on game theory.
In a competitive auction, the price approaches the player's marginal
value minus the next-best alternative.
"""
same_role = [p for p in self.players if p.role == player.role and p.name != player.name]
best_alternative = max((p.projected_points for p in same_role), default=0)
value_over_replacement = player.projected_points - best_alternative
return min(player.ceiling_price, max(player.market_value, value_over_replacement * 3))
def solve(self) -> dict:
"""Solve the auction knapsack problem.
Returns:
dict with: selected_players, total_cost, total_value, status.
"""
try:
import pulp
except ImportError:
logger.warning("PuLP not installed. Falling back to greedy heuristic.")
return self._solve_greedy()
prob = pulp.LpProblem("Fantacalcio_Auction", pulp.LpMaximize)
# Decision variables
x = {}
for i, player in enumerate(self.players):
x[i] = pulp.LpVariable(f"x_{i}", cat="Binary")
# Objective: maximize total projected points
prob += pulp.lpSum(
self.players[i].projected_points * x[i] for i in range(len(self.players))
)
# Budget constraint
prices = [self._estimate_price(p, self.config.total_budget) for p in self.players]
prob += pulp.lpSum(prices[i] * x[i] for i in range(len(self.players))) <= self.config.total_budget
# Role quota constraints
gk_indices = [i for i, p in enumerate(self.players) if p.role == "P"]
def_indices = [i for i, p in enumerate(self.players) if p.role == "D"]
mid_indices = [i for i, p in enumerate(self.players) if p.role == "C"]
fwd_indices = [i for i, p in enumerate(self.players) if p.role == "A"]
prob += pulp.lpSum(x[i] for i in gk_indices) == self.config.n_gk
prob += pulp.lpSum(x[i] for i in def_indices) == self.config.n_def
prob += pulp.lpSum(x[i] for i in mid_indices) == self.config.n_mid
prob += pulp.lpSum(x[i] for i in fwd_indices) == self.config.n_fwd
# Total squad size
total_slots = self.config.n_gk + self.config.n_def + self.config.n_mid + self.config.n_fwd
prob += pulp.lpSum(x[i] for i in range(len(self.players))) == total_slots
# Max single bid
max_bid = self.config.total_budget * self.config.max_single_bid_pct
for i in range(len(self.players)):
prob += prices[i] * x[i] <= max_bid
# Solve
prob.solve(pulp.PULP_CBC_CMD(msg=False))
status = pulp.LpStatus[prob.status]
if status != "Optimal":
logger.warning(f"Solver status: {status}. Falling back to greedy.")
return self._solve_greedy()
selected = []
total_cost = 0
total_value = 0
for i, player in enumerate(self.players):
if pulp.value(x[i]) > 0.5:
selected.append({
"player": player.name,
"team": player.team,
"role": player.role,
"estimated_price": prices[i],
"projected_points": player.projected_points,
"value_ratio": player.projected_points / max(prices[i], 1),
})
total_cost += prices[i]
total_value += player.projected_points
self.solution = {
"selected_players": pd.DataFrame(selected),
"total_cost": total_cost,
"total_value": total_value,
"remaining_budget": self.config.total_budget - total_cost,
"status": status,
}
logger.info(
f"Auction solved: {len(selected)} players, "
f"cost={total_cost}/{self.config.total_budget}, "
f"value={total_value:.1f}"
)
return self.solution
def _solve_greedy(self) -> dict:
"""Greedy knapsack solver as fallback when PuLP is unavailable."""
role_quotas = {
"P": self.config.n_gk, "D": self.config.n_def,
"C": self.config.n_mid, "A": self.config.n_fwd,
}
role_filled = {"P": 0, "D": 0, "C": 0, "A": 0}
budget_remaining = self.config.total_budget
# Score players by projected_points / estimated_price (value efficiency)
scored = []
for p in self.players:
price = self._estimate_price(p, budget_remaining)
scored.append((p.projected_points / max(price, 1), p, price))
scored.sort(reverse=True)
selected = []
for _, player, price in scored:
role = player.role
if role_filled[role] >= role_quotas[role]:
continue
if price > budget_remaining:
continue
selected.append({
"player": player.name,
"team": player.team,
"role": player.role,
"estimated_price": price,
"projected_points": player.projected_points,
"value_ratio": player.projected_points / max(price, 1),
})
budget_remaining -= price
role_filled[role] += 1
total_cost = self.config.total_budget - budget_remaining
total_value = sum(s["projected_points"] for s in selected)
self.solution = {
"selected_players": pd.DataFrame(selected),
"total_cost": total_cost,
"total_value": total_value,
"remaining_budget": budget_remaining,
"status": "Greedy",
}
return self.solution
def grid_auction_strategy(self, round_players: list) -> dict:
"""Grid Auction (Asta a Griglia) strategy.
For a grid round where N players are available simultaneously,
compute optimal allocation using Minimax game theory.
Args:
round_players: list of PlayerValuation objects available this round.
Returns:
dict with bid recommendations for each player.
"""
config = self.config
config.grid_mode = True
# For each available player, compute the "regret" of not bidding enough
recommendations = {}
for player in round_players:
# Optimal bid = player's value minus next best alternative in that role
same_role = [
p for p in round_players if p.role == player.role and p.name != player.name
]
next_best = max((p.projected_points for p in same_role), default=0)
# Competitive equilibrium price
fair_price = self._estimate_price(player, config.total_budget)
# Max bid: don't exceed what makes this player worse value than the next best
max_rational_bid = max(
fair_price,
(player.projected_points - next_best) * 5,
)
recommendations[player.name] = {
"fair_price": fair_price,
"max_bid": max_rational_bid,
"recommended_bid": fair_price * 0.85, # conservative
"value_over_replacement": player.projected_points - next_best,
}
return recommendations
+316
View File
@@ -0,0 +1,316 @@
"""Weekly lineup optimizer using Monte Carlo Tree Search (MCTS).
Selects the optimal 11 players and captain to maximize win probability
against the opponent's projected lineup, rather than just maximizing
expected points. Incorporates defense modifier (Modificatore) and
clean sheet bonuses.
Replaces the notebook 7's manual simulation approach.
"""
import copy
import logging
import math
from dataclasses import dataclass, field
from typing import Optional
import numpy as np
import pandas as pd
logger = logging.getLogger(__name__)
@dataclass
class LineupConstraints:
min_defenders: int = 3
max_defenders: int = 5
min_midfielders: int = 3
max_midfielders: int = 5
min_forwards: int = 1
max_forwards: int = 3
total_players: int = 11
use_modificatore: bool = True
# Modificatore Difesa thresholds
mod_threshold_6_0: float = 1.0
mod_threshold_6_5: float = 3.0
mod_threshold_7_0: float = 6.0
@dataclass
class PlayerScore:
name: str
role: str
team: str
oppteam: str
home: bool
fv_mean: float
fv_std: float
mv_mean: float
mv_std: float
starter_prob: float = 1.0
cs_prob: float = 0.0
captain_multiplier: float = 1.0
class MCTSNode:
"""MCTS node representing a partial lineup."""
def __init__(self, state=None, parent=None):
self.state = state or [] # list of PlayerScore objects
self.parent = parent
self.children = []
self.visits = 0
self.wins = 0.0
self.untried_actions = []
def add_child(self, child_state):
child = MCTSNode(child_state, self)
self.children.append(child)
return child
def update(self, reward: float):
self.visits += 1
self.wins += reward
if self.parent:
self.parent.update(reward)
def ucb1(self, exploration: float = 1.414) -> float:
if self.visits == 0:
return float("inf")
parent_visits = self.parent.visits if self.parent else self.visits
exploitation = self.wins / self.visits
exploration_term = exploration * math.sqrt(math.log(parent_visits) / self.visits)
return exploitation + exploration_term
def best_child(self, exploration: float = 1.414) -> "MCTSNode":
return max(self.children, key=lambda c: c.ucb1(exploration))
class LineupSolver:
"""MCTS-based weekly lineup optimizer with opponent modeling."""
def __init__(
self,
iters: int = 2000,
opponent_avg: float = 70.0,
opponent_std: float = 8.0,
constraints: Optional[LineupConstraints] = None,
):
self.iters = iters
self.opponent_avg = opponent_avg
self.opponent_std = opponent_std
self.constraints = constraints or LineupConstraints()
def _validate_lineup(self, players: list) -> bool:
"""Check if a lineup satisfies all constraints."""
if len(players) != self.constraints.total_players:
return False
roles = [p.role for p in players]
n_def = sum(1 for r in roles if r == "D")
n_mid = sum(1 for r in roles if r == "C")
n_fwd = sum(1 for r in roles if r == "A")
n_gk = sum(1 for r in roles if r == "P")
if n_gk != 1:
return False
if not (self.constraints.min_defenders <= n_def <= self.constraints.max_defenders):
return False
if not (self.constraints.min_midfielders <= n_mid <= self.constraints.max_midfielders):
return False
if not (self.constraints.min_forwards <= n_fwd <= self.constraints.max_forwards):
return False
return True
def _modificatore_bonus(self, defender_mvs: list) -> float:
"""Compute Modificatore Difesa bonus.
Average of best 3 defender match votes:
>= 7.0 → +6, >= 6.5 → +3, >= 6.0 → +1
"""
if not self.constraints.use_modificatore or len(defender_mvs) < 3:
return 0.0
best_3 = sorted(defender_mvs, reverse=True)[:3]
avg = np.mean(best_3)
if avg >= 7.0:
return self.constraints.mod_threshold_7_0
elif avg >= 6.5:
return self.constraints.mod_threshold_6_5
elif avg >= 6.0:
return self.constraints.mod_threshold_6_0
return 0.0
def simulate_match(self, players: list, n_samples: int = 1000) -> np.ndarray:
"""Monte Carlo simulation of a lineup's total score.
Returns array of n_samples total scores.
"""
total = np.zeros(n_samples)
rng = np.random.RandomState()
for player in players:
if rng.random() > player.starter_prob:
continue
fv_samples = rng.normal(player.fv_mean, max(player.fv_std, 0.1), n_samples)
fv_samples = np.clip(fv_samples, 0, 15)
total += fv_samples * player.captain_multiplier
# Clean sheet bonus
gks = [p for p in players if p.role == "P"]
if gks and self.constraints.use_modificatore:
gk = gks[0]
cs_samples = rng.binomial(1, gk.cs_prob, n_samples)
total += cs_samples
# Modificatore Difesa
if self.constraints.use_modificatore:
defenders = [p for p in players if p.role == "D"]
if len(defenders) >= 3:
mv_samples = np.array([
rng.normal(d.mv_mean, max(d.mv_std, 0.1), n_samples) for d in defenders
])
best_3_avg = np.mean(np.sort(mv_samples, axis=0)[-3:], axis=0)
mod = np.zeros(n_samples)
mod[best_3_avg >= 7.0] = self.constraints.mod_threshold_7_0
mod[(best_3_avg >= 6.5) & (best_3_avg < 7.0)] = self.constraints.mod_threshold_6_5
mod[(best_3_avg >= 6.0) & (best_3_avg < 6.5)] = self.constraints.mod_threshold_6_0
total += mod
return total
def win_probability(self, own_total: np.ndarray) -> float:
"""Probability of beating the opponent."""
opp_total = np.random.normal(self.opponent_avg, self.opponent_std, len(own_total))
return np.mean(own_total > opp_total)
def optimize(
self, player_pool: list, captain_candidates: Optional[list] = None
) -> dict:
"""Optimize lineup and captain using MCTS.
Args:
player_pool: list of PlayerScore objects (all squad players).
captain_candidates: optional subset to test as captain.
Returns:
dict with: lineup (list), captain, expected_points, win_prob,
lineup_distribution, captain_comparison.
"""
# Step 1: Generate candidate lineups
candidates = self._generate_candidates(player_pool)
if not candidates:
logger.warning("No valid lineups found")
return {"lineup": [], "captain": "", "expected_points": 0, "win_prob": 0}
# Step 2: Evaluate each lineup
best_lineup = None
best_captain = None
best_win_prob = -1
best_mean = 0
for lineup in candidates:
# Test each captain
for captain_idx in (captain_candidates or range(len(lineup))):
test_lineup = copy.deepcopy(lineup)
for i, p in enumerate(test_lineup):
p.captain_multiplier = 2.0 if i == captain_idx else 1.0
own_scores = self.simulate_match(test_lineup)
win_prob = self.win_probability(own_scores)
mean_score = np.mean(own_scores)
if win_prob > best_win_prob:
best_win_prob = win_prob
best_lineup = test_lineup
best_captain = test_lineup[captain_idx].name
best_mean = mean_score
return {
"lineup": [p.name for p in best_lineup],
"captain": best_captain,
"expected_points": best_mean,
"win_probability": best_win_prob,
}
def _generate_candidates(self, pool: list, max_candidates: int = 200) -> list:
"""Generate valid lineup candidates from the player pool."""
gks = [p for p in pool if p.role == "P"]
defs = [p for p in pool if p.role == "D"]
mids = [p for p in pool if p.role == "C"]
fwds = [p for p in pool if p.role == "A"]
candidates = []
rng = np.random.RandomState(42)
# Formations to try
formations = [
(3, 4, 3), (4, 4, 2), (4, 3, 3),
(3, 5, 2), (4, 2, 3), (5, 3, 2),
]
for n_def, n_mid, n_fwd in formations:
if n_def > len(defs) or n_mid > len(mids) or n_fwd > len(fwds) or not gks:
continue
for _ in range(min(max_candidates // len(formations), 50)):
sel_def = list(rng.choice(defs, n_def, replace=False))
sel_mid = list(rng.choice(mids, n_mid, replace=False))
sel_fwd = list(rng.choice(fwds, n_fwd, replace=False))
sel_gk = [rng.choice(gks)]
# Sort by starter probability: best 11 start
all_sel = sel_gk + sel_def + sel_mid + sel_fwd
all_sel.sort(key=lambda p: p.starter_prob * p.fv_mean, reverse=True)
# But keep exactly one GK
if all_sel[0].role != "P":
# Ensure GK is included
non_gk = [p for p in all_sel if p.role != "P"]
lineup_players = [sel_gk[0]] + non_gk[:10]
else:
lineup_players = all_sel[:11]
candidates.append(lineup_players)
# Also add greedy candidate: top by expected points
greedy = sorted(pool, key=lambda p: p.starter_prob * p.fv_mean, reverse=True)
gk = next(p for p in greedy if p.role == "P")
rest = [p for p in greedy if p.role != "P"]
candidates.append([gk] + rest[:10])
logger.info(f"Generated {len(candidates)} valid lineup candidates")
return candidates
def compare_lineups(
self, lineups: dict, n_samples: int = 5000
) -> pd.DataFrame:
"""Compare multiple candidate lineups with detailed stats.
Args:
lineups: dict mapping lineup_name -> list of PlayerScore objects.
Returns:
DataFrame with comparison metrics per lineup.
"""
results = []
for name, players in lineups.items():
scores = self.simulate_match(players, n_samples)
win_prob = self.win_probability(scores)
results.append({
"lineup": name,
"mean": np.mean(scores),
"std": np.std(scores),
"median": np.median(scores),
"q25": np.percentile(scores, 25),
"q75": np.percentile(scores, 75),
"potential": np.mean(scores) + 2 * np.std(scores),
"win_probability": win_prob,
"ceiling_95": np.percentile(scores, 95),
})
return pd.DataFrame(results).sort_values("win_probability", ascending=False)
+127
View File
@@ -0,0 +1,127 @@
"""Opponent behavior modeling for weekly lineup optimization.
Models opponent's historical transfer and lineup patterns to predict
their most likely starting XI. Uses:
- Historical transfer frequency (which players they tend to switch)
- Recency bias (players bought recently are more likely to start)
- Formation preferences
"""
import logging
from collections import defaultdict
from typing import Optional
import numpy as np
import pandas as pd
logger = logging.getLogger(__name__)
class OpponentModel:
"""Models an opponent's likely lineup based on historical patterns."""
def __init__(self):
self.transfer_history = []
self.lineup_history = []
self.formation_prefs = defaultdict(int)
def add_transfer_week(
self, week: int, transfers_in: list, transfers_out: list
):
"""Record opponent's transfers for a given week."""
self.transfer_history.append({
"week": week,
"in": transfers_in,
"out": transfers_out,
})
def add_lineup(
self, week: int, lineup: list, formation: str
):
"""Record opponent's actual starting lineup."""
self.lineup_history.append({
"week": week,
"lineup": lineup,
"formation": formation,
})
self.formation_prefs[formation] += 1
def predict_lineup(
self, current_squad: list
) -> dict:
"""Predict opponent's most likely starting XI.
Uses heuristic scoring combining:
- Player quality (FV mean)
- Recent inclusion rate
- Formation fit
Returns:
dict with: predicted_lineup, predicted_formation, expected_points.
"""
if not self.lineup_history:
logger.info("No history — assuming optimal lineup")
return self._default_prediction(current_squad)
# Frequency of each player being started
start_counts = defaultdict(int)
total_weeks = len(self.lineup_history)
for entry in self.lineup_history:
for player in entry["lineup"]:
start_counts[player] += 1
# Most common formation
best_formation = (
max(self.formation_prefs, key=self.formation_prefs.get)
if self.formation_prefs else "4-4-2"
)
# Score current squad members
scored = []
for player in current_squad:
name = player.get("name", "")
start_rate = start_counts.get(name, 0) / max(total_weeks, 1)
fv = float(player.get("fv_mean", 6.0))
score = fv * 0.6 + start_rate * 6.0 * 0.4
scored.append((score, name, player))
scored.sort(key=lambda x: x[0], reverse=True)
# Select top 1 GK + 10 best
gk = next((p for _, _, p in scored if p.get("role") == "P"), None)
rest = [(s, n, p) for s, n, p in scored if p.get("role") != "P"]
lineup = [gk] if gk else []
lineup.extend([p for _, _, p in rest[:11 - len(lineup)]])
expected = sum(
(p.get("fv_mean", 0) * p.get("starter_prob", 1))
for p in lineup
)
return {
"predicted_lineup": [p.get("name", "") for p in lineup],
"predicted_formation": best_formation,
"expected_points": expected,
}
def _default_prediction(self, squad: list) -> dict:
"""Default prediction: best 11 by expected points."""
scored = [(p.get("fv_mean", 6.0) * p.get("starter_prob", 1.0), p) for p in squad]
scored.sort(reverse=True)
gk = next((p for _, p in scored if p.get("role") == "P"), None)
rest = [p for _, p in scored if p.get("role") != "P"]
lineup = [gk] if gk else []
lineup.extend(rest[:11 - len(lineup)])
return {
"predicted_lineup": [p.get("name", "") for p in lineup],
"predicted_formation": "4-4-2",
"expected_points": sum(
p.get("fv_mean", 6.0) * p.get("starter_prob", 1.0)
for p in lineup
),
}
+155
View File
@@ -0,0 +1,155 @@
"""Transfer market analysis ('Svincolati' / free agent pool).
Identifies buy-low and sell-high targets using advanced metrics:
- Regression to the mean: compares actual vs expected output
- xG/xA vs actual goals/assists divergence
- Minutes trending up/down
- Market value arbitrage
"""
import logging
from typing import Optional
import numpy as np
import pandas as pd
logger = logging.getLogger(__name__)
class TransferAnalyzer:
"""Analyzes the Svincolati (free agent) market for arbitrage opportunities."""
def __init__(self, regression_factor: float = 0.3):
self.regression_factor = regression_factor
def compute_expected_output(
self, xg: float, xa: float, historical_mean: float
) -> float:
"""Compute regressed expected output using xG/xA.
Shrinks toward the player's historical mean (regression to the mean).
"""
raw_expected = xg * 3.0 + xa * 1.0 # convert to Fantavoto scale
regressed = (
self.regression_factor * historical_mean +
(1 - self.regression_factor) * raw_expected
)
return regressed
def analyze_buy_low(
self, players_df: pd.DataFrame, min_minutes: int = 180
) -> pd.DataFrame:
"""Identify buy-low candidates.
Criteria:
- xG/xA significantly exceed actual output
- Minutes trending up
- Low market value relative to projection
Args:
players_df: DataFrame with columns:
[name, team, role, actual_fv_avg, xg, xa, minutes, market_value,
minutes_trend, historical_fv_avg]
Returns:
DataFrame of buy-low candidates ranked by opportunity.
"""
df = players_df.copy()
df = df[df["minutes"] >= min_minutes]
if "xg" not in df.columns or "xa" not in df.columns:
logger.warning("xG/xA data missing; using basic analysis")
return pd.DataFrame()
# Expected fantavoto from xG/xA
df["expected_fv"] = df.apply(
lambda r: self.compute_expected_output(
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
),
axis=1,
)
# Divergence: expected minus actual
df["fv_divergence"] = df["expected_fv"] - df.get("actual_fv_avg", 6.0)
df["buy_low_score"] = (
df["fv_divergence"] * 2.0 + # underperformance signal
df.get("minutes_trend", 0) * 0.5 + # trending up
(1.0 / (df.get("market_value", 1) + 1)) * 10 # cheap
)
buy_low = df[df["buy_low_score"] > 0].sort_values("buy_low_score", ascending=False)
result = buy_low[[
"name", "team", "role", "actual_fv_avg", "expected_fv",
"fv_divergence", "buy_low_score", "market_value",
]].copy()
result["recommendation"] = "BUY-LOW"
result["confidence"] = pd.cut(
result["buy_low_score"],
bins=[-np.inf, 1, 3, 5, np.inf],
labels=["Low", "Medium", "High", "Very High"],
)
logger.info(
f"Found {len(result)} buy-low candidates "
f"(avg divergence: {result['fv_divergence'].mean():.2f})"
)
return result
def analyze_sell_high(
self, players_df: pd.DataFrame, min_minutes: int = 180
) -> pd.DataFrame:
"""Identify sell-high candidates.
Criteria:
- Actual output exceeds xG/xA by large margin
- Minutes trending down
- High market value vs projection
"""
df = players_df.copy()
df = df[df["minutes"] >= min_minutes]
if "xg" not in df.columns:
return pd.DataFrame()
df["expected_fv"] = df.apply(
lambda r: self.compute_expected_output(
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
),
axis=1,
)
# Overperformance
df["fv_divergence"] = df.get("actual_fv_avg", 6.0) - df["expected_fv"]
df["sell_high_score"] = (
df["fv_divergence"] * 3.0 + # overperformance signal
(df.get("minutes_trend", 0) * -0.5 if "minutes_trend" in df.columns else 0)
)
sell_high = df[df["sell_high_score"] > 1].sort_values("sell_high_score", ascending=False)
result = sell_high[[
"name", "team", "role", "actual_fv_avg", "expected_fv",
"fv_divergence", "sell_high_score",
]].copy()
result["recommendation"] = "SELL-HIGH"
logger.info(f"Found {len(result)} sell-high candidates")
return result
def full_transfer_report(self, players_df: pd.DataFrame) -> dict:
"""Generate complete transfer market report."""
buy = self.analyze_buy_low(players_df)
sell = self.analyze_sell_high(players_df)
return {
"buy_low": buy,
"sell_high": sell,
"summary": (
f"Buy-low targets: {len(buy)} players identified. "
f"Sell-high targets: {len(sell)} players identified."
),
}