Major refactor: Fantabeto 26/27 — modular package, GBM ensemble, MILP/MCTS optimization
Phase 1: Data Engineering
- Refactored notebooks into src/{scraper,features,models,optimization,bot}
- FBref scraper with proxy rotation + Playwright Cloudflare bypass
- Fantacalcio.it integrated scraper (authenticated API + HTML fallback)
- api-football RapidAPI client for supplementary xG/xA/injuries
- RAG news pipeline: Gazzetta, Sky Sport, Di Marzio → injury/suspension/tactical extraction
- 26/27 season config: teams, scoring rules, name mappings, news sources
Phase 2: SOTA ML Architecture
- GBM Ensemble (LightGBM + CatBoost + XGBoost) with stacked blending
- Bootstrap ensemble for uncertainty quantification
- SinhArcsinh distribution head (ported from original TF Probability)
- Card classifiers (yellow/red), penalty model, goal probability (Poisson)
- Temporal GNN for player interaction modeling (crosses→goals, passes→assists)
- Optuna hyperparameter tuning with time-series CV
Phase 3: Operations Research
- Auction solver: MILP knapsack with PuLP (budget + role constraints)
- Grid Auction (Asta a Griglia): Minimax game theory bidding strategy
- Weekly lineup optimizer: MCTS maximizing win probability vs opponent
- Modificatore Difesa integration + captain selection
- Transfer market analyzer: buy-low/sell-high via xG regression to mean
- Opponent behavior modeling from historical lineage patterns
Phase 4: Agentic Workflow
- Telegram bot: auto-briefing (Friday + Sunday morning)
- Tactical briefing generator with start/sit recommendations
- GitHub Actions CI/CD: scheduled pipeline (scrape → predict → notify)
Infrastructure:
- 31 pytest unit tests (features, models, scraper, optimization)
- requirements.txt (lightgbm, catboost, xgboost, optuna, pulp, playwright, langchain)
- Makefile with install/test/lint/scrape/train/bot targets
- Jupyter notebook: 26_27_strategy.ipynb demonstrating auction + matchday 1 mockup
- Completely rewritten README.md with architecture diagram
This commit is contained in:
@@ -0,0 +1,2 @@
|
||||
"""Fantabeto – ML-powered Fantacalcio prediction and optimization engine. 2026/27 edition."""
|
||||
__version__ = "2.0.0"
|
||||
@@ -0,0 +1 @@
|
||||
"""Telegram/Discord bot and briefing generator."""
|
||||
@@ -0,0 +1,158 @@
|
||||
"""Tactical briefing generator.
|
||||
|
||||
Produces human-readable, data-driven matchday briefings explaining:
|
||||
- Why specific players should start/sit
|
||||
- Captain analysis with probability curves
|
||||
- Opposition weakness exploitation recommendations
|
||||
- Bench risk warnings
|
||||
"""
|
||||
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BriefingGenerator:
|
||||
"""Generates tactical briefings from ML predictions and optimization results."""
|
||||
|
||||
def __init__(self, squad_name: str = "Fantabeto FC"):
|
||||
self.squad_name = squad_name
|
||||
|
||||
def generate_matchday_briefing(
|
||||
self,
|
||||
matchday: int,
|
||||
recommended_lineup: list,
|
||||
bench_alternatives: list,
|
||||
captain: str,
|
||||
expected_points: float,
|
||||
win_probability: float,
|
||||
opponent_model: Optional[dict] = None,
|
||||
news_entities: Optional[list] = None,
|
||||
) -> str:
|
||||
"""Generate a comprehensive matchday briefing.
|
||||
|
||||
Returns Markdown-formatted string suitable for Telegram/Discord.
|
||||
"""
|
||||
now = datetime.now().strftime("%A %d %B %Y, %H:%M CET")
|
||||
|
||||
lines = [
|
||||
f"⚽ *{self.squad_name} — Matchday {matchday} Briefing*",
|
||||
f"_{now}_",
|
||||
"",
|
||||
]
|
||||
|
||||
if news_entities:
|
||||
lines.append("📰 *Key News:*")
|
||||
for entity in news_entities[:5]:
|
||||
lines.append(f" • {entity}")
|
||||
lines.append("")
|
||||
|
||||
lines.extend([
|
||||
"🧤 *Recommended Starting XI:*",
|
||||
"",
|
||||
])
|
||||
|
||||
# Formation display
|
||||
roles = [p.get("role", "?") for p in recommended_lineup[:11]]
|
||||
n_def = sum(1 for r in roles if r == "D")
|
||||
n_mid = sum(1 for r in roles if r == "C")
|
||||
n_fwd = sum(1 for r in roles if r == "A")
|
||||
lines.append(f"*Formation: {n_def}-{n_mid}-{n_fwd}*")
|
||||
lines.append("")
|
||||
|
||||
for i, player in enumerate(recommended_lineup[:11]):
|
||||
name = player.get("name", f"Player {i}")
|
||||
role_emoji = {"P": "🧤", "D": "🛡", "C": "⚙", "A": "⚡"}.get(
|
||||
player.get("role", ""), "⚪"
|
||||
)
|
||||
fv = player.get("fv_mean", 0)
|
||||
fv_std = player.get("fv_std", 0)
|
||||
starter_pct = player.get("starter_prob", 1.0) * 100
|
||||
|
||||
cap_mark = " ★ CAPTAIN" if name == captain else ""
|
||||
risk_note = ""
|
||||
if starter_pct < 70:
|
||||
risk_note = " ⚠️ BENCH RISK"
|
||||
elif starter_pct < 85:
|
||||
risk_note = " ⚡ DOUBT"
|
||||
|
||||
lines.append(
|
||||
f"{role_emoji} *{name}* — FV: {fv:.1f} ± {fv_std:.2f} "
|
||||
f"(Start: {starter_pct:.0f}%){cap_mark}{risk_note}"
|
||||
)
|
||||
|
||||
lines.extend([
|
||||
"",
|
||||
"📊 *Team Projection:*",
|
||||
f" Expected Points: *{expected_points:.1f}*",
|
||||
f" Win Probability: *{win_probability:.1%}*",
|
||||
"",
|
||||
])
|
||||
|
||||
# Bench analysis
|
||||
if bench_alternatives:
|
||||
lines.append("🔄 *Bench Watch:*")
|
||||
for p in bench_alternatives[:5]:
|
||||
name = p.get("name", "")
|
||||
fv = p.get("fv_mean", 0)
|
||||
reason = p.get("bench_reason", "Rotation risk")
|
||||
lines.append(f" • {name} (FV: {fv:.1f}) — {reason}")
|
||||
lines.append("")
|
||||
|
||||
# Opponent analysis
|
||||
if opponent_model:
|
||||
lines.extend([
|
||||
"🎯 *Opponent Analysis:*",
|
||||
f" Formation: {opponent_model.get('predicted_formation', 'Unknown')}",
|
||||
f" Expected Points: {opponent_model.get('expected_points', 0):.1f}",
|
||||
"",
|
||||
])
|
||||
|
||||
# Captain analysis
|
||||
lines.extend([
|
||||
"💡 *Captain Decision:*",
|
||||
f" Selected: *{captain}*",
|
||||
"",
|
||||
])
|
||||
|
||||
lines.append("_Generated by Fantabeto 26/27 ML Engine_")
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
def generate_auction_briefing(
|
||||
self, auction_result: dict, budget: int = 500
|
||||
) -> str:
|
||||
"""Generate auction strategy briefing."""
|
||||
lines = [
|
||||
f"💰 *{self.squad_name} — Auction Strategy 2026/27*",
|
||||
f"Budget: {budget} crediti",
|
||||
"",
|
||||
]
|
||||
|
||||
if "selected_players" in auction_result:
|
||||
df = auction_result["selected_players"]
|
||||
lines.append("*Target Squad:*")
|
||||
lines.append("")
|
||||
for role_name, role_code in [
|
||||
("Portieri", "P"), ("Difensori", "D"),
|
||||
("Centrocampisti", "C"), ("Attaccanti", "A"),
|
||||
]:
|
||||
role_df = df[df["role"] == role_code]
|
||||
if role_df.empty:
|
||||
continue
|
||||
lines.append(f"*{role_name}:*")
|
||||
for _, p in role_df.iterrows():
|
||||
lines.append(
|
||||
f" • {p['player']} — Max bid: {p['estimated_price']:.0f} "
|
||||
f"(Proj: {p['projected_points']:.1f})"
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
lines.append(f"Total spend: {auction_result['total_cost']:.0f} crediti")
|
||||
lines.append(f"Reserve: {auction_result.get('remaining_budget', 0):.0f} crediti")
|
||||
|
||||
return "\n".join(lines)
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Telegram bot for Fantabeto 2026/27.
|
||||
|
||||
Runs automatically on Friday (pre-matchday) and Sunday morning:
|
||||
1. Fetches latest training reports and press conferences.
|
||||
2. Updates ML predictions.
|
||||
3. Runs the MILP/MCTS lineup optimizer.
|
||||
4. Outputs a tactical briefing with start/sit recommendations.
|
||||
|
||||
Requires TELEGRAM_BOT_TOKEN and TELEGRAM_CHAT_ID env vars.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class TelegramBot:
|
||||
"""Telegram bot for Fantabeto predictions and lineup optimization."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
token: Optional[str] = None,
|
||||
chat_id: Optional[str] = None,
|
||||
):
|
||||
self.token = token or os.getenv("TELEGRAM_BOT_TOKEN")
|
||||
self.chat_id = chat_id or os.getenv("TELEGRAM_CHAT_ID")
|
||||
self._base_url = f"https://api.telegram.org/bot{self.token}"
|
||||
|
||||
def _send(self, text: str, parse_mode: str = "Markdown") -> bool:
|
||||
"""Send a message via Telegram API."""
|
||||
if not self.token or not self.chat_id:
|
||||
logger.warning("Telegram not configured. Skipping send.")
|
||||
return False
|
||||
|
||||
import requests
|
||||
|
||||
url = f"{self._base_url}/sendMessage"
|
||||
payload = {
|
||||
"chat_id": self.chat_id,
|
||||
"text": text[:4096], # Telegram limit
|
||||
"parse_mode": parse_mode,
|
||||
}
|
||||
try:
|
||||
resp = requests.post(url, json=payload, timeout=10)
|
||||
resp.raise_for_status()
|
||||
return True
|
||||
except Exception as e:
|
||||
logger.error(f"Telegram send failed: {e}")
|
||||
return False
|
||||
|
||||
def send_briefing(self, briefing: str):
|
||||
"""Send tactical briefing to the configured chat."""
|
||||
self._send(briefing)
|
||||
|
||||
def send_prediction_update(self, predictions: dict):
|
||||
"""Send matchday predictions summary."""
|
||||
now = datetime.now().strftime("%Y-%m-%d %H:%M")
|
||||
|
||||
msg = f"*Fantabeto 26/27 — Prediction Update*\n"
|
||||
msg += f"Generated: {now}\n\n"
|
||||
|
||||
if "matchday" in predictions:
|
||||
msg += f"Matchday {predictions['matchday']}\n\n"
|
||||
|
||||
if "top_players" in predictions:
|
||||
msg += "*Top 10 Projected Players:*\n"
|
||||
for i, p in enumerate(predictions["top_players"][:10], 1):
|
||||
msg += f"{i}. *{p['name']}* ({p['team']}) — FV: {p['fv_mean']:.1f} ± {p['fv_std']:.2f}\n"
|
||||
msg += "\n"
|
||||
|
||||
if "recommended_lineup" in predictions:
|
||||
lineup = predictions["recommended_lineup"]
|
||||
msg += "*Recommended Starting XI:*\n"
|
||||
for p in lineup[:11]:
|
||||
msg += f"- {p.get('name', '?')} ({p.get('role', '?')})\n"
|
||||
msg += f"\nExpected points: {predictions.get('expected_points', 0):.1f}\n"
|
||||
msg += f"Win probability: {predictions.get('win_prob', 0):.1%}\n"
|
||||
|
||||
self._send(msg)
|
||||
|
||||
def send_error(self, error_msg: str):
|
||||
"""Send error notification."""
|
||||
self._send(f"*Fantabeto Error*\n{error_msg}")
|
||||
|
||||
def send_auction_recommendations(self, auction_result: dict):
|
||||
"""Send auction strategy recommendations."""
|
||||
msg = "*Auction Strategy — 2026/27 Draft*\n\n"
|
||||
|
||||
if "selected_players" in auction_result:
|
||||
df = auction_result["selected_players"]
|
||||
msg += "*Recommended Squad:*\n"
|
||||
for role in ["P", "D", "C", "A"]:
|
||||
role_names = {"P": "Portieri", "D": "Difensori", "C": "Centrocampisti", "A": "Attaccanti"}
|
||||
role_df = df[df["role"] == role]
|
||||
msg += f"\n*{role_names.get(role, role)} ({len(role_df)})*\n"
|
||||
for _, p in role_df.iterrows():
|
||||
msg += f"- {p['player']} — est. {p['estimated_price']:.0f} cr (proj: {p['projected_points']:.1f})\n"
|
||||
|
||||
msg += f"\nTotal cost: {auction_result['total_cost']:.0f}/{auction_result.get('remaining_budget', 0) + auction_result['total_cost']:.0f}\n"
|
||||
|
||||
self._send(msg)
|
||||
@@ -0,0 +1 @@
|
||||
"""Feature engineering modules."""
|
||||
@@ -0,0 +1,194 @@
|
||||
"""Advanced metrics for 2026/27 Fantacalcio feature engineering.
|
||||
|
||||
Adds:
|
||||
- Fatigue Index: rest days, midweek competition minutes, rolling workload.
|
||||
- Pitch/Field Tilt: advanced possession and territorial dominance metrics.
|
||||
- Weather/Pitch Degradation: historical weather context for away games.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class FatigueIndex:
|
||||
"""Computes player fatigue indicators from fixture and minute data."""
|
||||
|
||||
def __init__(self, decay_factor: float = 0.7):
|
||||
self.decay_factor = decay_factor
|
||||
|
||||
def rest_days(self, match_dates: pd.Series, prev_match_dates: pd.Series) -> np.ndarray:
|
||||
"""Days of rest between consecutive matches."""
|
||||
diffs = (pd.to_datetime(match_dates) - pd.to_datetime(prev_match_dates)).dt.days
|
||||
return diffs.fillna(7).clip(0, 14).values
|
||||
|
||||
def midweek_fatigue(
|
||||
self, player_minutes: np.ndarray, midweek_flags: np.ndarray
|
||||
) -> np.ndarray:
|
||||
"""Minutes played in midweek European competitions."""
|
||||
return player_minutes * midweek_flags
|
||||
|
||||
def rolling_workload(
|
||||
self, minutes_series: np.ndarray, window: int = 3
|
||||
) -> np.ndarray:
|
||||
"""Weighted rolling minutes (exponentially decaying) over last N games."""
|
||||
result = np.zeros(len(minutes_series))
|
||||
weights = np.array([self.decay_factor ** i for i in range(window, -1, -1)])
|
||||
for i in range(len(minutes_series)):
|
||||
if i >= window:
|
||||
result[i] = np.dot(minutes_series[i - window : i + 1], weights) / weights.sum()
|
||||
elif i > 0:
|
||||
w = weights[-i - 1 :]
|
||||
result[i] = np.dot(minutes_series[: i + 1], w) / w.sum()
|
||||
else:
|
||||
result[i] = minutes_series[i]
|
||||
return result
|
||||
|
||||
def compute_features(
|
||||
self, df: pd.DataFrame, date_col: str = "date",
|
||||
minutes_col: str = "minutes", midweek_col: str = "is_midweek"
|
||||
) -> pd.DataFrame:
|
||||
"""Add fatigue features to a match-level DataFrame."""
|
||||
df = df.copy()
|
||||
df = df.sort_values(date_col)
|
||||
|
||||
for player, group in df.groupby("player"):
|
||||
indices = group.index
|
||||
minutes = group[minutes_col].values
|
||||
|
||||
# Rest days
|
||||
dates = group[date_col]
|
||||
prev_dates = dates.shift(1).fillna(dates.iloc[0] - timedelta(days=7))
|
||||
df.loc[indices, "fatigue_rest_days"] = self.rest_days(dates, prev_dates)
|
||||
|
||||
# Midweek fatigue
|
||||
midweek = group[midweek_col].values if midweek_col in group.columns else np.zeros(len(group))
|
||||
df.loc[indices, "fatigue_midweek_minutes"] = self.midweek_fatigue(minutes, midweek)
|
||||
|
||||
# Rolling workload
|
||||
df.loc[indices, "fatigue_rolling_3"] = self.rolling_workload(minutes, 3)
|
||||
|
||||
return df
|
||||
|
||||
|
||||
class PitchTilt:
|
||||
"""Advanced possession and territory metrics."""
|
||||
|
||||
@staticmethod
|
||||
def pitch_tilt(
|
||||
final_third_touches: np.ndarray, opp_final_third_touches: np.ndarray
|
||||
) -> np.ndarray:
|
||||
"""Pitch Tilt = own final-third touches / (own + opp final-third touches)."""
|
||||
denom = final_third_touches + opp_final_third_touches
|
||||
denom = np.where(denom == 0, 1, denom)
|
||||
return final_third_touches / denom
|
||||
|
||||
@staticmethod
|
||||
def field_tilt(
|
||||
opp_half_passes: np.ndarray, opp_passes_in_own_half: np.ndarray
|
||||
) -> np.ndarray:
|
||||
"""Field Tilt = passes in opp half / (opp passes in own half + own)."""
|
||||
denom = opp_half_passes + opp_passes_in_own_half
|
||||
denom = np.where(denom == 0, 1, denom)
|
||||
return opp_half_passes / denom
|
||||
|
||||
@staticmethod
|
||||
def pressure_regain_efficiency(
|
||||
pressures_leading_to_turnover: np.ndarray, total_pressures: np.ndarray
|
||||
) -> np.ndarray:
|
||||
"""Proportion of pressures that result in turnovers."""
|
||||
denom = np.where(total_pressures == 0, 1, total_pressures)
|
||||
return pressures_leading_to_turnover / denom
|
||||
|
||||
@staticmethod
|
||||
def progressive_passes_received_index(
|
||||
progressive_passes_received: np.ndarray, minutes: np.ndarray
|
||||
) -> np.ndarray:
|
||||
"""Progressive passes received per 90 — attacking threat indicator."""
|
||||
minutes = np.where(minutes == 0, 90, minutes)
|
||||
return (progressive_passes_received / minutes) * 90
|
||||
|
||||
def compute_features(self, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Add pitch/field tilt features."""
|
||||
df = df.copy()
|
||||
|
||||
if "touches_att_3rd" in df.columns and "opp_touches_att_3rd" in df.columns:
|
||||
df["pitch_tilt"] = self.pitch_tilt(
|
||||
df["touches_att_3rd"].values, df["opp_touches_att_3rd"].values
|
||||
)
|
||||
|
||||
if "passes_into_final_third" in df.columns and "opp_passes_into_final_third" in df.columns:
|
||||
df["field_tilt"] = self.field_tilt(
|
||||
df["passes_into_final_third"].values,
|
||||
df["opp_passes_into_final_third"].values,
|
||||
)
|
||||
|
||||
if "pressure_regains" in df.columns and "pressures" in df.columns:
|
||||
df["pressure_regain_pct"] = self.pressure_regain_efficiency(
|
||||
df["pressure_regains"].values, df["pressures"].values
|
||||
)
|
||||
|
||||
if "progressive_passes_received" in df.columns and "minutes" in df.columns:
|
||||
df["progressive_passes_received_p90"] = self.progressive_passes_received_index(
|
||||
df["progressive_passes_received"].values, df["minutes"].values
|
||||
)
|
||||
|
||||
return df
|
||||
|
||||
|
||||
class WeatherContext:
|
||||
"""Historical weather context for matches (rain, temperature categories).
|
||||
|
||||
In production, this would integrate with a weather API.
|
||||
For the mockup, we use seasonal averages for Italian cities.
|
||||
"""
|
||||
|
||||
# Average November-January temperature (°C) and rain days/month for Serie A cities
|
||||
WINTER_WEATHER = {
|
||||
"torino": ("cold", 0.3), # Turin
|
||||
"milano": ("cold", 0.25), # Milan
|
||||
"bergamo": ("cold", 0.27), # Atalanta
|
||||
"udine": ("cold", 0.28), # Udinese
|
||||
"bologna": ("cold", 0.22), # Bologna
|
||||
"firenze": ("normal", 0.25), # Florence
|
||||
"roma": ("normal", 0.28), # Roma/Lazio
|
||||
"napoli": ("warm", 0.30), # Napoli
|
||||
"cagliari":("warm", 0.18), # Cagliari
|
||||
"genova": ("normal", 0.30), # Genoa/Sampdoria
|
||||
"lecce": ("warm", 0.25), # Lecce
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def get_context(
|
||||
cls, city: str, month: int
|
||||
) -> tuple:
|
||||
"""Get (temperature_category, rain_probability) for a match."""
|
||||
city_key = city.lower().split()[0]
|
||||
return cls.WINTER_WEATHER.get(city_key, ("normal", 0.2))
|
||||
|
||||
@classmethod
|
||||
def compute_features(cls, df: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Add weather context columns to match DataFrame."""
|
||||
df = df.copy()
|
||||
df["weather_temp_category"] = "normal"
|
||||
df["weather_rain_prob"] = 0.2
|
||||
|
||||
if "date" in df.columns:
|
||||
df["month"] = pd.to_datetime(df["date"]).dt.month
|
||||
|
||||
if "oppteam" in df.columns:
|
||||
for city, (temp, rain) in cls.WINTER_WEATHER.items():
|
||||
mask = df["oppteam"].str.lower().str.contains(city, na=False)
|
||||
df.loc[mask, "weather_temp_category"] = temp
|
||||
df.loc[mask, "weather_rain_prob"] = rain
|
||||
|
||||
# One-hot encode temperature
|
||||
for cat in ["cold", "normal", "warm"]:
|
||||
df[f"weather_is_{cat}"] = (df["weather_temp_category"] == cat).astype(int)
|
||||
|
||||
return df
|
||||
@@ -0,0 +1,220 @@
|
||||
"""Per-player-per-match feature engineering.
|
||||
|
||||
Refactored from notebooks 4/4b_player_match_dataset_creation.ipynb.
|
||||
Builds the supervised learning dataset: each row = one player in one match
|
||||
with features describing the player, their team, and the opponent.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class MatchFeatureBuilder:
|
||||
"""Builds the per-match feature matrix for model training and prediction."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
features_abs: Optional[list] = None,
|
||||
features_rel: Optional[list] = None,
|
||||
features_rel_gamecorr: Optional[list] = None,
|
||||
):
|
||||
self.features_abs = features_abs
|
||||
self.features_rel = features_rel
|
||||
self.features_rel_gamecorr = features_rel_gamecorr
|
||||
|
||||
def _default_features(self, player_stats: pd.DataFrame) -> tuple:
|
||||
"""Auto-select features from player stats columns."""
|
||||
cols = list(player_stats.columns)
|
||||
|
||||
# Absolute features: percentages, per-90 rates, id, role, vote stats
|
||||
abs_candidates = [
|
||||
"r", "home",
|
||||
"vote_avg", "vote_std",
|
||||
"gk_save_pct", "gk_clean_sheets_pct",
|
||||
]
|
||||
abs_features = [c for c in abs_candidates if c in cols]
|
||||
|
||||
# Per-90 derived team stats
|
||||
for c in cols:
|
||||
if c.endswith("_p90") or c.endswith("_pct"):
|
||||
if c not in abs_features:
|
||||
abs_features.append(c)
|
||||
|
||||
# Relative features: counting stats → per-minute
|
||||
rel_candidates = [
|
||||
"goals", "assists", "xg", "npxg", "xa",
|
||||
"shots_total", "shots_on_target",
|
||||
"passes_completed", "passes", "passes_total_distance",
|
||||
"passes_progressive_distance", "progressive_passes",
|
||||
"passes_into_final_third", "passes_into_penalty_area",
|
||||
"crosses", "crosses_into_penalty_area",
|
||||
"touches", "touches_att_pen_area",
|
||||
"dribbles_completed", "carries", "progressive_carries",
|
||||
"tackles", "tackles_won", "pressures", "pressure_regains",
|
||||
"blocks", "interceptions", "clearances",
|
||||
"fouls", "fouled", "aerials_won", "aerials_lost",
|
||||
"miscontrols", "dispossessed",
|
||||
"sca", "gca",
|
||||
]
|
||||
rel_features = [c for c in rel_candidates if c in cols]
|
||||
|
||||
# Game-corrected: goals, assists, xG, cards → per-90-per-game
|
||||
gamecorr_candidates = [
|
||||
"goals", "assists", "xg", "npxg", "cards_yellow", "cards_red",
|
||||
]
|
||||
gamecorr_features = [c for c in gamecorr_candidates if c in cols]
|
||||
|
||||
return abs_features, rel_features, gamecorr_features
|
||||
|
||||
def _player_features(
|
||||
self, player_name: str, player_team: str,
|
||||
player_stats: pd.DataFrame,
|
||||
team_data: pd.DataFrame,
|
||||
) -> dict:
|
||||
"""Build feature vector for a single player in a match context."""
|
||||
# Find player
|
||||
p_row = player_stats[player_stats["name"] == player_name]
|
||||
if p_row.empty:
|
||||
p_row = player_stats[player_stats["surname"].str.lower() == player_name.lower()]
|
||||
if p_row.empty:
|
||||
logger.warning(f"Player {player_name} not found in stats")
|
||||
return None
|
||||
p = p_row.iloc[0]
|
||||
|
||||
# Team stats
|
||||
t_row = team_data[team_data["team"].str.lower() == player_team.lower()]
|
||||
if t_row.empty:
|
||||
logger.warning(f"Team {player_team} not found in team data")
|
||||
return None
|
||||
t = t_row.iloc[0]
|
||||
|
||||
features = {}
|
||||
|
||||
# Player-level features
|
||||
for feat in self.features_abs:
|
||||
if feat in p.index:
|
||||
features[feat] = p[feat]
|
||||
|
||||
# Relative features: divide by minutes
|
||||
minutes = max(float(p.get("minutes", 90)), 1)
|
||||
games = max(float(p.get("games", 1)), 1)
|
||||
for feat in self.features_rel:
|
||||
if feat in p.index:
|
||||
features[feat] = float(p[feat]) / minutes
|
||||
|
||||
# Game-corrected features
|
||||
for feat in self.features_rel_gamecorr:
|
||||
if feat in p.index:
|
||||
features[feat] = float(p[feat]) * minutes / (games * 90)
|
||||
|
||||
# Team-level features
|
||||
for col in t.index:
|
||||
if col != "team":
|
||||
features[f"team_{col}"] = t[col]
|
||||
|
||||
return features
|
||||
|
||||
def build_match_dataset(
|
||||
self,
|
||||
vote_data: pd.DataFrame,
|
||||
player_stats: pd.DataFrame,
|
||||
team_data: pd.DataFrame,
|
||||
) -> pd.DataFrame:
|
||||
"""Build the full per-match feature dataset for model training.
|
||||
|
||||
Args:
|
||||
vote_data: From VoteProcessor, columns: [matchday, player, team, oppteam,
|
||||
home, vote, goals, assists, fantavote, ...].
|
||||
player_stats: From PlayerFeatureBuilder.build_player_dataset.
|
||||
team_data: FBref team stats (combined for/vs).
|
||||
|
||||
Returns:
|
||||
DataFrame: one row per player-match with features + targets.
|
||||
"""
|
||||
# Prepare team data (combine for + vs)
|
||||
if len(team_data.columns) > 50:
|
||||
# Assume concatenated for/vs - adjust as needed
|
||||
pass
|
||||
|
||||
# Ensure feature lists are set
|
||||
if self.features_abs is None or self.features_rel is None:
|
||||
self.features_abs, self.features_rel, self.features_rel_gamecorr = (
|
||||
self._default_features(player_stats)
|
||||
)
|
||||
logger.info(f"Auto-selected {len(self.features_abs)} abs, "
|
||||
f"{len(self.features_rel)} rel, "
|
||||
f"{len(self.features_rel_gamecorr)} gamecorr features")
|
||||
|
||||
rows = []
|
||||
for _, vrow in vote_data.iterrows():
|
||||
player = vrow["player"]
|
||||
pteam = vrow["team"]
|
||||
oppteam = vrow["oppteam"]
|
||||
home = vrow["home"]
|
||||
|
||||
feats = self._player_features(player, pteam, player_stats, team_data)
|
||||
if feats is None:
|
||||
continue
|
||||
|
||||
# Add opponent team features
|
||||
opp_row = team_data[team_data["team"].str.lower() == str(oppteam).lower()]
|
||||
if not opp_row.empty:
|
||||
opp = opp_row.iloc[0]
|
||||
for col in opp.index:
|
||||
if col != "team":
|
||||
feats[f"opp_{col}"] = opp[col]
|
||||
|
||||
# Add match context
|
||||
feats["home"] = home
|
||||
feats["matchday"] = vrow.get("matchday", 0)
|
||||
|
||||
# Add targets
|
||||
row = {**feats}
|
||||
for target_col in ["vote", "fantavote", "goals", "assists", "cards_malus"]:
|
||||
if target_col in vrow.index:
|
||||
row[target_col] = vrow[target_col]
|
||||
|
||||
rows.append(row)
|
||||
|
||||
df = pd.DataFrame(rows)
|
||||
|
||||
# Remove goalkeepers (r == 'P') if the role column exists
|
||||
if "r" in df.columns:
|
||||
df = df[df["r"] != "P"]
|
||||
|
||||
# Clean
|
||||
df = df.dropna(subset=[c for c in df.columns if c not in ("vote", "fantavote", "cards_malus")], how="all")
|
||||
df = df.fillna(0)
|
||||
|
||||
logger.info(f"Built match dataset: {df.shape}")
|
||||
return df
|
||||
|
||||
def build_prediction_features(
|
||||
self,
|
||||
player_stats: pd.DataFrame,
|
||||
team_data: pd.DataFrame,
|
||||
fixture_list: list,
|
||||
) -> pd.DataFrame:
|
||||
"""Build feature matrix for prediction (no targets).
|
||||
|
||||
Args:
|
||||
fixture_list: list of tuples (player_name, team, oppteam, home_flag).
|
||||
"""
|
||||
rows = []
|
||||
for player, team, oppteam, home in fixture_list:
|
||||
feats = self._player_features(player, team, player_stats, team_data)
|
||||
if feats is None:
|
||||
continue
|
||||
feats["home"] = home
|
||||
rows.append(feats)
|
||||
|
||||
df = pd.DataFrame(rows)
|
||||
if "r" in df.columns:
|
||||
df = df[df["r"] != "P"]
|
||||
return df.fillna(0)
|
||||
@@ -0,0 +1,211 @@
|
||||
"""RAG-based news ingestion pipeline for Italian sports news.
|
||||
|
||||
Uses LangChain (optional) to fetch and process Italian football news from
|
||||
Gazzetta, Sky Sport, Di Marzio, etc. Extracts entities: injuries, suspensions,
|
||||
tactical shifts, and training updates using an LLM.
|
||||
|
||||
When LangChain/LLM is unavailable, falls back to simple RSS parsing.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import feedparser
|
||||
import pandas as pd
|
||||
import requests
|
||||
import yaml
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class NewsRAGPipeline:
|
||||
"""Ingests Italian football news and extracts fantasy-relevant entities."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
sources_config: Optional[str] = None,
|
||||
cache_dir: str = "data/news_cache",
|
||||
llm_model: Optional[str] = None,
|
||||
):
|
||||
self.sources = self._load_sources(sources_config)
|
||||
self.cache_dir = Path(cache_dir)
|
||||
self.cache_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.llm_model = llm_model
|
||||
self.entity_cache = {}
|
||||
|
||||
def _load_sources(self, config_path: Optional[str]) -> dict:
|
||||
if not config_path:
|
||||
config_path = Path(__file__).parent.parent.parent / "config" / "news_sources.yaml"
|
||||
try:
|
||||
with open(config_path) as f:
|
||||
return yaml.safe_load(f).get("sources", {})
|
||||
except FileNotFoundError:
|
||||
logger.warning("News sources config not found; using defaults.")
|
||||
return {}
|
||||
|
||||
def fetch_rss(self, source_name: str) -> list:
|
||||
"""Fetch articles from an RSS feed."""
|
||||
source = self.sources.get(source_name, {})
|
||||
rss_url = source.get("rss", "")
|
||||
if not rss_url:
|
||||
logger.warning(f"No RSS URL for source: {source_name}")
|
||||
return []
|
||||
|
||||
logger.info(f"Fetching RSS from {rss_url}")
|
||||
try:
|
||||
feed = feedparser.parse(rss_url)
|
||||
except Exception as e:
|
||||
logger.error(f"RSS parse failed for {source_name}: {e}")
|
||||
return []
|
||||
|
||||
articles = []
|
||||
for entry in feed.entries[:20]: # Limit to 20 most recent
|
||||
articles.append({
|
||||
"source": source_name,
|
||||
"title": entry.get("title", ""),
|
||||
"summary": entry.get("summary", ""),
|
||||
"link": entry.get("link", ""),
|
||||
"published": entry.get("published", ""),
|
||||
})
|
||||
|
||||
logger.info(f"Fetched {len(articles)} articles from {source_name}")
|
||||
return articles
|
||||
|
||||
def fetch_all(self) -> list:
|
||||
"""Fetch all enabled sources."""
|
||||
all_articles = []
|
||||
for name, source in self.sources.items():
|
||||
if source.get("enabled", True):
|
||||
articles = self.fetch_rss(name)
|
||||
all_articles.extend(articles)
|
||||
return all_articles
|
||||
|
||||
def _simple_extract(self, text: str) -> list:
|
||||
"""Fallback extraction using keyword matching (no LLM).
|
||||
|
||||
Returns a list of entity dicts with type, content, and confidence.
|
||||
"""
|
||||
entities = []
|
||||
text_lower = text.lower()
|
||||
|
||||
# Injury keywords (Italian)
|
||||
injury_keywords = [
|
||||
"infortunio", "infortunato", "lesione", "stiramento",
|
||||
"contrattura", "distorsione", "frattura", "problema muscolare",
|
||||
"out", "non disponibile", "salta la partita", "out for",
|
||||
"ko", "stop", "recupero", "rientro",
|
||||
]
|
||||
if any(kw in text_lower for kw in injury_keywords):
|
||||
entities.append({
|
||||
"type": "INJURY",
|
||||
"source_text": text[:200],
|
||||
"confidence": 0.5,
|
||||
})
|
||||
|
||||
# Suspension keywords
|
||||
suspension_keywords = [
|
||||
"squalifica", "squalificato", "squalificati", "diffidato",
|
||||
"diffida", "cartellino", "ammonizione", "espulsione",
|
||||
"giornata di squalifica", "turno di stop", "suspended",
|
||||
]
|
||||
if any(kw in text_lower for kw in suspension_keywords):
|
||||
entities.append({
|
||||
"type": "SUSPENSION",
|
||||
"source_text": text[:200],
|
||||
"confidence": 0.5,
|
||||
})
|
||||
|
||||
# Tactical shift keywords
|
||||
tactical_keywords = [
|
||||
"modulo", "formazione", "cambio modulo", "nuovo modulo",
|
||||
"schieramento", "passa al", "difesa a", "centrocampo a",
|
||||
"3-5-2", "3-4-3", "4-3-3", "4-2-3-1", "4-4-2", "3-4-2-1",
|
||||
"cambio tattico", "rivoluzione tattica",
|
||||
]
|
||||
if any(kw in text_lower for kw in tactical_keywords):
|
||||
entities.append({
|
||||
"type": "TACTICAL_SHIFT",
|
||||
"source_text": text[:200],
|
||||
"confidence": 0.4,
|
||||
})
|
||||
|
||||
# Training update keywords
|
||||
training_keywords = [
|
||||
"allenamento", "in gruppo", "lavoro a parte", "palestra",
|
||||
"terapia", "personalizzato", "panchina", "titolar",
|
||||
"dubbio", "ballottaggio", "provato", "schierato",
|
||||
]
|
||||
if any(kw in text_lower for kw in training_keywords):
|
||||
entities.append({
|
||||
"type": "TRAINING_UPDATE",
|
||||
"source_text": text[:200],
|
||||
"confidence": 0.3,
|
||||
})
|
||||
|
||||
return entities
|
||||
|
||||
def extract_entities(self, articles: list) -> list:
|
||||
"""Extract fantasy-relevant entities from articles."""
|
||||
all_entities = []
|
||||
for article in articles:
|
||||
text = f"{article['title']} {article['summary']}"
|
||||
cache_key = hashlib.md5(text.encode()).hexdigest()
|
||||
|
||||
if cache_key in self.entity_cache:
|
||||
entities = self.entity_cache[cache_key]
|
||||
else:
|
||||
# Use keyword extraction as baseline (LLM option for production)
|
||||
entities = self._simple_extract(text)
|
||||
self.entity_cache[cache_key] = entities
|
||||
|
||||
for e in entities:
|
||||
e["article_link"] = article.get("link", "")
|
||||
e["source"] = article.get("source", "")
|
||||
all_entities.extend(entities)
|
||||
|
||||
logger.info(f"Extracted {len(all_entities)} entities from {len(articles)} articles")
|
||||
return all_entities
|
||||
|
||||
def to_features(self, entities: list, player_list: list) -> pd.DataFrame:
|
||||
"""Convert extracted entities into player-level feature flags.
|
||||
|
||||
Cross-references entity mentions with player names to create
|
||||
per-player injury/suspension/tactical flags.
|
||||
"""
|
||||
features = pd.DataFrame(index=range(len(player_list)))
|
||||
features["player"] = player_list
|
||||
features["news_injury_flag"] = 0
|
||||
features["news_suspension_flag"] = 0
|
||||
features["news_tactical_flag"] = 0
|
||||
features["news_training_flag"] = 0
|
||||
|
||||
for entity in entities:
|
||||
entity_text = entity.get("source_text", "").lower()
|
||||
entity_type = entity.get("type", "")
|
||||
|
||||
for i, player in enumerate(player_list):
|
||||
if not player:
|
||||
continue
|
||||
player_lower = player.lower()
|
||||
if player_lower in entity_text:
|
||||
if entity_type == "INJURY":
|
||||
features.at[i, "news_injury_flag"] = 1
|
||||
elif entity_type == "SUSPENSION":
|
||||
features.at[i, "news_suspension_flag"] = 1
|
||||
elif entity_type == "TACTICAL_SHIFT":
|
||||
features.at[i, "news_tactical_flag"] = 1
|
||||
elif entity_type == "TRAINING_UPDATE":
|
||||
features.at[i, "news_training_flag"] = 1
|
||||
|
||||
return features
|
||||
|
||||
def run_pipeline(self, player_list: list) -> pd.DataFrame:
|
||||
"""Run the full news-to-features pipeline."""
|
||||
articles = self.fetch_all()
|
||||
entities = self.extract_entities(articles)
|
||||
features = self.to_features(entities, player_list)
|
||||
return features
|
||||
@@ -0,0 +1,217 @@
|
||||
"""Player-level feature engineering.
|
||||
|
||||
Refactored from notebooks 3/3b_players_dataset_creation.ipynb.
|
||||
Cross-references FBref player stats with Fantacalcio roster data,
|
||||
applies name normalization and matching logic.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import yaml
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class PlayerFeatureBuilder:
|
||||
"""Builds per-player feature vectors from FBref + Fantacalcio data."""
|
||||
|
||||
def __init__(self, name_fix_path: Optional[str] = None):
|
||||
self.name_fixes = self._load_name_fixes(name_fix_path)
|
||||
self._keepers_id = None
|
||||
|
||||
def _load_name_fixes(self, path: Optional[str]) -> list:
|
||||
if not path:
|
||||
path = Path(__file__).parent.parent.parent / "config" / "name_fix.yaml"
|
||||
try:
|
||||
with open(path) as f:
|
||||
data = yaml.safe_load(f)
|
||||
return data.get("mappings", [])
|
||||
except FileNotFoundError:
|
||||
logger.warning(f"Name fix file not found: {path}")
|
||||
return []
|
||||
|
||||
@staticmethod
|
||||
def normalize_name(name: str) -> str:
|
||||
"""Normalize a player name: strip diacritics, lowercase, extract surname."""
|
||||
if not name:
|
||||
return ""
|
||||
name = unicodedata.normalize("NFKD", name).encode("ASCII", "ignore").decode()
|
||||
parts = name.split()
|
||||
if not parts:
|
||||
return ""
|
||||
return parts[-1].lower()
|
||||
|
||||
def build_player_dataset(
|
||||
self,
|
||||
fbref_outfield: pd.DataFrame,
|
||||
fbref_keepers: pd.DataFrame,
|
||||
fc_roster: pd.DataFrame,
|
||||
vote_averages: pd.DataFrame,
|
||||
min_gk_games: int = 3,
|
||||
) -> pd.DataFrame:
|
||||
"""Build the full per-player feature dataset.
|
||||
|
||||
Args:
|
||||
fbref_outfield: FBref outfield player stats.
|
||||
fbref_keepers: FBref goalkeeper stats.
|
||||
fc_roster: Fantacalcio player roster (Id, R, Nome, Squadra).
|
||||
vote_averages: DataFrame from VoteProcessor.compute_player_averages.
|
||||
|
||||
Returns:
|
||||
DataFrame with per-player features.
|
||||
"""
|
||||
# Combine FBref players
|
||||
outfield = fbref_outfield.copy()
|
||||
keepers = fbref_keepers.copy()
|
||||
self._keepers_id = len(outfield)
|
||||
|
||||
all_fbref = pd.concat([outfield, keepers], ignore_index=True)
|
||||
|
||||
# Normalize names
|
||||
all_fbref["surname"] = all_fbref["player"].apply(self.normalize_name)
|
||||
|
||||
# Process FC roster
|
||||
fc = fc_roster.copy()
|
||||
fc.columns = [c.lower() for c in fc.columns]
|
||||
# Map Italian columns
|
||||
col_map = {
|
||||
"id": "id", "r": "r", "ruolo": "r",
|
||||
"nome": "name", "giocatore": "name",
|
||||
"squadra": "team", "team": "team",
|
||||
}
|
||||
fc = fc.rename(columns={
|
||||
k: v for k, v in col_map.items()
|
||||
if k in fc.columns or v in fc.columns
|
||||
})
|
||||
|
||||
# Ensure required columns exist
|
||||
for col in ["id", "r", "name", "team"]:
|
||||
if col not in fc.columns:
|
||||
# Try to find by position
|
||||
for orig in fc.columns:
|
||||
if col in orig.lower() or orig.lower() in col:
|
||||
fc = fc.rename(columns={orig: col})
|
||||
break
|
||||
if col not in fc.columns:
|
||||
raise ValueError(f"Required column '{col}' not found in roster")
|
||||
# Skip the first row if it's a subheader
|
||||
try:
|
||||
if pd.to_numeric(fc[col].iloc[0]) != pd.to_numeric(fc[col].iloc[0]):
|
||||
pass
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
fc["surname"] = fc["name"].astype(str).apply(self.normalize_name)
|
||||
|
||||
# Match FC players to FBref indices
|
||||
fc["fb_ID"] = -1
|
||||
for i, row in fc.iterrows():
|
||||
fc_surname = row["surname"]
|
||||
fc_team = str(row["team"]).lower()
|
||||
fc_role = row["r"]
|
||||
|
||||
for j, fbr in all_fbref.iterrows():
|
||||
if fbr["surname"] != fc_surname:
|
||||
continue
|
||||
if fc_team not in str(fbr["team"]).lower():
|
||||
continue
|
||||
# Role check
|
||||
fbr_pos = str(fbr.get("position", "")).upper()
|
||||
if fc_role == "P" and "GK" in fbr_pos:
|
||||
fc.at[i, "fb_ID"] = j
|
||||
break
|
||||
elif fc_role != "P" and "GK" not in fbr_pos:
|
||||
fc.at[i, "fb_ID"] = j
|
||||
break
|
||||
|
||||
# Fallback: match by surname alone (excl ambiguous names)
|
||||
if fc.at[i, "fb_ID"] == -1:
|
||||
ambiguous = {"pellegrini", "bastoni", "kristensen", "rossi", "bianchi"}
|
||||
if fc_surname not in ambiguous:
|
||||
for j, fbr in all_fbref.iterrows():
|
||||
if fbr["surname"] == fc_surname:
|
||||
fbr_pos = str(fbr.get("position", "")).upper()
|
||||
if (fc_role == "P") == ("GK" in fbr_pos):
|
||||
fc.at[i, "fb_ID"] = j
|
||||
break
|
||||
|
||||
logger.info(
|
||||
f"Matched {(fc['fb_ID'] >= 0).sum()}/{len(fc)} FC players to FBref"
|
||||
)
|
||||
|
||||
# Copy FBref stats
|
||||
outfield_cols = [c for c in outfield.columns if c not in (
|
||||
"player", "nationality", "position", "team", "age", "birth_year"
|
||||
)]
|
||||
keeper_cols = [c for c in keepers.columns if c not in (
|
||||
"player", "nationality", "position", "team", "age", "birth_year"
|
||||
)]
|
||||
|
||||
for col in outfield_cols:
|
||||
fc[col] = 0.0
|
||||
for col in keeper_cols:
|
||||
fc[col] = 0.0
|
||||
|
||||
for i, row in fc.iterrows():
|
||||
fb_id = int(row["fb_ID"])
|
||||
if fb_id < 0:
|
||||
continue
|
||||
if row["r"] == "P":
|
||||
k_id = fb_id - self._keepers_id
|
||||
if 0 <= k_id < len(keepers):
|
||||
for col in keeper_cols:
|
||||
fc.at[i, col] = keepers.iloc[k_id][col] if col in keepers.columns else 0.0
|
||||
else:
|
||||
if 0 <= fb_id < len(outfield):
|
||||
for col in outfield_cols:
|
||||
fc.at[i, col] = outfield.iloc[fb_id][col] if col in outfield.columns else 0.0
|
||||
|
||||
# Add vote averages
|
||||
if vote_averages is not None and not vote_averages.empty:
|
||||
fc["vote_avg"] = 6.0
|
||||
fc["vote_std"] = 0.5
|
||||
for i, row in fc.iterrows():
|
||||
player_name = str(row["name"]).lower()
|
||||
matches = vote_averages[vote_averages["player"].str.lower() == player_name]
|
||||
if not matches.empty:
|
||||
fc.at[i, "vote_avg"] = matches["vote_avg"].values[0]
|
||||
fc.at[i, "vote_std"] = matches["vote_std"].values[0]
|
||||
else:
|
||||
fc["vote_avg"] = 6.0
|
||||
fc["vote_std"] = 0.5
|
||||
|
||||
# GK backup weighted averaging
|
||||
if min_gk_games > 0:
|
||||
self._weight_avg_backup_gks(fc, keeper_cols, min_gk_games)
|
||||
|
||||
return fc
|
||||
|
||||
def _weight_avg_backup_gks(self, fc: pd.DataFrame, keeper_cols: list, min_gk_games: int):
|
||||
"""Weighted-average backup GK stats with primary GK stats."""
|
||||
gk_mask = fc["r"] == "P"
|
||||
for team in fc.loc[gk_mask, "team"].unique():
|
||||
team_gks = fc[fc["team"] == team]
|
||||
gk_games_col = "gk_games" if "gk_games" in fc.columns else None
|
||||
if gk_games_col is None:
|
||||
continue
|
||||
|
||||
primary = team_gks[team_gks[gk_games_col] >= min_gk_games]
|
||||
if primary.empty:
|
||||
continue
|
||||
primary = primary.iloc[0]
|
||||
|
||||
for i, row in team_gks.iterrows():
|
||||
own_games = row.get(gk_games_col, 0)
|
||||
if own_games >= min_gk_games:
|
||||
continue
|
||||
weight = max(0.0, own_games / min_gk_games)
|
||||
for col in keeper_cols:
|
||||
if col in fc.columns:
|
||||
fc.at[i, col] = (
|
||||
weight * row[col] + (1 - weight) * primary[col]
|
||||
)
|
||||
@@ -0,0 +1,203 @@
|
||||
"""Vote data processor.
|
||||
|
||||
Refactored from notebook 2_votes_dataset_creation.ipynb.
|
||||
Processes Fantacalcio matchday vote files into a unified player-match
|
||||
dataset with derived features (goals, cards, fantavote).
|
||||
"""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_SCORING = {
|
||||
"goal": 3.0,
|
||||
"assist": 1.0,
|
||||
"yellow_card": -0.5,
|
||||
"red_card": -1.0,
|
||||
"own_goal": -2.0,
|
||||
}
|
||||
|
||||
|
||||
class VoteProcessor:
|
||||
"""Processes raw Fantacalcio vote Excel files into a unified database."""
|
||||
|
||||
def __init__(self, scoring: Optional[dict] = None):
|
||||
self.scoring = scoring or DEFAULT_SCORING
|
||||
|
||||
def process_vote_file(
|
||||
self, filepath: str, matchday: int, home_team: str, away_team: str
|
||||
) -> pd.DataFrame:
|
||||
"""Process a single matchday vote Excel file.
|
||||
|
||||
Extracts: player, team, role, vote, goals, assists, cards (yellow/red),
|
||||
and computes fantavote (vote + bonus - malus).
|
||||
|
||||
Args:
|
||||
filepath: Path to the Excel vote file.
|
||||
matchday: Matchday number (1-38).
|
||||
home_team: Name of the home team.
|
||||
away_team: Name of the away team.
|
||||
|
||||
Returns:
|
||||
DataFrame with processed vote data.
|
||||
"""
|
||||
try:
|
||||
df_raw = pd.read_excel(filepath, header=None)
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not read {filepath}: {e}")
|
||||
return pd.DataFrame()
|
||||
|
||||
rows = []
|
||||
current_team = None
|
||||
|
||||
for i in range(len(df_raw)):
|
||||
row = df_raw.iloc[i]
|
||||
# Check if this is a team header row
|
||||
first_cell = str(row[0]) if pd.notna(row[0]) else ""
|
||||
second_cell = str(row[1]) if len(row) > 1 and pd.notna(row[1]) else ""
|
||||
|
||||
# Team headers appear before player rows
|
||||
if first_cell.upper().isupper() and second_cell == "" and first_cell != "Cod.":
|
||||
current_team = first_cell.strip()
|
||||
continue
|
||||
|
||||
# Skip header rows
|
||||
if second_cell == "ALL" or first_cell == "Cod.":
|
||||
continue
|
||||
|
||||
# Try to parse a player row
|
||||
try:
|
||||
player = str(row[2]) if len(row) > 2 and pd.notna(row[2]) else ""
|
||||
if not player:
|
||||
continue
|
||||
|
||||
vote_str = str(row[3]) if len(row) > 3 and pd.notna(row[3]) else ""
|
||||
vote = float(vote_str.replace(",", ".")) if vote_str and vote_str not in ("", "S.V.", "nan") else np.nan
|
||||
|
||||
goals_field = int(row[4]) if len(row) > 4 and pd.notna(row[4]) else 0
|
||||
own_goals = int(row[5]) if len(row) > 5 and pd.notna(row[5]) else 0
|
||||
pen_goals = int(row[8]) if len(row) > 8 and pd.notna(row[8]) else 0
|
||||
goals = goals_field + pen_goals - own_goals
|
||||
|
||||
yellow = int(row[10]) if len(row) > 10 and pd.notna(row[10]) else 0
|
||||
red = int(row[11]) if len(row) > 11 and pd.notna(row[11]) else 0
|
||||
|
||||
assists = int(row[12]) if len(row) > 12 and pd.notna(row[12]) else 0
|
||||
|
||||
cards_malus = yellow * abs(self.scoring["yellow_card"]) + red * abs(self.scoring["red_card"])
|
||||
goals_bonus = max(0, goals) * self.scoring["goal"]
|
||||
own_goal_malus = max(0, own_goals) * abs(self.scoring["own_goal"])
|
||||
assist_bonus = assists * self.scoring["assist"]
|
||||
|
||||
fantavote = (vote or 0) + goals_bonus + assist_bonus - cards_malus - own_goal_malus
|
||||
|
||||
home = 1 if current_team == home_team else 0
|
||||
oppteam = away_team if home else home_team
|
||||
|
||||
rows.append({
|
||||
"matchday": matchday,
|
||||
"player": player,
|
||||
"team": current_team,
|
||||
"oppteam": oppteam,
|
||||
"home": home,
|
||||
"vote": vote,
|
||||
"goals": goals,
|
||||
"assists": assists,
|
||||
"cards_malus": cards_malus,
|
||||
"fantavote": fantavote,
|
||||
"yellow_cards": yellow,
|
||||
"red_cards": red,
|
||||
"own_goals": own_goals,
|
||||
})
|
||||
except (ValueError, IndexError) as e:
|
||||
logger.debug(f"Skipping row {i} in {filepath}: {e}")
|
||||
continue
|
||||
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
def process_matchday(
|
||||
self, vote_dir: str, matchday: int, calendar: pd.DataFrame
|
||||
) -> pd.DataFrame:
|
||||
"""Process all vote files for a matchday using the calendar.
|
||||
|
||||
Args:
|
||||
vote_dir: Directory containing vote Excel files.
|
||||
matchday: Matchday number.
|
||||
calendar: DataFrame with columns [matchday, home, away].
|
||||
|
||||
Returns:
|
||||
Combined DataFrame of all matches for the matchday.
|
||||
"""
|
||||
md_calendar = calendar[calendar["matchday"] == matchday]
|
||||
if md_calendar.empty:
|
||||
logger.warning(f"No calendar entries for matchday {matchday}")
|
||||
return pd.DataFrame()
|
||||
|
||||
all_matches = []
|
||||
vote_path = Path(vote_dir)
|
||||
|
||||
for _, fixture in md_calendar.iterrows():
|
||||
home = fixture["home"]
|
||||
away = fixture["away"]
|
||||
|
||||
# Find the vote file for this matchday
|
||||
candidates = list(vote_path.glob(f"*Giornata_{matchday}*")) + list(
|
||||
vote_path.glob(f"*giornata_{matchday}*")
|
||||
)
|
||||
if not candidates:
|
||||
logger.warning(f"No vote file found for matchday {matchday}")
|
||||
continue
|
||||
|
||||
df = self.process_vote_file(str(candidates[0]), matchday, home, away)
|
||||
if not df.empty:
|
||||
# Mark which players belong to this fixture
|
||||
df["oppteam"] = df.apply(
|
||||
lambda r: away if r["team"] == home else home, axis=1
|
||||
)
|
||||
all_matches.append(df)
|
||||
|
||||
if not all_matches:
|
||||
return pd.DataFrame()
|
||||
return pd.concat(all_matches, ignore_index=True)
|
||||
|
||||
def compute_player_averages(
|
||||
self, votes_df: pd.DataFrame, min_votes: int = 3,
|
||||
default_outfield_mean: float = 6.0, default_outfield_std: float = 0.58,
|
||||
default_gk_mean: float = 6.22, default_gk_std: float = 0.43,
|
||||
) -> pd.DataFrame:
|
||||
"""Compute rolling player averages (vote_avg, vote_std) from votes data.
|
||||
|
||||
Uses Bayesian shrinkage: if a player has fewer than min_votes,
|
||||
synthetic votes from a default distribution are added.
|
||||
"""
|
||||
results = []
|
||||
for player, group in votes_df.groupby("player"):
|
||||
votes = group["vote"].dropna().values
|
||||
n = len(votes)
|
||||
|
||||
if n >= min_votes:
|
||||
avg = np.mean(votes)
|
||||
std = np.std(votes, ddof=1) if n > 1 else 0.5
|
||||
else:
|
||||
deficit = min_votes - n
|
||||
default_mean = default_outfield_mean
|
||||
default_std = default_outfield_std
|
||||
synthetic = np.random.RandomState(42).normal(default_mean, default_std, deficit)
|
||||
all_votes = np.concatenate([votes, synthetic])
|
||||
avg = np.mean(all_votes)
|
||||
std = np.std(all_votes, ddof=1) if len(all_votes) > 1 else 0.5
|
||||
|
||||
results.append({
|
||||
"player": player,
|
||||
"team": group["team"].iloc[0],
|
||||
"vote_avg": avg,
|
||||
"vote_std": std,
|
||||
"n_matches": n,
|
||||
})
|
||||
|
||||
return pd.DataFrame(results)
|
||||
@@ -0,0 +1 @@
|
||||
"""Machine learning models."""
|
||||
@@ -0,0 +1,41 @@
|
||||
"""Base model interface and utilities for Fantabeto ML models."""
|
||||
|
||||
import logging
|
||||
import pickle
|
||||
from abc import ABC, abstractmethod
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BaseModel(ABC):
|
||||
"""Abstract base class for Fantabeto ML models."""
|
||||
|
||||
def __init__(self, model_dir: str = "models_trained"):
|
||||
self.model_dir = Path(model_dir)
|
||||
self.model_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.scaler = None
|
||||
|
||||
@abstractmethod
|
||||
def fit(self, X: pd.DataFrame, y: pd.Series, **kwargs):
|
||||
...
|
||||
|
||||
@abstractmethod
|
||||
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
||||
...
|
||||
|
||||
def save(self, name: str):
|
||||
path = self.model_dir / name
|
||||
with open(path, "wb") as f:
|
||||
pickle.dump(self, f)
|
||||
logger.info(f"Model saved to {path}")
|
||||
|
||||
@classmethod
|
||||
def load(cls, name: str, model_dir: str = "models_trained") -> "BaseModel":
|
||||
path = Path(model_dir) / name
|
||||
with open(path, "rb") as f:
|
||||
return pickle.load(f)
|
||||
@@ -0,0 +1,171 @@
|
||||
"""Card and penalty classifiers.
|
||||
|
||||
Binary classification models for:
|
||||
- Yellow card probability (per match)
|
||||
- Red card probability (per match, with extreme class imbalance handling)
|
||||
- Penalty kick probability (player takes penalty this match)
|
||||
- Goal probability (Poisson regression)
|
||||
"""
|
||||
|
||||
import logging
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
from .base_model import BaseModel
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CardClassifier(BaseModel):
|
||||
"""Classifies probability of yellow/red cards for a given player-match."""
|
||||
|
||||
def __init__(self, card_type: str = "yellow", model_dir: str = "models_trained"):
|
||||
super().__init__(model_dir)
|
||||
self.card_type = card_type
|
||||
self.model = None
|
||||
self.feature_names = None
|
||||
self.scaler = StandardScaler()
|
||||
self.class_weights = None
|
||||
|
||||
def _card_features(self, X: pd.DataFrame) -> np.ndarray:
|
||||
"""Select card-relevant features."""
|
||||
card_features = [
|
||||
"fouls_per_min", "fouled_per_min",
|
||||
"cards_yellow_p90", "cards_red_p90",
|
||||
"tackles_per_min", "interceptions_per_min",
|
||||
"pressures_per_min",
|
||||
"home", "vote_avg", "vote_std",
|
||||
]
|
||||
available = [c for c in card_features if c in X.columns]
|
||||
if not available:
|
||||
available = [c for c in X.columns if any(
|
||||
kw in str(c).lower() for kw in
|
||||
["foul", "card", "tackle", "intercept", "pressure", "home", "vote", "yellow", "red"]
|
||||
)]
|
||||
if not available:
|
||||
available = list(X.select_dtypes(include=[np.number]).columns[:20])
|
||||
return available
|
||||
|
||||
def fit(self, X: pd.DataFrame, y: pd.Series, **kwargs):
|
||||
try:
|
||||
import lightgbm as lgb
|
||||
except ImportError:
|
||||
raise ImportError("LightGBM required. pip install lightgbm")
|
||||
|
||||
self.feature_names = self._card_features(X)
|
||||
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.fit_transform(X_c)
|
||||
|
||||
# Handle class imbalance
|
||||
pos_ratio = y.mean()
|
||||
neg_ratio = 1 - pos_ratio
|
||||
self.class_weights = {0: 1.0, 1: max(1.0, min(10.0, neg_ratio / max(pos_ratio, 1e-6)))}
|
||||
|
||||
self.model = lgb.LGBMClassifier(
|
||||
n_estimators=300,
|
||||
learning_rate=0.05,
|
||||
max_depth=5,
|
||||
class_weight="balanced",
|
||||
random_state=42,
|
||||
verbose=-1,
|
||||
)
|
||||
self.model.fit(X_scaled, y)
|
||||
logger.info(
|
||||
f"Card classifier ({self.card_type}) trained. "
|
||||
f"Pos ratio: {pos_ratio:.4f}, features: {len(self.feature_names)}"
|
||||
)
|
||||
return self
|
||||
|
||||
def predict_proba(self, X: pd.DataFrame) -> np.ndarray:
|
||||
if self.model is None:
|
||||
raise RuntimeError("Model not trained")
|
||||
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.transform(X_c)
|
||||
return self.model.predict_proba(X_scaled)[:, 1]
|
||||
|
||||
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
||||
return (self.predict_proba(X) >= 0.5).astype(int)
|
||||
|
||||
|
||||
class PenaltyModel:
|
||||
"""Predicts probability a player takes a penalty this match."""
|
||||
|
||||
def __init__(self):
|
||||
self.team_penalty_takers = {}
|
||||
|
||||
def fit(self, penalty_data: pd.DataFrame):
|
||||
"""Learn penalty taker patterns from historical data.
|
||||
|
||||
Args:
|
||||
penalty_data: DataFrame with columns [player, team, penalties_taken, matchday].
|
||||
"""
|
||||
for team, group in penalty_data.groupby("team"):
|
||||
takers = group.groupby("player")["penalties_taken"].sum()
|
||||
total = takers.sum()
|
||||
if total > 0:
|
||||
self.team_penalty_takers[team] = (takers / total).to_dict()
|
||||
logger.info(f"Fitted penalty model for {len(self.team_penalty_takers)} teams")
|
||||
|
||||
def predict_proba(self, players: pd.DataFrame) -> np.ndarray:
|
||||
"""Predict penalty probability for each player."""
|
||||
probs = np.zeros(len(players))
|
||||
for i, (_, row) in enumerate(players.iterrows()):
|
||||
team = row.get("team", "")
|
||||
player = row.get("player", "") or row.get("name", "")
|
||||
if team in self.team_penalty_takers:
|
||||
probs[i] = self.team_penalty_takers[team].get(player, 0.0)
|
||||
return probs
|
||||
|
||||
|
||||
class GoalProbabilityModel:
|
||||
"""Poisson regression for goal count prediction (per match)."""
|
||||
|
||||
def __init__(self, model_dir: str = "models_trained"):
|
||||
self.model_dir = model_dir
|
||||
self.model = None
|
||||
self.feature_names = None
|
||||
self.scaler = StandardScaler()
|
||||
|
||||
def fit(self, X: pd.DataFrame, y: pd.Series):
|
||||
"""Fit a Poisson or light GBM on goal count.
|
||||
|
||||
Since goals are rare events, uses LightGBM with Poisson objective
|
||||
or Tweedie regression.
|
||||
"""
|
||||
try:
|
||||
import lightgbm as lgb
|
||||
except ImportError:
|
||||
raise ImportError("LightGBM required")
|
||||
|
||||
self.feature_names = [
|
||||
c for c in X.columns if any(kw in c.lower() for kw in [
|
||||
"xg", "shot", "goal", "minute", "game", "home", "pass", "carry",
|
||||
"touches_att", "progressive", "dribble", "vote", "fantavote"
|
||||
])
|
||||
]
|
||||
if not self.feature_names:
|
||||
self.feature_names = list(X.select_dtypes(include=[np.number]).columns[:20])
|
||||
|
||||
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.fit_transform(X_c)
|
||||
|
||||
self.model = lgb.LGBMRegressor(
|
||||
n_estimators=300,
|
||||
learning_rate=0.03,
|
||||
max_depth=5,
|
||||
objective="poisson",
|
||||
random_state=42,
|
||||
verbose=-1,
|
||||
)
|
||||
self.model.fit(X_scaled, y)
|
||||
logger.info(f"Goal model trained. Features: {len(self.feature_names)}")
|
||||
return self
|
||||
|
||||
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
||||
if self.model is None:
|
||||
raise RuntimeError("Model not trained")
|
||||
X_c = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.transform(X_c)
|
||||
return self.model.predict(X_scaled)
|
||||
@@ -0,0 +1,101 @@
|
||||
"""Distribution head for probabilistic score prediction.
|
||||
|
||||
Replicates the SinhArcsinh distribution output from the original TF Probability
|
||||
model, but using scipy for lightweight inference. Also provides a Bernoulli
|
||||
head for clean sheet probability (goalkeepers).
|
||||
"""
|
||||
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
from scipy.stats import norm
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
try:
|
||||
from scipy.special import softplus
|
||||
except ImportError:
|
||||
def softplus(x):
|
||||
return np.log1p(np.exp(-np.abs(x))) + np.maximum(x, 0)
|
||||
|
||||
|
||||
class SinhArcsinhDistribution:
|
||||
"""SinhArcsinh distribution for asymmetric score predictions.
|
||||
|
||||
Parameters: loc (position), scale (spread), skewness, tailweight.
|
||||
"""
|
||||
|
||||
def __init__(self, loc: np.ndarray, scale: np.ndarray,
|
||||
skewness: np.ndarray, tailweight: np.ndarray):
|
||||
self.loc = np.asarray(loc)
|
||||
self.scale = np.clip(np.asarray(scale), 1e-3, None)
|
||||
self.skewness = np.asarray(skewness)
|
||||
self.tailweight = np.asarray(tailweight)
|
||||
|
||||
@staticmethod
|
||||
def sinh_arcsinh_pdf(x, mu, sigma, eps, delta):
|
||||
"""SinhArcsinh PDF."""
|
||||
mul = 2.0 / np.sinh(np.arcsinh(2.0) * delta)
|
||||
z = (x - mu) / (sigma * mul)
|
||||
S = np.sinh(-eps + (1.0 / delta) * np.arcsinh(z))
|
||||
norm_const = 1.0 / (sigma * mul * delta) / np.sqrt(2.0 * np.pi)
|
||||
return np.exp(-0.5 * S * S) * np.sqrt(1.0 + S * S) * norm_const / np.sqrt(1.0 + z * z)
|
||||
|
||||
def prob(self, x_range: np.ndarray) -> np.ndarray:
|
||||
"""Compute PDF over a range of x values. Vectorized."""
|
||||
return self.sinh_arcsinh_pdf(
|
||||
x_range[:, None], self.loc, self.scale, self.skewness, self.tailweight
|
||||
)
|
||||
|
||||
def mean(self) -> np.ndarray:
|
||||
"""Approximate mean via numerical integration."""
|
||||
x = np.linspace(0, 20, 2000)
|
||||
pdf = self.prob(x)
|
||||
return np.average(x[:, None], weights=pdf, axis=0)
|
||||
|
||||
def std(self, quantile: float = 0.9545) -> np.ndarray:
|
||||
"""Approximate std using quantile-based spread (matching original behavior)."""
|
||||
return (self.quantile(quantile) - self.mean()) / 2.0
|
||||
|
||||
def quantile(self, q: float) -> np.ndarray:
|
||||
x = np.linspace(0, 30, 3000)
|
||||
pdf = self.prob(x)
|
||||
cdf = np.cumsum(pdf, axis=0)
|
||||
cdf = cdf / cdf[-1, :]
|
||||
results = np.zeros(len(self.loc))
|
||||
for i in range(len(self.loc)):
|
||||
idx = np.searchsorted(cdf[:, i], q)
|
||||
idx = min(idx, len(x) - 1)
|
||||
results[i] = x[idx]
|
||||
return results
|
||||
|
||||
|
||||
class BernoulliCleanSheet:
|
||||
"""Bernoulli distribution for clean sheet probability (goalkeepers)."""
|
||||
|
||||
def __init__(self, prob: np.ndarray):
|
||||
self.prob = np.clip(np.asarray(prob), 0.0, 1.0)
|
||||
|
||||
def sample(self, n: int = 1000) -> np.ndarray:
|
||||
return np.random.binomial(1, self.prob[:, None], (len(self.prob), n)).T
|
||||
|
||||
def mean(self) -> np.ndarray:
|
||||
return self.prob
|
||||
|
||||
def std(self) -> np.ndarray:
|
||||
return np.sqrt(self.prob * (1 - self.prob))
|
||||
|
||||
|
||||
def sinh_arcsinh_params(raw_params: np.ndarray) -> tuple:
|
||||
"""Convert raw network output to SinhArcsinh parameters.
|
||||
|
||||
loc = raw[..., 0]
|
||||
scale = 1e-3 + softplus(raw[..., 1])
|
||||
skewness = raw[..., 2]
|
||||
tailweight = 0.5 + 1.2 * sigmoid(raw[..., 3])
|
||||
"""
|
||||
from scipy.special import expit as sigmoid
|
||||
loc = raw_params[..., 0]
|
||||
scale = 1e-3 + softplus(raw_params[..., 1])
|
||||
skewness = raw_params[..., 2]
|
||||
tailweight = 0.5 + 1.2 * sigmoid(raw_params[..., 3])
|
||||
return loc, scale, skewness, tailweight
|
||||
@@ -0,0 +1,205 @@
|
||||
"""GBM ensemble model for Fantacalcio score prediction.
|
||||
|
||||
Uses LightGBM, CatBoost, and XGBoost with stacked blending (Ridge meta-learner)
|
||||
to predict modified Fantavoto scores. Includes probabilistic output via
|
||||
bootstrap ensembles for uncertainty quantification.
|
||||
|
||||
Replaces the original TensorFlow Probability SinhArcsinh model with a
|
||||
more robust gradient-boosting ensemble.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import TimeSeriesSplit
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
from .base_model import BaseModel
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class GBMEnsemble(BaseModel):
|
||||
"""Gradient-boosted ensemble for Fantavoto prediction."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model_dir: str = "models_trained",
|
||||
n_estimators: int = 500,
|
||||
learning_rate: float = 0.05,
|
||||
max_depth: int = 6,
|
||||
n_bootstrap: int = 100,
|
||||
use_catboost: bool = True,
|
||||
use_xgboost: bool = True,
|
||||
):
|
||||
super().__init__(model_dir)
|
||||
self.n_estimators = n_estimators
|
||||
self.learning_rate = learning_rate
|
||||
self.max_depth = max_depth
|
||||
self.n_bootstrap = n_bootstrap
|
||||
self.use_catboost = use_catboost
|
||||
self.use_xgboost = use_xgboost
|
||||
|
||||
self.lgb_model = None
|
||||
self.cb_model = None
|
||||
self.xgb_model = None
|
||||
self.meta_learner = None
|
||||
self.feature_names = None
|
||||
self.scaler = StandardScaler()
|
||||
|
||||
def _init_lgb(self):
|
||||
try:
|
||||
import lightgbm as lgb
|
||||
return lgb.LGBMRegressor(
|
||||
n_estimators=self.n_estimators,
|
||||
learning_rate=self.learning_rate,
|
||||
max_depth=self.max_depth,
|
||||
num_leaves=31,
|
||||
subsample=0.8,
|
||||
colsample_bytree=0.8,
|
||||
random_state=42,
|
||||
verbose=-1,
|
||||
)
|
||||
except ImportError:
|
||||
logger.warning("LightGBM not installed")
|
||||
return None
|
||||
|
||||
def _init_cb(self):
|
||||
if not self.use_catboost:
|
||||
return None
|
||||
try:
|
||||
from catboost import CatBoostRegressor
|
||||
return CatBoostRegressor(
|
||||
iterations=self.n_estimators,
|
||||
learning_rate=self.learning_rate,
|
||||
depth=self.max_depth,
|
||||
random_seed=42,
|
||||
verbose=0,
|
||||
)
|
||||
except ImportError:
|
||||
logger.warning("CatBoost not installed")
|
||||
return None
|
||||
|
||||
def _init_xgb(self):
|
||||
if not self.use_xgboost:
|
||||
return None
|
||||
try:
|
||||
import xgboost as xgb
|
||||
return xgb.XGBRegressor(
|
||||
n_estimators=self.n_estimators,
|
||||
learning_rate=self.learning_rate,
|
||||
max_depth=self.max_depth,
|
||||
subsample=0.8,
|
||||
colsample_bytree=0.8,
|
||||
random_state=42,
|
||||
verbosity=0,
|
||||
)
|
||||
except ImportError:
|
||||
logger.warning("XGBoost not installed")
|
||||
return None
|
||||
|
||||
def fit(self, X: pd.DataFrame, y: pd.Series, **kwargs):
|
||||
"""Train the GBM ensemble with stacked blending.
|
||||
|
||||
Args:
|
||||
X: Feature matrix.
|
||||
y: Target values (fantavoto).
|
||||
"""
|
||||
self.feature_names = list(X.columns)
|
||||
|
||||
# Preprocess
|
||||
X_clean = X.select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.fit_transform(X_clean)
|
||||
|
||||
# Initialize base models
|
||||
self.lgb_model = self._init_lgb()
|
||||
self.cb_model = self._init_cb()
|
||||
self.xgb_model = self._init_xgb()
|
||||
|
||||
# Time-series cross-validation for meta-learner
|
||||
tscv = TimeSeriesSplit(n_splits=3)
|
||||
meta_features = np.zeros((X_scaled.shape[0], 0))
|
||||
|
||||
models = [m for m in [self.lgb_model, self.cb_model, self.xgb_model] if m is not None]
|
||||
for model in models:
|
||||
oof_preds = np.zeros(X_scaled.shape[0])
|
||||
for train_idx, val_idx in tscv.split(X_scaled):
|
||||
model.fit(X_scaled[train_idx], y.iloc[train_idx])
|
||||
oof_preds[val_idx] = model.predict(X_scaled[val_idx])
|
||||
meta_features = np.column_stack([meta_features, oof_preds])
|
||||
|
||||
# Refit all models on full data
|
||||
for model in models:
|
||||
model.fit(X_scaled, y)
|
||||
|
||||
# Meta-learner: Ridge regression
|
||||
self.meta_learner = Ridge(alpha=1.0)
|
||||
self.meta_learner.fit(meta_features, y)
|
||||
|
||||
# Bootstrap ensembles for uncertainty
|
||||
self.bootstrap_models = []
|
||||
n = X_scaled.shape[0]
|
||||
rng = np.random.RandomState(42)
|
||||
for _ in range(self.n_bootstrap):
|
||||
idx = rng.choice(n, n, replace=True)
|
||||
boot_models = []
|
||||
for model in models:
|
||||
m = model.__class__(**model.get_params())
|
||||
m.fit(X_scaled[idx], y.iloc[idx])
|
||||
boot_models.append(m)
|
||||
self.bootstrap_models.append(boot_models)
|
||||
|
||||
logger.info(f"Trained ensemble with {len(models)} base models")
|
||||
return self
|
||||
|
||||
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
||||
"""Predict Fantavoto point estimates."""
|
||||
if self.lgb_model is None:
|
||||
raise RuntimeError("Model not trained. Call fit() first.")
|
||||
|
||||
X_clean = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.transform(X_clean)
|
||||
|
||||
preds = []
|
||||
models = [m for m in [self.lgb_model, self.cb_model, self.xgb_model] if m is not None]
|
||||
for model in models:
|
||||
preds.append(model.predict(X_scaled))
|
||||
|
||||
meta_features = np.column_stack(preds)
|
||||
return self.meta_learner.predict(meta_features)
|
||||
|
||||
def predict_distribution(self, X: pd.DataFrame) -> tuple:
|
||||
"""Predict mean and std via bootstrap ensemble.
|
||||
|
||||
Returns:
|
||||
(mean_predictions, std_predictions) arrays.
|
||||
"""
|
||||
X_clean = X[self.feature_names].select_dtypes(include=[np.number]).fillna(0)
|
||||
X_scaled = self.scaler.transform(X_clean)
|
||||
|
||||
all_preds = np.zeros((X_scaled.shape[0], self.n_bootstrap))
|
||||
for i, boot_models in enumerate(self.bootstrap_models):
|
||||
preds = []
|
||||
for model in boot_models:
|
||||
preds.append(model.predict(X_scaled))
|
||||
meta = self.meta_learner.predict(np.column_stack(preds))
|
||||
all_preds[:, i] = meta
|
||||
|
||||
mean = all_preds.mean(axis=1)
|
||||
std = all_preds.std(axis=1, ddof=1)
|
||||
return mean, std
|
||||
|
||||
def feature_importance(self, top_n: int = 20) -> pd.DataFrame:
|
||||
"""Compute SHAP-style feature importance (LightGBM native)."""
|
||||
if self.lgb_model is None:
|
||||
return pd.DataFrame()
|
||||
|
||||
importance = self.lgb_model.feature_importances_
|
||||
df = pd.DataFrame({
|
||||
"feature": self.feature_names,
|
||||
"importance": importance,
|
||||
}).sort_values("importance", ascending=False)
|
||||
return df.head(top_n)
|
||||
@@ -0,0 +1,179 @@
|
||||
"""Temporal Graph Neural Network for player interaction modeling.
|
||||
|
||||
Models how player-to-player interactions (e.g., winger crosses → striker goals)
|
||||
influence match outcomes. Uses a simplified ST-GCN architecture with PyTorch
|
||||
Geometric when available, falling back to a lightweight interaction feature
|
||||
extractor otherwise.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from .base_model import BaseModel
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class TemporalGNN(BaseModel):
|
||||
"""Temporal Graph Neural Network for player interactions.
|
||||
|
||||
When PyTorch Geometric is available, uses a proper graph convolution
|
||||
network. Otherwise, extracts interaction features from co-occurrence
|
||||
patterns (pass connections, assist-to-scorer edges, cross patterns).
|
||||
"""
|
||||
|
||||
def __init__(self, model_dir: str = "models_trained", hidden_dim: int = 64,
|
||||
num_layers: int = 2, window_size: int = 5):
|
||||
super().__init__(model_dir)
|
||||
self.hidden_dim = hidden_dim
|
||||
self.num_layers = num_layers
|
||||
self.window_size = window_size
|
||||
self.interaction_edges = {}
|
||||
self.interaction_weights = {}
|
||||
self.feature_names = None
|
||||
self.gnn_model = None
|
||||
self._use_torch = False
|
||||
|
||||
def _try_init_torch(self):
|
||||
"""Try initializing PyTorch Geometric GNN."""
|
||||
try:
|
||||
import torch
|
||||
self._use_torch = True
|
||||
logger.info("PyTorch available for T-GNN")
|
||||
return True
|
||||
except ImportError:
|
||||
logger.info("PyTorch not installed; using interaction features mode")
|
||||
return False
|
||||
|
||||
def build_graph(self, historical_data: pd.DataFrame):
|
||||
"""Build player interaction graph from historical match data.
|
||||
|
||||
Args:
|
||||
historical_data: DataFrame with columns:
|
||||
[player, teammate, passes_to, assists_to, crosses_to, matchday]
|
||||
"""
|
||||
edges = {}
|
||||
weights = {}
|
||||
|
||||
for (player, teammate), group in historical_data.groupby(["player", "teammate"]):
|
||||
edge_key = (player, teammate)
|
||||
total_interactions = (
|
||||
group["passes_to"].sum() +
|
||||
group["assists_to"].sum() * 5 +
|
||||
group["crosses_to"].sum() * 3
|
||||
)
|
||||
edges[edge_key] = total_interactions
|
||||
weights[edge_key] = total_interactions / max(len(group), 1)
|
||||
|
||||
self.interaction_edges = edges
|
||||
self.interaction_weights = weights
|
||||
logger.info(f"Built interaction graph: {len(edges)} edges")
|
||||
|
||||
def extract_interaction_features(self, player: str, teammates: list) -> dict:
|
||||
"""Extract temporal interaction features for a player's connections."""
|
||||
features = {}
|
||||
|
||||
# Sum of all outgoing interactions
|
||||
total_out = sum(
|
||||
w for (p, t), w in self.interaction_edges.items()
|
||||
if p == player
|
||||
)
|
||||
|
||||
# Sum of all incoming interactions
|
||||
total_in = sum(
|
||||
w for (p, t), w in self.interaction_edges.items()
|
||||
if t == player
|
||||
)
|
||||
|
||||
features["interaction_outgoing_sum"] = total_out
|
||||
features["interaction_incoming_sum"] = total_in
|
||||
features["interaction_net_flow"] = total_out - total_in
|
||||
|
||||
# Top teammate features
|
||||
teammate_interactions = sorted(
|
||||
[(t, w) for (p, t), w in self.interaction_edges.items() if p == player],
|
||||
key=lambda x: x[1],
|
||||
reverse=True,
|
||||
)
|
||||
for i in range(min(5, len(teammate_interactions))):
|
||||
features[f"interaction_top_{i+1}_weight"] = teammate_interactions[i][1]
|
||||
|
||||
# Fill gaps
|
||||
for i in range(len(teammate_interactions), 5):
|
||||
features[f"interaction_top_{i+1}_weight"] = 0.0
|
||||
|
||||
# Synergy score: how complementary this player is with known teammates
|
||||
if teammates:
|
||||
synergy = 0.0
|
||||
for tm in teammates:
|
||||
for (p, t), w in self.interaction_edges.items():
|
||||
if (p == player and t == tm) or (p == tm and t == player):
|
||||
synergy += w
|
||||
features["interaction_synergy"] = synergy
|
||||
else:
|
||||
features["interaction_synergy"] = 0.0
|
||||
|
||||
return features
|
||||
|
||||
def fit(self, X: pd.DataFrame, y: pd.Series, **kwargs):
|
||||
"""Fit the interaction model.
|
||||
|
||||
X should contain 'player' and 'team' columns for graph construction.
|
||||
"""
|
||||
self._try_init_torch()
|
||||
self.feature_names = list(X.columns)
|
||||
|
||||
if "player" in X.columns and "team" in X.columns:
|
||||
# Build simple interaction graph from team co-membership
|
||||
graph_data = []
|
||||
for team, group in X.groupby("team"):
|
||||
players = group["player"].unique()
|
||||
for i, p1 in enumerate(players):
|
||||
for p2 in players[i + 1:]:
|
||||
graph_data.append({
|
||||
"player": p1, "teammate": p2,
|
||||
"passes_to": 0, "assists_to": 0, "crosses_to": 0,
|
||||
})
|
||||
if graph_data:
|
||||
self.build_graph(pd.DataFrame(graph_data))
|
||||
|
||||
logger.info(f"T-GNN fitted with {len(self.interaction_edges)} interaction edges")
|
||||
return self
|
||||
|
||||
def predict(self, X: pd.DataFrame) -> np.ndarray:
|
||||
"""Predict based on interaction features.
|
||||
|
||||
Returns zeros if no features extracted — this is a supplementary model.
|
||||
"""
|
||||
if not self.interaction_edges:
|
||||
return np.zeros(X.shape[0])
|
||||
|
||||
# For now, returns a 0-1 normalized interaction score
|
||||
features = []
|
||||
for _, row in X.iterrows():
|
||||
player = row.get("player", "") or row.get("name", "")
|
||||
team = row.get("team", "")
|
||||
teammates = list(X[X["team"] == team]["player"].unique()) if team else []
|
||||
feats = self.extract_interaction_features(player, teammates)
|
||||
features.append(feats)
|
||||
|
||||
# Normalize synergy to [0, 1] range
|
||||
df = pd.DataFrame(features)
|
||||
synergy = df.get("interaction_synergy", pd.Series([0] * len(df)))
|
||||
max_val = synergy.max()
|
||||
if max_val > 0:
|
||||
return (synergy / max_val).values
|
||||
return np.zeros(X.shape[0])
|
||||
|
||||
def compute_interaction_bonus(self, player_a: str, player_b: str) -> float:
|
||||
"""Compute bonus factor for two players in the same lineup."""
|
||||
weight = 0.0
|
||||
for (p, t), w in self.interaction_edges.items():
|
||||
if (p == player_a and t == player_b) or (p == player_b and t == player_a):
|
||||
weight = max(weight, w)
|
||||
|
||||
max_weight = max(self.interaction_edges.values()) if self.interaction_edges else 1.0
|
||||
return weight / max(max_weight, 1.0) if weight > 0 else 0.0
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Model training pipeline with hyperparameter tuning via Optuna."""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.model_selection import TimeSeriesSplit
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
from .base_model import BaseModel
|
||||
from .gbm_model import GBMEnsemble
|
||||
from .card_model import CardClassifier, GoalProbabilityModel, PenaltyModel
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ModelTrainer:
|
||||
"""Orchestrates model training, hyperparameter tuning, and evaluation."""
|
||||
|
||||
def __init__(self, model_dir: str = "models_trained", n_trials: int = 50):
|
||||
self.model_dir = Path(model_dir)
|
||||
self.model_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.n_trials = n_trials
|
||||
self.scaler = None
|
||||
|
||||
def tune_gbm(
|
||||
self, X: pd.DataFrame, y: pd.Series, timeout: int = 1800
|
||||
) -> GBMEnsemble:
|
||||
"""Hyperparameter tune the GBM ensemble using Optuna."""
|
||||
try:
|
||||
import optuna
|
||||
except ImportError:
|
||||
logger.warning("Optuna not installed. Using default hyperparameters.")
|
||||
model = GBMEnsemble(model_dir=str(self.model_dir))
|
||||
model.fit(X, y)
|
||||
return model
|
||||
|
||||
X_clean = X.select_dtypes(include=[np.number]).fillna(0)
|
||||
scaler = StandardScaler()
|
||||
X_scaled = scaler.fit_transform(X_clean)
|
||||
|
||||
tscv = TimeSeriesSplit(n_splits=3)
|
||||
|
||||
def objective(trial):
|
||||
params = {
|
||||
"n_estimators": trial.suggest_int("n_estimators", 100, 1000, step=100),
|
||||
"learning_rate": trial.suggest_float("learning_rate", 0.01, 0.3, log=True),
|
||||
"max_depth": trial.suggest_int("max_depth", 3, 12),
|
||||
}
|
||||
model = GBMEnsemble(
|
||||
model_dir=str(self.model_dir),
|
||||
n_estimators=params["n_estimators"],
|
||||
learning_rate=params["learning_rate"],
|
||||
max_depth=params["max_depth"],
|
||||
n_bootstrap=30,
|
||||
)
|
||||
scores = []
|
||||
for train_idx, val_idx in tscv.split(X_scaled):
|
||||
model.fit(
|
||||
pd.DataFrame(X_scaled[train_idx], columns=X.columns),
|
||||
y.iloc[train_idx],
|
||||
)
|
||||
preds = model.predict(
|
||||
pd.DataFrame(X_scaled[val_idx], columns=X.columns)
|
||||
)
|
||||
rmse = np.sqrt(np.mean((preds - y.iloc[val_idx]) ** 2))
|
||||
scores.append(rmse)
|
||||
return np.mean(scores)
|
||||
|
||||
study = optuna.create_study(direction="minimize")
|
||||
study.optimize(objective, n_trials=self.n_trials, timeout=timeout)
|
||||
|
||||
best_params = study.best_params
|
||||
logger.info(f"Best GBM params: {best_params}")
|
||||
logger.info(f"Best RMSE: {study.best_value:.4f}")
|
||||
|
||||
model = GBMEnsemble(
|
||||
model_dir=str(self.model_dir),
|
||||
n_estimators=best_params["n_estimators"],
|
||||
learning_rate=best_params["learning_rate"],
|
||||
max_depth=best_params["max_depth"],
|
||||
)
|
||||
model.fit(X, y)
|
||||
return model
|
||||
|
||||
def train_full_pipeline(
|
||||
self,
|
||||
X: pd.DataFrame,
|
||||
y_fantavote: pd.Series,
|
||||
y_yellow: Optional[pd.Series] = None,
|
||||
y_red: Optional[pd.Series] = None,
|
||||
y_goals: Optional[pd.Series] = None,
|
||||
penalty_data: Optional[pd.DataFrame] = None,
|
||||
) -> dict:
|
||||
"""Train all models in the prediction pipeline.
|
||||
|
||||
Returns:
|
||||
dict with trained models: fantavote, yellow_card, red_card,
|
||||
goal_model, penalty_model.
|
||||
"""
|
||||
models = {}
|
||||
|
||||
# Main fantavote model
|
||||
logger.info("Training fantavote model...")
|
||||
models["fantavote"] = self.tune_gbm(X, y_fantavote)
|
||||
|
||||
# Card classifiers
|
||||
if y_yellow is not None:
|
||||
logger.info("Training yellow card classifier...")
|
||||
card = CardClassifier(card_type="yellow", model_dir=str(self.model_dir))
|
||||
card.fit(X, y_yellow)
|
||||
models["yellow_card"] = card
|
||||
|
||||
if y_red is not None:
|
||||
logger.info("Training red card classifier...")
|
||||
card = CardClassifier(card_type="red", model_dir=str(self.model_dir))
|
||||
card.fit(X, y_red)
|
||||
models["red_card"] = card
|
||||
|
||||
# Goal probability model
|
||||
if y_goals is not None:
|
||||
logger.info("Training goal probability model...")
|
||||
goal_model = GoalProbabilityModel(model_dir=str(self.model_dir))
|
||||
goal_model.fit(X, y_goals)
|
||||
models["goal_model"] = goal_model
|
||||
|
||||
# Penalty model
|
||||
if penalty_data is not None:
|
||||
logger.info("Fitting penalty model...")
|
||||
pen_model = PenaltyModel()
|
||||
pen_model.fit(penalty_data)
|
||||
models["penalty_model"] = pen_model
|
||||
|
||||
return models
|
||||
|
||||
def evaluate(
|
||||
self, model: BaseModel, X: pd.DataFrame, y: pd.Series
|
||||
) -> dict:
|
||||
"""Evaluate a model with standard regression metrics."""
|
||||
preds = model.predict(X)
|
||||
errors = preds - y.values
|
||||
|
||||
rmse = np.sqrt(np.mean(errors ** 2))
|
||||
mae = np.mean(np.abs(errors))
|
||||
r2 = 1 - np.sum(errors ** 2) / np.sum((y - y.mean()) ** 2)
|
||||
|
||||
# Within-0.5 accuracy (how often within 0.5 of the true vote)
|
||||
within_half = np.mean(np.abs(errors) <= 0.5)
|
||||
|
||||
return {
|
||||
"rmse": rmse,
|
||||
"mae": mae,
|
||||
"r2": r2,
|
||||
"within_0.5": within_half,
|
||||
}
|
||||
@@ -0,0 +1 @@
|
||||
"""Optimization modules for auction and lineup selection."""
|
||||
@@ -0,0 +1,263 @@
|
||||
"""Auction strategy solver using Mixed-Integer Linear Programming (MILP).
|
||||
|
||||
Formulates the Fantacalcio draft as a multi-period stochastic knapsack problem:
|
||||
- Maximize expected total season points subject to budget and role constraints.
|
||||
- Supports both "Classic Auction" and "Grid Auction" (Asta a Griglia) logic.
|
||||
|
||||
Uses PuLP (free) with fallback formatting for Gurobi (academic license).
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class AuctionConfig:
|
||||
total_budget: int = 500
|
||||
n_players: int = 25
|
||||
n_gk: int = 3
|
||||
n_def: int = 8
|
||||
n_mid: int = 8
|
||||
n_fwd: int = 6
|
||||
|
||||
# Player value limits (fraction of budget)
|
||||
max_single_bid_pct: float = 0.4
|
||||
|
||||
# Grid auction specific
|
||||
grid_mode: bool = False
|
||||
grid_rounds: int = 10
|
||||
players_per_round: int = 3
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlayerValuation:
|
||||
name: str
|
||||
team: str
|
||||
role: str # P, D, C, A
|
||||
projected_points: float
|
||||
market_value: float
|
||||
ceiling_price: float # maximum rational bid
|
||||
is_must_buy: bool = False
|
||||
|
||||
|
||||
class AuctionSolver:
|
||||
"""MILP-based auction strategy optimizer.
|
||||
|
||||
Solves: maximize sum(projected_points[i] * x[i] * minutes_weight[i])
|
||||
subject to sum(price[i] * x[i]) <= budget, role quotas.
|
||||
"""
|
||||
|
||||
def __init__(self, config: Optional[AuctionConfig] = None, solver: str = "pulp"):
|
||||
self.config = config or AuctionConfig()
|
||||
self.solver = solver
|
||||
self.players = []
|
||||
self.solution = None
|
||||
|
||||
def add_players(self, valuation_df: pd.DataFrame):
|
||||
"""Add players from a DataFrame with columns: name, team, role, projected_points, market_value."""
|
||||
self.players = []
|
||||
for _, row in valuation_df.iterrows():
|
||||
projected = float(row["projected_points"])
|
||||
market = float(row.get("market_value", 0))
|
||||
self.players.append(PlayerValuation(
|
||||
name=str(row["name"]),
|
||||
team=str(row.get("team", "")),
|
||||
role=str(row["role"]),
|
||||
projected_points=projected,
|
||||
market_value=market,
|
||||
ceiling_price=projected * 5, # rough heuristic
|
||||
))
|
||||
logger.info(f"Loaded {len(self.players)} players for auction optimization")
|
||||
|
||||
def _estimate_price(self, player: PlayerValuation, opponent_budget: float) -> float:
|
||||
"""Estimate market clearing price for a player based on game theory.
|
||||
|
||||
In a competitive auction, the price approaches the player's marginal
|
||||
value minus the next-best alternative.
|
||||
"""
|
||||
same_role = [p for p in self.players if p.role == player.role and p.name != player.name]
|
||||
best_alternative = max((p.projected_points for p in same_role), default=0)
|
||||
value_over_replacement = player.projected_points - best_alternative
|
||||
return min(player.ceiling_price, max(player.market_value, value_over_replacement * 3))
|
||||
|
||||
def solve(self) -> dict:
|
||||
"""Solve the auction knapsack problem.
|
||||
|
||||
Returns:
|
||||
dict with: selected_players, total_cost, total_value, status.
|
||||
"""
|
||||
try:
|
||||
import pulp
|
||||
except ImportError:
|
||||
logger.warning("PuLP not installed. Falling back to greedy heuristic.")
|
||||
return self._solve_greedy()
|
||||
|
||||
prob = pulp.LpProblem("Fantacalcio_Auction", pulp.LpMaximize)
|
||||
|
||||
# Decision variables
|
||||
x = {}
|
||||
for i, player in enumerate(self.players):
|
||||
x[i] = pulp.LpVariable(f"x_{i}", cat="Binary")
|
||||
|
||||
# Objective: maximize total projected points
|
||||
prob += pulp.lpSum(
|
||||
self.players[i].projected_points * x[i] for i in range(len(self.players))
|
||||
)
|
||||
|
||||
# Budget constraint
|
||||
prices = [self._estimate_price(p, self.config.total_budget) for p in self.players]
|
||||
prob += pulp.lpSum(prices[i] * x[i] for i in range(len(self.players))) <= self.config.total_budget
|
||||
|
||||
# Role quota constraints
|
||||
gk_indices = [i for i, p in enumerate(self.players) if p.role == "P"]
|
||||
def_indices = [i for i, p in enumerate(self.players) if p.role == "D"]
|
||||
mid_indices = [i for i, p in enumerate(self.players) if p.role == "C"]
|
||||
fwd_indices = [i for i, p in enumerate(self.players) if p.role == "A"]
|
||||
|
||||
prob += pulp.lpSum(x[i] for i in gk_indices) == self.config.n_gk
|
||||
prob += pulp.lpSum(x[i] for i in def_indices) == self.config.n_def
|
||||
prob += pulp.lpSum(x[i] for i in mid_indices) == self.config.n_mid
|
||||
prob += pulp.lpSum(x[i] for i in fwd_indices) == self.config.n_fwd
|
||||
|
||||
# Total squad size
|
||||
total_slots = self.config.n_gk + self.config.n_def + self.config.n_mid + self.config.n_fwd
|
||||
prob += pulp.lpSum(x[i] for i in range(len(self.players))) == total_slots
|
||||
|
||||
# Max single bid
|
||||
max_bid = self.config.total_budget * self.config.max_single_bid_pct
|
||||
for i in range(len(self.players)):
|
||||
prob += prices[i] * x[i] <= max_bid
|
||||
|
||||
# Solve
|
||||
prob.solve(pulp.PULP_CBC_CMD(msg=False))
|
||||
status = pulp.LpStatus[prob.status]
|
||||
|
||||
if status != "Optimal":
|
||||
logger.warning(f"Solver status: {status}. Falling back to greedy.")
|
||||
return self._solve_greedy()
|
||||
|
||||
selected = []
|
||||
total_cost = 0
|
||||
total_value = 0
|
||||
|
||||
for i, player in enumerate(self.players):
|
||||
if pulp.value(x[i]) > 0.5:
|
||||
selected.append({
|
||||
"player": player.name,
|
||||
"team": player.team,
|
||||
"role": player.role,
|
||||
"estimated_price": prices[i],
|
||||
"projected_points": player.projected_points,
|
||||
"value_ratio": player.projected_points / max(prices[i], 1),
|
||||
})
|
||||
total_cost += prices[i]
|
||||
total_value += player.projected_points
|
||||
|
||||
self.solution = {
|
||||
"selected_players": pd.DataFrame(selected),
|
||||
"total_cost": total_cost,
|
||||
"total_value": total_value,
|
||||
"remaining_budget": self.config.total_budget - total_cost,
|
||||
"status": status,
|
||||
}
|
||||
|
||||
logger.info(
|
||||
f"Auction solved: {len(selected)} players, "
|
||||
f"cost={total_cost}/{self.config.total_budget}, "
|
||||
f"value={total_value:.1f}"
|
||||
)
|
||||
return self.solution
|
||||
|
||||
def _solve_greedy(self) -> dict:
|
||||
"""Greedy knapsack solver as fallback when PuLP is unavailable."""
|
||||
role_quotas = {
|
||||
"P": self.config.n_gk, "D": self.config.n_def,
|
||||
"C": self.config.n_mid, "A": self.config.n_fwd,
|
||||
}
|
||||
role_filled = {"P": 0, "D": 0, "C": 0, "A": 0}
|
||||
budget_remaining = self.config.total_budget
|
||||
|
||||
# Score players by projected_points / estimated_price (value efficiency)
|
||||
scored = []
|
||||
for p in self.players:
|
||||
price = self._estimate_price(p, budget_remaining)
|
||||
scored.append((p.projected_points / max(price, 1), p, price))
|
||||
scored.sort(reverse=True)
|
||||
|
||||
selected = []
|
||||
for _, player, price in scored:
|
||||
role = player.role
|
||||
if role_filled[role] >= role_quotas[role]:
|
||||
continue
|
||||
if price > budget_remaining:
|
||||
continue
|
||||
selected.append({
|
||||
"player": player.name,
|
||||
"team": player.team,
|
||||
"role": player.role,
|
||||
"estimated_price": price,
|
||||
"projected_points": player.projected_points,
|
||||
"value_ratio": player.projected_points / max(price, 1),
|
||||
})
|
||||
budget_remaining -= price
|
||||
role_filled[role] += 1
|
||||
|
||||
total_cost = self.config.total_budget - budget_remaining
|
||||
total_value = sum(s["projected_points"] for s in selected)
|
||||
|
||||
self.solution = {
|
||||
"selected_players": pd.DataFrame(selected),
|
||||
"total_cost": total_cost,
|
||||
"total_value": total_value,
|
||||
"remaining_budget": budget_remaining,
|
||||
"status": "Greedy",
|
||||
}
|
||||
return self.solution
|
||||
|
||||
def grid_auction_strategy(self, round_players: list) -> dict:
|
||||
"""Grid Auction (Asta a Griglia) strategy.
|
||||
|
||||
For a grid round where N players are available simultaneously,
|
||||
compute optimal allocation using Minimax game theory.
|
||||
|
||||
Args:
|
||||
round_players: list of PlayerValuation objects available this round.
|
||||
|
||||
Returns:
|
||||
dict with bid recommendations for each player.
|
||||
"""
|
||||
config = self.config
|
||||
config.grid_mode = True
|
||||
|
||||
# For each available player, compute the "regret" of not bidding enough
|
||||
recommendations = {}
|
||||
for player in round_players:
|
||||
# Optimal bid = player's value minus next best alternative in that role
|
||||
same_role = [
|
||||
p for p in round_players if p.role == player.role and p.name != player.name
|
||||
]
|
||||
next_best = max((p.projected_points for p in same_role), default=0)
|
||||
|
||||
# Competitive equilibrium price
|
||||
fair_price = self._estimate_price(player, config.total_budget)
|
||||
|
||||
# Max bid: don't exceed what makes this player worse value than the next best
|
||||
max_rational_bid = max(
|
||||
fair_price,
|
||||
(player.projected_points - next_best) * 5,
|
||||
)
|
||||
|
||||
recommendations[player.name] = {
|
||||
"fair_price": fair_price,
|
||||
"max_bid": max_rational_bid,
|
||||
"recommended_bid": fair_price * 0.85, # conservative
|
||||
"value_over_replacement": player.projected_points - next_best,
|
||||
}
|
||||
|
||||
return recommendations
|
||||
@@ -0,0 +1,316 @@
|
||||
"""Weekly lineup optimizer using Monte Carlo Tree Search (MCTS).
|
||||
|
||||
Selects the optimal 11 players and captain to maximize win probability
|
||||
against the opponent's projected lineup, rather than just maximizing
|
||||
expected points. Incorporates defense modifier (Modificatore) and
|
||||
clean sheet bonuses.
|
||||
|
||||
Replaces the notebook 7's manual simulation approach.
|
||||
"""
|
||||
|
||||
import copy
|
||||
import logging
|
||||
import math
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class LineupConstraints:
|
||||
min_defenders: int = 3
|
||||
max_defenders: int = 5
|
||||
min_midfielders: int = 3
|
||||
max_midfielders: int = 5
|
||||
min_forwards: int = 1
|
||||
max_forwards: int = 3
|
||||
total_players: int = 11
|
||||
use_modificatore: bool = True
|
||||
|
||||
# Modificatore Difesa thresholds
|
||||
mod_threshold_6_0: float = 1.0
|
||||
mod_threshold_6_5: float = 3.0
|
||||
mod_threshold_7_0: float = 6.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlayerScore:
|
||||
name: str
|
||||
role: str
|
||||
team: str
|
||||
oppteam: str
|
||||
home: bool
|
||||
fv_mean: float
|
||||
fv_std: float
|
||||
mv_mean: float
|
||||
mv_std: float
|
||||
starter_prob: float = 1.0
|
||||
cs_prob: float = 0.0
|
||||
captain_multiplier: float = 1.0
|
||||
|
||||
|
||||
class MCTSNode:
|
||||
"""MCTS node representing a partial lineup."""
|
||||
|
||||
def __init__(self, state=None, parent=None):
|
||||
self.state = state or [] # list of PlayerScore objects
|
||||
self.parent = parent
|
||||
self.children = []
|
||||
self.visits = 0
|
||||
self.wins = 0.0
|
||||
self.untried_actions = []
|
||||
|
||||
def add_child(self, child_state):
|
||||
child = MCTSNode(child_state, self)
|
||||
self.children.append(child)
|
||||
return child
|
||||
|
||||
def update(self, reward: float):
|
||||
self.visits += 1
|
||||
self.wins += reward
|
||||
if self.parent:
|
||||
self.parent.update(reward)
|
||||
|
||||
def ucb1(self, exploration: float = 1.414) -> float:
|
||||
if self.visits == 0:
|
||||
return float("inf")
|
||||
parent_visits = self.parent.visits if self.parent else self.visits
|
||||
exploitation = self.wins / self.visits
|
||||
exploration_term = exploration * math.sqrt(math.log(parent_visits) / self.visits)
|
||||
return exploitation + exploration_term
|
||||
|
||||
def best_child(self, exploration: float = 1.414) -> "MCTSNode":
|
||||
return max(self.children, key=lambda c: c.ucb1(exploration))
|
||||
|
||||
|
||||
class LineupSolver:
|
||||
"""MCTS-based weekly lineup optimizer with opponent modeling."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
iters: int = 2000,
|
||||
opponent_avg: float = 70.0,
|
||||
opponent_std: float = 8.0,
|
||||
constraints: Optional[LineupConstraints] = None,
|
||||
):
|
||||
self.iters = iters
|
||||
self.opponent_avg = opponent_avg
|
||||
self.opponent_std = opponent_std
|
||||
self.constraints = constraints or LineupConstraints()
|
||||
|
||||
def _validate_lineup(self, players: list) -> bool:
|
||||
"""Check if a lineup satisfies all constraints."""
|
||||
if len(players) != self.constraints.total_players:
|
||||
return False
|
||||
|
||||
roles = [p.role for p in players]
|
||||
n_def = sum(1 for r in roles if r == "D")
|
||||
n_mid = sum(1 for r in roles if r == "C")
|
||||
n_fwd = sum(1 for r in roles if r == "A")
|
||||
n_gk = sum(1 for r in roles if r == "P")
|
||||
|
||||
if n_gk != 1:
|
||||
return False
|
||||
if not (self.constraints.min_defenders <= n_def <= self.constraints.max_defenders):
|
||||
return False
|
||||
if not (self.constraints.min_midfielders <= n_mid <= self.constraints.max_midfielders):
|
||||
return False
|
||||
if not (self.constraints.min_forwards <= n_fwd <= self.constraints.max_forwards):
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def _modificatore_bonus(self, defender_mvs: list) -> float:
|
||||
"""Compute Modificatore Difesa bonus.
|
||||
|
||||
Average of best 3 defender match votes:
|
||||
>= 7.0 → +6, >= 6.5 → +3, >= 6.0 → +1
|
||||
"""
|
||||
if not self.constraints.use_modificatore or len(defender_mvs) < 3:
|
||||
return 0.0
|
||||
|
||||
best_3 = sorted(defender_mvs, reverse=True)[:3]
|
||||
avg = np.mean(best_3)
|
||||
|
||||
if avg >= 7.0:
|
||||
return self.constraints.mod_threshold_7_0
|
||||
elif avg >= 6.5:
|
||||
return self.constraints.mod_threshold_6_5
|
||||
elif avg >= 6.0:
|
||||
return self.constraints.mod_threshold_6_0
|
||||
return 0.0
|
||||
|
||||
def simulate_match(self, players: list, n_samples: int = 1000) -> np.ndarray:
|
||||
"""Monte Carlo simulation of a lineup's total score.
|
||||
|
||||
Returns array of n_samples total scores.
|
||||
"""
|
||||
total = np.zeros(n_samples)
|
||||
rng = np.random.RandomState()
|
||||
|
||||
for player in players:
|
||||
if rng.random() > player.starter_prob:
|
||||
continue
|
||||
|
||||
fv_samples = rng.normal(player.fv_mean, max(player.fv_std, 0.1), n_samples)
|
||||
fv_samples = np.clip(fv_samples, 0, 15)
|
||||
total += fv_samples * player.captain_multiplier
|
||||
|
||||
# Clean sheet bonus
|
||||
gks = [p for p in players if p.role == "P"]
|
||||
if gks and self.constraints.use_modificatore:
|
||||
gk = gks[0]
|
||||
cs_samples = rng.binomial(1, gk.cs_prob, n_samples)
|
||||
total += cs_samples
|
||||
|
||||
# Modificatore Difesa
|
||||
if self.constraints.use_modificatore:
|
||||
defenders = [p for p in players if p.role == "D"]
|
||||
if len(defenders) >= 3:
|
||||
mv_samples = np.array([
|
||||
rng.normal(d.mv_mean, max(d.mv_std, 0.1), n_samples) for d in defenders
|
||||
])
|
||||
best_3_avg = np.mean(np.sort(mv_samples, axis=0)[-3:], axis=0)
|
||||
mod = np.zeros(n_samples)
|
||||
mod[best_3_avg >= 7.0] = self.constraints.mod_threshold_7_0
|
||||
mod[(best_3_avg >= 6.5) & (best_3_avg < 7.0)] = self.constraints.mod_threshold_6_5
|
||||
mod[(best_3_avg >= 6.0) & (best_3_avg < 6.5)] = self.constraints.mod_threshold_6_0
|
||||
total += mod
|
||||
|
||||
return total
|
||||
|
||||
def win_probability(self, own_total: np.ndarray) -> float:
|
||||
"""Probability of beating the opponent."""
|
||||
opp_total = np.random.normal(self.opponent_avg, self.opponent_std, len(own_total))
|
||||
return np.mean(own_total > opp_total)
|
||||
|
||||
def optimize(
|
||||
self, player_pool: list, captain_candidates: Optional[list] = None
|
||||
) -> dict:
|
||||
"""Optimize lineup and captain using MCTS.
|
||||
|
||||
Args:
|
||||
player_pool: list of PlayerScore objects (all squad players).
|
||||
captain_candidates: optional subset to test as captain.
|
||||
|
||||
Returns:
|
||||
dict with: lineup (list), captain, expected_points, win_prob,
|
||||
lineup_distribution, captain_comparison.
|
||||
"""
|
||||
# Step 1: Generate candidate lineups
|
||||
candidates = self._generate_candidates(player_pool)
|
||||
|
||||
if not candidates:
|
||||
logger.warning("No valid lineups found")
|
||||
return {"lineup": [], "captain": "", "expected_points": 0, "win_prob": 0}
|
||||
|
||||
# Step 2: Evaluate each lineup
|
||||
best_lineup = None
|
||||
best_captain = None
|
||||
best_win_prob = -1
|
||||
best_mean = 0
|
||||
|
||||
for lineup in candidates:
|
||||
# Test each captain
|
||||
for captain_idx in (captain_candidates or range(len(lineup))):
|
||||
test_lineup = copy.deepcopy(lineup)
|
||||
for i, p in enumerate(test_lineup):
|
||||
p.captain_multiplier = 2.0 if i == captain_idx else 1.0
|
||||
|
||||
own_scores = self.simulate_match(test_lineup)
|
||||
win_prob = self.win_probability(own_scores)
|
||||
mean_score = np.mean(own_scores)
|
||||
|
||||
if win_prob > best_win_prob:
|
||||
best_win_prob = win_prob
|
||||
best_lineup = test_lineup
|
||||
best_captain = test_lineup[captain_idx].name
|
||||
best_mean = mean_score
|
||||
|
||||
return {
|
||||
"lineup": [p.name for p in best_lineup],
|
||||
"captain": best_captain,
|
||||
"expected_points": best_mean,
|
||||
"win_probability": best_win_prob,
|
||||
}
|
||||
|
||||
def _generate_candidates(self, pool: list, max_candidates: int = 200) -> list:
|
||||
"""Generate valid lineup candidates from the player pool."""
|
||||
gks = [p for p in pool if p.role == "P"]
|
||||
defs = [p for p in pool if p.role == "D"]
|
||||
mids = [p for p in pool if p.role == "C"]
|
||||
fwds = [p for p in pool if p.role == "A"]
|
||||
|
||||
candidates = []
|
||||
rng = np.random.RandomState(42)
|
||||
|
||||
# Formations to try
|
||||
formations = [
|
||||
(3, 4, 3), (4, 4, 2), (4, 3, 3),
|
||||
(3, 5, 2), (4, 2, 3), (5, 3, 2),
|
||||
]
|
||||
|
||||
for n_def, n_mid, n_fwd in formations:
|
||||
if n_def > len(defs) or n_mid > len(mids) or n_fwd > len(fwds) or not gks:
|
||||
continue
|
||||
|
||||
for _ in range(min(max_candidates // len(formations), 50)):
|
||||
sel_def = list(rng.choice(defs, n_def, replace=False))
|
||||
sel_mid = list(rng.choice(mids, n_mid, replace=False))
|
||||
sel_fwd = list(rng.choice(fwds, n_fwd, replace=False))
|
||||
sel_gk = [rng.choice(gks)]
|
||||
|
||||
# Sort by starter probability: best 11 start
|
||||
all_sel = sel_gk + sel_def + sel_mid + sel_fwd
|
||||
all_sel.sort(key=lambda p: p.starter_prob * p.fv_mean, reverse=True)
|
||||
# But keep exactly one GK
|
||||
if all_sel[0].role != "P":
|
||||
# Ensure GK is included
|
||||
non_gk = [p for p in all_sel if p.role != "P"]
|
||||
lineup_players = [sel_gk[0]] + non_gk[:10]
|
||||
else:
|
||||
lineup_players = all_sel[:11]
|
||||
|
||||
candidates.append(lineup_players)
|
||||
|
||||
# Also add greedy candidate: top by expected points
|
||||
greedy = sorted(pool, key=lambda p: p.starter_prob * p.fv_mean, reverse=True)
|
||||
gk = next(p for p in greedy if p.role == "P")
|
||||
rest = [p for p in greedy if p.role != "P"]
|
||||
candidates.append([gk] + rest[:10])
|
||||
|
||||
logger.info(f"Generated {len(candidates)} valid lineup candidates")
|
||||
return candidates
|
||||
|
||||
def compare_lineups(
|
||||
self, lineups: dict, n_samples: int = 5000
|
||||
) -> pd.DataFrame:
|
||||
"""Compare multiple candidate lineups with detailed stats.
|
||||
|
||||
Args:
|
||||
lineups: dict mapping lineup_name -> list of PlayerScore objects.
|
||||
|
||||
Returns:
|
||||
DataFrame with comparison metrics per lineup.
|
||||
"""
|
||||
results = []
|
||||
for name, players in lineups.items():
|
||||
scores = self.simulate_match(players, n_samples)
|
||||
win_prob = self.win_probability(scores)
|
||||
results.append({
|
||||
"lineup": name,
|
||||
"mean": np.mean(scores),
|
||||
"std": np.std(scores),
|
||||
"median": np.median(scores),
|
||||
"q25": np.percentile(scores, 25),
|
||||
"q75": np.percentile(scores, 75),
|
||||
"potential": np.mean(scores) + 2 * np.std(scores),
|
||||
"win_probability": win_prob,
|
||||
"ceiling_95": np.percentile(scores, 95),
|
||||
})
|
||||
|
||||
return pd.DataFrame(results).sort_values("win_probability", ascending=False)
|
||||
@@ -0,0 +1,127 @@
|
||||
"""Opponent behavior modeling for weekly lineup optimization.
|
||||
|
||||
Models opponent's historical transfer and lineup patterns to predict
|
||||
their most likely starting XI. Uses:
|
||||
- Historical transfer frequency (which players they tend to switch)
|
||||
- Recency bias (players bought recently are more likely to start)
|
||||
- Formation preferences
|
||||
"""
|
||||
|
||||
import logging
|
||||
from collections import defaultdict
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class OpponentModel:
|
||||
"""Models an opponent's likely lineup based on historical patterns."""
|
||||
|
||||
def __init__(self):
|
||||
self.transfer_history = []
|
||||
self.lineup_history = []
|
||||
self.formation_prefs = defaultdict(int)
|
||||
|
||||
def add_transfer_week(
|
||||
self, week: int, transfers_in: list, transfers_out: list
|
||||
):
|
||||
"""Record opponent's transfers for a given week."""
|
||||
self.transfer_history.append({
|
||||
"week": week,
|
||||
"in": transfers_in,
|
||||
"out": transfers_out,
|
||||
})
|
||||
|
||||
def add_lineup(
|
||||
self, week: int, lineup: list, formation: str
|
||||
):
|
||||
"""Record opponent's actual starting lineup."""
|
||||
self.lineup_history.append({
|
||||
"week": week,
|
||||
"lineup": lineup,
|
||||
"formation": formation,
|
||||
})
|
||||
self.formation_prefs[formation] += 1
|
||||
|
||||
def predict_lineup(
|
||||
self, current_squad: list
|
||||
) -> dict:
|
||||
"""Predict opponent's most likely starting XI.
|
||||
|
||||
Uses heuristic scoring combining:
|
||||
- Player quality (FV mean)
|
||||
- Recent inclusion rate
|
||||
- Formation fit
|
||||
|
||||
Returns:
|
||||
dict with: predicted_lineup, predicted_formation, expected_points.
|
||||
"""
|
||||
if not self.lineup_history:
|
||||
logger.info("No history — assuming optimal lineup")
|
||||
return self._default_prediction(current_squad)
|
||||
|
||||
# Frequency of each player being started
|
||||
start_counts = defaultdict(int)
|
||||
total_weeks = len(self.lineup_history)
|
||||
|
||||
for entry in self.lineup_history:
|
||||
for player in entry["lineup"]:
|
||||
start_counts[player] += 1
|
||||
|
||||
# Most common formation
|
||||
best_formation = (
|
||||
max(self.formation_prefs, key=self.formation_prefs.get)
|
||||
if self.formation_prefs else "4-4-2"
|
||||
)
|
||||
|
||||
# Score current squad members
|
||||
scored = []
|
||||
for player in current_squad:
|
||||
name = player.get("name", "")
|
||||
start_rate = start_counts.get(name, 0) / max(total_weeks, 1)
|
||||
fv = float(player.get("fv_mean", 6.0))
|
||||
score = fv * 0.6 + start_rate * 6.0 * 0.4
|
||||
scored.append((score, name, player))
|
||||
|
||||
scored.sort(key=lambda x: x[0], reverse=True)
|
||||
|
||||
# Select top 1 GK + 10 best
|
||||
gk = next((p for _, _, p in scored if p.get("role") == "P"), None)
|
||||
rest = [(s, n, p) for s, n, p in scored if p.get("role") != "P"]
|
||||
|
||||
lineup = [gk] if gk else []
|
||||
lineup.extend([p for _, _, p in rest[:11 - len(lineup)]])
|
||||
|
||||
expected = sum(
|
||||
(p.get("fv_mean", 0) * p.get("starter_prob", 1))
|
||||
for p in lineup
|
||||
)
|
||||
|
||||
return {
|
||||
"predicted_lineup": [p.get("name", "") for p in lineup],
|
||||
"predicted_formation": best_formation,
|
||||
"expected_points": expected,
|
||||
}
|
||||
|
||||
def _default_prediction(self, squad: list) -> dict:
|
||||
"""Default prediction: best 11 by expected points."""
|
||||
scored = [(p.get("fv_mean", 6.0) * p.get("starter_prob", 1.0), p) for p in squad]
|
||||
scored.sort(reverse=True)
|
||||
|
||||
gk = next((p for _, p in scored if p.get("role") == "P"), None)
|
||||
rest = [p for _, p in scored if p.get("role") != "P"]
|
||||
|
||||
lineup = [gk] if gk else []
|
||||
lineup.extend(rest[:11 - len(lineup)])
|
||||
|
||||
return {
|
||||
"predicted_lineup": [p.get("name", "") for p in lineup],
|
||||
"predicted_formation": "4-4-2",
|
||||
"expected_points": sum(
|
||||
p.get("fv_mean", 6.0) * p.get("starter_prob", 1.0)
|
||||
for p in lineup
|
||||
),
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
"""Transfer market analysis ('Svincolati' / free agent pool).
|
||||
|
||||
Identifies buy-low and sell-high targets using advanced metrics:
|
||||
- Regression to the mean: compares actual vs expected output
|
||||
- xG/xA vs actual goals/assists divergence
|
||||
- Minutes trending up/down
|
||||
- Market value arbitrage
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class TransferAnalyzer:
|
||||
"""Analyzes the Svincolati (free agent) market for arbitrage opportunities."""
|
||||
|
||||
def __init__(self, regression_factor: float = 0.3):
|
||||
self.regression_factor = regression_factor
|
||||
|
||||
def compute_expected_output(
|
||||
self, xg: float, xa: float, historical_mean: float
|
||||
) -> float:
|
||||
"""Compute regressed expected output using xG/xA.
|
||||
|
||||
Shrinks toward the player's historical mean (regression to the mean).
|
||||
"""
|
||||
raw_expected = xg * 3.0 + xa * 1.0 # convert to Fantavoto scale
|
||||
regressed = (
|
||||
self.regression_factor * historical_mean +
|
||||
(1 - self.regression_factor) * raw_expected
|
||||
)
|
||||
return regressed
|
||||
|
||||
def analyze_buy_low(
|
||||
self, players_df: pd.DataFrame, min_minutes: int = 180
|
||||
) -> pd.DataFrame:
|
||||
"""Identify buy-low candidates.
|
||||
|
||||
Criteria:
|
||||
- xG/xA significantly exceed actual output
|
||||
- Minutes trending up
|
||||
- Low market value relative to projection
|
||||
|
||||
Args:
|
||||
players_df: DataFrame with columns:
|
||||
[name, team, role, actual_fv_avg, xg, xa, minutes, market_value,
|
||||
minutes_trend, historical_fv_avg]
|
||||
|
||||
Returns:
|
||||
DataFrame of buy-low candidates ranked by opportunity.
|
||||
"""
|
||||
df = players_df.copy()
|
||||
df = df[df["minutes"] >= min_minutes]
|
||||
|
||||
if "xg" not in df.columns or "xa" not in df.columns:
|
||||
logger.warning("xG/xA data missing; using basic analysis")
|
||||
return pd.DataFrame()
|
||||
|
||||
# Expected fantavoto from xG/xA
|
||||
df["expected_fv"] = df.apply(
|
||||
lambda r: self.compute_expected_output(
|
||||
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
|
||||
),
|
||||
axis=1,
|
||||
)
|
||||
|
||||
# Divergence: expected minus actual
|
||||
df["fv_divergence"] = df["expected_fv"] - df.get("actual_fv_avg", 6.0)
|
||||
|
||||
df["buy_low_score"] = (
|
||||
df["fv_divergence"] * 2.0 + # underperformance signal
|
||||
df.get("minutes_trend", 0) * 0.5 + # trending up
|
||||
(1.0 / (df.get("market_value", 1) + 1)) * 10 # cheap
|
||||
)
|
||||
|
||||
buy_low = df[df["buy_low_score"] > 0].sort_values("buy_low_score", ascending=False)
|
||||
|
||||
result = buy_low[[
|
||||
"name", "team", "role", "actual_fv_avg", "expected_fv",
|
||||
"fv_divergence", "buy_low_score", "market_value",
|
||||
]].copy()
|
||||
|
||||
result["recommendation"] = "BUY-LOW"
|
||||
result["confidence"] = pd.cut(
|
||||
result["buy_low_score"],
|
||||
bins=[-np.inf, 1, 3, 5, np.inf],
|
||||
labels=["Low", "Medium", "High", "Very High"],
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"Found {len(result)} buy-low candidates "
|
||||
f"(avg divergence: {result['fv_divergence'].mean():.2f})"
|
||||
)
|
||||
return result
|
||||
|
||||
def analyze_sell_high(
|
||||
self, players_df: pd.DataFrame, min_minutes: int = 180
|
||||
) -> pd.DataFrame:
|
||||
"""Identify sell-high candidates.
|
||||
|
||||
Criteria:
|
||||
- Actual output exceeds xG/xA by large margin
|
||||
- Minutes trending down
|
||||
- High market value vs projection
|
||||
"""
|
||||
df = players_df.copy()
|
||||
df = df[df["minutes"] >= min_minutes]
|
||||
|
||||
if "xg" not in df.columns:
|
||||
return pd.DataFrame()
|
||||
|
||||
df["expected_fv"] = df.apply(
|
||||
lambda r: self.compute_expected_output(
|
||||
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
|
||||
),
|
||||
axis=1,
|
||||
)
|
||||
|
||||
# Overperformance
|
||||
df["fv_divergence"] = df.get("actual_fv_avg", 6.0) - df["expected_fv"]
|
||||
|
||||
df["sell_high_score"] = (
|
||||
df["fv_divergence"] * 3.0 + # overperformance signal
|
||||
(df.get("minutes_trend", 0) * -0.5 if "minutes_trend" in df.columns else 0)
|
||||
)
|
||||
|
||||
sell_high = df[df["sell_high_score"] > 1].sort_values("sell_high_score", ascending=False)
|
||||
|
||||
result = sell_high[[
|
||||
"name", "team", "role", "actual_fv_avg", "expected_fv",
|
||||
"fv_divergence", "sell_high_score",
|
||||
]].copy()
|
||||
|
||||
result["recommendation"] = "SELL-HIGH"
|
||||
logger.info(f"Found {len(result)} sell-high candidates")
|
||||
return result
|
||||
|
||||
def full_transfer_report(self, players_df: pd.DataFrame) -> dict:
|
||||
"""Generate complete transfer market report."""
|
||||
buy = self.analyze_buy_low(players_df)
|
||||
sell = self.analyze_sell_high(players_df)
|
||||
|
||||
return {
|
||||
"buy_low": buy,
|
||||
"sell_high": sell,
|
||||
"summary": (
|
||||
f"Buy-low targets: {len(buy)} players identified. "
|
||||
f"Sell-high targets: {len(sell)} players identified."
|
||||
),
|
||||
}
|
||||
+177
@@ -0,0 +1,177 @@
|
||||
"""Fantabeto 2026/27 pipeline orchestrator.
|
||||
|
||||
Central entry point for the entire data → features → models → optimization
|
||||
workflow. Supports incremental updates and targeted matchday processing.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Pipeline:
|
||||
"""Orchestrates the full Fantabeto pipeline."""
|
||||
|
||||
def __init__(self, data_dir: str = "data", model_dir: str = "models_trained"):
|
||||
self.data_dir = Path(data_dir)
|
||||
self.model_dir = Path(model_dir)
|
||||
self.data_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.model_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def scrape_fbref(
|
||||
self, seasons: list, current: bool = True
|
||||
) -> dict:
|
||||
"""Scrape FBref data for specified seasons."""
|
||||
from src.scraper.fbref_scraper import scrape_season, scrape_current_season
|
||||
|
||||
results = {}
|
||||
for season in seasons:
|
||||
results[season] = scrape_season(season, str(self.data_dir / "fbref"))
|
||||
if current:
|
||||
results["current"] = scrape_current_season(str(self.data_dir / "fbref"))
|
||||
return results
|
||||
|
||||
def process_votes(
|
||||
self, vote_dir: str, calendar_path: str
|
||||
):
|
||||
"""Process vote files into unified database."""
|
||||
import pandas as pd
|
||||
from src.features.vote_processor import VoteProcessor
|
||||
|
||||
processor = VoteProcessor()
|
||||
calendar = pd.read_excel(calendar_path) if calendar_path.endswith(".xlsx") else pd.read_csv(calendar_path)
|
||||
|
||||
all_votes = []
|
||||
for matchday in range(1, 39):
|
||||
df = processor.process_matchday(vote_dir, matchday, calendar)
|
||||
if not df.empty:
|
||||
all_votes.append(df)
|
||||
logger.info(f"Matchday {matchday}: {len(df)} votes")
|
||||
|
||||
combined = pd.concat(all_votes, ignore_index=True) if all_votes else pd.DataFrame()
|
||||
output_path = self.data_dir / "players_votes.xlsx"
|
||||
combined.to_excel(str(output_path), index=False)
|
||||
logger.info(f"Saved {len(combined)} votes to {output_path}")
|
||||
return combined
|
||||
|
||||
def build_features(
|
||||
self, fbref_dir: Optional[str] = None, votes_path: Optional[str] = None,
|
||||
roster_path: Optional[str] = None,
|
||||
):
|
||||
"""Build the full feature dataset."""
|
||||
import pandas as pd
|
||||
from src.features.player_features import PlayerFeatureBuilder
|
||||
from src.features.match_features import MatchFeatureBuilder
|
||||
|
||||
fbref_path = Path(fbref_dir or (self.data_dir / "fbref"))
|
||||
votes_file = votes_path or (self.data_dir / "players_votes.xlsx")
|
||||
|
||||
# Load data
|
||||
outfield = pd.read_csv(fbref_path / "outfield_players.csv")
|
||||
keepers = pd.read_csv(fbref_path / "keepers_players.csv")
|
||||
roster = pd.read_excel(roster_path) if roster_path else None
|
||||
votes = pd.read_excel(votes_file)
|
||||
|
||||
# Build player features
|
||||
pfb = PlayerFeatureBuilder()
|
||||
vote_avgs = votes.groupby("player").agg(
|
||||
vote_avg=("vote", "mean"),
|
||||
vote_std=("vote", "std"),
|
||||
).reset_index()
|
||||
|
||||
if roster is not None:
|
||||
players = pfb.build_player_dataset(outfield, keepers, roster, vote_avgs)
|
||||
else:
|
||||
players = pd.concat([outfield, keepers], ignore_index=True)
|
||||
players["vote_avg"] = 6.0
|
||||
players["vote_std"] = 0.5
|
||||
|
||||
# Build match features
|
||||
mfb = MatchFeatureBuilder()
|
||||
team_data = pd.concat([outfield, keepers], ignore_index=True)
|
||||
|
||||
dataset = mfb.build_match_dataset(votes, players, team_data)
|
||||
|
||||
output_path = self.data_dir / "match_dataset.xlsx"
|
||||
dataset.to_excel(str(output_path), index=False)
|
||||
logger.info(f"Saved features to {output_path}")
|
||||
return dataset
|
||||
|
||||
def train_models(self, X: pd.DataFrame, y: pd.Series):
|
||||
"""Train all ML models."""
|
||||
from src.models.train import ModelTrainer
|
||||
|
||||
trainer = ModelTrainer(model_dir=str(self.model_dir))
|
||||
model = trainer.tune_gbm(X, y)
|
||||
model.save("fantavote_model.pkl")
|
||||
return model
|
||||
|
||||
def optimize_lineup(self, predictions_path: str, squad_path: str):
|
||||
"""Run lineup optimization."""
|
||||
import pandas as pd
|
||||
from src.optimization.lineup_solver import LineupSolver, PlayerScore
|
||||
|
||||
preds = pd.read_excel(predictions_path)
|
||||
squad = pd.read_excel(squad_path)
|
||||
|
||||
pool = []
|
||||
for _, row in squad.iterrows():
|
||||
p_row = preds[preds["name"] == row["player"]]
|
||||
if p_row.empty:
|
||||
continue
|
||||
p = p_row.iloc[0]
|
||||
pool.append(PlayerScore(
|
||||
name=str(row["player"]),
|
||||
role=str(row.get("role", "C")),
|
||||
team=str(row.get("team", "")),
|
||||
oppteam=str(p.get("oppteam", "")),
|
||||
home=bool(p.get("home", 0)),
|
||||
fv_mean=float(p.get("fv_mean", 6.0)),
|
||||
fv_std=float(p.get("fv_std", 1.0)),
|
||||
mv_mean=float(p.get("mv_mean", 6.0)),
|
||||
mv_std=float(p.get("mv_std", 0.5)),
|
||||
starter_prob=float(p.get("starter_prob", 0.9)),
|
||||
cs_prob=float(p.get("cs_prob", 0.0)),
|
||||
))
|
||||
|
||||
solver = LineupSolver()
|
||||
result = solver.optimize(pool)
|
||||
|
||||
logger.info(f"Optimized lineup: {result}")
|
||||
return result
|
||||
|
||||
def run_full_pipeline(
|
||||
self, seasons: Optional[list] = None,
|
||||
vote_dir: Optional[str] = None,
|
||||
roster_path: Optional[str] = None,
|
||||
):
|
||||
"""Run the full pipeline end-to-end."""
|
||||
seasons = seasons or ["2024-2025", "2025-2026"]
|
||||
|
||||
logger.info("=" * 50)
|
||||
logger.info("Fantabeto 26/27 Pipeline — Starting")
|
||||
logger.info("=" * 50)
|
||||
|
||||
# Step 1: Scrape
|
||||
logger.info("Step 1: Scraping FBref data...")
|
||||
self.scrape_fbref(seasons, current=True)
|
||||
|
||||
# Step 2: Process votes
|
||||
if vote_dir:
|
||||
logger.info("Step 2: Processing votes...")
|
||||
self.process_votes(vote_dir, str(Path(vote_dir) / "calendar.xlsx"))
|
||||
|
||||
# Step 3: Build features
|
||||
logger.info("Step 3: Building features...")
|
||||
dataset = self.build_features(roster_path=roster_path)
|
||||
|
||||
# Step 4: Train models
|
||||
if "fantavote" in dataset.columns:
|
||||
logger.info("Step 4: Training models...")
|
||||
y = dataset["fantavote"]
|
||||
X = dataset.drop(columns=["fantavote", "vote", "matchday", "player", "team", "oppteam"], errors="ignore")
|
||||
self.train_models(X, y)
|
||||
|
||||
logger.info("Pipeline complete!")
|
||||
@@ -0,0 +1 @@
|
||||
"""Scraping modules for FBref, Fantacalcio.it, and api-football."""
|
||||
@@ -0,0 +1,118 @@
|
||||
"""api-football integration via RapidAPI.
|
||||
|
||||
Supplements FBref data with real-time injury info, expected goals (xG),
|
||||
expected assists (xA), and fixture data.
|
||||
|
||||
Requires RAPIDAPI_KEY environment variable.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
API_BASE = "https://api-football-v1.p.rapidapi.com/v3"
|
||||
LEAGUE_ID = 135 # Serie A
|
||||
|
||||
|
||||
class APIFootballClient:
|
||||
def __init__(self, api_key: Optional[str] = None):
|
||||
self.api_key = api_key or os.getenv("RAPIDAPI_KEY")
|
||||
if not self.api_key:
|
||||
logger.warning("No RAPIDAPI_KEY found. API calls will fail.")
|
||||
self.session = requests.Session()
|
||||
self.session.headers.update({
|
||||
"x-rapidapi-key": self.api_key or "",
|
||||
"x-rapidapi-host": "api-football-v1.p.rapidapi.com",
|
||||
})
|
||||
|
||||
def _get(self, endpoint: str, params: Optional[dict] = None) -> dict:
|
||||
if not self.api_key:
|
||||
raise ValueError("RAPIDAPI_KEY not configured")
|
||||
url = f"{API_BASE}/{endpoint}"
|
||||
resp = self.session.get(url, params=params, timeout=15)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
if data.get("errors"):
|
||||
logger.error(f"API error: {data['errors']}")
|
||||
return data
|
||||
|
||||
def get_fixtures(self, season: int, matchday: Optional[int] = None) -> pd.DataFrame:
|
||||
"""Get Serie A fixtures for a season. Optionally filter by matchday."""
|
||||
params = {"league": LEAGUE_ID, "season": season}
|
||||
if matchday:
|
||||
params["round"] = f"Regular Season - {matchday}"
|
||||
|
||||
data = self._get("fixtures", params)
|
||||
fixtures = data.get("response", [])
|
||||
rows = []
|
||||
for fix in fixtures:
|
||||
f = fix["fixture"]
|
||||
teams = fix["teams"]
|
||||
rows.append({
|
||||
"fixture_id": f["id"],
|
||||
"date": f["date"],
|
||||
"matchday": f.get("round", "").replace("Regular Season - ", ""),
|
||||
"home_team": teams["home"]["name"],
|
||||
"away_team": teams["away"]["name"],
|
||||
"home_goals": fix.get("goals", {}).get("home"),
|
||||
"away_goals": fix.get("goals", {}).get("away"),
|
||||
})
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
def get_team_statistics(self, season: int, team_id: int) -> dict:
|
||||
"""Get team-level statistics including xG, formations, etc."""
|
||||
data = self._get("teams/statistics", {
|
||||
"league": LEAGUE_ID, "season": season, "team": team_id,
|
||||
})
|
||||
return data.get("response", {})
|
||||
|
||||
def get_player_statistics(self, season: int, team_id: int, page: int = 1) -> pd.DataFrame:
|
||||
"""Get player statistics for a team in a given season."""
|
||||
data = self._get("players", {
|
||||
"league": LEAGUE_ID, "season": season, "team": team_id, "page": page,
|
||||
})
|
||||
players = data.get("response", [])
|
||||
rows = []
|
||||
for p in players:
|
||||
player = p["player"]
|
||||
stats = p["statistics"][0] if p.get("statistics") else {}
|
||||
rows.append({
|
||||
"player_id": player["id"],
|
||||
"player_name": player["name"],
|
||||
"position": stats.get("games", {}).get("position", ""),
|
||||
"appearences": stats.get("games", {}).get("appearences", 0),
|
||||
"minutes": stats.get("games", {}).get("minutes", 0),
|
||||
"goals": stats.get("goals", {}).get("total", 0),
|
||||
"assists": stats.get("goals", {}).get("assists", 0),
|
||||
"yellow_cards": stats.get("cards", {}).get("yellow", 0),
|
||||
"red_cards": stats.get("cards", {}).get("red", 0),
|
||||
})
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
def get_injuries(self, season: int, team_id: Optional[int] = None) -> pd.DataFrame:
|
||||
"""Get current injury list."""
|
||||
params = {"league": LEAGUE_ID, "season": season}
|
||||
if team_id:
|
||||
params["team"] = team_id
|
||||
|
||||
data = self._get("injuries", params)
|
||||
injuries = data.get("response", [])
|
||||
rows = []
|
||||
for inj in injuries:
|
||||
player = inj["player"]
|
||||
row = {
|
||||
"player_id": player["id"],
|
||||
"player_name": player["name"],
|
||||
"team": inj["team"]["name"],
|
||||
"injury_type": inj.get("player", {}).get("type", ""),
|
||||
"reason": inj.get("player", {}).get("reason", ""),
|
||||
}
|
||||
rows.append(row)
|
||||
return pd.DataFrame(rows)
|
||||
@@ -0,0 +1,110 @@
|
||||
"""Browser-based scraping fallback using Playwright.
|
||||
|
||||
Activates when standard requests are blocked by Cloudflare or similar
|
||||
anti-bot protections. Uses Playwright stealth mode to mimic a real browser.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BrowserFallback:
|
||||
"""Playwright-based scraper for Cloudflare-protected pages."""
|
||||
|
||||
def __init__(self, headless: bool = True, timeout: int = 30000):
|
||||
self.headless = headless
|
||||
self.timeout = timeout
|
||||
self._browser = None
|
||||
self._context = None
|
||||
self._initialized = False
|
||||
|
||||
def _ensure_browser(self):
|
||||
if self._initialized:
|
||||
return
|
||||
try:
|
||||
from playwright.sync_api import sync_playwright
|
||||
except ImportError:
|
||||
raise ImportError(
|
||||
"Playwright is required for browser fallback. "
|
||||
"Install with: pip install playwright && playwright install chromium"
|
||||
)
|
||||
|
||||
self._pw = sync_playwright().start()
|
||||
self._browser = self._pw.chromium.launch(headless=self.headless)
|
||||
self._context = self._browser.new_context(
|
||||
user_agent=(
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
),
|
||||
viewport={"width": 1920, "height": 1080},
|
||||
locale="en-US",
|
||||
)
|
||||
self._initialized = True
|
||||
logger.info("Playwright browser initialized")
|
||||
|
||||
def fetch(self, url: str, wait_selector: Optional[str] = None, wait_time: float = 3.0) -> str:
|
||||
"""Fetch a page using Playwright and return HTML content.
|
||||
|
||||
Args:
|
||||
url: The URL to fetch.
|
||||
wait_selector: CSS selector to wait for before extracting content.
|
||||
wait_time: Extra seconds to wait for dynamic content to load.
|
||||
|
||||
Returns:
|
||||
HTML content as string.
|
||||
"""
|
||||
self._ensure_browser()
|
||||
|
||||
page = self._context.new_page()
|
||||
try:
|
||||
logger.debug(f"Browser fetching {url}")
|
||||
page.goto(url, timeout=self.timeout, wait_until="domcontentloaded")
|
||||
|
||||
if wait_selector:
|
||||
page.wait_for_selector(wait_selector, timeout=self.timeout)
|
||||
if wait_time:
|
||||
time.sleep(wait_time)
|
||||
|
||||
content = page.content()
|
||||
logger.debug(f"Got {len(content)} bytes from {url}")
|
||||
return content
|
||||
except Exception as e:
|
||||
logger.error(f"Browser fetch failed for {url}: {e}")
|
||||
raise
|
||||
finally:
|
||||
page.close()
|
||||
|
||||
def close(self):
|
||||
if self._context:
|
||||
self._context.close()
|
||||
if self._browser:
|
||||
self._browser.close()
|
||||
if hasattr(self, "_pw"):
|
||||
self._pw.stop()
|
||||
self._initialized = False
|
||||
logger.info("Playwright browser closed")
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *args):
|
||||
self.close()
|
||||
|
||||
|
||||
def is_cloudflare_blocked(response_text: str) -> bool:
|
||||
"""Check if a response indicates Cloudflare blocking."""
|
||||
text_lower = response_text.lower()
|
||||
return any(
|
||||
marker in text_lower
|
||||
for marker in [
|
||||
"cf-browser-verification",
|
||||
"checking your browser",
|
||||
"cloudflare",
|
||||
"attention required",
|
||||
"just a moment",
|
||||
"enable javascript",
|
||||
]
|
||||
)
|
||||
@@ -0,0 +1,242 @@
|
||||
"""Fantacalcio.it data scraper.
|
||||
|
||||
Handles:
|
||||
- Votes (match ratings) via authenticated Excel API or HTML fallback
|
||||
- Probable lineups via HTML scraping
|
||||
- Player roster/quotazioni via authenticated API or HTML fallback
|
||||
- Calendar data
|
||||
|
||||
Supports season IDs: 21 = 2026/27, 20 = 2025/26, 19 = 2024/25, etc.
|
||||
"""
|
||||
|
||||
import io
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
FANTACALCIO_BASE = "https://www.fantacalcio.it"
|
||||
API_VOTES = f"{FANTACALCIO_BASE}/api/v1/Excel/votes"
|
||||
API_STATS = f"{FANTACALCIO_BASE}/api/v1/Excel/stats"
|
||||
API_CALENDAR = f"{FANTACALCIO_BASE}/api/v1/Excel/calendar"
|
||||
PROBABILI_URL = f"{FANTACALCIO_BASE}/probabili-formazioni-serie-a"
|
||||
|
||||
HEADERS = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
),
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "it-IT,it;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"Referer": f"{FANTACALCIO_BASE}/",
|
||||
}
|
||||
|
||||
|
||||
class FantacalcioScraper:
|
||||
"""Scraper for Fantacalcio.it data with API + HTML fallback."""
|
||||
|
||||
def __init__(self, auth_token: Optional[str] = None):
|
||||
self.auth_token = auth_token or os.getenv("FANTACALCIO_TOKEN")
|
||||
self.session = requests.Session()
|
||||
self.session.headers.update(HEADERS)
|
||||
if self.auth_token:
|
||||
self.session.headers["Authorization"] = f"Bearer {self.auth_token}"
|
||||
|
||||
# ─── Votes API ─────────────────────────────────────────────────
|
||||
|
||||
def fetch_votes_matchday(self, season_id: int, matchday: int) -> pd.DataFrame:
|
||||
"""Fetch votes for a specific matchday via the Excel API.
|
||||
|
||||
Requires authentication (FANTACALCIO_TOKEN env var).
|
||||
Falls back to HTML scraping if not authenticated.
|
||||
"""
|
||||
if self.auth_token:
|
||||
return self._fetch_votes_api(season_id, matchday)
|
||||
return self._fetch_votes_html(season_id, matchday)
|
||||
|
||||
def _fetch_votes_api(self, season_id: int, matchday: int) -> pd.DataFrame:
|
||||
url = f"{API_VOTES}/{season_id}/{matchday}"
|
||||
logger.info(f"Fetching votes from API: {url}")
|
||||
resp = self.session.get(url, timeout=30)
|
||||
if resp.status_code == 401:
|
||||
logger.warning("API returned 401 (unauthorized). Set FANTACALCIO_TOKEN for API access.")
|
||||
return pd.DataFrame()
|
||||
resp.raise_for_status()
|
||||
return pd.read_excel(io.BytesIO(resp.content))
|
||||
|
||||
def _fetch_votes_html(self, season_id: int, matchday: int) -> pd.DataFrame:
|
||||
"""Fallback: scrape votes from HTML page."""
|
||||
url = f"{FANTACALCIO_BASE}/voti-serie-a-giornata-{matchday}"
|
||||
logger.info(f"Fetching votes from HTML: {url}")
|
||||
resp = self.session.get(url, timeout=30, allow_redirects=True)
|
||||
if resp.status_code != 200:
|
||||
logger.warning(f"HTML votes page returned {resp.status_code}")
|
||||
return pd.DataFrame()
|
||||
|
||||
soup = BeautifulSoup(resp.text, "lxml")
|
||||
rows = self._parse_votes_html(soup)
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
def _parse_votes_html(self, soup) -> list:
|
||||
"""Parse votes from HTML table structure."""
|
||||
results = []
|
||||
for table in soup.select("table.voti-table"):
|
||||
team_name = ""
|
||||
for row in table.select("tr"):
|
||||
cols = row.select("td")
|
||||
if not cols:
|
||||
continue
|
||||
if len(cols) == 1 and cols[0].get("colspan"):
|
||||
team_name = cols[0].text.strip()
|
||||
continue
|
||||
if len(cols) >= 4:
|
||||
player_name = cols[0].text.strip()
|
||||
try:
|
||||
vote = float(cols[1].text.strip())
|
||||
except (ValueError, TypeError):
|
||||
vote = None
|
||||
gf = cols[2].text.strip() if len(cols) > 2 else "0"
|
||||
gs = cols[3].text.strip() if len(cols) > 3 else "0"
|
||||
results.append({
|
||||
"player": player_name,
|
||||
"team": team_name,
|
||||
"vote": vote,
|
||||
"goals_scored": gf,
|
||||
"goals_conceded": gs,
|
||||
})
|
||||
return results
|
||||
|
||||
# ─── Player Stats / Roster ─────────────────────────────────────
|
||||
|
||||
def fetch_player_roster(self, season_id: int) -> pd.DataFrame:
|
||||
"""Fetch the full player roster for a season via the stats API.
|
||||
|
||||
The endpoint /Excel/stats/{season}/{matchday=1} returns the full player
|
||||
list with FVM (Fantacalcio Market Value) quotazioni.
|
||||
"""
|
||||
url = f"{API_STATS}/{season_id}/1"
|
||||
logger.info(f"Fetching player roster from: {url}")
|
||||
resp = self.session.get(url, timeout=30)
|
||||
if resp.status_code == 401 or resp.status_code == 404:
|
||||
logger.warning(f"Player roster API returned {resp.status_code}. Trying HTML fallback.")
|
||||
return self._fetch_roster_html()
|
||||
resp.raise_for_status()
|
||||
return pd.read_excel(io.BytesIO(resp.content))
|
||||
|
||||
def _fetch_roster_html(self) -> pd.DataFrame:
|
||||
"""Fallback: scrape quotazioni/roster from HTML page."""
|
||||
url = f"{FANTACALCIO_BASE}/quotazioni-fantacalcio"
|
||||
logger.info(f"Fetching roster from HTML: {url}")
|
||||
resp = self.session.get(url, timeout=30, allow_redirects=True)
|
||||
soup = BeautifulSoup(resp.text, "lxml")
|
||||
rows = []
|
||||
for tr in soup.select("table tbody tr"):
|
||||
cols = tr.select("td")
|
||||
if len(cols) >= 5:
|
||||
rows.append({
|
||||
"player": cols[0].text.strip(),
|
||||
"role": cols[1].text.strip(),
|
||||
"team": cols[2].text.strip(),
|
||||
"value": cols[3].text.strip(),
|
||||
})
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
# ─── Probable Lineups ──────────────────────────────────────────
|
||||
|
||||
def fetch_probable_lineups(self) -> pd.DataFrame:
|
||||
"""Scrape probable starting lineups and player probabilities.
|
||||
|
||||
Returns DataFrame with columns: player, team, starter, percentage
|
||||
"""
|
||||
logger.info(f"Fetching probable lineups from: {PROBABILI_URL}")
|
||||
resp = self.session.get(PROBABILI_URL, timeout=30)
|
||||
resp.raise_for_status()
|
||||
soup = BeautifulSoup(resp.text, "lxml")
|
||||
|
||||
rows = []
|
||||
for match_section in soup.select(".match-row"):
|
||||
home_team = match_section.select_one(".home-team .team-name")
|
||||
away_team = match_section.select_one(".away-team .team-name")
|
||||
home = home_team.text.strip() if home_team else ""
|
||||
away = away_team.text.strip() if away_team else ""
|
||||
|
||||
for player_list in match_section.select("ul.player-list"):
|
||||
classes = player_list.get("class", [])
|
||||
is_starter = "starters" in classes
|
||||
|
||||
for player_link in player_list.select("a.player-name"):
|
||||
name = player_link.text.strip()
|
||||
prob_bar = player_list.select_one(".progress-bar")
|
||||
percentage = 0.0
|
||||
if prob_bar:
|
||||
try:
|
||||
percentage = float(prob_bar.get("aria-valuenow", 0))
|
||||
except (ValueError, TypeError):
|
||||
percentage = 0.0
|
||||
|
||||
# Determine team
|
||||
parent = player_link.parent
|
||||
team = home if "home" in str(parent.parent).lower() else away
|
||||
if not team:
|
||||
team = home if player_link.parent.get("class") and "home" in str(player_link.parent.get("class")) else away
|
||||
|
||||
rows.append({
|
||||
"player": name,
|
||||
"team": team,
|
||||
"starter": 1.0 if is_starter else percentage / 100.0,
|
||||
"percentage": percentage,
|
||||
})
|
||||
|
||||
if not rows:
|
||||
rows = self._parse_probables_fallback(soup)
|
||||
|
||||
return pd.DataFrame(rows).drop_duplicates(subset=["player"])
|
||||
|
||||
def _parse_probables_fallback(self, soup) -> list:
|
||||
"""More aggressive parsing for probable lineups."""
|
||||
rows = []
|
||||
for ul in soup.select("ul.player-list"):
|
||||
classes = ul.get("class", [])
|
||||
is_starter = "starters" in classes
|
||||
for a in ul.select("a.player-name"):
|
||||
name = a.text.strip()
|
||||
bar = ul.select_one(".progress-bar")
|
||||
perc = float(bar.get("aria-valuenow", 0)) if bar else 0.0
|
||||
rows.append({
|
||||
"player": name,
|
||||
"team": "",
|
||||
"starter": 1.0 if is_starter else perc / 100.0,
|
||||
"percentage": perc,
|
||||
})
|
||||
return rows
|
||||
|
||||
# ─── Calendar ──────────────────────────────────────────────────
|
||||
|
||||
def fetch_calendar(self) -> pd.DataFrame:
|
||||
"""Fetch the Serie A match calendar.
|
||||
|
||||
Falls back to scraping the schedule page if API unavailable.
|
||||
"""
|
||||
url = f"{FANTACALCIO_BASE}/calendario-serie-a"
|
||||
resp = self.session.get(url, timeout=30, allow_redirects=True)
|
||||
soup = BeautifulSoup(resp.text, "lxml")
|
||||
rows = []
|
||||
for match in soup.select(".match-row"):
|
||||
md_elem = match.select_one(".matchday")
|
||||
home_elem = match.select_one(".home-team .team-name")
|
||||
away_elem = match.select_one(".away-team .team-name")
|
||||
if home_elem and away_elem:
|
||||
rows.append({
|
||||
"matchday": md_elem.text.strip() if md_elem else "",
|
||||
"home": home_elem.text.strip(),
|
||||
"away": away_elem.text.strip(),
|
||||
})
|
||||
return pd.DataFrame(rows)
|
||||
@@ -0,0 +1,295 @@
|
||||
"""FBref.com Serie A data scraper with proxy rotation and browser fallback.
|
||||
|
||||
Refactored from notebook 1_scraping_fbref.ipynb.
|
||||
Scrapes player stats (outfield + goalkeeper) and team stats for a given season.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# Constants
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
FBREF_BASE = "https://fbref.com/en/comps/11/"
|
||||
HEADERS = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
}
|
||||
|
||||
REQUEST_DELAY = 4.0 # seconds between requests to be kind to FBref
|
||||
|
||||
STAT_CATEGORIES = [
|
||||
"stats", "shooting", "passing", "passing_types",
|
||||
"gca", "defense", "possession", "misc",
|
||||
]
|
||||
|
||||
KEEPER_CATEGORIES = ["keepers", "keepersadv"]
|
||||
|
||||
# Columns that should be stripped from player tables (not needed)
|
||||
INFO_COLS = [
|
||||
"player", "nationality", "position", "team",
|
||||
"age", "birth_year", "games", "minutes", "birth_year",
|
||||
]
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# Core scraping functions (adapted from parth1902's scraper)
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _clean_html(html: str) -> str:
|
||||
"""Remove HTML comments that can break BeautifulSoup table parsing."""
|
||||
return re.sub(r"<!--|-->", "", html)
|
||||
|
||||
|
||||
def _get_tables(html: str):
|
||||
"""Parse HTML and return player + team tables from the stats page."""
|
||||
soup = BeautifulSoup(_clean_html(html), "lxml")
|
||||
all_tables = soup.findAll("tbody")
|
||||
|
||||
if len(all_tables) < 3:
|
||||
raise ValueError(f"Expected at least 3 tables, found {len(all_tables)}")
|
||||
|
||||
player_table = all_tables[2]
|
||||
team_table_for = all_tables[0]
|
||||
team_table_vs = all_tables[1]
|
||||
return player_table, team_table_for, team_table_vs
|
||||
|
||||
|
||||
def _get_frame(features: list, table) -> pd.DataFrame:
|
||||
"""Extract player-level data from a <tbody> element into a DataFrame."""
|
||||
rows = table.find_all("tr")
|
||||
if not rows:
|
||||
return pd.DataFrame()
|
||||
|
||||
pre_df = {col: [] for col in ["player", "nationality", "position", "team", "age", "birth_year"] + features}
|
||||
info_keys = {"player", "nationality", "position", "team", "age", "birth_year"}
|
||||
|
||||
for row in rows:
|
||||
cells = row.find_all("td")
|
||||
if not cells:
|
||||
continue
|
||||
for col, cell in zip(pre_df.keys(), cells):
|
||||
text = cell.text.strip()
|
||||
key = cell.get("data-stat", "")
|
||||
if key in info_keys or col in info_keys:
|
||||
pre_df[col].append(text)
|
||||
elif key in features:
|
||||
try:
|
||||
pre_df[col].append(float(text.replace(",", "")))
|
||||
except (ValueError, TypeError):
|
||||
pre_df[col].append(0.0)
|
||||
|
||||
df = pd.DataFrame(pre_df)
|
||||
# Ensure consistent columns
|
||||
for col in ["player", "nationality", "position", "team", "age", "birth_year"]:
|
||||
if col not in df.columns:
|
||||
df[col] = ""
|
||||
for f in features:
|
||||
if f not in df.columns:
|
||||
df[f] = 0.0
|
||||
return df
|
||||
|
||||
|
||||
def _get_frame_team(features: list, table, text: str = "for") -> pd.DataFrame:
|
||||
"""Extract team-level data from the stats table."""
|
||||
rows = table.find_all("tr")
|
||||
if not rows:
|
||||
return pd.DataFrame()
|
||||
|
||||
# Determine prefix for columns
|
||||
prefix = "vs " if text == "vs" else ""
|
||||
|
||||
pre_df_team = {"team": []}
|
||||
for f in features:
|
||||
pre_df_team[f"{prefix}{f}"] = []
|
||||
|
||||
team_rows = table.find_all("tr")
|
||||
for row in team_rows:
|
||||
cells = row.find_all("td")
|
||||
if not cells:
|
||||
continue
|
||||
|
||||
team_name = row.find("th", {"data-stat": "team"}).text.strip()
|
||||
if team_name == "":
|
||||
continue
|
||||
|
||||
pre_df_team["team"].append(team_name)
|
||||
for cell in cells:
|
||||
key = cell.get("data-stat", "")
|
||||
cell_text = cell.text.strip()
|
||||
if key in features:
|
||||
try:
|
||||
val = float(cell_text.replace(",", ""))
|
||||
except (ValueError, TypeError):
|
||||
val = 0.0
|
||||
pre_df_team[f"{prefix}{key}"].append(val)
|
||||
|
||||
df = pd.DataFrame(pre_df_team)
|
||||
if "team" not in df.columns:
|
||||
return pd.DataFrame()
|
||||
return df
|
||||
|
||||
|
||||
def _frame_for_category(category: str, base_url: str, suffix: str) -> tuple:
|
||||
"""Scrape a single stat category page and return (player_df, team_for_df, team_vs_df)."""
|
||||
url = f"{base_url}{category}{suffix}"
|
||||
logger.debug(f"Fetching {url}")
|
||||
time.sleep(REQUEST_DELAY)
|
||||
|
||||
resp = requests.get(url, headers=HEADERS, timeout=30)
|
||||
resp.raise_for_status()
|
||||
|
||||
player_table, team_for, team_vs = _get_tables(resp.text)
|
||||
|
||||
features = []
|
||||
for row in player_table.find_all("tr"):
|
||||
for cell in row.find_all("td"):
|
||||
stat = cell.get("data-stat", "")
|
||||
if stat and stat not in features and stat not in INFO_COLS:
|
||||
features.append(stat)
|
||||
if features:
|
||||
break
|
||||
|
||||
player_df = _get_frame(features, player_table)
|
||||
team_for_df = _get_frame_team(features, team_for, "for")
|
||||
team_vs_df = _get_frame_team(features, team_vs, "vs")
|
||||
return player_df, team_for_df, team_vs_df
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
# Public API
|
||||
# ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
def scrape_outfield_players(base_url: str, suffix: str) -> pd.DataFrame:
|
||||
"""Scrape outfield player stats across all categories."""
|
||||
players = None
|
||||
for cat in STAT_CATEGORIES:
|
||||
pdf, _, _ = _frame_for_category(cat, base_url, suffix)
|
||||
if players is None:
|
||||
players = pdf
|
||||
else:
|
||||
players = pd.concat([players, pdf.drop(columns=["player", "nationality", "position", "team", "age", "birth_year"], errors="ignore")], axis=1)
|
||||
if players is not None:
|
||||
players = players.loc[:, ~players.columns.duplicated()]
|
||||
return players
|
||||
|
||||
|
||||
def scrape_keeper_players(base_url: str, suffix: str) -> pd.DataFrame:
|
||||
"""Scrape goalkeeper stats."""
|
||||
keepers = None
|
||||
for cat in KEEPER_CATEGORIES:
|
||||
pdf, _, _ = _frame_for_category(cat, base_url, suffix)
|
||||
if keepers is None:
|
||||
keepers = pdf
|
||||
else:
|
||||
keepers = pd.concat([keepers, pdf.drop(columns=["player", "nationality", "position", "team", "age", "birth_year"], errors="ignore")], axis=1)
|
||||
if keepers is not None:
|
||||
keepers = keepers.loc[:, ~keepers.columns.duplicated()]
|
||||
keepers = keepers[keepers["position"] == "GK"]
|
||||
return keepers
|
||||
|
||||
|
||||
def scrape_team_stats(base_url: str, suffix: str) -> tuple:
|
||||
"""Scrape team 'for' and 'vs' stats across all categories."""
|
||||
all_cats = STAT_CATEGORIES + KEEPER_CATEGORIES
|
||||
teams_for = None
|
||||
teams_vs = None
|
||||
|
||||
for cat in all_cats:
|
||||
_, tf, tv = _frame_for_category(cat, base_url, suffix)
|
||||
if teams_for is None:
|
||||
teams_for = tf
|
||||
teams_vs = tv
|
||||
else:
|
||||
if tf is not None and "team" in tf.columns:
|
||||
teams_for = pd.merge(teams_for, tf, on="team", how="outer") if "team" in teams_for.columns else tf
|
||||
if tv is not None and "team" in tv.columns:
|
||||
teams_vs = pd.merge(teams_vs, tv, on="team", how="outer") if "team" in teams_vs.columns else tv
|
||||
|
||||
if teams_for is not None:
|
||||
teams_for = teams_for.loc[:, ~teams_for.columns.duplicated()]
|
||||
if teams_vs is not None:
|
||||
teams_vs = teams_vs.loc[:, ~teams_vs.columns.duplicated()]
|
||||
return teams_for, teams_vs
|
||||
|
||||
|
||||
def scrape_season(season_str: str, output_dir: str = "data/fbref") -> dict:
|
||||
"""Scrape a full Serie A season from FBref and save CSV files.
|
||||
|
||||
Args:
|
||||
season_str: e.g. "2026-2027" for the 26/27 season.
|
||||
output_dir: Base directory for output. Files saved to {output_dir}/season{YY}/.
|
||||
|
||||
Returns:
|
||||
dict with keys: outfield_players, keepers_players, teams, teams_vs, output_path
|
||||
"""
|
||||
short = season_str[2:4] + season_str[7:9] # "2627"
|
||||
season_dir = Path(output_dir) / f"season{short}"
|
||||
season_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
base_url = f"{FBREF_BASE}{season_str}/"
|
||||
suffix = f"/{season_str}-Serie-A-Stats"
|
||||
|
||||
logger.info(f"Scraping {season_str} from FBref...")
|
||||
logger.info(f"Base URL: {base_url}")
|
||||
logger.info(f"Suffix: {suffix}")
|
||||
|
||||
outfield = scrape_outfield_players(base_url, suffix)
|
||||
keepers = scrape_keeper_players(base_url, suffix)
|
||||
teams_for, teams_vs = scrape_team_stats(base_url, suffix)
|
||||
|
||||
outfield.to_csv(season_dir / "outfield_players.csv", index=False)
|
||||
keepers.to_csv(season_dir / "keepers_players.csv", index=False)
|
||||
teams_for.to_csv(season_dir / "teams.csv", index=False)
|
||||
teams_vs.to_csv(season_dir / "teams_vs.csv", index=False)
|
||||
|
||||
logger.info(f"Saved to {season_dir}/ (outfield={outfield.shape}, keepers={keepers.shape})")
|
||||
|
||||
return {
|
||||
"outfield_players": outfield,
|
||||
"keepers_players": keepers,
|
||||
"teams": teams_for,
|
||||
"teams_vs": teams_vs,
|
||||
"output_path": str(season_dir),
|
||||
}
|
||||
|
||||
|
||||
def scrape_current_season(output_dir: str = "data/fbref") -> dict:
|
||||
"""Scrape the current (live) Serie A season from FBref."""
|
||||
logger.info("Scraping current season from FBref...")
|
||||
base_url = FBREF_BASE
|
||||
suffix = "/Serie-A-Stats"
|
||||
|
||||
outfield = scrape_outfield_players(base_url, suffix)
|
||||
keepers = scrape_keeper_players(base_url, suffix)
|
||||
teams_for, teams_vs = scrape_team_stats(base_url, suffix)
|
||||
|
||||
current_dir = Path(output_dir) / "current"
|
||||
current_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
outfield.to_csv(current_dir / "outfield_players.csv", index=False)
|
||||
keepers.to_csv(current_dir / "keepers_players.csv", index=False)
|
||||
teams_for.to_csv(current_dir / "teams.csv", index=False)
|
||||
teams_vs.to_csv(current_dir / "teams_vs.csv", index=False)
|
||||
|
||||
logger.info(f"Saved current season to {current_dir}/")
|
||||
return {
|
||||
"outfield_players": outfield,
|
||||
"keepers_players": keepers,
|
||||
"teams": teams_for,
|
||||
"teams_vs": teams_vs,
|
||||
"output_path": str(current_dir),
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
"""Rotating proxy management for web scraping.
|
||||
|
||||
Manages a pool of HTTP/HTTPS proxies with health checks and
|
||||
automatic rotation to avoid rate-limiting and IP bans.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import random
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
import requests
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Proxy:
|
||||
url: str
|
||||
failures: int = 0
|
||||
last_used: float = 0.0
|
||||
cooldown_until: float = 0.0
|
||||
max_failures: int = 3
|
||||
base_cooldown: float = 60.0
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
return time.time() > self.cooldown_until and self.failures < self.max_failures
|
||||
|
||||
def mark_success(self):
|
||||
self.failures = max(0, self.failures - 1)
|
||||
self.cooldown_until = 0.0
|
||||
self.last_used = time.time()
|
||||
|
||||
def mark_failure(self):
|
||||
self.failures += 1
|
||||
cooldown = self.base_cooldown * (2 ** (self.failures - 1))
|
||||
self.cooldown_until = time.time() + cooldown
|
||||
self.last_used = time.time()
|
||||
|
||||
|
||||
@dataclass
|
||||
class ProxyManager:
|
||||
proxies: list[Proxy] = field(default_factory=list)
|
||||
test_url: str = "https://httpbin.org/ip"
|
||||
test_timeout: int = 10
|
||||
min_rotation_interval: float = 5.0
|
||||
_last_rotation: float = 0.0
|
||||
|
||||
def add_proxy(self, proxy_url: str):
|
||||
self.proxies.append(Proxy(url=proxy_url))
|
||||
logger.debug(f"Added proxy: {proxy_url}")
|
||||
|
||||
def add_proxies_from_file(self, filepath: str):
|
||||
with open(filepath) as f:
|
||||
for line in f:
|
||||
url = line.strip()
|
||||
if url and not url.startswith("#"):
|
||||
self.add_proxy(url)
|
||||
logger.info(f"Loaded {len(self.proxies)} proxies from {filepath}")
|
||||
|
||||
def add_proxies_from_env(self, env_var: str = "PROXY_LIST"):
|
||||
import os
|
||||
val = os.getenv(env_var, "")
|
||||
if val:
|
||||
for url in val.split(","):
|
||||
url = url.strip()
|
||||
if url:
|
||||
self.add_proxy(url)
|
||||
|
||||
def get_proxy(self) -> Optional[dict]:
|
||||
available = [p for p in self.proxies if p.available]
|
||||
if not available:
|
||||
logger.warning("No proxies available")
|
||||
return None
|
||||
|
||||
now = time.time()
|
||||
if now - self._last_rotation < self.min_rotation_interval and len(available) > 1:
|
||||
# Filter out the most recently used proxy if possible
|
||||
most_recent = max(available, key=lambda p: p.last_used)
|
||||
available = [p for p in available if p != most_recent] or available
|
||||
|
||||
proxy = random.choice(available)
|
||||
self._last_rotation = now
|
||||
return {"http": proxy.url, "https": proxy.url}
|
||||
|
||||
def report_success(self, proxy_url: str):
|
||||
for p in self.proxies:
|
||||
if p.url == proxy_url:
|
||||
p.mark_success()
|
||||
return
|
||||
|
||||
def report_failure(self, proxy_url: str):
|
||||
for p in self.proxies:
|
||||
if p.url == proxy_url:
|
||||
p.mark_failure()
|
||||
logger.warning(f"Proxy {p.url} failed ({p.failures}/{p.max_failures}), cooldown until {p.cooldown_until}")
|
||||
return
|
||||
|
||||
def health_check(self):
|
||||
for p in self.proxies:
|
||||
try:
|
||||
proxies = {"http": p.url, "https": p.url}
|
||||
r = requests.get(self.test_url, proxies=proxies, timeout=self.test_timeout)
|
||||
if r.ok:
|
||||
p.mark_success()
|
||||
else:
|
||||
p.mark_failure()
|
||||
except Exception:
|
||||
p.mark_failure()
|
||||
available = [p for p in self.proxies if p.available]
|
||||
total = len(self.proxies)
|
||||
logger.info(f"Proxy health: {len(available)}/{total} available")
|
||||
|
||||
@property
|
||||
def has_proxies(self) -> bool:
|
||||
return any(p.available for p in self.proxies)
|
||||
|
||||
@property
|
||||
def pool_size(self) -> int:
|
||||
return len(self.proxies)
|
||||
Reference in New Issue
Block a user