3b065775f5
Phase 1: Data Engineering
- Refactored notebooks into src/{scraper,features,models,optimization,bot}
- FBref scraper with proxy rotation + Playwright Cloudflare bypass
- Fantacalcio.it integrated scraper (authenticated API + HTML fallback)
- api-football RapidAPI client for supplementary xG/xA/injuries
- RAG news pipeline: Gazzetta, Sky Sport, Di Marzio → injury/suspension/tactical extraction
- 26/27 season config: teams, scoring rules, name mappings, news sources
Phase 2: SOTA ML Architecture
- GBM Ensemble (LightGBM + CatBoost + XGBoost) with stacked blending
- Bootstrap ensemble for uncertainty quantification
- SinhArcsinh distribution head (ported from original TF Probability)
- Card classifiers (yellow/red), penalty model, goal probability (Poisson)
- Temporal GNN for player interaction modeling (crosses→goals, passes→assists)
- Optuna hyperparameter tuning with time-series CV
Phase 3: Operations Research
- Auction solver: MILP knapsack with PuLP (budget + role constraints)
- Grid Auction (Asta a Griglia): Minimax game theory bidding strategy
- Weekly lineup optimizer: MCTS maximizing win probability vs opponent
- Modificatore Difesa integration + captain selection
- Transfer market analyzer: buy-low/sell-high via xG regression to mean
- Opponent behavior modeling from historical lineage patterns
Phase 4: Agentic Workflow
- Telegram bot: auto-briefing (Friday + Sunday morning)
- Tactical briefing generator with start/sit recommendations
- GitHub Actions CI/CD: scheduled pipeline (scrape → predict → notify)
Infrastructure:
- 31 pytest unit tests (features, models, scraper, optimization)
- requirements.txt (lightgbm, catboost, xgboost, optuna, pulp, playwright, langchain)
- Makefile with install/test/lint/scrape/train/bot targets
- Jupyter notebook: 26_27_strategy.ipynb demonstrating auction + matchday 1 mockup
- Completely rewritten README.md with architecture diagram
178 lines
6.4 KiB
Python
178 lines
6.4 KiB
Python
"""Fantabeto 2026/27 pipeline orchestrator.
|
|
|
|
Central entry point for the entire data → features → models → optimization
|
|
workflow. Supports incremental updates and targeted matchday processing.
|
|
"""
|
|
|
|
import logging
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class Pipeline:
|
|
"""Orchestrates the full Fantabeto pipeline."""
|
|
|
|
def __init__(self, data_dir: str = "data", model_dir: str = "models_trained"):
|
|
self.data_dir = Path(data_dir)
|
|
self.model_dir = Path(model_dir)
|
|
self.data_dir.mkdir(parents=True, exist_ok=True)
|
|
self.model_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
def scrape_fbref(
|
|
self, seasons: list, current: bool = True
|
|
) -> dict:
|
|
"""Scrape FBref data for specified seasons."""
|
|
from src.scraper.fbref_scraper import scrape_season, scrape_current_season
|
|
|
|
results = {}
|
|
for season in seasons:
|
|
results[season] = scrape_season(season, str(self.data_dir / "fbref"))
|
|
if current:
|
|
results["current"] = scrape_current_season(str(self.data_dir / "fbref"))
|
|
return results
|
|
|
|
def process_votes(
|
|
self, vote_dir: str, calendar_path: str
|
|
):
|
|
"""Process vote files into unified database."""
|
|
import pandas as pd
|
|
from src.features.vote_processor import VoteProcessor
|
|
|
|
processor = VoteProcessor()
|
|
calendar = pd.read_excel(calendar_path) if calendar_path.endswith(".xlsx") else pd.read_csv(calendar_path)
|
|
|
|
all_votes = []
|
|
for matchday in range(1, 39):
|
|
df = processor.process_matchday(vote_dir, matchday, calendar)
|
|
if not df.empty:
|
|
all_votes.append(df)
|
|
logger.info(f"Matchday {matchday}: {len(df)} votes")
|
|
|
|
combined = pd.concat(all_votes, ignore_index=True) if all_votes else pd.DataFrame()
|
|
output_path = self.data_dir / "players_votes.xlsx"
|
|
combined.to_excel(str(output_path), index=False)
|
|
logger.info(f"Saved {len(combined)} votes to {output_path}")
|
|
return combined
|
|
|
|
def build_features(
|
|
self, fbref_dir: Optional[str] = None, votes_path: Optional[str] = None,
|
|
roster_path: Optional[str] = None,
|
|
):
|
|
"""Build the full feature dataset."""
|
|
import pandas as pd
|
|
from src.features.player_features import PlayerFeatureBuilder
|
|
from src.features.match_features import MatchFeatureBuilder
|
|
|
|
fbref_path = Path(fbref_dir or (self.data_dir / "fbref"))
|
|
votes_file = votes_path or (self.data_dir / "players_votes.xlsx")
|
|
|
|
# Load data
|
|
outfield = pd.read_csv(fbref_path / "outfield_players.csv")
|
|
keepers = pd.read_csv(fbref_path / "keepers_players.csv")
|
|
roster = pd.read_excel(roster_path) if roster_path else None
|
|
votes = pd.read_excel(votes_file)
|
|
|
|
# Build player features
|
|
pfb = PlayerFeatureBuilder()
|
|
vote_avgs = votes.groupby("player").agg(
|
|
vote_avg=("vote", "mean"),
|
|
vote_std=("vote", "std"),
|
|
).reset_index()
|
|
|
|
if roster is not None:
|
|
players = pfb.build_player_dataset(outfield, keepers, roster, vote_avgs)
|
|
else:
|
|
players = pd.concat([outfield, keepers], ignore_index=True)
|
|
players["vote_avg"] = 6.0
|
|
players["vote_std"] = 0.5
|
|
|
|
# Build match features
|
|
mfb = MatchFeatureBuilder()
|
|
team_data = pd.concat([outfield, keepers], ignore_index=True)
|
|
|
|
dataset = mfb.build_match_dataset(votes, players, team_data)
|
|
|
|
output_path = self.data_dir / "match_dataset.xlsx"
|
|
dataset.to_excel(str(output_path), index=False)
|
|
logger.info(f"Saved features to {output_path}")
|
|
return dataset
|
|
|
|
def train_models(self, X: pd.DataFrame, y: pd.Series):
|
|
"""Train all ML models."""
|
|
from src.models.train import ModelTrainer
|
|
|
|
trainer = ModelTrainer(model_dir=str(self.model_dir))
|
|
model = trainer.tune_gbm(X, y)
|
|
model.save("fantavote_model.pkl")
|
|
return model
|
|
|
|
def optimize_lineup(self, predictions_path: str, squad_path: str):
|
|
"""Run lineup optimization."""
|
|
import pandas as pd
|
|
from src.optimization.lineup_solver import LineupSolver, PlayerScore
|
|
|
|
preds = pd.read_excel(predictions_path)
|
|
squad = pd.read_excel(squad_path)
|
|
|
|
pool = []
|
|
for _, row in squad.iterrows():
|
|
p_row = preds[preds["name"] == row["player"]]
|
|
if p_row.empty:
|
|
continue
|
|
p = p_row.iloc[0]
|
|
pool.append(PlayerScore(
|
|
name=str(row["player"]),
|
|
role=str(row.get("role", "C")),
|
|
team=str(row.get("team", "")),
|
|
oppteam=str(p.get("oppteam", "")),
|
|
home=bool(p.get("home", 0)),
|
|
fv_mean=float(p.get("fv_mean", 6.0)),
|
|
fv_std=float(p.get("fv_std", 1.0)),
|
|
mv_mean=float(p.get("mv_mean", 6.0)),
|
|
mv_std=float(p.get("mv_std", 0.5)),
|
|
starter_prob=float(p.get("starter_prob", 0.9)),
|
|
cs_prob=float(p.get("cs_prob", 0.0)),
|
|
))
|
|
|
|
solver = LineupSolver()
|
|
result = solver.optimize(pool)
|
|
|
|
logger.info(f"Optimized lineup: {result}")
|
|
return result
|
|
|
|
def run_full_pipeline(
|
|
self, seasons: Optional[list] = None,
|
|
vote_dir: Optional[str] = None,
|
|
roster_path: Optional[str] = None,
|
|
):
|
|
"""Run the full pipeline end-to-end."""
|
|
seasons = seasons or ["2024-2025", "2025-2026"]
|
|
|
|
logger.info("=" * 50)
|
|
logger.info("Fantabeto 26/27 Pipeline — Starting")
|
|
logger.info("=" * 50)
|
|
|
|
# Step 1: Scrape
|
|
logger.info("Step 1: Scraping FBref data...")
|
|
self.scrape_fbref(seasons, current=True)
|
|
|
|
# Step 2: Process votes
|
|
if vote_dir:
|
|
logger.info("Step 2: Processing votes...")
|
|
self.process_votes(vote_dir, str(Path(vote_dir) / "calendar.xlsx"))
|
|
|
|
# Step 3: Build features
|
|
logger.info("Step 3: Building features...")
|
|
dataset = self.build_features(roster_path=roster_path)
|
|
|
|
# Step 4: Train models
|
|
if "fantavote" in dataset.columns:
|
|
logger.info("Step 4: Training models...")
|
|
y = dataset["fantavote"]
|
|
X = dataset.drop(columns=["fantavote", "vote", "matchday", "player", "team", "oppteam"], errors="ignore")
|
|
self.train_models(X, y)
|
|
|
|
logger.info("Pipeline complete!")
|