Major refactor: Fantabeto 26/27 — modular package, GBM ensemble, MILP/MCTS optimization
Phase 1: Data Engineering
- Refactored notebooks into src/{scraper,features,models,optimization,bot}
- FBref scraper with proxy rotation + Playwright Cloudflare bypass
- Fantacalcio.it integrated scraper (authenticated API + HTML fallback)
- api-football RapidAPI client for supplementary xG/xA/injuries
- RAG news pipeline: Gazzetta, Sky Sport, Di Marzio → injury/suspension/tactical extraction
- 26/27 season config: teams, scoring rules, name mappings, news sources
Phase 2: SOTA ML Architecture
- GBM Ensemble (LightGBM + CatBoost + XGBoost) with stacked blending
- Bootstrap ensemble for uncertainty quantification
- SinhArcsinh distribution head (ported from original TF Probability)
- Card classifiers (yellow/red), penalty model, goal probability (Poisson)
- Temporal GNN for player interaction modeling (crosses→goals, passes→assists)
- Optuna hyperparameter tuning with time-series CV
Phase 3: Operations Research
- Auction solver: MILP knapsack with PuLP (budget + role constraints)
- Grid Auction (Asta a Griglia): Minimax game theory bidding strategy
- Weekly lineup optimizer: MCTS maximizing win probability vs opponent
- Modificatore Difesa integration + captain selection
- Transfer market analyzer: buy-low/sell-high via xG regression to mean
- Opponent behavior modeling from historical lineage patterns
Phase 4: Agentic Workflow
- Telegram bot: auto-briefing (Friday + Sunday morning)
- Tactical briefing generator with start/sit recommendations
- GitHub Actions CI/CD: scheduled pipeline (scrape → predict → notify)
Infrastructure:
- 31 pytest unit tests (features, models, scraper, optimization)
- requirements.txt (lightgbm, catboost, xgboost, optuna, pulp, playwright, langchain)
- Makefile with install/test/lint/scrape/train/bot targets
- Jupyter notebook: 26_27_strategy.ipynb demonstrating auction + matchday 1 mockup
- Completely rewritten README.md with architecture diagram
This commit is contained in:
+177
@@ -0,0 +1,177 @@
|
||||
"""Fantabeto 2026/27 pipeline orchestrator.
|
||||
|
||||
Central entry point for the entire data → features → models → optimization
|
||||
workflow. Supports incremental updates and targeted matchday processing.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Pipeline:
|
||||
"""Orchestrates the full Fantabeto pipeline."""
|
||||
|
||||
def __init__(self, data_dir: str = "data", model_dir: str = "models_trained"):
|
||||
self.data_dir = Path(data_dir)
|
||||
self.model_dir = Path(model_dir)
|
||||
self.data_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.model_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def scrape_fbref(
|
||||
self, seasons: list, current: bool = True
|
||||
) -> dict:
|
||||
"""Scrape FBref data for specified seasons."""
|
||||
from src.scraper.fbref_scraper import scrape_season, scrape_current_season
|
||||
|
||||
results = {}
|
||||
for season in seasons:
|
||||
results[season] = scrape_season(season, str(self.data_dir / "fbref"))
|
||||
if current:
|
||||
results["current"] = scrape_current_season(str(self.data_dir / "fbref"))
|
||||
return results
|
||||
|
||||
def process_votes(
|
||||
self, vote_dir: str, calendar_path: str
|
||||
):
|
||||
"""Process vote files into unified database."""
|
||||
import pandas as pd
|
||||
from src.features.vote_processor import VoteProcessor
|
||||
|
||||
processor = VoteProcessor()
|
||||
calendar = pd.read_excel(calendar_path) if calendar_path.endswith(".xlsx") else pd.read_csv(calendar_path)
|
||||
|
||||
all_votes = []
|
||||
for matchday in range(1, 39):
|
||||
df = processor.process_matchday(vote_dir, matchday, calendar)
|
||||
if not df.empty:
|
||||
all_votes.append(df)
|
||||
logger.info(f"Matchday {matchday}: {len(df)} votes")
|
||||
|
||||
combined = pd.concat(all_votes, ignore_index=True) if all_votes else pd.DataFrame()
|
||||
output_path = self.data_dir / "players_votes.xlsx"
|
||||
combined.to_excel(str(output_path), index=False)
|
||||
logger.info(f"Saved {len(combined)} votes to {output_path}")
|
||||
return combined
|
||||
|
||||
def build_features(
|
||||
self, fbref_dir: Optional[str] = None, votes_path: Optional[str] = None,
|
||||
roster_path: Optional[str] = None,
|
||||
):
|
||||
"""Build the full feature dataset."""
|
||||
import pandas as pd
|
||||
from src.features.player_features import PlayerFeatureBuilder
|
||||
from src.features.match_features import MatchFeatureBuilder
|
||||
|
||||
fbref_path = Path(fbref_dir or (self.data_dir / "fbref"))
|
||||
votes_file = votes_path or (self.data_dir / "players_votes.xlsx")
|
||||
|
||||
# Load data
|
||||
outfield = pd.read_csv(fbref_path / "outfield_players.csv")
|
||||
keepers = pd.read_csv(fbref_path / "keepers_players.csv")
|
||||
roster = pd.read_excel(roster_path) if roster_path else None
|
||||
votes = pd.read_excel(votes_file)
|
||||
|
||||
# Build player features
|
||||
pfb = PlayerFeatureBuilder()
|
||||
vote_avgs = votes.groupby("player").agg(
|
||||
vote_avg=("vote", "mean"),
|
||||
vote_std=("vote", "std"),
|
||||
).reset_index()
|
||||
|
||||
if roster is not None:
|
||||
players = pfb.build_player_dataset(outfield, keepers, roster, vote_avgs)
|
||||
else:
|
||||
players = pd.concat([outfield, keepers], ignore_index=True)
|
||||
players["vote_avg"] = 6.0
|
||||
players["vote_std"] = 0.5
|
||||
|
||||
# Build match features
|
||||
mfb = MatchFeatureBuilder()
|
||||
team_data = pd.concat([outfield, keepers], ignore_index=True)
|
||||
|
||||
dataset = mfb.build_match_dataset(votes, players, team_data)
|
||||
|
||||
output_path = self.data_dir / "match_dataset.xlsx"
|
||||
dataset.to_excel(str(output_path), index=False)
|
||||
logger.info(f"Saved features to {output_path}")
|
||||
return dataset
|
||||
|
||||
def train_models(self, X: pd.DataFrame, y: pd.Series):
|
||||
"""Train all ML models."""
|
||||
from src.models.train import ModelTrainer
|
||||
|
||||
trainer = ModelTrainer(model_dir=str(self.model_dir))
|
||||
model = trainer.tune_gbm(X, y)
|
||||
model.save("fantavote_model.pkl")
|
||||
return model
|
||||
|
||||
def optimize_lineup(self, predictions_path: str, squad_path: str):
|
||||
"""Run lineup optimization."""
|
||||
import pandas as pd
|
||||
from src.optimization.lineup_solver import LineupSolver, PlayerScore
|
||||
|
||||
preds = pd.read_excel(predictions_path)
|
||||
squad = pd.read_excel(squad_path)
|
||||
|
||||
pool = []
|
||||
for _, row in squad.iterrows():
|
||||
p_row = preds[preds["name"] == row["player"]]
|
||||
if p_row.empty:
|
||||
continue
|
||||
p = p_row.iloc[0]
|
||||
pool.append(PlayerScore(
|
||||
name=str(row["player"]),
|
||||
role=str(row.get("role", "C")),
|
||||
team=str(row.get("team", "")),
|
||||
oppteam=str(p.get("oppteam", "")),
|
||||
home=bool(p.get("home", 0)),
|
||||
fv_mean=float(p.get("fv_mean", 6.0)),
|
||||
fv_std=float(p.get("fv_std", 1.0)),
|
||||
mv_mean=float(p.get("mv_mean", 6.0)),
|
||||
mv_std=float(p.get("mv_std", 0.5)),
|
||||
starter_prob=float(p.get("starter_prob", 0.9)),
|
||||
cs_prob=float(p.get("cs_prob", 0.0)),
|
||||
))
|
||||
|
||||
solver = LineupSolver()
|
||||
result = solver.optimize(pool)
|
||||
|
||||
logger.info(f"Optimized lineup: {result}")
|
||||
return result
|
||||
|
||||
def run_full_pipeline(
|
||||
self, seasons: Optional[list] = None,
|
||||
vote_dir: Optional[str] = None,
|
||||
roster_path: Optional[str] = None,
|
||||
):
|
||||
"""Run the full pipeline end-to-end."""
|
||||
seasons = seasons or ["2024-2025", "2025-2026"]
|
||||
|
||||
logger.info("=" * 50)
|
||||
logger.info("Fantabeto 26/27 Pipeline — Starting")
|
||||
logger.info("=" * 50)
|
||||
|
||||
# Step 1: Scrape
|
||||
logger.info("Step 1: Scraping FBref data...")
|
||||
self.scrape_fbref(seasons, current=True)
|
||||
|
||||
# Step 2: Process votes
|
||||
if vote_dir:
|
||||
logger.info("Step 2: Processing votes...")
|
||||
self.process_votes(vote_dir, str(Path(vote_dir) / "calendar.xlsx"))
|
||||
|
||||
# Step 3: Build features
|
||||
logger.info("Step 3: Building features...")
|
||||
dataset = self.build_features(roster_path=roster_path)
|
||||
|
||||
# Step 4: Train models
|
||||
if "fantavote" in dataset.columns:
|
||||
logger.info("Step 4: Training models...")
|
||||
y = dataset["fantavote"]
|
||||
X = dataset.drop(columns=["fantavote", "vote", "matchday", "player", "team", "oppteam"], errors="ignore")
|
||||
self.train_models(X, y)
|
||||
|
||||
logger.info("Pipeline complete!")
|
||||
Reference in New Issue
Block a user