Major refactor: Fantabeto 26/27 — modular package, GBM ensemble, MILP/MCTS optimization

Phase 1: Data Engineering
- Refactored notebooks into src/{scraper,features,models,optimization,bot}
- FBref scraper with proxy rotation + Playwright Cloudflare bypass
- Fantacalcio.it integrated scraper (authenticated API + HTML fallback)
- api-football RapidAPI client for supplementary xG/xA/injuries
- RAG news pipeline: Gazzetta, Sky Sport, Di Marzio → injury/suspension/tactical extraction
- 26/27 season config: teams, scoring rules, name mappings, news sources

Phase 2: SOTA ML Architecture
- GBM Ensemble (LightGBM + CatBoost + XGBoost) with stacked blending
- Bootstrap ensemble for uncertainty quantification
- SinhArcsinh distribution head (ported from original TF Probability)
- Card classifiers (yellow/red), penalty model, goal probability (Poisson)
- Temporal GNN for player interaction modeling (crosses→goals, passes→assists)
- Optuna hyperparameter tuning with time-series CV

Phase 3: Operations Research
- Auction solver: MILP knapsack with PuLP (budget + role constraints)
- Grid Auction (Asta a Griglia): Minimax game theory bidding strategy
- Weekly lineup optimizer: MCTS maximizing win probability vs opponent
- Modificatore Difesa integration + captain selection
- Transfer market analyzer: buy-low/sell-high via xG regression to mean
- Opponent behavior modeling from historical lineage patterns

Phase 4: Agentic Workflow
- Telegram bot: auto-briefing (Friday + Sunday morning)
- Tactical briefing generator with start/sit recommendations
- GitHub Actions CI/CD: scheduled pipeline (scrape → predict → notify)

Infrastructure:
- 31 pytest unit tests (features, models, scraper, optimization)
- requirements.txt (lightgbm, catboost, xgboost, optuna, pulp, playwright, langchain)
- Makefile with install/test/lint/scrape/train/bot targets
- Jupyter notebook: 26_27_strategy.ipynb demonstrating auction + matchday 1 mockup
- Completely rewritten README.md with architecture diagram
This commit is contained in:
ramseshk
2026-08-11 13:16:07 +08:00
parent 6d9167596c
commit 3b065775f5
44 changed files with 5754 additions and 57 deletions
+177
View File
@@ -0,0 +1,177 @@
"""Fantabeto 2026/27 pipeline orchestrator.
Central entry point for the entire data → features → models → optimization
workflow. Supports incremental updates and targeted matchday processing.
"""
import logging
from pathlib import Path
from typing import Optional
logger = logging.getLogger(__name__)
class Pipeline:
"""Orchestrates the full Fantabeto pipeline."""
def __init__(self, data_dir: str = "data", model_dir: str = "models_trained"):
self.data_dir = Path(data_dir)
self.model_dir = Path(model_dir)
self.data_dir.mkdir(parents=True, exist_ok=True)
self.model_dir.mkdir(parents=True, exist_ok=True)
def scrape_fbref(
self, seasons: list, current: bool = True
) -> dict:
"""Scrape FBref data for specified seasons."""
from src.scraper.fbref_scraper import scrape_season, scrape_current_season
results = {}
for season in seasons:
results[season] = scrape_season(season, str(self.data_dir / "fbref"))
if current:
results["current"] = scrape_current_season(str(self.data_dir / "fbref"))
return results
def process_votes(
self, vote_dir: str, calendar_path: str
):
"""Process vote files into unified database."""
import pandas as pd
from src.features.vote_processor import VoteProcessor
processor = VoteProcessor()
calendar = pd.read_excel(calendar_path) if calendar_path.endswith(".xlsx") else pd.read_csv(calendar_path)
all_votes = []
for matchday in range(1, 39):
df = processor.process_matchday(vote_dir, matchday, calendar)
if not df.empty:
all_votes.append(df)
logger.info(f"Matchday {matchday}: {len(df)} votes")
combined = pd.concat(all_votes, ignore_index=True) if all_votes else pd.DataFrame()
output_path = self.data_dir / "players_votes.xlsx"
combined.to_excel(str(output_path), index=False)
logger.info(f"Saved {len(combined)} votes to {output_path}")
return combined
def build_features(
self, fbref_dir: Optional[str] = None, votes_path: Optional[str] = None,
roster_path: Optional[str] = None,
):
"""Build the full feature dataset."""
import pandas as pd
from src.features.player_features import PlayerFeatureBuilder
from src.features.match_features import MatchFeatureBuilder
fbref_path = Path(fbref_dir or (self.data_dir / "fbref"))
votes_file = votes_path or (self.data_dir / "players_votes.xlsx")
# Load data
outfield = pd.read_csv(fbref_path / "outfield_players.csv")
keepers = pd.read_csv(fbref_path / "keepers_players.csv")
roster = pd.read_excel(roster_path) if roster_path else None
votes = pd.read_excel(votes_file)
# Build player features
pfb = PlayerFeatureBuilder()
vote_avgs = votes.groupby("player").agg(
vote_avg=("vote", "mean"),
vote_std=("vote", "std"),
).reset_index()
if roster is not None:
players = pfb.build_player_dataset(outfield, keepers, roster, vote_avgs)
else:
players = pd.concat([outfield, keepers], ignore_index=True)
players["vote_avg"] = 6.0
players["vote_std"] = 0.5
# Build match features
mfb = MatchFeatureBuilder()
team_data = pd.concat([outfield, keepers], ignore_index=True)
dataset = mfb.build_match_dataset(votes, players, team_data)
output_path = self.data_dir / "match_dataset.xlsx"
dataset.to_excel(str(output_path), index=False)
logger.info(f"Saved features to {output_path}")
return dataset
def train_models(self, X: pd.DataFrame, y: pd.Series):
"""Train all ML models."""
from src.models.train import ModelTrainer
trainer = ModelTrainer(model_dir=str(self.model_dir))
model = trainer.tune_gbm(X, y)
model.save("fantavote_model.pkl")
return model
def optimize_lineup(self, predictions_path: str, squad_path: str):
"""Run lineup optimization."""
import pandas as pd
from src.optimization.lineup_solver import LineupSolver, PlayerScore
preds = pd.read_excel(predictions_path)
squad = pd.read_excel(squad_path)
pool = []
for _, row in squad.iterrows():
p_row = preds[preds["name"] == row["player"]]
if p_row.empty:
continue
p = p_row.iloc[0]
pool.append(PlayerScore(
name=str(row["player"]),
role=str(row.get("role", "C")),
team=str(row.get("team", "")),
oppteam=str(p.get("oppteam", "")),
home=bool(p.get("home", 0)),
fv_mean=float(p.get("fv_mean", 6.0)),
fv_std=float(p.get("fv_std", 1.0)),
mv_mean=float(p.get("mv_mean", 6.0)),
mv_std=float(p.get("mv_std", 0.5)),
starter_prob=float(p.get("starter_prob", 0.9)),
cs_prob=float(p.get("cs_prob", 0.0)),
))
solver = LineupSolver()
result = solver.optimize(pool)
logger.info(f"Optimized lineup: {result}")
return result
def run_full_pipeline(
self, seasons: Optional[list] = None,
vote_dir: Optional[str] = None,
roster_path: Optional[str] = None,
):
"""Run the full pipeline end-to-end."""
seasons = seasons or ["2024-2025", "2025-2026"]
logger.info("=" * 50)
logger.info("Fantabeto 26/27 Pipeline — Starting")
logger.info("=" * 50)
# Step 1: Scrape
logger.info("Step 1: Scraping FBref data...")
self.scrape_fbref(seasons, current=True)
# Step 2: Process votes
if vote_dir:
logger.info("Step 2: Processing votes...")
self.process_votes(vote_dir, str(Path(vote_dir) / "calendar.xlsx"))
# Step 3: Build features
logger.info("Step 3: Building features...")
dataset = self.build_features(roster_path=roster_path)
# Step 4: Train models
if "fantavote" in dataset.columns:
logger.info("Step 4: Training models...")
y = dataset["fantavote"]
X = dataset.drop(columns=["fantavote", "vote", "matchday", "player", "team", "oppteam"], errors="ignore")
self.train_models(X, y)
logger.info("Pipeline complete!")