"""Fantabeto 2026/27 pipeline orchestrator. Central entry point for the entire data → features → models → optimization workflow. Supports incremental updates and targeted matchday processing. """ import logging from pathlib import Path from typing import Optional logger = logging.getLogger(__name__) class Pipeline: """Orchestrates the full Fantabeto pipeline.""" def __init__(self, data_dir: str = "data", model_dir: str = "models_trained"): self.data_dir = Path(data_dir) self.model_dir = Path(model_dir) self.data_dir.mkdir(parents=True, exist_ok=True) self.model_dir.mkdir(parents=True, exist_ok=True) def scrape_fbref( self, seasons: list, current: bool = True ) -> dict: """Scrape FBref data for specified seasons.""" from src.scraper.fbref_scraper import scrape_season, scrape_current_season results = {} for season in seasons: results[season] = scrape_season(season, str(self.data_dir / "fbref")) if current: results["current"] = scrape_current_season(str(self.data_dir / "fbref")) return results def process_votes( self, vote_dir: str, calendar_path: str ): """Process vote files into unified database.""" import pandas as pd from src.features.vote_processor import VoteProcessor processor = VoteProcessor() calendar = pd.read_excel(calendar_path) if calendar_path.endswith(".xlsx") else pd.read_csv(calendar_path) all_votes = [] for matchday in range(1, 39): df = processor.process_matchday(vote_dir, matchday, calendar) if not df.empty: all_votes.append(df) logger.info(f"Matchday {matchday}: {len(df)} votes") combined = pd.concat(all_votes, ignore_index=True) if all_votes else pd.DataFrame() output_path = self.data_dir / "players_votes.xlsx" combined.to_excel(str(output_path), index=False) logger.info(f"Saved {len(combined)} votes to {output_path}") return combined def build_features( self, fbref_dir: Optional[str] = None, votes_path: Optional[str] = None, roster_path: Optional[str] = None, ): """Build the full feature dataset.""" import pandas as pd from src.features.player_features import PlayerFeatureBuilder from src.features.match_features import MatchFeatureBuilder fbref_path = Path(fbref_dir or (self.data_dir / "fbref")) votes_file = votes_path or (self.data_dir / "players_votes.xlsx") # Load data outfield = pd.read_csv(fbref_path / "outfield_players.csv") keepers = pd.read_csv(fbref_path / "keepers_players.csv") roster = pd.read_excel(roster_path) if roster_path else None votes = pd.read_excel(votes_file) # Build player features pfb = PlayerFeatureBuilder() vote_avgs = votes.groupby("player").agg( vote_avg=("vote", "mean"), vote_std=("vote", "std"), ).reset_index() if roster is not None: players = pfb.build_player_dataset(outfield, keepers, roster, vote_avgs) else: players = pd.concat([outfield, keepers], ignore_index=True) players["vote_avg"] = 6.0 players["vote_std"] = 0.5 # Build match features mfb = MatchFeatureBuilder() team_data = pd.concat([outfield, keepers], ignore_index=True) dataset = mfb.build_match_dataset(votes, players, team_data) output_path = self.data_dir / "match_dataset.xlsx" dataset.to_excel(str(output_path), index=False) logger.info(f"Saved features to {output_path}") return dataset def train_models(self, X: pd.DataFrame, y: pd.Series): """Train all ML models.""" from src.models.train import ModelTrainer trainer = ModelTrainer(model_dir=str(self.model_dir)) model = trainer.tune_gbm(X, y) model.save("fantavote_model.pkl") return model def optimize_lineup(self, predictions_path: str, squad_path: str): """Run lineup optimization.""" import pandas as pd from src.optimization.lineup_solver import LineupSolver, PlayerScore preds = pd.read_excel(predictions_path) squad = pd.read_excel(squad_path) pool = [] for _, row in squad.iterrows(): p_row = preds[preds["name"] == row["player"]] if p_row.empty: continue p = p_row.iloc[0] pool.append(PlayerScore( name=str(row["player"]), role=str(row.get("role", "C")), team=str(row.get("team", "")), oppteam=str(p.get("oppteam", "")), home=bool(p.get("home", 0)), fv_mean=float(p.get("fv_mean", 6.0)), fv_std=float(p.get("fv_std", 1.0)), mv_mean=float(p.get("mv_mean", 6.0)), mv_std=float(p.get("mv_std", 0.5)), starter_prob=float(p.get("starter_prob", 0.9)), cs_prob=float(p.get("cs_prob", 0.0)), )) solver = LineupSolver() result = solver.optimize(pool) logger.info(f"Optimized lineup: {result}") return result def run_full_pipeline( self, seasons: Optional[list] = None, vote_dir: Optional[str] = None, roster_path: Optional[str] = None, ): """Run the full pipeline end-to-end.""" seasons = seasons or ["2024-2025", "2025-2026"] logger.info("=" * 50) logger.info("Fantabeto 26/27 Pipeline — Starting") logger.info("=" * 50) # Step 1: Scrape logger.info("Step 1: Scraping FBref data...") self.scrape_fbref(seasons, current=True) # Step 2: Process votes if vote_dir: logger.info("Step 2: Processing votes...") self.process_votes(vote_dir, str(Path(vote_dir) / "calendar.xlsx")) # Step 3: Build features logger.info("Step 3: Building features...") dataset = self.build_features(roster_path=roster_path) # Step 4: Train models if "fantavote" in dataset.columns: logger.info("Step 4: Training models...") y = dataset["fantavote"] X = dataset.drop(columns=["fantavote", "vote", "matchday", "player", "team", "oppteam"], errors="ignore") self.train_models(X, y) logger.info("Pipeline complete!")