Major refactor: Fantabeto 26/27 — modular package, GBM ensemble, MILP/MCTS optimization
Phase 1: Data Engineering
- Refactored notebooks into src/{scraper,features,models,optimization,bot}
- FBref scraper with proxy rotation + Playwright Cloudflare bypass
- Fantacalcio.it integrated scraper (authenticated API + HTML fallback)
- api-football RapidAPI client for supplementary xG/xA/injuries
- RAG news pipeline: Gazzetta, Sky Sport, Di Marzio → injury/suspension/tactical extraction
- 26/27 season config: teams, scoring rules, name mappings, news sources
Phase 2: SOTA ML Architecture
- GBM Ensemble (LightGBM + CatBoost + XGBoost) with stacked blending
- Bootstrap ensemble for uncertainty quantification
- SinhArcsinh distribution head (ported from original TF Probability)
- Card classifiers (yellow/red), penalty model, goal probability (Poisson)
- Temporal GNN for player interaction modeling (crosses→goals, passes→assists)
- Optuna hyperparameter tuning with time-series CV
Phase 3: Operations Research
- Auction solver: MILP knapsack with PuLP (budget + role constraints)
- Grid Auction (Asta a Griglia): Minimax game theory bidding strategy
- Weekly lineup optimizer: MCTS maximizing win probability vs opponent
- Modificatore Difesa integration + captain selection
- Transfer market analyzer: buy-low/sell-high via xG regression to mean
- Opponent behavior modeling from historical lineage patterns
Phase 4: Agentic Workflow
- Telegram bot: auto-briefing (Friday + Sunday morning)
- Tactical briefing generator with start/sit recommendations
- GitHub Actions CI/CD: scheduled pipeline (scrape → predict → notify)
Infrastructure:
- 31 pytest unit tests (features, models, scraper, optimization)
- requirements.txt (lightgbm, catboost, xgboost, optuna, pulp, playwright, langchain)
- Makefile with install/test/lint/scrape/train/bot targets
- Jupyter notebook: 26_27_strategy.ipynb demonstrating auction + matchday 1 mockup
- Completely rewritten README.md with architecture diagram
This commit is contained in:
@@ -0,0 +1,155 @@
|
||||
"""Transfer market analysis ('Svincolati' / free agent pool).
|
||||
|
||||
Identifies buy-low and sell-high targets using advanced metrics:
|
||||
- Regression to the mean: compares actual vs expected output
|
||||
- xG/xA vs actual goals/assists divergence
|
||||
- Minutes trending up/down
|
||||
- Market value arbitrage
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class TransferAnalyzer:
|
||||
"""Analyzes the Svincolati (free agent) market for arbitrage opportunities."""
|
||||
|
||||
def __init__(self, regression_factor: float = 0.3):
|
||||
self.regression_factor = regression_factor
|
||||
|
||||
def compute_expected_output(
|
||||
self, xg: float, xa: float, historical_mean: float
|
||||
) -> float:
|
||||
"""Compute regressed expected output using xG/xA.
|
||||
|
||||
Shrinks toward the player's historical mean (regression to the mean).
|
||||
"""
|
||||
raw_expected = xg * 3.0 + xa * 1.0 # convert to Fantavoto scale
|
||||
regressed = (
|
||||
self.regression_factor * historical_mean +
|
||||
(1 - self.regression_factor) * raw_expected
|
||||
)
|
||||
return regressed
|
||||
|
||||
def analyze_buy_low(
|
||||
self, players_df: pd.DataFrame, min_minutes: int = 180
|
||||
) -> pd.DataFrame:
|
||||
"""Identify buy-low candidates.
|
||||
|
||||
Criteria:
|
||||
- xG/xA significantly exceed actual output
|
||||
- Minutes trending up
|
||||
- Low market value relative to projection
|
||||
|
||||
Args:
|
||||
players_df: DataFrame with columns:
|
||||
[name, team, role, actual_fv_avg, xg, xa, minutes, market_value,
|
||||
minutes_trend, historical_fv_avg]
|
||||
|
||||
Returns:
|
||||
DataFrame of buy-low candidates ranked by opportunity.
|
||||
"""
|
||||
df = players_df.copy()
|
||||
df = df[df["minutes"] >= min_minutes]
|
||||
|
||||
if "xg" not in df.columns or "xa" not in df.columns:
|
||||
logger.warning("xG/xA data missing; using basic analysis")
|
||||
return pd.DataFrame()
|
||||
|
||||
# Expected fantavoto from xG/xA
|
||||
df["expected_fv"] = df.apply(
|
||||
lambda r: self.compute_expected_output(
|
||||
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
|
||||
),
|
||||
axis=1,
|
||||
)
|
||||
|
||||
# Divergence: expected minus actual
|
||||
df["fv_divergence"] = df["expected_fv"] - df.get("actual_fv_avg", 6.0)
|
||||
|
||||
df["buy_low_score"] = (
|
||||
df["fv_divergence"] * 2.0 + # underperformance signal
|
||||
df.get("minutes_trend", 0) * 0.5 + # trending up
|
||||
(1.0 / (df.get("market_value", 1) + 1)) * 10 # cheap
|
||||
)
|
||||
|
||||
buy_low = df[df["buy_low_score"] > 0].sort_values("buy_low_score", ascending=False)
|
||||
|
||||
result = buy_low[[
|
||||
"name", "team", "role", "actual_fv_avg", "expected_fv",
|
||||
"fv_divergence", "buy_low_score", "market_value",
|
||||
]].copy()
|
||||
|
||||
result["recommendation"] = "BUY-LOW"
|
||||
result["confidence"] = pd.cut(
|
||||
result["buy_low_score"],
|
||||
bins=[-np.inf, 1, 3, 5, np.inf],
|
||||
labels=["Low", "Medium", "High", "Very High"],
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"Found {len(result)} buy-low candidates "
|
||||
f"(avg divergence: {result['fv_divergence'].mean():.2f})"
|
||||
)
|
||||
return result
|
||||
|
||||
def analyze_sell_high(
|
||||
self, players_df: pd.DataFrame, min_minutes: int = 180
|
||||
) -> pd.DataFrame:
|
||||
"""Identify sell-high candidates.
|
||||
|
||||
Criteria:
|
||||
- Actual output exceeds xG/xA by large margin
|
||||
- Minutes trending down
|
||||
- High market value vs projection
|
||||
"""
|
||||
df = players_df.copy()
|
||||
df = df[df["minutes"] >= min_minutes]
|
||||
|
||||
if "xg" not in df.columns:
|
||||
return pd.DataFrame()
|
||||
|
||||
df["expected_fv"] = df.apply(
|
||||
lambda r: self.compute_expected_output(
|
||||
r.get("xg", 0), r.get("xa", 0), r.get("historical_fv_avg", 6.0)
|
||||
),
|
||||
axis=1,
|
||||
)
|
||||
|
||||
# Overperformance
|
||||
df["fv_divergence"] = df.get("actual_fv_avg", 6.0) - df["expected_fv"]
|
||||
|
||||
df["sell_high_score"] = (
|
||||
df["fv_divergence"] * 3.0 + # overperformance signal
|
||||
(df.get("minutes_trend", 0) * -0.5 if "minutes_trend" in df.columns else 0)
|
||||
)
|
||||
|
||||
sell_high = df[df["sell_high_score"] > 1].sort_values("sell_high_score", ascending=False)
|
||||
|
||||
result = sell_high[[
|
||||
"name", "team", "role", "actual_fv_avg", "expected_fv",
|
||||
"fv_divergence", "sell_high_score",
|
||||
]].copy()
|
||||
|
||||
result["recommendation"] = "SELL-HIGH"
|
||||
logger.info(f"Found {len(result)} sell-high candidates")
|
||||
return result
|
||||
|
||||
def full_transfer_report(self, players_df: pd.DataFrame) -> dict:
|
||||
"""Generate complete transfer market report."""
|
||||
buy = self.analyze_buy_low(players_df)
|
||||
sell = self.analyze_sell_high(players_df)
|
||||
|
||||
return {
|
||||
"buy_low": buy,
|
||||
"sell_high": sell,
|
||||
"summary": (
|
||||
f"Buy-low targets: {len(buy)} players identified. "
|
||||
f"Sell-high targets: {len(sell)} players identified."
|
||||
),
|
||||
}
|
||||
Reference in New Issue
Block a user