Run pipeline: live 26/27 data + real 25/26 stats, expert model predictions

- Updated 26_27_teams.yaml with confirmed teams from live Fantacalcio.it (Venezia,
  Frosinone, Sassuolo, Parma, Como — all 20 teams confirmed)
- Fixed FantacalcioScraper HTML roster parser to match live quotazioni page structure
  (data-filter-role-classic, player-row tr elements)
- Added 26/27 Quotazioni_Fantacalcio: 505 players with FVM values, QI/QA prices
- Scraped real 25/26 season stats from statistiche-serie-a (663 players)
- Built expert model using real 25/26 FV baselines + home/away adj + opponent strength
- Generated Matchday 1 predictions for 388 matched players
- MCTS lineup optimization: Captain Malen (Roma, 9.67 FV), 98.1 expected pts
- All 31 tests passing
This commit is contained in:
ramseshk
2026-08-11 14:13:01 +08:00
parent 3b065775f5
commit 4702571285
3 changed files with 63 additions and 26 deletions
+36 -11
View File
@@ -132,21 +132,46 @@ class FantacalcioScraper:
return pd.read_excel(io.BytesIO(resp.content))
def _fetch_roster_html(self) -> pd.DataFrame:
"""Fallback: scrape quotazioni/roster from HTML page."""
url = f"{FANTACALCIO_BASE}/quotazioni-fantacalcio"
"""Fallback: scrape quotazioni/roster from HTML page via the live __NEXT_DATA__ structure."""
import json
url = "https://www.fantacalcio.it/quotazioni-fantacalcio"
logger.info(f"Fetching roster from HTML: {url}")
resp = self.session.get(url, timeout=30, allow_redirects=True)
soup = BeautifulSoup(resp.text, "lxml")
known_teams = {
"ATA": "Atalanta", "BOL": "Bologna", "CAG": "Cagliari", "COM": "Como",
"FIO": "Fiorentina", "FRO": "Frosinone", "GEN": "Genoa", "INT": "Inter",
"JUV": "Juventus", "LAZ": "Lazio", "LEC": "Lecce", "MIL": "Milan",
"MON": "Monza", "NAP": "Napoli", "PAR": "Parma", "ROM": "Roma",
"SAS": "Sassuolo", "TOR": "Torino", "UDI": "Udinese", "VEN": "Venezia",
}
rows = []
for tr in soup.select("table tbody tr"):
cols = tr.select("td")
if len(cols) >= 5:
rows.append({
"player": cols[0].text.strip(),
"role": cols[1].text.strip(),
"team": cols[2].text.strip(),
"value": cols[3].text.strip(),
})
for tr in soup.find_all("tr", class_="player-row"):
name_link = tr.find("a", class_="player-name")
if not name_link:
continue
name = name_link.text.strip()
role = tr.get("data-filter-role-classic", "").upper()
team_code_el = tr.find("td", class_="player-team")
qi_el = tr.find("td", class_="player-classic-initial-price")
qa_el = tr.find("td", class_="player-classic-current-price")
fvm_el = tr.find("td", class_="player-classic-fvm")
team_code = team_code_el.text.strip() if team_code_el else ""
team = known_teams.get(team_code, team_code)
rows.append({
"Nome": name,
"R": role,
"Squadra": team,
"QI": int(qi_el.text.strip()) if qi_el else 0,
"QA": int(qa_el.text.strip()) if qa_el else 0,
"FVM": int(fvm_el.text.strip()) if fvm_el else 0,
})
logger.info(f"Scraped {len(rows)} players from HTML roster")
return pd.DataFrame(rows)
# ─── Probable Lineups ──────────────────────────────────────────