Run pipeline: live 26/27 data + real 25/26 stats, expert model predictions
- Updated 26_27_teams.yaml with confirmed teams from live Fantacalcio.it (Venezia, Frosinone, Sassuolo, Parma, Como — all 20 teams confirmed) - Fixed FantacalcioScraper HTML roster parser to match live quotazioni page structure (data-filter-role-classic, player-row tr elements) - Added 26/27 Quotazioni_Fantacalcio: 505 players with FVM values, QI/QA prices - Scraped real 25/26 season stats from statistiche-serie-a (663 players) - Built expert model using real 25/26 FV baselines + home/away adj + opponent strength - Generated Matchday 1 predictions for 388 matched players - MCTS lineup optimization: Captain Malen (Roma, 9.67 FV), 98.1 expected pts - All 31 tests passing
This commit is contained in:
@@ -132,21 +132,46 @@ class FantacalcioScraper:
|
||||
return pd.read_excel(io.BytesIO(resp.content))
|
||||
|
||||
def _fetch_roster_html(self) -> pd.DataFrame:
|
||||
"""Fallback: scrape quotazioni/roster from HTML page."""
|
||||
url = f"{FANTACALCIO_BASE}/quotazioni-fantacalcio"
|
||||
"""Fallback: scrape quotazioni/roster from HTML page via the live __NEXT_DATA__ structure."""
|
||||
import json
|
||||
url = "https://www.fantacalcio.it/quotazioni-fantacalcio"
|
||||
logger.info(f"Fetching roster from HTML: {url}")
|
||||
resp = self.session.get(url, timeout=30, allow_redirects=True)
|
||||
soup = BeautifulSoup(resp.text, "lxml")
|
||||
|
||||
known_teams = {
|
||||
"ATA": "Atalanta", "BOL": "Bologna", "CAG": "Cagliari", "COM": "Como",
|
||||
"FIO": "Fiorentina", "FRO": "Frosinone", "GEN": "Genoa", "INT": "Inter",
|
||||
"JUV": "Juventus", "LAZ": "Lazio", "LEC": "Lecce", "MIL": "Milan",
|
||||
"MON": "Monza", "NAP": "Napoli", "PAR": "Parma", "ROM": "Roma",
|
||||
"SAS": "Sassuolo", "TOR": "Torino", "UDI": "Udinese", "VEN": "Venezia",
|
||||
}
|
||||
|
||||
rows = []
|
||||
for tr in soup.select("table tbody tr"):
|
||||
cols = tr.select("td")
|
||||
if len(cols) >= 5:
|
||||
rows.append({
|
||||
"player": cols[0].text.strip(),
|
||||
"role": cols[1].text.strip(),
|
||||
"team": cols[2].text.strip(),
|
||||
"value": cols[3].text.strip(),
|
||||
})
|
||||
for tr in soup.find_all("tr", class_="player-row"):
|
||||
name_link = tr.find("a", class_="player-name")
|
||||
if not name_link:
|
||||
continue
|
||||
name = name_link.text.strip()
|
||||
role = tr.get("data-filter-role-classic", "").upper()
|
||||
team_code_el = tr.find("td", class_="player-team")
|
||||
qi_el = tr.find("td", class_="player-classic-initial-price")
|
||||
qa_el = tr.find("td", class_="player-classic-current-price")
|
||||
fvm_el = tr.find("td", class_="player-classic-fvm")
|
||||
|
||||
team_code = team_code_el.text.strip() if team_code_el else ""
|
||||
team = known_teams.get(team_code, team_code)
|
||||
|
||||
rows.append({
|
||||
"Nome": name,
|
||||
"R": role,
|
||||
"Squadra": team,
|
||||
"QI": int(qi_el.text.strip()) if qi_el else 0,
|
||||
"QA": int(qa_el.text.strip()) if qa_el else 0,
|
||||
"FVM": int(fvm_el.text.strip()) if fvm_el else 0,
|
||||
})
|
||||
|
||||
logger.info(f"Scraped {len(rows)} players from HTML roster")
|
||||
return pd.DataFrame(rows)
|
||||
|
||||
# ─── Probable Lineups ──────────────────────────────────────────
|
||||
|
||||
Reference in New Issue
Block a user