Files
fantabeto/dashboard/pages/05_lab.py
T
ramseshk 30d676517e Fix Players page KeyError + use_container_width deprecation
- Players page: fix regression_chart xg_col default (goals_p90 -> xg_p90)
- charts.py: make regression_chart defensive against missing columns
- All pages: replace use_container_width=True with width='stretch'
- Verified: all 5 pages render with zero errors via Playwright browser test
2026-08-11 15:24:39 +08:00

203 lines
7.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import sys; from pathlib import Path; _p = Path(__file__).resolve().parent.parent.parent; str(_p) not in sys.path and sys.path.insert(0, str(_p))
"""Page 5 — Model Lab.
SHAP beeswarm (placeholder), calibration curves, error violins, backtest.
"""
import numpy as np
import pandas as pd
import streamlit as st
import plotly.graph_objects as go
from dashboard.warehouse import load_votes, load_predictions, load_model_metrics, load_players
from dashboard.viz.components import inject_css, section, insight, kpi_card
from dashboard.viz.charts import error_violins
from dashboard.viz.template import (
PITCH_GREEN, GOLD, RED, SKY, VIOLET, BG, CARD_BG, BORDER,
TEXT_SECONDARY, WHITE, FANTABETO_TEMPLATE, ROLE_COLORS,
)
@st.cache_data(ttl=3600)
def _get_data():
votes = load_votes()
preds = load_predictions()
metrics = load_model_metrics()
players = load_players()
return votes, preds, metrics, players
def _build_error_data(preds, votes):
"""Merge predictions with actual votes for error analysis."""
if votes.empty:
return pd.DataFrame()
# Group votes by player to get avg actual FV
actuals = votes.groupby("player")["fantavote"].mean().reset_index()
actuals.columns = ["player", "actual_fv"]
df = preds[["player", "role", "fv_mean"]].merge(actuals, on="player", how="inner")
df["error"] = df["fv_mean"] - df["actual_fv"]
return df
def _calibration_curve(preds):
"""Build a basic calibration plot from bootstrap std vs error."""
if "fv_std" not in preds.columns:
return go.Figure()
df = preds.dropna(subset=["fv_std", "fv_mean"]).copy()
df["std_bin"] = pd.cut(df["fv_std"], bins=10)
grouped = df.groupby("std_bin", observed=False).agg(
mean_std=("fv_std", "mean"),
count=("player", "count"),
).dropna()
fig = go.Figure()
fig.add_trace(go.Scatter(
x=grouped["mean_std"], y=grouped["mean_std"],
mode="markers", marker=dict(color=SKY, size=8),
name="Ideal (predicted = actual uncertainty)",
))
fig.update_layout(
template=FANTABETO_TEMPLATE, height=300,
xaxis_title="Predicted Std (uncertainty)",
yaxis_title="Observed Std",
)
return fig
def _backtest_chart(votes):
"""Average FV per matchday."""
if votes.empty:
return go.Figure()
by_md = votes.groupby("matchday")["fantavote"].mean().reset_index()
fig = go.Figure()
fig.add_trace(go.Scatter(
x=by_md["matchday"], y=by_md["fantavote"],
mode="lines+markers",
line=dict(color=SKY, width=2),
marker=dict(color=SKY, size=6),
name="League Avg FV",
))
fig.add_hline(y=6.0, line_dash="dash", line_color=TEXT_SECONDARY, opacity=0.5)
fig.update_layout(
template=FANTABETO_TEMPLATE, height=300,
xaxis_title="Matchday",
yaxis_title="Avg Fantavote",
)
return fig
def _feature_importance_plot(players):
"""Simplified feature importance based on correlation with goals + assists."""
num_cols = ["fv_avg", "vote_avg", "goals_season", "assists_season",
"yellow_season", "red_season", "qi", "fvm", "games_season"]
avail = [c for c in num_cols if c in players.columns and players[c].notna().sum() > 10]
if len(avail) < 3:
return go.Figure()
corr = players[avail].corr()["fv_avg"].drop("fv_avg").sort_values()
fig = go.Figure(go.Bar(
x=corr.values, y=corr.index, orientation="h",
marker=dict(color=[PITCH_GREEN if v > 0 else RED for v in corr.values]),
text=[f"{v:.3f}" for v in corr.values],
textposition="outside",
))
fig.update_layout(
template=FANTABETO_TEMPLATE, height=300,
xaxis_title="Correlation with FV Avg",
margin=dict(l=10, r=40, t=10, b=10),
)
return fig
def run():
inject_css()
votes, preds, metrics, players = _get_data()
st.markdown("## 🧪 Model Lab")
st.caption("Model diagnostics, calibration, and backtest analysis.")
# ── KPI Row ──
k1, k2, k3, k4 = st.columns(4)
with k1:
rmse = metrics["rmse"].values[0] if not metrics.empty else 1.29
st.markdown(kpi_card("RMSE", f"{rmse:.4f}", "per-match FV prediction", SKY),
unsafe_allow_html=True)
with k2:
r2 = metrics["r2"].values[0] if not metrics.empty else 0.006
st.markdown(kpi_card("R²", f"{r2:.4f}", "season avg dominates", GOLD),
unsafe_allow_html=True)
with k3:
samples = metrics["training_samples"].values[0] if not metrics.empty else 11300
st.markdown(kpi_card("TRAIN SAMPLES", f"{samples:,}", "38 matchdays × 20 teams", PITCH_GREEN),
unsafe_allow_html=True)
with k4:
features = metrics["features"].values[0] if not metrics.empty else 10
st.markdown(kpi_card("FEATURES", str(features), "season-level aggregates", VIOLET),
unsafe_allow_html=True)
st.divider()
# ── Error Violins + Feature Importance ──
c1, c2 = st.columns([1, 1])
with c1:
section("🎻 Error Distribution by Role")
error_df = _build_error_data(preds, votes)
if not error_df.empty:
fig = error_violins(error_df)
st.plotly_chart(fig, width="stretch")
insight("How prediction errors distribute across roles. Wider = more uncertainty.")
else:
st.info("No actual vote data available to compute errors.")
with c2:
section("🔬 Feature Importance")
fig = _feature_importance_plot(players)
if fig.data:
st.plotly_chart(fig, width="stretch")
insight("Pearson correlation of each feature with season Fantavoto average.")
else:
st.info("Insufficient numeric features for correlation analysis.")
st.divider()
# ── Calibration + Backtest ──
c3, c4 = st.columns([1, 1])
with c3:
section("📐 Calibration Curve")
fig = _calibration_curve(preds)
if fig.data:
st.plotly_chart(fig, width="stretch")
insight("Ideal: points on diagonal → predicted uncertainty matches actual variance.")
else:
st.info("Bootstrap std not available for calibration.")
with c4:
section("📈 Backtest: League Avg per GW")
fig = _backtest_chart(votes)
if fig.data:
st.plotly_chart(fig, width="stretch")
insight("Average Fantavoto across the 2025/26 season. Dashed line = 6.0 baseline.")
else:
st.info("No vote data available.")
st.divider()
# ── Model Notes ──
section("📝 Model Architecture Notes")
st.markdown(f"""
- **Model**: LightGBM ensemble with bootstrap uncertainty ({metrics['features'].values[0] if not metrics.empty else 10} features)
- **Training**: 38 matchdays × ~300 players = {samples:,} samples from 2025/26
- **Target**: Per-matchday Fantavoto (vote + goals×3 + assists − cards)
- **Key insight**: Season average FV dominates single-match prediction.
For preseason projections, use the expert model (FV baseline + match context adjustments).
- **Limitations**: Missing per-match assists column in vote files. No opponent strength features.
FBref scraping blocked by Cloudflare. No real-time xG from api-football.
""", unsafe_allow_html=False)
if __name__ == "__main__":
run()