FPL / app.py
Seewhatson's picture
feat: add get_player MCP tool and UI tab for single player search
d63ef2c
Raw History Blame Contribute Delete
40.4 kB
"""
FPL Point Predictor with Gradio MCP Support
Exposes FPL predictions as MCP tools for Claude Desktop
"""
import math
import pandas as pd
import numpy as np
import requests
import json
import os
import time
import threading
import gradio as gr
from concurrent.futures import ThreadPoolExecutor, as_completed
from sklearn.linear_model import Ridge
from sklearn.metrics import mean_squared_error
from sklearn.preprocessing import StandardScaler
from xgboost import XGBRegressor
from typing import Optional
import warnings
warnings.filterwarnings('ignore')
MANAGER_ID = int(os.environ.get("FPL_MANAGER_ID", "8679881"))
POSITION_MAP = {1: 'Goalkeeper', 2: 'Defender', 3: 'Midfielder', 4: 'Forward'}
VALID_POSITIONS = set(POSITION_MAP.values())
_data_lock = threading.RLock()
# Globals β€” populated by load_and_process_data()
final_predictions_df: pd.DataFrame = pd.DataFrame()
team_performance_df: pd.DataFrame = pd.DataFrame()
manager_players_df: pd.DataFrame = pd.DataFrame()
gw_status_df: pd.DataFrame = pd.DataFrame()
model_metrics_df: pd.DataFrame = pd.DataFrame()
current_gameweek: int = 0
NEXT_GW: int = 1
class CachedFPL_API:
def __init__(self, base_url="https://fantasy.premierleague.com/api/", cache_dir="data", cache_ttl=3600):
self.base_url = base_url
self.cache_dir = cache_dir
self.cache_ttl = cache_ttl
os.makedirs(cache_dir, exist_ok=True)
def fetch_data(self, endpoint):
filename = endpoint.replace("/", "_").replace("?", "_").replace("&", "_") + ".json"
filepath = os.path.join(self.cache_dir, filename)
if os.path.exists(filepath):
if time.time() - os.path.getmtime(filepath) < self.cache_ttl:
with open(filepath, 'r', encoding='utf-8') as f:
return json.load(f)
url = f"{self.base_url}{endpoint}"
try:
response = requests.get(url, timeout=10)
response.raise_for_status()
data = response.json()
with open(filepath, 'w', encoding='utf-8') as f:
json.dump(data, f)
return data
except Exception as e:
print(f"Error fetching {url}: {e}")
if os.path.exists(filepath):
with open(filepath, 'r', encoding='utf-8') as f:
return json.load(f)
return None
def get_bootstrap_static(self):
return self.fetch_data("bootstrap-static/")
def get_fixtures(self):
return self.fetch_data("fixtures/")
def get_player_summary(self, player_id):
return self.fetch_data(f"element-summary/{player_id}/")
def get_gameweek_fixtures(self, event_id):
return self.fetch_data(f"fixtures/?event={event_id}")
def get_manager_team(self, manager_id=None, gameweek=None):
mid = manager_id or MANAGER_ID
gw = gameweek if gameweek is not None else 1
return self.fetch_data(f"entry/{mid}/event/{gw}/picks/")
def trend_emoji(short_avg, long_avg):
if short_avg > 0 and long_avg > 0:
return '⬆️' if short_avg > long_avg else ('⬇️' if short_avg < long_avg else '➑️')
return '➑️' if short_avg > 0 or long_avg > 0 else 'βšͺ'
def load_and_process_data():
"""Load FPL data, train models, compute metrics, and generate predictions."""
FPL = CachedFPL_API()
bootstrap_data = FPL.get_bootstrap_static()
events = bootstrap_data['events']
teams = bootstrap_data['teams']
elements = bootstrap_data['elements']
current_gameweek = next((e['id'] for e in events if e['is_current']), None)
if not current_gameweek:
current_gameweek = next((e['id'] for e in events if e['is_next']), 2) - 1
all_players = pd.DataFrame(elements)
all_players['full_name'] = all_players['first_name'] + " " + all_players['second_name']
all_players['selected_by_percent'] = pd.to_numeric(all_players['selected_by_percent'], errors='coerce').fillna(0.0)
# Lookup dicts for efficient mapping
team_name_map = {t['id']: t['name'] for t in teams}
player_team_map = all_players.set_index('id')['team'].to_dict()
player_type_map = all_players.set_index('id')['element_type'].to_dict()
player_name_map = all_players.set_index('id')['web_name'].to_dict()
ownership_map = all_players.set_index('id')['selected_by_percent'].to_dict()
# Fetch player histories concurrently
def fetch_player(pid):
history = FPL.get_player_summary(pid)
if history and 'history' in history:
df = pd.DataFrame(history['history'])
df['element'] = pid
return df
return None
player_histories = []
with ThreadPoolExecutor(max_workers=20) as executor:
futures = {executor.submit(fetch_player, pid): pid for pid in all_players['id'].tolist()}
for future in as_completed(futures):
result = future.result()
if result is not None:
player_histories.append(result)
all_fpl_players_games = pd.concat(player_histories, ignore_index=True)
# Preprocessing
cols_to_numeric = ['influence', 'creativity', 'threat', 'ict_index', 'expected_goals',
'expected_assists', 'expected_goal_involvements', 'expected_goals_conceded']
for col in cols_to_numeric:
all_fpl_players_games[col] = pd.to_numeric(all_fpl_players_games[col], errors='coerce').fillna(0)
if 'kickoff_time' in all_fpl_players_games.columns:
all_fpl_players_games = all_fpl_players_games.drop('kickoff_time', axis=1)
# Team Strength
team_stats_data = []
for gw in range(1, current_gameweek + 1):
fixtures = FPL.get_gameweek_fixtures(gw)
if fixtures:
for fixture in fixtures:
if fixture['finished']:
team_stats_data.append({
'team_id': fixture['team_h'], 'round': gw,
'goals_scored': fixture['team_h_score'], 'goals_conceded': fixture['team_a_score'],
'was_home': True
})
team_stats_data.append({
'team_id': fixture['team_a'], 'round': gw,
'goals_scored': fixture['team_a_score'], 'goals_conceded': fixture['team_h_score'],
'was_home': False
})
team_stats_df = pd.DataFrame(team_stats_data)
team_stats_df = team_stats_df.sort_values(by=['team_id', 'round'])
team_stats_df['attack_strength'] = team_stats_df.groupby('team_id')['goals_scored'].transform(
lambda x: x.rolling(5, min_periods=1).mean().shift(1)
)
team_stats_df['defence_strength'] = team_stats_df.groupby('team_id')['goals_conceded'].transform(
lambda x: x.rolling(5, min_periods=1).mean().shift(1)
)
team_stats_df.fillna(0, inplace=True)
opponent_strength = team_stats_df[['team_id', 'round', 'attack_strength', 'defence_strength']].rename(
columns={'attack_strength': 'opponent_attack_strength', 'defence_strength': 'opponent_defence_strength'}
)
final_df = pd.merge(all_fpl_players_games, opponent_strength,
left_on=['opponent_team', 'round'], right_on=['team_id', 'round'], how='left')
final_df.drop('team_id', axis=1, inplace=True)
# Fixture difficulty
all_fixtures = FPL.get_fixtures()
fixture_difficulty_map = {
f['id']: {'h': f['team_h_difficulty'], 'a': f['team_a_difficulty']}
for f in all_fixtures
}
def get_difficulty(row):
fid = row['fixture']
if fid in fixture_difficulty_map:
return fixture_difficulty_map[fid]['h'] if row['was_home'] else fixture_difficulty_map[fid]['a']
return 3
final_df['difficulty'] = final_df.apply(get_difficulty, axis=1)
# Rolling stats β€” includes expected_goal_involvements for xG model
stats_to_roll = ['minutes', 'goals_scored', 'assists', 'bps', 'influence', 'creativity',
'threat', 'expected_goals', 'expected_assists', 'expected_goal_involvements',
'expected_goals_conceded', 'total_points']
final_df = final_df.sort_values(by=['element', 'round'])
for stat in stats_to_roll:
final_df[f'{stat}_RA5'] = final_df.groupby('element')[stat].transform(
lambda x: x.rolling(window=5, min_periods=1).mean().shift(1)
)
final_df['total_points_SD5'] = final_df.groupby('element')['total_points'].transform(
lambda x: x.rolling(window=5, min_periods=1).std().shift(1)
)
final_df['total_points_RA10'] = final_df.groupby('element')['total_points'].transform(
lambda x: x.rolling(window=10, min_periods=1).mean().shift(1)
)
final_df['Trend'] = final_df.apply(
lambda row: trend_emoji(row['total_points_RA5'], row['total_points_RA10']), axis=1
)
final_df.fillna(0, inplace=True)
final_df['value'] = final_df['value'] / 10.0
features = ['was_home', 'value', 'difficulty', 'opponent_attack_strength',
'opponent_defence_strength', 'total_points_SD5'] + [f'{stat}_RA5' for stat in stats_to_roll]
xg_features = ['was_home', 'difficulty', 'expected_goals_RA5', 'expected_assists_RA5',
'expected_goal_involvements_RA5', 'expected_goals_conceded_RA5', 'minutes_RA5']
target = 'total_points'
# Train / test split
final_df = final_df.sort_values(by='round')
unique_rounds = final_df['round'].unique()
split_round = unique_rounds[int(len(unique_rounds) * 0.8)]
train_df = final_df[final_df['round'] <= split_round]
test_df = final_df[final_df['round'] > split_round]
X_train, y_train = train_df[features], train_df[target]
X_test, y_test = test_df[features], test_df[target]
# Full-feature models
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train)
X_test_scaled = scaler.transform(X_test)
ridge_model = Ridge()
ridge_model.fit(X_train_scaled, y_train)
xgb_model = XGBRegressor(n_estimators=100, learning_rate=0.1, random_state=42)
xgb_model.fit(X_train_scaled, y_train)
# xG-only model with its own scaler
xg_scaler = StandardScaler()
X_train_xg_scaled = xg_scaler.fit_transform(train_df[xg_features])
X_test_xg_scaled = xg_scaler.transform(test_df[xg_features])
xgb_xg_model = XGBRegressor(n_estimators=100, learning_rate=0.1, random_state=42)
xgb_xg_model.fit(X_train_xg_scaled, y_train)
# Model accuracy metrics on held-out test gameweeks
ridge_rmse = math.sqrt(mean_squared_error(y_test, ridge_model.predict(X_test_scaled).clip(min=0)))
xgb_rmse = math.sqrt(mean_squared_error(y_test, xgb_model.predict(X_test_scaled).clip(min=0)))
xg_rmse = math.sqrt(mean_squared_error(y_test, xgb_xg_model.predict(X_test_xg_scaled).clip(min=0)))
model_metrics_df = pd.DataFrame({
'Model': ['Ridge', 'XGB (Full Features)', 'XGB (xG Only)'],
'Test RMSE': [round(ridge_rmse, 4), round(xgb_rmse, 4), round(xg_rmse, 4)],
'Train Rounds': [int(split_round), int(split_round), int(split_round)],
'Test Rows': [len(test_df), len(test_df), len(test_df)],
})
# Generate next GW predictions
NEXT_GW = current_gameweek + 1
next_gw_fixtures = FPL.get_gameweek_fixtures(NEXT_GW)
prediction_rows = []
if next_gw_fixtures:
team_fixture_map = {}
for f in next_gw_fixtures:
team_fixture_map[f['team_h']] = {'opponent': f['team_a'], 'was_home': True, 'difficulty': f['team_h_difficulty']}
team_fixture_map[f['team_a']] = {'opponent': f['team_h'], 'was_home': False, 'difficulty': f['team_a_difficulty']}
for _, player in all_players.iterrows():
team_id = player['team']
if team_id in team_fixture_map:
fix_info = team_fixture_map[team_id]
prediction_rows.append({
'element': player['id'], 'web_name': player['web_name'], 'team': team_id,
'value': player['now_cost'], 'was_home': fix_info['was_home'],
'opponent_team': fix_info['opponent'], 'difficulty': fix_info['difficulty']
})
# Blank / double GW detection
fixture_count_per_team = {}
if next_gw_fixtures:
for f in next_gw_fixtures:
fixture_count_per_team[f['team_h']] = fixture_count_per_team.get(f['team_h'], 0) + 1
fixture_count_per_team[f['team_a']] = fixture_count_per_team.get(f['team_a'], 0) + 1
def gw_label(count):
if count == 0:
return 'Blank GW'
if count >= 2:
return 'Double GW'
return 'Normal'
gw_status_rows = []
for t in teams:
tid = t['id']
count = fixture_count_per_team.get(tid, 0)
gw_status_rows.append({
'team_id': tid,
'team_name': t['name'],
'fixtures': count,
'GW Status': gw_label(count),
})
gw_status_df = pd.DataFrame(gw_status_rows)
gw_status_sort_key = {'Double GW': 0, 'Normal': 1, 'Blank GW': 2}
gw_status_df['_sort'] = gw_status_df['GW Status'].map(gw_status_sort_key)
gw_status_df = gw_status_df.sort_values(['_sort', 'team_name']).drop(columns='_sort').reset_index(drop=True)
gw_status_map = gw_status_df.set_index('team_id')['GW Status'].to_dict()
prediction_df = pd.DataFrame(prediction_rows)
latest_team_stats = team_stats_df.sort_values('round').groupby('team_id').tail(1)
opponent_stats = latest_team_stats[['team_id', 'attack_strength', 'defence_strength']].rename(
columns={'team_id': 'opponent_team', 'attack_strength': 'opponent_attack_strength',
'defence_strength': 'opponent_defence_strength'}
)
prediction_df = pd.merge(prediction_df, opponent_stats, on='opponent_team', how='left')
prediction_df.fillna(0, inplace=True)
# Latest rolling stats per player
player_stats_latest = final_df.sort_values(by=['element', 'round']).copy()
for stat in stats_to_roll:
player_stats_latest[f'{stat}_RA5_latest'] = player_stats_latest.groupby('element')[stat].transform(
lambda x: x.rolling(window=5, min_periods=1).mean()
)
player_stats_latest['total_points_SD5_latest'] = player_stats_latest.groupby('element')['total_points'].transform(
lambda x: x.rolling(window=5, min_periods=1).std()
)
player_stats_latest['total_points_RA10_latest'] = player_stats_latest.groupby('element')['total_points'].transform(
lambda x: x.rolling(window=10, min_periods=1).mean()
)
player_stats_latest['Trend'] = player_stats_latest.apply(
lambda row: trend_emoji(row['total_points_RA5_latest'], row['total_points_RA10_latest']), axis=1
)
player_stats_latest = player_stats_latest.groupby('element').tail(1)
cols_to_merge = ['element', 'total_points_SD5_latest', 'Trend'] + [f'{stat}_RA5_latest' for stat in stats_to_roll]
prediction_df = pd.merge(prediction_df, player_stats_latest[cols_to_merge], on='element', how='left')
rename_dict = {f'{stat}_RA5_latest': f'{stat}_RA5' for stat in stats_to_roll}
rename_dict['total_points_SD5_latest'] = 'total_points_SD5'
prediction_df.rename(columns=rename_dict, inplace=True)
prediction_df['value'] = prediction_df['value'] / 10.0
prediction_df.fillna(0, inplace=True)
# Run all three models
X_pred_scaled = scaler.transform(prediction_df[features])
prediction_df['Ridge Pred'] = ridge_model.predict(X_pred_scaled).clip(min=0)
prediction_df['XGB Pred'] = xgb_model.predict(X_pred_scaled).clip(min=0)
prediction_df['xG Pred'] = xgb_xg_model.predict(
xg_scaler.transform(prediction_df[xg_features])
).clip(min=0)
# Assemble final predictions dataframe
final_predictions_df = prediction_df.copy()
final_predictions_df['player_name'] = final_predictions_df['web_name']
final_predictions_df['position'] = final_predictions_df['element'].map(player_type_map).map(POSITION_MAP)
final_predictions_df['team_name'] = final_predictions_df['team'].map(team_name_map)
final_predictions_df['Next Opponent'] = final_predictions_df['opponent_team'].map(team_name_map)
final_predictions_df['GW Status'] = final_predictions_df['team'].map(gw_status_map).fillna('Normal')
final_predictions_df['ownership'] = final_predictions_df['element'].map(ownership_map).astype(float)
cols_from_players = all_players[['id', 'total_points', 'event_points']].rename(
columns={'id': 'element', 'total_points': 'Total Pts', 'event_points': 'Last GW Pts'}
)
for col in ['Total Pts', 'Last GW Pts']:
if col in final_predictions_df.columns:
final_predictions_df.drop(columns=[col], inplace=True)
final_predictions_df = pd.merge(final_predictions_df, cols_from_players, on='element', how='left')
final_predictions_df.rename(columns={
'total_points_RA5': 'Pts Mean (5GW)',
'total_points_SD5': 'Pts Std Dev (5GW)',
}, inplace=True)
for col in ['Ridge Pred', 'XGB Pred', 'xG Pred', 'Pts Mean (5GW)', 'Pts Std Dev (5GW)']:
if col in final_predictions_df.columns:
final_predictions_df[col] = final_predictions_df[col].astype(float).round(2)
# Consistency label β€” CV < 0.4 = High, 0.4–0.75 = Medium, > 0.75 = Low
cv = (final_predictions_df['Pts Std Dev (5GW)'] /
final_predictions_df['Pts Mean (5GW)'].replace(0, np.nan))
final_predictions_df['Consistency'] = pd.cut(
cv.fillna(1.0),
bins=[-np.inf, 0.4, 0.75, np.inf],
labels=['High', 'Medium', 'Low']
).astype(str)
# Differential score β€” how much predicted output per % of managers who own the player
final_predictions_df['Differential Score'] = (
final_predictions_df['XGB Pred'] /
final_predictions_df['ownership'].replace(0, np.nan)
).fillna(0).round(4)
# Team Performance
team_performance_df = team_stats_df.groupby('team_id').agg(
overall_goals_scored=('goals_scored', 'sum'),
overall_goals_conceded=('goals_conceded', 'sum')
).reset_index()
team_performance_df['team_name'] = team_performance_df['team_id'].map(team_name_map)
team_performance_df['goals_scored_per_game'] = team_performance_df['overall_goals_scored'] / current_gameweek
team_performance_df['goals_conceded_per_game'] = team_performance_df['overall_goals_conceded'] / current_gameweek
team_performance_df['attack_rank'] = team_performance_df['goals_scored_per_game'].rank(ascending=False)
team_performance_df['defense_rank'] = team_performance_df['goals_conceded_per_game'].rank(ascending=True)
# Manager Team
try:
manager_picks = FPL.get_manager_team(gameweek=current_gameweek)
if manager_picks and 'picks' in manager_picks:
picks_df = pd.DataFrame(manager_picks['picks'])
manager_players_df = picks_df.copy()
manager_players_df['player_name'] = manager_players_df['element'].map(player_name_map).fillna('Unknown')
manager_players_df['team_name'] = (
manager_players_df['element'].map(player_team_map).map(team_name_map).fillna('Unknown')
)
manager_players_df['position'] = (
manager_players_df['element'].map(player_type_map).map(POSITION_MAP).fillna('Unknown')
)
# Defensive check for selling_price
if 'selling_price' in picks_df.columns:
manager_players_df['sell_price'] = picks_df['selling_price'] / 10.0
elif 'purchase_price' in picks_df.columns:
manager_players_df['sell_price'] = picks_df['purchase_price'] / 10.0
else:
manager_players_df['sell_price'] = 0.0
manager_players_df = pd.merge(
manager_players_df,
final_predictions_df[['element', 'XGB Pred', 'xG Pred', 'Total Pts', 'Last GW Pts',
'difficulty', 'Next Opponent', 'GW Status']],
on='element', how='left'
)
else:
manager_players_df = pd.DataFrame(
columns=['element', 'player_name', 'team_name', 'position', 'sell_price', 'XGB Pred', 'xG Pred']
)
except Exception as e:
print(f"Could not fetch manager team: {e}")
manager_players_df = pd.DataFrame(
columns=['element', 'player_name', 'team_name', 'position', 'sell_price', 'XGB Pred', 'xG Pred']
)
return (
final_predictions_df, team_performance_df, manager_players_df,
current_gameweek, NEXT_GW, gw_status_df, model_metrics_df
)
# Load data at startup
print("Loading FPL data and training models...")
(final_predictions_df, team_performance_df, manager_players_df,
current_gameweek, NEXT_GW, gw_status_df, model_metrics_df) = load_and_process_data()
print("Data loaded successfully!")
def refresh_data():
global final_predictions_df, team_performance_df, manager_players_df
global current_gameweek, NEXT_GW, gw_status_df, model_metrics_df
print("Refreshing FPL data...")
new_data = load_and_process_data()
with _data_lock:
(final_predictions_df, team_performance_df, manager_players_df,
current_gameweek, NEXT_GW, gw_status_df, model_metrics_df) = new_data
print("Data refreshed!")
return f"Data refreshed β€” GW {current_gameweek} loaded, predicting GW {NEXT_GW}."
def refresh_and_reload():
"""Refresh data and repopulate all auto-loadable tab outputs with their defaults."""
status = refresh_data()
return (
status,
get_top_predictions(),
get_team_stats(),
get_value_picks(),
get_my_team(),
get_captaincy_picks(),
get_ownership_differentials(),
get_gw_status(),
get_model_accuracy(),
)
# ── MCP Tool Functions ────────────────────────────────────────────────────────
# All functions are named (no lambdas) so Gradio MCP exposes them with proper
# tool names and docstrings that Claude can read and use.
def get_top_predictions(top_n: int = 10, position: str = "") -> pd.DataFrame:
"""
Return the top N FPL players predicted to score the most points in the next gameweek,
ranked by XGBoost model prediction. Optionally filter to a single position.
Args:
top_n (int): How many players to return (default 10, max 50).
position (str): Position to filter by β€” one of: Goalkeeper, Defender, Midfielder, Forward.
Leave blank to include all positions.
Returns:
pd.DataFrame: Players ranked by predicted points, with price, form, trend, GW status,
and next opponent.
"""
pos = position.strip() or None
if pos and pos not in VALID_POSITIONS:
return pd.DataFrame({"error": [f"Invalid position '{pos}'. Choose from: {', '.join(sorted(VALID_POSITIONS))}"]})
with _data_lock:
df = final_predictions_df.copy()
if pos:
df = df[df['position'] == pos]
df = df.sort_values('XGB Pred', ascending=False).head(top_n)
cols = ['player_name', 'team_name', 'position', 'value', 'XGB Pred', 'xG Pred',
'Total Pts', 'Last GW Pts', 'Pts Mean (5GW)', 'Trend', 'Consistency',
'GW Status', 'difficulty', 'Next Opponent']
return df[[c for c in cols if c in df.columns]]
def get_player(player_name: str) -> pd.DataFrame:
"""
Get detailed FPL information for a single player by name or partial name.
Includes predicted points, current price, season total, recent form,
consistency, and next fixture details.
Args:
player_name (str): The name or partial name of the player (e.g., 'Salah', 'Haaland').
Returns:
pd.DataFrame: Detailed stats for the matched player(s).
"""
with _data_lock:
df = final_predictions_df.copy()
hits = df[df['player_name'].str.contains(player_name, case=False, na=False)]
if hits.empty:
return pd.DataFrame({"error": [f"No player found matching '{player_name}'"]})
cols = ['player_name', 'team_name', 'position', 'value', 'XGB Pred', 'xG Pred',
'Ridge Pred', 'Total Pts', 'Last GW Pts', 'Pts Mean (5GW)',
'Pts Std Dev (5GW)', 'Consistency', 'Trend', 'GW Status', 'difficulty', 'Next Opponent', 'ownership']
return hits[[c for c in cols if c in df.columns]]
def compare_players(player1_name: str, player2_name: str, player3_name: str = "") -> pd.DataFrame:
"""
Compare two or three FPL players side-by-side: predicted points, price, season total,
recent form, consistency (std dev), fixture difficulty, and next opponent.
Partial name matching is supported (e.g. "Salah" matches "Mohamed Salah").
Args:
player1_name (str): Name or partial name of the first player.
player2_name (str): Name or partial name of the second player.
player3_name (str): Optional name or partial name of a third player.
Returns:
pd.DataFrame: Side-by-side stats for the matched players.
"""
with _data_lock:
df = final_predictions_df.copy()
names = [player1_name, player2_name]
if player3_name.strip():
names.append(player3_name.strip())
matched = []
for name in names:
hits = df[df['player_name'].str.contains(name, case=False, na=False)]
if not hits.empty:
matched.append(hits.iloc[0])
if not matched:
return pd.DataFrame({"error": ["No players found β€” try a partial surname like 'Salah' or 'Haaland'"]})
cols = ['player_name', 'team_name', 'position', 'value', 'XGB Pred', 'xG Pred',
'Ridge Pred', 'Total Pts', 'Last GW Pts', 'Pts Mean (5GW)',
'Pts Std Dev (5GW)', 'Consistency', 'Trend', 'GW Status', 'difficulty', 'Next Opponent']
return pd.DataFrame(matched)[[c for c in cols if c in df.columns]]
def get_captaincy_picks(top_n: int = 5, position: str = "") -> pd.DataFrame:
"""
Rank the best captaincy options for the next gameweek. A captain's points are doubled,
so prioritise players with high predicted points AND high consistency (low variance).
Includes an xG-based prediction as a second opinion for each candidate.
Args:
top_n (int): Number of captaincy candidates to return (default 5).
position (str): Filter by position β€” Goalkeeper, Defender, Midfielder, Forward.
Leave blank for all positions (attackers/midfielders are usually best).
Returns:
pd.DataFrame: Top candidates ranked by XGB Pred, with consistency rating,
xG prediction, fixture difficulty, and next opponent.
"""
pos = position.strip() or None
if pos and pos not in VALID_POSITIONS:
return pd.DataFrame({"error": [f"Invalid position '{pos}'. Choose from: {', '.join(sorted(VALID_POSITIONS))}"]})
with _data_lock:
df = final_predictions_df.copy()
if pos:
df = df[df['position'] == pos]
df = df.sort_values('XGB Pred', ascending=False).head(top_n)
cols = ['player_name', 'team_name', 'position', 'value', 'XGB Pred', 'xG Pred',
'Pts Mean (5GW)', 'Pts Std Dev (5GW)', 'Consistency', 'Trend',
'GW Status', 'difficulty', 'Next Opponent']
return df[[c for c in cols if c in df.columns]]
def get_transfer_suggestions(budget_extra: float = 0.0, top_n: int = 3) -> pd.DataFrame:
"""
For each player in the manager's squad, find the best available transfer targets
at the same position that are affordable within (sell price + budget_extra Β£m).
Results are ranked by predicted points gain, making it easy to spot the highest-impact transfer.
Args:
budget_extra (float): Additional budget in Β£m above each player's sell price (default 0.0).
Set to your free transfer budget to find realistic moves.
top_n (int): Maximum number of suggestions to return per squad player (default 3).
Returns:
pd.DataFrame: Transfer options sorted by points gain, showing current vs. suggested player,
prices, predicted points, and next fixture.
"""
with _data_lock:
squad = manager_players_df.copy()
pool = final_predictions_df.copy()
if squad.empty or 'sell_price' not in squad.columns:
return pd.DataFrame({"error": ["Squad data unavailable β€” check FPL_MANAGER_ID is set correctly."]})
squad_ids = set(squad['element'].tolist())
rows = []
for _, player in squad.iterrows():
if pd.isna(player.get('sell_price')):
continue
budget = float(player['sell_price']) + budget_extra
pos = player['position']
current_xgb = float(player.get('XGB Pred', 0))
alternatives = pool[
(pool['position'] == pos) &
(pool['value'] <= budget) &
(~pool['element'].isin(squad_ids))
].sort_values('XGB Pred', ascending=False).head(top_n)
for _, alt in alternatives.iterrows():
rows.append({
'Out': player['player_name'],
'Out Price (Β£m)': round(float(player['sell_price']), 1),
'Out XGB Pred': round(current_xgb, 2),
'In': alt['player_name'],
'In Team': alt['team_name'],
'In Price (Β£m)': round(float(alt['value']), 1),
'In XGB Pred': round(float(alt['XGB Pred']), 2),
'In xG Pred': round(float(alt['xG Pred']), 2),
'Points Gain': round(float(alt['XGB Pred']) - current_xgb, 2),
'Cost Diff (Β£m)': round(float(alt['value']) - float(player['sell_price']), 1),
'Position': pos,
'GW Status': alt.get('GW Status', ''),
'Next Opponent': alt.get('Next Opponent', ''),
})
if not rows:
return pd.DataFrame({"message": ["No upgrades found within budget. Try increasing budget_extra."]})
return (pd.DataFrame(rows)
.sort_values('Points Gain', ascending=False)
.reset_index(drop=True))
def get_ownership_differentials(max_ownership: float = 10.0, min_predicted: float = 4.0, top_n: int = 15) -> pd.DataFrame:
"""
Find low-ownership players with high predicted points β€” the best differentials to gain
rank on your mini-league rivals. Differential Score = XGB Pred / ownership %, so players
who are both highly predicted and thinly owned rank highest.
Args:
max_ownership (float): Maximum ownership % to qualify as a differential (default 10.0).
min_predicted (float): Minimum XGB Pred points to filter out noise (default 4.0).
top_n (int): Number of results to return (default 15).
Returns:
pd.DataFrame: Differentials ranked by Differential Score, with ownership %,
both model predictions, form, and fixture info.
"""
with _data_lock:
df = final_predictions_df.copy()
df = df[
(df['ownership'] <= max_ownership) &
(df['XGB Pred'] >= min_predicted)
]
df = df.sort_values('Differential Score', ascending=False).head(top_n)
cols = ['player_name', 'team_name', 'position', 'value', 'ownership', 'XGB Pred', 'xG Pred',
'Differential Score', 'Pts Mean (5GW)', 'Consistency', 'Trend',
'GW Status', 'difficulty', 'Next Opponent']
return df[[c for c in cols if c in df.columns]]
def get_team_stats(team_name: str = "") -> pd.DataFrame:
"""
Return Premier League team performance stats: goals scored and conceded per game,
plus attack and defense rankings across the season so far.
Useful for assessing fixture difficulty and identifying teams to target or avoid.
Args:
team_name (str): Team name or partial name to filter (e.g. 'Arsenal', 'City').
Leave blank to return all 20 teams sorted by attack rank.
Returns:
pd.DataFrame: Goals per game, attack rank, and defence rank for each team.
"""
with _data_lock:
df = team_performance_df.copy()
name = team_name.strip()
if name:
df = df[df['team_name'].str.contains(name, case=False, na=False)]
df = df.sort_values('attack_rank')
cols = ['team_name', 'goals_scored_per_game', 'goals_conceded_per_game', 'attack_rank', 'defense_rank']
return df[cols]
def get_value_picks(max_price: float = 10.0, min_predicted: float = 4.0, top_n: int = 10) -> pd.DataFrame:
"""
Find the best-value FPL players: high predicted points within a price cap.
Useful for identifying budget options or cheap enablers with good fixtures.
Args:
max_price (float): Maximum player price in Β£m (default 10.0).
min_predicted (float): Minimum XGBoost predicted points to qualify (default 4.0).
top_n (int): Number of players to return (default 10).
Returns:
pd.DataFrame: Budget players ranked by predicted points with form and fixture info.
"""
with _data_lock:
df = final_predictions_df.copy()
df = df[(df['value'] <= max_price) & (df['XGB Pred'] >= min_predicted)]
df = df.sort_values('XGB Pred', ascending=False).head(top_n)
cols = ['player_name', 'team_name', 'position', 'value', 'XGB Pred', 'xG Pred',
'Total Pts', 'Last GW Pts', 'Pts Mean (5GW)', 'Consistency', 'Trend',
'GW Status', 'difficulty', 'Next Opponent']
return df[[c for c in cols if c in df.columns]]
def get_my_team() -> pd.DataFrame:
"""
Return the next-gameweek predictions for the configured manager's current FPL squad.
Shows both XGB and xG predictions, fixture difficulty, GW type, and next opponent.
Returns:
pd.DataFrame: Manager's squad with predictions and fixture info.
"""
with _data_lock:
return manager_players_df.copy()
def get_gw_status() -> pd.DataFrame:
"""
Show which Premier League teams have a Blank, Normal, or Double Gameweek next round.
Use this before buying or selling players β€” avoid blanking players and target doubles.
Returns:
pd.DataFrame: All 20 teams with their fixture count and GW type label,
sorted with Double GW teams first, Blank GW teams last.
"""
with _data_lock:
df = gw_status_df.copy()
return df[['team_name', 'fixtures', 'GW Status']]
def get_model_accuracy() -> pd.DataFrame:
"""
Return the RMSE of all three trained prediction models, evaluated on held-out test
gameweeks (the most recent 20% of the season). Lower RMSE = better point predictions.
Compare Ridge, full-feature XGB, and xG-only XGB to understand which to trust most.
Returns:
pd.DataFrame: Model name, test RMSE, training rounds used, and test row count.
"""
with _data_lock:
return model_metrics_df.copy()
# ── Gradio Interface ──────────────────────────────────────────────────────────
with gr.Blocks(title=f"FPL GW {NEXT_GW} Predictor (MCP)") as demo:
gr.Markdown(f"# FPL Gameweek {NEXT_GW} Point Predictor")
gr.Markdown("""
**MCP-Enabled**: This app exposes prediction tools for Claude Desktop!
Use the tabs below or connect via Claude Desktop to query FPL data.
""")
with gr.Row():
refresh_btn = gr.Button("Refresh Data", variant="secondary")
refresh_status = gr.Textbox(label="", interactive=False, show_label=False)
with gr.Tab("Top Predictions"):
with gr.Row():
top_n_input = gr.Slider(5, 50, value=10, step=5, label="Number of Players")
position_input = gr.Dropdown(
choices=["", "Goalkeeper", "Defender", "Midfielder", "Forward"],
label="Position Filter", value=""
)
top_pred_btn = gr.Button("Get Top Predictions")
top_pred_output = gr.Dataframe()
top_pred_btn.click(fn=get_top_predictions, inputs=[top_n_input, position_input], outputs=top_pred_output)
with gr.Tab("Player Search"):
player_search_input = gr.Textbox(label="Player Name", placeholder="e.g. Salah")
player_search_btn = gr.Button("Search Player")
player_search_output = gr.Dataframe()
player_search_btn.click(fn=get_player, inputs=player_search_input, outputs=player_search_output)
with gr.Tab("Captaincy Picks"):
with gr.Row():
cap_top_n = gr.Slider(3, 20, value=5, step=1, label="Top N Candidates")
cap_position = gr.Dropdown(
choices=["", "Goalkeeper", "Defender", "Midfielder", "Forward"],
label="Position Filter", value=""
)
cap_btn = gr.Button("Get Captaincy Picks")
cap_output = gr.Dataframe()
cap_btn.click(fn=get_captaincy_picks, inputs=[cap_top_n, cap_position], outputs=cap_output)
with gr.Tab("Compare Players"):
player1 = gr.Textbox(label="Player 1", value="Salah")
player2 = gr.Textbox(label="Player 2", value="Haaland")
player3 = gr.Textbox(label="Player 3 (Optional)", value="")
compare_btn = gr.Button("Compare Players")
compare_output = gr.Dataframe()
compare_btn.click(fn=compare_players, inputs=[player1, player2, player3], outputs=compare_output)
with gr.Tab("My Team"):
mgr_btn = gr.Button("Load My Team")
mgr_output = gr.Dataframe()
mgr_btn.click(fn=get_my_team, outputs=mgr_output)
with gr.Tab("Transfer Suggester"):
with gr.Row():
budget_extra_input = gr.Slider(0.0, 5.0, value=0.0, step=0.5, label="Extra Budget Β£m (above each player's sell price)")
transfer_top_n = gr.Slider(1, 5, value=3, step=1, label="Max alternatives per player")
transfer_btn = gr.Button("Find Transfer Suggestions")
transfer_output = gr.Dataframe()
transfer_btn.click(fn=get_transfer_suggestions, inputs=[budget_extra_input, transfer_top_n], outputs=transfer_output)
with gr.Tab("Budget Picks"):
with gr.Row():
max_price = gr.Slider(4.0, 15.0, value=8.0, step=0.5, label="Max Price (Β£m)")
min_pred = gr.Slider(0.0, 10.0, value=4.0, step=0.5, label="Min Predicted Points")
val_top_n = gr.Slider(5, 30, value=10, step=5, label="Results")
val_btn = gr.Button("Find Budget Picks")
val_output = gr.Dataframe()
val_btn.click(fn=get_value_picks, inputs=[max_price, min_pred, val_top_n], outputs=val_output)
with gr.Tab("Ownership Differentials"):
with gr.Row():
own_max = gr.Slider(1.0, 30.0, value=10.0, step=1.0, label="Max Ownership %")
own_min_pred = gr.Slider(0.0, 10.0, value=4.0, step=0.5, label="Min XGB Predicted Pts")
own_top_n = gr.Slider(5, 30, value=15, step=5, label="Results")
own_btn = gr.Button("Find Ownership Differentials")
own_output = gr.Dataframe()
own_btn.click(fn=get_ownership_differentials, inputs=[own_max, own_min_pred, own_top_n], outputs=own_output)
with gr.Tab("Team Stats"):
team_input = gr.Textbox(label="Team Name (optional)", value="")
team_btn = gr.Button("Get Team Stats")
team_output = gr.Dataframe()
team_btn.click(fn=get_team_stats, inputs=team_input, outputs=team_output)
with gr.Tab("GW Fixture Status"):
gw_btn = gr.Button("Show GW Status")
gw_output = gr.Dataframe()
gw_btn.click(fn=get_gw_status, outputs=gw_output)
with gr.Tab("Model Accuracy"):
acc_btn = gr.Button("Show Model Accuracy")
acc_output = gr.Dataframe()
acc_btn.click(fn=get_model_accuracy, outputs=acc_output)
# Bound after all outputs exist β€” repopulates all auto-loadable tabs on refresh
refresh_btn.click(
fn=refresh_and_reload,
outputs=[
refresh_status,
top_pred_output,
team_output,
val_output,
mgr_output,
cap_output,
own_output,
gw_output,
acc_output,
],
)
if __name__ == "__main__":
demo.launch(mcp_server=True)