Spaces:
Running
Running
File size: 5,767 Bytes
cafe7ea f038138 cafe7ea | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 | import pandas as pd
from model_handler import TASK_CATEGORIES, TASK_DISPLAY_NAMES
# Column labels for display
COLUMN_LABELS = {
"model_name": "Model",
"NER": "NER",
"POS": "POS",
"Reading Comprehension": "Reading",
"Classification": "Classification",
"MCQA": "MCQA",
"Generation": "Generation",
"Translation": "Translation",
"Exams": "Exams",
"Text Processing": "Text Proc.",
"MMLU": "MMLU",
"Average": "Average",
}
# Category columns for computing average
CATEGORY_COLS = list(TASK_CATEGORIES.keys())
def format_model_name(full_model_name: str) -> str:
"""Keep the full model name with org/model format.
Args:
full_model_name: Full model identifier like 'google/gemini-2.5-pro'
Returns:
The same model name with org/model format
"""
return full_model_name
def get_task_scores(detailed_results: dict) -> dict:
"""Calculate average scores for each task across all models.
Returns:
dict: Maps task display name to average score
"""
tasks_df = detailed_results.get("tasks", pd.DataFrame())
if tasks_df.empty:
return {}
task_scores = {}
for task_key, display_name in TASK_DISPLAY_NAMES.items():
if display_name in tasks_df.columns:
# Calculate average for this task
avg_score = tasks_df[display_name].mean()
if not pd.isna(avg_score):
task_scores[display_name] = avg_score
return task_scores
def prepare_leaderboard(df: pd.DataFrame) -> pd.DataFrame:
"""Prepare LLM benchmark leaderboard from raw results DataFrame."""
if df.empty:
return df
df = df.copy()
# Format model names
df["model_name"] = df["model_name"].apply(format_model_name)
# Calculate overall average if not present
available_cols = [c for c in CATEGORY_COLS if c in df.columns]
if available_cols and "Average" not in df.columns:
df["Average"] = df[available_cols].mean(axis=1)
# Sort by average
if "Average" in df.columns:
df = df.sort_values(by="Average", ascending=False).reset_index(drop=True)
df.insert(0, "Rank", range(1, len(df) + 1))
# Select columns for display (Average first, then categories)
display_cols = ["Rank", "model_name", "Size", "Average"] + available_cols
df = df[[c for c in display_cols if c in df.columns]]
# Round numeric columns
df = df.round(4)
# Rename columns for display
df = df.rename(columns=COLUMN_LABELS)
return df
def prepare_detailed_leaderboard(
detailed_results: dict,
leaderboard_df: pd.DataFrame = None,
use_multiindex: bool = True,
) -> pd.DataFrame:
"""Prepare detailed task-level leaderboard with hierarchical columns.
Args:
detailed_results: Dict with 'tasks' DataFrame from ModelHandler.get_detailed_results()
leaderboard_df: Optional leaderboard DataFrame to match model order
use_multiindex: If True, return DataFrame with MultiIndex columns for proper
hierarchical display (merged headers in HTML/Gradio).
Returns:
pd.DataFrame: Combined table with category names as hierarchical column headers
"""
tasks_df = detailed_results.get("tasks", pd.DataFrame())
if tasks_df.empty:
return pd.DataFrame()
# Format model names
tasks_df = tasks_df.copy()
tasks_df["model_name"] = tasks_df["model_name"].apply(format_model_name)
# Build combined dataframe with hierarchical columns
combined = tasks_df[["model_name"]].copy().rename(columns={"model_name": "Model"})
column_tuples = [("", "Model")]
# Group tasks by category
for category, task_keys in TASK_CATEGORIES.items():
for task_key in task_keys:
display_name = TASK_DISPLAY_NAMES.get(task_key, task_key)
if display_name in tasks_df.columns:
col_name = f"{category} | {display_name}"
column_tuples.append((category, display_name))
combined[col_name] = tasks_df[display_name]
# Round numeric columns
combined = combined.round(4)
# Sort by leaderboard order if provided, otherwise by row average
if leaderboard_df is not None and "Model" in leaderboard_df.columns:
# Extract model names from leaderboard (skip "Rank" column)
model_order = leaderboard_df["Model"].tolist()
# Create a mapping of model names to their rank
model_rank = {name: idx for idx, name in enumerate(model_order)}
# Sort combined dataframe by leaderboard order
combined["_sort_rank"] = combined["Model"].map(model_rank)
combined = combined.sort_values(by="_sort_rank", na_position="last")
combined = combined.drop(columns=["_sort_rank"])
else:
# Fallback: sort by average of category columns
category_col_names = []
for category, task_keys in TASK_CATEGORIES.items():
for task_key in task_keys:
display_name = TASK_DISPLAY_NAMES.get(task_key, task_key)
col_name = f"{category} | {display_name}"
if col_name in combined.columns:
category_col_names.append(col_name)
if category_col_names:
combined["_sort_avg"] = combined[category_col_names].mean(axis=1)
combined = combined.sort_values(
by="_sort_avg", ascending=False, na_position="last"
)
combined = combined.drop(columns=["_sort_avg"])
combined = combined.reset_index(drop=True)
combined.insert(0, "#", range(1, len(combined) + 1))
column_tuples.insert(0, ("", "#"))
if use_multiindex:
combined.columns = pd.MultiIndex.from_tuples(column_tuples)
return combined
|