ArmBench-LLM / data_handler.py
Zaruhi's picture
bug fixes
3bfe826
Raw
History Blame Contribute Delete
5.77 kB
import pandas as pd
from model_handler import TASK_CATEGORIES, TASK_DISPLAY_NAMES
# Column labels for display
COLUMN_LABELS = {
"model_name": "Model",
"NER": "NER",
"POS": "POS",
"Reading Comprehension": "Reading",
"Classification": "Classification",
"MCQA": "MCQA",
"Generation": "Generation",
"Translation": "Translation",
"Exams": "Exams",
"Text Processing": "Text Proc.",
"MMLU": "MMLU",
"Average": "Average",
}
# Category columns for computing average
CATEGORY_COLS = list(TASK_CATEGORIES.keys())
def format_model_name(full_model_name: str) -> str:
"""Keep the full model name with org/model format.
Args:
full_model_name: Full model identifier like 'google/gemini-2.5-pro'
Returns:
The same model name with org/model format
"""
return full_model_name
def get_task_scores(detailed_results: dict) -> dict:
"""Calculate average scores for each task across all models.
Returns:
dict: Maps task display name to average score
"""
tasks_df = detailed_results.get("tasks", pd.DataFrame())
if tasks_df.empty:
return {}
task_scores = {}
for task_key, display_name in TASK_DISPLAY_NAMES.items():
if display_name in tasks_df.columns:
# Calculate average for this task
avg_score = tasks_df[display_name].mean()
if not pd.isna(avg_score):
task_scores[display_name] = avg_score
return task_scores
def prepare_leaderboard(df: pd.DataFrame) -> pd.DataFrame:
"""Prepare LLM benchmark leaderboard from raw results DataFrame."""
if df.empty:
return df
df = df.copy()
# Format model names
df["model_name"] = df["model_name"].apply(format_model_name)
# Calculate overall average if not present
available_cols = [c for c in CATEGORY_COLS if c in df.columns]
if available_cols and "Average" not in df.columns:
df["Average"] = df[available_cols].mean(axis=1)
# Sort by average
if "Average" in df.columns:
df = df.sort_values(by="Average", ascending=False).reset_index(drop=True)
df.insert(0, "Rank", range(1, len(df) + 1))
# Select columns for display (Average first, then categories)
display_cols = ["Rank", "model_name", "Size", "Average"] + available_cols
df = df[[c for c in display_cols if c in df.columns]]
# Round numeric columns
df = df.round(4)
# Rename columns for display
df = df.rename(columns=COLUMN_LABELS)
return df
def prepare_detailed_leaderboard(
detailed_results: dict,
leaderboard_df: pd.DataFrame = None,
use_multiindex: bool = True,
) -> pd.DataFrame:
"""Prepare detailed task-level leaderboard with hierarchical columns.
Args:
detailed_results: Dict with 'tasks' DataFrame from ModelHandler.get_detailed_results()
leaderboard_df: Optional leaderboard DataFrame to match model order
use_multiindex: If True, return DataFrame with MultiIndex columns for proper
hierarchical display (merged headers in HTML/Gradio).
Returns:
pd.DataFrame: Combined table with category names as hierarchical column headers
"""
tasks_df = detailed_results.get("tasks", pd.DataFrame())
if tasks_df.empty:
return pd.DataFrame()
# Format model names
tasks_df = tasks_df.copy()
tasks_df["model_name"] = tasks_df["model_name"].apply(format_model_name)
# Build combined dataframe with hierarchical columns
combined = tasks_df[["model_name"]].copy().rename(columns={"model_name": "Model"})
column_tuples = [("", "Model")]
# Group tasks by category
for category, task_keys in TASK_CATEGORIES.items():
for task_key in task_keys:
display_name = TASK_DISPLAY_NAMES.get(task_key, task_key)
if display_name in tasks_df.columns:
col_name = f"{category} | {display_name}"
column_tuples.append((category, display_name))
combined[col_name] = tasks_df[display_name]
# Round numeric columns
combined = combined.round(4)
# Sort by leaderboard order if provided, otherwise by row average
if leaderboard_df is not None and "Model" in leaderboard_df.columns:
# Extract model names from leaderboard (skip "Rank" column)
model_order = leaderboard_df["Model"].tolist()
# Create a mapping of model names to their rank
model_rank = {name: idx for idx, name in enumerate(model_order)}
# Sort combined dataframe by leaderboard order
combined["_sort_rank"] = combined["Model"].map(model_rank)
combined = combined.sort_values(by="_sort_rank", na_position="last")
combined = combined.drop(columns=["_sort_rank"])
else:
# Fallback: sort by average of category columns
category_col_names = []
for category, task_keys in TASK_CATEGORIES.items():
for task_key in task_keys:
display_name = TASK_DISPLAY_NAMES.get(task_key, task_key)
col_name = f"{category} | {display_name}"
if col_name in combined.columns:
category_col_names.append(col_name)
if category_col_names:
combined["_sort_avg"] = combined[category_col_names].mean(axis=1)
combined = combined.sort_values(
by="_sort_avg", ascending=False, na_position="last"
)
combined = combined.drop(columns=["_sort_avg"])
combined = combined.reset_index(drop=True)
combined.insert(0, "#", range(1, len(combined) + 1))
column_tuples.insert(0, ("", "#"))
if use_multiindex:
combined.columns = pd.MultiIndex.from_tuples(column_tuples)
return combined