Spaces:
Running
Running
| import pandas as pd | |
| from model_handler import TASK_CATEGORIES, TASK_DISPLAY_NAMES | |
| # Column labels for display | |
| COLUMN_LABELS = { | |
| "model_name": "Model", | |
| "NER": "NER", | |
| "POS": "POS", | |
| "Reading Comprehension": "Reading", | |
| "Classification": "Classification", | |
| "MCQA": "MCQA", | |
| "Generation": "Generation", | |
| "Translation": "Translation", | |
| "Exams": "Exams", | |
| "Text Processing": "Text Proc.", | |
| "MMLU": "MMLU", | |
| "Average": "Average", | |
| } | |
| # Category columns for computing average | |
| CATEGORY_COLS = list(TASK_CATEGORIES.keys()) | |
| def format_model_name(full_model_name: str) -> str: | |
| """Keep the full model name with org/model format. | |
| Args: | |
| full_model_name: Full model identifier like 'google/gemini-2.5-pro' | |
| Returns: | |
| The same model name with org/model format | |
| """ | |
| return full_model_name | |
| def get_task_scores(detailed_results: dict) -> dict: | |
| """Calculate average scores for each task across all models. | |
| Returns: | |
| dict: Maps task display name to average score | |
| """ | |
| tasks_df = detailed_results.get("tasks", pd.DataFrame()) | |
| if tasks_df.empty: | |
| return {} | |
| task_scores = {} | |
| for task_key, display_name in TASK_DISPLAY_NAMES.items(): | |
| if display_name in tasks_df.columns: | |
| # Calculate average for this task | |
| avg_score = tasks_df[display_name].mean() | |
| if not pd.isna(avg_score): | |
| task_scores[display_name] = avg_score | |
| return task_scores | |
| def prepare_leaderboard(df: pd.DataFrame) -> pd.DataFrame: | |
| """Prepare LLM benchmark leaderboard from raw results DataFrame.""" | |
| if df.empty: | |
| return df | |
| df = df.copy() | |
| # Format model names | |
| df["model_name"] = df["model_name"].apply(format_model_name) | |
| # Calculate overall average if not present | |
| available_cols = [c for c in CATEGORY_COLS if c in df.columns] | |
| if available_cols and "Average" not in df.columns: | |
| df["Average"] = df[available_cols].mean(axis=1) | |
| # Sort by average | |
| if "Average" in df.columns: | |
| df = df.sort_values(by="Average", ascending=False).reset_index(drop=True) | |
| df.insert(0, "Rank", range(1, len(df) + 1)) | |
| # Select columns for display (Average first, then categories) | |
| display_cols = ["Rank", "model_name", "Size", "Average"] + available_cols | |
| df = df[[c for c in display_cols if c in df.columns]] | |
| # Round numeric columns | |
| df = df.round(4) | |
| # Rename columns for display | |
| df = df.rename(columns=COLUMN_LABELS) | |
| return df | |
| def prepare_detailed_leaderboard( | |
| detailed_results: dict, | |
| leaderboard_df: pd.DataFrame = None, | |
| use_multiindex: bool = True, | |
| ) -> pd.DataFrame: | |
| """Prepare detailed task-level leaderboard with hierarchical columns. | |
| Args: | |
| detailed_results: Dict with 'tasks' DataFrame from ModelHandler.get_detailed_results() | |
| leaderboard_df: Optional leaderboard DataFrame to match model order | |
| use_multiindex: If True, return DataFrame with MultiIndex columns for proper | |
| hierarchical display (merged headers in HTML/Gradio). | |
| Returns: | |
| pd.DataFrame: Combined table with category names as hierarchical column headers | |
| """ | |
| tasks_df = detailed_results.get("tasks", pd.DataFrame()) | |
| if tasks_df.empty: | |
| return pd.DataFrame() | |
| # Format model names | |
| tasks_df = tasks_df.copy() | |
| tasks_df["model_name"] = tasks_df["model_name"].apply(format_model_name) | |
| # Build combined dataframe with hierarchical columns | |
| combined = tasks_df[["model_name"]].copy().rename(columns={"model_name": "Model"}) | |
| column_tuples = [("", "Model")] | |
| # Group tasks by category | |
| for category, task_keys in TASK_CATEGORIES.items(): | |
| for task_key in task_keys: | |
| display_name = TASK_DISPLAY_NAMES.get(task_key, task_key) | |
| if display_name in tasks_df.columns: | |
| col_name = f"{category} | {display_name}" | |
| column_tuples.append((category, display_name)) | |
| combined[col_name] = tasks_df[display_name] | |
| # Round numeric columns | |
| combined = combined.round(4) | |
| # Sort by leaderboard order if provided, otherwise by row average | |
| if leaderboard_df is not None and "Model" in leaderboard_df.columns: | |
| # Extract model names from leaderboard (skip "Rank" column) | |
| model_order = leaderboard_df["Model"].tolist() | |
| # Create a mapping of model names to their rank | |
| model_rank = {name: idx for idx, name in enumerate(model_order)} | |
| # Sort combined dataframe by leaderboard order | |
| combined["_sort_rank"] = combined["Model"].map(model_rank) | |
| combined = combined.sort_values(by="_sort_rank", na_position="last") | |
| combined = combined.drop(columns=["_sort_rank"]) | |
| else: | |
| # Fallback: sort by average of category columns | |
| category_col_names = [] | |
| for category, task_keys in TASK_CATEGORIES.items(): | |
| for task_key in task_keys: | |
| display_name = TASK_DISPLAY_NAMES.get(task_key, task_key) | |
| col_name = f"{category} | {display_name}" | |
| if col_name in combined.columns: | |
| category_col_names.append(col_name) | |
| if category_col_names: | |
| combined["_sort_avg"] = combined[category_col_names].mean(axis=1) | |
| combined = combined.sort_values( | |
| by="_sort_avg", ascending=False, na_position="last" | |
| ) | |
| combined = combined.drop(columns=["_sort_avg"]) | |
| combined = combined.reset_index(drop=True) | |
| combined.insert(0, "#", range(1, len(combined) + 1)) | |
| column_tuples.insert(0, ("", "#")) | |
| if use_multiindex: | |
| combined.columns = pd.MultiIndex.from_tuples(column_tuples) | |
| return combined | |