File size: 5,767 Bytes
cafe7ea
 
 
 
 
 
 
 
 
 
 
f038138
cafe7ea
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
import pandas as pd

from model_handler import TASK_CATEGORIES, TASK_DISPLAY_NAMES

# Column labels for display
COLUMN_LABELS = {
    "model_name": "Model",
    "NER": "NER",
    "POS": "POS",
    "Reading Comprehension": "Reading",
    "Classification": "Classification",
    "MCQA": "MCQA",
    "Generation": "Generation",
    "Translation": "Translation",
    "Exams": "Exams",
    "Text Processing": "Text Proc.",
    "MMLU": "MMLU",
    "Average": "Average",
}

# Category columns for computing average
CATEGORY_COLS = list(TASK_CATEGORIES.keys())


def format_model_name(full_model_name: str) -> str:
    """Keep the full model name with org/model format.

    Args:
        full_model_name: Full model identifier like 'google/gemini-2.5-pro'

    Returns:
        The same model name with org/model format
    """
    return full_model_name


def get_task_scores(detailed_results: dict) -> dict:
    """Calculate average scores for each task across all models.

    Returns:
        dict: Maps task display name to average score
    """
    tasks_df = detailed_results.get("tasks", pd.DataFrame())

    if tasks_df.empty:
        return {}

    task_scores = {}
    for task_key, display_name in TASK_DISPLAY_NAMES.items():
        if display_name in tasks_df.columns:
            # Calculate average for this task
            avg_score = tasks_df[display_name].mean()
            if not pd.isna(avg_score):
                task_scores[display_name] = avg_score

    return task_scores


def prepare_leaderboard(df: pd.DataFrame) -> pd.DataFrame:
    """Prepare LLM benchmark leaderboard from raw results DataFrame."""
    if df.empty:
        return df

    df = df.copy()

    # Format model names
    df["model_name"] = df["model_name"].apply(format_model_name)

    # Calculate overall average if not present
    available_cols = [c for c in CATEGORY_COLS if c in df.columns]
    if available_cols and "Average" not in df.columns:
        df["Average"] = df[available_cols].mean(axis=1)

    # Sort by average
    if "Average" in df.columns:
        df = df.sort_values(by="Average", ascending=False).reset_index(drop=True)

    df.insert(0, "Rank", range(1, len(df) + 1))

    # Select columns for display (Average first, then categories)
    display_cols = ["Rank", "model_name", "Size", "Average"] + available_cols
    df = df[[c for c in display_cols if c in df.columns]]

    # Round numeric columns
    df = df.round(4)

    # Rename columns for display
    df = df.rename(columns=COLUMN_LABELS)
    return df


def prepare_detailed_leaderboard(
    detailed_results: dict,
    leaderboard_df: pd.DataFrame = None,
    use_multiindex: bool = True,
) -> pd.DataFrame:
    """Prepare detailed task-level leaderboard with hierarchical columns.

    Args:
        detailed_results: Dict with 'tasks' DataFrame from ModelHandler.get_detailed_results()
        leaderboard_df: Optional leaderboard DataFrame to match model order
        use_multiindex: If True, return DataFrame with MultiIndex columns for proper
                        hierarchical display (merged headers in HTML/Gradio).

    Returns:
        pd.DataFrame: Combined table with category names as hierarchical column headers
    """
    tasks_df = detailed_results.get("tasks", pd.DataFrame())

    if tasks_df.empty:
        return pd.DataFrame()

    # Format model names
    tasks_df = tasks_df.copy()
    tasks_df["model_name"] = tasks_df["model_name"].apply(format_model_name)

    # Build combined dataframe with hierarchical columns
    combined = tasks_df[["model_name"]].copy().rename(columns={"model_name": "Model"})
    column_tuples = [("", "Model")]

    # Group tasks by category
    for category, task_keys in TASK_CATEGORIES.items():
        for task_key in task_keys:
            display_name = TASK_DISPLAY_NAMES.get(task_key, task_key)
            if display_name in tasks_df.columns:
                col_name = f"{category} | {display_name}"
                column_tuples.append((category, display_name))
                combined[col_name] = tasks_df[display_name]

    # Round numeric columns
    combined = combined.round(4)

    # Sort by leaderboard order if provided, otherwise by row average
    if leaderboard_df is not None and "Model" in leaderboard_df.columns:
        # Extract model names from leaderboard (skip "Rank" column)
        model_order = leaderboard_df["Model"].tolist()
        # Create a mapping of model names to their rank
        model_rank = {name: idx for idx, name in enumerate(model_order)}
        # Sort combined dataframe by leaderboard order
        combined["_sort_rank"] = combined["Model"].map(model_rank)
        combined = combined.sort_values(by="_sort_rank", na_position="last")
        combined = combined.drop(columns=["_sort_rank"])
    else:
        # Fallback: sort by average of category columns
        category_col_names = []
        for category, task_keys in TASK_CATEGORIES.items():
            for task_key in task_keys:
                display_name = TASK_DISPLAY_NAMES.get(task_key, task_key)
                col_name = f"{category} | {display_name}"
                if col_name in combined.columns:
                    category_col_names.append(col_name)

        if category_col_names:
            combined["_sort_avg"] = combined[category_col_names].mean(axis=1)
            combined = combined.sort_values(
                by="_sort_avg", ascending=False, na_position="last"
            )
            combined = combined.drop(columns=["_sort_avg"])

    combined = combined.reset_index(drop=True)
    combined.insert(0, "#", range(1, len(combined) + 1))
    column_tuples.insert(0, ("", "#"))

    if use_multiindex:
        combined.columns = pd.MultiIndex.from_tuples(column_tuples)

    return combined