Download src/data_loader.py from EmbodiedCity/EmbodiedCity-Leaderboard: direct link, hf CLI and curl.
- Browser
- Download file 5.04 kB
-
https://huggingface.co/spaces/EmbodiedCity/EmbodiedCity-Leaderboard/resolve/main/src/data_loader.py
- Command line
-
hf download hf://spaces/EmbodiedCity/EmbodiedCity-Leaderboard/src/data_loader.py
-
curl -L -o data_loader.py https://huggingface.co/spaces/EmbodiedCity/EmbodiedCity-Leaderboard/resolve/main/src/data_loader.py
5.04 kB
| from __future__ import annotations | |
| import json | |
| from pathlib import Path | |
| from typing import Any, Dict, List, Optional | |
| import pandas as pd | |
| RADAR_METRIC_KEYS: Dict[str, List[str]] = { | |
| "urbanvideo": ["recall_avg", "perception_avg", "reasoning_avg", "navigation_avg"], | |
| "embodiednav": ["short_sr", "middle_sr", "long_sr"], | |
| } | |
| ENTRY_TYPE_CHOICES = ["Reference Baseline", "Organizer Evaluation"] | |
| class DataLoader: | |
| def __init__(self, benchmark_id: str, data_file: str = "./data/leaderboards.json"): | |
| self.benchmark_id = benchmark_id | |
| self.data_file = Path(data_file) | |
| self.df_all: Optional[pd.DataFrame] = None | |
| self.benchmark_config: Optional[dict[str, Any]] = None | |
| self.metric_display_map: dict[str, str] = {} | |
| self.display_to_internal_map: dict[str, str] = {} | |
| self.lower_better: list[str] = [] | |
| self.dimension_metrics: list[str] = [] | |
| self.dimension_display_labels: list[str] = [] | |
| self.BASIC_METRICS: list[str] = [] | |
| self.DIMENSION_METRICS: list[str] = [] | |
| self.DIMENSION_MAP: dict[str, list[str]] = {} | |
| self.ALL_METRICS: list[str] = [] | |
| self.METRIC_CHOICES: list[str] = [] | |
| def load_results(self) -> pd.DataFrame: | |
| payload = json.loads(self.data_file.read_text(encoding="utf-8")) | |
| benchmark = next( | |
| (item for item in payload["benchmarks"] if item["id"] == self.benchmark_id), | |
| None, | |
| ) | |
| if benchmark is None: | |
| raise ValueError(f"Benchmark '{self.benchmark_id}' not found in {self.data_file}") | |
| self.benchmark_config = benchmark | |
| self.metric_display_map = { | |
| **benchmark["metricLabels"], | |
| "entry_type": "Entry Type", | |
| } | |
| self.display_to_internal_map = { | |
| display_name: internal_name | |
| for internal_name, display_name in self.metric_display_map.items() | |
| } | |
| self.lower_better = benchmark.get("lowerBetter", []) | |
| self.dimension_metrics = [ | |
| metric for metric in RADAR_METRIC_KEYS.get(self.benchmark_id, []) if metric in self.metric_display_map | |
| ] | |
| self.dimension_display_labels = [ | |
| self.metric_display_map.get(metric, metric) for metric in self.dimension_metrics | |
| ] | |
| self.DIMENSION_METRICS = list(self.dimension_metrics) | |
| self.DIMENSION_MAP = {metric: [metric] for metric in self.dimension_metrics} | |
| self.ALL_METRICS = self._flatten_metric_keys(benchmark) | |
| self.METRIC_CHOICES = list(self.ALL_METRICS) | |
| self.BASIC_METRICS = [metric for metric in self.ALL_METRICS if metric not in self.DIMENSION_METRICS] | |
| rows: list[dict[str, Any]] = [] | |
| for row in benchmark["rows"]: | |
| metrics = row.get("metrics", {}) | |
| flattened: dict[str, Any] = { | |
| "Model": row["model"], | |
| "entry_type": row.get("entryType", benchmark.get("defaultEntryType", "Reference Baseline")), | |
| "note": row.get("note", ""), | |
| } | |
| for metric in self.ALL_METRICS: | |
| flattened[metric] = metrics.get(metric) | |
| rows.append(flattened) | |
| df = pd.DataFrame(rows) | |
| for metric in self.ALL_METRICS: | |
| if metric in df.columns: | |
| df[metric] = pd.to_numeric(df[metric], errors="coerce").round(2) | |
| return df | |
| def reload_data(self) -> str: | |
| self.df_all = self.load_results() | |
| return f"Loaded {len(self.df_all)} models for {self.benchmark_config['name']}" | |
| def get_entry_type_choices(self) -> List[str]: | |
| if self.df_all is None or "entry_type" not in self.df_all.columns: | |
| return ["All"] | |
| values = set(self.df_all["entry_type"].dropna()) | |
| return ["All"] + [choice for choice in ENTRY_TYPE_CHOICES if choice in values] | |
| def get_metric_choices(self) -> List[str]: | |
| return list(self.METRIC_CHOICES) | |
| def get_display_metric_choices(self) -> List[str]: | |
| return [self.metric_display_map.get(metric, metric) for metric in self.METRIC_CHOICES] | |
| def to_internal_metric(self, display_metric: str) -> str: | |
| return self.display_to_internal_map.get(display_metric, display_metric) | |
| def get_metric_label(self, metric: str) -> str: | |
| return self.metric_display_map.get(metric, metric) | |
| def get_dimension_dataframe(self, displayed_models: List[str]) -> pd.DataFrame: | |
| if self.df_all is None or not displayed_models or not self.dimension_metrics: | |
| return pd.DataFrame() | |
| columns = ["Model"] + [metric for metric in self.dimension_metrics if metric in self.df_all.columns] | |
| df = self.df_all[self.df_all["Model"].isin(displayed_models)][columns].copy() | |
| return df | |
| def _flatten_metric_keys(self, benchmark: dict[str, Any]) -> List[str]: | |
| ordered: list[str] = [] | |
| for group in benchmark["metricGroups"]: | |
| for key in group["keys"]: | |
| if key not in ordered: | |
| ordered.append(key) | |
| return ordered | |