primo-eval / scoring.py
karimox's picture
Rank all-NaN entries last; tidy truncated blurbs
be5ca40 verified
Raw
History Blame Contribute Delete
3.97 kB
"""Scoring policy for the PRIMO benchmark: predictions -> per-category numbers.
Everything that turns a model's out-of-fold predictions into leaderboard
numbers lives here, kept apart from the probe and from any I/O so it stays easy
to change as the benchmark grows.
The scoring unit is a task = (dataset, target). Tasks are grouped by their
``category`` (``treatment_outcome`` / ``clinical_scores`` / ``endotype``). A
category uses a SINGLE metric (enforced in the registry), so its leaderboard
number is a plain mean of that metric -- AUROC and Pearson are never averaged
together inside a category column.
``sort_key`` is the one place they are averaged, to give the board a single
order. It is shown as the ``Mean`` column, labelled as a cross-metric average so
nobody reads it as a metric in its own right; the per-category columns remain
the numbers to compare on.
Pure numpy / sklearn-metrics -- no huggingface, no file I/O, so it unit-tests
without a network and is safe to rework mid-project.
"""
from collections import defaultdict
from collections.abc import Callable
from dataclasses import dataclass
import numpy as np
from sklearn.metrics import roc_auc_score
def compute_auroc(y_true: np.ndarray, y_pred: np.ndarray) -> float:
"""AUROC from class probabilities (``y_pred`` is an ``(n, n_classes)`` matrix)."""
classes = np.unique(y_true)
if len(classes) == 2:
return float(roc_auc_score(y_true, y_pred[:, 1]))
return float(
roc_auc_score(
y_true, y_pred, multi_class="ovr", average="weighted", labels=classes
)
)
def compute_pearson(y_true: np.ndarray, y_pred: np.ndarray) -> float:
"""Pearson r between predictions and targets; NaN if either is constant."""
if np.std(y_pred) == 0 or np.std(y_true) == 0:
return float("nan")
return float(np.corrcoef(y_pred, y_true)[0, 1])
METRICS: dict[str, Callable[[np.ndarray, np.ndarray], float]] = {
"auroc": compute_auroc,
"pearson": compute_pearson,
}
@dataclass(frozen=True)
class TaskScore:
"""One task's result: a raw metric plus the category it is grouped under."""
task_id: str
dataset_id: str
category: str
metric: str
score: float
n_samples: int
def category_means(scores: list[TaskScore]) -> dict[str, dict]:
"""Mean of the native metric per task category.
A category uses one metric, so this is a plain mean of that metric -- never a
mix of AUROC and Pearson. Degenerate (non-finite) task scores are dropped from
the mean. Returns ``{category: {metric, mean, n_tasks}}`` for the categories
present in ``scores``.
"""
by_category: dict[str, list[TaskScore]] = defaultdict(list)
for score in scores:
by_category[score.category].append(score)
out = {}
for category, items in by_category.items():
finite = [s.score for s in items if np.isfinite(s.score)]
out[category] = {
"metric": items[0].metric,
"mean": float(np.mean(finite)) if finite else float("nan"),
"n_tasks": len(items),
}
return out
def sort_key(categories: dict[str, dict]) -> float:
"""Leaderboard ranking key: mean of the per-category means.
Orders the rows, and is shown as the ``Mean`` column on boards holding more
than one category. It does average across metrics (AUROC + Pearson), a
deliberate compromise for a single order; swap for a per-category-normalized
mean if the ranking needs to be metric-fair.
An entry with nothing finite to average sorts LAST, not at zero: a constant
embedding scores NaN on every Pearson task, and zero would float it above a
model that merely correlates negatively -- ranking "no score" over "a bad
score". Nothing finite means nothing to show, so the table renders it blank.
"""
means = [c["mean"] for c in categories.values() if np.isfinite(c["mean"])]
return float(np.mean(means)) if means else float("-inf")