Zero-Shot Classification
Safetensors
PEFT
English
openjev
classification
decision-model
listwise
gemma4
research
Instructions to use bambamdevs/openjev-e4b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use bambamdevs/openjev-e4b with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
Download eval/analysis/paired_analysis.py from bambamdevs/openjev-e4b: direct link, hf CLI and curl.
- Browser
- Download file 6.79 kB
-
https://huggingface.co/bambamdevs/openjev-e4b/resolve/main/eval/analysis/paired_analysis.py
- Command line
-
hf download hf://bambamdevs/openjev-e4b/eval/analysis/paired_analysis.py
-
curl -L -o paired_analysis.py https://huggingface.co/bambamdevs/openjev-e4b/resolve/main/eval/analysis/paired_analysis.py
6.79 kB
| """Row-level paired analysis for public-benchmark runs. | |
| Usage: | |
| python scripts/paired_analysis.py OUT_DIR NAME=details.jsonl [NAME=details.jsonl ...] | |
| For every run and task: accuracy with a Wilson 95% CI, an exact binomial test | |
| against chance, NLL against the ln(k) uniform reference, ECE (15 equal-width | |
| bins, as in eval/harness) with a percentile-bootstrap 95% CI, and the | |
| distribution of predicted positions next to the gold distribution. | |
| For every pair of runs: the paired accuracy delta with a paired-bootstrap 95% | |
| CI, the exact McNemar test, and the 2x2 flip matrix. | |
| """ | |
| from __future__ import annotations | |
| import itertools | |
| import json | |
| import math | |
| import sys | |
| from collections import Counter, defaultdict | |
| from pathlib import Path | |
| import numpy as np | |
| from scipy.stats import binomtest | |
| BOOTSTRAP = 2000 | |
| BINS = 15 | |
| SEED = 0 | |
| def wilson(k: int, n: int, z: float = 1.959964) -> tuple[float, float]: | |
| p = k / n | |
| d = 1 + z * z / n | |
| c = (p + z * z / (2 * n)) / d | |
| h = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / d | |
| return c - h, c + h | |
| def ece(conf: np.ndarray, correct: np.ndarray) -> float: | |
| total = 0.0 | |
| for b in range(BINS): | |
| lo, hi = b / BINS, (b + 1) / BINS | |
| m = (conf >= lo) & ((conf < hi) if b < BINS - 1 else (conf <= hi)) | |
| if m.any(): | |
| total += m.mean() * abs(conf[m].mean() - correct[m].mean()) | |
| return float(total) | |
| def mcnemar_exact(b: int, c: int) -> float: | |
| n = b + c | |
| return 1.0 if n == 0 else float(binomtest(min(b, c), n, 0.5).pvalue) | |
| def load(path: Path) -> dict[str, dict[int, dict]]: | |
| by: dict[str, dict[int, dict]] = defaultdict(dict) | |
| with path.open(encoding="utf-8") as fp: | |
| for line in fp: | |
| r = json.loads(line) | |
| by[r["task"]][int(r["index"])] = r | |
| return by | |
| def run_stats(rows: list[dict], rng: np.random.Generator) -> dict: | |
| n = len(rows) | |
| k = int(rows[0]["option_count"]) | |
| correct = np.array([bool(r["correct"]) for r in rows], dtype=float) | |
| conf = np.array([float(r["confidence"]) for r in rows]) | |
| nll = np.array([-math.log(max(float(r["p_gold"]), 1e-12)) for r in rows]) | |
| hits = int(correct.sum()) | |
| lo, hi = wilson(hits, n) | |
| idx = rng.integers(0, n, size=(BOOTSTRAP, n)) | |
| eces = [ece(conf[i], correct[i]) for i in idx] | |
| gold = Counter(int(r["gold"]) for r in rows) | |
| pred = Counter(int(r["pred"]) for r in rows) | |
| return { | |
| "n": n, | |
| "k": k, | |
| "accuracy": hits / n, | |
| "accuracy_ci95": [lo, hi], | |
| "chance": 1 / k, | |
| "binom_p_vs_chance_two_sided": float(binomtest(hits, n, 1 / k).pvalue), | |
| "nll": float(nll.mean()), | |
| "ln_k": math.log(k), | |
| "nll_worse_than_uniform": bool(nll.mean() > math.log(k)), | |
| "ece_15bin": ece(conf, correct), | |
| "ece_15bin_ci95": [float(np.percentile(eces, 2.5)), float(np.percentile(eces, 97.5))], | |
| "gold_position_share": [gold[i] / n for i in range(k)], | |
| "pred_position_share": [pred[i] / n for i in range(k)], | |
| "max_pred_position_share": max(pred[i] / n for i in range(k)), | |
| } | |
| def pair_stats(a: list[dict], b: list[dict], rng: np.random.Generator) -> dict: | |
| ca = np.array([bool(r["correct"]) for r in a], dtype=float) | |
| cb = np.array([bool(r["correct"]) for r in b], dtype=float) | |
| n = len(ca) | |
| diff = cb - ca | |
| idx = rng.integers(0, n, size=(BOOTSTRAP, n)) | |
| boots = diff[idx].mean(axis=1) | |
| both = int(((ca == 1) & (cb == 1)).sum()) | |
| a_only = int(((ca == 1) & (cb == 0)).sum()) | |
| b_only = int(((ca == 0) & (cb == 1)).sum()) | |
| neither = int(((ca == 0) & (cb == 0)).sum()) | |
| return { | |
| "n": n, | |
| "delta_pp": 100 * float(diff.mean()), | |
| "delta_pp_ci95": [100 * float(np.percentile(boots, 2.5)), 100 * float(np.percentile(boots, 97.5))], | |
| "flip": {"both": both, "a_only": a_only, "b_only": b_only, "neither": neither}, | |
| "net_rows": b_only - a_only, | |
| "mcnemar_exact_p": mcnemar_exact(a_only, b_only), | |
| } | |
| def main() -> None: | |
| out_dir = Path(sys.argv[1]) | |
| runs = {} | |
| for arg in sys.argv[2:]: | |
| name, path = arg.split("=", 1) | |
| runs[name] = load(Path(path)) | |
| out_dir.mkdir(parents=True, exist_ok=True) | |
| rng = np.random.default_rng(SEED) | |
| tasks = sorted(set.intersection(*(set(r) for r in runs.values())), key=list(next(iter(runs.values()))).index) | |
| report = {"bootstrap": BOOTSTRAP, "ece_bins": BINS, "seed": SEED, "runs": {}, "pairs": {}} | |
| for name, by in runs.items(): | |
| report["runs"][name] = {t: run_stats([by[t][i] for i in sorted(by[t])], rng) for t in tasks} | |
| for a, b in itertools.combinations(runs, 2): | |
| key = f"{a}->{b}" | |
| report["pairs"][key] = {} | |
| for t in tasks: | |
| ids = sorted(set(runs[a][t]) & set(runs[b][t])) | |
| for i in ids: | |
| if runs[a][t][i]["gold"] != runs[b][t][i]["gold"]: | |
| raise SystemExit(f"gold mismatch {t}:{i} between {a} and {b}") | |
| report["pairs"][key][t] = pair_stats([runs[a][t][i] for i in ids], [runs[b][t][i] for i in ids], rng) | |
| (out_dir / "paired_analysis.json").write_text(json.dumps(report, indent=2), encoding="utf-8") | |
| lines = ["# Paired analysis", "", f"Bootstrap resamples: {BOOTSTRAP}; ECE: {BINS} equal-width bins; seed {SEED}.", ""] | |
| for name, stats in report["runs"].items(): | |
| lines += [f"## {name}", "", "| Task | n | k | Acc | 95% CI | p vs chance | NLL | ln k | ECE (95% CI) | max pred-position share |", | |
| "|---|---:|---:|---:|---|---:|---:|---:|---|---:|"] | |
| for t, s in stats.items(): | |
| lines.append( | |
| f"| {t} | {s['n']} | {s['k']} | {100*s['accuracy']:.2f}% | [{100*s['accuracy_ci95'][0]:.1f}, {100*s['accuracy_ci95'][1]:.1f}] " | |
| f"| {s['binom_p_vs_chance_two_sided']:.2g} | {s['nll']:.3f}{' (>ln k)' if s['nll_worse_than_uniform'] else ''} | {s['ln_k']:.3f} " | |
| f"| {s['ece_15bin']:.3f} [{s['ece_15bin_ci95'][0]:.3f}, {s['ece_15bin_ci95'][1]:.3f}] | {100*s['max_pred_position_share']:.0f}% |" | |
| ) | |
| lines.append("") | |
| for key, stats in report["pairs"].items(): | |
| a, b = key.split("->") | |
| lines += [f"## {a} -> {b}", "", f"| Task | Δ pp | 95% CI | {a} only | {b} only | net rows | McNemar p |", "|---|---:|---|---:|---:|---:|---:|"] | |
| for t, s in stats.items(): | |
| lines.append( | |
| f"| {t} | {s['delta_pp']:+.2f} | [{s['delta_pp_ci95'][0]:+.2f}, {s['delta_pp_ci95'][1]:+.2f}] " | |
| f"| {s['flip']['a_only']} | {s['flip']['b_only']} | {s['net_rows']:+d} | {s['mcnemar_exact_p']:.2g} |" | |
| ) | |
| lines.append("") | |
| (out_dir / "paired_analysis.md").write_text("\n".join(lines), encoding="utf-8") | |
| print("\n".join(lines)) | |
| if __name__ == "__main__": | |
| main() | |