Zero-Shot Classification
Safetensors
PEFT
English
openjev
classification
decision-model
listwise
gemma4
research
Instructions to use bambamdevs/openjev-e4b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use bambamdevs/openjev-e4b with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
File size: 6,792 Bytes
03223d7 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 | """Row-level paired analysis for public-benchmark runs.
Usage:
python scripts/paired_analysis.py OUT_DIR NAME=details.jsonl [NAME=details.jsonl ...]
For every run and task: accuracy with a Wilson 95% CI, an exact binomial test
against chance, NLL against the ln(k) uniform reference, ECE (15 equal-width
bins, as in eval/harness) with a percentile-bootstrap 95% CI, and the
distribution of predicted positions next to the gold distribution.
For every pair of runs: the paired accuracy delta with a paired-bootstrap 95%
CI, the exact McNemar test, and the 2x2 flip matrix.
"""
from __future__ import annotations
import itertools
import json
import math
import sys
from collections import Counter, defaultdict
from pathlib import Path
import numpy as np
from scipy.stats import binomtest
BOOTSTRAP = 2000
BINS = 15
SEED = 0
def wilson(k: int, n: int, z: float = 1.959964) -> tuple[float, float]:
p = k / n
d = 1 + z * z / n
c = (p + z * z / (2 * n)) / d
h = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / d
return c - h, c + h
def ece(conf: np.ndarray, correct: np.ndarray) -> float:
total = 0.0
for b in range(BINS):
lo, hi = b / BINS, (b + 1) / BINS
m = (conf >= lo) & ((conf < hi) if b < BINS - 1 else (conf <= hi))
if m.any():
total += m.mean() * abs(conf[m].mean() - correct[m].mean())
return float(total)
def mcnemar_exact(b: int, c: int) -> float:
n = b + c
return 1.0 if n == 0 else float(binomtest(min(b, c), n, 0.5).pvalue)
def load(path: Path) -> dict[str, dict[int, dict]]:
by: dict[str, dict[int, dict]] = defaultdict(dict)
with path.open(encoding="utf-8") as fp:
for line in fp:
r = json.loads(line)
by[r["task"]][int(r["index"])] = r
return by
def run_stats(rows: list[dict], rng: np.random.Generator) -> dict:
n = len(rows)
k = int(rows[0]["option_count"])
correct = np.array([bool(r["correct"]) for r in rows], dtype=float)
conf = np.array([float(r["confidence"]) for r in rows])
nll = np.array([-math.log(max(float(r["p_gold"]), 1e-12)) for r in rows])
hits = int(correct.sum())
lo, hi = wilson(hits, n)
idx = rng.integers(0, n, size=(BOOTSTRAP, n))
eces = [ece(conf[i], correct[i]) for i in idx]
gold = Counter(int(r["gold"]) for r in rows)
pred = Counter(int(r["pred"]) for r in rows)
return {
"n": n,
"k": k,
"accuracy": hits / n,
"accuracy_ci95": [lo, hi],
"chance": 1 / k,
"binom_p_vs_chance_two_sided": float(binomtest(hits, n, 1 / k).pvalue),
"nll": float(nll.mean()),
"ln_k": math.log(k),
"nll_worse_than_uniform": bool(nll.mean() > math.log(k)),
"ece_15bin": ece(conf, correct),
"ece_15bin_ci95": [float(np.percentile(eces, 2.5)), float(np.percentile(eces, 97.5))],
"gold_position_share": [gold[i] / n for i in range(k)],
"pred_position_share": [pred[i] / n for i in range(k)],
"max_pred_position_share": max(pred[i] / n for i in range(k)),
}
def pair_stats(a: list[dict], b: list[dict], rng: np.random.Generator) -> dict:
ca = np.array([bool(r["correct"]) for r in a], dtype=float)
cb = np.array([bool(r["correct"]) for r in b], dtype=float)
n = len(ca)
diff = cb - ca
idx = rng.integers(0, n, size=(BOOTSTRAP, n))
boots = diff[idx].mean(axis=1)
both = int(((ca == 1) & (cb == 1)).sum())
a_only = int(((ca == 1) & (cb == 0)).sum())
b_only = int(((ca == 0) & (cb == 1)).sum())
neither = int(((ca == 0) & (cb == 0)).sum())
return {
"n": n,
"delta_pp": 100 * float(diff.mean()),
"delta_pp_ci95": [100 * float(np.percentile(boots, 2.5)), 100 * float(np.percentile(boots, 97.5))],
"flip": {"both": both, "a_only": a_only, "b_only": b_only, "neither": neither},
"net_rows": b_only - a_only,
"mcnemar_exact_p": mcnemar_exact(a_only, b_only),
}
def main() -> None:
out_dir = Path(sys.argv[1])
runs = {}
for arg in sys.argv[2:]:
name, path = arg.split("=", 1)
runs[name] = load(Path(path))
out_dir.mkdir(parents=True, exist_ok=True)
rng = np.random.default_rng(SEED)
tasks = sorted(set.intersection(*(set(r) for r in runs.values())), key=list(next(iter(runs.values()))).index)
report = {"bootstrap": BOOTSTRAP, "ece_bins": BINS, "seed": SEED, "runs": {}, "pairs": {}}
for name, by in runs.items():
report["runs"][name] = {t: run_stats([by[t][i] for i in sorted(by[t])], rng) for t in tasks}
for a, b in itertools.combinations(runs, 2):
key = f"{a}->{b}"
report["pairs"][key] = {}
for t in tasks:
ids = sorted(set(runs[a][t]) & set(runs[b][t]))
for i in ids:
if runs[a][t][i]["gold"] != runs[b][t][i]["gold"]:
raise SystemExit(f"gold mismatch {t}:{i} between {a} and {b}")
report["pairs"][key][t] = pair_stats([runs[a][t][i] for i in ids], [runs[b][t][i] for i in ids], rng)
(out_dir / "paired_analysis.json").write_text(json.dumps(report, indent=2), encoding="utf-8")
lines = ["# Paired analysis", "", f"Bootstrap resamples: {BOOTSTRAP}; ECE: {BINS} equal-width bins; seed {SEED}.", ""]
for name, stats in report["runs"].items():
lines += [f"## {name}", "", "| Task | n | k | Acc | 95% CI | p vs chance | NLL | ln k | ECE (95% CI) | max pred-position share |",
"|---|---:|---:|---:|---|---:|---:|---:|---|---:|"]
for t, s in stats.items():
lines.append(
f"| {t} | {s['n']} | {s['k']} | {100*s['accuracy']:.2f}% | [{100*s['accuracy_ci95'][0]:.1f}, {100*s['accuracy_ci95'][1]:.1f}] "
f"| {s['binom_p_vs_chance_two_sided']:.2g} | {s['nll']:.3f}{' (>ln k)' if s['nll_worse_than_uniform'] else ''} | {s['ln_k']:.3f} "
f"| {s['ece_15bin']:.3f} [{s['ece_15bin_ci95'][0]:.3f}, {s['ece_15bin_ci95'][1]:.3f}] | {100*s['max_pred_position_share']:.0f}% |"
)
lines.append("")
for key, stats in report["pairs"].items():
a, b = key.split("->")
lines += [f"## {a} -> {b}", "", f"| Task | Δ pp | 95% CI | {a} only | {b} only | net rows | McNemar p |", "|---|---:|---|---:|---:|---:|---:|"]
for t, s in stats.items():
lines.append(
f"| {t} | {s['delta_pp']:+.2f} | [{s['delta_pp_ci95'][0]:+.2f}, {s['delta_pp_ci95'][1]:+.2f}] "
f"| {s['flip']['a_only']} | {s['flip']['b_only']} | {s['net_rows']:+d} | {s['mcnemar_exact_p']:.2g} |"
)
lines.append("")
(out_dir / "paired_analysis.md").write_text("\n".join(lines), encoding="utf-8")
print("\n".join(lines))
if __name__ == "__main__":
main()
|