MERNIK / scripts /ledger.py
wepiqx's picture
hub README + gates (393adcc)
6d7c232 verified
Raw History Blame Contribute Delete
14.6 kB
#!/usr/bin/env python3
"""ledger.py — standings tables GENERATED from the artefacts, not typed.
Every standings table in the ledgers is hand-transcribed, and they have
already drifted apart: MTP.md says x1.53 where SAGA.md says 1.51x,
README §6 claims a 5-23% draft acceptance that MTP.md never measured, and
RINIQ-NEXT.md still has an older table reading "N4c ... HE running" for a
build whose verdict is printed six lines above it. Every one of those is a
typing error, and none of them is a measurement error.
This renders the same tables from eval_results + the manifest, so:
* a number can only appear if the file says so
* every number travels with its identity status, so a reader cannot
mistake a contemporaneous measurement for an established build
* `--check` diffs a ledger against the artefacts and lists drift
Usage:
python scripts/ledger.py # standings
python scripts/ledger.py --vs riniqn2 # crown duel, all 3 columns
python scripts/ledger.py --check README.md # what that ledger gets wrong
python scripts/ledger.py --json
"""
import argparse
import json
import math
import os
import re
import sys
from collections import Counter
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from manifest import (tags, collect, attach_provenance, PERMANENT, ROOT,
RESULTS)
LEDGER_DIR = ROOT
# The predecessor project's battery archive. MERNIK's own ledgers quote
# numbers from it (README §5 "Reference Results I/II" are all NeoHorse and
# OxCoder, produced under ASHQ1), and nothing in this repo pointed there —
# so `ledger.py --check` called them non-reproducible when they are merely
# filed elsewhere. Read-only, never used for the standings table: those
# builds belong to another era and another model family.
EXTERNAL_RESULTS = ["/mnt/Vsio/ASHQ1 battlefield/eval_results"]
def archive_hits(score):
"""External tags carrying the same N/164, for orphan resolution."""
hits = []
for d in EXTERNAL_RESULTS:
if not os.path.isdir(d):
continue
for f in sorted(os.listdir(d)):
if not f.endswith(".jsonl_results.jsonl"):
continue
ok = n = 0
try:
for line in open(os.path.join(d, f)):
if not line.strip():
continue
r = json.loads(line)
n += 1
ok += bool(r.get("passed"))
except Exception:
continue
if ok == score and n:
hits.append("%s (%s)" % (f.replace("humaneval_", "")
.replace(".jsonl_results.jsonl", ""),
os.path.basename(d.rstrip("/"))))
return hits
def load_all():
recs = [collect(t) for t in tags()]
attach_provenance(recs)
return recs
def _n(rec, key):
v = rec["counts"].get(key)
return int(v.split("/")[0]) if v else None
def mcnemar_p(a, b, xa, xb):
common = sorted(set(a) & set(b))
ab = sum(1 for t in common if a[t] and not b[t])
ba = sum(1 for t in common if b[t] and not a[t])
n = ab + ba
if n == 0:
return 1.0, ab, ba, len(common)
k = min(ab, ba)
p = min(1.0, 2 * sum(math.comb(n, i) for i in range(k + 1)) / 2 ** n)
return p, ab, ba, len(common)
def _vecs(tag, results_dir=RESULTS):
"""Per-task vectors for a tag, the three columns the way duel.py reads
them (human_eval for HE, evalplus for HE+, completions for empties)."""
p = os.path.join(results_dir, "humaneval_%s" % tag)
out = {}
f = p + ".jsonl_results.jsonl"
if os.path.exists(f):
out["base"] = {json.loads(l)["task_id"]: bool(json.loads(l)["passed"])
for l in open(f) if l.strip() and json.loads(l).get("passed") is not None}
f = p + "_eval_results.json"
if os.path.exists(f):
ev = json.load(open(f)).get("eval", {})
out["plus"] = {t: (r[0] if isinstance(r, list) else r).get("plus_status") == "pass"
for t, r in ev.items()
if (r[0] if isinstance(r, list) else r).get("plus_status") is not None}
f = p + ".jsonl"
if os.path.exists(f):
out["empties"] = {json.loads(l)["task_id"]: not json.loads(l)["completion"].strip()
for l in open(f) if l.strip()}
return out
COL_P = ("HE", "HE+", "empty")
def load_vectors(recs, results_dir=RESULTS):
"""Read every battery's three per-task vectors ONCE. 1711 pairs × 3
columns of file reads would take minutes; this takes about one second."""
out = {}
for r in recs:
out[r["tag"]] = _vecs(r["tag"], results_dir)
return out
def pair_verdict(va, vb):
"""Three-column paired test. Returns (verdict, per-column detail).
verdict: 'SIGNIFICANT' every column separates them, 'MIXED' some do
(columns disagree — say which, never average), 'NOISE' none do.
"""
details, sig = {}, []
for col in ("base", "plus", "empties"):
if col not in va or col not in vb:
continue
a, b = va[col], vb[col]
common = sorted(set(a) & set(b))
if not common:
continue
xa = sum(1 for t in common if a[t])
xb = sum(1 for t in common if b[t])
p, ab, ba, n = mcnemar_p(a, b, xa, xb)
hit = p < 0.05
if hit:
sig.append(col)
details[COL_P[("base", "plus", "empties").index(col)]] = {
"A": xa, "B": xb, "n": n, "delta": xa - xb,
"p": p, "A_only": ab, "B_only": ba, "sig": hit}
if not details:
return "NO-DATA", {}
if len(sig) == len(details):
v = "SIGNIFICANT"
elif sig:
v = "MIXED"
else:
v = "NOISE"
return v, details
def pairs_report(recs, vectors, min_delta=3, family_filter=None):
"""All within-family pairs, ranked. Cross-family pairs are deliberately
NOT computed: comparing a MiMo build to a Prism build measures the base
model, not the allocation — the same objection that killed the ASHQ1
archive idea. The count of suppressed pairs is reported, not hidden."""
from manifest import family_of
fams = {}
for r in recs:
if not r["counts"].get("HE"):
continue
fams.setdefault(family_of(r["tag"]) or "?", []).append(r["tag"])
rows, suppressed = [], 0
for fam, tags in sorted(fams.items()):
if family_filter and fam.lower() != family_filter.lower():
continue
tags.sort()
for i in range(len(tags)):
for j in range(i + 1, len(tags)):
a, b = tags[i], tags[j]
v, det = pair_verdict(vectors.get(a, {}), vectors.get(b, {}))
if v == "NO-DATA":
continue
d = det.get("HE", {}).get("delta", 0)
rows.append({"family": fam, "A": a, "B": b, "verdict": v,
"delta": d, "cols": det})
all_tags = [t for ts in fams.values() for t in ts]
suppressed = len(all_tags) * (len(all_tags) - 1) // 2 - len(rows)
rows.sort(key=lambda r: (-abs(r["delta"]), r["A"]))
by_verdict = Counter(r["verdict"] for r in rows)
sig_rows = [r for r in rows if r["verdict"] == "SIGNIFICANT"]
mixed_rows = [r for r in rows if r["verdict"] == "MIXED"]
head = (f"{'family':<7} {'A':<22} {'B':<22} {'dHE':>4} {'p(HE)':>7} "
f"{'dHE+':>5} {'p(HE+)':>7} verdict")
print(head)
print("-" * len(head))
for r in rows:
if abs(r["delta"]) < min_delta and r["verdict"] == "NOISE":
continue
c = r["cols"]
d1 = c.get("HE", {}).get("delta", 0)
d2 = c.get("HE+", {}).get("delta", 0)
p1 = c.get("HE", {}).get("p", float("nan"))
p2 = c.get("HE+", {}).get("p", float("nan"))
mark = {"SIGNIFICANT": "***", "MIXED": "* ", "NOISE": " "}[r["verdict"]]
print(f"{r['family']:<7} {r['A']:<22} {r['B']:<22} {d1:>+4d} {p1:>7.4f} "
f"{d2:>+5d} {p2:>7.4f} {mark} {r['verdict']}")
print(f"\n{len(rows)} within-family pairs computed, {suppressed} cross-family "
f"pairs suppressed by design (they would measure the base model).")
print(f" SIGNIFICANT (3/3 columns): {by_verdict.get('SIGNIFICANT', 0)}")
print(f" MIXED (columns disagree): {by_verdict.get('MIXED', 0)}")
print(f" NOISE (0/3): {by_verdict.get('NOISE', 0)}")
if sig_rows:
smallest = min(abs(r["delta"]) for r in sig_rows)
print(f"\n Smallest |delta| that reached 3/3 significance: {smallest} tasks "
f"({smallest/164*100:.1f}pp) — the instrument's real resolution, "
f"measured from the lab's own data.")
big_noise = [r for r in rows if r["verdict"] == "NOISE" and abs(r["delta"]) >= 8]
if big_noise:
print(f" ...and {len(big_noise)} pairs with |delta| >= 8 tasks that are "
f"still NOISE — the exact width of the noise band.")
mix = [r for r in rows if r["verdict"] == "MIXED"]
if mix:
print("\n MIXED broken down by the EXACT set of significant columns "
"(alpha=0.05, two-sided exact McNemar):")
buckets = Counter()
for r in mix:
sig = tuple(c for c in COL_P if r["cols"].get(c, {}).get("sig"))
buckets[sig or ("none",)] += 1
for sig, cnt in buckets.most_common():
label = " + ".join(sig)
print(f" only {label:<14} {cnt:3d} pairs")
empt_only = [r for r in mix
if r["cols"].get("empty", {}).get("sig")
and not r["cols"].get("HE", {}).get("sig")
and not r["cols"].get("HE+", {}).get("sig")]
if empt_only:
print(f" -> in {len(empt_only)} of them empties is the ONLY column "
f"that speaks.")
print("\n" + "=" * 72)
print("Where only empties separates two builds, capability is silent and")
print("decisiveness talks — the verdict column cannot rank those, and the")
print("empties column is the single loudest signal in the lab. — BIG,")
print(" asked verbatim to be put here; his line, my footer.")
return rows
def standings(recs, limit=None):
rows = [r for r in recs if r["counts"].get("HE")]
rows.sort(key=lambda r: -_n(r, "HE"))
if limit:
rows = rows[:limit]
head = (f"{'build':<26} {'HE':>9} {'HE+':>9} {'empty':>7} identity")
lines = [head, "-" * len(head)]
for r in rows:
idn = r["identity"]
mark = "ok" if idn in ("verified", "reconstructed") else PERMANENT
lines.append(f"{r['tag']:<26} {r['counts']['HE']:>9} "
f"{r['counts'].get('HE+','-'):>9} "
f"{r['counts'].get('empties','-'):>7} {mark}")
return "\n".join(lines)
def duel(recs, ref, other):
by = {r["tag"]: r for r in recs}
for t in (ref, other):
if t not in by:
raise SystemExit("ledger: unknown tag %r" % t)
va, vb = _vecs(ref), _vecs(other)
print(f"{ref} vs {other}\n")
sig = 0
for col, label in (("base", "HE "), ("plus", "HE+ "), ("empties", "empty")):
if col not in va or col not in vb:
print(f" {label} n/a")
continue
xa = sum(1 for t in va[col] if va[col][t])
xb = sum(1 for t in vb[col] if vb[col][t])
p, ab, ba, n = mcnemar_p(va[col], vb[col], xa, xb)
s = p < 0.05
sig += s
extra = " (fewer is better)" if col == "empties" else ""
print(f" {label} {xa:3d}/{n} vs {xb:3d}/{n} p={p:.4f} "
f"{'SIGNIFICANT' if s else 'noise'}{extra}")
print(f"\n VERDICT: {sig} of 3 columns separate them."
+ (" Crown may move." if sig == 3 else " Crown stays."))
for t in (ref, other):
print(f" {t:<24} identity: {by[t]['identity']}")
def check(recs, ledger):
path = os.path.join(LEDGER_DIR, ledger) if not os.path.isabs(ledger) else ledger
if not os.path.exists(path):
raise SystemExit("ledger: no such file %s" % path)
text = open(path, encoding="utf-8", errors="replace").read()
quoted = set(re.findall(r"\d{1,3}/164", text))
have = {}
for r in recs:
he = r["counts"].get("HE")
if he:
have.setdefault(he, []).append(r["tag"])
unlogged = [r["tag"] for r in recs
if r["counts"].get("HE") and r["counts"]["HE"] not in quoted]
orphan = sorted(q for q in quoted if q not in have)
print(f"{ledger}: {len(quoted)} distinct N/164 values quoted, "
f"{len(have)} exist in the artefacts")
print(f"\n scored here, NOT quoted in that ledger: {len(unlogged)}")
for t in unlogged:
print(f" {t:<28} {next(r for r in recs if r['tag']==t)['counts']['HE']}")
print(f"\n quoted in that ledger, NOT reproducible from THIS repo: {len(orphan)}")
for q in sorted(orphan, key=lambda s: -int(s.split("/")[0])):
n = int(q.split("/")[0])
hits = archive_hits(n)
if hits:
print(f" {q:>9} -> filed in the ASHQ1 archive: {', '.join(hits)}")
else:
print(f" {q:>9} -> NOT FOUND anywhere, including the archive")
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--json", action="store_true")
ap.add_argument("--vs", nargs=2, metavar=("REF", "OTHER"))
ap.add_argument("--check", metavar="LEDGER")
ap.add_argument("--top", type=int, default=0)
ap.add_argument("--pairs", action="store_true",
help="every within-family pair, three columns, verdict")
ap.add_argument("--family", default=None,
help="restrict --pairs to one family (mimo/riniq/prism/...)")
ap.add_argument("--min-delta", type=int, default=3,
help="--pairs: hide |d| below this for NOISE rows")
args = ap.parse_args()
recs = load_all()
if args.vs:
duel(recs, *args.vs)
return
if args.check:
check(recs, args.check)
return
if args.pairs:
pairs_report(recs, load_vectors(recs), args.min_delta, args.family)
return
if args.json:
print(json.dumps(recs, indent=2))
return
print(standings(recs, args.top or None))
perm = sum(1 for r in recs if r["identity"] == PERMANENT)
print(f"\n{len(recs)} batteries; {perm} permanently unverifiable — their "
f"scores are contemporaneous measurements, the served build is not.")
if __name__ == "__main__":
main()