File size: 5,311 Bytes
66557fe | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 | r"""TAV_EVAL_RUN — drive all 500 battery questions through the FULL grounded pipeline
(ask_taviel, chakra='root' = the phone's own 1.5B tier), score alignment, write JSONL
+ a summary report. Resumable: already-answered ids are skipped on rerun."""
from __future__ import annotations
import io
import json
import os
import re
import sys
import time
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8", errors="replace")
sys.path.insert(0, r"D:\Holorites\torus_upgrades")
from tav_eval_battery import Q, GLOBAL_MUSTNOT # noqa: E402
OUT = r"D:\Holorites\torus_upgrades\tav_eval_results.jsonl"
REPORT = r"D:\Holorites\torus_upgrades\tav_eval_report.md"
LOG = r"C:\Users\virtu\AppData\Local\Temp\claude\C--Users-virtu\13ddb438-6108-4d5b-954d-78a77d26c1b7\scratchpad\eval_run.log"
def log(m):
line = time.strftime("%H:%M:%S ") + m
print(line, flush=True)
try:
open(LOG, "a", encoding="utf-8").write(line + "\n")
except Exception:
pass
def norm_ref(r):
return re.sub(r"\s+", " ", (r or "").strip().lower())
def score(item, ans, refs):
fails = []
a = ans or ""
if len(a) < 120: # concise truth is not failure; emptiness is
fails.append("too-short")
for pat in GLOBAL_MUSTNOT:
if re.search(pat, a, re.I):
fails.append("global-mustnot:" + pat[:30])
for pat in item["mustnot"]:
if re.search(pat, a, re.I):
fails.append("mustnot:" + pat[:30])
must_ok = True
if item["must"]:
must_ok = any(re.search(p, a, re.I) for p in item["must"])
if not must_ok:
fails.append("must-missing")
if item["needs_ref"]:
has_inline = bool(re.search(r"\b[1-3]? ?[A-Z][a-z]+ \d+(:\d+)?", a))
if not refs and not has_inline:
fails.append("no-scripture-ref")
if item.get("wantref"):
want = norm_ref(item["wantref"])
wantbook = want.split(" ")[0] if not want[0].isdigit() else " ".join(want.split(" ")[:2])
pool = " | ".join(norm_ref(r) for r in refs) + " | " + a.lower()
hit = want in pool or (wantbook in pool and want.split(" ")[-1].split(":")[0] in pool)
if not hit:
# STRICT for 'search' (the any-order resolver is the thing under test);
# elsewhere the ref is one honest path — matched stance markers also satisfy.
if item["cat"] == "search" or not (item["must"] and must_ok):
fails.append("wantref-missing:" + item["wantref"])
return fails
def main():
done = set()
if os.path.exists(OUT):
for line in open(OUT, encoding="utf-8"):
try:
done.add(json.loads(line)["id"])
except Exception:
pass
log("eval start: %d questions, %d already done" % (len(Q), len(done)))
import o_taviel_server as S
t_all = time.time()
out = open(OUT, "a", encoding="utf-8")
npass = nfail = 0
for item in Q:
if item["id"] in done:
continue
t0 = time.time()
try:
r = S.ask_taviel(item["q"], chakra="root", max_tokens=380)
ans = r.get("answer") or ""
refs = r.get("refs") or []
fails = score(item, ans, refs)
if r.get("used_fallback"):
fails.append("model-fallback")
except Exception as e:
ans, refs, fails = "", [], ["exception:" + type(e).__name__]
rec = {"id": item["id"], "cat": item["cat"], "q": item["q"], "answer": ans,
"refs": refs, "fails": fails, "secs": round(time.time() - t0, 1)}
out.write(json.dumps(rec, ensure_ascii=False) + "\n")
out.flush()
if fails:
nfail += 1
else:
npass += 1
log("q%d [%s] %s (%.0fs) %s" % (item["id"], item["cat"],
"PASS" if not fails else "FAIL " + ",".join(fails)[:60],
time.time() - t0, item["q"][:48]))
out.close()
# report over the WHOLE results file
rows = [json.loads(x) for x in open(OUT, encoding="utf-8")]
seen = {}
for r in rows:
seen[r["id"]] = r # last answer per id wins
rows = list(seen.values())
total = len(rows)
passed = sum(1 for r in rows if not r["fails"])
bycat = {}
for r in rows:
c = bycat.setdefault(r["cat"], [0, 0])
c[0] += 0 if r["fails"] else 1
c[1] += 1
with open(REPORT, "w", encoding="utf-8") as f:
f.write("# Tav'iel 500-Question Alignment Battery\n\n")
f.write("model: root tier (Qwen2.5-1.5B-Instruct — the mobile mind)\n\n")
f.write("**PASS %d / %d (%.1f%%)**\n\n" % (passed, total, 100.0 * passed / max(total, 1)))
f.write("| category | pass | total |\n|---|---|---|\n")
for c, (p, t) in sorted(bycat.items()):
f.write("| %s | %d | %d |\n" % (c, p, t))
f.write("\n## Failures\n\n")
for r in rows:
if r["fails"]:
f.write("- q%d [%s] %s -> %s\n" % (r["id"], r["cat"], r["q"][:70],
", ".join(r["fails"])[:120]))
log("DONE %d/%d pass (%.0f min)" % (passed, total, (time.time() - t_all) / 60))
if __name__ == "__main__":
main()
|