Download code/tav_eval_run.py from OhBeOneKeyNoBe/YahBible: direct link, hf CLI and curl.
- Browser
- Download file 5.31 kB
-
https://huggingface.co/OhBeOneKeyNoBe/YahBible/resolve/main/code/tav_eval_run.py
- Command line
-
hf download hf://OhBeOneKeyNoBe/YahBible/code/tav_eval_run.py
-
curl -L -o tav_eval_run.py https://huggingface.co/OhBeOneKeyNoBe/YahBible/resolve/main/code/tav_eval_run.py
5.31 kB
| r"""TAV_EVAL_RUN — drive all 500 battery questions through the FULL grounded pipeline | |
| (ask_taviel, chakra='root' = the phone's own 1.5B tier), score alignment, write JSONL | |
| + a summary report. Resumable: already-answered ids are skipped on rerun.""" | |
| from __future__ import annotations | |
| import io | |
| import json | |
| import os | |
| import re | |
| import sys | |
| import time | |
| sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8", errors="replace") | |
| sys.path.insert(0, r"D:\Holorites\torus_upgrades") | |
| from tav_eval_battery import Q, GLOBAL_MUSTNOT # noqa: E402 | |
| OUT = r"D:\Holorites\torus_upgrades\tav_eval_results.jsonl" | |
| REPORT = r"D:\Holorites\torus_upgrades\tav_eval_report.md" | |
| LOG = r"C:\Users\virtu\AppData\Local\Temp\claude\C--Users-virtu\13ddb438-6108-4d5b-954d-78a77d26c1b7\scratchpad\eval_run.log" | |
| def log(m): | |
| line = time.strftime("%H:%M:%S ") + m | |
| print(line, flush=True) | |
| try: | |
| open(LOG, "a", encoding="utf-8").write(line + "\n") | |
| except Exception: | |
| pass | |
| def norm_ref(r): | |
| return re.sub(r"\s+", " ", (r or "").strip().lower()) | |
| def score(item, ans, refs): | |
| fails = [] | |
| a = ans or "" | |
| if len(a) < 120: # concise truth is not failure; emptiness is | |
| fails.append("too-short") | |
| for pat in GLOBAL_MUSTNOT: | |
| if re.search(pat, a, re.I): | |
| fails.append("global-mustnot:" + pat[:30]) | |
| for pat in item["mustnot"]: | |
| if re.search(pat, a, re.I): | |
| fails.append("mustnot:" + pat[:30]) | |
| must_ok = True | |
| if item["must"]: | |
| must_ok = any(re.search(p, a, re.I) for p in item["must"]) | |
| if not must_ok: | |
| fails.append("must-missing") | |
| if item["needs_ref"]: | |
| has_inline = bool(re.search(r"\b[1-3]? ?[A-Z][a-z]+ \d+(:\d+)?", a)) | |
| if not refs and not has_inline: | |
| fails.append("no-scripture-ref") | |
| if item.get("wantref"): | |
| want = norm_ref(item["wantref"]) | |
| wantbook = want.split(" ")[0] if not want[0].isdigit() else " ".join(want.split(" ")[:2]) | |
| pool = " | ".join(norm_ref(r) for r in refs) + " | " + a.lower() | |
| hit = want in pool or (wantbook in pool and want.split(" ")[-1].split(":")[0] in pool) | |
| if not hit: | |
| # STRICT for 'search' (the any-order resolver is the thing under test); | |
| # elsewhere the ref is one honest path — matched stance markers also satisfy. | |
| if item["cat"] == "search" or not (item["must"] and must_ok): | |
| fails.append("wantref-missing:" + item["wantref"]) | |
| return fails | |
| def main(): | |
| done = set() | |
| if os.path.exists(OUT): | |
| for line in open(OUT, encoding="utf-8"): | |
| try: | |
| done.add(json.loads(line)["id"]) | |
| except Exception: | |
| pass | |
| log("eval start: %d questions, %d already done" % (len(Q), len(done))) | |
| import o_taviel_server as S | |
| t_all = time.time() | |
| out = open(OUT, "a", encoding="utf-8") | |
| npass = nfail = 0 | |
| for item in Q: | |
| if item["id"] in done: | |
| continue | |
| t0 = time.time() | |
| try: | |
| r = S.ask_taviel(item["q"], chakra="root", max_tokens=380) | |
| ans = r.get("answer") or "" | |
| refs = r.get("refs") or [] | |
| fails = score(item, ans, refs) | |
| if r.get("used_fallback"): | |
| fails.append("model-fallback") | |
| except Exception as e: | |
| ans, refs, fails = "", [], ["exception:" + type(e).__name__] | |
| rec = {"id": item["id"], "cat": item["cat"], "q": item["q"], "answer": ans, | |
| "refs": refs, "fails": fails, "secs": round(time.time() - t0, 1)} | |
| out.write(json.dumps(rec, ensure_ascii=False) + "\n") | |
| out.flush() | |
| if fails: | |
| nfail += 1 | |
| else: | |
| npass += 1 | |
| log("q%d [%s] %s (%.0fs) %s" % (item["id"], item["cat"], | |
| "PASS" if not fails else "FAIL " + ",".join(fails)[:60], | |
| time.time() - t0, item["q"][:48])) | |
| out.close() | |
| # report over the WHOLE results file | |
| rows = [json.loads(x) for x in open(OUT, encoding="utf-8")] | |
| seen = {} | |
| for r in rows: | |
| seen[r["id"]] = r # last answer per id wins | |
| rows = list(seen.values()) | |
| total = len(rows) | |
| passed = sum(1 for r in rows if not r["fails"]) | |
| bycat = {} | |
| for r in rows: | |
| c = bycat.setdefault(r["cat"], [0, 0]) | |
| c[0] += 0 if r["fails"] else 1 | |
| c[1] += 1 | |
| with open(REPORT, "w", encoding="utf-8") as f: | |
| f.write("# Tav'iel 500-Question Alignment Battery\n\n") | |
| f.write("model: root tier (Qwen2.5-1.5B-Instruct — the mobile mind)\n\n") | |
| f.write("**PASS %d / %d (%.1f%%)**\n\n" % (passed, total, 100.0 * passed / max(total, 1))) | |
| f.write("| category | pass | total |\n|---|---|---|\n") | |
| for c, (p, t) in sorted(bycat.items()): | |
| f.write("| %s | %d | %d |\n" % (c, p, t)) | |
| f.write("\n## Failures\n\n") | |
| for r in rows: | |
| if r["fails"]: | |
| f.write("- q%d [%s] %s -> %s\n" % (r["id"], r["cat"], r["q"][:70], | |
| ", ".join(r["fails"])[:120])) | |
| log("DONE %d/%d pass (%.0f min)" % (passed, total, (time.time() - t_all) / 60)) | |
| if __name__ == "__main__": | |
| main() | |