AUBIN-E4B-Control / code /make_learn_report.py
emrevrg's picture
AUBIN-Learn: instant self-learning memory, fast skills, self-calibration + reports
5daf54a verified
Raw History Blame Contribute Delete
10.7 kB
"""learn_report.json → AUBIN_LEARN.md (HF'ye giden rapor). Dürüst: seçim dev'de, test bir kez; Kev referansları yanında.
python make_learn_report.py runs/learn/learn_report.json release/AUBIN_LEARN.md
"""
import json, sys
def pct(x):
return "—" if x is None else f"{100 * x:.1f}"
def glance(online_path, R):
"""Sonuçlar bir bakışta — dürüst: işe yarayan ve yaramayan birlikte."""
out = ["## Results at a glance", ""]
if online_path:
O = json.load(open(online_path, encoding="utf-8"))["runs"]
gains = [(rn, rr["kev_test"]["model"], rr["kev_test"]["aubin_learn"]) for rn, rr in O.items() if "kev_test" in rr]
tg = [(rn, rr["kev_transfer_test"]["model"], rr["kev_transfer_test"]["aubin_learn"]) for rn, rr in O.items() if "kev_transfer_test" in rr]
best = max(gains, key=lambda g: g[2])
out += [f"* **Learning from feedback works:** on the kev_test stream every one of {len(gains)} AUBIN variants improved "
f"(+{100 * min(b - a for _, a, b in gains):.1f} to +{100 * max(b - a for _, a, b in gains):.1f} points); best "
f"`{best[0]}` {pct(best[1])} → **{pct(best[2])}**. Online protocol (labels revealed after each answer) — not a static test score.",
f"* **Never-seen sources (transfer, memory starts empty):** change "
f"{100 * min(b - a for _, a, b in tg):+.1f} to {100 * max(b - a for _, a, b in tg):+.1f} points — the self-calibrator keeps "
"trust on the model when memory is not yet useful, so it does not hurt; a few hundred feedbacks per source are not enough to help."]
st = []
for emb, mems in R.get("embeds", {}).items():
for mname, res in mems.items():
for rn, rr in res["runs"].items():
if "kev_test" in rr:
st.append(rr["kev_test"]["fused"] - rr["kev_test"]["model"])
if st:
out.append(f"* **Static locked test (no feedback):** fusing a frozen memory into an already-fine-tuned AUBIN changes kev_test by "
f"{100 * min(st):+.1f} to {100 * max(st):+.1f} points — no reliable gain; AUBIN already learned these sources.")
out += ["* Latency: writing a case to memory takes well under a millisecond with the hashing embedder (tens of ms with a "
"sentence embedder); recall is a matrix product over the partition.", ""]
return out
def main():
R = json.load(open(sys.argv[1], encoding="utf-8")); out = sys.argv[2]
ref = R.get("reference", {})
L = ["# AUBIN-Learn — instant self-learning memory for AUBIN (Norovox core)", "",
"AUBIN-Learn plugs the Norovox self-learning core into AUBIN's fast decision loop. Knowledge is not baked into "
"weights: verified cases live in an external **decision memory**. When AUBIN is unsure, it recalls the most similar "
"solved cases of the same question type and fuses their vote with its own probabilities; when it is told the right "
"answer, `learn()` writes it to memory and the very next decision uses it — no gradient step, no retraining.", "",
"```python", "from aubin.learn import DecisionMemory, AubinLearner",
"mem = DecisionMemory() # or DecisionMemory(embed_fn=<sentence embedder>)",
"agent = AubinLearner(score_fn=scorer.score, memory=mem, weights=dev_tuned_weights, margin=4.0,",
" research_fn=Retrieval().fetch) # Norovox Retrieval: local KB -> web, when memory has nothing",
"choice, logprobs, evidence = agent.decide(item)", "agent.feedback(item, correct_index) # learned instantly", "```", "",
*glance(sys.argv[3] if len(sys.argv) > 3 else "", R),
"## Protocol", "",
"* Memory = Kev decision-v7 **train** (+ kev_augment extra rows). Test / development / calibration / transfer items never enter memory.",
"* Model log-probabilities come from previously measured AUBIN runs (same item order; labels verified to match).",
"* Fusion weight (per source, by log-loss), neighbour count and temperature are selected on development data only "
"(kev_dev; cal-300 when a run has no dev); kev_test and kev_transfer_test are reported once.",
"* Also measured: a learning curve (memory 0% → 100%), latency, and an online-feedback stream (separate protocol).", "",
f"Memory: {R['memory']['kev_train_items']} Kev decision-v7 train items + {R['memory']['extra_items']} leak-free extra items "
"(kev_augment.py: same HF source datasets, rows not used anywhere in Kev's suites).", ""]
for emb, mems in R.get("embeds", {}).items():
for mname, res in mems.items():
L += [f"## Embedding `{emb}`, memory `{mname}` ({res.get('memory_items', '?')} items)", "",
f"Latency: learn {res['latency_ms']['learn_median']} ms (median, incl. embedding) · recall "
f"{res['latency_ms']['recall_mean_incl_embed']} ms/query (incl. embedding).", "",
"| model predictions | tuned on | kev_test model | kev_test + memory | memory alone | kev_transfer_test model | + memory |",
"|---|---|---|---|---|---|---|"]
for rn, rr in res["runs"].items():
t, x = rr.get("kev_test", {}), rr.get("kev_transfer_test", {})
L.append(f"| {rn} | {rr['dev_used']} | {pct(t.get('model'))} | {pct(t.get('fused'))} | {pct(t.get('memory_only'))} | "
f"{pct(x.get('model'))} | {pct(x.get('fused'))} |")
best = max(res["runs"].items(), key=lambda kv: kv[1].get("kev_test", {}).get("fused", 0))
rn, rr = best
L += ["", f"Reference on kev_test (Kev's own training sources): Kev-9B {pct(ref.get('Kev-9B kev_test'))}, "
f"Kev-27B {pct(ref.get('Kev-27B kev_test'))}. Best row above: `{rn}` {pct(rr['kev_test']['fused'])}.", ""]
if "kev_test" in rr:
L += ["Per source (kev_test), model → model + memory:", "", "| source | n | model | + memory |", "|---|---|---|---|"]
for s, (a, n) in rr["kev_test"]["by_source_model"].items():
L.append(f"| {s} | {n} | {pct(a)} | {pct(rr['kev_test']['by_source_fused'][s][0])} |")
L.append("")
if "learning_curve" in rr:
L += ["Learning curve — the same model, memory grown from 0% to 100% (no retraining):", "",
"| memory fraction | items | kev_test |", "|---|---|---|"]
for f, c in rr["learning_curve"].items():
L.append(f"| {f} | {c['memory_items']} | {pct(c['kev_test_fused'])} |")
L.append("")
if "online_feedback_kev_test" in rr:
o = rr["online_feedback_kev_test"]
L += [f"Online feedback (separate protocol, not a static test score): answering kev_test as a stream and writing "
f"each correct label to memory after answering gives {pct(o['accuracy'])} overall, "
f"{pct(o['second_half'])} on the second half.", ""]
if len(sys.argv) > 4 and sys.argv[4]: # hızlı beceri (bellekten saniyede öğrenilen doğrusal sınıflandırıcı)
S = json.load(open(sys.argv[4], encoding="utf-8"))
L += ["## Fast skills (learned from memory in seconds)", "",
"The Norovox skill-library idea applied to decisions: for every source, a small softmax classifier is fitted on the "
"memory's embeddings (labels by name) in seconds on a CPU, and fused with AUBIN's probabilities. Regularisation and "
"per-source fusion weights were selected on kev_dev only; test is reported once. A skill is switched on for a source "
"only if, on kev_dev, it raises accuracy by at least 2 questions **and** lowers log-loss; everywhere else it stays off. "
"(An ungated variant that picked weights by log-loss alone gave no gain — `reports/learn_skill_ungated.json`.)", "",
"| predictions | kev_test model | kev_test + skills | kev_transfer_test model | + skills |", "|---|---|---|---|---|"]
for rn, rr in S["runs"].items():
t, x = rr.get("kev_test", {}), rr.get("kev_transfer_test", {})
L.append(f"| {rn} | {pct(t.get('model'))} | {pct(t.get('fused'))} | {pct(x.get('model'))} | {pct(x.get('fused'))} |")
best = max(S["runs"].items(), key=lambda kv: kv[1].get("kev_test", {}).get("fused", 0))
L += ["", f"Per source for `{best[0]}` (kev_test): n · model · + skills", ""]
for s, (n, m, f) in best[1]["kev_test"]["by_source"].items():
L.append(f"* {s}: {n} · {pct(m)} · {pct(f)}")
L.append("")
if len(sys.argv) > 3 and sys.argv[3]: # çevrimiçi (geri bildirimli) akış deneyi
O = json.load(open(sys.argv[3], encoding="utf-8"))
L += ["## Online self-learning (feedback stream)", "",
"Questions arrive one by one; after each answer the correct label is revealed. AUBIN-Learn writes it to memory "
"(instant) and its **self-calibrator** (Hedge / multiplicative weights over three experts — model, memory, fused) "
"shifts trust per source towards whichever has been more accurate. On kev_transfer_test the memory starts **empty**: "
"these sources were never seen in training by AUBIN or Kev. Hyper-parameters (η, initial trust) were selected on the "
"development streams only. This is a separate protocol, not a static locked-test score.", "",
"| predictions | stream | model | AUBIN-Learn | model, 2nd half | AUBIN-Learn, 2nd half |", "|---|---|---|---|---|---|"]
for rn, rr in O["runs"].items():
for s in ("kev_transfer_test", "kev_test"):
if s in rr:
x = rr[s]
L.append(f"| {rn} | {s} | {pct(x['model'])} | {pct(x['aubin_learn'])} | {pct(x['model_2nd_half'])} | {pct(x['aubin_learn_2nd_half'])} |")
L.append("")
L += ["## Honest notes", "",
"* Fusion weights, neighbour count and temperature were chosen on development data only; test numbers are reported once.",
"* Memory never contains test, development, calibration or transfer items.",
"* Sources unseen by the memory (all of transfer-v4) fall back to the model unchanged — memory cannot hurt them.",
"* Any comparison with Kev uses the same public Kev suites; Kev models are not part of AUBIN-Learn."]
open(out, "w", encoding="utf-8").write("\n".join(L) + "\n")
print(out)
if __name__ == "__main__":
main()