"""learn_report.json → AUBIN_LEARN.md (HF'ye giden rapor). Dürüst: seçim dev'de, test bir kez; Kev referansları yanında. python make_learn_report.py runs/learn/learn_report.json release/AUBIN_LEARN.md """ import json, sys def pct(x): return "—" if x is None else f"{100 * x:.1f}" def glance(online_path, R): """Sonuçlar bir bakışta — dürüst: işe yarayan ve yaramayan birlikte.""" out = ["## Results at a glance", ""] if online_path: O = json.load(open(online_path, encoding="utf-8"))["runs"] gains = [(rn, rr["kev_test"]["model"], rr["kev_test"]["aubin_learn"]) for rn, rr in O.items() if "kev_test" in rr] tg = [(rn, rr["kev_transfer_test"]["model"], rr["kev_transfer_test"]["aubin_learn"]) for rn, rr in O.items() if "kev_transfer_test" in rr] best = max(gains, key=lambda g: g[2]) out += [f"* **Learning from feedback works:** on the kev_test stream every one of {len(gains)} AUBIN variants improved " f"(+{100 * min(b - a for _, a, b in gains):.1f} to +{100 * max(b - a for _, a, b in gains):.1f} points); best " f"`{best[0]}` {pct(best[1])} → **{pct(best[2])}**. Online protocol (labels revealed after each answer) — not a static test score.", f"* **Never-seen sources (transfer, memory starts empty):** change " f"{100 * min(b - a for _, a, b in tg):+.1f} to {100 * max(b - a for _, a, b in tg):+.1f} points — the self-calibrator keeps " "trust on the model when memory is not yet useful, so it does not hurt; a few hundred feedbacks per source are not enough to help."] st = [] for emb, mems in R.get("embeds", {}).items(): for mname, res in mems.items(): for rn, rr in res["runs"].items(): if "kev_test" in rr: st.append(rr["kev_test"]["fused"] - rr["kev_test"]["model"]) if st: out.append(f"* **Static locked test (no feedback):** fusing a frozen memory into an already-fine-tuned AUBIN changes kev_test by " f"{100 * min(st):+.1f} to {100 * max(st):+.1f} points — no reliable gain; AUBIN already learned these sources.") out += ["* Latency: writing a case to memory takes well under a millisecond with the hashing embedder (tens of ms with a " "sentence embedder); recall is a matrix product over the partition.", ""] return out def main(): R = json.load(open(sys.argv[1], encoding="utf-8")); out = sys.argv[2] ref = R.get("reference", {}) L = ["# AUBIN-Learn — instant self-learning memory for AUBIN (Norovox core)", "", "AUBIN-Learn plugs the Norovox self-learning core into AUBIN's fast decision loop. Knowledge is not baked into " "weights: verified cases live in an external **decision memory**. When AUBIN is unsure, it recalls the most similar " "solved cases of the same question type and fuses their vote with its own probabilities; when it is told the right " "answer, `learn()` writes it to memory and the very next decision uses it — no gradient step, no retraining.", "", "```python", "from aubin.learn import DecisionMemory, AubinLearner", "mem = DecisionMemory() # or DecisionMemory(embed_fn=)", "agent = AubinLearner(score_fn=scorer.score, memory=mem, weights=dev_tuned_weights, margin=4.0,", " research_fn=Retrieval().fetch) # Norovox Retrieval: local KB -> web, when memory has nothing", "choice, logprobs, evidence = agent.decide(item)", "agent.feedback(item, correct_index) # learned instantly", "```", "", *glance(sys.argv[3] if len(sys.argv) > 3 else "", R), "## Protocol", "", "* Memory = Kev decision-v7 **train** (+ kev_augment extra rows). Test / development / calibration / transfer items never enter memory.", "* Model log-probabilities come from previously measured AUBIN runs (same item order; labels verified to match).", "* Fusion weight (per source, by log-loss), neighbour count and temperature are selected on development data only " "(kev_dev; cal-300 when a run has no dev); kev_test and kev_transfer_test are reported once.", "* Also measured: a learning curve (memory 0% → 100%), latency, and an online-feedback stream (separate protocol).", "", f"Memory: {R['memory']['kev_train_items']} Kev decision-v7 train items + {R['memory']['extra_items']} leak-free extra items " "(kev_augment.py: same HF source datasets, rows not used anywhere in Kev's suites).", ""] for emb, mems in R.get("embeds", {}).items(): for mname, res in mems.items(): L += [f"## Embedding `{emb}`, memory `{mname}` ({res.get('memory_items', '?')} items)", "", f"Latency: learn {res['latency_ms']['learn_median']} ms (median, incl. embedding) · recall " f"{res['latency_ms']['recall_mean_incl_embed']} ms/query (incl. embedding).", "", "| model predictions | tuned on | kev_test model | kev_test + memory | memory alone | kev_transfer_test model | + memory |", "|---|---|---|---|---|---|---|"] for rn, rr in res["runs"].items(): t, x = rr.get("kev_test", {}), rr.get("kev_transfer_test", {}) L.append(f"| {rn} | {rr['dev_used']} | {pct(t.get('model'))} | {pct(t.get('fused'))} | {pct(t.get('memory_only'))} | " f"{pct(x.get('model'))} | {pct(x.get('fused'))} |") best = max(res["runs"].items(), key=lambda kv: kv[1].get("kev_test", {}).get("fused", 0)) rn, rr = best L += ["", f"Reference on kev_test (Kev's own training sources): Kev-9B {pct(ref.get('Kev-9B kev_test'))}, " f"Kev-27B {pct(ref.get('Kev-27B kev_test'))}. Best row above: `{rn}` {pct(rr['kev_test']['fused'])}.", ""] if "kev_test" in rr: L += ["Per source (kev_test), model → model + memory:", "", "| source | n | model | + memory |", "|---|---|---|---|"] for s, (a, n) in rr["kev_test"]["by_source_model"].items(): L.append(f"| {s} | {n} | {pct(a)} | {pct(rr['kev_test']['by_source_fused'][s][0])} |") L.append("") if "learning_curve" in rr: L += ["Learning curve — the same model, memory grown from 0% to 100% (no retraining):", "", "| memory fraction | items | kev_test |", "|---|---|---|"] for f, c in rr["learning_curve"].items(): L.append(f"| {f} | {c['memory_items']} | {pct(c['kev_test_fused'])} |") L.append("") if "online_feedback_kev_test" in rr: o = rr["online_feedback_kev_test"] L += [f"Online feedback (separate protocol, not a static test score): answering kev_test as a stream and writing " f"each correct label to memory after answering gives {pct(o['accuracy'])} overall, " f"{pct(o['second_half'])} on the second half.", ""] if len(sys.argv) > 4 and sys.argv[4]: # hızlı beceri (bellekten saniyede öğrenilen doğrusal sınıflandırıcı) S = json.load(open(sys.argv[4], encoding="utf-8")) L += ["## Fast skills (learned from memory in seconds)", "", "The Norovox skill-library idea applied to decisions: for every source, a small softmax classifier is fitted on the " "memory's embeddings (labels by name) in seconds on a CPU, and fused with AUBIN's probabilities. Regularisation and " "per-source fusion weights were selected on kev_dev only; test is reported once. A skill is switched on for a source " "only if, on kev_dev, it raises accuracy by at least 2 questions **and** lowers log-loss; everywhere else it stays off. " "(An ungated variant that picked weights by log-loss alone gave no gain — `reports/learn_skill_ungated.json`.)", "", "| predictions | kev_test model | kev_test + skills | kev_transfer_test model | + skills |", "|---|---|---|---|---|"] for rn, rr in S["runs"].items(): t, x = rr.get("kev_test", {}), rr.get("kev_transfer_test", {}) L.append(f"| {rn} | {pct(t.get('model'))} | {pct(t.get('fused'))} | {pct(x.get('model'))} | {pct(x.get('fused'))} |") best = max(S["runs"].items(), key=lambda kv: kv[1].get("kev_test", {}).get("fused", 0)) L += ["", f"Per source for `{best[0]}` (kev_test): n · model · + skills", ""] for s, (n, m, f) in best[1]["kev_test"]["by_source"].items(): L.append(f"* {s}: {n} · {pct(m)} · {pct(f)}") L.append("") if len(sys.argv) > 3 and sys.argv[3]: # çevrimiçi (geri bildirimli) akış deneyi O = json.load(open(sys.argv[3], encoding="utf-8")) L += ["## Online self-learning (feedback stream)", "", "Questions arrive one by one; after each answer the correct label is revealed. AUBIN-Learn writes it to memory " "(instant) and its **self-calibrator** (Hedge / multiplicative weights over three experts — model, memory, fused) " "shifts trust per source towards whichever has been more accurate. On kev_transfer_test the memory starts **empty**: " "these sources were never seen in training by AUBIN or Kev. Hyper-parameters (η, initial trust) were selected on the " "development streams only. This is a separate protocol, not a static locked-test score.", "", "| predictions | stream | model | AUBIN-Learn | model, 2nd half | AUBIN-Learn, 2nd half |", "|---|---|---|---|---|---|"] for rn, rr in O["runs"].items(): for s in ("kev_transfer_test", "kev_test"): if s in rr: x = rr[s] L.append(f"| {rn} | {s} | {pct(x['model'])} | {pct(x['aubin_learn'])} | {pct(x['model_2nd_half'])} | {pct(x['aubin_learn_2nd_half'])} |") L.append("") L += ["## Honest notes", "", "* Fusion weights, neighbour count and temperature were chosen on development data only; test numbers are reported once.", "* Memory never contains test, development, calibration or transfer items.", "* Sources unseen by the memory (all of transfer-v4) fall back to the model unchanged — memory cannot hurt them.", "* Any comparison with Kev uses the same public Kev suites; Kev models are not part of AUBIN-Learn."] open(out, "w", encoding="utf-8").write("\n".join(L) + "\n") print(out) if __name__ == "__main__": main()