Instructions to use Falln87/clerk-memory with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use Falln87/clerk-memory with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
Download clerk/generator.py from Falln87/clerk-memory: direct link, hf CLI and curl.
- Browser
- Download file 20.3 kB
-
https://huggingface.co/Falln87/clerk-memory/resolve/main/clerk/generator.py
- Command line
-
hf download hf://Falln87/clerk-memory/clerk/generator.py
-
curl -L -o generator.py https://huggingface.co/Falln87/clerk-memory/resolve/main/clerk/generator.py
20.3 kB
| """CLERK synthetic evolving-session generator. | |
| Generates personas with evolving facts across multiple dialogue sessions. | |
| Because every fact event (ADD / UPDATE / TOMBSTONE) is generated | |
| programmatically, the gold ledger state after every session is known by | |
| construction, which yields exact supervision for the consolidation policy — | |
| no LLM judge anywhere in the pipeline. | |
| Invariants the code maintains: | |
| * Questions are only asked about predicates the ledger can actually | |
| support: never-stated (unknown), currently-valid (stale/superseded), | |
| or currently-tombstoned (negated/temporal). Active-but-EVICTED | |
| predicates are never asked, so supervision never demands recall of | |
| what the memory no longer holds. | |
| * n_uses (the salience signal) is incremented at the moment a question | |
| is asked, so the eviction oracle at session t only sees evidence | |
| available up to t. | |
| * Gold programs are always executable: if an ADD would overflow a full | |
| ledger, the oracle emits EVICT (lowest salience) first, then the ADD. | |
| * Everything is seeded: the same --seed yields byte-identical data. | |
| Author: Justin Wolcott (fallnai-research.org) | |
| License: Apache-2.0 | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import random | |
| from typing import Any, Dict, List, Optional, Tuple | |
| from clerk.common import ( | |
| LEDGER_BUDGET_DEFAULT, | |
| Slot, | |
| apply_ops, | |
| empty_ledger, | |
| salience, | |
| serialize_ledger, | |
| ) | |
| # ---------------------------------------------------------------- predicates | |
| # p: ledger key | h: human phrase | v: value pool | |
| # add/upd/tmb: statement templates ({v} = new value). | |
| PREDICATES: List[Dict[str, Any]] = [ | |
| {"p": "job", "h": "job", "v": ["barista", "chef", "nurse", "high school teacher", "librarian", "electrician", "accountant", "software engineer", "pharmacist", "bus driver"], | |
| "add": "Big news — I started a new job as a {v}.", "upd": "I actually switched jobs again. I'm a {v} now.", "tmb": "I left my job — between things right now."}, | |
| {"p": "city", "h": "city", "v": ["Denver", "Portland", "Nashville", "Austin", "Madison", "Boise", "Tucson", "Salem", "Reno", "Spokane"], | |
| "add": "I just moved to {v}!", "upd": "We relocated again — I'm in {v} now.", "tmb": "I'm between cities at the moment, staying with family."}, | |
| {"p": "pet_kind", "h": "pet", "v": ["cat", "dog", "rabbit", "parakeet", "hamster", "iguana"], | |
| "add": "I adopted a {v}.", "upd": "Long story, but I rehomed the old one and got a {v} instead.", "tmb": "My pet passed away last month. It's been rough."}, | |
| {"p": "partner", "h": "partner", "v": ["Sam", "Jordan", "Priya", "Diego", "Nina", "Marcus", "Ellie", "Omar"], | |
| "add": "I've been seeing someone — their name is {v}.", "upd": "That ended, and I'm dating {v} now.", "tmb": "We broke up, so I'm single again."}, | |
| {"p": "car", "h": "car", "v": ["a red Civic", "a blue Outback", "a gray Silverado", "a used Miata", "an old Corolla", "a white Jetta"], | |
| "add": "I finally bought a car — {v}.", "upd": "I traded it in for {v}.", "tmb": "I sold the car — biking everywhere now."}, | |
| {"p": "hobby", "h": "main hobby", "v": ["bouldering", "watercolor painting", "chess", "birdwatching", "pottery", "kickboxing", "sourdough baking", "trail running"], | |
| "add": "I've gotten really into {v}.", "upd": "I switched hobbies — now it's all about {v}.", "tmb": "I had to drop the hobby, no time lately."}, | |
| {"p": "favorite_food", "h": "favorite food", "v": ["pad thai", "dumplings", "tacos", "ramen", "lasagna", "pho", "shawarma", "pancakes"], | |
| "add": "If I had to pick a favorite food, it's {v}.", "upd": "I've changed my favorite food to {v}, believe it or not.", "tmb": "I can't eat that anymore — no favorite food these days."}, | |
| {"p": "gym", "h": "gym", "v": ["RiseFit", "IronWorks", "FlexCity", "PeakBarre", "the Strength Lab", "Northside Athletic"], | |
| "add": "I joined a new gym called {v}.", "upd": "I switched gyms — now I'm at {v}.", "tmb": "I cancelled my gym membership."}, | |
| {"p": "boss", "h": "boss", "v": ["Dana", "Victor", "Renee", "Kofi", "Beatrice", "Hank"], | |
| "add": "My new boss's name is {v}.", "upd": "There's a new boss over me now: {v}.", "tmb": "My boss left and they haven't replaced them."}, | |
| {"p": "coworker", "h": "closest coworker", "v": ["Lena", "Theo", "Yusuf", "Marge", "Pete", "Iris"], | |
| "add": "I've become friends with a coworker named {v}.", "upd": "I'm on a new team now — my closest coworker there is {v}.", "tmb": "That coworker transferred away."}, | |
| {"p": "project", "h": "big work project", "v": ["the website redesign", "the database migration", "the product launch", "the security audit", "the onboarding revamp"], | |
| "add": "I'm leading {v} at work.", "upd": "I got moved off that and onto {v}.", "tmb": "The project got cancelled, honestly."}, | |
| {"p": "deadline", "h": "big deadline", "v": ["March 14", "April 2", "May 9", "June 21", "July 3", "August 30"], | |
| "add": "My big deadline is {v}.", "upd": "The deadline moved — it's {v} now.", "tmb": "The deadline got pushed indefinitely."}, | |
| {"p": "instrument", "h": "instrument", "v": ["guitar", "violin", "accordion", "cello", "drums", "trumpet"], | |
| "add": "I've been learning the {v}.", "upd": "I switched instruments — playing the {v} now.", "tmb": "I gave up the instrument, my wrists couldn't take it."}, | |
| {"p": "team", "h": "sports team I follow", "v": ["the Larks", "the Comets", "the Harbors", "the Northside Pistons", "the Rovers"], | |
| "add": "I've started following {v}.", "upd": "I switched allegiances — it's {v} now.", "tmb": "I stopped following sports entirely."}, | |
| {"p": "allergy", "h": "allergy", "v": ["shellfish", "peanuts", "latex", "penicillin", "kiwi"], | |
| "add": "Heads up: I'm allergic to {v}.", "upd": "New allergy diagnosis — {v} this time.", "tmb": "Good news, the allergy testing came back clear."}, | |
| {"p": "vacation_plan", "h": "next vacation spot", "v": ["Lisbon", "Kyoto", "Banff", "Oaxaca", "the Azores", "Marrakesh"], | |
| "add": "I'm planning a trip to {v}.", "upd": "Change of plans — we're going to {v} instead.", "tmb": "The trip is off."}, | |
| {"p": "hobby_class", "h": "evening class", "v": ["a pottery class", "a Spanish class", "a jazz piano class", "a photography class", "a welding class"], | |
| "add": "I signed up for {v}.", "upd": "I dropped that and enrolled in {v}.", "tmb": "I withdrew from the class."}, | |
| {"p": "neighbor", "h": "neighbor", "v": ["Mr. Alvarez", "Ms. Chen", "the Kowalskis", "Pastor Jim", "Dr. Patel"], | |
| "add": "My new neighbor is {v}.", "upd": "They moved out — my neighbor now is {v}.", "tmb": "My neighbor moved away recently."}, | |
| ] | |
| TRAIN_NAMES = ["Alex", "Bailey", "Casey", "Drew", "Emery", "Frankie", "Gray", "Harper", | |
| "Indigo", "Jesse", "Kai", "Logan", "Marley", "Noel", "Oakley", "Parker", | |
| "Quinn", "Reese", "Sage", "Tatum", "Val", "Wren", "Xander", "Yosef"] | |
| TEST_NAMES = ["Ari", "Bellamy", "Crew", "Dallas", "East", "Flynn", "Greer", "Hollis", | |
| "Ira", "Jules", "Kit", "Lennon", "Morgan", "Nico", "Onyx", "Perry", | |
| "Quince", "Rory", "Sailor", "Tao", "Uri", "Vesper", "Wilder", "Yael"] | |
| CHATTER = [ | |
| "Anyway — how's your week going?", | |
| "Sorry, I just needed to vent a little.", | |
| "This weather is something else lately.", | |
| "Did you catch the game last night?", | |
| "I should probably get more sleep.", | |
| "Anyway, enough about that.", | |
| "My commute was ridiculous today.", | |
| "Coffee count today: three. Send help.", | |
| ] | |
| ACKS = ["Got it — noted.", "That's good to hear.", "Thanks for the update!", | |
| "Ah, interesting.", "Makes sense.", "Oh wow, okay.", "Sounds like a plan.", | |
| "Noted, thanks for telling me."] | |
| Q_TEMPLATES = [ | |
| "Do you remember what my {h} is?", | |
| "Quick quiz: what's my {h} again?", | |
| "Remind me — what's my current {h}?", | |
| "What about my {h}? Do you have that on record?", | |
| ] | |
| Q_TEMPLATES_TEMPORAL = [ | |
| "What was my {h} back then, before it changed?", | |
| "What did my {h} used to be?", | |
| ] | |
| ANSWER_NEGATED = "Not anymore — you mentioned that's no longer the case." | |
| ANSWER_UNKNOWN = "I don't know — you haven't told me about that yet." | |
| ANSWER_TEMPORAL = "Back then it was {v}, though it has changed since." | |
| def spec_of(pred: str) -> Dict[str, Any]: | |
| return next(p for p in PREDICATES if p["p"] == pred) | |
| def find_fact(persona: Dict[str, Any], pred: str) -> Optional[Dict[str, Any]]: | |
| for f in persona["facts"]: | |
| if f["pred"] == pred: | |
| return f | |
| return None | |
| def make_persona(rng: random.Random, name: str, n_init: int = 8) -> Dict[str, Any]: | |
| preds = rng.sample(PREDICATES, n_init) | |
| facts = [] | |
| for spec in preds: | |
| facts.append({ | |
| "pred": spec["p"], "value": rng.choice(spec["v"]), | |
| "born_session": 0, "state": "active", "history": [], | |
| }) | |
| return {"name": name, "facts": facts} | |
| # ---------------------------------------------------------------- sessions | |
| def render_session( | |
| rng: random.Random, | |
| persona: Dict[str, Any], | |
| ledger_before: List[Slot], | |
| session_idx: int, | |
| n_add: int, n_upd: int, n_tmb: int, n_q: int, | |
| ) -> Tuple[List[Dict[str, str]], List[Dict[str, Any]], List[Dict[str, Any]]]: | |
| """Render one session. Returns (turns, write_events, qas). | |
| Question targets are constrained to what the ledger supports: | |
| predicate never stated .................... unknown | |
| predicate has a slot in ledger (v=1) ...... stale / superseded | |
| predicate has a tombstoned slot (v=0) ..... negated / temporal | |
| predicate active but EVICTED from ledger .. never asked | |
| """ | |
| turns: List[Dict[str, str]] = [] | |
| qas: List[Dict[str, Any]] = [] | |
| def say(user: str, assistant: Optional[str] = None) -> None: | |
| turns.append({"role": "user", "content": user}) | |
| turns.append({"role": "assistant", | |
| "content": rng.choice(ACKS) if assistant is None else assistant}) | |
| # ---- write events (chosen against persona state at session start) | |
| events: List[Dict[str, Any]] = [] | |
| for _ in range(n_add): | |
| open_specs = [p for p in PREDICATES if find_fact(persona, p["p"]) is None] | |
| if not open_specs: | |
| break | |
| spec = rng.choice(open_specs) | |
| value = rng.choice(spec["v"]) | |
| events.append({"kind": "ADD", "pred": spec["p"], "value": value, | |
| "text": spec["add"].replace("{v}", value)}) | |
| for _ in range(n_upd): | |
| updatable = [f for f in persona["facts"] | |
| if f["state"] == "active" and f["born_session"] < session_idx] | |
| if not updatable: | |
| break | |
| f = rng.choice(updatable) | |
| spec = spec_of(f["pred"]) | |
| value = rng.choice([v for v in spec["v"] if v != f["value"]]) | |
| events.append({"kind": "UPDATE", "pred": f["pred"], "value": value, | |
| "text": spec["upd"].replace("{v}", value)}) | |
| for _ in range(n_tmb): | |
| tmbable = [f for f in persona["facts"] if f["state"] == "active"] | |
| if not tmbable: | |
| break | |
| f = rng.choice(tmbable) | |
| spec = spec_of(f["pred"]) | |
| events.append({"kind": "TOMBSTONE", "pred": f["pred"], "value": None, | |
| "text": spec["tmb"]}) | |
| rng.shuffle(events) | |
| # ---- questions, answered from ledger_before (pre-consolidation state) | |
| in_ledger = {s.predicate for s in ledger_before if s.occupied} | |
| askable = [p for p in PREDICATES | |
| if find_fact(persona, p["p"]) is None or p["p"] in in_ledger] | |
| rng.shuffle(askable) | |
| for spec in askable[:n_q]: | |
| f = find_fact(persona, spec["p"]) | |
| q = rng.choice(Q_TEMPLATES).replace("{h}", spec["h"]) | |
| if f is None: | |
| qas.append({"q": q, "a": ANSWER_UNKNOWN, "type": "unknown", | |
| "pred": spec["p"], "value": None}) | |
| elif f["state"] == "tombstoned": | |
| old = f["value"] # value the tombstoned slot holds | |
| if old and rng.random() < 0.5: | |
| q2 = rng.choice(Q_TEMPLATES_TEMPORAL).replace("{h}", spec["h"]) | |
| qas.append({"q": q2, "a": ANSWER_TEMPORAL.replace("{v}", old), | |
| "type": "temporal", "pred": spec["p"], "value": old}) | |
| else: | |
| qas.append({"q": q, "a": ANSWER_NEGATED, "type": "negated", | |
| "pred": spec["p"], "value": None}) | |
| elif f["state"] == "updated": | |
| qas.append({"q": q, "a": f["value"], "type": "superseded", | |
| "pred": spec["p"], "value": f["value"]}) | |
| else: | |
| qas.append({"q": q, "a": f["value"], "type": "stale", | |
| "pred": spec["p"], "value": f["value"]}) | |
| # ---- render turns: statements (+ chatter), then the Q&A block | |
| n_chatter = rng.randint(1, 3) | |
| for ev in events: | |
| say(ev["text"]) | |
| if rng.random() < 0.35 and n_chatter > 0: | |
| say(rng.choice(CHATTER)) | |
| n_chatter -= 1 | |
| for qa in qas: | |
| say(qa["q"], qa["a"]) | |
| return turns, events, qas | |
| def apply_event_to_persona(persona: Dict[str, Any], ev: Dict[str, Any], | |
| session_idx: int) -> None: | |
| if ev["kind"] == "ADD": | |
| persona["facts"].append({"pred": ev["pred"], "value": ev["value"], | |
| "born_session": session_idx, "state": "active", | |
| "history": []}) | |
| elif ev["kind"] == "UPDATE": | |
| f = find_fact(persona, ev["pred"]) | |
| f["history"].append({"value": f["value"], "until_session": session_idx}) | |
| f["value"] = ev["value"] | |
| f["state"] = "updated" | |
| elif ev["kind"] == "TOMBSTONE": | |
| f = find_fact(persona, ev["pred"]) | |
| f["state"] = "tombstoned" | |
| # ------------------------------------------------------- gold op programs | |
| # | |
| # Invariants: | |
| # * ``ledger_now`` always equals the reducer applied to ``ledger`` plus | |
| # the ops appended so far. No re-simulation anywhere: every decision | |
| # (slot resolution, free-slot check, eviction victim) reads the same | |
| # single state the emitted program will produce. | |
| def gold_ops_for_events( | |
| ledger: List[Slot], | |
| persona_name: str, | |
| events: List[Dict[str, Any]], | |
| session_idx: int, | |
| ) -> Tuple[List[Dict[str, Any]], List[Slot]]: | |
| """Build the gold edit program for one session's events, in transcript | |
| order, then apply budget pressure with the salience oracle. | |
| Always-executable: if an ADD would overflow a full ledger, the program | |
| first EVICTs the lowest-salience occupied slot, then ADDs. If a | |
| predicate's slot was already evicted, an UPDATE degrades to an ADD and a | |
| TOMBSTONE of an absent slot is skipped.""" | |
| ops: List[Dict[str, Any]] = [] | |
| ledger_now = [Slot(**vars(s)) for s in ledger] | |
| for ev in events: | |
| if ev["kind"] == "ADD": | |
| if not any(not s.occupied for s in ledger_now): | |
| victim = min((s for s in ledger_now if s.occupied), | |
| key=lambda s: salience(s, session_idx)) | |
| evict = {"op": "EVICT", "i": victim.id} | |
| ops.append(evict) | |
| ledger_now, _ = apply_ops(ledger_now, [evict], session_idx) | |
| op = {"op": "ADD", "s": persona_name, "p": ev["pred"], "o": ev["value"]} | |
| elif ev["kind"] == "UPDATE": | |
| sid = next((s.id for s in ledger_now | |
| if s.occupied and s.subject == persona_name | |
| and s.predicate == ev["pred"]), None) | |
| if sid is None: | |
| # slot was evicted earlier: re-remember the fact as an ADD | |
| if not any(not s.occupied for s in ledger_now): | |
| victim = min((s for s in ledger_now if s.occupied), | |
| key=lambda s: salience(s, session_idx)) | |
| evict = {"op": "EVICT", "i": victim.id} | |
| ops.append(evict) | |
| ledger_now, _ = apply_ops(ledger_now, [evict], session_idx) | |
| op = {"op": "ADD", "s": persona_name, "p": ev["pred"], | |
| "o": ev["value"]} | |
| else: | |
| op = {"op": "UPDATE", "i": sid, "s": persona_name, | |
| "p": ev["pred"], "o": ev["value"]} | |
| else: # TOMBSTONE | |
| sid = next((s.id for s in ledger_now | |
| if s.occupied and s.subject == persona_name | |
| and s.predicate == ev["pred"]), None) | |
| if sid is None: | |
| continue | |
| op = {"op": "TOMBSTONE", "i": sid} | |
| ops.append(op) | |
| ledger_now, _ = apply_ops(ledger_now, [op], session_idx) | |
| # budget pressure at session end (normally already satisfied by the | |
| # evict-first rule, kept as a final guard) | |
| occupied = [s for s in ledger_now if s.occupied] | |
| for _ in range(max(0, len(occupied) - len(ledger_now))): | |
| victim = min((s for s in ledger_now if s.occupied), | |
| key=lambda s: salience(s, session_idx)) | |
| ops.append({"op": "EVICT", "i": victim.id}) | |
| ledger_now[victim.id] = Slot(id=victim.id) | |
| return ops, ledger_now | |
| # ---------------------------------------------------------------- timeline | |
| def generate_persona_timeline( | |
| rng: random.Random, name: str, n_sessions: int, budget: int, | |
| ) -> Dict[str, Any]: | |
| persona = make_persona(rng, name) | |
| ledger = empty_ledger(budget) | |
| sessions: List[Dict[str, Any]] = [] | |
| for s_idx in range(n_sessions): | |
| if s_idx == 0: | |
| # intro session: persona states the initial facts | |
| events = [{"kind": "ADD", "pred": f["pred"], "value": f["value"], | |
| "text": spec_of(f["pred"])["add"].replace("{v}", f["value"])} | |
| for f in persona["facts"]] | |
| rng.shuffle(events) | |
| turns: List[Dict[str, str]] = [] | |
| for ev in events: | |
| turns.append({"role": "user", "content": ev["text"]}) | |
| turns.append({"role": "assistant", "content": rng.choice(ACKS)}) | |
| qas: List[Dict[str, Any]] = [] | |
| else: | |
| turns, events, qas = render_session( | |
| rng, persona, ledger, s_idx, | |
| rng.randint(1, 2), rng.randint(0, 2), rng.randint(0, 1), | |
| rng.randint(1, 3)) | |
| for ev in events: | |
| apply_event_to_persona(persona, ev, s_idx) | |
| for qa in qas: | |
| for sl in ledger: | |
| if sl.occupied and sl.predicate == qa["pred"]: | |
| sl.n_uses += 1 | |
| ledger_before = serialize_ledger(ledger) | |
| ops, ledger_after = gold_ops_for_events(ledger, name, events, s_idx) | |
| ledger = ledger_after | |
| sessions.append({ | |
| "session_idx": s_idx, | |
| "turns": turns, | |
| "events": events, | |
| "qas": qas, | |
| "gold_ops": ops, | |
| "ledger_before": ledger_before, | |
| "ledger_after": serialize_ledger(ledger), | |
| }) | |
| return {"name": name, "sessions": sessions, | |
| "final_ledger": serialize_ledger(ledger)} | |
| def main() -> None: | |
| ap = argparse.ArgumentParser(description="CLERK synthetic session generator") | |
| ap.add_argument("--personas", type=int, default=100) | |
| ap.add_argument("--sessions", type=int, default=8) | |
| ap.add_argument("--budget", type=int, default=12, | |
| help="ledger budget; 12 guarantees eviction pressure " | |
| "(personas accumulate up to ~18 facts)") | |
| ap.add_argument("--seed", type=int, default=137) | |
| ap.add_argument("--split", choices=["train", "test"], default="train") | |
| ap.add_argument("--out", type=str, required=True) | |
| args = ap.parse_args() | |
| rng = random.Random(args.seed) | |
| names = TRAIN_NAMES if args.split == "train" else TEST_NAMES | |
| n = 0 | |
| with open(args.out, "w") as f: | |
| for i in range(args.personas): | |
| suffix = "" if i < len(names) else f" {i // len(names)}" | |
| name = names[i % len(names)] + suffix | |
| tl = generate_persona_timeline(rng, name, args.sessions, args.budget) | |
| f.write(json.dumps(tl, ensure_ascii=False) + "\n") | |
| n += 1 | |
| print(f"wrote {n} personas ({args.split}) to {args.out}") | |
| if __name__ == "__main__": | |
| main() |