"""decider-4b on ZeroGPU, in TypeSafe's /v1/systemone shape. Endpoint signature deliberately mirrors dkappe/IDU's `systemone` (state, questions, three temperature overrides) and returns the same (table, raw-json-string) tuple, so the existing jev-bench runner drives this Space without a special case. Scoring path is the authors' own reference implementation, vendored in `decider/` from https://huggingface.co/Mapika/decider-4b (Apache-2.0): prompt.py, model.py, systemone.py, infer.py. Three deviations from the reference, all noted here: * the checkpoint is loaded once at module scope and moved to CUDA eagerly, which is the ZeroGPU pattern the official decider demos use; * temperature is an explicit per-type argument rather than instance state, so concurrent requests on one worker cannot race each other. decider_config.json carries temperature_by_type {choice: 1.11, noul: 1.56, score: 1.287}, fitted per answer type, and serving every answer at the single 1.099 value would flatten the yes/no and Score probabilities; * questions are scored independently (one row each). The packed single-row path is not implemented because per-type temperature is applied per slot. A caller passes 0 for a temperature to keep the checkpoint's fitted value, which is the same sentinel IDU uses. """ import os os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") import spaces # noqa: E402 — must precede torch / transformers import json # noqa: E402 import time # noqa: E402 import gradio as gr # noqa: E402 import torch # noqa: E402 from huggingface_hub import hf_hub_download # noqa: E402 from decider.infer import Example, Q, neutralize_options # noqa: E402 from decider.model import DecisionModel, collate # noqa: E402 from decider.prompt import MAX_OPTIONS, build # noqa: E402 from decider.systemone import ( # noqa: E402 assemble, plan_rows, render_question, render_state, row_types, ) MODEL_ID = "Mapika/decider-4b" SPACE_REVISION = "0.1.0" CFG = json.load(open(hf_hub_download(MODEL_ID, "decider_config.json"))) VERSION = CFG.get("version", "dev") DEFAULT_T = float(CFG.get("temperature", 1.0)) T_BY_TYPE = {k: float(v) for k, v in (CFG.get("temperature_by_type") or {}).items()} NEUTRALIZE_NONE = bool(CFG.get("neutralize_none", True)) DEFAULT_ISOLATED = bool(CFG.get("isolated_levels", False)) MAX_STATE_TOKENS = int(CFG.get("max_state_tokens", 32768)) MAX_FWD_TOKENS = 65536 # Loaded once at module scope, eagerly moved to the (ZeroGPU-intercepted) CUDA device. MODEL = DecisionModel(MODEL_ID, dtype=torch.bfloat16, grad_ckpt=False).to("cuda").eval() TOK = MODEL.tok class _Keep: """The reference's rng stand-in: keep the caller's option order, never shuffle.""" def shuffle(self, x): pass def sample(self, xs, k): return xs[:k] def _temps(tc, ts, tn): """Per-answer-type temperatures. 0 means 'keep the checkpoint's fitted value'.""" d = dict(T_BY_TYPE) for k in ("choice", "noul", "score"): d.setdefault(k, DEFAULT_T) if tc and float(tc) > 0: d["choice"] = float(tc) if ts and float(ts) > 0: d["noul"] = float(ts) if tn and float(tn) > 0: d["score"] = float(tn) return d def _system_one(state, questions, temps, isolated=None): """Reference `Decider.system_one`, eager path, with per-slot temperature.""" if isolated is None: isolated = DEFAULT_ISOLATED ctx = render_state(state) rqs = {k: render_question(v) for k, v in questions.items()} opts = ((lambda r: neutralize_options(r["options"])[0]) if NEUTRALIZE_NONE else (lambda r: list(r["options"]))) flat, index = plan_rows(rqs, isolated) items = [ build(Example(ctx, [Q(r["question"], opts(r), 0)]), TOK, _Keep(), max_options=MAX_OPTIONS, max_ctx_tokens=MAX_STATE_TOKENS, layout="state_first") for r in flat ] rtypes = row_types(rqs, index) probs = [] with torch.no_grad(): per = max(1, MAX_FWD_TOKENS // max(len(it["ids"]) for it in items)) for i in range(0, len(items), per): chunk = items[i:i + per] bt = collate(chunk, TOK.pad_token_id) lg = MODEL.slot_logits( *[bt[k].to("cuda") for k in ("input_ids", "attention_mask", "slot_idx", "slot_batch", "nopts")] ) # one slot per item, so slot order follows item order tvec = torch.tensor([temps.get(rtypes[i + j], DEFAULT_T) for j in range(len(chunk))], device=lg.device, dtype=lg.dtype) pr = torch.softmax(lg / tvec.unsqueeze(-1), -1).float().cpu() c = 0 for it in chunk: probs.append(pr[c:c + len(it["slots"])]) c += len(it["slots"]) flatp = [p.tolist() for ps in probs for p in ps] answers = assemble(rqs, index, flatp) n_tokens = sum(len(it["ids"]) for it in items) return { "model": f"decider-4b-{VERSION}", "answers": answers, "usage": {"input_tokens": n_tokens, "output_tokens": 0, "rows": len(items)}, } def _table(raw): rows = [] for k, a in raw["answers"].items(): if a.get("type") == "choice": rows.append([k, a.get("choice"), a.get("confidence")]) elif a.get("type") == "score": rows.append([k, a.get("score"), a.get("confidence")]) else: rows.append([k, a.get("noul"), None]) return {"headers": ["question", "answer", "confidence"], "data": rows, "metadata": None} @spaces.GPU(duration=60) def answer_systemone(state_text: str, q_text: str, tc: float = 0, ts: float = 0, tn: float = 0) -> tuple: """Score typed questions against a state and return the Jev shape. Args: state_text: the state, free text or JSON. q_text: the questions object, as JSON. tc, ts, tn: temperature overrides for choice / noul / score. 0 keeps the checkpoint's fitted value. """ t0 = time.time() if isinstance(q_text, str): questions = json.loads(q_text) else: questions = q_text if not isinstance(questions, dict) or not questions: return ({"headers": ["question", "answer", "confidence"], "data": [], "metadata": None}, json.dumps({"error": "questions must be a non-empty JSON object"})) if isinstance(state_text, str) and state_text.strip().startswith(("{", "[")): try: state = json.loads(state_text) except Exception: state = state_text else: state = state_text raw = _system_one(state, questions, _temps(tc, ts, tn)) raw["space"] = f"dkappe/decider4b {SPACE_REVISION}" raw["device"] = str(MODEL.model.device) if hasattr(MODEL, "model") else "cuda" raw["elapsed_s"] = round(time.time() - t0, 2) return _table(raw), json.dumps(raw, ensure_ascii=False, default=str) @spaces.GPU(duration=5) def health() -> str: return (f"decider-4b {VERSION} on ZeroGPU\n" f"temperatures: base {DEFAULT_T}, by type {T_BY_TYPE}\n" f"isolated_levels={DEFAULT_ISOLATED} neutralize_none={NEUTRALIZE_NONE}\n" f"max_state_tokens={MAX_STATE_TOKENS} max_options={MAX_OPTIONS}") DEMO_Q = { "department": { "type": "choice", "instructions": "Which team should handle this?", "criteria": { "billing": "Money: charges, invoices, refunds, duplicate payments.", "technical": "The software misbehaving: bugs, errors, crashes, outages.", "account": "Logging in, passwords, 2FA, locked accounts, permissions.", "returns": "Physical goods: exchanges, damaged items, shipping, delivery.", "other": "None of the above: legal, press, partnerships, hiring.", }, }, "frustration": { "type": "score", "instructions": "How frustrated is the writer?", "criteria": [ "Making a request or asking a question, no problem reported.", "Reporting a problem and staying civil.", "Angry and escalating: demands immediate action, threatens to cancel.", ], }, "refund_requested": { "type": "noul", "instructions": "Is the writer asking for money back?", }, } DEMO_STATE = ("I was charged twice for order C-26. Please refund the duplicate charge.\n" "Order history: order C-26 charged $310 twice on 2026-09-19.\n" "Refund policy: duplicate charges are refunded when the same order is " "charged twice.") with gr.Blocks(title="decider-4b systemone") as demo: gr.Markdown( "# decider-4b, Jev-shaped decisions\n" "`choice` / `score` / `noul` questions scored in one forward pass, with calibrated " "probabilities. 4B dense, Qwen3.5-4B-Base, Apache 2.0. " "The endpoint mirrors `dkappe/IDU`'s `systemone` so one benchmark harness drives " "both. A temperature of 0 keeps the checkpoint's fitted per-type value." ) with gr.Tab("systemone"): st = gr.Code(label="state", language="json", value=DEMO_STATE, lines=6) qt = gr.Code(label="questions", language="json", value=json.dumps(DEMO_Q, indent=2), lines=18) with gr.Row(): tc = gr.Number(label="choice temp (0 = fitted)", value=0) ts = gr.Number(label="noul temp (0 = fitted)", value=0) tn = gr.Number(label="score temp (0 = fitted)", value=0) run = gr.Button("decide", variant="primary") out_rows = gr.Dataframe(label="answers") out_raw = gr.Code(label="raw response", language="json", lines=16) run.click(answer_systemone, [st, qt, tc, ts, tn], [out_rows, out_raw], api_name="systemone") with gr.Tab("health"): hb = gr.Button("check") ho = gr.Textbox(label="status", lines=6) hb.click(health, None, ho, api_name="health") if __name__ == "__main__": demo.launch()