Download app.py from dkappe/decider4b: direct link, hf CLI and curl.
- Browser
- Download file 10.2 kB
-
https://huggingface.co/spaces/dkappe/decider4b/resolve/main/app.py
- Command line
-
hf download hf://spaces/dkappe/decider4b/app.py
-
curl -L -o app.py https://huggingface.co/spaces/dkappe/decider4b/resolve/main/app.py
10.2 kB
| """decider-4b on ZeroGPU, in TypeSafe's /v1/systemone shape. | |
| Endpoint signature deliberately mirrors dkappe/IDU's `systemone` (state, questions, | |
| three temperature overrides) and returns the same (table, raw-json-string) tuple, so the | |
| existing jev-bench runner drives this Space without a special case. | |
| Scoring path is the authors' own reference implementation, vendored in `decider/` from | |
| https://huggingface.co/Mapika/decider-4b (Apache-2.0): prompt.py, model.py, systemone.py, | |
| infer.py. Three deviations from the reference, all noted here: | |
| * the checkpoint is loaded once at module scope and moved to CUDA eagerly, which is the | |
| ZeroGPU pattern the official decider demos use; | |
| * temperature is an explicit per-type argument rather than instance state, so | |
| concurrent requests on one worker cannot race each other. decider_config.json carries | |
| temperature_by_type {choice: 1.11, noul: 1.56, score: 1.287}, fitted per answer type, | |
| and serving every answer at the single 1.099 value would flatten the yes/no and Score | |
| probabilities; | |
| * questions are scored independently (one row each). The packed single-row path is not | |
| implemented because per-type temperature is applied per slot. | |
| A caller passes 0 for a temperature to keep the checkpoint's fitted value, which is the | |
| same sentinel IDU uses. | |
| """ | |
| import os | |
| os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") | |
| import spaces # noqa: E402 — must precede torch / transformers | |
| import json # noqa: E402 | |
| import time # noqa: E402 | |
| import gradio as gr # noqa: E402 | |
| import torch # noqa: E402 | |
| from huggingface_hub import hf_hub_download # noqa: E402 | |
| from decider.infer import Example, Q, neutralize_options # noqa: E402 | |
| from decider.model import DecisionModel, collate # noqa: E402 | |
| from decider.prompt import MAX_OPTIONS, build # noqa: E402 | |
| from decider.systemone import ( # noqa: E402 | |
| assemble, | |
| plan_rows, | |
| render_question, | |
| render_state, | |
| row_types, | |
| ) | |
| MODEL_ID = "Mapika/decider-4b" | |
| SPACE_REVISION = "0.1.0" | |
| CFG = json.load(open(hf_hub_download(MODEL_ID, "decider_config.json"))) | |
| VERSION = CFG.get("version", "dev") | |
| DEFAULT_T = float(CFG.get("temperature", 1.0)) | |
| T_BY_TYPE = {k: float(v) for k, v in (CFG.get("temperature_by_type") or {}).items()} | |
| NEUTRALIZE_NONE = bool(CFG.get("neutralize_none", True)) | |
| DEFAULT_ISOLATED = bool(CFG.get("isolated_levels", False)) | |
| MAX_STATE_TOKENS = int(CFG.get("max_state_tokens", 32768)) | |
| MAX_FWD_TOKENS = 65536 | |
| # Loaded once at module scope, eagerly moved to the (ZeroGPU-intercepted) CUDA device. | |
| MODEL = DecisionModel(MODEL_ID, dtype=torch.bfloat16, grad_ckpt=False).to("cuda").eval() | |
| TOK = MODEL.tok | |
| class _Keep: | |
| """The reference's rng stand-in: keep the caller's option order, never shuffle.""" | |
| def shuffle(self, x): | |
| pass | |
| def sample(self, xs, k): | |
| return xs[:k] | |
| def _temps(tc, ts, tn): | |
| """Per-answer-type temperatures. 0 means 'keep the checkpoint's fitted value'.""" | |
| d = dict(T_BY_TYPE) | |
| for k in ("choice", "noul", "score"): | |
| d.setdefault(k, DEFAULT_T) | |
| if tc and float(tc) > 0: | |
| d["choice"] = float(tc) | |
| if ts and float(ts) > 0: | |
| d["noul"] = float(ts) | |
| if tn and float(tn) > 0: | |
| d["score"] = float(tn) | |
| return d | |
| def _system_one(state, questions, temps, isolated=None): | |
| """Reference `Decider.system_one`, eager path, with per-slot temperature.""" | |
| if isolated is None: | |
| isolated = DEFAULT_ISOLATED | |
| ctx = render_state(state) | |
| rqs = {k: render_question(v) for k, v in questions.items()} | |
| opts = ((lambda r: neutralize_options(r["options"])[0]) if NEUTRALIZE_NONE | |
| else (lambda r: list(r["options"]))) | |
| flat, index = plan_rows(rqs, isolated) | |
| items = [ | |
| build(Example(ctx, [Q(r["question"], opts(r), 0)]), TOK, _Keep(), | |
| max_options=MAX_OPTIONS, max_ctx_tokens=MAX_STATE_TOKENS, | |
| layout="state_first") | |
| for r in flat | |
| ] | |
| rtypes = row_types(rqs, index) | |
| probs = [] | |
| with torch.no_grad(): | |
| per = max(1, MAX_FWD_TOKENS // max(len(it["ids"]) for it in items)) | |
| for i in range(0, len(items), per): | |
| chunk = items[i:i + per] | |
| bt = collate(chunk, TOK.pad_token_id) | |
| lg = MODEL.slot_logits( | |
| *[bt[k].to("cuda") for k in | |
| ("input_ids", "attention_mask", "slot_idx", "slot_batch", "nopts")] | |
| ) | |
| # one slot per item, so slot order follows item order | |
| tvec = torch.tensor([temps.get(rtypes[i + j], DEFAULT_T) | |
| for j in range(len(chunk))], | |
| device=lg.device, dtype=lg.dtype) | |
| pr = torch.softmax(lg / tvec.unsqueeze(-1), -1).float().cpu() | |
| c = 0 | |
| for it in chunk: | |
| probs.append(pr[c:c + len(it["slots"])]) | |
| c += len(it["slots"]) | |
| flatp = [p.tolist() for ps in probs for p in ps] | |
| answers = assemble(rqs, index, flatp) | |
| n_tokens = sum(len(it["ids"]) for it in items) | |
| return { | |
| "model": f"decider-4b-{VERSION}", | |
| "answers": answers, | |
| "usage": {"input_tokens": n_tokens, "output_tokens": 0, "rows": len(items)}, | |
| } | |
| def _table(raw): | |
| rows = [] | |
| for k, a in raw["answers"].items(): | |
| if a.get("type") == "choice": | |
| rows.append([k, a.get("choice"), a.get("confidence")]) | |
| elif a.get("type") == "score": | |
| rows.append([k, a.get("score"), a.get("confidence")]) | |
| else: | |
| rows.append([k, a.get("noul"), None]) | |
| return {"headers": ["question", "answer", "confidence"], "data": rows, | |
| "metadata": None} | |
| def answer_systemone(state_text: str, q_text: str, tc: float = 0, ts: float = 0, | |
| tn: float = 0) -> tuple: | |
| """Score typed questions against a state and return the Jev shape. | |
| Args: | |
| state_text: the state, free text or JSON. | |
| q_text: the questions object, as JSON. | |
| tc, ts, tn: temperature overrides for choice / noul / score. 0 keeps the | |
| checkpoint's fitted value. | |
| """ | |
| t0 = time.time() | |
| if isinstance(q_text, str): | |
| questions = json.loads(q_text) | |
| else: | |
| questions = q_text | |
| if not isinstance(questions, dict) or not questions: | |
| return ({"headers": ["question", "answer", "confidence"], "data": [], "metadata": None}, | |
| json.dumps({"error": "questions must be a non-empty JSON object"})) | |
| if isinstance(state_text, str) and state_text.strip().startswith(("{", "[")): | |
| try: | |
| state = json.loads(state_text) | |
| except Exception: | |
| state = state_text | |
| else: | |
| state = state_text | |
| raw = _system_one(state, questions, _temps(tc, ts, tn)) | |
| raw["space"] = f"dkappe/decider4b {SPACE_REVISION}" | |
| raw["device"] = str(MODEL.model.device) if hasattr(MODEL, "model") else "cuda" | |
| raw["elapsed_s"] = round(time.time() - t0, 2) | |
| return _table(raw), json.dumps(raw, ensure_ascii=False, default=str) | |
| def health() -> str: | |
| return (f"decider-4b {VERSION} on ZeroGPU\n" | |
| f"temperatures: base {DEFAULT_T}, by type {T_BY_TYPE}\n" | |
| f"isolated_levels={DEFAULT_ISOLATED} neutralize_none={NEUTRALIZE_NONE}\n" | |
| f"max_state_tokens={MAX_STATE_TOKENS} max_options={MAX_OPTIONS}") | |
| DEMO_Q = { | |
| "department": { | |
| "type": "choice", | |
| "instructions": "Which team should handle this?", | |
| "criteria": { | |
| "billing": "Money: charges, invoices, refunds, duplicate payments.", | |
| "technical": "The software misbehaving: bugs, errors, crashes, outages.", | |
| "account": "Logging in, passwords, 2FA, locked accounts, permissions.", | |
| "returns": "Physical goods: exchanges, damaged items, shipping, delivery.", | |
| "other": "None of the above: legal, press, partnerships, hiring.", | |
| }, | |
| }, | |
| "frustration": { | |
| "type": "score", | |
| "instructions": "How frustrated is the writer?", | |
| "criteria": [ | |
| "Making a request or asking a question, no problem reported.", | |
| "Reporting a problem and staying civil.", | |
| "Angry and escalating: demands immediate action, threatens to cancel.", | |
| ], | |
| }, | |
| "refund_requested": { | |
| "type": "noul", | |
| "instructions": "Is the writer asking for money back?", | |
| }, | |
| } | |
| DEMO_STATE = ("I was charged twice for order C-26. Please refund the duplicate charge.\n" | |
| "Order history: order C-26 charged $310 twice on 2026-09-19.\n" | |
| "Refund policy: duplicate charges are refunded when the same order is " | |
| "charged twice.") | |
| with gr.Blocks(title="decider-4b systemone") as demo: | |
| gr.Markdown( | |
| "# decider-4b, Jev-shaped decisions\n" | |
| "`choice` / `score` / `noul` questions scored in one forward pass, with calibrated " | |
| "probabilities. 4B dense, Qwen3.5-4B-Base, Apache 2.0. " | |
| "The endpoint mirrors `dkappe/IDU`'s `systemone` so one benchmark harness drives " | |
| "both. A temperature of 0 keeps the checkpoint's fitted per-type value." | |
| ) | |
| with gr.Tab("systemone"): | |
| st = gr.Code(label="state", language="json", value=DEMO_STATE, lines=6) | |
| qt = gr.Code(label="questions", language="json", | |
| value=json.dumps(DEMO_Q, indent=2), lines=18) | |
| with gr.Row(): | |
| tc = gr.Number(label="choice temp (0 = fitted)", value=0) | |
| ts = gr.Number(label="noul temp (0 = fitted)", value=0) | |
| tn = gr.Number(label="score temp (0 = fitted)", value=0) | |
| run = gr.Button("decide", variant="primary") | |
| out_rows = gr.Dataframe(label="answers") | |
| out_raw = gr.Code(label="raw response", language="json", lines=16) | |
| run.click(answer_systemone, [st, qt, tc, ts, tn], [out_rows, out_raw], | |
| api_name="systemone") | |
| with gr.Tab("health"): | |
| hb = gr.Button("check") | |
| ho = gr.Textbox(label="status", lines=6) | |
| hb.click(health, None, ho, api_name="health") | |
| if __name__ == "__main__": | |
| demo.launch() | |