decider4b / app.py
dkappe's picture
decider-4b systemone wrapper on ZeroGPU
0a16466 verified
Raw History Blame Contribute Delete
10.2 kB
"""decider-4b on ZeroGPU, in TypeSafe's /v1/systemone shape.
Endpoint signature deliberately mirrors dkappe/IDU's `systemone` (state, questions,
three temperature overrides) and returns the same (table, raw-json-string) tuple, so the
existing jev-bench runner drives this Space without a special case.
Scoring path is the authors' own reference implementation, vendored in `decider/` from
https://huggingface.co/Mapika/decider-4b (Apache-2.0): prompt.py, model.py, systemone.py,
infer.py. Three deviations from the reference, all noted here:
* the checkpoint is loaded once at module scope and moved to CUDA eagerly, which is the
ZeroGPU pattern the official decider demos use;
* temperature is an explicit per-type argument rather than instance state, so
concurrent requests on one worker cannot race each other. decider_config.json carries
temperature_by_type {choice: 1.11, noul: 1.56, score: 1.287}, fitted per answer type,
and serving every answer at the single 1.099 value would flatten the yes/no and Score
probabilities;
* questions are scored independently (one row each). The packed single-row path is not
implemented because per-type temperature is applied per slot.
A caller passes 0 for a temperature to keep the checkpoint's fitted value, which is the
same sentinel IDU uses.
"""
import os
os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
import spaces # noqa: E402 — must precede torch / transformers
import json # noqa: E402
import time # noqa: E402
import gradio as gr # noqa: E402
import torch # noqa: E402
from huggingface_hub import hf_hub_download # noqa: E402
from decider.infer import Example, Q, neutralize_options # noqa: E402
from decider.model import DecisionModel, collate # noqa: E402
from decider.prompt import MAX_OPTIONS, build # noqa: E402
from decider.systemone import ( # noqa: E402
assemble,
plan_rows,
render_question,
render_state,
row_types,
)
MODEL_ID = "Mapika/decider-4b"
SPACE_REVISION = "0.1.0"
CFG = json.load(open(hf_hub_download(MODEL_ID, "decider_config.json")))
VERSION = CFG.get("version", "dev")
DEFAULT_T = float(CFG.get("temperature", 1.0))
T_BY_TYPE = {k: float(v) for k, v in (CFG.get("temperature_by_type") or {}).items()}
NEUTRALIZE_NONE = bool(CFG.get("neutralize_none", True))
DEFAULT_ISOLATED = bool(CFG.get("isolated_levels", False))
MAX_STATE_TOKENS = int(CFG.get("max_state_tokens", 32768))
MAX_FWD_TOKENS = 65536
# Loaded once at module scope, eagerly moved to the (ZeroGPU-intercepted) CUDA device.
MODEL = DecisionModel(MODEL_ID, dtype=torch.bfloat16, grad_ckpt=False).to("cuda").eval()
TOK = MODEL.tok
class _Keep:
"""The reference's rng stand-in: keep the caller's option order, never shuffle."""
def shuffle(self, x):
pass
def sample(self, xs, k):
return xs[:k]
def _temps(tc, ts, tn):
"""Per-answer-type temperatures. 0 means 'keep the checkpoint's fitted value'."""
d = dict(T_BY_TYPE)
for k in ("choice", "noul", "score"):
d.setdefault(k, DEFAULT_T)
if tc and float(tc) > 0:
d["choice"] = float(tc)
if ts and float(ts) > 0:
d["noul"] = float(ts)
if tn and float(tn) > 0:
d["score"] = float(tn)
return d
def _system_one(state, questions, temps, isolated=None):
"""Reference `Decider.system_one`, eager path, with per-slot temperature."""
if isolated is None:
isolated = DEFAULT_ISOLATED
ctx = render_state(state)
rqs = {k: render_question(v) for k, v in questions.items()}
opts = ((lambda r: neutralize_options(r["options"])[0]) if NEUTRALIZE_NONE
else (lambda r: list(r["options"])))
flat, index = plan_rows(rqs, isolated)
items = [
build(Example(ctx, [Q(r["question"], opts(r), 0)]), TOK, _Keep(),
max_options=MAX_OPTIONS, max_ctx_tokens=MAX_STATE_TOKENS,
layout="state_first")
for r in flat
]
rtypes = row_types(rqs, index)
probs = []
with torch.no_grad():
per = max(1, MAX_FWD_TOKENS // max(len(it["ids"]) for it in items))
for i in range(0, len(items), per):
chunk = items[i:i + per]
bt = collate(chunk, TOK.pad_token_id)
lg = MODEL.slot_logits(
*[bt[k].to("cuda") for k in
("input_ids", "attention_mask", "slot_idx", "slot_batch", "nopts")]
)
# one slot per item, so slot order follows item order
tvec = torch.tensor([temps.get(rtypes[i + j], DEFAULT_T)
for j in range(len(chunk))],
device=lg.device, dtype=lg.dtype)
pr = torch.softmax(lg / tvec.unsqueeze(-1), -1).float().cpu()
c = 0
for it in chunk:
probs.append(pr[c:c + len(it["slots"])])
c += len(it["slots"])
flatp = [p.tolist() for ps in probs for p in ps]
answers = assemble(rqs, index, flatp)
n_tokens = sum(len(it["ids"]) for it in items)
return {
"model": f"decider-4b-{VERSION}",
"answers": answers,
"usage": {"input_tokens": n_tokens, "output_tokens": 0, "rows": len(items)},
}
def _table(raw):
rows = []
for k, a in raw["answers"].items():
if a.get("type") == "choice":
rows.append([k, a.get("choice"), a.get("confidence")])
elif a.get("type") == "score":
rows.append([k, a.get("score"), a.get("confidence")])
else:
rows.append([k, a.get("noul"), None])
return {"headers": ["question", "answer", "confidence"], "data": rows,
"metadata": None}
@spaces.GPU(duration=60)
def answer_systemone(state_text: str, q_text: str, tc: float = 0, ts: float = 0,
tn: float = 0) -> tuple:
"""Score typed questions against a state and return the Jev shape.
Args:
state_text: the state, free text or JSON.
q_text: the questions object, as JSON.
tc, ts, tn: temperature overrides for choice / noul / score. 0 keeps the
checkpoint's fitted value.
"""
t0 = time.time()
if isinstance(q_text, str):
questions = json.loads(q_text)
else:
questions = q_text
if not isinstance(questions, dict) or not questions:
return ({"headers": ["question", "answer", "confidence"], "data": [], "metadata": None},
json.dumps({"error": "questions must be a non-empty JSON object"}))
if isinstance(state_text, str) and state_text.strip().startswith(("{", "[")):
try:
state = json.loads(state_text)
except Exception:
state = state_text
else:
state = state_text
raw = _system_one(state, questions, _temps(tc, ts, tn))
raw["space"] = f"dkappe/decider4b {SPACE_REVISION}"
raw["device"] = str(MODEL.model.device) if hasattr(MODEL, "model") else "cuda"
raw["elapsed_s"] = round(time.time() - t0, 2)
return _table(raw), json.dumps(raw, ensure_ascii=False, default=str)
@spaces.GPU(duration=5)
def health() -> str:
return (f"decider-4b {VERSION} on ZeroGPU\n"
f"temperatures: base {DEFAULT_T}, by type {T_BY_TYPE}\n"
f"isolated_levels={DEFAULT_ISOLATED} neutralize_none={NEUTRALIZE_NONE}\n"
f"max_state_tokens={MAX_STATE_TOKENS} max_options={MAX_OPTIONS}")
DEMO_Q = {
"department": {
"type": "choice",
"instructions": "Which team should handle this?",
"criteria": {
"billing": "Money: charges, invoices, refunds, duplicate payments.",
"technical": "The software misbehaving: bugs, errors, crashes, outages.",
"account": "Logging in, passwords, 2FA, locked accounts, permissions.",
"returns": "Physical goods: exchanges, damaged items, shipping, delivery.",
"other": "None of the above: legal, press, partnerships, hiring.",
},
},
"frustration": {
"type": "score",
"instructions": "How frustrated is the writer?",
"criteria": [
"Making a request or asking a question, no problem reported.",
"Reporting a problem and staying civil.",
"Angry and escalating: demands immediate action, threatens to cancel.",
],
},
"refund_requested": {
"type": "noul",
"instructions": "Is the writer asking for money back?",
},
}
DEMO_STATE = ("I was charged twice for order C-26. Please refund the duplicate charge.\n"
"Order history: order C-26 charged $310 twice on 2026-09-19.\n"
"Refund policy: duplicate charges are refunded when the same order is "
"charged twice.")
with gr.Blocks(title="decider-4b systemone") as demo:
gr.Markdown(
"# decider-4b, Jev-shaped decisions\n"
"`choice` / `score` / `noul` questions scored in one forward pass, with calibrated "
"probabilities. 4B dense, Qwen3.5-4B-Base, Apache 2.0. "
"The endpoint mirrors `dkappe/IDU`'s `systemone` so one benchmark harness drives "
"both. A temperature of 0 keeps the checkpoint's fitted per-type value."
)
with gr.Tab("systemone"):
st = gr.Code(label="state", language="json", value=DEMO_STATE, lines=6)
qt = gr.Code(label="questions", language="json",
value=json.dumps(DEMO_Q, indent=2), lines=18)
with gr.Row():
tc = gr.Number(label="choice temp (0 = fitted)", value=0)
ts = gr.Number(label="noul temp (0 = fitted)", value=0)
tn = gr.Number(label="score temp (0 = fitted)", value=0)
run = gr.Button("decide", variant="primary")
out_rows = gr.Dataframe(label="answers")
out_raw = gr.Code(label="raw response", language="json", lines=16)
run.click(answer_systemone, [st, qt, tc, ts, tn], [out_rows, out_raw],
api_name="systemone")
with gr.Tab("health"):
hb = gr.Button("check")
ho = gr.Textbox(label="status", lines=6)
hb.click(health, None, ho, api_name="health")
if __name__ == "__main__":
demo.launch()