Download app.py from dkappe/IDU: direct link, hf CLI and curl.
- Browser
- Download file 38.6 kB
-
https://huggingface.co/spaces/dkappe/IDU/resolve/main/app.py
- Command line
-
hf download hf://spaces/dkappe/IDU/app.py
-
curl -L -o app.py https://huggingface.co/spaces/dkappe/IDU/resolve/main/app.py
38.6 kB
| #!/usr/bin/env python3 | |
| """ | |
| IDU -- a Jev-shaped decision service. | |
| NAME: JEV shifted one letter back, the same shift that turns IBM into HAL. Jev is | |
| TypeSafe's typed-decision model; IDU is the same request/response shape built on open | |
| weights you can run yourself. | |
| WHAT IT DOES | |
| Give it a `state` (text or JSON) and typed `questions`; it returns typed answers with | |
| probabilities and confidence. Three primitives, matching Jev exactly: | |
| choice -> pick one option, probabilities across all options, confidence | |
| score -> position on an ordered rubric (can land BETWEEN levels), distribution | |
| noul -> one number: the probability the statement is true | |
| BACKENDS (two, same request shape) | |
| laya convaiinnovations/laya (Apache 2.0). 421M ModernBERT-large encoder with a | |
| decision head, trained by RL against strictly proper scoring rules. Chosen for | |
| three concrete reasons: | |
| * It IS the Jev shape -- `state` + typed questions in, typed decisions out. | |
| No adapter layer, no tool-calling workaround, no triggers. | |
| * Every option is scored at its own [MASK] token, then softmaxed over that | |
| question's options, so options are judged independently. That is what Jev | |
| documents ("each level is judged on its own against the state") and it makes | |
| the polarity problem in a tool-calling model disappear: a noul question is | |
| just two markers, not two competing armed tools. | |
| * All questions in one call are answered in ONE forward pass, and it is | |
| PyTorch, so it fits the ZeroGPU runtime natively. | |
| decider Mapika/decider-4b (Apache 2.0). 4B dense on Qwen3.5-4B-Base, one pass, no | |
| decoding. Added because it takes the SAME TypeSafe request shape -- the authors | |
| ship `systemone.py` implementing /v1/systemone -- so the whole adapter is the | |
| backend branch below, and because it has no small option-count cap (255 options, | |
| 32k state tokens) where other open decision models stop at 16 options. It also | |
| needs transformers>=5.17.0, which is why this Space requires that floor. | |
| Both backends are selectable per call. They are NOT comparable for latency: both run on | |
| the same ZeroGPU pool, so wall time measures queueing as much as compute. Accuracy is the | |
| comparison; timing is not. | |
| IMPORT ORDER IS LOAD-BEARING, NOT STYLE | |
| `spaces` MUST be imported before torch. The spaces library patches torch's CUDA | |
| initialisation so a GPU can be attached per call; if torch is imported first the patch | |
| lands too late and every decorated call fails inside worker_init with | |
| "RuntimeError: No CUDA GPUs are available". Nothing that pulls in torch -- gradio, | |
| transformers, laya, safetensors -- may be imported above that block. This is copied | |
| deliberately from the model's own demo Space, which documents the same trap. | |
| """ | |
| from __future__ import annotations | |
| import os | |
| # ---------------------------------------------------------------- import order guard | |
| DEVICE_MODE = os.environ.get("IDU_DEVICE", "zerogpu") # "zerogpu" or "cpu" | |
| ZERO = False | |
| if DEVICE_MODE == "zerogpu": | |
| try: | |
| import spaces # noqa: E402 MUST precede torch | |
| GPU = spaces.GPU(duration=120) | |
| os.environ["IDU_CUDA"] = "1" | |
| ZERO = True | |
| except Exception as _e: # pragma: no cover | |
| print(f"ZeroGPU unavailable ({type(_e).__name__}), running on CPU", flush=True) | |
| if not ZERO: | |
| os.environ.pop("IDU_CUDA", None) | |
| def GPU(fn): | |
| return fn | |
| import json # noqa: E402 | |
| import sys # noqa: E402 | |
| import time # noqa: E402 | |
| import traceback # noqa: E402 | |
| from typing import Any # noqa: E402 | |
| print(f"import order: spaces before torch = {'torch' not in sys.modules}", flush=True) | |
| import gradio as gr # noqa: E402 (pulls torch) | |
| MODEL_ID = "convaiinnovations/laya" | |
| DECIDER_ID = "Mapika/decider-4b" | |
| SPACE_REVISION = "0.2.0" | |
| _AGENT: dict[str, Any] = {} | |
| # Temperature: the checkpoint SHIPS FITTED, not raw. | |
| # rl_agent_config.json: temperature [1.6369, 1.2514, 1.9834] for [choice, score, noul], | |
| # plus temperature_by_options buckets ("choice:2", "choice:3-5", "noul:2", ...). | |
| # The model card reports mean ECE 0.466 -> 0.081 after fitting, and that fitting is already | |
| # baked into these files. So the defaults here are the calibrated path; overriding them is | |
| # for experimenting, not for fixing something broken. | |
| # | |
| # The app reads the real values from the loaded agent (see _current_temperature) instead of | |
| # hardcoding TEMP_DEFAULT as truth -- an earlier version reported [1.0, 1.0, 1.0] here, which | |
| # was simply false about what was in force. | |
| TEMP_LABELS = ("choice", "score", "noul") | |
| def _current_temperature(a) -> list: | |
| """The per-type temperature actually in force, from the agent.""" | |
| t = getattr(a, "temperature", None) | |
| if t is None: | |
| return [1.0, 1.0, 1.0] | |
| return [float(x) for x in t] | |
| def get_agent(): | |
| """Build the Laya agent once, at module scope, ON CPU. | |
| WHY CPU AT MODULE SCOPE, EVEN ON ZEROGPU | |
| ZeroGPU attaches a GPU only while a decorated function runs, and it does so in a | |
| worker process. Building the model with device="cuda" at STARTUP therefore runs with | |
| no GPU present: torch reports "No CUDA GPUs are available" and the checkpoint ends up | |
| bound to CPU anyway. Worse, the mismatch then surfaces INSIDE the decorated call as | |
| `RuntimeError: No CUDA GPUs are available` raised from `torch._C._cuda_init()`. | |
| So the model is built once on CPU here, and `answer_systemone` moves it onto the GPU | |
| from inside the decorated context, where a device actually exists. The weights are | |
| downloaded and the tokenizer built at startup either way, which is the part that | |
| matters for first-click latency. | |
| """ | |
| if _AGENT.get("a") is not None: | |
| return _AGENT["a"] | |
| import laya | |
| a = laya.load(MODEL_ID, device="cpu") | |
| _AGENT["a"] = a | |
| _AGENT["laya"] = laya | |
| return a | |
| def _move_to_gpu(a) -> str: | |
| """Move the agent to CUDA from inside a decorated call. Returns the device in use. | |
| Safe to call repeatedly; a no-op once the model is already on the GPU. Falls back to | |
| CPU with a printed reason rather than raising, so a GPU we cannot use degrades to a | |
| slower answer instead of an error. | |
| NOTE ON WORKER PROCESSES | |
| ZeroGPU runs each decorated call in its own worker process. A module-level cache of | |
| the agent is therefore NOT reliably shared across calls: the child may start with a | |
| fresh interpreter where `_AGENT` is empty, or inherit a copy-on-write view whose | |
| tensors were never on the GPU. The symptom is `No CUDA GPUs are available` from | |
| `torch._C._cuda_init()` INSIDE the call, even though startup looked healthy. | |
| `get_agent()` therefore rebuilds if the cache is missing rather than assuming it is | |
| there, and this function re-attaches the device every call. | |
| """ | |
| import torch | |
| if not ZERO: | |
| return str(a.device) | |
| try: | |
| if not torch.cuda.is_available(): | |
| print("ZeroGPU: no CUDA visible inside the call; staying on CPU", flush=True) | |
| return str(a.device) | |
| if a.device.type != "cuda" or next(a.model.parameters()).device.type != "cuda": | |
| a.device = torch.device("cuda") | |
| # bf16 is the checkpoint's amp dtype; the Space GPU supports it. | |
| if getattr(a, "dtype", None) is not None: | |
| try: | |
| a.dtype = torch.bfloat16 | |
| except Exception: | |
| pass | |
| a.model.to(a.device) | |
| print(f"ZeroGPU: model on {a.device}", flush=True) | |
| except Exception as e: # pragma: no cover | |
| print(f"ZeroGPU: move failed ({type(e).__name__}: {e}); staying on CPU", flush=True) | |
| return str(a.device) | |
| def warmup() -> dict: | |
| """One real call so weights are resident and kernels hot before the UI opens. | |
| Decorated, so it runs in the same ZeroGPU worker context as a real request. The agent | |
| is built on CPU at module scope and moved to the GPU inside this call, which is the only | |
| place a device exists. | |
| """ | |
| a = get_agent() | |
| dev = _move_to_gpu(a) | |
| t0 = time.perf_counter() | |
| out = a.predict( | |
| {"body": "I was charged twice for order A-104. Please refund the duplicate."}, | |
| {"department": {"type": "choice", | |
| "instructions": "Which team should handle this?", | |
| "criteria": {"billing": "payments and refunds", | |
| "technical": "bugs and outages", | |
| "account": "login and access"}}}) | |
| return {"ms": round((time.perf_counter() - t0) * 1000, 1), | |
| "device": dev, | |
| "warm_choice": out["answers"]["department"]["choice"]} | |
| # ------------------------------------------------------------------ decider backend | |
| # Mapika/decider-4b. The authors' own inference package is vendored verbatim in | |
| # `decider/` from the model repository (Apache 2.0), so the scoring path -- prompt layout, | |
| # per-question row isolation, isolated Score levels, restricting the answer logits to the | |
| # option-label tokens, the -inf masking of unused slots -- is theirs, not a reimplementation. | |
| # | |
| # Two deviations, both copied from the official decider demo Space: | |
| # * the checkpoint is built at module scope, on CPU, and moved onto the GPU from inside a | |
| # decorated call. Same reason as Laya: no CUDA device exists at import time, so building | |
| # on "cuda" at startup binds the model to CPU and then raises from torch._C._cuda_init() | |
| # inside the call. | |
| # * temperature is passed per answer type instead of being instance state, so concurrent | |
| # requests in one worker cannot race each other. | |
| DECIDER_CFG: dict[str, Any] = {} | |
| DECIDER_MODEL = None | |
| DECIDER_TOK = None | |
| DECIDER_T: dict[str, float] = {} | |
| def get_decider(): | |
| """Build the decider model once, at module scope, ON CPU.""" | |
| global DECIDER_MODEL, DECIDER_TOK, DECIDER_CFG, DECIDER_T | |
| if DECIDER_MODEL is not None: | |
| return DECIDER_MODEL | |
| import torch | |
| from huggingface_hub import hf_hub_download | |
| from decider.model import DecisionModel | |
| try: | |
| DECIDER_CFG = json.load(open(hf_hub_download(DECIDER_ID, "decider_config.json"))) | |
| except Exception as e: # pragma: no cover | |
| print(f"decider: no decider_config.json ({type(e).__name__}); defaults in force", | |
| flush=True) | |
| DECIDER_CFG = {} | |
| base = float(DECIDER_CFG.get("temperature", 1.0)) | |
| DECIDER_T = {k: float(v) for k, v in (DECIDER_CFG.get("temperature_by_type") or {}).items()} | |
| for k in ("choice", "noul", "score"): | |
| DECIDER_T.setdefault(k, base) | |
| DECIDER_MODEL = DecisionModel(DECIDER_ID, dtype=torch.bfloat16, grad_ckpt=False).cpu().eval() | |
| DECIDER_TOK = DECIDER_MODEL.tok | |
| return DECIDER_MODEL | |
| class _KeepOrder: | |
| """decider's rng stand-in: keep the caller's option order, never shuffle.""" | |
| def shuffle(self, x): | |
| pass | |
| def sample(self, xs, k): | |
| return xs[:k] | |
| def _decider_temps(t_choice=None, t_score=None, t_noul=None) -> dict: | |
| """Per-answer-type temperatures. 0 / None means 'keep the checkpoint's fitted value', | |
| the same sentinel Laya uses, so the two backends take the same three numbers. | |
| The per-type map matters: decider_config.json ships {choice: 1.11, noul: 1.56, | |
| score: 1.287}, and serving every answer at the single 1.099 flattens the yes/no and | |
| Score probabilities.""" | |
| t = dict(DECIDER_T) | |
| for key, v in (("choice", t_choice), ("score", t_score), ("noul", t_noul)): | |
| if v is None: | |
| continue | |
| v = float(v) | |
| if v > 0: | |
| t[key] = v | |
| return t | |
| def _decider_move_to_gpu() -> str: | |
| """Move the decider model onto CUDA from inside a decorated call. Same contract as the | |
| Laya path: degrade to CPU rather than raise, so an unusable GPU is slow, not fatal.""" | |
| global DECIDER_MODEL | |
| if DECIDER_MODEL is None: | |
| return "unloaded" | |
| import torch | |
| if not ZERO: | |
| return str(next(DECIDER_MODEL.parameters()).device) | |
| try: | |
| if not torch.cuda.is_available(): | |
| print("decider/ZeroGPU: no CUDA inside the call; staying on CPU", flush=True) | |
| return str(next(DECIDER_MODEL.parameters()).device) | |
| if next(DECIDER_MODEL.parameters()).device.type != "cuda": | |
| DECIDER_MODEL = DECIDER_MODEL.to("cuda") | |
| print("decider/ZeroGPU: model on cuda", flush=True) | |
| except Exception as e: # pragma: no cover | |
| print(f"decider/ZeroGPU: move failed ({type(e).__name__}: {e}); CPU", flush=True) | |
| return str(next(DECIDER_MODEL.parameters()).device) | |
| def _decider_systemone(state: Any, questions: dict, temps: dict) -> dict: | |
| """decider.infer.Decider.system_one's eager path, with an explicit per-type temperature. | |
| Every question gets its own row, so adding or removing a question cannot change another | |
| answer. The packed single-row path is not used because the temperature is applied per | |
| slot and the rows carry different answer types. | |
| """ | |
| import torch | |
| from decider.infer import Example, Q, neutralize_options | |
| from decider.model import collate | |
| from decider.prompt import MAX_OPTIONS, build | |
| from decider.systemone import (assemble, plan_rows, render_question, | |
| render_state, row_types) | |
| ctx = render_state(state) | |
| rqs = {k: render_question(v) for k, v in questions.items()} | |
| neutralize = bool(DECIDER_CFG.get("neutralize_none", True)) | |
| opts = ((lambda r: neutralize_options(r["options"])[0]) if neutralize | |
| else (lambda r: list(r["options"]))) | |
| isolated = bool(DECIDER_CFG.get("isolated_levels", False)) | |
| max_state = int(DECIDER_CFG.get("max_state_tokens", 32768)) | |
| flat, index = plan_rows(rqs, isolated) | |
| items = [ | |
| build(Example(ctx, [Q(r["question"], opts(r), 0)]), DECIDER_TOK, _KeepOrder(), | |
| max_options=MAX_OPTIONS, max_ctx_tokens=max_state, layout="state_first") | |
| for r in flat | |
| ] | |
| rtypes = row_types(rqs, index) | |
| base_t = float(DECIDER_CFG.get("temperature", 1.0)) | |
| # Take the device from the MODEL, not from torch.cuda.is_available(). The first version | |
| # of this sent tensors to "cuda" whenever a GPU was visible, while the module-scope | |
| # model was still on CPU -- which fails at startup with "Expected all tensors to be on | |
| # the same device, but found at least two devices, cuda:0 (ZeroGPU) and cpu!". | |
| dev = next(DECIDER_MODEL.parameters()).device | |
| probs = [] | |
| with torch.no_grad(): | |
| per = max(1, 65536 // max(len(it["ids"]) for it in items)) | |
| for i in range(0, len(items), per): | |
| chunk = items[i:i + per] | |
| bt = collate(chunk, DECIDER_TOK.pad_token_id) | |
| lg = DECIDER_MODEL.slot_logits( | |
| *[bt[k].to(dev) for k in ("input_ids", "attention_mask", "slot_idx", | |
| "slot_batch", "nopts")]) | |
| tvec = torch.tensor([temps.get(rtypes[i + j], base_t) | |
| for j in range(len(chunk))], | |
| device=lg.device, dtype=lg.dtype) | |
| pr = torch.softmax(lg / tvec.unsqueeze(-1), -1).float().cpu() | |
| c = 0 | |
| for it in chunk: | |
| probs.append(pr[c:c + len(it["slots"])]) | |
| c += len(it["slots"]) | |
| flatp = [p.tolist() for ps in probs for p in ps] | |
| answers = assemble(rqs, index, flatp) | |
| return {"model": f"decider-4b-{DECIDER_CFG.get('version', 'dev')}", | |
| "answers": answers, | |
| "usage": {"input_tokens": sum(len(it["ids"]) for it in items), | |
| "output_tokens": 0, "rows": len(items)}} | |
| def _normalise_questions(questions: dict) -> dict: | |
| """Accept Jev's shape directly; tolerate a couple of harmless variants. | |
| Jev sends `criteria` for choice (map of option -> description) and for score (ordered | |
| list). Both are what Laya expects, so the main job is not to mangle them. `noul` may | |
| carry optional `yes_text`/`no_text`; Laya takes those as the criterion text for the | |
| true/false markers, which changes what the two markers actually say -- worth passing | |
| through rather than dropping. | |
| """ | |
| out: dict[str, Any] = {} | |
| for qid, q in (questions or {}).items(): | |
| q = q or {} | |
| t = q.get("type") | |
| item: dict[str, Any] = {"type": t, | |
| "instructions": q.get("instructions", "")} | |
| if t in ("choice", "score"): | |
| item["criteria"] = q.get("criteria") or ({} if t == "choice" else []) | |
| elif t == "noul": | |
| yes, no = q.get("yes_text"), q.get("no_text") | |
| if yes or no: | |
| item["criteria"] = {k: v for k, v in | |
| (("true", yes), ("false", no)) if v} | |
| else: | |
| raise ValueError(f"unsupported question type: {t!r}") | |
| out[qid] = item | |
| return out | |
| def _apply_temperature(a, t_choice=None, t_score=None, t_noul=None) -> list: | |
| """Set per-type temperatures, but ONLY where the caller gave a real value. | |
| A value of 0 (or None) means "leave this one as the checkpoint configured it". That | |
| sentinel exists because of a bug this function caused: the Playground's number widgets | |
| defaulted to 1.0 and were applied on every call, which SILENTLY OVERWROTE the model's | |
| fitted temperatures and ran it in the uncalibrated regime the card explicitly warns | |
| about. The fitted values are the calibrated path; overriding them must be deliberate. | |
| """ | |
| t = _current_temperature(a) | |
| for i, v in enumerate((t_choice, t_score, t_noul)): | |
| if v is None: | |
| continue | |
| v = float(v) | |
| if v <= 0: # sentinel: keep the checkpoint's value | |
| continue | |
| t[i] = v | |
| a.temperature = t | |
| return t | |
| def answer_systemone(state: Any, questions: dict, temperature: list | None = None, | |
| backend: str = "laya") -> dict: | |
| """The single decorated entry point for every backend. | |
| Everything shares one GPU-decorated function: the start-up warm-up then covers all | |
| callers instead of each one paying its own first-call GPU attach. | |
| `temperature` is [choice, score, noul] with 0 meaning "keep the checkpoint's fitted | |
| value". Both backends take the same three numbers, but they are NOT the same | |
| temperatures: laya's fitted values live in rl_agent_config.json and decider's in | |
| decider_config.json, and each is left alone unless the caller overrides it. | |
| """ | |
| t0 = time.perf_counter() | |
| temps = list(temperature or (0, 0, 0)) | |
| while len(temps) < 3: | |
| temps.append(0) | |
| if backend == "decider": | |
| get_decider() | |
| device_in_use = _decider_move_to_gpu() | |
| dt = _decider_temps(temps[0], temps[1], temps[2]) | |
| res = _decider_systemone(state, questions, dt) | |
| res["_idu"] = { | |
| "space": f"dkappe/IDU {SPACE_REVISION}", | |
| "backend": DECIDER_ID, | |
| "device": device_in_use, | |
| "zero_gpu": ZERO, | |
| "latency_ms": round((time.perf_counter() - t0) * 1000, 1), | |
| "temperature": [dt["choice"], dt["score"], dt["noul"]], | |
| "note": ("decider-4b reads the option-label logits at one answer slot per " | |
| "question and never generates. Per-type temperatures are the " | |
| "checkpoint's fitted values unless overridden."), | |
| } | |
| return res | |
| a = get_agent() | |
| # Move onto the GPU HERE, inside the decorated call, where ZeroGPU has actually | |
| # attached one. Doing this at module scope cannot work: no device exists then. | |
| device_in_use = _move_to_gpu(a) | |
| if temperature: | |
| _apply_temperature(a, *temperature) | |
| qs = _normalise_questions(questions) | |
| res = a.predict(state, qs) | |
| res["_idu"] = { | |
| "space": f"dkappe/IDU {SPACE_REVISION}", | |
| "backend": MODEL_ID, | |
| "device": device_in_use, | |
| "zero_gpu": ZERO, | |
| "latency_ms": round((time.perf_counter() - t0) * 1000, 1), | |
| "temperature": _current_temperature(a), | |
| "note": ("Probabilities are the checkpoint's CALIBRATED path by default: its config " | |
| "ships fitted per-type and per-option-count temperatures. Override only " | |
| "deliberately."), | |
| } | |
| return res | |
| def _safe_answer(state, questions, temperature=None, backend="laya"): | |
| """Wrap the GPU call so failures show in the UI instead of killing the app.""" | |
| try: | |
| return answer_systemone(state, questions, temperature, backend) | |
| except Exception as e: # pragma: no cover | |
| traceback.print_exc() | |
| return {"error": f"{type(e).__name__}: {e}", | |
| "hint": "ZeroGPU attach can fail; the log line above has the cause."} | |
| # ------------------------------------------------------------------ presets / helpers | |
| DEPARTMENTS = {"billing": "invoices, payments, refunds", | |
| "technical": "bugs, outages, system errors", | |
| "account": "login, passwords, access", | |
| "other": "everything else"} | |
| TRIAGE = { | |
| "department": {"type": "choice", | |
| "instructions": "Which department should handle this request?", | |
| "criteria": DEPARTMENTS}, | |
| "frustration": {"type": "score", | |
| "instructions": "How frustrated is the writer?", | |
| "criteria": ["calm", "annoyed", "angry", "furious"]}, | |
| "refund_requested": {"type": "noul", | |
| "instructions": "Does the writer explicitly request a refund?"}, | |
| "churn_risk": {"type": "noul", | |
| "instructions": "Does the writer threaten to cancel or leave?"}, | |
| } | |
| PLAYGROUND_STATE = json.dumps({ | |
| "ticket": {"subject": "Duplicate charge", | |
| "messages": [{"from": "customer", | |
| "text": "I was charged twice for order A-104 and nobody " | |
| "has replied in three days. Refund the duplicate " | |
| "today or we are cancelling."}]}, | |
| "refund_policy": "Duplicate charges are eligible for a refund."}, indent=2) | |
| PLAYGROUND_Q = json.dumps({ | |
| "department": {"type": "choice", | |
| "instructions": "Which team should handle this?", | |
| "criteria": {"billing": "payments and refunds", | |
| "technical": "bugs and outages", | |
| "account": "login and access"}}, | |
| "frustration": {"type": "score", | |
| "instructions": "How frustrated is the customer?", | |
| "criteria": ["calm", "annoyed", "very angry"]}, | |
| "refund_requested": {"type": "noul", | |
| "instructions": "Does the ticket request a refund?"}, | |
| "policy_supports_refund": {"type": "noul", | |
| "instructions": "Does `refund_policy` allow the " | |
| "requested refund?"}}, indent=2) | |
| def as_rows(res: dict) -> list[list]: | |
| rows = [] | |
| for qid, ans in (res.get("answers") or {}).items(): | |
| if ans.get("type") == "choice": | |
| val = ans.get("choice") | |
| elif ans.get("type") == "score": | |
| val = ans.get("score") | |
| else: | |
| val = ans.get("noul") | |
| rows.append([qid, val, ans.get("confidence")]) | |
| return rows | |
| CARD = """ | |
| # IDU | |
| **JEV** shifted one letter back, the same shift that turns **IBM** into **HAL**. | |
| A typed-decision service: give it a `state` and typed `questions`, get typed answers with | |
| probabilities and confidence. No text generation, so there is nothing to parse and nothing | |
| to hallucinate. Three primitives, matching Jev's shape exactly: | |
| | type | question | answer | | |
| |---|---|---| | |
| | **choice** | which of these options? | the option, a probability per option, confidence | | |
| | **score** | where on this rubric? | a position along your levels, distribution, confidence | | |
| | **noul** | is this true? | one number: the probability that it is | | |
| **Backend:** [`convaiinnovations/laya`](https://huggingface.co/convaiinnovations/laya) -- | |
| 421M ModernBERT-large with a decision head, Apache 2.0. Every option is scored at its own | |
| `[MASK]` token and softmaxed over that question's options, and **all questions in a call are | |
| answered in one forward pass.** | |
| **Honest status.** The base checkpoints are near chance zero-shot on their own | |
| typed-decisions benchmark (0.362 vs a 0.461 majority-class baseline); the card's 0.766 is a | |
| checkpoint fine-tuned on that benchmark's training split. Laya is a fast base to specialise, | |
| not a finished decision engine. Its shipped probabilities are already temperature-fitted | |
| (`rl_agent_config.json` carries per-type and per-option-count temperatures), which is the | |
| 0.466 -> 0.081 ECE improvement the card describes — so the numbers here are the calibrated | |
| path by default, and the temperature controls are for experimenting rather than repair. | |
| """ | |
| def build_ui(): | |
| with gr.Blocks(title="IDU") as demo: | |
| gr.Markdown(CARD) | |
| with gr.Tab("Triage (one call, four questions)"): | |
| msg = gr.Textbox(lines=7, label="state", | |
| value="I was charged twice for invoice 4411 and nobody has " | |
| "answered for three days. Refund the duplicate today " | |
| "or we are cancelling our plan.") | |
| thr = gr.Slider(0.3, 0.95, 0.85, step=0.05, | |
| label="confidence needed to act without a human") | |
| go = gr.Button("Ask", variant="primary") | |
| with gr.Row(): | |
| t_rows = gr.Dataframe(headers=["question", "answer", "confidence"], | |
| col_count=(3, "fixed"), label="answers", wrap=True) | |
| with gr.Column(): | |
| t_act = gr.Markdown(label="action (plain code reading the numbers)") | |
| t_lat = gr.Markdown() | |
| def run_triage(text, threshold): | |
| r = _safe_answer(text, TRIAGE) | |
| if "error" in r: | |
| return [], f"**{r['error']}**", "" | |
| rows = as_rows(r) | |
| dept = (r["answers"]["department"] or {}) | |
| conf = dept.get("confidence") or 0.0 | |
| action = (f"route to **{dept.get('choice')}** automatically" | |
| if conf >= threshold else | |
| f"escalate to a human (confidence {conf:.2f} < {threshold:.2f})") | |
| return rows, f"{action} · {r['_idu']['latency_ms']} ms", json.dumps(r, indent=2) | |
| go.click(run_triage, [msg, thr], [t_rows, t_act, t_lat], api_name="triage") | |
| with gr.Tab("Playground"): | |
| gr.Markdown("Any state, any questions -- the same request shape as Jev's API.") | |
| p_state = gr.Code(label="state (JSON or plain text)", language="json", | |
| value=PLAYGROUND_STATE) | |
| p_q = gr.Code(label="questions", language="json", value=PLAYGROUND_Q) | |
| gr.Markdown("Temperature: **0 = keep the checkpoint's fitted value** (the default " | |
| "and the calibrated path). Set a positive number only to override " | |
| "deliberately — the shipped values are already fitted, and forcing " | |
| "1.0 runs the model in the uncalibrated regime.") | |
| with gr.Row(): | |
| t1 = gr.Number(value=0, label="temp: choice (0 = fitted)") | |
| t2 = gr.Number(value=0, label="temp: score (0 = fitted)") | |
| t3 = gr.Number(value=0, label="temp: noul (0 = fitted)") | |
| p_go = gr.Button("Ask", variant="primary") | |
| p_backend = gr.Radio(choices=["laya", "decider"], value="laya", | |
| label="backend", | |
| info="laya = 421M ModernBERT + decision head. " | |
| "decider = 4B Qwen3.5, 255 options / 32k state. " | |
| "Do not read latency as a comparison: both share " | |
| "one ZeroGPU pool.") | |
| p_rows = gr.Dataframe(headers=["question", "answer", "confidence"], | |
| col_count=(3, "fixed"), label="answers", wrap=True) | |
| p_raw = gr.Code(label="raw response", language="json") | |
| def run_playground(state_text, q_text, tc, ts, tn, backend): | |
| # Both boxes are gr.Code, but Gradio may hand us either a JSON string or an | |
| # already-parsed object depending on version and input source. Handle both; | |
| # the previous version only tried json.loads and blew up on a dict. | |
| def coerce(v): | |
| if isinstance(v, str): | |
| try: | |
| return json.loads(v), None | |
| except Exception: | |
| return v, None # plain text state is legitimate | |
| return v, None | |
| state, _ = coerce(state_text) | |
| if isinstance(q_text, str): | |
| try: | |
| qs = json.loads(q_text) | |
| except Exception as e: | |
| return [], f"{type(e).__name__}: questions is not valid JSON -- {e}" | |
| else: | |
| qs = q_text | |
| if not isinstance(qs, dict) or not qs: | |
| return [], "questions must be a non-empty JSON object" | |
| r = _safe_answer(state, qs, [tc, ts, tn], backend or "laya") | |
| return as_rows(r) if "answers" in r else [], json.dumps(r, indent=2) | |
| p_go.click(run_playground, [p_state, p_q, t1, t2, t3, p_backend], | |
| [p_rows, p_raw], api_name="systemone") | |
| with gr.Tab("Compare vs Needle 3"): | |
| gr.Markdown( | |
| "Same four questions, same three cases, through IDU and through the tuned " | |
| "Needle 3 adapter on `dkappe/needle3-gpu`. Both polarities of `noul` are " | |
| "exercised, so a 'yes to everything' bug cannot hide. The Needle side is " | |
| "called remotely and may be cold.") | |
| cs = gr.Button("run head-to-head", variant="primary") | |
| c_rows = gr.Dataframe(headers=["case", "expected", "IDU", "IDU ok", | |
| "Needle", "Needle ok"], | |
| col_count=(6, "fixed"), label="comparison", wrap=True) | |
| c_raw = gr.Code(label="detail", language="json") | |
| cs.click(compare_with_needle, None, [c_rows, c_raw], api_name="compare") | |
| with gr.Tab("Health"): | |
| h_btn = gr.Button("check") | |
| h_out = gr.JSON(label="health") | |
| h_btn.click(health, None, h_out, api_name="health") | |
| demo.load(lambda: None, None, None) # no auto-work on page load | |
| return demo | |
| # --------------------------------------------------------------------- comparison | |
| NEEDLE_SPACE = "dkappe/needle3-gpu" | |
| COMPARE_CASES = [ | |
| ("I was charged twice for order A-104. Please refund the duplicate.", | |
| {"department": "billing", "refund_requested": 1}), | |
| ("I cannot log in, my password is rejected.", | |
| {"department": "account", "refund_requested": 0}), | |
| ("The export button throws a 500 error.", | |
| {"department": "technical", "refund_requested": 0}), | |
| ] | |
| COMPARE_Q = { | |
| "department": {"type": "choice", "instructions": "Which team?", | |
| "criteria": {"billing": "Payment, charges, invoices, refunds.", | |
| "technical": "Bugs, errors, outages.", | |
| "account": "Login, password, access."}}, | |
| "refund_requested": {"type": "noul", | |
| "instructions": "Is the writer asking for a refund?"}, | |
| } | |
| def _needle_call(state: str, questions: dict) -> dict: | |
| """Call the tuned adapter on the Needle Space over the network.""" | |
| from gradio_client import Client | |
| tok = os.environ.get("HF_TOKEN") or None | |
| c = Client(NEEDLE_SPACE, token=tok, httpx_kwargs={"timeout": 900}) | |
| q = dict(questions) | |
| q["_request_text"] = state | |
| return c.predict(state, q, api_name="/tuned") | |
| def compare_with_needle(): | |
| rows, detail = [], {} | |
| for state, expect in COMPARE_CASES: | |
| # IDU first: it is local and cheap. | |
| r = _safe_answer(state, COMPARE_Q) | |
| idu_ch = idu_nl = None | |
| if "answers" in r: | |
| idu_ch = r["answers"]["department"]["choice"] | |
| idu_nl = int(round(r["answers"]["refund_requested"]["noul"])) | |
| idu_ok = (idu_ch == expect["department"]) and (idu_nl == expect["refund_requested"]) | |
| nd_ch = nd_nl = None | |
| nd_err = "" | |
| try: | |
| nr = _needle_call(state, COMPARE_Q) | |
| nd_ch = nr["answers"]["department"]["choice"] | |
| nd_nl = int(round(nr["answers"]["refund_requested"]["noul"])) | |
| except Exception as e: | |
| nd_err = f"{type(e).__name__}: {e}" | |
| nd_ok = (nd_ch == expect["department"]) and (nd_nl == expect["refund_requested"]) | |
| rows.append([state[:44], f"{expect['department']}/{expect['refund_requested']}", | |
| f"{idu_ch}/{idu_nl}", str(idu_ok), | |
| f"{nd_ch}/{nd_nl}" + (f" ({nd_err})" if nd_err else ""), str(nd_ok)]) | |
| detail[state[:40]] = {"expect": expect, "idu": idu_ch, "idu_noul": idu_nl, | |
| "needle": nd_ch, "needle_noul": nd_nl, | |
| "needle_error": nd_err} | |
| return rows, json.dumps(detail, indent=2) | |
| def health(): | |
| """Report the truth about where each model actually ran. | |
| DELIBERATELY DOES NOT ASSERT ok=... ON THE WEB-PROCESS CACHE. | |
| An earlier version returned ok=(_AGENT["a"] is not None). On ZeroGPU that is a lie: | |
| the model lives in the GPU worker, not in the web process, so the check reported | |
| ok=False while calls were succeeding on cuda. A health flag that says "broken" when | |
| the service is working is worse than no flag. | |
| So `ok` reflects the last OBSERVED WARM-UP result, which is evidence rather than | |
| bookkeeping, and `model_in_web_process` is reported separately for diagnosis. | |
| """ | |
| warm = _AGENT.get("warmup") or {} | |
| dwarm = _AGENT.get("warmup_decider") or {} | |
| a = _AGENT.get("a") | |
| import transformers as _tf | |
| return {"ok": bool(warm.get("warm_choice")), | |
| "space": f"dkappe/IDU {SPACE_REVISION}", | |
| "backends": {"laya": MODEL_ID, "decider": DECIDER_ID}, | |
| "last_warmup": warm, | |
| "warmup_device": warm.get("device"), | |
| "decider_warmup": dwarm, | |
| "transformers": _tf.__version__, | |
| "decider_config_version": DECIDER_CFG.get("version"), | |
| "zero_gpu": ZERO, | |
| "device_mode_env": DEVICE_MODE, | |
| # diagnostic only -- False on ZeroGPU is expected, not a fault | |
| "model_in_web_process": a is not None, | |
| "decider_in_web_process": DECIDER_MODEL is not None, | |
| "temperature_in_worker": _current_temperature(a) if a else None, | |
| "import_order_ok": "torch" in sys.modules, | |
| "note": ("ok reflects the last warm-up, because on ZeroGPU the model lives in " | |
| "the GPU worker and model_in_web_process will read False by design.")} | |
| def warmup_decider() -> dict: | |
| """Build decider-4b and run one real decision, so both backends are warm at startup. | |
| DECORATED WITH @GPU, which is not optional here. Without it this runs outside any GPU | |
| context: on ZeroGPU `torch.cuda.is_available()` can still report True while no device is | |
| actually attached, so moving the model fails inside the CUDA runtime with | |
| "no CUDA-capable device is detected". The decorated call is the only place a device | |
| exists. | |
| Kept separate from `warmup` so a decider failure cannot take Laya down with it -- the | |
| two are independent services sharing a UI. | |
| """ | |
| try: | |
| get_decider() | |
| dev = _decider_move_to_gpu() | |
| t0 = time.perf_counter() | |
| dt = _decider_temps() | |
| out = _decider_systemone( | |
| "I was charged twice for order A-104. Please refund the duplicate.", | |
| {"department": {"type": "choice", | |
| "instructions": "Which team should handle this?", | |
| "criteria": {"billing": "payments and refunds", | |
| "technical": "bugs and outages", | |
| "account": "login and access"}}}, | |
| dt) | |
| return {"ms": round((time.perf_counter() - t0) * 1000, 1), | |
| "device": dev, | |
| "warm_choice": out["answers"]["department"]["choice"], | |
| "temperature": [dt["choice"], dt["score"], dt["noul"]]} | |
| except Exception as e: # pragma: no cover | |
| traceback.print_exc() | |
| return {"error": f"{type(e).__name__}: {e}"} | |
| demo = build_ui() # safe here: every handler it references is defined above | |
| def _startup(): | |
| """Warm both backends BEFORE the UI is reachable. | |
| Gradio runs this file as __main__, so a warm-up placed in an `else:` branch would never | |
| execute. Laya's own demo does the same thing for the same reason. Each backend is | |
| warmed independently: a failure in one is reported and does not stop the other. | |
| """ | |
| try: | |
| _AGENT["warmup"] = warmup() | |
| print(f"[startup] warm laya: {_AGENT['warmup']}", flush=True) | |
| except Exception as _e: # pragma: no cover | |
| print(f"[startup] warmup failed: {type(_e).__name__}: {_e}", flush=True) | |
| try: | |
| _AGENT["warmup_decider"] = warmup_decider() | |
| print(f"[startup] warm decider: {_AGENT['warmup_decider']}", flush=True) | |
| except Exception as _e: # pragma: no cover | |
| print(f"[startup] warmup_decider failed: {type(_e).__name__}: {_e}", flush=True) | |
| if __name__ == "__main__": | |
| _startup() | |
| demo.queue(max_size=8).launch(server_name="0.0.0.0", server_port=7860) | |
| else: | |
| _startup() | |