{ "model_name": "Decision-2.0-Kai-0.6B", "source": "vllm-sr/Decision-2.0-Kai-0.6B", "source_revision": "cd49ea3813fd8ba0928a9a23ef6c9a0f2f0cd764", "package": "Decision2KaiPacked.mlpackage", "functions": {"L256_N32": 14.3, "L512_N64": 27.8, "L1024_N128": 58.1, "L2048_N256": 142.3}, "functions_note": "function name -> measured predict p50 ms on M5 Pro GPU, used to choose chunking", "pad_token_id": 151643, "inputs": { "input_ids": "[1, L] int32: shared prefix, then each question's suffix, then padding", "position_ids": "[1, L] int32: RoPE positions; a suffix continues from the prefix length", "mask": "[1, 1, L, L] float16 additive: 0 where allowed, -1e4 elsewhere", "cand_idx": "[N] int32: packed index of each option's last token", "query_idx": "[N] int32: packed index of that question's final token" }, "outputs": {"logits": "[N] float32: one raw logit per option (before Score offsets and softmax)"} }