Download coreml_config.json from FluidInference/decision-2.0-kai-coreml: direct link, hf CLI and curl.
- Browser
- Download file 951 Bytes
-
https://huggingface.co/FluidInference/decision-2.0-kai-coreml/resolve/main/coreml_config.json
- Command line
-
hf download hf://FluidInference/decision-2.0-kai-coreml/coreml_config.json
-
curl -L -o coreml_config.json https://huggingface.co/FluidInference/decision-2.0-kai-coreml/resolve/main/coreml_config.json
951 Bytes
| { | |
| "model_name": "Decision-2.0-Kai-0.6B", | |
| "source": "vllm-sr/Decision-2.0-Kai-0.6B", | |
| "source_revision": "cd49ea3813fd8ba0928a9a23ef6c9a0f2f0cd764", | |
| "package": "Decision2KaiPacked.mlpackage", | |
| "functions": {"L256_N32": 14.3, "L512_N64": 27.8, "L1024_N128": 58.1, "L2048_N256": 142.3}, | |
| "functions_note": "function name -> measured predict p50 ms on M5 Pro GPU, used to choose chunking", | |
| "pad_token_id": 151643, | |
| "inputs": { | |
| "input_ids": "[1, L] int32: shared prefix, then each question's suffix, then padding", | |
| "position_ids": "[1, L] int32: RoPE positions; a suffix continues from the prefix length", | |
| "mask": "[1, 1, L, L] float16 additive: 0 where allowed, -1e4 elsewhere", | |
| "cand_idx": "[N] int32: packed index of each option's last token", | |
| "query_idx": "[N] int32: packed index of that question's final token" | |
| }, | |
| "outputs": {"logits": "[N] float32: one raw logit per option (before Score offsets and softmax)"} | |
| } | |