File size: 1,713 Bytes
7bf8323 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 | {
"model_name": "Decision-2.0-Eos-0.8B",
"source": "vllm-sr/Decision-2.0-Eos-0.8B",
"source_revision": "3594047d69f476f1d01cf84c593e213fc3a4dfe0",
"backbone": "qwen3_5",
"package": "Decision2EosPacked.mlpackage",
"functions": {"S128_C256_N32": 34.3, "S256_C512_N64": 64.1, "S512_C768_N96": 108.6, "S512_C1024_N128": 131.0, "S1024_C1024_N128": 176.0},
"functions_note": "function name -> measured predict p50 ms on M5 Pro GPU, used to choose chunking. S = shared-prefix tokens, C = packed question tokens (multiple of 64), N = option slots",
"pad_token_id": 248044,
"rope": {"rotary_dim": 64, "rope_theta": 10000000},
"inputs": {
"input_ids": "[1, S+C] int32: shared prefix right-padded to S, then every question's suffix, then padding",
"cos, sin": "[S+C, 64] float16: RoPE tables (prefix positions 0.., each suffix continues from the prefix length)",
"valid": "[S] float16: 1 for real prefix tokens",
"tail_onehot": "[3, S] float16: selects the prefix's last 3 pre-convolution inputs (conv history)",
"segment": "[C, C] float16: 1 where j <= i within the same question (padding = own segment)",
"lag_keep, lag_tail": "[3, C], [3, C, 3] float16: a question's first 3 conv lags read the prefix tail",
"seg_chunks": "[C/64, 64, 64] float16: segment restricted to each 64-token chunk",
"cont": "[C] float16: 1 if the token's question started in an earlier chunk",
"last_seg": "[C/64, 64] float16: tokens in the question of each chunk's last token",
"cand_idx, query_idx": "[N] int32: packed-region index of each option's last token and of its question's final token"
},
"outputs": {"logits": "[N] float32: one raw logit per option (before softmax)"}
}
|