/** * Which `--reasoning-parser` a model needs, decided from its chat template. * * The JavaScript half of `tools/quantize/parser_advice.py`. Both are kept * honest by `parity.mjs`, which runs the same fixtures through each and * diffs — the rules took a false positive to get right, and a port that * drifts from them would reintroduce it silently. */ /** * Markers read out of vLLM's own reasoning parsers, not from documentation. * * Only markers that DISCRIMINATE a reasoning section are listed. The first * version of this table carried every literal each parser mentions, and * recommended gpt_oss for Phi-4-mini — not a reasoning model at all — * because Phi's template contains `<|end|>`, which is gpt_oss's turn * terminator. That recommendation would have cost the reader empty * answers, which is the exact failure this tool exists to prevent. */ export const PARSER_MARKERS = { deepseek_r1: ['', ''], hunyuan_v3: ['', ''], kimi_k2: ['', ''], minimax_m2: ['', ''], olmo3: ['', ''], step3: ['', ''], ernie45: ['', ''], minimax_m3: ['', ''], gemma4: ['<|channel>', ''], // gptoss_reasoning_parser.py triggers on this exact phrase, not on // `<|message|>` or `<|end|>` alone. gpt_oss: ['<|channel|>analysis'], cohere_command: ['<|START_THINKING|>', '<|END_THINKING|>'], }; /** * `qwen3` is registered in vLLM but holds its markers as token ids, so it * never appears in a scan for string literals. It closes on the same * `` pair as the deepseek family — stated here rather than * inferred, because an unstated special case is how a checker starts * lying. */ export const TOKEN_ID_PARSERS = { qwen3: ['', ''] }; export const ALL_MARKERS = { ...PARSER_MARKERS, ...TOKEN_ID_PARSERS }; /** Every parser whose discriminating markers appear in this template. */ export function candidates(template) { const hits = {}; for (const [parser, marks] of Object.entries(ALL_MARKERS)) { const found = marks.filter((m) => template.includes(m)); if (found.length) hits[parser] = found; } return hits; } /** * The verdict for one template. * * `none` is a real answer, not an absence of one: a template with no * reasoning marker must be served WITHOUT a parser, because adding one * claims the whole answer and returns empty content. */ export function advise(template) { if (template === null || template === undefined) { return { verdict: 'unknown', candidates: {} }; } const hits = candidates(template); return { verdict: Object.keys(hits).length ? 'candidates' : 'none', candidates: hits, }; } /** * Pull a repo's chat template, and say which file it came from. * * Repos publish it in either place and sometimes both. Which one was read * matters: a `chat_template.jinja` added later can disagree with the copy * embedded in `tokenizer_config.json`, and someone debugging an empty * answer needs to know which one this advice is about. */ export async function fetchTemplate(repoId) { const base = `https://huggingface.co/${repoId}/raw/main`; const jinja = await fetch(`${base}/chat_template.jinja`); if (jinja.ok) return { template: await jinja.text(), source: 'chat_template.jinja' }; const cfg = await fetch(`${base}/tokenizer_config.json`); if (cfg.ok) { let data; try { data = await cfg.json(); } catch { return { template: null, source: 'tokenizer_config.json (unparseable)' }; } let tpl = data.chat_template; if (Array.isArray(tpl)) tpl = tpl.map((t) => t.template || '').join('\n'); if (tpl) return { template: tpl, source: 'tokenizer_config.json' }; } if (jinja.status === 401 || jinja.status === 403 || cfg.status === 401 || cfg.status === 403) { return { template: null, source: 'gated or private' }; } return { template: null, source: 'not published' }; }