reasoning-parser-advisor / parser_advice.js
aleada's picture
reasoning-parser advisor
48879bc verified
Raw History Blame Contribute Delete
4 kB
/**
* Which `--reasoning-parser` a model needs, decided from its chat template.
*
* The JavaScript half of `tools/quantize/parser_advice.py`. Both are kept
* honest by `parity.mjs`, which runs the same fixtures through each and
* diffs — the rules took a false positive to get right, and a port that
* drifts from them would reintroduce it silently.
*/
/**
* Markers read out of vLLM's own reasoning parsers, not from documentation.
*
* Only markers that DISCRIMINATE a reasoning section are listed. The first
* version of this table carried every literal each parser mentions, and
* recommended gpt_oss for Phi-4-mini — not a reasoning model at all —
* because Phi's template contains `<|end|>`, which is gpt_oss's turn
* terminator. That recommendation would have cost the reader empty
* answers, which is the exact failure this tool exists to prevent.
*/
export const PARSER_MARKERS = {
deepseek_r1: ['<think>', '</think>'],
hunyuan_v3: ['<think>', '</think>'],
kimi_k2: ['<think>', '</think>'],
minimax_m2: ['<think>', '</think>'],
olmo3: ['<think>', '</think>'],
step3: ['<think>', '</think>'],
ernie45: ['<think>', '</think>'],
minimax_m3: ['<mm:think>', '</mm:think>'],
gemma4: ['<|channel>', '<channel|>'],
// gptoss_reasoning_parser.py triggers on this exact phrase, not on
// `<|message|>` or `<|end|>` alone.
gpt_oss: ['<|channel|>analysis'],
cohere_command: ['<|START_THINKING|>', '<|END_THINKING|>'],
};
/**
* `qwen3` is registered in vLLM but holds its markers as token ids, so it
* never appears in a scan for string literals. It closes on the same
* `<think>` pair as the deepseek family — stated here rather than
* inferred, because an unstated special case is how a checker starts
* lying.
*/
export const TOKEN_ID_PARSERS = { qwen3: ['<think>', '</think>'] };
export const ALL_MARKERS = { ...PARSER_MARKERS, ...TOKEN_ID_PARSERS };
/** Every parser whose discriminating markers appear in this template. */
export function candidates(template) {
const hits = {};
for (const [parser, marks] of Object.entries(ALL_MARKERS)) {
const found = marks.filter((m) => template.includes(m));
if (found.length) hits[parser] = found;
}
return hits;
}
/**
* The verdict for one template.
*
* `none` is a real answer, not an absence of one: a template with no
* reasoning marker must be served WITHOUT a parser, because adding one
* claims the whole answer and returns empty content.
*/
export function advise(template) {
if (template === null || template === undefined) {
return { verdict: 'unknown', candidates: {} };
}
const hits = candidates(template);
return {
verdict: Object.keys(hits).length ? 'candidates' : 'none',
candidates: hits,
};
}
/**
* Pull a repo's chat template, and say which file it came from.
*
* Repos publish it in either place and sometimes both. Which one was read
* matters: a `chat_template.jinja` added later can disagree with the copy
* embedded in `tokenizer_config.json`, and someone debugging an empty
* answer needs to know which one this advice is about.
*/
export async function fetchTemplate(repoId) {
const base = `https://huggingface.co/${repoId}/raw/main`;
const jinja = await fetch(`${base}/chat_template.jinja`);
if (jinja.ok) return { template: await jinja.text(), source: 'chat_template.jinja' };
const cfg = await fetch(`${base}/tokenizer_config.json`);
if (cfg.ok) {
let data;
try {
data = await cfg.json();
} catch {
return { template: null, source: 'tokenizer_config.json (unparseable)' };
}
let tpl = data.chat_template;
if (Array.isArray(tpl)) tpl = tpl.map((t) => t.template || '').join('\n');
if (tpl) return { template: tpl, source: 'tokenizer_config.json' };
}
if (jinja.status === 401 || jinja.status === 403 || cfg.status === 401 || cfg.status === 403) {
return { template: null, source: 'gated or private' };
}
return { template: null, source: 'not published' };
}