File size: 4,003 Bytes
48879bc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
/**
 * Which `--reasoning-parser` a model needs, decided from its chat template.
 *
 * The JavaScript half of `tools/quantize/parser_advice.py`. Both are kept
 * honest by `parity.mjs`, which runs the same fixtures through each and
 * diffs — the rules took a false positive to get right, and a port that
 * drifts from them would reintroduce it silently.
 */

/**
 * Markers read out of vLLM's own reasoning parsers, not from documentation.
 *
 * Only markers that DISCRIMINATE a reasoning section are listed. The first
 * version of this table carried every literal each parser mentions, and
 * recommended gpt_oss for Phi-4-mini — not a reasoning model at all —
 * because Phi's template contains `<|end|>`, which is gpt_oss's turn
 * terminator. That recommendation would have cost the reader empty
 * answers, which is the exact failure this tool exists to prevent.
 */
export const PARSER_MARKERS = {
  deepseek_r1: ['<think>', '</think>'],
  hunyuan_v3: ['<think>', '</think>'],
  kimi_k2: ['<think>', '</think>'],
  minimax_m2: ['<think>', '</think>'],
  olmo3: ['<think>', '</think>'],
  step3: ['<think>', '</think>'],
  ernie45: ['<think>', '</think>'],
  minimax_m3: ['<mm:think>', '</mm:think>'],
  gemma4: ['<|channel>', '<channel|>'],
  // gptoss_reasoning_parser.py triggers on this exact phrase, not on
  // `<|message|>` or `<|end|>` alone.
  gpt_oss: ['<|channel|>analysis'],
  cohere_command: ['<|START_THINKING|>', '<|END_THINKING|>'],
};

/**
 * `qwen3` is registered in vLLM but holds its markers as token ids, so it
 * never appears in a scan for string literals. It closes on the same
 * `<think>` pair as the deepseek family — stated here rather than
 * inferred, because an unstated special case is how a checker starts
 * lying.
 */
export const TOKEN_ID_PARSERS = { qwen3: ['<think>', '</think>'] };

export const ALL_MARKERS = { ...PARSER_MARKERS, ...TOKEN_ID_PARSERS };

/** Every parser whose discriminating markers appear in this template. */
export function candidates(template) {
  const hits = {};
  for (const [parser, marks] of Object.entries(ALL_MARKERS)) {
    const found = marks.filter((m) => template.includes(m));
    if (found.length) hits[parser] = found;
  }
  return hits;
}

/**
 * The verdict for one template.
 *
 * `none` is a real answer, not an absence of one: a template with no
 * reasoning marker must be served WITHOUT a parser, because adding one
 * claims the whole answer and returns empty content.
 */
export function advise(template) {
  if (template === null || template === undefined) {
    return { verdict: 'unknown', candidates: {} };
  }
  const hits = candidates(template);
  return {
    verdict: Object.keys(hits).length ? 'candidates' : 'none',
    candidates: hits,
  };
}

/**
 * Pull a repo's chat template, and say which file it came from.
 *
 * Repos publish it in either place and sometimes both. Which one was read
 * matters: a `chat_template.jinja` added later can disagree with the copy
 * embedded in `tokenizer_config.json`, and someone debugging an empty
 * answer needs to know which one this advice is about.
 */
export async function fetchTemplate(repoId) {
  const base = `https://huggingface.co/${repoId}/raw/main`;

  const jinja = await fetch(`${base}/chat_template.jinja`);
  if (jinja.ok) return { template: await jinja.text(), source: 'chat_template.jinja' };

  const cfg = await fetch(`${base}/tokenizer_config.json`);
  if (cfg.ok) {
    let data;
    try {
      data = await cfg.json();
    } catch {
      return { template: null, source: 'tokenizer_config.json (unparseable)' };
    }
    let tpl = data.chat_template;
    if (Array.isArray(tpl)) tpl = tpl.map((t) => t.template || '').join('\n');
    if (tpl) return { template: tpl, source: 'tokenizer_config.json' };
  }

  if (jinja.status === 401 || jinja.status === 403 || cfg.status === 401 || cfg.status === 403) {
    return { template: null, source: 'gated or private' };
  }
  return { template: null, source: 'not published' };
}