Janus-35B-HERETIC / scripts /strip_mtp.py
FoolDev's picture
Claude Opus 5 (1M context)
release 0.9.9: second audit tier — scripts that lied, and card claims that were false
e023532
Raw History Blame Contribute Delete
9.14 kB
#!/usr/bin/env python3
"""
Strip the MTP / NextN layer (the extra last block) out of a qwen35moe GGUF so it
loads on stock llama.cpp / Ollama.
MTP-preserving GGUF conversions keep the MTP tensors as one extra NextN block,
so block_count = <transformer layers> + 1 MTP layer — here 40 + 1 = 41, with the
NextN block at index 40. Stock loaders treat that last block as a normal hybrid
layer and abort with `missing tensor 'blk.<last>.ssm_conv1d.weight'`. This drops
all `blk.<last>.*` tensors, sets block_count down by one (back to 40), and
removes nextn_predict_layers.
For the blob this repo bundles, that is a no-op passthrough (hardlink, no extra
disk), not a required build step. Janus-35B-A3B.Q4_K_M.gguf comes from
llmfan46/Qwen3.6-35B-A3B-uncensored-heretic-GGUF and is already MTP-clean:
block_count 40, blk.0…blk.39 only, no nextn_predict_layers. The head exists
upstream (the base's config.json declares mtp_num_hidden_layers 1) and llmfan46
publishes it as a separate *-Native-MTP-Preserved-GGUF whose Q4_K_M reports
block_count 41 — that variant is what this script is for, and this repo
deliberately does not ship it. So the strip is kept purely as a safety net, the
way the dense sibling FoolDev/Thanatos-27B-HERETIC needs it for its
MTP-preserved heretic-ara base: it guards against a future llmfan46 re-quant
shipping the MTP layer. Wanting the MTP head means running the upstream
safetensors under vLLM / SGLang, or taking the unstripped GGUF straight from
that variant repo.
Every KEPT tensor is copied byte-for-byte (raw quantized data) — no
re-quantization, so no quality change to the model itself.
Optionally also restamp the embedded chat template in the same pass:
--chat-template FILE replace tokenizer.chat_template with FILE's contents
That matters because llama.cpp, LM Studio, Jan and KoboldCpp all read the
template baked into the GGUF, so replacing it is the only way to change what
those loaders see. This repo no longer carries a chat_template.jinja — the Qwen
3.6 base's own embedded template governs the llama.cpp path — so nothing here
passes the flag today: build.sh calls this script without it, and the bundled
blob keeps the template it shipped with.
With --chat-template and no MTP block to strip, the file is still rewritten (the
usual hardlink passthrough would not carry the new template).
Usage:
python3 scripts/strip_mtp.py IN.gguf OUT.gguf
python3 scripts/strip_mtp.py IN.gguf OUT.gguf --chat-template chat_template.jinja
"""
from __future__ import annotations
import argparse
import sys
import numpy as np
from gguf import GGUFReader, GGUFWriter, GGUFValueType
def decode_field(field):
if not field or not field.types:
return None
t = field.types[0]
if t == GGUFValueType.ARRAY:
sub = field.types[-1]
if sub == GGUFValueType.STRING:
return [str(bytes(field.parts[i]), encoding="utf-8") for i in field.data]
return [pv for i in field.data for pv in field.parts[i].tolist()]
if t == GGUFValueType.STRING:
return str(bytes(field.parts[-1]), encoding="utf-8")
return field.parts[-1].tolist()[0]
def main() -> int:
ap = argparse.ArgumentParser(
description="Strip the MTP/NextN block from a GGUF; optionally restamp its chat template.")
ap.add_argument("src", metavar="IN.gguf")
ap.add_argument("dst", metavar="OUT.gguf")
ap.add_argument("--chat-template", metavar="FILE", default=None,
help="replace tokenizer.chat_template with the contents of FILE")
args = ap.parse_args()
src, dst = args.src, args.dst
new_template = None
if args.chat_template:
try:
with open(args.chat_template, encoding="utf-8") as fh:
new_template = fh.read()
except OSError as exc:
print(f"[!] cannot read {args.chat_template}: {exc}", file=sys.stderr)
return 1
if not new_template.strip():
print(f"[!] {args.chat_template} is empty; refusing to stamp a blank template",
file=sys.stderr)
return 1
reader = GGUFReader(src, "r")
arch = decode_field(reader.get_field("general.architecture"))
if arch is None:
print("[!] no general.architecture", file=sys.stderr)
return 1
bc_field = reader.get_field(f"{arch}.block_count")
block_count = decode_field(bc_field)
# Same guard the general.architecture lookup above has: decode_field returns
# None for a missing key, and `block_count - 1` on None raises a TypeError
# traceback from inside a build.sh run instead of this diagnostic.
if block_count is None:
print(f"[!] no {arch}.block_count", file=sys.stderr)
return 1
last = block_count - 1
prefix = f"blk.{last}."
# Only strip when the last block is actually an MTP/NextN layer. On an
# already-clean GGUF this is a no-op: hardlink (or copy) src -> dst so the
# caller always gets a valid output without touching real transformer layers.
has_nextn = decode_field(reader.get_field(f"{arch}.nextn_predict_layers")) or 0
last_is_mtp = any(t.name.startswith(prefix) and "nextn" in t.name.lower() for t in reader.tensors)
strip = bool(has_nextn or last_is_mtp)
# The hardlink passthrough cannot carry a new chat template, so it only
# applies when there is nothing at all to rewrite.
if not strip and new_template is None:
print(f"[=] no MTP/NextN layer (block_count={block_count}); copying through unchanged")
if src != dst:
import os
if os.path.exists(dst):
os.remove(dst)
try:
os.link(src, dst)
except OSError:
import shutil
shutil.copyfile(src, dst)
return 0
# We only ever drop the last block, so it MUST be the MTP/NextN block. If
# nextn_predict_layers is set but blk.<last>.* holds no NextN tensor, the MTP
# block isn't where we assume — refuse rather than drop a real transformer layer.
if strip and not last_is_mtp:
print(f"[!] {arch}.nextn_predict_layers is set but blk.{last}.* has no NextN tensor; "
"refusing to strip (would drop a real transformer layer).", file=sys.stderr)
return 1
# This script drops exactly ONE block and decrements block_count by one, so a
# model declaring more than one MTP head would be silently half-stripped and
# reported as a success. Refuse instead: there is no artifact here to test a
# multi-block path against, and the sibling refusal above already treats a
# broken assumption as fatal rather than guessing.
if strip and has_nextn and has_nextn != 1:
print(f"[!] {arch}.nextn_predict_layers={has_nextn}; this script only removes a single "
"MTP block. Refusing rather than half-stripping.", file=sys.stderr)
return 1
if strip:
print(f"[*] arch={arch} block_count={block_count} -> stripping MTP block {last}, new block_count={block_count-1}")
else:
print(f"[=] arch={arch} block_count={block_count}; no MTP block, rewriting to restamp the chat template")
# Writer manages general.architecture itself; block_count/nextn we override/drop.
skip_keys = {
"general.architecture",
f"{arch}.block_count",
f"{arch}.nextn_predict_layers",
}
if new_template is not None:
skip_keys.add("tokenizer.chat_template")
writer = GGUFWriter(dst, arch=arch)
writer.add_block_count(block_count - 1 if strip else block_count)
n_kv = 0
for name, field in reader.fields.items():
# GGUF.version / GGUF.tensor_count / GGUF.kv_count are header pseudo-fields
# the writer emits itself — re-adding them yields a duplicate key.
if name in skip_keys or name.startswith("GGUF."):
continue
vtype = field.types[0]
val = decode_field(field)
if vtype == GGUFValueType.ARRAY:
writer.add_array(name, val)
elif vtype == GGUFValueType.STRING:
writer.add_string(name, val)
else:
writer.add_key_value(name, val, vtype)
n_kv += 1
if new_template is not None:
writer.add_string("tokenizer.chat_template", new_template)
n_kv += 1
print(f"[*] chat template restamped from {args.chat_template} "
f"({len(new_template)} chars)")
kept = [t for t in reader.tensors if not (strip and t.name.startswith(prefix))]
dropped = [t.name for t in reader.tensors if strip and t.name.startswith(prefix)]
print(f"[*] copied {n_kv} KV keys; keeping {len(kept)} tensors, dropping {len(dropped)} (block {last})")
for t in kept:
writer.add_tensor_info(t.name, t.data.shape, t.data.dtype, t.data.nbytes, t.tensor_type)
writer.write_header_to_file()
writer.write_kv_data_to_file()
writer.write_ti_data_to_file()
for t in kept:
writer.write_tensor_data(np.ascontiguousarray(t.data))
writer.close()
print(f"[+] wrote {dst}")
return 0
if __name__ == "__main__":
sys.exit(main())