#!/usr/bin/env python3 """ Strip the MTP / NextN layer (the extra last block) out of a qwen35moe GGUF so it loads on stock llama.cpp / Ollama. MTP-preserving GGUF conversions keep the MTP tensors as one extra NextN block, so block_count = + 1 MTP layer — here 40 + 1 = 41, with the NextN block at index 40. Stock loaders treat that last block as a normal hybrid layer and abort with `missing tensor 'blk..ssm_conv1d.weight'`. This drops all `blk..*` tensors, sets block_count down by one (back to 40), and removes nextn_predict_layers. For the blob this repo bundles, that is a no-op passthrough (hardlink, no extra disk), not a required build step. Janus-35B-A3B.Q4_K_M.gguf comes from llmfan46/Qwen3.6-35B-A3B-uncensored-heretic-GGUF and is already MTP-clean: block_count 40, blk.0…blk.39 only, no nextn_predict_layers. The head exists upstream (the base's config.json declares mtp_num_hidden_layers 1) and llmfan46 publishes it as a separate *-Native-MTP-Preserved-GGUF whose Q4_K_M reports block_count 41 — that variant is what this script is for, and this repo deliberately does not ship it. So the strip is kept purely as a safety net, the way the dense sibling FoolDev/Thanatos-27B-HERETIC needs it for its MTP-preserved heretic-ara base: it guards against a future llmfan46 re-quant shipping the MTP layer. Wanting the MTP head means running the upstream safetensors under vLLM / SGLang, or taking the unstripped GGUF straight from that variant repo. Every KEPT tensor is copied byte-for-byte (raw quantized data) — no re-quantization, so no quality change to the model itself. Optionally also restamp the embedded chat template in the same pass: --chat-template FILE replace tokenizer.chat_template with FILE's contents That matters because llama.cpp, LM Studio, Jan and KoboldCpp all read the template baked into the GGUF, so replacing it is the only way to change what those loaders see. This repo no longer carries a chat_template.jinja — the Qwen 3.6 base's own embedded template governs the llama.cpp path — so nothing here passes the flag today: build.sh calls this script without it, and the bundled blob keeps the template it shipped with. With --chat-template and no MTP block to strip, the file is still rewritten (the usual hardlink passthrough would not carry the new template). Usage: python3 scripts/strip_mtp.py IN.gguf OUT.gguf python3 scripts/strip_mtp.py IN.gguf OUT.gguf --chat-template chat_template.jinja """ from __future__ import annotations import argparse import sys import numpy as np from gguf import GGUFReader, GGUFWriter, GGUFValueType def decode_field(field): if not field or not field.types: return None t = field.types[0] if t == GGUFValueType.ARRAY: sub = field.types[-1] if sub == GGUFValueType.STRING: return [str(bytes(field.parts[i]), encoding="utf-8") for i in field.data] return [pv for i in field.data for pv in field.parts[i].tolist()] if t == GGUFValueType.STRING: return str(bytes(field.parts[-1]), encoding="utf-8") return field.parts[-1].tolist()[0] def main() -> int: ap = argparse.ArgumentParser( description="Strip the MTP/NextN block from a GGUF; optionally restamp its chat template.") ap.add_argument("src", metavar="IN.gguf") ap.add_argument("dst", metavar="OUT.gguf") ap.add_argument("--chat-template", metavar="FILE", default=None, help="replace tokenizer.chat_template with the contents of FILE") args = ap.parse_args() src, dst = args.src, args.dst new_template = None if args.chat_template: try: with open(args.chat_template, encoding="utf-8") as fh: new_template = fh.read() except OSError as exc: print(f"[!] cannot read {args.chat_template}: {exc}", file=sys.stderr) return 1 if not new_template.strip(): print(f"[!] {args.chat_template} is empty; refusing to stamp a blank template", file=sys.stderr) return 1 reader = GGUFReader(src, "r") arch = decode_field(reader.get_field("general.architecture")) if arch is None: print("[!] no general.architecture", file=sys.stderr) return 1 bc_field = reader.get_field(f"{arch}.block_count") block_count = decode_field(bc_field) # Same guard the general.architecture lookup above has: decode_field returns # None for a missing key, and `block_count - 1` on None raises a TypeError # traceback from inside a build.sh run instead of this diagnostic. if block_count is None: print(f"[!] no {arch}.block_count", file=sys.stderr) return 1 last = block_count - 1 prefix = f"blk.{last}." # Only strip when the last block is actually an MTP/NextN layer. On an # already-clean GGUF this is a no-op: hardlink (or copy) src -> dst so the # caller always gets a valid output without touching real transformer layers. has_nextn = decode_field(reader.get_field(f"{arch}.nextn_predict_layers")) or 0 last_is_mtp = any(t.name.startswith(prefix) and "nextn" in t.name.lower() for t in reader.tensors) strip = bool(has_nextn or last_is_mtp) # The hardlink passthrough cannot carry a new chat template, so it only # applies when there is nothing at all to rewrite. if not strip and new_template is None: print(f"[=] no MTP/NextN layer (block_count={block_count}); copying through unchanged") if src != dst: import os if os.path.exists(dst): os.remove(dst) try: os.link(src, dst) except OSError: import shutil shutil.copyfile(src, dst) return 0 # We only ever drop the last block, so it MUST be the MTP/NextN block. If # nextn_predict_layers is set but blk..* holds no NextN tensor, the MTP # block isn't where we assume — refuse rather than drop a real transformer layer. if strip and not last_is_mtp: print(f"[!] {arch}.nextn_predict_layers is set but blk.{last}.* has no NextN tensor; " "refusing to strip (would drop a real transformer layer).", file=sys.stderr) return 1 # This script drops exactly ONE block and decrements block_count by one, so a # model declaring more than one MTP head would be silently half-stripped and # reported as a success. Refuse instead: there is no artifact here to test a # multi-block path against, and the sibling refusal above already treats a # broken assumption as fatal rather than guessing. if strip and has_nextn and has_nextn != 1: print(f"[!] {arch}.nextn_predict_layers={has_nextn}; this script only removes a single " "MTP block. Refusing rather than half-stripping.", file=sys.stderr) return 1 if strip: print(f"[*] arch={arch} block_count={block_count} -> stripping MTP block {last}, new block_count={block_count-1}") else: print(f"[=] arch={arch} block_count={block_count}; no MTP block, rewriting to restamp the chat template") # Writer manages general.architecture itself; block_count/nextn we override/drop. skip_keys = { "general.architecture", f"{arch}.block_count", f"{arch}.nextn_predict_layers", } if new_template is not None: skip_keys.add("tokenizer.chat_template") writer = GGUFWriter(dst, arch=arch) writer.add_block_count(block_count - 1 if strip else block_count) n_kv = 0 for name, field in reader.fields.items(): # GGUF.version / GGUF.tensor_count / GGUF.kv_count are header pseudo-fields # the writer emits itself — re-adding them yields a duplicate key. if name in skip_keys or name.startswith("GGUF."): continue vtype = field.types[0] val = decode_field(field) if vtype == GGUFValueType.ARRAY: writer.add_array(name, val) elif vtype == GGUFValueType.STRING: writer.add_string(name, val) else: writer.add_key_value(name, val, vtype) n_kv += 1 if new_template is not None: writer.add_string("tokenizer.chat_template", new_template) n_kv += 1 print(f"[*] chat template restamped from {args.chat_template} " f"({len(new_template)} chars)") kept = [t for t in reader.tensors if not (strip and t.name.startswith(prefix))] dropped = [t.name for t in reader.tensors if strip and t.name.startswith(prefix)] print(f"[*] copied {n_kv} KV keys; keeping {len(kept)} tensors, dropping {len(dropped)} (block {last})") for t in kept: writer.add_tensor_info(t.name, t.data.shape, t.data.dtype, t.data.nbytes, t.tensor_type) writer.write_header_to_file() writer.write_kv_data_to_file() writer.write_ti_data_to_file() for t in kept: writer.write_tensor_data(np.ascontiguousarray(t.data)) writer.close() print(f"[+] wrote {dst}") return 0 if __name__ == "__main__": sys.exit(main())