# wrapper: run llama.cpp converter on the pruned model, forcing the qwen35 pre-tokenizer (hash differs after vocab pruning) import os, sys, os, runpy L=os.environ.get("LLAMA_CPP","../llama.cpp") sys.path.insert(0, L); sys.path.insert(0, f"{L}/gguf-py") import conversion.base as base for c in vars(base).values(): if isinstance(c, type) and "get_vocab_base_pre" in vars(c): c.get_vocab_base_pre = lambda self, tokenizer: "qwen35" sys.argv=["convert_hf_to_gguf.py"]+sys.argv[1:] runpy.run_path(f"{L}/convert_hf_to_gguf.py", run_name="__main__")