# /// script # requires-python = ">=3.10" # dependencies = [ # "huggingface_hub", # "llama-cpp-python", # ] # /// """Download the merged SecureCoder safetensors repo, convert to GGUF via llama.cpp's convert_hf_to_gguf.py, and quantise to Q4_K_M. Runs on cpu-performance (no GPU needed for the conversion path). Inputs: --merged-repo Taimwe/securecoder-30b-pro-merged (default) --gguf-repo Taimwe/securecoder-30b-pro-GGUF (default) --quant Q4_K_M (default) This deliberately uses the standard llama.cpp converter (rather than Unsloth's quantise_gguf_model) so the GGUF can be loaded by llama.cpp, Ollama, LM Studio, Jan, and text-generation-webui without Unsloth's patched kernels. """ from __future__ import annotations import argparse import json import logging import os import subprocess import sys import time from pathlib import Path logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s") log = logging.getLogger("quantise") def parse_args() -> argparse.Namespace: p = argparse.ArgumentParser(description="Convert merged safetensors -> GGUF") p.add_argument("--merged-repo", default="Taimwe/securecoder-30b-pro-merged") p.add_argument("--gguf-repo", default="Taimwe/securecoder-30b-pro-GGUF") p.add_argument("--quant", default="Q4_K_M") p.add_argument("--private", action="store_true") p.add_argument("--work-dir", default="/data/securecoder-gguf") return p.parse_args() def main() -> int: args = parse_args() token = os.environ.get("HF_TOKEN") if not token: log.error("HF_TOKEN not set") return 1 from huggingface_hub import HfApi, snapshot_download work = Path(args.work_dir) work.mkdir(parents=True, exist_ok=True) log.info("downloading %s ...", args.merged_repo) started = time.time() src = snapshot_download( args.merged_repo, local_dir=work / "merged", token=token, allow_patterns=["*.json", "*.txt", "*.safetensors", "*.tiktoken", "*.jinja"], ) log.info("downloaded %s in %.1f min", src, (time.time() - started) / 60) src_path = Path(src) # Pull llama.cpp's converter + quantiser. We need cmake + make to build # llama-quantize; the container does not ship them, so apt-install them # up front. log.info("ensuring cmake / make are present ...") try: subprocess.run(["cmake", "--version"], check=True, stdout=subprocess.DEVNULL) except Exception: # noqa: BLE001 log.info("installing cmake via apt ...") subprocess.run(["apt-get", "update", "-qq"], check=True) subprocess.run(["apt-get", "install", "-y", "-qq", "cmake", "build-essential"], check=True) log.info("fetching llama.cpp ...") llama_dir = work / "llama.cpp" subprocess.run([ "git", "clone", "--depth=1", "--branch", "master", "https://github.com/ggml-org/llama.cpp", str(llama_dir), ], check=True) subprocess.run(["pip", "install", "-q", "-r", str(llama_dir / "requirements" / "requirements-convert_hf_to_gguf.txt")], check=True) f16_dir = work / "gguf-f16" f16_dir.mkdir(exist_ok=True) log.info("converting safetensors -> GGUF F16 ...") subprocess.run([ sys.executable, str(llama_dir / "convert_hf_to_gguf.py"), str(src_path), "--outfile", str(f16_dir / "model.gguf"), "--outtype", "f16", ], check=True) log.info("quantising F16 -> %s ...", args.quant) qbin = llama_dir / "llama-quantize" # build only if present if not qbin.exists() or not os.access(qbin, os.X_OK): log.info("building llama-quantize binary ...") subprocess.run([ "cmake", "-S", str(llama_dir), "-B", str(llama_dir / "build"), "-DBUILD_SHARED_LIBS=OFF", ], check=True) subprocess.run([ "cmake", "--build", str(llama_dir / "build"), "--target", "llama-quantize", "--config", "Release", ], check=True) # Find the produced binary candidates = [ llama_dir / "build" / "bin" / "llama-quantize", llama_dir / "build" / "llama-quantize", llama_dir / "llama-quantize", ] qbin = next((p for p in candidates if p.exists()), None) if qbin is None: log.error("could not find llama-quantize after build") return 1 out_dir = work / "gguf-out" out_dir.mkdir(exist_ok=True) subprocess.run([str(qbin), str(f16_dir / "model.gguf"), str(out_dir / f"model-{args.quant}.gguf"), args.quant], check=True) # Write a README so the GGUF repo isn't an empty shell readme = f"""--- base_model: Taimwe/securecoder-30b-pro-merged license: apache-2.0 pipeline_tag: text-generation tags: - gguf - llama.cpp - q4_k_m - qwen3 - code - tool-calling - security --- # SecureCoder (GGUF Q4_K_M) GGUF export of [Taimwe/securecoder-30b-pro-merged](https://huggingface.co/Taimwe/securecoder-30b-pro-merged), a LoRA fine-tune of `unsloth/Qwen3-Coder-30B-A3B-Instruct` for code + tool calling + cybersecurity (offence and defence). ## Files | File | Notes | | --- | --- | | `model-{args.quant}.gguf` | {args.quant} (~7 GB) | | `README.md` | this card | ## Usage (llama.cpp / Ollama / LM Studio) ```bash # llama.cpp server llama-server -m model-{args.quant}.gguf --host 0.0.0.0 --port 8080 -ngl 99 # Ollama ollama create securecoder -f Modelfile ollama run securecoder "Write a binary search in Rust." ``` See the parent repo for full provenance, training data, and the limitations section. """ (out_dir / "README.md").write_text(readme, encoding="utf-8") api = HfApi(token=token) api.create_repo(args.gguf_repo, repo_type="model", exist_ok=True, private=args.private) api.upload_folder(folder_path=str(out_dir), repo_id=args.gguf_repo, repo_type="model", commit_message=f"Add {args.quant} GGUF export") log.info("GGUF live: https://huggingface.co/%s", args.gguf_repo) print("=" * 78) print("QUANTISE COMPLETE") print(f" gguf: https://huggingface.co/{args.gguf_repo}") print("=" * 78) return 0 if __name__ == "__main__": raise SystemExit(main())