securecoder-scripts / quantise_securecoder.py
Taimwe's picture
Quantise: apt-install cmake before building llama-quantize
9fa3ddd verified
Raw History Blame Contribute Delete
6.44 kB
# /// script
# requires-python = ">=3.10"
# dependencies = [
# "huggingface_hub",
# "llama-cpp-python",
# ]
# ///
"""Download the merged SecureCoder safetensors repo, convert to GGUF via
llama.cpp's convert_hf_to_gguf.py, and quantise to Q4_K_M. Runs on
cpu-performance (no GPU needed for the conversion path).
Inputs:
--merged-repo Taimwe/securecoder-30b-pro-merged (default)
--gguf-repo Taimwe/securecoder-30b-pro-GGUF (default)
--quant Q4_K_M (default)
This deliberately uses the standard llama.cpp converter (rather than Unsloth's
quantise_gguf_model) so the GGUF can be loaded by llama.cpp, Ollama, LM Studio,
Jan, and text-generation-webui without Unsloth's patched kernels.
"""
from __future__ import annotations
import argparse
import json
import logging
import os
import subprocess
import sys
import time
from pathlib import Path
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
log = logging.getLogger("quantise")
def parse_args() -> argparse.Namespace:
p = argparse.ArgumentParser(description="Convert merged safetensors -> GGUF")
p.add_argument("--merged-repo", default="Taimwe/securecoder-30b-pro-merged")
p.add_argument("--gguf-repo", default="Taimwe/securecoder-30b-pro-GGUF")
p.add_argument("--quant", default="Q4_K_M")
p.add_argument("--private", action="store_true")
p.add_argument("--work-dir", default="/data/securecoder-gguf")
return p.parse_args()
def main() -> int:
args = parse_args()
token = os.environ.get("HF_TOKEN")
if not token:
log.error("HF_TOKEN not set")
return 1
from huggingface_hub import HfApi, snapshot_download
work = Path(args.work_dir)
work.mkdir(parents=True, exist_ok=True)
log.info("downloading %s ...", args.merged_repo)
started = time.time()
src = snapshot_download(
args.merged_repo,
local_dir=work / "merged",
token=token,
allow_patterns=["*.json", "*.txt", "*.safetensors", "*.tiktoken", "*.jinja"],
)
log.info("downloaded %s in %.1f min", src, (time.time() - started) / 60)
src_path = Path(src)
# Pull llama.cpp's converter + quantiser. We need cmake + make to build
# llama-quantize; the container does not ship them, so apt-install them
# up front.
log.info("ensuring cmake / make are present ...")
try:
subprocess.run(["cmake", "--version"], check=True, stdout=subprocess.DEVNULL)
except Exception: # noqa: BLE001
log.info("installing cmake via apt ...")
subprocess.run(["apt-get", "update", "-qq"], check=True)
subprocess.run(["apt-get", "install", "-y", "-qq", "cmake", "build-essential"], check=True)
log.info("fetching llama.cpp ...")
llama_dir = work / "llama.cpp"
subprocess.run([
"git", "clone", "--depth=1", "--branch", "master",
"https://github.com/ggml-org/llama.cpp", str(llama_dir),
], check=True)
subprocess.run(["pip", "install", "-q", "-r", str(llama_dir / "requirements" / "requirements-convert_hf_to_gguf.txt")],
check=True)
f16_dir = work / "gguf-f16"
f16_dir.mkdir(exist_ok=True)
log.info("converting safetensors -> GGUF F16 ...")
subprocess.run([
sys.executable, str(llama_dir / "convert_hf_to_gguf.py"),
str(src_path),
"--outfile", str(f16_dir / "model.gguf"),
"--outtype", "f16",
], check=True)
log.info("quantising F16 -> %s ...", args.quant)
qbin = llama_dir / "llama-quantize" # build only if present
if not qbin.exists() or not os.access(qbin, os.X_OK):
log.info("building llama-quantize binary ...")
subprocess.run([
"cmake", "-S", str(llama_dir), "-B", str(llama_dir / "build"),
"-DBUILD_SHARED_LIBS=OFF",
], check=True)
subprocess.run([
"cmake", "--build", str(llama_dir / "build"), "--target", "llama-quantize", "--config", "Release",
], check=True)
# Find the produced binary
candidates = [
llama_dir / "build" / "bin" / "llama-quantize",
llama_dir / "build" / "llama-quantize",
llama_dir / "llama-quantize",
]
qbin = next((p for p in candidates if p.exists()), None)
if qbin is None:
log.error("could not find llama-quantize after build")
return 1
out_dir = work / "gguf-out"
out_dir.mkdir(exist_ok=True)
subprocess.run([str(qbin), str(f16_dir / "model.gguf"),
str(out_dir / f"model-{args.quant}.gguf"), args.quant], check=True)
# Write a README so the GGUF repo isn't an empty shell
readme = f"""---
base_model: Taimwe/securecoder-30b-pro-merged
license: apache-2.0
pipeline_tag: text-generation
tags:
- gguf
- llama.cpp
- q4_k_m
- qwen3
- code
- tool-calling
- security
---
# SecureCoder (GGUF Q4_K_M)
GGUF export of [Taimwe/securecoder-30b-pro-merged](https://huggingface.co/Taimwe/securecoder-30b-pro-merged),
a LoRA fine-tune of `unsloth/Qwen3-Coder-30B-A3B-Instruct` for code + tool
calling + cybersecurity (offence and defence).
## Files
| File | Notes |
| --- | --- |
| `model-{args.quant}.gguf` | {args.quant} (~7 GB) |
| `README.md` | this card |
## Usage (llama.cpp / Ollama / LM Studio)
```bash
# llama.cpp server
llama-server -m model-{args.quant}.gguf --host 0.0.0.0 --port 8080 -ngl 99
# Ollama
ollama create securecoder -f Modelfile
ollama run securecoder "Write a binary search in Rust."
```
See the parent repo for full provenance, training data, and the limitations
section.
"""
(out_dir / "README.md").write_text(readme, encoding="utf-8")
api = HfApi(token=token)
api.create_repo(args.gguf_repo, repo_type="model", exist_ok=True, private=args.private)
api.upload_folder(folder_path=str(out_dir), repo_id=args.gguf_repo, repo_type="model",
commit_message=f"Add {args.quant} GGUF export")
log.info("GGUF live: https://huggingface.co/%s", args.gguf_repo)
print("=" * 78)
print("QUANTISE COMPLETE")
print(f" gguf: https://huggingface.co/{args.gguf_repo}")
print("=" * 78)
return 0
if __name__ == "__main__":
raise SystemExit(main())