Taimwe commited on
Commit
d66ab11
·
verified ·
1 Parent(s): 9ad52d0

Quantise safetensors -> GGUF (llama.cpp path)

Browse files
Files changed (1) hide show
  1. quantise_securecoder.py +176 -0
quantise_securecoder.py ADDED
@@ -0,0 +1,176 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # /// script
2
+ # requires-python = ">=3.10"
3
+ # dependencies = [
4
+ # "huggingface_hub",
5
+ # "llama-cpp-python",
6
+ # ]
7
+ # ///
8
+ """Download the merged SecureCoder safetensors repo, convert to GGUF via
9
+ llama.cpp's convert_hf_to_gguf.py, and quantise to Q4_K_M. Runs on
10
+ cpu-performance (no GPU needed for the conversion path).
11
+
12
+ Inputs:
13
+ --merged-repo Taimwe/securecoder-30b-pro-merged (default)
14
+ --gguf-repo Taimwe/securecoder-30b-pro-GGUF (default)
15
+ --quant Q4_K_M (default)
16
+
17
+ This deliberately uses the standard llama.cpp converter (rather than Unsloth's
18
+ quantise_gguf_model) so the GGUF can be loaded by llama.cpp, Ollama, LM Studio,
19
+ Jan, and text-generation-webui without Unsloth's patched kernels.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import argparse
25
+ import json
26
+ import logging
27
+ import os
28
+ import subprocess
29
+ import sys
30
+ import time
31
+ from pathlib import Path
32
+
33
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
34
+ log = logging.getLogger("quantise")
35
+
36
+
37
+ def parse_args() -> argparse.Namespace:
38
+ p = argparse.ArgumentParser(description="Convert merged safetensors -> GGUF")
39
+ p.add_argument("--merged-repo", default="Taimwe/securecoder-30b-pro-merged")
40
+ p.add_argument("--gguf-repo", default="Taimwe/securecoder-30b-pro-GGUF")
41
+ p.add_argument("--quant", default="Q4_K_M")
42
+ p.add_argument("--private", action="store_true")
43
+ p.add_argument("--work-dir", default="/data/securecoder-gguf")
44
+ return p.parse_args()
45
+
46
+
47
+ def main() -> int:
48
+ args = parse_args()
49
+ token = os.environ.get("HF_TOKEN")
50
+ if not token:
51
+ log.error("HF_TOKEN not set")
52
+ return 1
53
+
54
+ from huggingface_hub import HfApi, snapshot_download
55
+
56
+ work = Path(args.work_dir)
57
+ work.mkdir(parents=True, exist_ok=True)
58
+
59
+ log.info("downloading %s ...", args.merged_repo)
60
+ started = time.time()
61
+ src = snapshot_download(
62
+ args.merged_repo,
63
+ local_dir=work / "merged",
64
+ token=token,
65
+ allow_patterns=["*.json", "*.txt", "*.safetensors", "*.tiktoken", "*.jinja"],
66
+ )
67
+ log.info("downloaded %s in %.1f min", src, (time.time() - started) / 60)
68
+
69
+ src_path = Path(src)
70
+
71
+ # Pull llama.cpp's converter + quantiser
72
+ log.info("fetching llama.cpp ...")
73
+ llama_dir = work / "llama.cpp"
74
+ subprocess.run([
75
+ "git", "clone", "--depth=1", "--branch", "master",
76
+ "https://github.com/ggml-org/llama.cpp", str(llama_dir),
77
+ ], check=True)
78
+ subprocess.run(["pip", "install", "-q", "-r", str(llama_dir / "requirements" / "requirements-convert_hf_to_gguf.txt")],
79
+ check=False)
80
+
81
+ f16_dir = work / "gguf-f16"
82
+ f16_dir.mkdir(exist_ok=True)
83
+ log.info("converting safetensors -> GGUF F16 ...")
84
+ subprocess.run([
85
+ sys.executable, str(llama_dir / "convert_hf_to_gguf.py"),
86
+ str(src_path),
87
+ "--outfile", str(f16_dir / "model.gguf"),
88
+ "--outtype", "f16",
89
+ ], check=True)
90
+
91
+ log.info("quantising F16 -> %s ...", args.quant)
92
+ qbin = llama_dir / "llama-quantize" # build only if present
93
+ if not qbin.exists() or not os.access(qbin, os.X_OK):
94
+ log.info("building llama-quantize binary ...")
95
+ subprocess.run([
96
+ "cmake", "-S", str(llama_dir), "-B", str(llama_dir / "build"),
97
+ "-DBUILD_SHARED_LIBS=OFF",
98
+ ], check=True)
99
+ subprocess.run([
100
+ "cmake", "--build", str(llama_dir / "build"), "--target", "llama-quantize", "--config", "Release",
101
+ ], check=True)
102
+ # Find the produced binary
103
+ candidates = [
104
+ llama_dir / "build" / "bin" / "llama-quantize",
105
+ llama_dir / "build" / "llama-quantize",
106
+ llama_dir / "llama-quantize",
107
+ ]
108
+ qbin = next((p for p in candidates if p.exists()), None)
109
+ if qbin is None:
110
+ log.error("could not find llama-quantize after build")
111
+ return 1
112
+
113
+ out_dir = work / "gguf-out"
114
+ out_dir.mkdir(exist_ok=True)
115
+ subprocess.run([str(qbin), str(f16_dir / "model.gguf"),
116
+ str(out_dir / f"model-{args.quant}.gguf"), args.quant], check=True)
117
+
118
+ # Write a README so the GGUF repo isn't an empty shell
119
+ readme = f"""---
120
+ base_model: Taimwe/securecoder-30b-pro-merged
121
+ license: apache-2.0
122
+ pipeline_tag: text-generation
123
+ tags:
124
+ - gguf
125
+ - llama.cpp
126
+ - q4_k_m
127
+ - qwen3
128
+ - code
129
+ - tool-calling
130
+ - security
131
+ ---
132
+
133
+ # SecureCoder (GGUF Q4_K_M)
134
+
135
+ GGUF export of [Taimwe/securecoder-30b-pro-merged](https://huggingface.co/Taimwe/securecoder-30b-pro-merged),
136
+ a LoRA fine-tune of `unsloth/Qwen3-Coder-30B-A3B-Instruct` for code + tool
137
+ calling + cybersecurity (offence and defence).
138
+
139
+ ## Files
140
+
141
+ | File | Notes |
142
+ | --- | --- |
143
+ | `model-{args.quant}.gguf` | {args.quant} (~7 GB) |
144
+ | `README.md` | this card |
145
+
146
+ ## Usage (llama.cpp / Ollama / LM Studio)
147
+
148
+ ```bash
149
+ # llama.cpp server
150
+ llama-server -m model-{args.quant}.gguf --host 0.0.0.0 --port 8080 -ngl 99
151
+
152
+ # Ollama
153
+ ollama create securecoder -f Modelfile
154
+ ollama run securecoder "Write a binary search in Rust."
155
+ ```
156
+
157
+ See the parent repo for full provenance, training data, and the limitations
158
+ section.
159
+ """
160
+ (out_dir / "README.md").write_text(readme, encoding="utf-8")
161
+
162
+ api = HfApi(token=token)
163
+ api.create_repo(args.gguf_repo, repo_type="model", exist_ok=True, private=args.private)
164
+ api.upload_folder(folder_path=str(out_dir), repo_id=args.gguf_repo, repo_type="model",
165
+ commit_message=f"Add {args.quant} GGUF export")
166
+ log.info("GGUF live: https://huggingface.co/%s", args.gguf_repo)
167
+
168
+ print("=" * 78)
169
+ print("QUANTISE COMPLETE")
170
+ print(f" gguf: https://huggingface.co/{args.gguf_repo}")
171
+ print("=" * 78)
172
+ return 0
173
+
174
+
175
+ if __name__ == "__main__":
176
+ raise SystemExit(main())