File size: 6,442 Bytes
d66ab11
 
 
 
 
 
 
a3fc135
 
 
d66ab11
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9fa3ddd
 
 
 
 
 
 
 
 
 
 
d66ab11
 
 
 
 
 
 
9fa3ddd
d66ab11
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
# /// script
# requires-python = ">=3.10"
# dependencies = [
#     "huggingface_hub",
#     "llama-cpp-python",
# ]
# ///



"""Download the merged SecureCoder safetensors repo, convert to GGUF via

llama.cpp's convert_hf_to_gguf.py, and quantise to Q4_K_M. Runs on

cpu-performance (no GPU needed for the conversion path).



Inputs:

    --merged-repo Taimwe/securecoder-30b-pro-merged   (default)

    --gguf-repo   Taimwe/securecoder-30b-pro-GGUF    (default)

    --quant       Q4_K_M                              (default)



This deliberately uses the standard llama.cpp converter (rather than Unsloth's

quantise_gguf_model) so the GGUF can be loaded by llama.cpp, Ollama, LM Studio,

Jan, and text-generation-webui without Unsloth's patched kernels.

"""

from __future__ import annotations

import argparse
import json
import logging
import os
import subprocess
import sys
import time
from pathlib import Path

logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
log = logging.getLogger("quantise")


def parse_args() -> argparse.Namespace:
    p = argparse.ArgumentParser(description="Convert merged safetensors -> GGUF")
    p.add_argument("--merged-repo", default="Taimwe/securecoder-30b-pro-merged")
    p.add_argument("--gguf-repo", default="Taimwe/securecoder-30b-pro-GGUF")
    p.add_argument("--quant", default="Q4_K_M")
    p.add_argument("--private", action="store_true")
    p.add_argument("--work-dir", default="/data/securecoder-gguf")
    return p.parse_args()


def main() -> int:
    args = parse_args()
    token = os.environ.get("HF_TOKEN")
    if not token:
        log.error("HF_TOKEN not set")
        return 1

    from huggingface_hub import HfApi, snapshot_download

    work = Path(args.work_dir)
    work.mkdir(parents=True, exist_ok=True)

    log.info("downloading %s ...", args.merged_repo)
    started = time.time()
    src = snapshot_download(
        args.merged_repo,
        local_dir=work / "merged",
        token=token,
        allow_patterns=["*.json", "*.txt", "*.safetensors", "*.tiktoken", "*.jinja"],
    )
    log.info("downloaded %s in %.1f min", src, (time.time() - started) / 60)

    src_path = Path(src)

    # Pull llama.cpp's converter + quantiser. We need cmake + make to build
    # llama-quantize; the container does not ship them, so apt-install them
    # up front.
    log.info("ensuring cmake / make are present ...")
    try:
        subprocess.run(["cmake", "--version"], check=True, stdout=subprocess.DEVNULL)
    except Exception:  # noqa: BLE001
        log.info("installing cmake via apt ...")
        subprocess.run(["apt-get", "update", "-qq"], check=True)
        subprocess.run(["apt-get", "install", "-y", "-qq", "cmake", "build-essential"], check=True)

    log.info("fetching llama.cpp ...")
    llama_dir = work / "llama.cpp"
    subprocess.run([
        "git", "clone", "--depth=1", "--branch", "master",
        "https://github.com/ggml-org/llama.cpp", str(llama_dir),
    ], check=True)
    subprocess.run(["pip", "install", "-q", "-r", str(llama_dir / "requirements" / "requirements-convert_hf_to_gguf.txt")],
                   check=True)

    f16_dir = work / "gguf-f16"
    f16_dir.mkdir(exist_ok=True)
    log.info("converting safetensors -> GGUF F16 ...")
    subprocess.run([
        sys.executable, str(llama_dir / "convert_hf_to_gguf.py"),
        str(src_path),
        "--outfile", str(f16_dir / "model.gguf"),
        "--outtype", "f16",
    ], check=True)

    log.info("quantising F16 -> %s ...", args.quant)
    qbin = llama_dir / "llama-quantize"  # build only if present
    if not qbin.exists() or not os.access(qbin, os.X_OK):
        log.info("building llama-quantize binary ...")
        subprocess.run([
            "cmake", "-S", str(llama_dir), "-B", str(llama_dir / "build"),
            "-DBUILD_SHARED_LIBS=OFF",
        ], check=True)
        subprocess.run([
            "cmake", "--build", str(llama_dir / "build"), "--target", "llama-quantize", "--config", "Release",
        ], check=True)
        # Find the produced binary
        candidates = [
            llama_dir / "build" / "bin" / "llama-quantize",
            llama_dir / "build" / "llama-quantize",
            llama_dir / "llama-quantize",
        ]
        qbin = next((p for p in candidates if p.exists()), None)
        if qbin is None:
            log.error("could not find llama-quantize after build")
            return 1

    out_dir = work / "gguf-out"
    out_dir.mkdir(exist_ok=True)
    subprocess.run([str(qbin), str(f16_dir / "model.gguf"),
                    str(out_dir / f"model-{args.quant}.gguf"), args.quant], check=True)

    # Write a README so the GGUF repo isn't an empty shell
    readme = f"""---

base_model: Taimwe/securecoder-30b-pro-merged

license: apache-2.0

pipeline_tag: text-generation

tags:

- gguf

- llama.cpp

- q4_k_m

- qwen3

- code

- tool-calling

- security

---



# SecureCoder (GGUF Q4_K_M)



GGUF export of [Taimwe/securecoder-30b-pro-merged](https://huggingface.co/Taimwe/securecoder-30b-pro-merged),

a LoRA fine-tune of `unsloth/Qwen3-Coder-30B-A3B-Instruct` for code + tool

calling + cybersecurity (offence and defence).



## Files



| File | Notes |

| --- | --- |

| `model-{args.quant}.gguf` | {args.quant} (~7 GB) |

| `README.md` | this card |



## Usage (llama.cpp / Ollama / LM Studio)



```bash

# llama.cpp server

llama-server -m model-{args.quant}.gguf --host 0.0.0.0 --port 8080 -ngl 99



# Ollama

ollama create securecoder -f Modelfile

ollama run securecoder "Write a binary search in Rust."

```



See the parent repo for full provenance, training data, and the limitations

section.

"""
    (out_dir / "README.md").write_text(readme, encoding="utf-8")

    api = HfApi(token=token)
    api.create_repo(args.gguf_repo, repo_type="model", exist_ok=True, private=args.private)
    api.upload_folder(folder_path=str(out_dir), repo_id=args.gguf_repo, repo_type="model",
                      commit_message=f"Add {args.quant} GGUF export")
    log.info("GGUF live: https://huggingface.co/%s", args.gguf_repo)

    print("=" * 78)
    print("QUANTISE COMPLETE")
    print(f"  gguf: https://huggingface.co/{args.gguf_repo}")
    print("=" * 78)
    return 0


if __name__ == "__main__":
    raise SystemExit(main())