Taimwe commited on
Commit
9ad52d0
·
verified ·
1 Parent(s): a2ba1ae

Merge only (GGUF in separate job)

Browse files
Files changed (1) hide show
  1. merge_securecoder.py +105 -141
merge_securecoder.py CHANGED
@@ -1,141 +1,105 @@
1
- # /// script
2
- # requires-python = ">=3.10"
3
- # dependencies = [
4
- # "unsloth",
5
- # "transformers>=4.57",
6
- # "huggingface_hub",
7
- # ]
8
- # ///
9
- """Merge the SecureCoder LoRA into the base Qwen3-Coder-30B-A3B-Instruct
10
- checkpoint, upload a 16-bit safetensors repo, then quantise to Q4_K_M GGUF.
11
-
12
- Default base: unsloth/Qwen3-Coder-30B-A3B-Instruct
13
- Default adapter: Taimwe/securecoder-30b-pro
14
-
15
- Run on HF Jobs (a100-large has the headroom to load 30B in 16-bit):
16
- hf jobs run -d --flavor a100-large --timeout 90m --secrets HF_TOKEN \\
17
- ghcr.io/astral-sh/uv:python3.12-bookworm \\
18
- uv run --no-project https://huggingface.co/Taimwe/securecoder-scripts/resolve/main/merge_securecoder.py \\
19
- -- --adapter Taimwe/securecoder-30b-pro \\
20
- --output-repo Taimwe/securecoder-30b-pro-merged \\
21
- --gguf-repo Taimwe/securecoder-30b-pro-GGUF
22
- """
23
-
24
- from __future__ import annotations
25
-
26
- import argparse
27
- import logging
28
- import os
29
- import shutil
30
- import sys
31
- import time
32
-
33
- logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
34
- log = logging.getLogger("merge")
35
-
36
-
37
- def parse_args() -> argparse.Namespace:
38
- p = argparse.ArgumentParser(description="Merge + GGUF + push the SecureCoder LoRA")
39
- p.add_argument("--base", default="unsloth/Qwen3-Coder-30B-A3B-Instruct")
40
- p.add_argument("--adapter", default="Taimwe/securecoder-30b-pro")
41
- p.add_argument("--output-repo", default="Taimwe/securecoder-30b-pro-merged")
42
- p.add_argument("--gguf-repo", default=None)
43
- p.add_argument("--gguf-quant", default="Q4_K_M")
44
- p.add_argument("--private", action="store_true")
45
- p.add_argument("--work-dir", default="/data/securecoder-merge")
46
- p.add_argument("--max-shard-size", default="5GB")
47
- return p.parse_args()
48
-
49
- def main() -> int:
50
- args = parse_args()
51
- token = os.environ.get("HF_TOKEN")
52
- if not token:
53
- log.error("HF_TOKEN not set")
54
- return 1
55
-
56
- import torch
57
- from huggingface_hub import HfApi
58
- from unsloth import FastLanguageModel
59
-
60
- if not torch.cuda.is_available():
61
- log.error("no CUDA - merge needs a GPU")
62
- return 1
63
- log.info("GPU: %s", torch.cuda.get_device_name(0))
64
-
65
- work = args.work_dir
66
- if os.path.exists(work):
67
- shutil.rmtree(work)
68
- os.makedirs(work, exist_ok=True)
69
-
70
- log.info("loading base %s in 16-bit ...", args.base)
71
- started = time.time()
72
- model, tokenizer = FastLanguageModel.from_pretrained(
73
- model_name=args.base,
74
- max_seq_length=2048,
75
- dtype=torch.bfloat16,
76
- load_in_4bit=False,
77
- )
78
-
79
- log.info("loading adapter %s ...", args.adapter)
80
- from peft import PeftModel
81
- model = PeftModel.from_pretrained(model, args.adapter, token=token)
82
- log.info("merging ...")
83
- model = model.merge_and_unload()
84
- log.info("merge done in %.1f min", (time.time() - started) / 60)
85
-
86
- out_dir = os.path.join(work, "merged")
87
- model.save_pretrained(out_dir, safe_serialization=True, max_shard_size=args.max_shard_size)
88
- tokenizer.save_pretrained(out_dir)
89
- log.info("saved merged model to %s", out_dir)
90
-
91
- api = HfApi(token=token)
92
- api.create_repo(args.output_repo, repo_type="model", exist_ok=True, private=args.private)
93
- log.info("uploading to %s ...", args.output_repo)
94
- api.upload_folder(folder_path=out_dir, repo_id=args.output_repo, repo_type="model",
95
- commit_message="Merge SecureCoder LoRA into base (16-bit)")
96
- log.info("merged model live: https://huggingface.co/%s", args.output_repo)
97
-
98
- if args.gguf_repo:
99
- log.info("re-loading merged model for GGUF export ...")
100
- from unsloth import FastLanguageModel as FLM
101
- model, tokenizer = FLM.from_pretrained(
102
- model_name=out_dir,
103
- max_seq_length=2048,
104
- dtype=torch.bfloat16,
105
- load_in_4bit=False,
106
- )
107
- gguf_path = os.path.join(work, "gguf")
108
- os.makedirs(gguf_path, exist_ok=True)
109
- log.info("quantising to %s ...", args.gguf_quant)
110
- try:
111
- model.quantize_gguf_model(save_dir=gguf_path, quantization=args.gguf_quant)
112
- except Exception as exc: # noqa: BLE001
113
- log.warning("model.quantize_gguf_model failed (%s); falling back to llama-quantize CLI", exc)
114
- from huggingface_hub import hf_hub_download
115
- from pathlib import Path as _P
116
- qbin = hf_hub_download(repo_id="unsloth/llama.cpp", filename="llama-quantize",
117
- repo_type="model", token=token)
118
- import subprocess
119
- subprocess.run(["chmod", "+x", qbin], check=False)
120
- src = next(_P(out_dir).glob("*.gguf"), None)
121
- if src is None:
122
- log.error("no GGUF produced by Unsloth quantise pass")
123
- return 1
124
- subprocess.run([qbin, str(src), str(_P(gguf_path) / src.name), args.gguf_quant], check=True)
125
-
126
- api.create_repo(args.gguf_repo, repo_type="model", exist_ok=True, private=args.private)
127
- api.upload_folder(folder_path=gguf_path, repo_id=args.gguf_repo, repo_type="model",
128
- commit_message=f"Add {args.gguf_quant} GGUF export")
129
- log.info("GGUF live: https://huggingface.co/%s", args.gguf_repo)
130
-
131
- print("=" * 78)
132
- print("MERGE COMPLETE")
133
- print(f" merged: https://huggingface.co/{args.output_repo}")
134
- if args.gguf_repo:
135
- print(f" gguf : https://huggingface.co/{args.gguf_repo}")
136
- print("=" * 78)
137
- return 0
138
-
139
-
140
- if __name__ == "__main__":
141
- raise SystemExit(main())
 
1
+ # /// script
2
+ # requires-python = ">=3.10"
3
+ # dependencies = [
4
+ # "unsloth",
5
+ # "transformers>=4.57",
6
+ # "huggingface_hub",
7
+ # ]
8
+ # ///
9
+ """Merge the SecureCoder LoRA into the base Qwen3-Coder-30B-A3B-Instruct
10
+ checkpoint and upload a 16-bit safetensors repo. GGUF is done in a separate
11
+ job (the 30B-A3B MoE does not fit on one 80 GB card when merge and quantise
12
+ both run in the same process).
13
+
14
+ Default base: unsloth/Qwen3-Coder-30B-A3B-Instruct
15
+ Default adapter: Taimwe/securecoder-30b-pro
16
+
17
+ Run on HF Jobs:
18
+ hf jobs run -d --flavor a100-large --timeout 90m --secrets HF_TOKEN \\
19
+ ghcr.io/astral-sh/uv:python3.12-bookworm \\
20
+ uv run --no-project https://huggingface.co/Taimwe/securecoder-scripts/resolve/main/merge_securecoder.py \\
21
+ -- --adapter Taimwe/securecoder-30b-pro \\
22
+ --output-repo Taimwe/securecoder-30b-pro-merged
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import argparse
28
+ import logging
29
+ import os
30
+ import shutil
31
+ import sys
32
+ import time
33
+
34
+ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
35
+ log = logging.getLogger("merge")
36
+ def parse_args() -> argparse.Namespace:
37
+ p = argparse.ArgumentParser(description="Merge + push the SecureCoder LoRA")
38
+ p.add_argument("--base", default="unsloth/Qwen3-Coder-30B-A3B-Instruct")
39
+ p.add_argument("--adapter", default="Taimwe/securecoder-30b-pro")
40
+ p.add_argument("--output-repo", default="Taimwe/securecoder-30b-pro-merged")
41
+ p.add_argument("--private", action="store_true")
42
+ p.add_argument("--work-dir", default="/data/securecoder-merge")
43
+ p.add_argument("--max-shard-size", default="5GB")
44
+ return p.parse_args()
45
+
46
+
47
+ def main() -> int:
48
+ args = parse_args()
49
+ token = os.environ.get("HF_TOKEN")
50
+ if not token:
51
+ log.error("HF_TOKEN not set")
52
+ return 1
53
+
54
+ import torch
55
+ from huggingface_hub import HfApi
56
+ from unsloth import FastLanguageModel
57
+
58
+ if not torch.cuda.is_available():
59
+ log.error("no CUDA - merge needs a GPU")
60
+ return 1
61
+ log.info("GPU: %s", torch.cuda.get_device_name(0))
62
+
63
+ work = args.work_dir
64
+ if os.path.exists(work):
65
+ shutil.rmtree(work)
66
+ os.makedirs(work, exist_ok=True)
67
+
68
+ log.info("loading base %s in 16-bit ...", args.base)
69
+ started = time.time()
70
+ model, tokenizer = FastLanguageModel.from_pretrained(
71
+ model_name=args.base,
72
+ max_seq_length=2048,
73
+ dtype=torch.bfloat16,
74
+ load_in_4bit=False,
75
+ )
76
+
77
+ log.info("loading adapter %s ...", args.adapter)
78
+ from peft import PeftModel
79
+ model = PeftModel.from_pretrained(model, args.adapter, token=token)
80
+ log.info("merging ...")
81
+ model = model.merge_and_unload()
82
+ log.info("merge done in %.1f min", (time.time() - started) / 60)
83
+
84
+ out_dir = os.path.join(work, "merged")
85
+ model.save_pretrained(out_dir, safe_serialization=True, max_shard_size=args.max_shard_size)
86
+ tokenizer.save_pretrained(out_dir)
87
+ log.info("saved merged model to %s", out_dir)
88
+
89
+ api = HfApi(token=token)
90
+ api.create_repo(args.output_repo, repo_type="model", exist_ok=True, private=args.private)
91
+ log.info("uploading to %s ...", args.output_repo)
92
+ api.upload_folder(folder_path=out_dir, repo_id=args.output_repo, repo_type="model",
93
+ commit_message="Merge SecureCoder LoRA into base (16-bit)")
94
+ log.info("merged model live: https://huggingface.co/%s", args.output_repo)
95
+
96
+ print("=" * 78)
97
+ print("MERGE COMPLETE")
98
+ print(f" merged: https://huggingface.co/{args.output_repo}")
99
+ print(" next : run quantise_securecoder.py separately for GGUF")
100
+ print("=" * 78)
101
+ return 0
102
+
103
+
104
+ if __name__ == "__main__":
105
+ raise SystemExit(main())