File size: 8,693 Bytes
0771951
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
50adde2
0771951
3bf965a
 
 
 
 
 
 
 
 
 
0771951
 
3bf965a
0771951
 
3bf965a
0771951
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""colab_deploy.py — Colab T4 单卡 m13_r10 部署链 (merge→convert→surgery→双量化→推仓→serving)
血统: 底模 Qwen/Qwen3.5-9B ⊕ m13_r10 adapter (r10 权重已含 r6..r10, PeftModel 续训链)
RAM 12G 策略: merge device_map=auto (GPU 13G + CPU 溢出); 磁盘时序: convert 后删 merged
幂等旗标: /content/k8b/DEPLOY_* ; 日志: /content/k8b/colab_deploy.log"""
import json, os, subprocess, sys, time, traceback

K8B = "/content/k8b"
K27 = "/content/k27"
BASE = K27 + "/Qwen3.5-9B"
MERGED = K8B + "/merged_r10_hf"
ADP = K8B + "/m12_adapters/m13_r10"
LC = "/content/lcbuild/llama.cpp"
BIN = "/content/lcbin"
F16 = K8B + "/m13_r10_f16.gguf"
Q4 = K8B + "/m13_r10_q4_k_m.gguf"
Q48 = K8B + "/m13_r10_mixbit_4x8.gguf"
MMPROJ = K8B + "/mmproj_m11.gguf"
LOG = K8B + "/colab_deploy.log"
os.makedirs(K8B, exist_ok=True)
log = open(LOG, "a", buffering=1)

def P(m):
    log.write("[%s] %s\n" % (time.strftime("%m-%d %H:%M:%S"), m))

def sh(c, t=7200):
    p = subprocess.run(c, shell=True, capture_output=True, text=True, timeout=t, errors="replace")
    return ((p.stdout or "") + (p.stderr or ""))[-1500:]

def flag(n, c="1"):
    open(K8B + "/" + n, "w").write(str(c)[:400])

def have(n):
    return os.path.exists(K8B + "/" + n)

def hf_token():
    return open("/root/.cache/huggingface/token").read().strip()

LD = "LD_LIBRARY_PATH=/content/lcbin:/usr/local/nvidia/lib64:/usr/lib/x86_64-linux-gnu"

def main():
    t0 = time.time()
    P("==== colab deploy start ====")
    try:
        # D1: merge (device_map auto: T4 16G 吃大头, RAM 溢出)
        if not have("DEPLOY_MERGE"):
            if not os.path.exists(BASE + "/config.json"):
                raise RuntimeError("base missing (download first)")
            import torch
            from transformers import AutoProcessor, AutoModelForImageTextToText, AutoConfig
            from peft import PeftModel
            from accelerate import infer_auto_device_map
            with torch.device("meta"):
                skel = AutoModelForImageTextToText.from_config(
                    AutoConfig.from_pretrained(BASE))
            dm = infer_auto_device_map(skel, max_memory={0: "13500MiB", "cpu": "48000MiB"})
            del skel
            n_disk = sum(1 for v in dm.values() if v == "disk")
            P("device_map: gpu+cpu, disk=%d" % n_disk)
            assert n_disk == 0, "device_map still has disk offload"
            kw = dict(low_cpu_mem_usage=True)
            try:
                m = AutoModelForImageTextToText.from_pretrained(
                    BASE, dtype=torch.float16, device_map=dm, **kw)
            except TypeError:
                m = AutoModelForImageTextToText.from_pretrained(
                    BASE, torch_dtype=torch.float16, device_map=dm, **kw)
            P("loaded; GPU %.1fG" % (torch.cuda.memory_allocated() / 1e9))
            m = PeftModel.from_pretrained(m, ADP)
            m = m.merge_and_unload()
            m.save_pretrained(MERGED, safe_serialization=True)
            try:
                AutoProcessor.from_pretrained(BASE).save_pretrained(MERGED)
            except Exception as e:
                P("proc save skip %s" % repr(e)[:80])
            del m
            import gc, torch as _t
            gc.collect(); _t.cuda.empty_cache()
            flag("DEPLOY_MERGE")
            P("merge ok")
        # D2: convert f16 (mmap 友好), 完成后删 merged 省盘
        if not have("DEPLOY_CONVERT"):
            if not os.path.exists(LC + "/convert_hf_to_gguf.py"):
                raise RuntimeError("llama.cpp src missing (compile chain clones it)")
            r = sh("cd %s && NO_LOCAL_GGUF=1 PYTHONPATH=%s/gguf-py python3 convert_hf_to_gguf.py "
                   "%s --outfile %s --outtype f16 > %s/convert.log 2>&1; echo RC=$?; tail -2 %s/convert.log"
                   % (LC, LC, MERGED, F16, K8B, K8B), 14400)
            P("convert: " + r[-250:])
            if "RC=0" not in r or not os.path.exists(F16) or os.path.getsize(F16) < 15e9:
                raise RuntimeError("convert failed")
            flag("DEPLOY_CONVERT")
            P("convert ok (%.1fG)" % (os.path.getsize(F16) / 1e9))
            sh("rm -rf %s" % MERGED, 900)
            P("merged_r10_hf removed (disk)")
        # D3: surgery v2 (bc->32, nx->0)
        if not have("DEPLOY_SURGERY"):
            r = subprocess.run([sys.executable, K8B + "/r12_surgery.py", F16],
                               capture_output=True, text=True, timeout=2400,
                               env=dict(os.environ, PYTHONPATH=LC + "/gguf-py"))
            out = r.stdout + r.stderr
            P("surgery tail: %s" % out[-300:])
            if '"R12_SURGERY_V2_DONE": true' not in out:
                raise RuntimeError("surgery failed")
            flag("DEPLOY_SURGERY")
        # D4: 双量化 (纯 Q4_K_M + 4x8 mixbit, flag 前置语法)
        if not have("DEPLOY_QUANT4"):
            if not os.path.exists(BIN + "/llama-quantize"):
                raise RuntimeError("llama-quantize missing (wait compile)")
            cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize %s %s q4_k_m"
                   % (BIN, F16, Q4))
            sh(cmd + " > %s/qtz4.log 2>&1" % K8B, 14400)
            if not os.path.exists(Q4) or os.path.getsize(Q4) < 4e9:
                raise RuntimeError("quant4 failed: " + sh("tail -6 %s/qtz4.log" % K8B)[-400:])
            flag("DEPLOY_QUANT4")
            P("quant4 ok (%.1fG)" % (os.path.getsize(Q4) / 1e9))
        if not have("DEPLOY_QUANT48"):
            ATTN = ("--tensor-type attn_q=q8_0 --tensor-type attn_k=q8_0 "
                    "--tensor-type attn_v=q8_0 --tensor-type attn_output=q8_0")
            cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize --token-embedding-type q8_0 "
                   "--output-tensor-type q8_0 %s %s %s q4_k_m" % (BIN, ATTN, F16, Q48))
            sh(cmd + " > %s/qtz48.log 2>&1" % K8B, 14400)
            if not os.path.exists(Q48) or os.path.getsize(Q48) < 4e9:
                raise RuntimeError("quant48 failed: " + sh("tail -6 %s/qtz48.log" % K8B)[-400:])
            flag("DEPLOY_QUANT48")
            P("quant48 ok (%.1fG)" % (os.path.getsize(Q48) / 1e9))
        # D5: 推仓 (两份 GGUF + sha, 第一时间)
        if not have("DEPLOY_PUSH"):
            import hashlib
            from huggingface_hub import HfApi
            api = HfApi(token=hf_token())
            for f, msg in ((Q4, "m13_r10_q4_k_m: pure 4bit (colab T4)"),
                           (Q48, "m13_r10_mixbit_4x8: R12 champion (colab T4)")):
                sha = hashlib.sha256(open(f, "rb").read()).hexdigest()
                name = os.path.basename(f)
                api.upload_file(path_or_fileobj=f, path_in_repo=name,
                                repo_id="tchbcb/samai-9b", repo_type="model",
                                commit_message=msg)
                open(K8B + "/" + name.replace(".gguf", "_sha256.txt"), "w").write(sha + "  " + name + "\n")
                api.upload_file(path_or_fileobj=K8B + "/" + name.replace(".gguf", "_sha256.txt"),
                                path_in_repo=name.replace(".gguf", "_sha256.txt"),
                                repo_id="tchbcb/samai-9b", repo_type="model")
                P("pushed %s sha=%s" % (name, sha[:12]))
            flag("DEPLOY_PUSH")
        # D6: serving (Q4_K_M 纯4bit, :8080, key=1234, mmproj 挂视觉)
        if not have("DEPLOY_SERVE"):
            sh("pkill -f '[l]lama-server' 2>/dev/null; sleep 1", 20)
            ctx = "32768"
            cmd = (LD + " nohup %s/llama-server -m %s --mmproj %s --host 127.0.0.1 --port 8080 "
                   "--api-key 1234 -c %s -ngl 99 -fa on --parallel 2 > %s/llama_server.log 2>&1 &"
                   % (BIN, Q4, MMPROJ, ctx, K8B))
            sh(cmd, 30)
            time.sleep(12)
            r = sh("curl -s -o /dev/null -w '%{http_code}' --max-time 5 -H 'Authorization: Bearer 1234' "
                   "http://127.0.0.1:8080/health; echo; nvidia-smi --query-gpu=memory.used --format=csv,noheader", 40)
            P("serve check: " + r)
            if "200" not in r:
                P("srv log: " + sh("tail -12 %s/llama_server.log" % K8B))
                raise RuntimeError("llama-server not healthy")
            flag("DEPLOY_SERVE")
            P("SERVE OK :8080 key=1234")
        P("==== DEPLOY_ALL_DONE in %.0fs ====" % (time.time() - t0))
        flag("DEPLOY_ALL_DONE", "%.0fs" % (time.time() - t0))
    except Exception as e:
        traceback.print_exc(file=log)
        flag("DEPLOY_FAIL", repr(e)[:200])
        P("==== DEPLOY_FAIL: %s ====" % repr(e)[:200])

if __name__ == "__main__":
    main()