samai-9b / artifacts /m15_scripts /colab_deploy.py
tchbcb's picture
deploy v6: AutoConfig import fix
50adde2 verified
Raw History Blame Contribute Delete
8.69 kB
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""colab_deploy.py — Colab T4 单卡 m13_r10 部署链 (merge→convert→surgery→双量化→推仓→serving)
血统: 底模 Qwen/Qwen3.5-9B ⊕ m13_r10 adapter (r10 权重已含 r6..r10, PeftModel 续训链)
RAM 12G 策略: merge device_map=auto (GPU 13G + CPU 溢出); 磁盘时序: convert 后删 merged
幂等旗标: /content/k8b/DEPLOY_* ; 日志: /content/k8b/colab_deploy.log"""
import json, os, subprocess, sys, time, traceback
K8B = "/content/k8b"
K27 = "/content/k27"
BASE = K27 + "/Qwen3.5-9B"
MERGED = K8B + "/merged_r10_hf"
ADP = K8B + "/m12_adapters/m13_r10"
LC = "/content/lcbuild/llama.cpp"
BIN = "/content/lcbin"
F16 = K8B + "/m13_r10_f16.gguf"
Q4 = K8B + "/m13_r10_q4_k_m.gguf"
Q48 = K8B + "/m13_r10_mixbit_4x8.gguf"
MMPROJ = K8B + "/mmproj_m11.gguf"
LOG = K8B + "/colab_deploy.log"
os.makedirs(K8B, exist_ok=True)
log = open(LOG, "a", buffering=1)
def P(m):
log.write("[%s] %s\n" % (time.strftime("%m-%d %H:%M:%S"), m))
def sh(c, t=7200):
p = subprocess.run(c, shell=True, capture_output=True, text=True, timeout=t, errors="replace")
return ((p.stdout or "") + (p.stderr or ""))[-1500:]
def flag(n, c="1"):
open(K8B + "/" + n, "w").write(str(c)[:400])
def have(n):
return os.path.exists(K8B + "/" + n)
def hf_token():
return open("/root/.cache/huggingface/token").read().strip()
LD = "LD_LIBRARY_PATH=/content/lcbin:/usr/local/nvidia/lib64:/usr/lib/x86_64-linux-gnu"
def main():
t0 = time.time()
P("==== colab deploy start ====")
try:
# D1: merge (device_map auto: T4 16G 吃大头, RAM 溢出)
if not have("DEPLOY_MERGE"):
if not os.path.exists(BASE + "/config.json"):
raise RuntimeError("base missing (download first)")
import torch
from transformers import AutoProcessor, AutoModelForImageTextToText, AutoConfig
from peft import PeftModel
from accelerate import infer_auto_device_map
with torch.device("meta"):
skel = AutoModelForImageTextToText.from_config(
AutoConfig.from_pretrained(BASE))
dm = infer_auto_device_map(skel, max_memory={0: "13500MiB", "cpu": "48000MiB"})
del skel
n_disk = sum(1 for v in dm.values() if v == "disk")
P("device_map: gpu+cpu, disk=%d" % n_disk)
assert n_disk == 0, "device_map still has disk offload"
kw = dict(low_cpu_mem_usage=True)
try:
m = AutoModelForImageTextToText.from_pretrained(
BASE, dtype=torch.float16, device_map=dm, **kw)
except TypeError:
m = AutoModelForImageTextToText.from_pretrained(
BASE, torch_dtype=torch.float16, device_map=dm, **kw)
P("loaded; GPU %.1fG" % (torch.cuda.memory_allocated() / 1e9))
m = PeftModel.from_pretrained(m, ADP)
m = m.merge_and_unload()
m.save_pretrained(MERGED, safe_serialization=True)
try:
AutoProcessor.from_pretrained(BASE).save_pretrained(MERGED)
except Exception as e:
P("proc save skip %s" % repr(e)[:80])
del m
import gc, torch as _t
gc.collect(); _t.cuda.empty_cache()
flag("DEPLOY_MERGE")
P("merge ok")
# D2: convert f16 (mmap 友好), 完成后删 merged 省盘
if not have("DEPLOY_CONVERT"):
if not os.path.exists(LC + "/convert_hf_to_gguf.py"):
raise RuntimeError("llama.cpp src missing (compile chain clones it)")
r = sh("cd %s && NO_LOCAL_GGUF=1 PYTHONPATH=%s/gguf-py python3 convert_hf_to_gguf.py "
"%s --outfile %s --outtype f16 > %s/convert.log 2>&1; echo RC=$?; tail -2 %s/convert.log"
% (LC, LC, MERGED, F16, K8B, K8B), 14400)
P("convert: " + r[-250:])
if "RC=0" not in r or not os.path.exists(F16) or os.path.getsize(F16) < 15e9:
raise RuntimeError("convert failed")
flag("DEPLOY_CONVERT")
P("convert ok (%.1fG)" % (os.path.getsize(F16) / 1e9))
sh("rm -rf %s" % MERGED, 900)
P("merged_r10_hf removed (disk)")
# D3: surgery v2 (bc->32, nx->0)
if not have("DEPLOY_SURGERY"):
r = subprocess.run([sys.executable, K8B + "/r12_surgery.py", F16],
capture_output=True, text=True, timeout=2400,
env=dict(os.environ, PYTHONPATH=LC + "/gguf-py"))
out = r.stdout + r.stderr
P("surgery tail: %s" % out[-300:])
if '"R12_SURGERY_V2_DONE": true' not in out:
raise RuntimeError("surgery failed")
flag("DEPLOY_SURGERY")
# D4: 双量化 (纯 Q4_K_M + 4x8 mixbit, flag 前置语法)
if not have("DEPLOY_QUANT4"):
if not os.path.exists(BIN + "/llama-quantize"):
raise RuntimeError("llama-quantize missing (wait compile)")
cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize %s %s q4_k_m"
% (BIN, F16, Q4))
sh(cmd + " > %s/qtz4.log 2>&1" % K8B, 14400)
if not os.path.exists(Q4) or os.path.getsize(Q4) < 4e9:
raise RuntimeError("quant4 failed: " + sh("tail -6 %s/qtz4.log" % K8B)[-400:])
flag("DEPLOY_QUANT4")
P("quant4 ok (%.1fG)" % (os.path.getsize(Q4) / 1e9))
if not have("DEPLOY_QUANT48"):
ATTN = ("--tensor-type attn_q=q8_0 --tensor-type attn_k=q8_0 "
"--tensor-type attn_v=q8_0 --tensor-type attn_output=q8_0")
cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize --token-embedding-type q8_0 "
"--output-tensor-type q8_0 %s %s %s q4_k_m" % (BIN, ATTN, F16, Q48))
sh(cmd + " > %s/qtz48.log 2>&1" % K8B, 14400)
if not os.path.exists(Q48) or os.path.getsize(Q48) < 4e9:
raise RuntimeError("quant48 failed: " + sh("tail -6 %s/qtz48.log" % K8B)[-400:])
flag("DEPLOY_QUANT48")
P("quant48 ok (%.1fG)" % (os.path.getsize(Q48) / 1e9))
# D5: 推仓 (两份 GGUF + sha, 第一时间)
if not have("DEPLOY_PUSH"):
import hashlib
from huggingface_hub import HfApi
api = HfApi(token=hf_token())
for f, msg in ((Q4, "m13_r10_q4_k_m: pure 4bit (colab T4)"),
(Q48, "m13_r10_mixbit_4x8: R12 champion (colab T4)")):
sha = hashlib.sha256(open(f, "rb").read()).hexdigest()
name = os.path.basename(f)
api.upload_file(path_or_fileobj=f, path_in_repo=name,
repo_id="tchbcb/samai-9b", repo_type="model",
commit_message=msg)
open(K8B + "/" + name.replace(".gguf", "_sha256.txt"), "w").write(sha + " " + name + "\n")
api.upload_file(path_or_fileobj=K8B + "/" + name.replace(".gguf", "_sha256.txt"),
path_in_repo=name.replace(".gguf", "_sha256.txt"),
repo_id="tchbcb/samai-9b", repo_type="model")
P("pushed %s sha=%s" % (name, sha[:12]))
flag("DEPLOY_PUSH")
# D6: serving (Q4_K_M 纯4bit, :8080, key=1234, mmproj 挂视觉)
if not have("DEPLOY_SERVE"):
sh("pkill -f '[l]lama-server' 2>/dev/null; sleep 1", 20)
ctx = "32768"
cmd = (LD + " nohup %s/llama-server -m %s --mmproj %s --host 127.0.0.1 --port 8080 "
"--api-key 1234 -c %s -ngl 99 -fa on --parallel 2 > %s/llama_server.log 2>&1 &"
% (BIN, Q4, MMPROJ, ctx, K8B))
sh(cmd, 30)
time.sleep(12)
r = sh("curl -s -o /dev/null -w '%{http_code}' --max-time 5 -H 'Authorization: Bearer 1234' "
"http://127.0.0.1:8080/health; echo; nvidia-smi --query-gpu=memory.used --format=csv,noheader", 40)
P("serve check: " + r)
if "200" not in r:
P("srv log: " + sh("tail -12 %s/llama_server.log" % K8B))
raise RuntimeError("llama-server not healthy")
flag("DEPLOY_SERVE")
P("SERVE OK :8080 key=1234")
P("==== DEPLOY_ALL_DONE in %.0fs ====" % (time.time() - t0))
flag("DEPLOY_ALL_DONE", "%.0fs" % (time.time() - t0))
except Exception as e:
traceback.print_exc(file=log)
flag("DEPLOY_FAIL", repr(e)[:200])
P("==== DEPLOY_FAIL: %s ====" % repr(e)[:200])
if __name__ == "__main__":
main()