#!/usr/bin/env python3 # -*- coding: utf-8 -*- """colab_deploy.py — Colab T4 单卡 m13_r10 部署链 (merge→convert→surgery→双量化→推仓→serving) 血统: 底模 Qwen/Qwen3.5-9B ⊕ m13_r10 adapter (r10 权重已含 r6..r10, PeftModel 续训链) RAM 12G 策略: merge device_map=auto (GPU 13G + CPU 溢出); 磁盘时序: convert 后删 merged 幂等旗标: /content/k8b/DEPLOY_* ; 日志: /content/k8b/colab_deploy.log""" import json, os, subprocess, sys, time, traceback K8B = "/content/k8b" K27 = "/content/k27" BASE = K27 + "/Qwen3.5-9B" MERGED = K8B + "/merged_r10_hf" ADP = K8B + "/m12_adapters/m13_r10" LC = "/content/lcbuild/llama.cpp" BIN = "/content/lcbin" F16 = K8B + "/m13_r10_f16.gguf" Q4 = K8B + "/m13_r10_q4_k_m.gguf" Q48 = K8B + "/m13_r10_mixbit_4x8.gguf" MMPROJ = K8B + "/mmproj_m11.gguf" LOG = K8B + "/colab_deploy.log" os.makedirs(K8B, exist_ok=True) log = open(LOG, "a", buffering=1) def P(m): log.write("[%s] %s\n" % (time.strftime("%m-%d %H:%M:%S"), m)) def sh(c, t=7200): p = subprocess.run(c, shell=True, capture_output=True, text=True, timeout=t, errors="replace") return ((p.stdout or "") + (p.stderr or ""))[-1500:] def flag(n, c="1"): open(K8B + "/" + n, "w").write(str(c)[:400]) def have(n): return os.path.exists(K8B + "/" + n) def hf_token(): return open("/root/.cache/huggingface/token").read().strip() LD = "LD_LIBRARY_PATH=/content/lcbin:/usr/local/nvidia/lib64:/usr/lib/x86_64-linux-gnu" def main(): t0 = time.time() P("==== colab deploy start ====") try: # D1: merge (device_map auto: T4 16G 吃大头, RAM 溢出) if not have("DEPLOY_MERGE"): if not os.path.exists(BASE + "/config.json"): raise RuntimeError("base missing (download first)") import torch from transformers import AutoProcessor, AutoModelForImageTextToText, AutoConfig from peft import PeftModel from accelerate import infer_auto_device_map with torch.device("meta"): skel = AutoModelForImageTextToText.from_config( AutoConfig.from_pretrained(BASE)) dm = infer_auto_device_map(skel, max_memory={0: "13500MiB", "cpu": "48000MiB"}) del skel n_disk = sum(1 for v in dm.values() if v == "disk") P("device_map: gpu+cpu, disk=%d" % n_disk) assert n_disk == 0, "device_map still has disk offload" kw = dict(low_cpu_mem_usage=True) try: m = AutoModelForImageTextToText.from_pretrained( BASE, dtype=torch.float16, device_map=dm, **kw) except TypeError: m = AutoModelForImageTextToText.from_pretrained( BASE, torch_dtype=torch.float16, device_map=dm, **kw) P("loaded; GPU %.1fG" % (torch.cuda.memory_allocated() / 1e9)) m = PeftModel.from_pretrained(m, ADP) m = m.merge_and_unload() m.save_pretrained(MERGED, safe_serialization=True) try: AutoProcessor.from_pretrained(BASE).save_pretrained(MERGED) except Exception as e: P("proc save skip %s" % repr(e)[:80]) del m import gc, torch as _t gc.collect(); _t.cuda.empty_cache() flag("DEPLOY_MERGE") P("merge ok") # D2: convert f16 (mmap 友好), 完成后删 merged 省盘 if not have("DEPLOY_CONVERT"): if not os.path.exists(LC + "/convert_hf_to_gguf.py"): raise RuntimeError("llama.cpp src missing (compile chain clones it)") r = sh("cd %s && NO_LOCAL_GGUF=1 PYTHONPATH=%s/gguf-py python3 convert_hf_to_gguf.py " "%s --outfile %s --outtype f16 > %s/convert.log 2>&1; echo RC=$?; tail -2 %s/convert.log" % (LC, LC, MERGED, F16, K8B, K8B), 14400) P("convert: " + r[-250:]) if "RC=0" not in r or not os.path.exists(F16) or os.path.getsize(F16) < 15e9: raise RuntimeError("convert failed") flag("DEPLOY_CONVERT") P("convert ok (%.1fG)" % (os.path.getsize(F16) / 1e9)) sh("rm -rf %s" % MERGED, 900) P("merged_r10_hf removed (disk)") # D3: surgery v2 (bc->32, nx->0) if not have("DEPLOY_SURGERY"): r = subprocess.run([sys.executable, K8B + "/r12_surgery.py", F16], capture_output=True, text=True, timeout=2400, env=dict(os.environ, PYTHONPATH=LC + "/gguf-py")) out = r.stdout + r.stderr P("surgery tail: %s" % out[-300:]) if '"R12_SURGERY_V2_DONE": true' not in out: raise RuntimeError("surgery failed") flag("DEPLOY_SURGERY") # D4: 双量化 (纯 Q4_K_M + 4x8 mixbit, flag 前置语法) if not have("DEPLOY_QUANT4"): if not os.path.exists(BIN + "/llama-quantize"): raise RuntimeError("llama-quantize missing (wait compile)") cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize %s %s q4_k_m" % (BIN, F16, Q4)) sh(cmd + " > %s/qtz4.log 2>&1" % K8B, 14400) if not os.path.exists(Q4) or os.path.getsize(Q4) < 4e9: raise RuntimeError("quant4 failed: " + sh("tail -6 %s/qtz4.log" % K8B)[-400:]) flag("DEPLOY_QUANT4") P("quant4 ok (%.1fG)" % (os.path.getsize(Q4) / 1e9)) if not have("DEPLOY_QUANT48"): ATTN = ("--tensor-type attn_q=q8_0 --tensor-type attn_k=q8_0 " "--tensor-type attn_v=q8_0 --tensor-type attn_output=q8_0") cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize --token-embedding-type q8_0 " "--output-tensor-type q8_0 %s %s %s q4_k_m" % (BIN, ATTN, F16, Q48)) sh(cmd + " > %s/qtz48.log 2>&1" % K8B, 14400) if not os.path.exists(Q48) or os.path.getsize(Q48) < 4e9: raise RuntimeError("quant48 failed: " + sh("tail -6 %s/qtz48.log" % K8B)[-400:]) flag("DEPLOY_QUANT48") P("quant48 ok (%.1fG)" % (os.path.getsize(Q48) / 1e9)) # D5: 推仓 (两份 GGUF + sha, 第一时间) if not have("DEPLOY_PUSH"): import hashlib from huggingface_hub import HfApi api = HfApi(token=hf_token()) for f, msg in ((Q4, "m13_r10_q4_k_m: pure 4bit (colab T4)"), (Q48, "m13_r10_mixbit_4x8: R12 champion (colab T4)")): sha = hashlib.sha256(open(f, "rb").read()).hexdigest() name = os.path.basename(f) api.upload_file(path_or_fileobj=f, path_in_repo=name, repo_id="tchbcb/samai-9b", repo_type="model", commit_message=msg) open(K8B + "/" + name.replace(".gguf", "_sha256.txt"), "w").write(sha + " " + name + "\n") api.upload_file(path_or_fileobj=K8B + "/" + name.replace(".gguf", "_sha256.txt"), path_in_repo=name.replace(".gguf", "_sha256.txt"), repo_id="tchbcb/samai-9b", repo_type="model") P("pushed %s sha=%s" % (name, sha[:12])) flag("DEPLOY_PUSH") # D6: serving (Q4_K_M 纯4bit, :8080, key=1234, mmproj 挂视觉) if not have("DEPLOY_SERVE"): sh("pkill -f '[l]lama-server' 2>/dev/null; sleep 1", 20) ctx = "32768" cmd = (LD + " nohup %s/llama-server -m %s --mmproj %s --host 127.0.0.1 --port 8080 " "--api-key 1234 -c %s -ngl 99 -fa on --parallel 2 > %s/llama_server.log 2>&1 &" % (BIN, Q4, MMPROJ, ctx, K8B)) sh(cmd, 30) time.sleep(12) r = sh("curl -s -o /dev/null -w '%{http_code}' --max-time 5 -H 'Authorization: Bearer 1234' " "http://127.0.0.1:8080/health; echo; nvidia-smi --query-gpu=memory.used --format=csv,noheader", 40) P("serve check: " + r) if "200" not in r: P("srv log: " + sh("tail -12 %s/llama_server.log" % K8B)) raise RuntimeError("llama-server not healthy") flag("DEPLOY_SERVE") P("SERVE OK :8080 key=1234") P("==== DEPLOY_ALL_DONE in %.0fs ====" % (time.time() - t0)) flag("DEPLOY_ALL_DONE", "%.0fs" % (time.time() - t0)) except Exception as e: traceback.print_exc(file=log) flag("DEPLOY_FAIL", repr(e)[:200]) P("==== DEPLOY_FAIL: %s ====" % repr(e)[:200]) if __name__ == "__main__": main()