#!/usr/bin/env python3 # -*- coding: utf-8 -*- """colab_deploy2.py — Colab T4 单卡 12G RAM 现实路径: 底模量化 + LoRA 运行时挂载 血统: W_eff = quant(B) + scale*Δ10 (Δ10=续训链完整累积 delta, 相对原始底模) 流程: E1 convert底模->f16 -> E2 surgery -> E3 双量化 -> E4 lora->gguf -> E5 推仓 -> E6 serve RAM 全程 <3G (convert/quantize 均 mmap), 绕开 cgroup OOM 幂等旗标: /content/k8b/D2_* ; 日志: /content/k8b/colab_deploy2.log""" import os, subprocess, sys, time, traceback K8B = "/content/k8b" K27 = "/content/k27" BASE = K27 + "/Qwen3.5-9B" LC = "/content/lcbuild/llama.cpp" BIN = "/content/lcbin" F16 = K8B + "/base_f16.gguf" Q4 = K8B + "/base_q4_k_m.gguf" Q48 = K8B + "/base_mixbit_4x8.gguf" ADP = K8B + "/m12_adapters/m13_r10" LORA = K8B + "/m13_r10_lora.gguf" MMPROJ = K8B + "/mmproj_m11.gguf" LOG = K8B + "/colab_deploy2.log" os.makedirs(K8B, exist_ok=True) log = open(LOG, "a", buffering=1) def P(m): log.write("[%s] %s\n" % (time.strftime("%m-%d %H:%M:%S"), m)) def sh(c, t=14400): p = subprocess.run(c, shell=True, capture_output=True, text=True, timeout=t, errors="replace") return ((p.stdout or "") + (p.stderr or ""))[-1500:] def flag(n, c="1"): open(K8B + "/" + n, "w").write(str(c)[:400]) def have(n): return os.path.exists(K8B + "/" + n) def hf_token(): return open("/root/.cache/huggingface/token").read().strip() LD = "LD_LIBRARY_PATH=/content/lcbin:/usr/lib64-nvidia:/usr/local/nvidia/lib64:/usr/lib/x86_64-linux-gnu" def main(): t0 = time.time() P("==== colab deploy2 (lora-mount path) start ====") try: # E1: convert 底模 -> f16 GGUF (mmap, RAM 友好) if not have("D2_CONVERT"): if not os.path.exists(LC + "/convert_hf_to_gguf.py"): raise RuntimeError("llama.cpp src missing") r = sh("cd %s && NO_LOCAL_GGUF=1 PYTHONPATH=%s/gguf-py python3 convert_hf_to_gguf.py " "%s --outfile %s --outtype f16 > %s/convert2.log 2>&1; echo RC=$?; tail -2 %s/convert2.log" % (LC, LC, BASE, F16, K8B, K8B), 14400) P("convert: " + r[-250:]) if not os.path.exists(F16) or os.path.getsize(F16) < 15e9: raise RuntimeError("convert failed") # [FIX] 只信文件, RC 被 tqdm stderr 淹没 (坑#69) flag("D2_CONVERT") P("convert ok (%.1fG, %.0fs)" % (os.path.getsize(F16) / 1e9, time.time() - t0)) # 删底模 HF 缓存 (symlink BASE 指向它, convert 后无用) if os.path.islink(BASE): tgt = os.path.realpath(BASE) os.remove(BASE) sh("rm -rf %s/hf/models--Qwen--Qwen3.5-9B %s" % (K27, tgt), 900) P("base cache freed") # E2: surgery v2 (bc->32, nx->0) if not have("D2_SURGERY"): r = subprocess.run([sys.executable, K8B + "/r12_surgery.py", F16], capture_output=True, text=True, timeout=2400, env=dict(os.environ, PYTHONPATH=LC + "/gguf-py")) outp = r.stdout + r.stderr P("surgery: %s" % outp[-300:]) if '"R12_SURGERY_V2_DONE": true' not in outp: raise RuntimeError("surgery failed") flag("D2_SURGERY") # E3: 双量化 (等编译产物) if not os.path.exists(BIN + "/llama-quantize"): P("llama-quantize 未就绪, 等编译 (build pct: %s)" % sh("grep -oE '\\[ *[0-9]+%\\]' /content/lcbuild/llama.cpp/build.log 2>/dev/null | tail -1", 15)) for i in range(360): time.sleep(30) if os.path.exists(BIN + "/llama-quantize"): break if have("COMPILE_FAIL"): raise RuntimeError("compile failed while waiting") if not os.path.exists(BIN + "/llama-quantize"): raise RuntimeError("quantize binary timeout") if not have("D2_QUANT4"): cmd = "CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize %s %s q4_k_m" % (BIN, F16, Q4) sh(cmd + " > %s/qtz4.log 2>&1" % K8B, 14400) if not os.path.exists(Q4) or os.path.getsize(Q4) < 4e9: raise RuntimeError("quant4 failed: " + sh("tail -6 %s/qtz4.log" % K8B)[-300:]) flag("D2_QUANT4") P("quant4 ok (%.1fG)" % (os.path.getsize(Q4) / 1e9)) if not have("D2_QUANT48"): ATTN = ("--tensor-type attn_q=q8_0 --tensor-type attn_k=q8_0 " "--tensor-type attn_v=q8_0 --tensor-type attn_output=q8_0") cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize --token-embedding-type q8_0 " "--output-tensor-type q8_0 %s %s %s q4_k_m" % (BIN, ATTN, F16, Q48)) sh(cmd + " > %s/qtz48.log 2>&1" % K8B, 14400) if not os.path.exists(Q48) or os.path.getsize(Q48) < 4e9: raise RuntimeError("quant48 failed: " + sh("tail -6 %s/qtz48.log" % K8B)[-300:]) flag("D2_QUANT48") P("quant48 ok (%.1fG)" % (os.path.getsize(Q48) / 1e9)) sh("rm -f %s" % F16, 30) # 释放 18.5G P("f16 freed") # E4: adapter -> lora.gguf (PEFT 242M, RAM 小) if not have("D2_LORA"): r = sh("cd %s && PYTHONPATH=%s/gguf-py python3 convert_lora_to_gguf.py " "%s --outfile %s > %s/lora_conv.log 2>&1; echo RC=$?; tail -3 %s/lora_conv.log" % (LC, LC, ADP, LORA, K8B, K8B), 3600) P("lora conv: " + r[-300:]) if not os.path.exists(LORA) or os.path.getsize(LORA) < 1e8: # [FIX] file-only check raise RuntimeError("lora convert failed") flag("D2_LORA") P("lora.gguf ok (%.0fM)" % (os.path.getsize(LORA) / 1e6)) # E5: 推仓 (第一时间: lora 先行, 再两份 base 盘 + sha) if not have("D2_PUSH"): import hashlib from huggingface_hub import HfApi api = HfApi(token=hf_token()) items = [(LORA, "m13_r10_lora.gguf: R12 adapter as gguf-lora (run-time mount)"), (Q4, "base_q4_k_m.gguf: pure 4bit base for lora mount (colab T4)"), (Q48, "base_mixbit_4x8.gguf: 4x8 base for lora mount (colab T4)")] for f, msg in items: sha = hashlib.sha256(open(f, "rb").read()).hexdigest() name = os.path.basename(f) api.upload_file(path_or_fileobj=f, path_in_repo=name, repo_id="tchbcb/samai-9b", repo_type="model", commit_message=msg) open(K8B + "/" + name + ".sha256", "w").write(sha + " " + name + "\n") api.upload_file(path_or_fileobj=K8B + "/" + name + ".sha256", path_in_repo=name + ".sha256", repo_id="tchbcb/samai-9b", repo_type="model") P("pushed %s sha=%s" % (name, sha[:12])) flag("D2_PUSH") # E6: serve (Q4_K_M + LoRA 挂载, :8080, key=1234) if not have("D2_SERVE"): sh("pkill -f '[l]lama-server' 2>/dev/null; sleep 1", 20) cmd = (LD + " nohup %s/llama-server -m %s --lora %s --mmproj %s " "--host 127.0.0.1 --port 8080 --api-key 1234 -c 32768 -ngl 99 -fa on " "--parallel 2 > %s/llama_server.log 2>&1 &" % (BIN, Q4, LORA, MMPROJ, K8B)) sh(cmd, 30) time.sleep(15) r = sh("curl -s -o /dev/null -w '%{http_code}' --max-time 5 -H 'Authorization: Bearer 1234' " "http://127.0.0.1:8080/health; echo; nvidia-smi --query-gpu=memory.used --format=csv,noheader", 40) P("serve check: " + r) if "200" not in r: P("srv log: " + sh("tail -12 %s/llama_server.log" % K8B)) raise RuntimeError("llama-server not healthy") flag("D2_SERVE") P("SERVE OK :8080 key=1234 (Q4 + lora mount)") P("==== DEPLOY2_ALL_DONE in %.0fs ====" % (time.time() - t0)) flag("D2_ALL_DONE", "%.0fs" % (time.time() - t0)) except Exception as e: traceback.print_exc(file=log) flag("D2_FAIL", repr(e)[:200]) P("==== D2_FAIL: %s ====" % repr(e)[:200]) if __name__ == "__main__": main()