samai-9b / artifacts /m15_scripts /colab_deploy2.py
tchbcb's picture
deploy2 v5: lora-scale arg removed (new llama.cpp)
2588ddd verified
Raw History Blame Contribute Delete
8.3 kB
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""colab_deploy2.py — Colab T4 单卡 12G RAM 现实路径: 底模量化 + LoRA 运行时挂载
血统: W_eff = quant(B) + scale*Δ10 (Δ10=续训链完整累积 delta, 相对原始底模)
流程: E1 convert底模->f16 -> E2 surgery -> E3 双量化 -> E4 lora->gguf -> E5 推仓 -> E6 serve
RAM 全程 <3G (convert/quantize 均 mmap), 绕开 cgroup OOM
幂等旗标: /content/k8b/D2_* ; 日志: /content/k8b/colab_deploy2.log"""
import os, subprocess, sys, time, traceback
K8B = "/content/k8b"
K27 = "/content/k27"
BASE = K27 + "/Qwen3.5-9B"
LC = "/content/lcbuild/llama.cpp"
BIN = "/content/lcbin"
F16 = K8B + "/base_f16.gguf"
Q4 = K8B + "/base_q4_k_m.gguf"
Q48 = K8B + "/base_mixbit_4x8.gguf"
ADP = K8B + "/m12_adapters/m13_r10"
LORA = K8B + "/m13_r10_lora.gguf"
MMPROJ = K8B + "/mmproj_m11.gguf"
LOG = K8B + "/colab_deploy2.log"
os.makedirs(K8B, exist_ok=True)
log = open(LOG, "a", buffering=1)
def P(m):
log.write("[%s] %s\n" % (time.strftime("%m-%d %H:%M:%S"), m))
def sh(c, t=14400):
p = subprocess.run(c, shell=True, capture_output=True, text=True, timeout=t, errors="replace")
return ((p.stdout or "") + (p.stderr or ""))[-1500:]
def flag(n, c="1"):
open(K8B + "/" + n, "w").write(str(c)[:400])
def have(n):
return os.path.exists(K8B + "/" + n)
def hf_token():
return open("/root/.cache/huggingface/token").read().strip()
LD = "LD_LIBRARY_PATH=/content/lcbin:/usr/lib64-nvidia:/usr/local/nvidia/lib64:/usr/lib/x86_64-linux-gnu"
def main():
t0 = time.time()
P("==== colab deploy2 (lora-mount path) start ====")
try:
# E1: convert 底模 -> f16 GGUF (mmap, RAM 友好)
if not have("D2_CONVERT"):
if not os.path.exists(LC + "/convert_hf_to_gguf.py"):
raise RuntimeError("llama.cpp src missing")
r = sh("cd %s && NO_LOCAL_GGUF=1 PYTHONPATH=%s/gguf-py python3 convert_hf_to_gguf.py "
"%s --outfile %s --outtype f16 > %s/convert2.log 2>&1; echo RC=$?; tail -2 %s/convert2.log"
% (LC, LC, BASE, F16, K8B, K8B), 14400)
P("convert: " + r[-250:])
if not os.path.exists(F16) or os.path.getsize(F16) < 15e9:
raise RuntimeError("convert failed") # [FIX] 只信文件, RC 被 tqdm stderr 淹没 (坑#69)
flag("D2_CONVERT")
P("convert ok (%.1fG, %.0fs)" % (os.path.getsize(F16) / 1e9, time.time() - t0))
# 删底模 HF 缓存 (symlink BASE 指向它, convert 后无用)
if os.path.islink(BASE):
tgt = os.path.realpath(BASE)
os.remove(BASE)
sh("rm -rf %s/hf/models--Qwen--Qwen3.5-9B %s" % (K27, tgt), 900)
P("base cache freed")
# E2: surgery v2 (bc->32, nx->0)
if not have("D2_SURGERY"):
r = subprocess.run([sys.executable, K8B + "/r12_surgery.py", F16],
capture_output=True, text=True, timeout=2400,
env=dict(os.environ, PYTHONPATH=LC + "/gguf-py"))
outp = r.stdout + r.stderr
P("surgery: %s" % outp[-300:])
if '"R12_SURGERY_V2_DONE": true' not in outp:
raise RuntimeError("surgery failed")
flag("D2_SURGERY")
# E3: 双量化 (等编译产物)
if not os.path.exists(BIN + "/llama-quantize"):
P("llama-quantize 未就绪, 等编译 (build pct: %s)" %
sh("grep -oE '\\[ *[0-9]+%\\]' /content/lcbuild/llama.cpp/build.log 2>/dev/null | tail -1", 15))
for i in range(360):
time.sleep(30)
if os.path.exists(BIN + "/llama-quantize"):
break
if have("COMPILE_FAIL"):
raise RuntimeError("compile failed while waiting")
if not os.path.exists(BIN + "/llama-quantize"):
raise RuntimeError("quantize binary timeout")
if not have("D2_QUANT4"):
cmd = "CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize %s %s q4_k_m" % (BIN, F16, Q4)
sh(cmd + " > %s/qtz4.log 2>&1" % K8B, 14400)
if not os.path.exists(Q4) or os.path.getsize(Q4) < 4e9:
raise RuntimeError("quant4 failed: " + sh("tail -6 %s/qtz4.log" % K8B)[-300:])
flag("D2_QUANT4")
P("quant4 ok (%.1fG)" % (os.path.getsize(Q4) / 1e9))
if not have("D2_QUANT48"):
ATTN = ("--tensor-type attn_q=q8_0 --tensor-type attn_k=q8_0 "
"--tensor-type attn_v=q8_0 --tensor-type attn_output=q8_0")
cmd = ("CUDA_VISIBLE_DEVICES= " + LD + " %s/llama-quantize --token-embedding-type q8_0 "
"--output-tensor-type q8_0 %s %s %s q4_k_m" % (BIN, ATTN, F16, Q48))
sh(cmd + " > %s/qtz48.log 2>&1" % K8B, 14400)
if not os.path.exists(Q48) or os.path.getsize(Q48) < 4e9:
raise RuntimeError("quant48 failed: " + sh("tail -6 %s/qtz48.log" % K8B)[-300:])
flag("D2_QUANT48")
P("quant48 ok (%.1fG)" % (os.path.getsize(Q48) / 1e9))
sh("rm -f %s" % F16, 30) # 释放 18.5G
P("f16 freed")
# E4: adapter -> lora.gguf (PEFT 242M, RAM 小)
if not have("D2_LORA"):
r = sh("cd %s && PYTHONPATH=%s/gguf-py python3 convert_lora_to_gguf.py "
"%s --outfile %s > %s/lora_conv.log 2>&1; echo RC=$?; tail -3 %s/lora_conv.log"
% (LC, LC, ADP, LORA, K8B, K8B), 3600)
P("lora conv: " + r[-300:])
if not os.path.exists(LORA) or os.path.getsize(LORA) < 1e8: # [FIX] file-only check
raise RuntimeError("lora convert failed")
flag("D2_LORA")
P("lora.gguf ok (%.0fM)" % (os.path.getsize(LORA) / 1e6))
# E5: 推仓 (第一时间: lora 先行, 再两份 base 盘 + sha)
if not have("D2_PUSH"):
import hashlib
from huggingface_hub import HfApi
api = HfApi(token=hf_token())
items = [(LORA, "m13_r10_lora.gguf: R12 adapter as gguf-lora (run-time mount)"),
(Q4, "base_q4_k_m.gguf: pure 4bit base for lora mount (colab T4)"),
(Q48, "base_mixbit_4x8.gguf: 4x8 base for lora mount (colab T4)")]
for f, msg in items:
sha = hashlib.sha256(open(f, "rb").read()).hexdigest()
name = os.path.basename(f)
api.upload_file(path_or_fileobj=f, path_in_repo=name,
repo_id="tchbcb/samai-9b", repo_type="model", commit_message=msg)
open(K8B + "/" + name + ".sha256", "w").write(sha + " " + name + "\n")
api.upload_file(path_or_fileobj=K8B + "/" + name + ".sha256",
path_in_repo=name + ".sha256", repo_id="tchbcb/samai-9b", repo_type="model")
P("pushed %s sha=%s" % (name, sha[:12]))
flag("D2_PUSH")
# E6: serve (Q4_K_M + LoRA 挂载, :8080, key=1234)
if not have("D2_SERVE"):
sh("pkill -f '[l]lama-server' 2>/dev/null; sleep 1", 20)
cmd = (LD + " nohup %s/llama-server -m %s --lora %s --mmproj %s "
"--host 127.0.0.1 --port 8080 --api-key 1234 -c 32768 -ngl 99 -fa on "
"--parallel 2 > %s/llama_server.log 2>&1 &" % (BIN, Q4, LORA, MMPROJ, K8B))
sh(cmd, 30)
time.sleep(15)
r = sh("curl -s -o /dev/null -w '%{http_code}' --max-time 5 -H 'Authorization: Bearer 1234' "
"http://127.0.0.1:8080/health; echo; nvidia-smi --query-gpu=memory.used --format=csv,noheader", 40)
P("serve check: " + r)
if "200" not in r:
P("srv log: " + sh("tail -12 %s/llama_server.log" % K8B))
raise RuntimeError("llama-server not healthy")
flag("D2_SERVE")
P("SERVE OK :8080 key=1234 (Q4 + lora mount)")
P("==== DEPLOY2_ALL_DONE in %.0fs ====" % (time.time() - t0))
flag("D2_ALL_DONE", "%.0fs" % (time.time() - t0))
except Exception as e:
traceback.print_exc(file=log)
flag("D2_FAIL", repr(e)[:200])
P("==== D2_FAIL: %s ====" % repr(e)[:200])
if __name__ == "__main__":
main()