Instructions to use tchbcb/samai-9b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- llama.cpp
How to use tchbcb/samai-9b with llama.cpp:
Install (macOS, Linux)
curl -LsSf https://llama.app/install.sh | sh # Start a local OpenAI-compatible server with a web UI: llama serve -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: llama cli -hf tchbcb/samai-9b:Q4_K_M
Install from WinGet (Windows)
winget install llama.cpp # Start a local OpenAI-compatible server with a web UI: llama serve -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: llama cli -hf tchbcb/samai-9b:Q4_K_M
Use pre-built binary
# Download pre-built binary from: # https://github.com/ggerganov/llama.cpp/releases # Start a local OpenAI-compatible server with a web UI: ./llama-server -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: ./llama-cli -hf tchbcb/samai-9b:Q4_K_M
Build from source code
git clone https://github.com/ggerganov/llama.cpp.git cd llama.cpp cmake -B build cmake --build build -j --target llama-server llama-cli # Start a local OpenAI-compatible server with a web UI: ./build/bin/llama-server -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: ./build/bin/llama-cli -hf tchbcb/samai-9b:Q4_K_M
Use Docker
docker model run hf.co/tchbcb/samai-9b:Q4_K_M
- LM Studio
- Jan
- Ollama
How to use tchbcb/samai-9b with Ollama:
ollama run hf.co/tchbcb/samai-9b:Q4_K_M
- Unsloth Desktop
- Pi
How to use tchbcb/samai-9b with Pi:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-9b:Q4_K_M
Configure the model in Pi
# Install Pi: npm install -g @earendil-works/pi-coding-agent # Add to ~/.pi/agent/models.json: { "providers": { "llama-cpp": { "baseUrl": "http://localhost:8080/v1", "api": "openai-completions", "apiKey": "none", "models": [ { "id": "tchbcb/samai-9b:Q4_K_M" } ] } } }Run Pi
# Start Pi in your project directory: pi
- Docker Model Runner
How to use tchbcb/samai-9b with Docker Model Runner:
docker model run hf.co/tchbcb/samai-9b:Q4_K_M
- Lemonade
How to use tchbcb/samai-9b with Lemonade:
Pull the model
# Download Lemonade from https://lemonade-server.ai/ lemonade pull tchbcb/samai-9b:Q4_K_M
Run and chat with the model
lemonade run user.samai-9b-Q4_K_M
List all available models
lemonade list
- Hermes Agent
How to use tchbcb/samai-9b with Hermes Agent:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-9b:Q4_K_M
Configure Hermes
# Install Hermes: curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash hermes setup # Point Hermes at the local server: hermes config set model.provider custom hermes config set model.base_url http://127.0.0.1:8080/v1 hermes config set model.default tchbcb/samai-9b:Q4_K_M
Run Hermes
hermes
- Atomic Chat
- OpenClaw
How to use tchbcb/samai-9b with OpenClaw:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-9b:Q4_K_M
Configure OpenClaw
# Install OpenClaw: npm install -g openclaw@latest # Register the local server and set it as the default model: openclaw onboard --non-interactive --mode local \ --auth-choice custom-api-key \ --custom-base-url http://127.0.0.1:8080/v1 \ --custom-model-id "tchbcb/samai-9b:Q4_K_M" \ --custom-provider-id llama-cpp \ --custom-compatibility openai \ --custom-text-input \ --accept-risk \ --skip-health
Run OpenClaw
openclaw agent --local --agent main --message "Hello from Hugging Face"
File size: 4,478 Bytes
3ba280a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 | #!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""colab_fetch.py — Colab T4 弹药回拉: adapter+mmproj+底模(18G) -> 自动点火 colab_deploy.py
坑#57 对策: snapshot_download 走 cache_dir + symlink, 避免 local_dir 嵌套
幂等旗标: /content/k8b/FETCH_* ; 日志: /content/k8b/colab_fetch.log"""
import os, subprocess, time, traceback
K8B = "/content/k8b"
K27 = "/content/k27"
BASE = K27 + "/Qwen3.5-9B"
LOG = K8B + "/colab_fetch.log"
os.makedirs(K8B, exist_ok=True)
log = open(LOG, "a", buffering=1)
def P(m):
log.write("[%s] %s\n" % (time.strftime("%m-%d %H:%M:%S"), m))
def sh(c, t=1800):
p = subprocess.run(c, shell=True, capture_output=True, text=True, timeout=t, errors="replace")
return ((p.stdout or "") + (p.stderr or ""))[-800:]
def flag(n, c="1"):
open(K8B + "/" + n, "w").write(str(c)[:400])
def have(n):
return os.path.exists(K8B + "/" + n)
def hf_token():
return open("/root/.cache/huggingface/token").read().strip()
def main():
t0 = time.time()
P("==== colab fetch start ====")
try:
from huggingface_hub import hf_hub_download, snapshot_download, HfApi
import shutil
tok = hf_token()
api = HfApi(token=tok)
# F1: 脚本弹药 (公仓 artifacts/m15_scripts/)
if not have("FETCH_SCRIPTS"):
os.makedirs(K8B, exist_ok=True)
for rp, dst in (
("artifacts/m15_scripts/colab_compile.py", K8B + "/colab_compile.py"),
("artifacts/m15_scripts/colab_deploy.py", K8B + "/colab_deploy.py"),
("artifacts/m15_scripts/r12_surgery.py", K8B + "/r12_surgery.py"),
):
if not os.path.exists(dst):
p = hf_hub_download("tchbcb/samai-9b", rp, token=tok)
shutil.copy(p, dst)
P("got %s (%d)" % (dst, os.path.getsize(dst)))
flag("FETCH_SCRIPTS")
# F2: r10 adapter (私仓 242M)
if not have("FETCH_ADP"):
for f in ("adapter_config.json", "adapter_model.safetensors",
"chat_template.jinja", "tokenizer.json", "tokenizer_config.json"):
d = K8B + "/m12_adapters/m13_r10"
os.makedirs(d, exist_ok=True)
if not os.path.exists(d + "/" + f):
p = hf_hub_download("tchbcb/samai-8b-M8", "m12_adapters/m13_r10/" + f, token=tok)
shutil.copy(p, d + "/" + f)
n = len(os.listdir(K8B + "/m12_adapters/m13_r10"))
if n < 5:
raise RuntimeError("adapter incomplete: %d files" % n)
P("adapter ok (%d files)" % n)
flag("FETCH_ADP", str(n))
# F3: mmproj (公仓 0.92G)
if not have("FETCH_MM"):
if not os.path.exists(K8B + "/mmproj_m11.gguf"):
p = hf_hub_download("tchbcb/samai-9b", "mmproj_m11.gguf", token=tok)
shutil.copy(p, K8B + "/mmproj_m11.gguf")
P("mmproj ok (%.2fG)" % (os.path.getsize(K8B + "/mmproj_m11.gguf") / 1e9))
flag("FETCH_MM")
# F4: 底模 18G (cache_dir 隔离 + symlink 防嵌套)
if not have("FETCH_BASE"):
if not os.path.exists(BASE + "/config.json"):
cache = K27 + "/hf"
os.makedirs(cache, exist_ok=True)
P("snapshot_download base (18G)...")
p = snapshot_download("Qwen/Qwen3.5-9B", cache_dir=cache, token=tok)
if os.path.islink(BASE):
os.remove(BASE)
elif os.path.isdir(BASE):
shutil.rmtree(BASE)
os.symlink(p, BASE)
total = sum(os.path.getsize(os.path.join(r, f))
for r, _, fs in os.walk(p) for f in fs)
P("base ok %s (%.1fG)" % (p, total / 1e9))
if total < 15e9:
raise RuntimeError("base too small: %.1fG" % (total / 1e9))
flag("FETCH_BASE")
flag("FETCH_ALL_DONE", "%.0fs" % (time.time() - t0))
P("==== FETCH_ALL_DONE, firing deploy ====")
# F5: 自动点火部署链 (分离)
sh("cd %s && rm -f DEPLOY_FAIL && setsid nohup python3 colab_deploy.py > /dev/null 2>&1 & sleep 1", 20)
P("deploy fired")
except Exception as e:
traceback.print_exc(file=log)
flag("FETCH_FAIL", repr(e)[:200])
P("==== FETCH_FAIL: %s ====" % repr(e)[:200])
if __name__ == "__main__":
main()
|