Instructions to use tchbcb/samai-9b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- llama.cpp
How to use tchbcb/samai-9b with llama.cpp:
Install (macOS, Linux)
curl -LsSf https://llama.app/install.sh | sh # Start a local OpenAI-compatible server with a web UI: llama serve -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: llama cli -hf tchbcb/samai-9b:Q4_K_M
Install from WinGet (Windows)
winget install llama.cpp # Start a local OpenAI-compatible server with a web UI: llama serve -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: llama cli -hf tchbcb/samai-9b:Q4_K_M
Use pre-built binary
# Download pre-built binary from: # https://github.com/ggerganov/llama.cpp/releases # Start a local OpenAI-compatible server with a web UI: ./llama-server -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: ./llama-cli -hf tchbcb/samai-9b:Q4_K_M
Build from source code
git clone https://github.com/ggerganov/llama.cpp.git cd llama.cpp cmake -B build cmake --build build -j --target llama-server llama-cli # Start a local OpenAI-compatible server with a web UI: ./build/bin/llama-server -hf tchbcb/samai-9b:Q4_K_M # Run inference directly in the terminal: ./build/bin/llama-cli -hf tchbcb/samai-9b:Q4_K_M
Use Docker
docker model run hf.co/tchbcb/samai-9b:Q4_K_M
- LM Studio
- Jan
- Ollama
How to use tchbcb/samai-9b with Ollama:
ollama run hf.co/tchbcb/samai-9b:Q4_K_M
- Unsloth Desktop
- Pi
How to use tchbcb/samai-9b with Pi:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-9b:Q4_K_M
Configure the model in Pi
# Install Pi: npm install -g @earendil-works/pi-coding-agent # Add to ~/.pi/agent/models.json: { "providers": { "llama-cpp": { "baseUrl": "http://localhost:8080/v1", "api": "openai-completions", "apiKey": "none", "models": [ { "id": "tchbcb/samai-9b:Q4_K_M" } ] } } }Run Pi
# Start Pi in your project directory: pi
- Docker Model Runner
How to use tchbcb/samai-9b with Docker Model Runner:
docker model run hf.co/tchbcb/samai-9b:Q4_K_M
- Lemonade
How to use tchbcb/samai-9b with Lemonade:
Pull the model
# Download Lemonade from https://lemonade-server.ai/ lemonade pull tchbcb/samai-9b:Q4_K_M
Run and chat with the model
lemonade run user.samai-9b-Q4_K_M
List all available models
lemonade list
- Hermes Agent
How to use tchbcb/samai-9b with Hermes Agent:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-9b:Q4_K_M
Configure Hermes
# Install Hermes: curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash hermes setup # Point Hermes at the local server: hermes config set model.provider custom hermes config set model.base_url http://127.0.0.1:8080/v1 hermes config set model.default tchbcb/samai-9b:Q4_K_M
Run Hermes
hermes
- Atomic Chat
- OpenClaw
How to use tchbcb/samai-9b with OpenClaw:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-9b:Q4_K_M
Configure OpenClaw
# Install OpenClaw: npm install -g openclaw@latest # Register the local server and set it as the default model: openclaw onboard --non-interactive --mode local \ --auth-choice custom-api-key \ --custom-base-url http://127.0.0.1:8080/v1 \ --custom-model-id "tchbcb/samai-9b:Q4_K_M" \ --custom-provider-id llama-cpp \ --custom-compatibility openai \ --custom-text-input \ --accept-risk \ --skip-health
Run OpenClaw
openclaw agent --local --agent main --message "Hello from Hugging Face"
Download artifacts/m15_scripts/colab_fetch.py from tchbcb/samai-9b: direct link, hf CLI and curl.
- Browser
- Download file 4.48 kB
-
https://huggingface.co/tchbcb/samai-9b/resolve/main/artifacts/m15_scripts/colab_fetch.py
- Command line
-
hf download hf://tchbcb/samai-9b/artifacts/m15_scripts/colab_fetch.py
-
curl -L -o colab_fetch.py https://huggingface.co/tchbcb/samai-9b/resolve/main/artifacts/m15_scripts/colab_fetch.py
4.48 kB
| #!/usr/bin/env python3 | |
| # -*- coding: utf-8 -*- | |
| """colab_fetch.py — Colab T4 弹药回拉: adapter+mmproj+底模(18G) -> 自动点火 colab_deploy.py | |
| 坑#57 对策: snapshot_download 走 cache_dir + symlink, 避免 local_dir 嵌套 | |
| 幂等旗标: /content/k8b/FETCH_* ; 日志: /content/k8b/colab_fetch.log""" | |
| import os, subprocess, time, traceback | |
| K8B = "/content/k8b" | |
| K27 = "/content/k27" | |
| BASE = K27 + "/Qwen3.5-9B" | |
| LOG = K8B + "/colab_fetch.log" | |
| os.makedirs(K8B, exist_ok=True) | |
| log = open(LOG, "a", buffering=1) | |
| def P(m): | |
| log.write("[%s] %s\n" % (time.strftime("%m-%d %H:%M:%S"), m)) | |
| def sh(c, t=1800): | |
| p = subprocess.run(c, shell=True, capture_output=True, text=True, timeout=t, errors="replace") | |
| return ((p.stdout or "") + (p.stderr or ""))[-800:] | |
| def flag(n, c="1"): | |
| open(K8B + "/" + n, "w").write(str(c)[:400]) | |
| def have(n): | |
| return os.path.exists(K8B + "/" + n) | |
| def hf_token(): | |
| return open("/root/.cache/huggingface/token").read().strip() | |
| def main(): | |
| t0 = time.time() | |
| P("==== colab fetch start ====") | |
| try: | |
| from huggingface_hub import hf_hub_download, snapshot_download, HfApi | |
| import shutil | |
| tok = hf_token() | |
| api = HfApi(token=tok) | |
| # F1: 脚本弹药 (公仓 artifacts/m15_scripts/) | |
| if not have("FETCH_SCRIPTS"): | |
| os.makedirs(K8B, exist_ok=True) | |
| for rp, dst in ( | |
| ("artifacts/m15_scripts/colab_compile.py", K8B + "/colab_compile.py"), | |
| ("artifacts/m15_scripts/colab_deploy.py", K8B + "/colab_deploy.py"), | |
| ("artifacts/m15_scripts/r12_surgery.py", K8B + "/r12_surgery.py"), | |
| ): | |
| if not os.path.exists(dst): | |
| p = hf_hub_download("tchbcb/samai-9b", rp, token=tok) | |
| shutil.copy(p, dst) | |
| P("got %s (%d)" % (dst, os.path.getsize(dst))) | |
| flag("FETCH_SCRIPTS") | |
| # F2: r10 adapter (私仓 242M) | |
| if not have("FETCH_ADP"): | |
| for f in ("adapter_config.json", "adapter_model.safetensors", | |
| "chat_template.jinja", "tokenizer.json", "tokenizer_config.json"): | |
| d = K8B + "/m12_adapters/m13_r10" | |
| os.makedirs(d, exist_ok=True) | |
| if not os.path.exists(d + "/" + f): | |
| p = hf_hub_download("tchbcb/samai-8b-M8", "m12_adapters/m13_r10/" + f, token=tok) | |
| shutil.copy(p, d + "/" + f) | |
| n = len(os.listdir(K8B + "/m12_adapters/m13_r10")) | |
| if n < 5: | |
| raise RuntimeError("adapter incomplete: %d files" % n) | |
| P("adapter ok (%d files)" % n) | |
| flag("FETCH_ADP", str(n)) | |
| # F3: mmproj (公仓 0.92G) | |
| if not have("FETCH_MM"): | |
| if not os.path.exists(K8B + "/mmproj_m11.gguf"): | |
| p = hf_hub_download("tchbcb/samai-9b", "mmproj_m11.gguf", token=tok) | |
| shutil.copy(p, K8B + "/mmproj_m11.gguf") | |
| P("mmproj ok (%.2fG)" % (os.path.getsize(K8B + "/mmproj_m11.gguf") / 1e9)) | |
| flag("FETCH_MM") | |
| # F4: 底模 18G (cache_dir 隔离 + symlink 防嵌套) | |
| if not have("FETCH_BASE"): | |
| if not os.path.exists(BASE + "/config.json"): | |
| cache = K27 + "/hf" | |
| os.makedirs(cache, exist_ok=True) | |
| P("snapshot_download base (18G)...") | |
| p = snapshot_download("Qwen/Qwen3.5-9B", cache_dir=cache, token=tok) | |
| if os.path.islink(BASE): | |
| os.remove(BASE) | |
| elif os.path.isdir(BASE): | |
| shutil.rmtree(BASE) | |
| os.symlink(p, BASE) | |
| total = sum(os.path.getsize(os.path.join(r, f)) | |
| for r, _, fs in os.walk(p) for f in fs) | |
| P("base ok %s (%.1fG)" % (p, total / 1e9)) | |
| if total < 15e9: | |
| raise RuntimeError("base too small: %.1fG" % (total / 1e9)) | |
| flag("FETCH_BASE") | |
| flag("FETCH_ALL_DONE", "%.0fs" % (time.time() - t0)) | |
| P("==== FETCH_ALL_DONE, firing deploy ====") | |
| # F5: 自动点火部署链 (分离) | |
| sh("cd %s && rm -f DEPLOY_FAIL && setsid nohup python3 colab_deploy.py > /dev/null 2>&1 & sleep 1", 20) | |
| P("deploy fired") | |
| except Exception as e: | |
| traceback.print_exc(file=log) | |
| flag("FETCH_FAIL", repr(e)[:200]) | |
| P("==== FETCH_FAIL: %s ====" % repr(e)[:200]) | |
| if __name__ == "__main__": | |
| main() | |