Instructions to use tchbcb/samai-8b-M8 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- llama.cpp
How to use tchbcb/samai-8b-M8 with llama.cpp:
Install (macOS, Linux)
curl -LsSf https://llama.app/install.sh | sh # Start a local OpenAI-compatible server with a web UI: llama serve -hf tchbcb/samai-8b-M8:Q4_K_M # Run inference directly in the terminal: llama cli -hf tchbcb/samai-8b-M8:Q4_K_M
Install from WinGet (Windows)
winget install llama.cpp # Start a local OpenAI-compatible server with a web UI: llama serve -hf tchbcb/samai-8b-M8:Q4_K_M # Run inference directly in the terminal: llama cli -hf tchbcb/samai-8b-M8:Q4_K_M
Use pre-built binary
# Download pre-built binary from: # https://github.com/ggerganov/llama.cpp/releases # Start a local OpenAI-compatible server with a web UI: ./llama-server -hf tchbcb/samai-8b-M8:Q4_K_M # Run inference directly in the terminal: ./llama-cli -hf tchbcb/samai-8b-M8:Q4_K_M
Build from source code
git clone https://github.com/ggerganov/llama.cpp.git cd llama.cpp cmake -B build cmake --build build -j --target llama-server llama-cli # Start a local OpenAI-compatible server with a web UI: ./build/bin/llama-server -hf tchbcb/samai-8b-M8:Q4_K_M # Run inference directly in the terminal: ./build/bin/llama-cli -hf tchbcb/samai-8b-M8:Q4_K_M
Use Docker
docker model run hf.co/tchbcb/samai-8b-M8:Q4_K_M
- LM Studio
- Jan
- Ollama
How to use tchbcb/samai-8b-M8 with Ollama:
ollama run hf.co/tchbcb/samai-8b-M8:Q4_K_M
- Unsloth Desktop
- Pi
How to use tchbcb/samai-8b-M8 with Pi:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-8b-M8:Q4_K_M
Configure the model in Pi
# Install Pi: npm install -g @earendil-works/pi-coding-agent # Add to ~/.pi/agent/models.json: { "providers": { "llama-cpp": { "baseUrl": "http://localhost:8080/v1", "api": "openai-completions", "apiKey": "none", "models": [ { "id": "tchbcb/samai-8b-M8:Q4_K_M" } ] } } }Run Pi
# Start Pi in your project directory: pi
- Docker Model Runner
How to use tchbcb/samai-8b-M8 with Docker Model Runner:
docker model run hf.co/tchbcb/samai-8b-M8:Q4_K_M
- Lemonade
How to use tchbcb/samai-8b-M8 with Lemonade:
Pull the model
# Download Lemonade from https://lemonade-server.ai/ lemonade pull tchbcb/samai-8b-M8:Q4_K_M
Run and chat with the model
lemonade run user.samai-8b-M8-Q4_K_M
List all available models
lemonade list
- Hermes Agent
How to use tchbcb/samai-8b-M8 with Hermes Agent:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-8b-M8:Q4_K_M
Configure Hermes
# Install Hermes: curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash hermes setup # Point Hermes at the local server: hermes config set model.provider custom hermes config set model.base_url http://127.0.0.1:8080/v1 hermes config set model.default tchbcb/samai-8b-M8:Q4_K_M
Run Hermes
hermes
- Atomic Chat
- OpenClaw
How to use tchbcb/samai-8b-M8 with OpenClaw:
Start the llama.cpp server
# Install llama.cpp: brew install llama.cpp # Start a local OpenAI-compatible server: llama serve -hf tchbcb/samai-8b-M8:Q4_K_M
Configure OpenClaw
# Install OpenClaw: npm install -g openclaw@latest # Register the local server and set it as the default model: openclaw onboard --non-interactive --mode local \ --auth-choice custom-api-key \ --custom-base-url http://127.0.0.1:8080/v1 \ --custom-model-id "tchbcb/samai-8b-M8:Q4_K_M" \ --custom-provider-id llama-cpp \ --custom-compatibility openai \ --custom-text-input \ --accept-risk \ --skip-health
Run OpenClaw
openclaw agent --local --agent main --message "Hello from Hugging Face"
Download artifacts/m15_scripts/zpatch_v3.py from tchbcb/samai-8b-M8: direct link, hf CLI and curl.
- Browser
- Download file 5.07 kB
-
https://huggingface.co/tchbcb/samai-8b-M8/resolve/main/artifacts/m15_scripts/zpatch_v3.py
- Command line
-
hf download hf://tchbcb/samai-8b-M8/artifacts/m15_scripts/zpatch_v3.py
-
curl -L -o zpatch_v3.py https://huggingface.co/tchbcb/samai-8b-M8/resolve/main/artifacts/m15_scripts/zpatch_v3.py
5.07 kB
| import os | |
| AG = "/tmp/k12/src/agent.go" | |
| WS = "/tmp/k12/src/web_stream.go" | |
| MG = "/tmp/k12/src/main.go" | |
| def rd(p): return open(p).read() | |
| def wr(p, s): open(p, "w").write(s) | |
| def enclosing(code, idx): | |
| st = code.rfind("\nfunc ", 0, idx) | |
| en = code.find("\nfunc ", idx + 10) | |
| return code[st:(en if en > 0 else len(code))] | |
| ALC = '''type antiLoopState struct { | |
| \tm map[string]int | |
| } | |
| var antiLoopMap sync.Map | |
| // antiLoopCheck returns (blocked, n). Tracks the last two distinct call | |
| // signatures per key, so both period-1 (A,A,A) and period-2 (A,B,A,B) loops | |
| // reach the block threshold. A third distinct signature resets the tracker. | |
| func antiLoopCheck(key, sig string) (bool, int) { | |
| \tany, _ := antiLoopMap.Load(key) | |
| \tif any == nil { | |
| \t\ts := &antiLoopState{m: make(map[string]int)} | |
| \t\tantiLoopMap.Store(key, s) | |
| \t\tany = s | |
| \t} | |
| \ts := any.(*antiLoopState) | |
| \tif _, seen := s.m[sig]; !seen && len(s.m) >= 2 { | |
| \t\ts.m = make(map[string]int) | |
| \t} | |
| \ts.m[sig]++ | |
| \treturn s.m[sig] >= 3, s.m[sig] | |
| } | |
| ''' | |
| def block_tmpl(ind, key): | |
| return ( | |
| ind + 'alSig, _ := json.Marshal(args)\n' | |
| + ind + 'if blocked, aln := antiLoopCheck(' + key + ', tc.Function.Name+"|"+truncate(string(alSig), 400)); blocked {\n' | |
| + ind + '\tlog.Printf("[ANTI-LOOP] blocked repeat #%d %s", aln, truncate(tc.Function.Name, 40))\n' | |
| + ind + '\tblockMsg := "HARD SYSTEM BLOCK: this exact tool call has already been executed " + fmt.Sprintf("%d", aln) + " times in a row with unchanged context. The path is DEAD. You MUST now either (a) switch to a completely different tool/source/approach, or (b) if your evidence is sufficient, output the final answer on the last line in the format: FINAL ANSWER: <answer>. Do NOT repeat this call."\n' | |
| + ind + '\thistory = append(history, LLMMessage{Role: "tool", Content: blockMsg, ToolCallID: tc.ID})\n' | |
| + ind + '\tcontinue\n' | |
| + ind + '}\n') | |
| def ws_tmpl(ind): | |
| return ( | |
| ind + '// [ANTI-LOOP FIX 2026-09-22]\n' | |
| + ind + 'alSig, _ := json.Marshal(args)\n' | |
| + ind + 'if blocked, aln := antiLoopCheck(conv.ID, tc.Function.Name+"|"+truncate(string(alSig), 400)); blocked {\n' | |
| + ind + '\tlog.Printf("[ANTI-LOOP] blocked repeat #%d %s", aln, truncate(tc.Function.Name, 40))\n' | |
| + ind + '\tblockMsg := "HARD SYSTEM BLOCK: this exact tool call has already been executed " + fmt.Sprintf("%d", aln) + " times in a row with unchanged context. The path is DEAD. You MUST now either (a) switch to a completely different tool/source/approach, or (b) if your evidence is sufficient, output the final answer on the last line in the format: FINAL ANSWER: <answer>. Do NOT repeat this call."\n' | |
| + ind + '\thistory = append(history, LLMMessage{Role: "tool", Content: blockMsg, ToolCallID: tc.ID})\n' | |
| + ind + '\tconv.Messages = append(conv.Messages, WebChatMessage{Role: "tool", ToolName: tc.Function.Name, ToolArgs: args, ToolResult: blockMsg, ToolOK: false, ToolCallID: tc.ID, Timestamp: nextConvTimestamp(conv)})\n' | |
| + ind + '\tcontinue\n' | |
| + ind + '}\n') | |
| # 1) agent.go defs | |
| code = rd(AG) | |
| assert "func antiLoopCheck" not in code, "agent.go not clean" | |
| anchor = "func (a *Agent) SyncLLMFromConfig() {" | |
| assert anchor in code | |
| code = code.replace(anchor, ALC + "\n" + anchor, 1) | |
| wr(AG, code) | |
| print("ALC_DEFS_INJECTED") | |
| # 2) agent.go sites — FRESH find each iteration | |
| code = rd(AG) | |
| needle = "result := a.toolCtx.Execute(tc.Function.Name, args)" | |
| prev = 0 | |
| for n in range(2): | |
| pos = code.find(needle, prev) | |
| assert pos > 0, "ag site %d missing" % (n + 1) | |
| prev = pos + 800 | |
| ls = code.rfind("\n", 0, pos) + 1 | |
| le = code.find("\n", pos) | |
| line = code[ls:le] | |
| ind = line[:len(line) - len(line.lstrip("\t"))] | |
| fn_scope = enclosing(code, pos) | |
| has_sl = "sourceLabel" in fn_scope | |
| key = ('"SYNC|"+sourceLabel' if n == 1 else "sourceLabel") if has_sl else ('"SYNC"' if n == 1 else '"AG"') | |
| print("AG_SITE%d ind=%d key=%s sl=%s" % (n + 1, len(ind), key, has_sl)) | |
| code = code[:ls] + block_tmpl(ind, key) + code[ls:] | |
| wr(AG, code) | |
| # 3) web_stream.go sites — FRESH find each iteration | |
| code = rd(WS) | |
| needle = "result := loopToolCtx.Execute(tc.Function.Name, args)" | |
| assert "antiLoopCheck(" not in code | |
| assert code.count("nextConvTimestamp(") >= 2 | |
| prev = 0 | |
| for n in range(2): | |
| pos = code.find(needle, prev) | |
| assert pos > 0, "ws site %d missing" % (n + 1) | |
| prev = pos + 800 | |
| ls = code.rfind("\n", 0, pos) + 1 | |
| le = code.find("\n", pos) | |
| line = code[ls:le] | |
| ind = line[:len(line) - len(line.lstrip("\t"))] | |
| print("WS_SITE%d ind=%d" % (n + 1, len(ind))) | |
| code = code[:ls] + ws_tmpl(ind) + code[ls:] | |
| wr(WS, code) | |
| print("WS_PATCH_DONE") | |
| # 4) main.go debug.Stack | |
| code = rd(MG) | |
| if "debug.Stack()" not in code: | |
| for old in ['PANIC in ws.Start(): %v", r)', 'PANIC in agent.Run(): %v", r)', 'PANIC in restarted agent.Run(): %v", r)']: | |
| code = code.replace(old, old.replace('%v", r)', '%v\\n%s", r, debug.Stack())')) | |
| if '"runtime/debug"' not in code: | |
| code = code.replace("import (", 'import (\n\t"runtime/debug"', 1) | |
| wr(MG, code) | |
| print("MAIN_STACKED") | |
| else: | |
| print("MAIN_ALREADY") | |
| print("PATCH_V2_ALL_DONE") | |