Instructions to use patdev/k3-a40-bootstrap with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- llama.cpp
How to use patdev/k3-a40-bootstrap with llama.cpp:
Install (macOS, Linux)
curl -LsSf https://llama.app/install.sh | sh # Start a local OpenAI-compatible server with a web UI: llama serve -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: llama cli -hf patdev/k3-a40-bootstrap:BF16
Install from WinGet (Windows)
winget install llama.cpp # Start a local OpenAI-compatible server with a web UI: llama serve -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: llama cli -hf patdev/k3-a40-bootstrap:BF16
Use pre-built binary
# Download pre-built binary from: # https://github.com/ggerganov/llama.cpp/releases # Start a local OpenAI-compatible server with a web UI: ./llama-server -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: ./llama-cli -hf patdev/k3-a40-bootstrap:BF16
Build from source code
git clone https://github.com/ggerganov/llama.cpp.git cd llama.cpp cmake -B build cmake --build build -j --target llama-server llama-cli # Start a local OpenAI-compatible server with a web UI: ./build/bin/llama-server -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: ./build/bin/llama-cli -hf patdev/k3-a40-bootstrap:BF16
Use Docker
docker model run hf.co/patdev/k3-a40-bootstrap:BF16
- LM Studio
- Jan
- Ollama
How to use patdev/k3-a40-bootstrap with Ollama:
ollama run hf.co/patdev/k3-a40-bootstrap:BF16
- Unsloth Desktop
- Docker Model Runner
How to use patdev/k3-a40-bootstrap with Docker Model Runner:
docker model run hf.co/patdev/k3-a40-bootstrap:BF16
- Lemonade
How to use patdev/k3-a40-bootstrap with Lemonade:
Pull the model
# Download Lemonade from https://lemonade-server.ai/ lemonade pull patdev/k3-a40-bootstrap:BF16
Run and chat with the model
lemonade run user.k3-a40-bootstrap-BF16
List all available models
lemonade list
- Atomic Chat
Download chien.py from patdev/k3-a40-bootstrap: direct link, hf CLI and curl.
- Browser
- Download file 7.43 kB
-
https://huggingface.co/patdev/k3-a40-bootstrap/resolve/main/chien.py
- Command line
-
hf download hf://patdev/k3-a40-bootstrap/chien.py
-
curl -L -o chien.py https://huggingface.co/patdev/k3-a40-bootstrap/resolve/main/chien.py
7.43 kB
| # Chien de garde du pont et du moteur. | |
| # | |
| # POURQUOI. Trois fois le 27/08, le pont a cesse de servir alors que le moteur | |
| # allait parfaitement bien. Signature invariable : | |
| # | |
| # /v1/models 200 <- servi localement, sans appel amont | |
| # /v1/messages 500 en ~10 s <- pool=10.0 de httpx | |
| # 127.0.0.1:8000 200 en 0,1 s <- LE MOTEUR VA BIEN | |
| # | |
| # La cause n'est PAS un plafond de temps : avec VL_TIMEOUT=240 la panne | |
| # revient, et au moment du blocage il y avait **zero socket** ouvert vers le | |
| # port 8000. Ce ne sont donc pas des connexions vivantes qui saturent le | |
| # bassin, ce sont des PLACES comptabilisees et jamais rendues -- des flux | |
| # abandonnes par le client dont le slot httpx n'est pas libere. La fuite est | |
| # definitive par flux, pas temporelle. | |
| # | |
| # Tant que le correctif de code n'est pas ecrit et eprouve, ce chien convertit | |
| # une panne silencieuse de plusieurs heures en une coupure de quelques secondes. | |
| # | |
| # IL NE SONDE JAMAIS /v1/models : c'est precisement l'endpoint qui ment. | |
| import json | |
| import os | |
| import subprocess | |
| import time | |
| import urllib.error | |
| import urllib.request | |
| PONT = "http://127.0.0.1:8080" | |
| MOTEUR = "http://127.0.0.1:8000" | |
| JOURNAL = "/travail/chien.log" | |
| PERIODE = 60 | |
| ECHECS_AVANT_ACTION = 2 # deux passes de suite : on ne relance pas sur un hoquet | |
| ENV_PONT = { | |
| "VL_UPSTREAM": "http://127.0.0.1:8000", | |
| "VL_SERVED_NAME": "flashnext", | |
| "VL_REAL_MODEL": "RadixArk/Qwen3.8-Flash-Next-NVFP4", | |
| "VL_MAX_OUTPUT": "32768", | |
| "VL_TIMEOUT": "240", | |
| } | |
| MOTEUR_CMD = [ | |
| "vllm", "serve", "RadixArk/Qwen3.8-Flash-Next-NVFP4", | |
| "--served-model-name", "flashnext", "--host", "127.0.0.1", "--port", "8000", | |
| "--max-model-len", "262144", "--max-num-seqs", "16", | |
| "--gpu-memory-utilization", "0.93", | |
| "--kv-cache-memory-bytes", "15569256448", # 14,5 GiB : marge pour l'encodeur d'images | |
| "--distributed-executor-backend", "mp", | |
| "--reasoning-parser", "qwen3", "--tool-call-parser", "qwen3_coder", | |
| # Le plafond a 0 rendait un 400 sur toute capture collee dans Claude Code | |
| # ("At most 0 image(s) may be provided..."). Il doit rester identique a | |
| # celui de lancer_flashnext.sh : sinon le chien "repare" en remettant les | |
| # images hors service, et la panne revient sans que rien ne l'explique. | |
| "--limit-mm-per-prompt", '{"image":4,"video":0}', | |
| "--trust-remote-code", "--enable-prefix-caching", "--enable-auto-tool-choice", | |
| "--enable-flashinfer-autotune", "--enable-prompt-tokens-details", | |
| ] | |
| def dire(m): | |
| l = "[%s] %s" % (time.strftime("%H:%M:%S"), m) | |
| print(l, flush=True) | |
| try: | |
| with open(JOURNAL, "a") as f: | |
| f.write(l + "\n") | |
| except Exception: | |
| pass | |
| def poste(url, corps, delai): | |
| """Rend (ok, detail). Une VRAIE generation, jamais /v1/models.""" | |
| try: | |
| r = urllib.request.Request( | |
| url, data=json.dumps(corps).encode(), | |
| headers={"content-type": "application/json", | |
| "anthropic-version": "2023-06-01", "x-api-key": "x"}) | |
| t = time.time() | |
| d = json.loads(urllib.request.urlopen(r, timeout=delai).read().decode()) | |
| return True, "%.2fs" % (time.time() - t) | |
| except urllib.error.HTTPError as e: | |
| return False, "HTTP %d" % e.code | |
| except Exception as e: | |
| return False, repr(e)[:90] | |
| def pont_vivant(): | |
| return poste(PONT + "/v1/messages", | |
| {"model": "flashnext", "max_tokens": 8, | |
| "thinking": {"type": "disabled"}, | |
| "messages": [{"role": "user", "content": "ping"}]}, 45) | |
| def moteur_vivant(): | |
| return poste(MOTEUR + "/v1/completions", | |
| {"model": "flashnext", "prompt": "ping", "max_tokens": 4, | |
| "temperature": 0}, 45) | |
| def tuer(motif): | |
| out = subprocess.run(["ps", "-eo", "pid,rss,comm", "--no-headers"], | |
| capture_output=True, text=True).stdout | |
| for l in out.splitlines(): | |
| p = l.split(None, 2) | |
| if len(p) != 3: | |
| continue | |
| pid, rss, comm = int(p[0]), int(p[1]), p[2].strip() | |
| import re as _re | |
| if _re.match(motif, comm) or (rss > 10_000_000 and _re.match(r"^python3?$", comm)): | |
| try: | |
| os.kill(pid, 9) | |
| except Exception: | |
| pass | |
| def relancer_pont(): | |
| """Le pont ne se relance JAMAIS depuis /proc/PID/cmdline : l'environnement | |
| n'y figure pas, et le defaut de VL_UPSTREAM est le port du pont lui-meme -- | |
| il se parlerait a lui-meme. L'env est donc explicite ici.""" | |
| p = subprocess.run(["pgrep", "-f", "uvicorn anthropic_proxy"], | |
| capture_output=True, text=True).stdout.split() | |
| for pid in p: | |
| try: | |
| os.kill(int(pid), 9) | |
| except Exception: | |
| pass | |
| time.sleep(3) | |
| env = dict(os.environ) | |
| env.update(ENV_PONT) | |
| with open("/travail/pont.log", "a") as s: | |
| subprocess.Popen(["python3", "-m", "uvicorn", "anthropic_proxy:app", | |
| "--host", "0.0.0.0", "--port", "8080", | |
| "--log-level", "warning"], | |
| cwd="/travail", stdout=s, stderr=subprocess.STDOUT, | |
| stdin=subprocess.DEVNULL, start_new_session=True, env=env) | |
| for _ in range(20): | |
| time.sleep(3) | |
| ok, d = pont_vivant() | |
| if ok: | |
| return True, d | |
| return False, "pas de generation apres 60 s" | |
| def relancer_moteur(): | |
| tuer(r"^(vllm|VLLM|Ple|EngineCore)") | |
| time.sleep(12) | |
| if os.path.exists("/travail/vllm.log"): | |
| os.replace("/travail/vllm.log", "/travail/vllm-chien-precedent.log") | |
| with open("/travail/vllm.log", "w") as s: | |
| subprocess.Popen(MOTEUR_CMD, stdout=s, stderr=subprocess.STDOUT, | |
| stdin=subprocess.DEVNULL, start_new_session=True, | |
| env=dict(os.environ, VLLM_PLE_CPU_OFFLOAD="1")) | |
| for _ in range(45): | |
| time.sleep(20) | |
| ok, d = moteur_vivant() | |
| if ok: | |
| return True, d | |
| return False, "moteur pas pret en 15 min" | |
| dire("chien de garde arme : generation reelle toutes les %d s, action apres " | |
| "%d echecs consecutifs" % (PERIODE, ECHECS_AVANT_ACTION)) | |
| echecs, reparations, tours = 0, 0, 0 | |
| while True: | |
| time.sleep(PERIODE) | |
| tours += 1 | |
| ok, detail = pont_vivant() | |
| if ok: | |
| if echecs: | |
| dire("retabli tout seul apres %d echec(s)" % echecs) | |
| echecs = 0 | |
| if tours % 15 == 0: | |
| dire("ok (%s) | %d reparation(s) depuis l'armement" % (detail, reparations)) | |
| continue | |
| echecs += 1 | |
| dire("pont KO (%s) -- echec %d/%d" % (detail, echecs, ECHECS_AVANT_ACTION)) | |
| if echecs < ECHECS_AVANT_ACTION: | |
| continue | |
| # On distingue TOUJOURS les deux couches avant d'agir : trois fois cette | |
| # nuit, le moteur allait bien et j'ai failli le tuer pour rien. | |
| mok, mdetail = moteur_vivant() | |
| if mok: | |
| dire("le moteur va bien (%s) -> c'est le pont : relance du pont seul" % mdetail) | |
| r, d = relancer_pont() | |
| dire(" pont : %s (%s)" % ("REPARE" if r else "ECHEC", d)) | |
| else: | |
| dire("le moteur aussi est KO (%s) -> relance moteur puis pont" % mdetail) | |
| r, d = relancer_moteur() | |
| dire(" moteur : %s (%s)" % ("REPARE" if r else "ECHEC", d)) | |
| r2, d2 = relancer_pont() | |
| dire(" pont : %s (%s)" % ("REPARE" if r2 else "ECHEC", d2)) | |
| reparations += 1 | |
| echecs = 0 | |