Instructions to use patdev/k3-a40-bootstrap with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- llama.cpp
How to use patdev/k3-a40-bootstrap with llama.cpp:
Install (macOS, Linux)
curl -LsSf https://llama.app/install.sh | sh # Start a local OpenAI-compatible server with a web UI: llama serve -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: llama cli -hf patdev/k3-a40-bootstrap:BF16
Install from WinGet (Windows)
winget install llama.cpp # Start a local OpenAI-compatible server with a web UI: llama serve -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: llama cli -hf patdev/k3-a40-bootstrap:BF16
Use pre-built binary
# Download pre-built binary from: # https://github.com/ggerganov/llama.cpp/releases # Start a local OpenAI-compatible server with a web UI: ./llama-server -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: ./llama-cli -hf patdev/k3-a40-bootstrap:BF16
Build from source code
git clone https://github.com/ggerganov/llama.cpp.git cd llama.cpp cmake -B build cmake --build build -j --target llama-server llama-cli # Start a local OpenAI-compatible server with a web UI: ./build/bin/llama-server -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: ./build/bin/llama-cli -hf patdev/k3-a40-bootstrap:BF16
Use Docker
docker model run hf.co/patdev/k3-a40-bootstrap:BF16
- LM Studio
- Jan
- Ollama
How to use patdev/k3-a40-bootstrap with Ollama:
ollama run hf.co/patdev/k3-a40-bootstrap:BF16
- Unsloth Desktop
- Docker Model Runner
How to use patdev/k3-a40-bootstrap with Docker Model Runner:
docker model run hf.co/patdev/k3-a40-bootstrap:BF16
- Lemonade
How to use patdev/k3-a40-bootstrap with Lemonade:
Pull the model
# Download Lemonade from https://lemonade-server.ai/ lemonade pull patdev/k3-a40-bootstrap:BF16
Run and chat with the model
lemonade run user.k3-a40-bootstrap-BF16
List all available models
lemonade list
- Atomic Chat
pont v72 : la speculation voyage dans le nom (ornith+dspark)
Browse files- anthropic_proxy.py +26 -13
anthropic_proxy.py
CHANGED
|
@@ -36,23 +36,28 @@ SWAP = dict(x.split(":", 1) for x in os.environ.get("VL_SWAP", "").split(",") if
|
|
| 36 |
SWAP_TIMEOUT = float(os.environ.get("VL_SWAP_TIMEOUT", "900"))
|
| 37 |
|
| 38 |
|
| 39 |
-
def _courant() -> tuple[str, str, str]:
|
| 40 |
-
"""(cle bootstrap, nom servi, depot) du modele charge."""
|
| 41 |
try:
|
| 42 |
with open(os.path.join(SWAP_DIR, "modele_courant"), encoding="utf-8") as f:
|
| 43 |
-
|
| 44 |
-
|
|
|
|
| 45 |
except Exception:
|
| 46 |
-
return os.environ.get("VL_MODEL_KEY", ""), MODEL, os.environ.get("VL_REAL_MODEL", MODEL)
|
| 47 |
|
| 48 |
|
| 49 |
-
def _cle_demandee(model: object) -> str | None:
|
|
|
|
|
|
|
| 50 |
if not isinstance(model, str):
|
| 51 |
return None
|
| 52 |
nom = model.removesuffix("[1m]")
|
| 53 |
if nom.startswith("claude-"):
|
| 54 |
nom = nom[len("claude-"):]
|
| 55 |
-
|
|
|
|
|
|
|
| 56 |
TIMEOUT = float(os.environ.get("VL_TIMEOUT", "1800"))
|
| 57 |
# Sortie maximale du modele. Claude Code demande couramment 64 k, ce que
|
| 58 |
# vLLM refuse d'un 400 portant sur max_tokens -- la requete entiere echoue
|
|
@@ -92,7 +97,7 @@ ALIASES = [
|
|
| 92 |
# est sa convention pour la fenetre du million ; VERIFIE : avec lui,
|
| 93 |
# l'avertissement "auto-compact will keep this session within 200k" disparait.
|
| 94 |
for _n in SWAP: # les modeles interchangeables, tous annonces
|
| 95 |
-
ALIASES += [_n, f"claude-{_n}"]
|
| 96 |
ALIASES = [a for pair in ((x, f"{x}[1m]") for x in ALIASES) for a in pair]
|
| 97 |
ALIASES = list(dict.fromkeys(ALIASES)) # dedoublonne en gardant l'ordre
|
| 98 |
|
|
@@ -103,21 +108,29 @@ _verrou_swap = asyncio.Lock()
|
|
| 103 |
|
| 104 |
async def _assurer_modele(model: object) -> str:
|
| 105 |
"""Rend le nom servi pour `model`, en declenchant le swap s'il le faut."""
|
| 106 |
-
|
| 107 |
courant = _courant()
|
| 108 |
-
if not
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 109 |
return courant[1]
|
| 110 |
async with _verrou_swap:
|
| 111 |
courant = _courant()
|
| 112 |
-
if
|
| 113 |
return courant[1]
|
| 114 |
with open(os.path.join(SWAP_DIR, "modele_demande"), "w", encoding="utf-8") as f:
|
| 115 |
-
f.write(cle
|
| 116 |
t0 = time.time()
|
| 117 |
while time.time() - t0 < SWAP_TIMEOUT:
|
| 118 |
await asyncio.sleep(3)
|
| 119 |
c = _courant()
|
| 120 |
-
if
|
| 121 |
continue
|
| 122 |
try:
|
| 123 |
r = await _client.get("/health", timeout=3.0)
|
|
|
|
| 36 |
SWAP_TIMEOUT = float(os.environ.get("VL_SWAP_TIMEOUT", "900"))
|
| 37 |
|
| 38 |
|
| 39 |
+
def _courant() -> tuple[str, str, str, str]:
|
| 40 |
+
"""(cle bootstrap, nom servi, depot, speculation) du modele charge."""
|
| 41 |
try:
|
| 42 |
with open(os.path.join(SWAP_DIR, "modele_courant"), encoding="utf-8") as f:
|
| 43 |
+
champs = f.read().split()
|
| 44 |
+
cle, nom, depot = champs[:3]
|
| 45 |
+
return cle, nom, depot, (champs[3] if len(champs) > 3 else "off")
|
| 46 |
except Exception:
|
| 47 |
+
return os.environ.get("VL_MODEL_KEY", ""), MODEL, os.environ.get("VL_REAL_MODEL", MODEL), "off"
|
| 48 |
|
| 49 |
|
| 50 |
+
def _cle_demandee(model: object) -> tuple[str, str] | None:
|
| 51 |
+
"""`ornith` -> (cle, "") ; `ornith+dspark` -> (cle, "dspark") : la speculation
|
| 52 |
+
voyage dans le nom du modele, pour changer de reglage sans recreer le pod."""
|
| 53 |
if not isinstance(model, str):
|
| 54 |
return None
|
| 55 |
nom = model.removesuffix("[1m]")
|
| 56 |
if nom.startswith("claude-"):
|
| 57 |
nom = nom[len("claude-"):]
|
| 58 |
+
nom, _, spec = nom.partition("+")
|
| 59 |
+
cle = SWAP.get(nom)
|
| 60 |
+
return (cle, spec) if cle else None
|
| 61 |
TIMEOUT = float(os.environ.get("VL_TIMEOUT", "1800"))
|
| 62 |
# Sortie maximale du modele. Claude Code demande couramment 64 k, ce que
|
| 63 |
# vLLM refuse d'un 400 portant sur max_tokens -- la requete entiere echoue
|
|
|
|
| 97 |
# est sa convention pour la fenetre du million ; VERIFIE : avec lui,
|
| 98 |
# l'avertissement "auto-compact will keep this session within 200k" disparait.
|
| 99 |
for _n in SWAP: # les modeles interchangeables, tous annonces
|
| 100 |
+
ALIASES += [_n, f"claude-{_n}", f"{_n}+dspark", f"{_n}+off"]
|
| 101 |
ALIASES = [a for pair in ((x, f"{x}[1m]") for x in ALIASES) for a in pair]
|
| 102 |
ALIASES = list(dict.fromkeys(ALIASES)) # dedoublonne en gardant l'ordre
|
| 103 |
|
|
|
|
| 108 |
|
| 109 |
async def _assurer_modele(model: object) -> str:
|
| 110 |
"""Rend le nom servi pour `model`, en declenchant le swap s'il le faut."""
|
| 111 |
+
dem = _cle_demandee(model)
|
| 112 |
courant = _courant()
|
| 113 |
+
if not dem:
|
| 114 |
+
return courant[1]
|
| 115 |
+
cle, spec = dem
|
| 116 |
+
|
| 117 |
+
def satisfait(c):
|
| 118 |
+
# sans "+spec" dans le nom, n'importe quelle speculation du bon modele convient
|
| 119 |
+
return c[0] == cle and (not spec or c[3] == spec)
|
| 120 |
+
|
| 121 |
+
if satisfait(courant):
|
| 122 |
return courant[1]
|
| 123 |
async with _verrou_swap:
|
| 124 |
courant = _courant()
|
| 125 |
+
if satisfait(courant):
|
| 126 |
return courant[1]
|
| 127 |
with open(os.path.join(SWAP_DIR, "modele_demande"), "w", encoding="utf-8") as f:
|
| 128 |
+
f.write(f"{cle} {spec}\n")
|
| 129 |
t0 = time.time()
|
| 130 |
while time.time() - t0 < SWAP_TIMEOUT:
|
| 131 |
await asyncio.sleep(3)
|
| 132 |
c = _courant()
|
| 133 |
+
if not satisfait(c):
|
| 134 |
continue
|
| 135 |
try:
|
| 136 |
r = await _client.get("/health", timeout=3.0)
|