Instructions to use patdev/k3-a40-bootstrap with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- llama.cpp
How to use patdev/k3-a40-bootstrap with llama.cpp:
Install (macOS, Linux)
curl -LsSf https://llama.app/install.sh | sh # Start a local OpenAI-compatible server with a web UI: llama serve -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: llama cli -hf patdev/k3-a40-bootstrap:BF16
Install from WinGet (Windows)
winget install llama.cpp # Start a local OpenAI-compatible server with a web UI: llama serve -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: llama cli -hf patdev/k3-a40-bootstrap:BF16
Use pre-built binary
# Download pre-built binary from: # https://github.com/ggerganov/llama.cpp/releases # Start a local OpenAI-compatible server with a web UI: ./llama-server -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: ./llama-cli -hf patdev/k3-a40-bootstrap:BF16
Build from source code
git clone https://github.com/ggerganov/llama.cpp.git cd llama.cpp cmake -B build cmake --build build -j --target llama-server llama-cli # Start a local OpenAI-compatible server with a web UI: ./build/bin/llama-server -hf patdev/k3-a40-bootstrap:BF16 # Run inference directly in the terminal: ./build/bin/llama-cli -hf patdev/k3-a40-bootstrap:BF16
Use Docker
docker model run hf.co/patdev/k3-a40-bootstrap:BF16
- LM Studio
- Jan
- Ollama
How to use patdev/k3-a40-bootstrap with Ollama:
ollama run hf.co/patdev/k3-a40-bootstrap:BF16
- Unsloth Desktop
- Docker Model Runner
How to use patdev/k3-a40-bootstrap with Docker Model Runner:
docker model run hf.co/patdev/k3-a40-bootstrap:BF16
- Lemonade
How to use patdev/k3-a40-bootstrap with Lemonade:
Pull the model
# Download Lemonade from https://lemonade-server.ai/ lemonade pull patdev/k3-a40-bootstrap:BF16
Run and chat with the model
lemonade run user.k3-a40-bootstrap-BF16
List all available models
lemonade list
- Atomic Chat
banc reecrit : sonde non bloquante, partiels gardes, raisonnement coupe
Browse files- aides/banc_t4.py +96 -104
aides/banc_t4.py
CHANGED
|
@@ -4,25 +4,32 @@
|
|
| 4 |
# POURQUOI SUR T4 ET PAS SUR LA 2060. La carte de l'utilisateur fait aussi
|
| 5 |
# tourner son bureau : un OOM GPU y est une session perdue, pas un test rate --
|
| 6 |
# c'est arrive. Le T4 est la MEME generation (sm75) avec 16 Go au lieu de 6 :
|
| 7 |
-
# on
|
| 8 |
# contrainte des 6 Go sur le papier.
|
| 9 |
#
|
| 10 |
-
#
|
| 11 |
-
#
|
| 12 |
-
#
|
| 13 |
-
#
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 14 |
import json
|
| 15 |
import os
|
| 16 |
import re
|
| 17 |
import subprocess
|
| 18 |
-
import sys
|
| 19 |
import time
|
| 20 |
import urllib.request
|
| 21 |
|
| 22 |
-
BIN = "/app"
|
| 23 |
RACINE = "/tmp/modeles"
|
| 24 |
-
VRAM_2060 = 6144
|
| 25 |
-
RESERVE = 700
|
| 26 |
BUDGET = VRAM_2060 - RESERVE
|
| 27 |
|
| 28 |
MODELES = [
|
|
@@ -30,10 +37,14 @@ MODELES = [
|
|
| 30 |
("Qwen3.5-4B", "unsloth/Qwen3.5-4B-GGUF", "*Q5_K_M*"),
|
| 31 |
]
|
| 32 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
|
| 34 |
def outil(nom):
|
| 35 |
-
for p in (os.path.join(BIN, nom), os.path.join(BIN, "bin", nom)
|
| 36 |
-
if os.path.exists(p)
|
| 37 |
return p
|
| 38 |
return nom
|
| 39 |
|
|
@@ -46,22 +57,20 @@ def trouver_gguf(d):
|
|
| 46 |
return None
|
| 47 |
|
| 48 |
|
| 49 |
-
def
|
| 50 |
-
"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 51 |
|
| 52 |
-
|
| 53 |
-
|
| 54 |
-
un timeout place DANS la boucle ne se declenche jamais. C'est ce qui a fige
|
| 55 |
-
le premier passage pendant quinze minutes. On sonde le fichier a l'horloge.
|
| 56 |
-
"""
|
| 57 |
journal = "/tmp/srv-%s-%s.log" % (ctx, kv)
|
| 58 |
-
cmd = [outil("llama-server"), "-m", chemin, "-ngl", "99", "-fa", "1",
|
| 59 |
-
"-c", str(ctx), "--host", "127.0.0.1", "--port", "8081",
|
| 60 |
-
"--no-webui"]
|
| 61 |
-
if kv != "f16":
|
| 62 |
-
cmd += ["-ctk", kv, "-ctv", kv]
|
| 63 |
f = open(journal, "w")
|
| 64 |
-
p = subprocess.Popen(
|
|
|
|
| 65 |
res = {"kv": None, "poids": None, "calcul": None, "erreur": None}
|
| 66 |
t0 = time.time()
|
| 67 |
try:
|
|
@@ -71,32 +80,15 @@ def charger(chemin, ctx, kv, timeout=240):
|
|
| 71 |
txt = open(journal, errors="replace").read()
|
| 72 |
except Exception:
|
| 73 |
continue
|
| 74 |
-
|
| 75 |
-
|
| 76 |
-
|
| 77 |
-
|
| 78 |
-
if m:
|
| 79 |
-
res["poids"] = float(m.group(1))
|
| 80 |
-
m = re.search(r"CUDA0 compute buffer size\s*=\s*([\d.]+) MiB", txt)
|
| 81 |
-
if m:
|
| 82 |
-
res["calcul"] = float(m.group(1))
|
| 83 |
bas = txt.lower()
|
| 84 |
if "out of memory" in bas or "failed to allocate" in bas:
|
| 85 |
res["erreur"] = "OOM"
|
| 86 |
break
|
| 87 |
-
# Le tampon de calcul n'est pas toujours annonce : exiger les TROIS
|
| 88 |
-
# faisait expirer la sonde alors que KV et poids etaient deja lus,
|
| 89 |
-
# et le script jetait ensuite les valeurs partielles. Deux jobs
|
| 90 |
-
# perdus pour ca.
|
| 91 |
if res["kv"] is not None and res["poids"] is not None:
|
| 92 |
-
time.sleep(3)
|
| 93 |
-
try:
|
| 94 |
-
txt2 = open(journal, errors="replace").read()
|
| 95 |
-
m2 = re.search(r"CUDA0 compute buffer size\s*=\s*([\d.]+) MiB", txt2)
|
| 96 |
-
if m2:
|
| 97 |
-
res["calcul"] = float(m2.group(1))
|
| 98 |
-
except Exception:
|
| 99 |
-
pass
|
| 100 |
break
|
| 101 |
if p.poll() is not None:
|
| 102 |
if res["kv"] is None:
|
|
@@ -104,31 +96,27 @@ def charger(chemin, ctx, kv, timeout=240):
|
|
| 104 |
break
|
| 105 |
else:
|
| 106 |
if res["kv"] is None:
|
| 107 |
-
|
| 108 |
try:
|
| 109 |
-
|
| 110 |
-
|
| 111 |
except Exception:
|
| 112 |
pass
|
| 113 |
-
res["erreur"] = "timeout %ds ; fin de log : %s" % (timeout,
|
| 114 |
finally:
|
| 115 |
try:
|
| 116 |
-
p.terminate()
|
|
|
|
| 117 |
except Exception:
|
| 118 |
p.kill()
|
| 119 |
f.close()
|
| 120 |
-
if res["erreur"] is None and res["kv"] is None:
|
| 121 |
-
res["erreur"] = "tampons non annonces"
|
| 122 |
return res
|
| 123 |
|
| 124 |
|
| 125 |
def serveur(chemin, ctx, kv):
|
| 126 |
-
|
| 127 |
-
|
| 128 |
-
|
| 129 |
-
cmd += ["-ctk", kv, "-ctv", kv]
|
| 130 |
-
p = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
| 131 |
-
for _ in range(120):
|
| 132 |
time.sleep(2)
|
| 133 |
try:
|
| 134 |
urllib.request.urlopen("http://127.0.0.1:8081/health", timeout=5)
|
|
@@ -140,42 +128,35 @@ def serveur(chemin, ctx, kv):
|
|
| 140 |
|
| 141 |
|
| 142 |
def demander(prompt, maxtok=640, temp=0.2):
|
| 143 |
-
|
| 144 |
-
|
| 145 |
-
|
| 146 |
-
|
| 147 |
-
|
| 148 |
-
conclure "radotage" -- alors que la sonde ne mesurait rien du tout.
|
| 149 |
-
"""
|
| 150 |
-
corps = json.dumps({"messages": [{"role": "user", "content": prompt}],
|
| 151 |
-
"max_tokens": maxtok, "temperature": temp,
|
| 152 |
-
"stream": False,
|
| 153 |
-
"chat_template_kwargs": {"enable_thinking": False}}).encode()
|
| 154 |
r = urllib.request.Request("http://127.0.0.1:8081/v1/chat/completions",
|
| 155 |
data=corps,
|
| 156 |
headers={"content-type": "application/json"})
|
| 157 |
t = time.time()
|
| 158 |
d = json.loads(urllib.request.urlopen(r, timeout=300).read())
|
| 159 |
-
|
|
|
|
| 160 |
texte = (msg.get("content") or "").strip()
|
| 161 |
raison = (msg.get("reasoning_content") or "").strip()
|
| 162 |
if not texte and raison:
|
| 163 |
texte = "[raisonnement seul] " + raison
|
| 164 |
-
fin = d["choices"][0].get("finish_reason")
|
| 165 |
u = d.get("usage") or {}
|
| 166 |
-
return texte, time.time() - t,
|
|
|
|
| 167 |
|
| 168 |
|
| 169 |
def distinct4(texte):
|
| 170 |
-
"""
|
| 171 |
-
|
| 172 |
-
|
| 173 |
-
|
| 174 |
-
lui, etait exact. On ne publie pas un debit sans ce ratio.
|
| 175 |
-
"""
|
| 176 |
mots = re.findall(r"\w+", texte.lower())
|
| 177 |
if len(mots) < 8:
|
| 178 |
-
return
|
| 179 |
g = [tuple(mots[i:i + 4]) for i in range(len(mots) - 3)]
|
| 180 |
return len(set(g)) / len(g)
|
| 181 |
|
|
@@ -185,20 +166,21 @@ EPREUVES = [
|
|
| 185 |
"Calcule 17 * 23 + 145. Reponds uniquement par le nombre.",
|
| 186 |
lambda r: "536" in r),
|
| 187 |
("code-python",
|
| 188 |
-
"Ecris une fonction Python
|
| 189 |
"premier. Code seulement, pas d'explication.",
|
| 190 |
-
lambda r: "
|
| 191 |
("lecture-erreur",
|
| 192 |
-
"Voici une erreur :
|
| 193 |
-
"'int' and 'str'
|
| 194 |
-
"
|
| 195 |
-
lambda r: any(k in r.lower() for k in
|
|
|
|
| 196 |
("instruction-stricte",
|
| 197 |
"Reponds exactement par le mot MANGUE, rien d'autre.",
|
| 198 |
lambda r: r.strip().strip(".").upper().startswith("MANGUE")),
|
| 199 |
("json",
|
| 200 |
'Donne un objet JSON avec les cles "nom" (chaine) et "age" (entier) pour '
|
| 201 |
-
'une personne nommee Ada
|
| 202 |
lambda r: "ada" in r.lower() and "36" in r),
|
| 203 |
]
|
| 204 |
|
|
@@ -210,7 +192,7 @@ def main():
|
|
| 210 |
"--format=csv,noheader"])
|
| 211 |
print("budget equivalent RTX 2060 : %d Mio (6144 - %d de reserve)"
|
| 212 |
% (BUDGET, RESERVE))
|
| 213 |
-
print("=" * 74)
|
| 214 |
|
| 215 |
for nom, repo, motif in MODELES:
|
| 216 |
d = os.path.join(RACINE, nom)
|
|
@@ -222,14 +204,13 @@ def main():
|
|
| 222 |
taille = os.path.getsize(chemin) / (1024 * 1024)
|
| 223 |
print("\n" + "=" * 74)
|
| 224 |
print("### %s (%.0f Mio sur disque)" % (nom, taille))
|
| 225 |
-
print("=" * 74)
|
| 226 |
|
| 227 |
-
# --- capacite : une mesure par dtype de cache, le reste se deduit
|
| 228 |
CTX_SONDE = 8192
|
| 229 |
for kv in ("f16", "q8_0"):
|
| 230 |
r = charger(chemin, CTX_SONDE, kv)
|
| 231 |
-
if r["
|
| 232 |
-
print(" KV %-5s : ECHEC %s" % (kv, r["erreur"]))
|
| 233 |
continue
|
| 234 |
par_jeton = r["kv"] * 1024 * 1024 / CTX_SONDE
|
| 235 |
fixe = (r["poids"] or 0) + (r["calcul"] or 0)
|
|
@@ -237,15 +218,16 @@ def main():
|
|
| 237 |
ctx_max = int(dispo * 1024 * 1024 / par_jeton) if dispo > 0 else 0
|
| 238 |
print(" KV %-5s : %6.0f o/jeton | poids %6.0f + calcul %5.0f = %6.0f Mio fixes"
|
| 239 |
% (kv, par_jeton, r["poids"] or 0, r["calcul"] or 0, fixe))
|
| 240 |
-
print(" ->
|
| 241 |
-
% (dispo, "{:,}".format(max(ctx_max, 0)).replace(",", " "))
|
|
|
|
| 242 |
|
| 243 |
-
|
| 244 |
-
|
| 245 |
try:
|
| 246 |
p = serveur(chemin, 8192, "q8_0")
|
| 247 |
except Exception as e:
|
| 248 |
-
print(" serveur KO :", repr(e)[:100])
|
| 249 |
continue
|
| 250 |
try:
|
| 251 |
score = 0
|
|
@@ -253,24 +235,34 @@ def main():
|
|
| 253 |
textes = []
|
| 254 |
for etiquette, prompt, verif in EPREUVES:
|
| 255 |
try:
|
| 256 |
-
rep, dt = demander(prompt)
|
| 257 |
except Exception as e:
|
| 258 |
-
print(" %-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 259 |
continue
|
| 260 |
textes.append(rep)
|
| 261 |
ok = bool(verif(rep))
|
| 262 |
score += ok
|
| 263 |
-
court = rep.
|
| 264 |
-
print(" %-
|
| 265 |
-
|
| 266 |
print(" ---")
|
| 267 |
juges = len(EPREUVES) - vides
|
|
|
|
| 268 |
if juges == 0:
|
| 269 |
print(" AUCUNE REPONSE EXPLOITABLE -- sonde cassee, pas de verdict")
|
| 270 |
else:
|
| 271 |
-
|
| 272 |
-
|
| 273 |
-
|
|
|
|
|
|
|
|
|
|
| 274 |
finally:
|
| 275 |
p.kill()
|
| 276 |
|
|
|
|
| 4 |
# POURQUOI SUR T4 ET PAS SUR LA 2060. La carte de l'utilisateur fait aussi
|
| 5 |
# tourner son bureau : un OOM GPU y est une session perdue, pas un test rate --
|
| 6 |
# c'est arrive. Le T4 est la MEME generation (sm75) avec 16 Go au lieu de 6 :
|
| 7 |
+
# on y sonde sans risque, on mesure l'allocation REELLE, puis on applique la
|
| 8 |
# contrainte des 6 Go sur le papier.
|
| 9 |
#
|
| 10 |
+
# TROIS DEFAUTS DU PREMIER PASSAGE, TOUS DANS LA SONDE ET PAS DANS LE MODELE :
|
| 11 |
+
# 1. `for ligne in p.stdout` bufferise et bloque dans readline, donc le
|
| 12 |
+
# timeout place DANS la boucle ne s'executait jamais -> 15 min de T4 fige.
|
| 13 |
+
# On ecrit desormais dans un FICHIER qu'on relit a l'horloge.
|
| 14 |
+
# 2. On exigeait les TROIS tampons pour conclure ; quand le tampon de calcul
|
| 15 |
+
# n'etait pas annonce, la sonde expirait et JETAIT le KV et les poids
|
| 16 |
+
# deja lus. On garde les partiels, et on affiche la fin du log en cas
|
| 17 |
+
# d'echec pour diagnostiquer sans relancer un job.
|
| 18 |
+
# 3. Ces modeles RAISONNENT. Sans couper le raisonnement, le budget partait
|
| 19 |
+
# dans <think> et `content` revenait VIDE ; le ratio de 4-grammes affichait
|
| 20 |
+
# alors 0,000 et j'ai failli conclure "radotage". Une reponse vide est une
|
| 21 |
+
# mesure ABSENTE, pas une mauvaise reponse -- elle ne se compte pas.
|
| 22 |
import json
|
| 23 |
import os
|
| 24 |
import re
|
| 25 |
import subprocess
|
|
|
|
| 26 |
import time
|
| 27 |
import urllib.request
|
| 28 |
|
| 29 |
+
BIN = "/app"
|
| 30 |
RACINE = "/tmp/modeles"
|
| 31 |
+
VRAM_2060 = 6144
|
| 32 |
+
RESERVE = 700
|
| 33 |
BUDGET = VRAM_2060 - RESERVE
|
| 34 |
|
| 35 |
MODELES = [
|
|
|
|
| 37 |
("Qwen3.5-4B", "unsloth/Qwen3.5-4B-GGUF", "*Q5_K_M*"),
|
| 38 |
]
|
| 39 |
|
| 40 |
+
RE_KV = re.compile(r"KV (?:self )?(?:buffer )?size\s*=\s*([\d.]+) MiB")
|
| 41 |
+
RE_POIDS = re.compile(r"CUDA0 model buffer size\s*=\s*([\d.]+) MiB")
|
| 42 |
+
RE_CALC = re.compile(r"CUDA0 compute buffer size\s*=\s*([\d.]+) MiB")
|
| 43 |
+
|
| 44 |
|
| 45 |
def outil(nom):
|
| 46 |
+
for p in (os.path.join(BIN, nom), os.path.join(BIN, "bin", nom)):
|
| 47 |
+
if os.path.exists(p):
|
| 48 |
return p
|
| 49 |
return nom
|
| 50 |
|
|
|
|
| 57 |
return None
|
| 58 |
|
| 59 |
|
| 60 |
+
def cmd_serveur(chemin, ctx, kv):
|
| 61 |
+
c = [outil("llama-server"), "-m", chemin, "-ngl", "99", "-fa", "1",
|
| 62 |
+
"-c", str(ctx), "--host", "127.0.0.1", "--port", "8081", "--no-webui"]
|
| 63 |
+
if kv != "f16":
|
| 64 |
+
c += ["-ctk", kv, "-ctv", kv]
|
| 65 |
+
return c
|
| 66 |
+
|
| 67 |
|
| 68 |
+
def charger(chemin, ctx, kv, timeout=240):
|
| 69 |
+
"""Rend les tailles de tampons annoncees au chargement, en Mio."""
|
|
|
|
|
|
|
|
|
|
| 70 |
journal = "/tmp/srv-%s-%s.log" % (ctx, kv)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 71 |
f = open(journal, "w")
|
| 72 |
+
p = subprocess.Popen(cmd_serveur(chemin, ctx, kv), stdout=f,
|
| 73 |
+
stderr=subprocess.STDOUT)
|
| 74 |
res = {"kv": None, "poids": None, "calcul": None, "erreur": None}
|
| 75 |
t0 = time.time()
|
| 76 |
try:
|
|
|
|
| 80 |
txt = open(journal, errors="replace").read()
|
| 81 |
except Exception:
|
| 82 |
continue
|
| 83 |
+
for cle, rx in (("kv", RE_KV), ("poids", RE_POIDS), ("calcul", RE_CALC)):
|
| 84 |
+
m = rx.search(txt)
|
| 85 |
+
if m:
|
| 86 |
+
res[cle] = float(m.group(1))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 87 |
bas = txt.lower()
|
| 88 |
if "out of memory" in bas or "failed to allocate" in bas:
|
| 89 |
res["erreur"] = "OOM"
|
| 90 |
break
|
|
|
|
|
|
|
|
|
|
|
|
|
| 91 |
if res["kv"] is not None and res["poids"] is not None:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 92 |
break
|
| 93 |
if p.poll() is not None:
|
| 94 |
if res["kv"] is None:
|
|
|
|
| 96 |
break
|
| 97 |
else:
|
| 98 |
if res["kv"] is None:
|
| 99 |
+
bout = ""
|
| 100 |
try:
|
| 101 |
+
brut = open(journal, errors="replace").read()[-400:]
|
| 102 |
+
bout = " | ".join(brut.splitlines()[-4:])
|
| 103 |
except Exception:
|
| 104 |
pass
|
| 105 |
+
res["erreur"] = "timeout %ds ; fin de log : %s" % (timeout, bout)
|
| 106 |
finally:
|
| 107 |
try:
|
| 108 |
+
p.terminate()
|
| 109 |
+
p.wait(timeout=20)
|
| 110 |
except Exception:
|
| 111 |
p.kill()
|
| 112 |
f.close()
|
|
|
|
|
|
|
| 113 |
return res
|
| 114 |
|
| 115 |
|
| 116 |
def serveur(chemin, ctx, kv):
|
| 117 |
+
p = subprocess.Popen(cmd_serveur(chemin, ctx, kv),
|
| 118 |
+
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
| 119 |
+
for _ in range(150):
|
|
|
|
|
|
|
|
|
|
| 120 |
time.sleep(2)
|
| 121 |
try:
|
| 122 |
urllib.request.urlopen("http://127.0.0.1:8081/health", timeout=5)
|
|
|
|
| 128 |
|
| 129 |
|
| 130 |
def demander(prompt, maxtok=640, temp=0.2):
|
| 131 |
+
corps = json.dumps({
|
| 132 |
+
"messages": [{"role": "user", "content": prompt}],
|
| 133 |
+
"max_tokens": maxtok, "temperature": temp, "stream": False,
|
| 134 |
+
"chat_template_kwargs": {"enable_thinking": False},
|
| 135 |
+
}).encode()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 136 |
r = urllib.request.Request("http://127.0.0.1:8081/v1/chat/completions",
|
| 137 |
data=corps,
|
| 138 |
headers={"content-type": "application/json"})
|
| 139 |
t = time.time()
|
| 140 |
d = json.loads(urllib.request.urlopen(r, timeout=300).read())
|
| 141 |
+
ch = d["choices"][0]
|
| 142 |
+
msg = ch["message"]
|
| 143 |
texte = (msg.get("content") or "").strip()
|
| 144 |
raison = (msg.get("reasoning_content") or "").strip()
|
| 145 |
if not texte and raison:
|
| 146 |
texte = "[raisonnement seul] " + raison
|
|
|
|
| 147 |
u = d.get("usage") or {}
|
| 148 |
+
return (texte, time.time() - t, ch.get("finish_reason"),
|
| 149 |
+
u.get("completion_tokens", 0))
|
| 150 |
|
| 151 |
|
| 152 |
def distinct4(texte):
|
| 153 |
+
"""Garde-fou contre la degenerescence : un depot de chez nous affichait un
|
| 154 |
+
debit honorable avec ce ratio a 0,000, du pur radotage. Rend None quand il
|
| 155 |
+
n'y a pas assez de mots pour conclure -- ne jamais confondre 'pas de
|
| 156 |
+
mesure' et 'mauvaise mesure'."""
|
|
|
|
|
|
|
| 157 |
mots = re.findall(r"\w+", texte.lower())
|
| 158 |
if len(mots) < 8:
|
| 159 |
+
return None
|
| 160 |
g = [tuple(mots[i:i + 4]) for i in range(len(mots) - 3)]
|
| 161 |
return len(set(g)) / len(g)
|
| 162 |
|
|
|
|
| 166 |
"Calcule 17 * 23 + 145. Reponds uniquement par le nombre.",
|
| 167 |
lambda r: "536" in r),
|
| 168 |
("code-python",
|
| 169 |
+
"Ecris une fonction Python est_premier(n) qui renvoie True si n est "
|
| 170 |
"premier. Code seulement, pas d'explication.",
|
| 171 |
+
lambda r: "est_premier" in r and "def" in r),
|
| 172 |
("lecture-erreur",
|
| 173 |
+
"Voici une erreur : TypeError: unsupported operand type(s) for +: "
|
| 174 |
+
"'int' and 'str', ligne 12 de app.py. En une phrase : la cause et le "
|
| 175 |
+
"correctif ?",
|
| 176 |
+
lambda r: any(k in r.lower() for k in
|
| 177 |
+
("str(", "int(", "conver", "cast", "chaine", "type"))),
|
| 178 |
("instruction-stricte",
|
| 179 |
"Reponds exactement par le mot MANGUE, rien d'autre.",
|
| 180 |
lambda r: r.strip().strip(".").upper().startswith("MANGUE")),
|
| 181 |
("json",
|
| 182 |
'Donne un objet JSON avec les cles "nom" (chaine) et "age" (entier) pour '
|
| 183 |
+
'une personne nommee Ada, 36 ans. JSON seul.',
|
| 184 |
lambda r: "ada" in r.lower() and "36" in r),
|
| 185 |
]
|
| 186 |
|
|
|
|
| 192 |
"--format=csv,noheader"])
|
| 193 |
print("budget equivalent RTX 2060 : %d Mio (6144 - %d de reserve)"
|
| 194 |
% (BUDGET, RESERVE))
|
| 195 |
+
print("=" * 74, flush=True)
|
| 196 |
|
| 197 |
for nom, repo, motif in MODELES:
|
| 198 |
d = os.path.join(RACINE, nom)
|
|
|
|
| 204 |
taille = os.path.getsize(chemin) / (1024 * 1024)
|
| 205 |
print("\n" + "=" * 74)
|
| 206 |
print("### %s (%.0f Mio sur disque)" % (nom, taille))
|
| 207 |
+
print("=" * 74, flush=True)
|
| 208 |
|
|
|
|
| 209 |
CTX_SONDE = 8192
|
| 210 |
for kv in ("f16", "q8_0"):
|
| 211 |
r = charger(chemin, CTX_SONDE, kv)
|
| 212 |
+
if r["kv"] is None:
|
| 213 |
+
print(" KV %-5s : ECHEC %s" % (kv, r["erreur"]), flush=True)
|
| 214 |
continue
|
| 215 |
par_jeton = r["kv"] * 1024 * 1024 / CTX_SONDE
|
| 216 |
fixe = (r["poids"] or 0) + (r["calcul"] or 0)
|
|
|
|
| 218 |
ctx_max = int(dispo * 1024 * 1024 / par_jeton) if dispo > 0 else 0
|
| 219 |
print(" KV %-5s : %6.0f o/jeton | poids %6.0f + calcul %5.0f = %6.0f Mio fixes"
|
| 220 |
% (kv, par_jeton, r["poids"] or 0, r["calcul"] or 0, fixe))
|
| 221 |
+
print(" -> il reste %6.0f Mio sur 6 Go => CONTEXTE MAX %s jetons"
|
| 222 |
+
% (dispo, "{:,}".format(max(ctx_max, 0)).replace(",", " ")),
|
| 223 |
+
flush=True)
|
| 224 |
|
| 225 |
+
print("\n --- qualite (contexte 8192, KV q8_0, raisonnement coupe) ---",
|
| 226 |
+
flush=True)
|
| 227 |
try:
|
| 228 |
p = serveur(chemin, 8192, "q8_0")
|
| 229 |
except Exception as e:
|
| 230 |
+
print(" serveur KO :", repr(e)[:100], flush=True)
|
| 231 |
continue
|
| 232 |
try:
|
| 233 |
score = 0
|
|
|
|
| 235 |
textes = []
|
| 236 |
for etiquette, prompt, verif in EPREUVES:
|
| 237 |
try:
|
| 238 |
+
rep, dt, fin, ntok = demander(prompt)
|
| 239 |
except Exception as e:
|
| 240 |
+
print(" %-19s ERREUR %s" % (etiquette, repr(e)[:60]))
|
| 241 |
+
vides += 1
|
| 242 |
+
continue
|
| 243 |
+
if not rep:
|
| 244 |
+
print(" %-19s VIDE %5.1fs (fin=%s, %d jetons emis)"
|
| 245 |
+
% (etiquette, dt, fin, ntok))
|
| 246 |
+
vides += 1
|
| 247 |
continue
|
| 248 |
textes.append(rep)
|
| 249 |
ok = bool(verif(rep))
|
| 250 |
score += ok
|
| 251 |
+
court = rep.replace("\n", " ")[:66]
|
| 252 |
+
print(" %-19s %-4s %5.1fs %s"
|
| 253 |
+
% (etiquette, "OK" if ok else "faux", dt, court))
|
| 254 |
print(" ---")
|
| 255 |
juges = len(EPREUVES) - vides
|
| 256 |
+
d4 = distinct4(" ".join(textes))
|
| 257 |
if juges == 0:
|
| 258 |
print(" AUCUNE REPONSE EXPLOITABLE -- sonde cassee, pas de verdict")
|
| 259 |
else:
|
| 260 |
+
etat = "(?)" if d4 is None else (
|
| 261 |
+
"(sain)" if d4 > 0.8 else "(!! RADOTAGE)" if d4 < 0.5
|
| 262 |
+
else "(a surveiller)")
|
| 263 |
+
print(" SCORE %d/%d juges (%d vides, non comptes) 4-grammes %s %s"
|
| 264 |
+
% (score, juges, vides,
|
| 265 |
+
"n/a" if d4 is None else "%.3f" % d4, etat))
|
| 266 |
finally:
|
| 267 |
p.kill()
|
| 268 |
|