Spaces:
Running on Zero
Running on Zero
Prevent llama output pipe deadlock
Browse files
app.py
CHANGED
|
@@ -265,12 +265,14 @@ def _run_bench_locked(model_filter=None):
|
|
| 265 |
to = 90 if mi["size_gb"] <= 6 else 180
|
| 266 |
command = [bench, "-m", mp, "-t", str(nt), "-ngl", "0", "-c", "256", "-p", "32", "-n", "64", "-r", "1", "--no-mmap", "-o", "json"]
|
| 267 |
_log(f"Running llama-bench for {mi['name']} on CPU with {nt} threads")
|
| 268 |
-
|
|
|
|
|
|
|
|
|
|
| 269 |
wt = time.time()-t0
|
| 270 |
diagnostics = (p.stderr or "").strip()
|
| 271 |
if p.returncode != 0:
|
| 272 |
raise RuntimeError(f"llama-bench exited with code {p.returncode}: {diagnostics[-1500:]}")
|
| 273 |
-
raw = (p.stdout or "").strip()
|
| 274 |
parsed = json.loads(raw)
|
| 275 |
row = parsed[0] if isinstance(parsed, list) and parsed else parsed
|
| 276 |
r.update({"status":"success", "total_time_sec":round(wt,2), "prompt_tokens":32, "completion_tokens":64, "prompt_tokens_per_sec":row.get("avg_ts", 0), "generation_tokens_per_sec":row.get("avg_ts", 0), "raw":row})
|
|
@@ -338,10 +340,10 @@ def _run_cpu_smoke(model_filter):
|
|
| 338 |
except Exception as exc:
|
| 339 |
result["phases"]["help"] = {"ok": False, "elapsed_sec": round(time.time() - help_t0, 2), "error": str(exc)}
|
| 340 |
|
| 341 |
-
command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "-fa", "0"]
|
| 342 |
t0 = time.time()
|
| 343 |
try:
|
| 344 |
-
run = subprocess.run(command,
|
| 345 |
result["phases"]["load_and_one_token"] = {
|
| 346 |
"ok": run.returncode == 0,
|
| 347 |
"elapsed_sec": round(time.time() - t0, 2),
|
|
@@ -373,7 +375,7 @@ def _run_cpu_process_probe(model_filter):
|
|
| 373 |
env["LD_LIBRARY_PATH"] = os.path.dirname(cli)
|
| 374 |
env["CUDA_VISIBLE_DEVICES"] = ""
|
| 375 |
env["GGML_CUDA"] = "0"
|
| 376 |
-
command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "-fa", "0"]
|
| 377 |
started = time.time()
|
| 378 |
child = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=env)
|
| 379 |
time.sleep(8)
|
|
@@ -389,8 +391,8 @@ def _run_cpu_process_probe(model_filter):
|
|
| 389 |
finally:
|
| 390 |
child.kill()
|
| 391 |
stdout, stderr = child.communicate(timeout=10)
|
| 392 |
-
probe["stdout"] = stdout[-2000:]
|
| 393 |
-
probe["stderr"] = stderr[-4000:]
|
| 394 |
probe["returncode"] = child.returncode
|
| 395 |
return probe
|
| 396 |
|
|
|
|
| 265 |
to = 90 if mi["size_gb"] <= 6 else 180
|
| 266 |
command = [bench, "-m", mp, "-t", str(nt), "-ngl", "0", "-c", "256", "-p", "32", "-n", "64", "-r", "1", "--no-mmap", "-o", "json"]
|
| 267 |
_log(f"Running llama-bench for {mi['name']} on CPU with {nt} threads")
|
| 268 |
+
with tempfile.TemporaryFile(mode="w+", encoding="utf-8") as output_file:
|
| 269 |
+
p = subprocess.run(command, stdout=output_file, stderr=subprocess.PIPE, text=True, timeout=to, env=env)
|
| 270 |
+
output_file.seek(0)
|
| 271 |
+
raw = output_file.read().strip()
|
| 272 |
wt = time.time()-t0
|
| 273 |
diagnostics = (p.stderr or "").strip()
|
| 274 |
if p.returncode != 0:
|
| 275 |
raise RuntimeError(f"llama-bench exited with code {p.returncode}: {diagnostics[-1500:]}")
|
|
|
|
| 276 |
parsed = json.loads(raw)
|
| 277 |
row = parsed[0] if isinstance(parsed, list) and parsed else parsed
|
| 278 |
r.update({"status":"success", "total_time_sec":round(wt,2), "prompt_tokens":32, "completion_tokens":64, "prompt_tokens_per_sec":row.get("avg_ts", 0), "generation_tokens_per_sec":row.get("avg_ts", 0), "raw":row})
|
|
|
|
| 340 |
except Exception as exc:
|
| 341 |
result["phases"]["help"] = {"ok": False, "elapsed_sec": round(time.time() - help_t0, 2), "error": str(exc)}
|
| 342 |
|
| 343 |
+
command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "--log-disable", "-fa", "0"]
|
| 344 |
t0 = time.time()
|
| 345 |
try:
|
| 346 |
+
run = subprocess.run(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, text=True, timeout=180, env=env)
|
| 347 |
result["phases"]["load_and_one_token"] = {
|
| 348 |
"ok": run.returncode == 0,
|
| 349 |
"elapsed_sec": round(time.time() - t0, 2),
|
|
|
|
| 375 |
env["LD_LIBRARY_PATH"] = os.path.dirname(cli)
|
| 376 |
env["CUDA_VISIBLE_DEVICES"] = ""
|
| 377 |
env["GGML_CUDA"] = "0"
|
| 378 |
+
command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "--log-disable", "-fa", "0"]
|
| 379 |
started = time.time()
|
| 380 |
child = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=env)
|
| 381 |
time.sleep(8)
|
|
|
|
| 391 |
finally:
|
| 392 |
child.kill()
|
| 393 |
stdout, stderr = child.communicate(timeout=10)
|
| 394 |
+
probe["stdout"] = (stdout or "")[-2000:]
|
| 395 |
+
probe["stderr"] = (stderr or "")[-4000:]
|
| 396 |
probe["returncode"] = child.returncode
|
| 397 |
return probe
|
| 398 |
|