cazyundee commited on
Commit
a8a716e
·
verified ·
1 Parent(s): 8ef48d6

Prevent llama output pipe deadlock

Browse files
Files changed (1) hide show
  1. app.py +9 -7
app.py CHANGED
@@ -265,12 +265,14 @@ def _run_bench_locked(model_filter=None):
265
  to = 90 if mi["size_gb"] <= 6 else 180
266
  command = [bench, "-m", mp, "-t", str(nt), "-ngl", "0", "-c", "256", "-p", "32", "-n", "64", "-r", "1", "--no-mmap", "-o", "json"]
267
  _log(f"Running llama-bench for {mi['name']} on CPU with {nt} threads")
268
- p = subprocess.run(command, capture_output=True, text=True, timeout=to, env=env)
 
 
 
269
  wt = time.time()-t0
270
  diagnostics = (p.stderr or "").strip()
271
  if p.returncode != 0:
272
  raise RuntimeError(f"llama-bench exited with code {p.returncode}: {diagnostics[-1500:]}")
273
- raw = (p.stdout or "").strip()
274
  parsed = json.loads(raw)
275
  row = parsed[0] if isinstance(parsed, list) and parsed else parsed
276
  r.update({"status":"success", "total_time_sec":round(wt,2), "prompt_tokens":32, "completion_tokens":64, "prompt_tokens_per_sec":row.get("avg_ts", 0), "generation_tokens_per_sec":row.get("avg_ts", 0), "raw":row})
@@ -338,10 +340,10 @@ def _run_cpu_smoke(model_filter):
338
  except Exception as exc:
339
  result["phases"]["help"] = {"ok": False, "elapsed_sec": round(time.time() - help_t0, 2), "error": str(exc)}
340
 
341
- command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "-fa", "0"]
342
  t0 = time.time()
343
  try:
344
- run = subprocess.run(command, capture_output=True, text=True, timeout=180, env=env)
345
  result["phases"]["load_and_one_token"] = {
346
  "ok": run.returncode == 0,
347
  "elapsed_sec": round(time.time() - t0, 2),
@@ -373,7 +375,7 @@ def _run_cpu_process_probe(model_filter):
373
  env["LD_LIBRARY_PATH"] = os.path.dirname(cli)
374
  env["CUDA_VISIBLE_DEVICES"] = ""
375
  env["GGML_CUDA"] = "0"
376
- command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "-fa", "0"]
377
  started = time.time()
378
  child = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=env)
379
  time.sleep(8)
@@ -389,8 +391,8 @@ def _run_cpu_process_probe(model_filter):
389
  finally:
390
  child.kill()
391
  stdout, stderr = child.communicate(timeout=10)
392
- probe["stdout"] = stdout[-2000:]
393
- probe["stderr"] = stderr[-4000:]
394
  probe["returncode"] = child.returncode
395
  return probe
396
 
 
265
  to = 90 if mi["size_gb"] <= 6 else 180
266
  command = [bench, "-m", mp, "-t", str(nt), "-ngl", "0", "-c", "256", "-p", "32", "-n", "64", "-r", "1", "--no-mmap", "-o", "json"]
267
  _log(f"Running llama-bench for {mi['name']} on CPU with {nt} threads")
268
+ with tempfile.TemporaryFile(mode="w+", encoding="utf-8") as output_file:
269
+ p = subprocess.run(command, stdout=output_file, stderr=subprocess.PIPE, text=True, timeout=to, env=env)
270
+ output_file.seek(0)
271
+ raw = output_file.read().strip()
272
  wt = time.time()-t0
273
  diagnostics = (p.stderr or "").strip()
274
  if p.returncode != 0:
275
  raise RuntimeError(f"llama-bench exited with code {p.returncode}: {diagnostics[-1500:]}")
 
276
  parsed = json.loads(raw)
277
  row = parsed[0] if isinstance(parsed, list) and parsed else parsed
278
  r.update({"status":"success", "total_time_sec":round(wt,2), "prompt_tokens":32, "completion_tokens":64, "prompt_tokens_per_sec":row.get("avg_ts", 0), "generation_tokens_per_sec":row.get("avg_ts", 0), "raw":row})
 
340
  except Exception as exc:
341
  result["phases"]["help"] = {"ok": False, "elapsed_sec": round(time.time() - help_t0, 2), "error": str(exc)}
342
 
343
+ command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "--log-disable", "-fa", "0"]
344
  t0 = time.time()
345
  try:
346
+ run = subprocess.run(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, text=True, timeout=180, env=env)
347
  result["phases"]["load_and_one_token"] = {
348
  "ok": run.returncode == 0,
349
  "elapsed_sec": round(time.time() - t0, 2),
 
375
  env["LD_LIBRARY_PATH"] = os.path.dirname(cli)
376
  env["CUDA_VISIBLE_DEVICES"] = ""
377
  env["GGML_CUDA"] = "0"
378
+ command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "--log-disable", "-fa", "0"]
379
  started = time.time()
380
  child = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=env)
381
  time.sleep(8)
 
391
  finally:
392
  child.kill()
393
  stdout, stderr = child.communicate(timeout=10)
394
+ probe["stdout"] = (stdout or "")[-2000:]
395
+ probe["stderr"] = (stderr or "")[-4000:]
396
  probe["returncode"] = child.returncode
397
  return probe
398