cazyundee commited on
Commit
a5de9d9
·
verified ·
1 Parent(s): 318e7cb

Add CPU model loader diagnostics

Browse files
Files changed (1) hide show
  1. app.py +62 -0
app.py CHANGED
@@ -360,6 +360,60 @@ def _run_cpu_smoke(model_filter):
360
  return result
361
 
362
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
363
  def _run_benchmark_job(job_id, kind, model_filter=None, force=False):
364
  try:
365
  if kind == "smoke":
@@ -464,6 +518,14 @@ def _patched_create_app(blocks, **kwargs):
464
  try: return JSONResponse(_search_web(q, backend=backend, max_results=max_results, extract_top=extract_top, region=region))
465
  except Exception as e: return JSONResponse({"error": str(e)}, status_code=502)
466
 
 
 
 
 
 
 
 
 
467
  @fa_app.get("/respite/benchmark/smoke")
468
  def _smoke(model: str="Gemma 4 E4B", job_id: str=""):
469
  if job_id:
 
360
  return result
361
 
362
 
363
+ def _run_cpu_diagnose(model_filter):
364
+ import subprocess
365
+ from huggingface_hub import hf_hub_download
366
+
367
+ model = next((m for m in _BENCH_MODELS if m["name"] == model_filter), None)
368
+ if model is None:
369
+ raise ValueError(f"Unknown model: {model_filter}")
370
+ cli = _ensure_llama()
371
+ path = hf_hub_download(repo_id=model["repo"], filename=model["file"], cache_dir=_BENCH_CACHE)
372
+ env = os.environ.copy()
373
+ env["LD_LIBRARY_PATH"] = os.path.dirname(cli)
374
+ env["CUDA_VISIBLE_DEVICES"] = ""
375
+ env["GGML_CUDA"] = "0"
376
+
377
+ def command_output(command, timeout=10):
378
+ try:
379
+ p = subprocess.run(command, capture_output=True, text=True, timeout=timeout, env=env)
380
+ return {"returncode": p.returncode, "output": (p.stdout + p.stderr)[-5000:]}
381
+ except Exception as exc:
382
+ return {"error": str(exc)}
383
+
384
+ result = {
385
+ "name": model["name"],
386
+ "model_path": path,
387
+ "model_size_bytes": os.path.getsize(path),
388
+ "model_magic": open(path, "rb").read(4).decode("ascii", "replace"),
389
+ "cpu_cores": _parse_cpu_count(),
390
+ "cpu_flags": command_output(["bash", "-lc", "grep -m1 '^flags' /proc/cpuinfo"], 5),
391
+ "filesystem": command_output(["df", "-T", path], 5),
392
+ "binary": command_output(["file", cli], 5),
393
+ "libraries": command_output(["ldd", cli], 5),
394
+ }
395
+ trace_path = os.path.join(_BENCH_CACHE, "llama-load.trace")
396
+ if shutil.which("strace"):
397
+ try:
398
+ p = subprocess.run(
399
+ ["strace", "-f", "-tt", "-o", trace_path, cli, "-m", path,
400
+ "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.",
401
+ "--no-display-prompt", "--no-warmup", "-fa", "0"],
402
+ capture_output=True, text=True, timeout=25, env=env,
403
+ )
404
+ result["load_trace"] = {"returncode": p.returncode, "stdout": (p.stdout or "")[-1000:], "stderr": (p.stderr or "")[-2000:]}
405
+ except subprocess.TimeoutExpired:
406
+ result["load_trace"] = {"timeout": True}
407
+ try:
408
+ with open(trace_path, "r", encoding="utf-8", errors="replace") as f:
409
+ result["trace_tail"] = f.read()[-12000:]
410
+ except OSError as exc:
411
+ result["trace_tail_error"] = str(exc)
412
+ else:
413
+ result["load_trace"] = {"available": False}
414
+ return result
415
+
416
+
417
  def _run_benchmark_job(job_id, kind, model_filter=None, force=False):
418
  try:
419
  if kind == "smoke":
 
518
  try: return JSONResponse(_search_web(q, backend=backend, max_results=max_results, extract_top=extract_top, region=region))
519
  except Exception as e: return JSONResponse({"error": str(e)}, status_code=502)
520
 
521
+ @fa_app.get("/respite/benchmark/diagnose")
522
+ def _diagnose(model: str="TinyLlama 1.1B (control)"):
523
+ try:
524
+ with _BENCH_LOCK:
525
+ return JSONResponse(_run_cpu_diagnose(model.strip()))
526
+ except Exception as e:
527
+ return JSONResponse({"error": str(e)}, status_code=500)
528
+
529
  @fa_app.get("/respite/benchmark/smoke")
530
  def _smoke(model: str="Gemma 4 E4B", job_id: str=""):
531
  if job_id: