Spaces:
Running on Zero
Running on Zero
| import os | |
| # Keep all model inference on CPU; this must be set before importing torch. | |
| os.environ["CUDA_VISIBLE_DEVICES"] = "" | |
| os.environ["CUDA_DEVICE_ORDER"] = "PCI_BUS_ID" | |
| import platform | |
| import sys | |
| import time | |
| import shutil | |
| import tempfile | |
| import threading | |
| import spaces | |
| import torch | |
| import torchaudio | |
| import gradio as gr | |
| from fastapi import Body | |
| from fastapi.responses import JSONResponse | |
| from einops import rearrange | |
| from huggingface_hub import login | |
| from stable_audio_3 import StableAudioModel | |
| import urllib.request | |
| import urllib.parse | |
| import json as _json | |
| import node_probe | |
| import node_runtime | |
| import searx_bridge | |
| import searx_runtime | |
| for _stream in (sys.stdout, sys.stderr): | |
| if hasattr(_stream, "reconfigure"): | |
| _stream.reconfigure(encoding="utf-8", errors="backslashreplace") | |
| hf_token = os.environ.get("HF_TOKEN") | |
| if hf_token: | |
| login(token=hf_token) | |
| def _gpu_startup_touch(): | |
| return "ok" | |
| # Keep the decorated symbol so ZeroGPU Spaces recognize the app, but do not call it. | |
| # The Respite API and all CPU benchmarks must not consume GPU time. | |
| def _log(msg): | |
| print(msg.encode("ascii", "backslashreplace").decode("ascii"), flush=True) | |
| def _read_cgroup(path): | |
| try: | |
| with open(path, "r", encoding="utf-8") as f: | |
| return f.read().strip() | |
| except (FileNotFoundError, OSError): | |
| return None | |
| def _parse_cpu_count(): | |
| v = _read_cgroup("/sys/fs/cgroup/cpu.max") | |
| if v: | |
| p = v.split() | |
| if len(p) == 2 and p[0] != "max": | |
| try: | |
| q, per = int(p[0]), int(p[1]) | |
| if per > 0: return q // per | |
| except: pass | |
| v = _read_cgroup("/sys/fs/cgroup/cpuset.cpus") | |
| if v: | |
| c = 0 | |
| for part in v.split(","): | |
| if "-" in part: | |
| s, e = part.split("-", 1) | |
| c += int(e) - int(s) + 1 | |
| else: | |
| c += 1 | |
| if c > 0: return c | |
| return os.cpu_count() | |
| def _parse_ram(): | |
| v = _read_cgroup("/sys/fs/cgroup/memory.max") | |
| if v and v != "max": | |
| try: | |
| val = int(v) | |
| if val < 2**62: return val | |
| except: pass | |
| try: | |
| with open("/proc/meminfo", "r", encoding="utf-8") as f: | |
| for line in f: | |
| if line.startswith("MemTotal:"): | |
| return int(line.split()[1]) * 1024 | |
| except: pass | |
| return None | |
| # --------------------------------------------------------------------------- | |
| # API metadata | |
| # --------------------------------------------------------------------------- | |
| API_RESOURCES = { | |
| "audio_generation": { | |
| "name": "Audio generation", | |
| "description": "Generate music or sound effects from a text prompt.", | |
| "endpoint": "/respite/audio/generate", | |
| "method": "POST", | |
| "input": { | |
| "prompt": "string", | |
| "duration": "number (1-120 seconds)", | |
| "steps": "integer (1-50)", | |
| "cfg_scale": "number (0-10)", | |
| "seed": "integer (-1 for random)", | |
| "model": "small-music | small-sfx", | |
| }, | |
| "output": "WAV audio file", | |
| }, | |
| "qwen38_inference": { | |
| "name": "Qwen3.8-27B text generation", | |
| "description": "Generate text with Qwen3.8-27B using the Space CPU; no GPU allocation.", | |
| "endpoint": "/respite/text/qwen38", | |
| "method": "POST", | |
| "input": { | |
| "prompt": "string", | |
| "max_new_tokens": "integer (1-256, default 64)", | |
| "temperature": "number (0-2, default 0.7)", | |
| "thinking": "boolean (default true; Qwen3.8 is a reasoning model)", | |
| }, | |
| "output": "JSON containing generated text and CPU timing", | |
| }, | |
| "tinyllama_inference": { | |
| "name": "TinyLlama text generation", | |
| "description": "Generate text with TinyLlama 1.1B using the Space CPU; no GPU allocation.", | |
| "endpoint": "/respite/text/generate", | |
| "method": "POST", | |
| "input": { | |
| "prompt": "string", | |
| "max_new_tokens": "integer (1-256, default 64)", | |
| "temperature": "number (0-2, default 0.7)", | |
| }, | |
| "output": "JSON containing generated text and CPU timing", | |
| }, | |
| "web_search": { | |
| "name": "Web search", | |
| "description": "Search the web via DuckDuckGo.", | |
| "endpoint": "/respite/search", | |
| "method": "GET", | |
| "input": { | |
| "q": "string", | |
| "backend": "'text', 'news', or 'images'", | |
| "max_results": "integer (1-50, default 10)", | |
| "extract_top": "integer (0-5, default 0)", | |
| "region": "string (default 'wt-wt')", | |
| }, | |
| "output": "JSON with title, url, content per result.", | |
| }, | |
| } | |
| API_SPECS = { | |
| "name": "Respite API", | |
| "version": "1.0.0", | |
| "description": "General-purpose AI API server with audio generation capabilities.", | |
| "base_path": "/respite", | |
| "authentication": "none", | |
| "content_types": ["application/json", "audio/wav"], | |
| "resources_endpoint": "/respite/resources", | |
| "specs_endpoint": "/respite/specs", | |
| "limits": { | |
| "max_concurrent_requests": 1, | |
| "max_queue_size": 4, | |
| "audio_max_duration_seconds": 120, | |
| }, | |
| } | |
| def _get_runtime_specs(): | |
| storage = shutil.disk_usage(os.getcwd()) | |
| hw = os.environ.get("SPACE_HARDWARE", "zero-a10g") | |
| is_zgpu = "zero" in hw.lower() | |
| return { | |
| "platform": platform.platform(), | |
| "python_version": platform.python_version(), | |
| "cpu_cores": _parse_cpu_count(), | |
| "host_cpu_cores": os.cpu_count(), | |
| "ram_bytes": _parse_ram(), | |
| "storage_total_bytes": storage.total, | |
| "storage_used_bytes": storage.used, | |
| "storage_free_bytes": storage.free, | |
| "hardware": hw, | |
| "gpu": {"type": "NVIDIA RTX Pro 6000 Blackwell", "vram_bytes": 48*1024**3, "shared_zero_gpu": True} if is_zgpu else None, | |
| } | |
| # --------------------------------------------------------------------------- | |
| # Web search | |
| # --------------------------------------------------------------------------- | |
| try: | |
| from ddgs import DDGS | |
| def _search_web(query, backend="text", max_results=10, extract_top=0, region="wt-wt"): | |
| extract_top = max(0, min(5, extract_top)) | |
| max_results = max(1, min(50, max_results)) | |
| with DDGS() as ddgs: | |
| if backend == "news": | |
| results = [{"title": r.get("title",""), "url": r.get("url",""), "content": r.get("body",""), "source": r.get("source",""), "date": r.get("date","")} for r in ddgs.news(query, max_results=max_results, region=region)] | |
| elif backend == "images": | |
| results = [{"title": r.get("title",""), "url": r.get("image",""), "source": r.get("source","")} for r in ddgs.images(query, max_results=max_results, region=region)] | |
| else: | |
| results = [{"title": r.get("title",""), "url": r.get("href",""), "content": r.get("body","")} for r in ddgs.text(query, max_results=max_results, region=region)] | |
| if extract_top > 0 and results and backend != "images": | |
| with DDGS() as ddgs: | |
| for url in [r["url"] for r in results[:extract_top] if r.get("url")]: | |
| try: | |
| ex = ddgs.extract(url) | |
| ct = ex.get("content","") | |
| if ct: | |
| for r in results: | |
| if r["url"] == url: | |
| r["extracted_content"] = ct[:8000] | |
| break | |
| except: pass | |
| return {"query": query, "backend": backend, "results": results, "total_results": len(results)} | |
| except ImportError: | |
| def _search_web(query, **kw): | |
| raise RuntimeError("Search not available") | |
| # --------------------------------------------------------------------------- | |
| # Benchmark | |
| # --------------------------------------------------------------------------- | |
| _BENCH_MODELS = [ | |
| {"name": "TinyLlama 1.1B (control)", "repo": "TheBloke/TinyLlama-1.1B-Chat-v1.0-GGUF", "file": "tinyllama-1.1b-chat-v1.0.Q4_K_M.gguf", "size_gb": 0.7}, | |
| {"name": "Gemma 4 E4B", "repo": "unsloth/gemma-4-E4B-it-GGUF", "file": "gemma-4-E4B-it-Q4_K_M.gguf", "size_gb": 2.5}, | |
| {"name": "Qwen3 8B", "repo": "Qwen/Qwen3-8B-GGUF", "file": "Qwen3-8B-Q4_K_M.gguf", "size_gb": 5.0}, | |
| {"name": "Qwen3.8 27B", "repo": "unsloth/Qwen3.8-27B-GGUF", "file": "Qwen3.8-27B-Q4_0.gguf", "size_gb": 16.0}, | |
| ] | |
| _BENCH_CACHE = os.path.join(tempfile.gettempdir(), "llm_bench_cpu_v2") | |
| _BENCH_RESULTS = [] | |
| _BENCH_LOCK = threading.Lock() | |
| _BENCH_JOBS = {} | |
| _BENCH_JOBS_LOCK = threading.Lock() | |
| def _ensure_llama(): | |
| d = os.path.join(_BENCH_CACHE, "bin") | |
| cli = os.path.join(d, "llama-cli") | |
| bench = os.path.join(d, "llama-bench") | |
| if os.path.isfile(cli) and os.path.isfile(bench): return cli | |
| import subprocess | |
| os.makedirs(d, exist_ok=True) | |
| src = os.path.join(_BENCH_CACHE, "src") | |
| if os.path.isdir(src): shutil.rmtree(src, ignore_errors=True) | |
| _log("Building llama.cpp...") | |
| subprocess.check_call(["git","clone","--depth=1","https://github.com/ggml-org/llama.cpp.git",src], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) | |
| bd = os.path.join(_BENCH_CACHE, "build") | |
| os.makedirs(bd, exist_ok=True) | |
| nt = _parse_cpu_count() or 8 | |
| subprocess.check_call(["cmake","-S",src,"-B",bd,"-DGGML_NATIVE=OFF","-DGGML_OPENMP=OFF","-DGGML_CUDA=OFF","-DGGML_AVX=ON","-DGGML_AVX2=ON","-DGGML_FMA=ON","-DGGML_AVX512=OFF","-DBUILD_SHARED_LIBS=OFF"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) | |
| subprocess.check_call(["cmake","--build",bd,"-j",str(nt),"--target","llama-cli","llama-bench"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) | |
| import glob as _g | |
| cands = [p for p in _g.glob(os.path.join(bd,"**","llama-cli"), recursive=True) if os.path.isfile(p)] | |
| bench_cands = [p for p in _g.glob(os.path.join(bd,"**","llama-bench"), recursive=True) if os.path.isfile(p)] | |
| if cands: | |
| shutil.copy2(cands[0], cli); os.chmod(cli, 0o755) | |
| else: raise RuntimeError("llama-cli not found") | |
| if bench_cands: | |
| bench = os.path.join(d, "llama-bench") | |
| shutil.copy2(bench_cands[0], bench); os.chmod(bench, 0o755) | |
| return cli | |
| def _run_bench(model_filter=None): | |
| with _BENCH_LOCK: | |
| return _run_bench_locked(model_filter) | |
| def _run_bench_locked(model_filter=None): | |
| global _BENCH_RESULTS | |
| if _BENCH_RESULTS and not model_filter: return _BENCH_RESULTS | |
| import subprocess, re | |
| from huggingface_hub import hf_hub_download | |
| os.makedirs(_BENCH_CACHE, exist_ok=True) | |
| cli = _ensure_llama() | |
| bench = os.path.join(os.path.dirname(cli), "llama-bench") | |
| if not os.path.isfile(bench): | |
| raise RuntimeError("llama-bench was not built; redeploy to rebuild llama.cpp") | |
| nt = _parse_cpu_count() or 8 | |
| models = [m for m in _BENCH_MODELS if m["name"]==model_filter] if model_filter else _BENCH_MODELS | |
| results = [] | |
| for mi in models: | |
| r = {"name": mi["name"], "status": "error", "engine": "llama-bench", "gpu_layers": 0} | |
| try: | |
| mp = hf_hub_download(repo_id=mi["repo"], filename=mi["file"], cache_dir=_BENCH_CACHE) | |
| env = os.environ.copy() | |
| env["LD_LIBRARY_PATH"] = os.path.dirname(cli) | |
| env["CUDA_VISIBLE_DEVICES"] = "" | |
| env["GGML_CUDA"] = "0" | |
| env["OMP_NUM_THREADS"] = str(nt) | |
| env["OMP_THREAD_LIMIT"] = str(nt) | |
| env["OMP_DYNAMIC"] = "FALSE" | |
| env["MKL_NUM_THREADS"] = str(nt) | |
| env["OPENBLAS_NUM_THREADS"] = str(nt) | |
| env["VECLIB_MAXIMUM_THREADS"] = str(nt) | |
| env["NUMEXPR_NUM_THREADS"] = str(nt) | |
| t0 = time.time() | |
| to = 90 if mi["size_gb"] <= 6 else 180 | |
| command = [bench, "-m", mp, "-t", str(nt), "-ngl", "0", "-c", "256", "-p", "32", "-n", "64", "-r", "1", "--no-mmap", "-o", "json"] | |
| _log(f"Running llama-bench for {mi['name']} on CPU with {nt} threads") | |
| with tempfile.TemporaryFile(mode="w+", encoding="utf-8") as output_file: | |
| p = subprocess.run(command, stdout=output_file, stderr=subprocess.PIPE, text=True, timeout=to, env=env) | |
| output_file.seek(0) | |
| raw = output_file.read().strip() | |
| wt = time.time()-t0 | |
| diagnostics = (p.stderr or "").strip() | |
| if p.returncode != 0: | |
| raise RuntimeError(f"llama-bench exited with code {p.returncode}: {diagnostics[-1500:]}") | |
| parsed = json.loads(raw) | |
| row = parsed[0] if isinstance(parsed, list) and parsed else parsed | |
| r.update({"status":"success", "total_time_sec":round(wt,2), "prompt_tokens":32, "completion_tokens":64, "prompt_tokens_per_sec":row.get("avg_ts", 0), "generation_tokens_per_sec":row.get("avg_ts", 0), "raw":row}) | |
| except subprocess.TimeoutExpired: | |
| r["error"] = f"llama-bench timed out after {to}s on CPU" | |
| results.append(r) | |
| continue | |
| except Exception as e: | |
| r["error"] = str(e) | |
| results.append(r) | |
| import glob as _g | |
| for f in _g.glob(os.path.join(_BENCH_CACHE,"**","*.gguf"), recursive=True): | |
| try: os.unlink(f) | |
| except: pass | |
| if not model_filter: _BENCH_RESULTS = results | |
| return results | |
| def _run_cpu_smoke(model_filter): | |
| """Run a short CPU-only loader/inference diagnostic for one model.""" | |
| import subprocess | |
| from huggingface_hub import hf_hub_download | |
| model = next((m for m in _BENCH_MODELS if m["name"] == model_filter), None) | |
| if model is None: | |
| raise ValueError(f"Unknown model: {model_filter}") | |
| cli = _ensure_llama() | |
| env = os.environ.copy() | |
| env["LD_LIBRARY_PATH"] = os.path.dirname(cli) | |
| env["CUDA_VISIBLE_DEVICES"] = "" | |
| env["GGML_CUDA"] = "0" | |
| env["OMP_NUM_THREADS"] = "1" | |
| env["OMP_THREAD_LIMIT"] = "1" | |
| env["OMP_DYNAMIC"] = "FALSE" | |
| env["MKL_NUM_THREADS"] = "1" | |
| env["OPENBLAS_NUM_THREADS"] = "1" | |
| env["VECLIB_MAXIMUM_THREADS"] = "1" | |
| env["NUMEXPR_NUM_THREADS"] = "1" | |
| result = { | |
| "name": model["name"], | |
| "engine": "llama-cli", | |
| "gpu_layers": 0, | |
| "threads": 1, | |
| "no_mmap": True, | |
| "phases": {}, | |
| } | |
| t0 = time.time() | |
| version = subprocess.run([cli, "--version"], capture_output=True, text=True, timeout=15, env=env) | |
| result["phases"]["binary"] = { | |
| "ok": version.returncode == 0, | |
| "elapsed_sec": round(time.time() - t0, 2), | |
| "returncode": version.returncode, | |
| "output": (version.stdout + version.stderr)[-1000:], | |
| } | |
| if version.returncode != 0: | |
| return result | |
| t0 = time.time() | |
| try: | |
| path = hf_hub_download(repo_id=model["repo"], filename=model["file"], cache_dir=_BENCH_CACHE) | |
| result["model_path"] = path | |
| result["phases"]["download"] = {"ok": True, "elapsed_sec": round(time.time() - t0, 2)} | |
| except Exception as exc: | |
| result["phases"]["download"] = {"ok": False, "elapsed_sec": round(time.time() - t0, 2), "error": str(exc)} | |
| return result | |
| help_t0 = time.time() | |
| try: | |
| help_run = subprocess.run([cli, "--help"], capture_output=True, text=True, timeout=15, env=env) | |
| result["phases"]["help"] = {"ok": help_run.returncode == 0, "elapsed_sec": round(time.time() - help_t0, 2), "returncode": help_run.returncode} | |
| except Exception as exc: | |
| result["phases"]["help"] = {"ok": False, "elapsed_sec": round(time.time() - help_t0, 2), "error": str(exc)} | |
| command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "--no-mmap", "--log-disable", "-fa", "0"] | |
| t0 = time.time() | |
| try: | |
| run = subprocess.run(command, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, text=True, timeout=180, env=env) | |
| result["phases"]["load_and_one_token"] = { | |
| "ok": run.returncode == 0, | |
| "elapsed_sec": round(time.time() - t0, 2), | |
| "returncode": run.returncode, | |
| "stdout": (run.stdout or "")[-1000:], | |
| "stderr": (run.stderr or "")[-3000:], | |
| } | |
| except subprocess.TimeoutExpired as exc: | |
| result["phases"]["load_and_one_token"] = { | |
| "ok": False, | |
| "elapsed_sec": round(time.time() - t0, 2), | |
| "timeout": True, | |
| "stdout": str(exc.stdout or "")[-1000:], | |
| "stderr": str(exc.stderr or "")[-3000:], | |
| } | |
| return result | |
| def _run_cpu_process_probe(model_filter): | |
| import subprocess | |
| from huggingface_hub import hf_hub_download | |
| model = next((m for m in _BENCH_MODELS if m["name"] == model_filter), None) | |
| if model is None: | |
| raise ValueError(f"Unknown model: {model_filter}") | |
| cli = _ensure_llama() | |
| path = hf_hub_download(repo_id=model["repo"], filename=model["file"], cache_dir=_BENCH_CACHE) | |
| env = os.environ.copy() | |
| env["LD_LIBRARY_PATH"] = os.path.dirname(cli) | |
| env["CUDA_VISIBLE_DEVICES"] = "" | |
| env["GGML_CUDA"] = "0" | |
| env["OMP_NUM_THREADS"] = "1" | |
| env["OMP_THREAD_LIMIT"] = "1" | |
| env["OMP_DYNAMIC"] = "FALSE" | |
| env["MKL_NUM_THREADS"] = "1" | |
| env["OPENBLAS_NUM_THREADS"] = "1" | |
| env["VECLIB_MAXIMUM_THREADS"] = "1" | |
| env["NUMEXPR_NUM_THREADS"] = "1" | |
| command = [cli, "-m", path, "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", "--no-display-prompt", "--no-warmup", "--no-mmap", "--log-disable", "-fa", "0"] | |
| started = time.time() | |
| child = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=env) | |
| time.sleep(8) | |
| probe = {"pid": child.pid, "elapsed_sec": round(time.time() - started, 2), "command": command, "gpu_layers": 0} | |
| try: | |
| probe["ps"] = subprocess.run(["ps", "-o", "pid,stat,pcpu,pmem,rss,wchan:40,cmd", "-p", str(child.pid)], capture_output=True, text=True, timeout=5).stdout | |
| with open(f"/proc/{child.pid}/io", "r", encoding="utf-8") as f: | |
| probe["io"] = f.read() | |
| with open(f"/proc/{child.pid}/status", "r", encoding="utf-8") as f: | |
| probe["status"] = "\n".join(line for line in f if line.startswith(("State:", "VmRSS:", "Threads:"))) | |
| except Exception as exc: | |
| probe["probe_error"] = str(exc) | |
| finally: | |
| child.kill() | |
| stdout, stderr = child.communicate(timeout=10) | |
| probe["stdout"] = (stdout or "")[-2000:] | |
| probe["stderr"] = (stderr or "")[-4000:] | |
| probe["returncode"] = child.returncode | |
| return probe | |
| def _run_cpu_diagnose(model_filter): | |
| import subprocess | |
| from huggingface_hub import hf_hub_download | |
| model = next((m for m in _BENCH_MODELS if m["name"] == model_filter), None) | |
| if model is None: | |
| raise ValueError(f"Unknown model: {model_filter}") | |
| cli = _ensure_llama() | |
| path = hf_hub_download(repo_id=model["repo"], filename=model["file"], cache_dir=_BENCH_CACHE) | |
| env = os.environ.copy() | |
| env["LD_LIBRARY_PATH"] = os.path.dirname(cli) | |
| env["CUDA_VISIBLE_DEVICES"] = "" | |
| env["GGML_CUDA"] = "0" | |
| def command_output(command, timeout=10): | |
| try: | |
| p = subprocess.run(command, capture_output=True, text=True, timeout=timeout, env=env) | |
| return {"returncode": p.returncode, "output": (p.stdout + p.stderr)[-5000:]} | |
| except Exception as exc: | |
| return {"error": str(exc)} | |
| result = { | |
| "name": model["name"], | |
| "model_path": path, | |
| "model_size_bytes": os.path.getsize(path), | |
| "model_magic": open(path, "rb").read(4).decode("ascii", "replace"), | |
| "cpu_cores": _parse_cpu_count(), | |
| "cpu_flags": command_output(["bash", "-lc", "grep -m1 '^flags' /proc/cpuinfo"], 5), | |
| "filesystem": command_output(["df", "-T", path], 5), | |
| "binary": command_output(["file", cli], 5), | |
| "libraries": command_output(["ldd", cli], 5), | |
| } | |
| trace_path = os.path.join(_BENCH_CACHE, "llama-load.trace") | |
| if shutil.which("strace"): | |
| try: | |
| p = subprocess.run( | |
| ["strace", "-f", "-tt", "-o", trace_path, cli, "-m", path, | |
| "-t", "1", "-ngl", "0", "-c", "256", "-n", "1", "-p", "Say hi.", | |
| "--no-display-prompt", "--no-warmup", "--no-mmap", "-fa", "0"], | |
| capture_output=True, text=True, timeout=25, env=env, | |
| ) | |
| result["load_trace"] = {"returncode": p.returncode, "stdout": (p.stdout or "")[-1000:], "stderr": (p.stderr or "")[-2000:]} | |
| except subprocess.TimeoutExpired: | |
| result["load_trace"] = {"timeout": True} | |
| try: | |
| with open(trace_path, "r", encoding="utf-8", errors="replace") as f: | |
| result["trace_tail"] = f.read()[-12000:] | |
| except OSError as exc: | |
| result["trace_tail_error"] = str(exc) | |
| else: | |
| result["load_trace"] = {"available": False} | |
| return result | |
| def _run_benchmark_job(job_id, kind, model_filter=None, force=False, request=None): | |
| try: | |
| if kind == "smoke": | |
| with _BENCH_LOCK: | |
| result = _run_cpu_smoke(model_filter) | |
| payload = {"status": "complete", "result": result} | |
| elif kind == "diagnose": | |
| with _BENCH_LOCK: | |
| result = _run_cpu_diagnose(model_filter) | |
| payload = {"status": "complete", "result": result} | |
| elif kind == "probe": | |
| with _BENCH_LOCK: | |
| result = _run_cpu_process_probe(model_filter) | |
| payload = {"status": "complete", "result": result} | |
| elif kind == "qwen38": | |
| request = request or _BENCH_JOBS.get(job_id, {}).get("request", {}) | |
| qwen_prompt = request.get("prompt") or model_filter or "" | |
| with _BENCH_LOCK: | |
| result = _generate_qwen38( | |
| qwen_prompt, | |
| request.get("max_new_tokens", 256), | |
| request.get("temperature", 0.7), | |
| request.get("thinking", False), | |
| ) | |
| payload = {"status": "complete", "result": result} | |
| else: | |
| if force: | |
| global _BENCH_RESULTS | |
| _BENCH_RESULTS = [] | |
| payload = { | |
| "status": "complete", | |
| "models": _run_bench(model_filter=model_filter), | |
| "system": { | |
| "cpu_cores": _parse_cpu_count(), | |
| "ram_bytes": _parse_ram(), | |
| "threads": _parse_cpu_count() or 8, | |
| }, | |
| } | |
| with _BENCH_JOBS_LOCK: | |
| _BENCH_JOBS[job_id] = payload | |
| except Exception as exc: | |
| with _BENCH_JOBS_LOCK: | |
| _BENCH_JOBS[job_id] = {"status": "error", "error": str(exc)} | |
| def _start_benchmark_job(kind, model_filter=None, force=False, request=None): | |
| job_id = f"bench-{int(time.time() * 1000)}-{threading.get_ident()}" | |
| with _BENCH_JOBS_LOCK: | |
| _BENCH_JOBS[job_id] = { | |
| "status": "running", | |
| "kind": kind, | |
| "model": model_filter, | |
| "started_at": time.time(), | |
| "request": request or {}, | |
| } | |
| threading.Thread( | |
| target=_run_benchmark_job, | |
| args=(job_id, kind, model_filter, force, request or {}), | |
| daemon=True, | |
| ).start() | |
| return job_id | |
| MODEL_CACHE = {} | |
| _TINYLLAMA = None | |
| _TINYLLAMA_TOKENIZER = None | |
| _TINYLLAMA_LOCK = threading.Lock() | |
| _TINYLLAMA_REPO = "TinyLlama/TinyLlama-1.1B-Chat-v1.0" | |
| _QWEN38 = None | |
| _QWEN38_PROCESSOR = None | |
| _QWEN38_LOCK = threading.Lock() | |
| _QWEN38_REPO = "Qwen/Qwen3.8-27B" | |
| def _load_qwen38(): | |
| global _QWEN38, _QWEN38_PROCESSOR | |
| with _QWEN38_LOCK: | |
| if _QWEN38 is not None: | |
| return _QWEN38_PROCESSOR, _QWEN38 | |
| from transformers import AutoModelForImageTextToText, AutoProcessor | |
| try: | |
| torch.set_num_threads(min(_parse_cpu_count() or 1, 16)) | |
| except RuntimeError: | |
| pass | |
| processor = AutoProcessor.from_pretrained(_QWEN38_REPO) | |
| model = AutoModelForImageTextToText.from_pretrained( | |
| _QWEN38_REPO, | |
| torch_dtype=torch.bfloat16, | |
| low_cpu_mem_usage=True, | |
| device_map="cpu", | |
| ) | |
| model.eval() | |
| _QWEN38_PROCESSOR = processor | |
| _QWEN38 = model | |
| return processor, model | |
| def _generate_qwen38(prompt, max_new_tokens=256, temperature=0.7, thinking=True): | |
| processor, model = _load_qwen38() | |
| max_new_tokens = max(1, min(256, int(max_new_tokens))) | |
| temperature = max(0.0, min(2.0, float(temperature))) | |
| content = str(prompt).strip() | |
| messages = [{"role": "user", "content": [{"type": "text", "text": content}]}] | |
| try: | |
| text = processor.apply_chat_template( | |
| messages, | |
| tokenize=False, | |
| add_generation_prompt=True, | |
| enable_thinking=True, | |
| ) | |
| except TypeError: | |
| text = processor.apply_chat_template(messages, tokenize=False, add_generation_prompt=True) | |
| inputs = processor(text=[text], return_tensors="pt") | |
| inputs = {key: value.to("cpu") if hasattr(value, "to") else value for key, value in inputs.items()} | |
| tokenizer = processor.tokenizer | |
| eos_id = tokenizer.eos_token_id | |
| pad_id = tokenizer.pad_token_id if tokenizer.pad_token_id is not None else eos_id | |
| input_tokens = int(inputs["input_ids"].shape[-1]) | |
| t0 = time.time() | |
| with torch.inference_mode(): | |
| output = model.generate( | |
| **inputs, | |
| max_new_tokens=max_new_tokens, | |
| do_sample=temperature > 0, | |
| temperature=max(temperature, 0.01), | |
| eos_token_id=eos_id, | |
| pad_token_id=pad_id, | |
| return_dict_in_generate=True, | |
| output_scores=True, | |
| ) | |
| elapsed = max(time.time() - t0, 0.001) | |
| sequences = output.sequences | |
| generated = sequences[0, input_tokens:] | |
| eos_positions = (generated == eos_id).nonzero(as_tuple=True)[0].tolist() if eos_id is not None else [] | |
| answer = tokenizer.decode(generated, skip_special_tokens=True).strip() | |
| return { | |
| "text": answer, | |
| "prompt_tokens": input_tokens, | |
| "completion_tokens": int(generated.shape[-1]), | |
| "generation_tokens_per_sec": round(float(generated.shape[-1]) / elapsed, 2), | |
| "inference_time_sec": round(elapsed, 3), | |
| "device": "cpu", | |
| "dtype": "bfloat16", | |
| "thinking": bool(thinking), | |
| "debug": { | |
| "rendered_chat_template": text, | |
| "input_ids": inputs["input_ids"][0].tolist(), | |
| "generated_token_ids": generated.tolist(), | |
| "eos_token_id": eos_id, | |
| "pad_token_id": pad_id, | |
| "eos_positions": eos_positions, | |
| "sequence_length": int(sequences.shape[-1]), | |
| "max_new_tokens": max_new_tokens, | |
| "scores_steps": len(output.scores) if output.scores is not None else 0, | |
| "finish_reason": "eos" if eos_positions else ("length" if generated.shape[-1] >= max_new_tokens else "unknown"), | |
| }, | |
| } | |
| def _load_tinyllama(): | |
| global _TINYLLAMA, _TINYLLAMA_TOKENIZER | |
| with _TINYLLAMA_LOCK: | |
| if _TINYLLAMA is not None: | |
| return _TINYLLAMA_TOKENIZER, _TINYLLAMA | |
| from transformers import AutoModelForCausalLM, AutoTokenizer | |
| try: | |
| torch.set_num_threads(min(_parse_cpu_count() or 1, 16)) | |
| except RuntimeError: | |
| # Gradio may already have started PyTorch work during app startup. | |
| pass | |
| tokenizer = AutoTokenizer.from_pretrained(_TINYLLAMA_REPO) | |
| model = AutoModelForCausalLM.from_pretrained( | |
| _TINYLLAMA_REPO, | |
| torch_dtype=torch.float32, | |
| low_cpu_mem_usage=True, | |
| ) | |
| model.to("cpu") | |
| model.eval() | |
| _TINYLLAMA_TOKENIZER = tokenizer | |
| _TINYLLAMA = model | |
| return tokenizer, model | |
| def _generate_tinyllama(prompt, max_new_tokens=64, temperature=0.7): | |
| tokenizer, model = _load_tinyllama() | |
| max_new_tokens = max(1, min(256, int(max_new_tokens))) | |
| temperature = max(0.0, min(2.0, float(temperature))) | |
| messages = [{"role": "user", "content": prompt}] | |
| if hasattr(tokenizer, "apply_chat_template"): | |
| templated = tokenizer.apply_chat_template( | |
| messages, add_generation_prompt=True, return_tensors="pt" | |
| ) | |
| if isinstance(templated, dict) or hasattr(templated, "input_ids"): | |
| inputs = templated["input_ids"] | |
| attention_mask = templated.get("attention_mask", torch.ones_like(inputs)) | |
| else: | |
| inputs = templated | |
| attention_mask = torch.ones_like(inputs) | |
| else: | |
| encoded = tokenizer(prompt, return_tensors="pt") | |
| inputs = encoded["input_ids"] | |
| attention_mask = encoded["attention_mask"] | |
| t0 = time.time() | |
| with torch.inference_mode(): | |
| output = model.generate( | |
| input_ids=inputs, | |
| attention_mask=attention_mask, | |
| max_new_tokens=max_new_tokens, | |
| do_sample=temperature > 0, | |
| temperature=max(temperature, 0.01), | |
| pad_token_id=tokenizer.eos_token_id, | |
| ) | |
| elapsed = max(time.time() - t0, 0.001) | |
| generated = output[0, inputs.shape[-1]:] | |
| text = tokenizer.decode(generated, skip_special_tokens=True) | |
| return {"text": text, "prompt_tokens": int(inputs.shape[-1]), "completion_tokens": int(generated.shape[-1]), "generation_tokens_per_sec": round(float(generated.shape[-1]) / elapsed, 2), "inference_time_sec": round(elapsed, 3), "device": "cpu"} | |
| def load_model(name): | |
| if name not in MODEL_CACHE: | |
| MODEL_CACHE[name] = StableAudioModel.from_pretrained(name, device="cpu") | |
| return MODEL_CACHE[name] | |
| def generate_audio(prompt, duration, steps, cfg_scale, seed, model_name): | |
| model = load_model(model_name) | |
| audio = model.generate(prompt=prompt, duration=duration, steps=steps, cfg_scale=cfg_scale, seed=seed, batch_size=1) | |
| audio = rearrange(audio, "b d n -> d (b n)") | |
| audio = audio.to(torch.float32).clamp(-1, 1).mul(32767).to(torch.int16).cpu() | |
| out = os.path.join(tempfile.gettempdir(), f"stable_audio_{seed}_{hash(prompt)&0xFFFFFFFF:08x}.wav") | |
| torchaudio.save(out, audio, 44100) | |
| return out | |
| # --------------------------------------------------------------------------- | |
| # Gradio UI | |
| # --------------------------------------------------------------------------- | |
| with gr.Blocks(title="Respite API") as demo: | |
| gr.Markdown("# Respite API") | |
| with gr.Row(): | |
| with gr.Column(): | |
| model_name = gr.Dropdown(choices=["small-music","small-sfx"], value="small-music", label="Model") | |
| prompt = gr.Textbox(label="Prompt", placeholder="Describe the audio...", lines=2) | |
| duration = gr.Slider(minimum=1, maximum=120, value=30, step=1, label="Duration") | |
| steps = gr.Slider(minimum=1, maximum=50, value=8, step=1, label="Steps") | |
| cfg_scale = gr.Slider(minimum=0.0, maximum=10.0, value=1.0, step=0.1, label="CFG Scale") | |
| seed = gr.Number(value=-1, label="Seed (-1=random)") | |
| btn = gr.Button("Generate", variant="primary") | |
| with gr.Column(): | |
| audio_output = gr.Audio(label="Generated Audio", type="filepath") | |
| btn.click(fn=generate_audio, inputs=[prompt, duration, steps, cfg_scale, seed, model_name], outputs=audio_output) | |
| # --------------------------------------------------------------------------- | |
| # Monkey-patch: intercept Gradio App creation to inject routes | |
| # --------------------------------------------------------------------------- | |
| from gradio.routes import App as GradioApp | |
| _orig_create_app = GradioApp.create_app | |
| def _patched_create_app(blocks, **kwargs): | |
| fa_app = _orig_create_app(blocks, **kwargs) | |
| def _r(): return JSONResponse({"resources": API_RESOURCES}) | |
| def _s(): return JSONResponse({**API_SPECS, "runtime": _get_runtime_specs()}) | |
| def _audio_api(payload: dict = Body(...)): | |
| prompt = str(payload.get("prompt", "")).strip() | |
| if not prompt: | |
| return JSONResponse({"error": "Missing 'prompt'"}, status_code=400) | |
| try: | |
| path = generate_audio( | |
| prompt, | |
| max(1, min(120, float(payload.get("duration", 30)))), | |
| max(1, min(50, int(payload.get("steps", 8)))), | |
| max(0.0, min(10.0, float(payload.get("cfg_scale", 1.0)))), | |
| int(payload.get("seed", -1)), | |
| str(payload.get("model", "small-music")), | |
| ) | |
| from fastapi.responses import FileResponse | |
| return FileResponse(path, media_type="audio/wav", filename=os.path.basename(path)) | |
| except Exception as e: | |
| return JSONResponse({"error": str(e)}, status_code=500) | |
| def _q(q: str="", backend: str="text", max_results: int=10, extract_top: int=0, region: str="wt-wt"): | |
| if not q: return JSONResponse({"error":"Missing 'q'"}, status_code=400) | |
| try: return JSONResponse(_search_web(q, backend=backend, max_results=max_results, extract_top=extract_top, region=region)) | |
| except Exception as e: return JSONResponse({"error": str(e)}, status_code=502) | |
| def _qwen38(payload: dict = Body(...)): | |
| prompt = str(payload.get("prompt", "")) | |
| if not prompt.strip(): | |
| return JSONResponse({"error": "Missing 'prompt'"}, status_code=400) | |
| job = _start_benchmark_job( | |
| "qwen38", | |
| None, | |
| request={ | |
| "prompt": prompt, | |
| "max_new_tokens": payload.get("max_new_tokens", 256), | |
| "temperature": payload.get("temperature", 0.7), | |
| "thinking": payload.get("thinking", True), | |
| }, | |
| ) | |
| return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/text/qwen38?job_id=" + job}, status_code=202) | |
| def _qwen38_status(job_id: str = ""): | |
| if not job_id: | |
| return JSONResponse({"error": "Missing 'job_id'"}, status_code=400) | |
| with _BENCH_JOBS_LOCK: | |
| return JSONResponse(_BENCH_JOBS.get(job_id, {"status": "not_found"})) | |
| def _tinyllama(payload: dict = Body(...)): | |
| prompt = str(payload.get("prompt", "")) | |
| if not prompt.strip(): | |
| return JSONResponse({"error": "Missing 'prompt'"}, status_code=400) | |
| try: | |
| return JSONResponse(_generate_tinyllama( | |
| prompt, | |
| payload.get("max_new_tokens", 64), | |
| payload.get("temperature", 0.7), | |
| )) | |
| except Exception as e: | |
| return JSONResponse({"error": str(e)}, status_code=500) | |
| def _probe(model: str="TinyLlama 1.1B (control)"): | |
| job = _start_benchmark_job("probe", model.strip()) | |
| return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/benchmark/diagnose?job_id=" + job}, status_code=202) | |
| def _diagnose(model: str="TinyLlama 1.1B (control)", job_id: str=""): | |
| if job_id: | |
| with _BENCH_JOBS_LOCK: | |
| return JSONResponse(_BENCH_JOBS.get(job_id, {"status": "not_found"})) | |
| job = _start_benchmark_job("diagnose", model.strip()) | |
| return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/benchmark/diagnose?job_id=" + job}, status_code=202) | |
| def _smoke(model: str="Gemma 4 E4B", job_id: str=""): | |
| if job_id: | |
| with _BENCH_JOBS_LOCK: | |
| return JSONResponse(_BENCH_JOBS.get(job_id, {"status": "not_found"})) | |
| job = _start_benchmark_job("smoke", model.strip()) | |
| return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/benchmark/smoke?job_id=" + job}, status_code=202) | |
| def _b(force: bool=False, model: str="", job_id: str=""): | |
| if job_id: | |
| with _BENCH_JOBS_LOCK: | |
| return JSONResponse(_BENCH_JOBS.get(job_id, {"status": "not_found"})) | |
| job = _start_benchmark_job("benchmark", model.strip() or None, force) | |
| return JSONResponse({"status": "accepted", "job_id": job, "poll": "/respite/benchmark?job_id=" + job}, status_code=202) | |
| # Node runtime capability probe + backend proxy | |
| try: | |
| node_probe.register_routes(fa_app) | |
| threading.Thread(target=node_runtime.start_node_backend, daemon=True).start() | |
| except Exception as _ne: | |
| _log(f"node probe registration failed: {_ne}") | |
| try: | |
| searx_bridge.register_routes(fa_app) | |
| threading.Thread(target=searx_runtime.start_searxng, daemon=True).start() | |
| except Exception as _se: | |
| _log(f"searxng bridge registration failed: {_se}") | |
| _log("Routes injected via create_app monkey-patch") | |
| return fa_app | |
| GradioApp.create_app = _patched_create_app | |
| # Disable file watcher | |
| import gradio.utils as _gr_utils | |
| if hasattr(_gr_utils, "watchfn_spaces"): | |
| _gr_utils.watchfn_spaces = lambda *a, **kw: None | |
| demo.queue(max_size=4, default_concurrency_limit=1) | |
| demo.launch(ssr_mode=False) | |