# @file app.py # @description AI Creative Studio & ZeroGPU LLM Hub โ€” High-Density Pro Dashboard # Showcases both HF PRO subscription resources: # โšก ZeroGPU (48GB Blackwell RTX PRO 6000) for local 27B LLMs & Tool Calling # ๐ŸŒ Serverless HF Inference API for Whisper, BGE-M3 & Qwen-VL. # # @changes # - [2026-07-08] [Composer] - Initial Gemma Heretic bucket-backed chat app # - [2026-08-12] [Jcode] - AI Creative Studio: multi-tab app combining ZeroGPU + Inference API # - [2026-08-15] [Jcode] - Add Qwen3.8-27B, 128k context, FlashAttention, OpenAI tool calling # - [2026-08-16] [Jcode] - High-density full-screen professional dashboard with live telemetry HUD import os import sys import shutil import subprocess import tempfile import time import hmac import json import re import uuid import gc import math from typing import Dict, List, Any, Optional from fastapi import FastAPI, Request, HTTPException from fastapi.responses import JSONResponse, StreamingResponse import uvicorn # Preload CUDA 12 runtime libs via ctypes try: import glob as _glob import ctypes as _ctypes import site _libs = [] for sp in site.getsitepackages(): _libs += _glob.glob(os.path.join(sp, "nvidia", "*", "lib", "*.so*")) for _p in _libs: try: _ctypes.CDLL(_p) except Exception: pass except Exception: pass try: import gradio as gr except ImportError: # Graceful fallback (same pattern as spaces/llama_cpp): allows offline # unit tests to import app.py without gradio installed. Mocks cover UI # builders, .Error, .themes, and event-handler chaining methods. class _MockComponent: def __init__(self, *args, **kwargs): pass def __enter__(self): return self def __exit__(self, *args): return False def _chain(self, *args, **kwargs): return self change = click = submit = select = upload = blur = _chain launch = _chain class _MockThemes: def __getattr__(self, name): return lambda *a, **kw: None class _MockGradio: Error = type("Error", (Exception,), {}) themes = _MockThemes() def __getattr__(self, name): return _MockComponent gr = _MockGradio() try: import spaces except ImportError: class _MockSpaces: @staticmethod def GPU(fn=None, *args, **kwargs): if fn is not None and callable(fn): return fn def decorator(f): return f return decorator spaces = _MockSpaces() from huggingface_hub import InferenceClient, hf_hub_download try: from llama_cpp import Llama except ImportError: Llama = None # โ”€โ”€โ”€ Config โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ SEARCH_DIRS = ["/tmp", "/data", "/models", "."] SUPPORTED_MODELS = ( "Qwen3.8-9B-Q8_0.gguf", "Qwen3.8-9B-Q4_K_M.gguf", "Qwen3.8-27B-Q4_K_M.gguf", "Qwen3.8-27B-Q6_K.gguf", "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf", "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf", "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf", "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf", "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf", "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf", "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", ) DEFAULT_MODEL = SUPPORTED_MODELS[0] MODEL_HUB_SOURCES = { "Qwen3.8-9B-Q8_0.gguf": { "repo_id": "empero-ai/Qwen3.8-9B-GGUF", "filename": "Qwen3.8-9B-Q8_0.gguf", }, "Qwen3.8-9B-Q4_K_M.gguf": { "repo_id": "empero-ai/Qwen3.8-9B-GGUF", "filename": "Qwen3.8-9B-Q4_K_M.gguf", }, "Qwen3.8-27B-Q4_K_M.gguf": { "repo_id": "unsloth/Qwen3.8-27B-GGUF", "filename": "Qwen3.8-27B-Q4_K_M.gguf", }, "Qwen3.8-27B-Q6_K.gguf": { "repo_id": "unsloth/Qwen3.8-27B-GGUF", "filename": "Qwen3.8-27B-Q6_K.gguf", }, "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf": { "repo_id": "mradermacher/Qwen3.8-27B-Uncensored-i1-GGUF", "filename": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", }, # The regular quant is deliberately selected over the MTP build. The # current llama-cpp-python integration does not enable draft decoding, so # MTP heads add ~0.4 GiB without providing an inference benefit. "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf": { "repo_id": "DavidAU/Qwen3.8-27B-TURBO-Fable-Cold-Fusion-735-882-Heretic-Uncensored-NEO-CODER-MAX-MTP-GGUF", "filename": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", }, "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf": { "repo_id": "petruhonk/Qwen3.8-9B-Distill-uncensored-heretic-GGUF", "filename": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", }, "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf": { "repo_id": "petruhonk/Qwen3.8-9B-Distill-uncensored-heretic-GGUF", "filename": "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf", }, "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf": { "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", }, "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf": { "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf", }, "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf": { "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf", }, "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf": { "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf", }, "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf": { "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf", }, "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf": { "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf", }, "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf": { "repo_id": "mradermacher/gemma-4-26B-A4B-it-ultra-uncensored-heretic-i1-GGUF", "filename": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", }, "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf": { "repo_id": "mradermacher/gemma-4-26B-A4B-it-ultra-uncensored-heretic-i1-GGUF", "filename": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", }, } import sys import shutil import random BASE_DIR = os.path.dirname(__file__) BREEZE_DIR = os.path.join(BASE_DIR, "zerogpu") # vendored breezeblue-ai/breeze-tts inference code if BREEZE_DIR not in sys.path: sys.path.insert(0, BREEZE_DIR) if BASE_DIR not in sys.path: sys.path.insert(0, BASE_DIR) BREEZE_MODEL_ID = "BreezeBlue/Breeze-TTS-2" _breeze_runtime_cache: Dict[str, Any] = {} def _breeze_available() -> bool: """Breeze TTS 2 needs the qwen-tts pin family (transformers==4.57.3), which is incompatible with diffusers Music3 ModularPipeline (huggingface_hub>=1.23). It is feature-gated: deps absent -> graceful 503 instead of build failure. breeze_infer is vendored under zerogpu/, so spec-checking it alone is not enough; qwen_tts is the pip-only package whose install implies the whole runtime dep chain (torchaudio, librosa, einops, onnxruntime) is present. Re-enable by adding qwen-tts + its deps back to requirements.txt once upstream unpins transformers.""" try: import importlib.util return ( importlib.util.find_spec("breeze_infer") is not None and importlib.util.find_spec("qwen_tts") is not None ) except Exception: return False # HF Serverless Inference API models # (FLUX_MODEL removed by policy: Space scope is Music + Music Videos only.) WHISPER_MODEL = "openai/whisper-large-v3" EMBEDDING_MODEL = "BAAI/bge-m3" VISION_MODEL = "Qwen/Qwen2.5-VL-72B-Instruct" # โ”€โ”€โ”€ Free Open-Weights ZeroGPU Model Registries ($0 Cost / 40 min A100 Quota) โ”€โ”€ VIDEO_MODELS = { "Wan 2.1 (Alibaba T2V 1.3B)": "Wan-AI/Wan2.1-T2V-1.3B", "Wan 2.1 (Alibaba I2V 1.3B)": "Wan-AI/Wan2.1-I2V-1.3B-480P", "LTX-Video (Lightricks Fast 0.9.1)": "Lightricks/LTX-Video", "CogVideoX-5B (THUDM 5B SOTA)": "THUDM/CogVideoX-5b", "CogVideoX-2B (THUDM SOTA)": "THUDM/CogVideoX-2b", "ZeroScope v2 (576w High-Res)": "cerspense/zeroscope_v2_576w", "ZeroScope v2 XL (1024w Upscaler)": "cerspense/zeroscope_v2_XL", "ModelScope Text-to-Video": "damo-vilab/modelscope-damo-text-to-video-synthesis", "AnimateDiff (Motion Adapter)": "guoyww/animatediff-motion-adapter", "I2VGen-XL (Image to Video)": "ali-vilab/i2vgen-xl", } AUDIO_MUSIC_MODELS = { # Full-song vocal model. Official card: diffusers ModularPipeline, CUDA, 24GB+ or CPU offload. "MiniMax Music 3 (full song + vocals)": "MiniMaxAI/MiniMax-Music3", "YuE Music Full-Song (vocal + instrumental)": "m-a-p/YuE-s1-7B-dpo", # YuE2-3B: frontier quality (WildSongBench 6.96), editable scores, 48kHz. CC BY-NC 4.0. "YuE2-3B (editable scores, 48kHz)": "m-a-p/YuE2-3B", # ACE-Step v1.5 XL Turbo: Apache 2.0, commercial-safe, <4GB VRAM, 48kHz. "ACE-Step v1.5 XL Turbo (commercial-safe)": "ACE-Step/acestep-v15-xl-turbo-diffusers", # Full-song instrumental / audio model. Access may require accepting HF terms. "Stable Audio 3 Medium (full structured music)": "stabilityai/stable-audio-3-medium", "MusicGen Large (instrumental SOTA)": "facebook/musicgen-large", "MusicGen Medium (instrumental)": "facebook/musicgen-medium", "MusicGen Small (instrumental)": "facebook/musicgen-small", "MusicGen Melody (instrumental)": "facebook/musicgen-melody", "AudioGen Medium (sound effects)": "facebook/audiogen-medium", "Stable Audio Open 1.0 (audio / loops)": "stabilityai/stable-audio-open-1.0", "Bark (speech / singing experiments)": "suno/bark", "F5-TTS (Voice Cloning SOTA)": "SWivid/F5-TTS", "Parler-TTS Mini (TTS expressive)": "parler-tts/parler-tts-mini-v1", "Whisper Large v3 Turbo (STT SOTA)": "openai/whisper-large-v3-turbo", # Experimental quality TTS (voice design / clone / direction). Weights are # RESEARCH AND NON-COMMERCIAL licensed; accepted for art experiments only. } # Breeze TTS 2 feature-gated: see _breeze_available() (qwen-tts pin conflict). if _breeze_available(): AUDIO_MUSIC_MODELS["Breeze TTS 2 (SOTA experimental TTS, EN/ZH)"] = "BreezeBlue/Breeze-TTS-2" # IMAGE_MODELS removed by policy (image generation out of Space scope). EMBEDDING_MODELS = { "BGE-M3 (Multilingual 100+ langs, 8k ctx)": "BAAI/bge-m3", "Nomic Embed Text v1.5 (8k Matryoshka)": "nomic-ai/nomic-embed-text-v1.5", "BGE Reranker Large (Cross-Encoder SOTA)": "BAAI/bge-reranker-large", } # Pipeline caches for ZeroGPU execution _video_pipeline_cache = {} _audio_pipeline_cache = {} # Keep model downloads in the persistent Space cache. Loading remains lazy: a # model is downloaded only when the user selects it, never during app startup. _HF_AUDIO_CACHE = os.environ.get("HF_HOME", "/data") os.environ.setdefault("HF_HOME", _HF_AUDIO_CACHE) os.environ.setdefault("HF_HUB_CACHE", os.path.join(_HF_AUDIO_CACHE, "hub")) _image_pipeline_cache = {} # โ”€โ”€โ”€ Startup Background Model Pre-caching on CPU โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ def _preload_heavy_models_on_cpu(): """Download weights to /data or standard cache on CPU so ZeroGPU lease is not eaten by network download.""" try: from huggingface_hub import snapshot_download cache_dir = "/data/hub" if (os.path.exists("/data") and os.access("/data", os.W_OK)) else os.environ.get("HF_HUB_CACHE", None) print("[Startup Pre-Cache] Pre-caching MiniMax Music 3 on CPU...", flush=True) snapshot_download( repo_id="MiniMaxAI/MiniMax-Music3", cache_dir=cache_dir, resume_download=True, ) print("[Startup Pre-Cache] MiniMax Music 3 weights cached successfully!", flush=True) except Exception as e: print(f"[Startup Pre-Cache] Note: Background preload skipped or failed: {e}", flush=True) # YuE2-3B (4.96GB) + ACE-Step v1.5 (9.73GB) pre-cache on CPU: first music # requests on a fresh container must not burn ZeroGPU lease on downloads. _music_precache_repos = ["m-a-p/YuE2-3B", "ACE-Step/acestep-v15-xl-turbo-diffusers", "m-a-p/YuE2-Vae"] for _repo in _music_precache_repos: try: from huggingface_hub import snapshot_download print(f"[Startup Pre-Cache] Pre-caching {_repo} on CPU...", flush=True) snapshot_download(repo_id=_repo, cache_dir=cache_dir, resume_download=True) print(f"[Startup Pre-Cache] {_repo} cached successfully!", flush=True) except Exception as e: print(f"[Startup Pre-Cache] Note: {_repo} preload skipped: {e}", flush=True) # Breeze TTS 2 (7.3GB shards + 651MB audio tokenizer) pre-cache on CPU so the # ZeroGPU lease is never spent on network downloads. if _breeze_available(): try: from huggingface_hub import snapshot_download cache_dir = "/data/hub" if (os.path.exists("/data") and os.access("/data", os.W_OK)) else os.environ.get("HF_HUB_CACHE", None) print("[Startup Pre-Cache] Pre-caching Breeze TTS 2 on CPU...", flush=True) snapshot_download( repo_id="BreezeBlue/Breeze-TTS-2", cache_dir=cache_dir, resume_download=True, ) print("[Startup Pre-Cache] Breeze TTS 2 weights cached successfully!", flush=True) except Exception as e: print(f"[Startup Pre-Cache] Note: Breeze preload skipped or failed: {e}", flush=True) else: print("[Startup Pre-Cache] Breeze TTS 2 disabled (qwen-tts deps absent; see _breeze_available).", flush=True) if ( os.environ.get("HF_SKIP_PRELOAD") != "1" and os.environ.get("PYTEST_CURRENT_TEST") is None and os.environ.get("CI") is None ): import threading threading.Thread(target=_preload_heavy_models_on_cpu, daemon=True).start() # HF token from Space secrets HF_TOKEN = os.environ.get("HF_TOKEN", None) # Inference API client api_client = InferenceClient(token=HF_TOKEN) # ZeroGPU model state _llm = None _loaded_file = None # ZeroGPU billable lease budget. Hugging Face denies any lease larger than the # remaining daily allowance, so the requested duration is clamped to whatever # headroom is actually left instead of always asking for the full window. GPU_LEASE_DEFAULT_SECONDS = int(os.environ.get("FLOW_GPU_LEASE_DEFAULT_S", "120")) GPU_LEASE_MIN_SECONDS = int(os.environ.get("FLOW_GPU_LEASE_MIN_S", "20")) GPU_LEASE_SAFETY_SECONDS = int(os.environ.get("FLOW_GPU_LEASE_SAFETY_S", "5")) def quota_ledger_path() -> "str | None": """Path of the persisted ZeroGPU quota ledger. The allowance belongs to the account and survives Space restarts, so the ledger must too: an in-memory counter silently reported a full ``40 min`` after every reload while Hugging Face was already refusing leases. Returns ``None`` under pytest or when no writable location exists, which keeps unit tests hermetic and the tracker usable without persistent disk. """ if os.environ.get("PYTEST_CURRENT_TEST"): return None override = os.environ.get("FLOW_QUOTA_LEDGER_PATH") if override: return override for candidate in ("/data", BASE_DIR): try: if candidate and os.path.isdir(candidate) and os.access(candidate, os.W_OK): return os.path.join(candidate, "zerogpu_quota_ledger.json") except OSError: continue return None class GPUUsageTracker: """In-memory ZeroGPU quota telemetry (GOAL.md Milestone 3, line 54). Tracks wall-clock GPU-lease seconds per call, queue wait, cold-start (model load inside a lease), daily rollups, and quota-exhaustion (HTTP 429) state. ZeroGPU bills the whole lease duration, so elapsed wall time of each @spaces.GPU call is the honest accounting unit. The ledger is persisted through ``state_path`` because the allowance is an account-level, 24-hour window that outlives any single Space process: an in-memory-only counter reported a full 40 minutes after every reload while the platform was already refusing leases. ``state_path=None`` keeps the tracker purely in-memory (unit tests, read-only filesystems). """ DAILY_QUOTA_SECONDS = 40 * 60 # 40 min/day PRO budget (ZeroGPU) # A quota error sticks only for this long; after the TTL the flag expires # because a request is always allowed to re-probe the platform. HF documents # the allowance as "daily" and it resets 24h after the first GPU usage, so a # refusal is evidence about the present moment, not about the whole window. QUOTA_FLAG_TTL_SECONDS = int( os.environ.get("FLOW_QUOTA_FLAG_TTL_S", "2700") ) WINDOW_SECONDS = 24 * 60 * 60 # HF: quota resets 24h after first GPU usage def __init__(self, clock=time.time, state_path: "str | None" = None): self._clock = clock self._calls: List[Dict[str, Any]] = [] self._active: Dict[str, float] = {} self._last_quota_error: Dict[str, Any] = {} self._state_path = state_path self._window_started_at: "float | None" = None self._load_state() # -- persistence --------------------------------------------------- def _load_state(self) -> None: """Restore the window anchor and today's billed seconds from disk.""" if not self._state_path: return try: with open(self._state_path, "r", encoding="utf-8") as fh: state = json.load(fh) except (OSError, ValueError): return if not isinstance(state, dict): return started = state.get("window_started_at") if isinstance(started, (int, float)) and started > 0: self._window_started_at = float(started) restored = state.get("calls") if isinstance(restored, list): for record in restored: if isinstance(record, dict) and "lease_seconds" in record: self._calls.append(record) last_error = state.get("last_quota_error") if isinstance(last_error, dict) and last_error.get("ts"): self._last_quota_error = last_error self._roll_window_if_elapsed() def _save_state(self) -> None: """Persist the ledger. Best-effort: telemetry must never break a call.""" if not self._state_path: return payload = { "updated_at": int(self._clock()), "window_started_at": self._window_started_at, "window_resets_at": self.window_resets_at(), "calls": self._calls[-500:], "last_quota_error": self._last_quota_error or None, } try: tmp = f"{self._state_path}.tmp" with open(tmp, "w", encoding="utf-8") as fh: json.dump(payload, fh) os.replace(tmp, self._state_path) except OSError: pass # -- quota window -------------------------------------------------- def _roll_window_if_elapsed(self) -> None: """Drop records from an elapsed 24h window so the counter stays honest.""" resets_at = self.window_resets_at() if resets_at is None or self._clock() < resets_at: return self._calls = [] self._last_quota_error = {} self._window_started_at = None def window_resets_at(self) -> "float | None": if self._window_started_at is None: return None return self._window_started_at + self.WINDOW_SECONDS # -- lease lifecycle --------------------------------------------- def lease_start(self, label: str) -> str: self._roll_window_if_elapsed() if self._window_started_at is None: # HF anchors the 24h window on the first GPU usage, not midnight. self._window_started_at = self._clock() call_id = f"{label}-{int(self._clock() * 1000)}-{len(self._calls)}" self._active[call_id] = self._clock() return call_id def lease_end(self, call_id: str, *, model: str = "", kind: str = "chat", queue_wait_s: float = 0.0, cold_start_s: float = 0.0, completion_tokens: int = 0, error: str = "") -> Dict[str, Any]: started = self._active.pop(call_id, None) if started is None: return {} lease_s = round(max(0.0, self._clock() - started), 3) record = { "kind": kind, "model": model, "lease_seconds": lease_s, "queue_wait_seconds": round(queue_wait_s, 3), "cold_start_seconds": round(cold_start_s, 3), "completion_tokens": int(completion_tokens), "error": error[:200], "ts": int(self._clock()), } self._calls.append(record) if error and ("limit" in error.lower() or "quota" in error.lower()): self._last_quota_error = {"ts": record["ts"], "detail": error[:200]} self._save_state() return record # -- rollup -------------------------------------------------------- def _day_key(self) -> str: return time.strftime("%Y-%m-%d", time.gmtime(self._clock())) def daily_usage(self) -> Dict[str, Any]: self._roll_window_if_elapsed() today = self._day_key() day_calls = [ c for c in self._calls if time.strftime("%Y-%m-%d", time.gmtime(c["ts"])) == today ] total_s = round(sum(c["lease_seconds"] for c in day_calls), 1) ok_calls = [c for c in day_calls if not c["error"]] quota_error_fresh = ( self._last_quota_error and self._clock() - self._last_quota_error["ts"] < self.QUOTA_FLAG_TTL_SECONDS ) # An observed refusal from the platform outranks this ledger: the ledger # only knows leases this process watched, so it can claim minutes are # left while Hugging Face is already denying every lease. Reporting the # refusal as zero headroom is what stops a client from showing # "31/40 min" next to a 429. ledger_remaining = max( 0.0, float(self.DAILY_QUOTA_SECONDS) - total_s ) reported_remaining = ( 0.0 if quota_error_fresh else round(ledger_remaining, 1) ) return { "date": today, "calls_total": len(day_calls), "calls_ok": len(ok_calls), "calls_failed": len(day_calls) - len(ok_calls), "gpu_seconds_used": total_s, "gpu_minutes_used": round(total_s / 60, 2), "daily_quota_minutes": self.DAILY_QUOTA_SECONDS // 60, "quota_consumed_percent": round( total_s / self.DAILY_QUOTA_SECONDS * 100, 1 ), "quota_remaining_seconds": reported_remaining, "quota_remaining_minutes": round(reported_remaining / 60, 2), "quota_headroom_known": not quota_error_fresh, "quota_exhausted_observed": bool(quota_error_fresh), "quota_window_started_at": ( int(self._window_started_at) if self._window_started_at else None ), "quota_window_resets_at": ( int(resets_at) if (resets_at := self.window_resets_at()) else None ), "quota_window_seconds_remaining": ( max(0, int(resets_at - self._clock())) if (resets_at := self.window_resets_at()) else self.WINDOW_SECONDS ), "ledger_persisted": bool(self._state_path), "last_quota_error": self._last_quota_error or None, } def summary(self) -> Dict[str, Any]: """Aggregate latency/throughput profile across recorded calls.""" ok = [c for c in self._calls if not c["error"]] chat_calls = [c for c in ok if c["kind"] == "chat"] total_comp_tokens = sum(c.get("completion_tokens", 0) for c in chat_calls) total_lease = sum(c["lease_seconds"] for c in chat_calls) or 1e-9 cold_starts = [c["cold_start_seconds"] for c in ok if c["cold_start_seconds"] > 0] queue_waits = [c["queue_wait_seconds"] for c in ok if c["queue_wait_seconds"] > 0] return { "total_calls": len(self._calls), "total_lease_seconds": round(sum(c["lease_seconds"] for c in self._calls), 1), "avg_queue_wait_seconds": ( round(sum(queue_waits) / len(queue_waits), 3) if queue_waits else 0.0 ), "cold_start_seconds": round(max(cold_starts), 3) if cold_starts else 0.0, "chat_throughput_tokens_per_gpu_second": ( round(total_comp_tokens / total_lease, 2) if chat_calls else 0.0 ), } def used_seconds_today(self) -> float: """Billed lease seconds recorded so far in the current UTC quota day.""" today = self._day_key() return round( sum( c["lease_seconds"] for c in self._calls if time.strftime("%Y-%m-%d", time.gmtime(c["ts"])) == today ), 3, ) def remaining_seconds(self) -> float: """Daily quota headroom this tracker can still afford to lease. ZeroGPU bills the whole requested lease, and Hugging Face refuses a lease that exceeds the remaining allowance. Tracking actual lease seconds (not attempts) is what makes the number comparable to the platform ceiling. """ self._roll_window_if_elapsed() return max(0.0, float(self.DAILY_QUOTA_SECONDS) - self.used_seconds_today()) def usage_snapshot(self) -> Dict[str, Any]: return {"daily": self.daily_usage(), "profile": self.summary()} GPU_USAGE = GPUUsageTracker(state_path=quota_ledger_path()) class ZeroGpuQuotaError(RuntimeError): """Raised when the Space cannot lease enough GPU time for the request.""" def __init__(self, message: str, *, requested_seconds: int = 0, remaining_seconds: float = 0.0): super().__init__(message) self.requested_seconds = int(requested_seconds) self.remaining_seconds = float(remaining_seconds) def requested_gpu_lease_seconds() -> int: """Largest ZeroGPU lease the remaining daily quota can accommodate. Hugging Face bills the whole requested lease and denies any request that does not fit the remaining allowance, so a fixed 120 s request is refused outright once the day is nearly spent โ€” even though a short completion would comfortably fit. Clamping to the remaining headroom lets a genuine request keep working instead of failing on a number the model never used. A floor is kept so a near-exhausted day still attempts one real lease: the platform is the source of truth about the rolling window, and a denied lease does not consume quota. """ remaining = GPU_USAGE.remaining_seconds() affordable = int(remaining - GPU_LEASE_SAFETY_SECONDS) return max(GPU_LEASE_MIN_SECONDS, min(GPU_LEASE_DEFAULT_SECONDS, affordable)) def _lease_duration(*_args: Any, **_kwargs: Any) -> int: """Dynamic @spaces.GPU duration: recomputed for every invocation.""" return requested_gpu_lease_seconds() def gpu_tracked_call(kind: str, fn, *args, model: str = "", **kwargs): """Run a ZeroGPU-leased callable with lease + queue accounting. The work runs inside the @spaces.GPU lease body, so submit->done wall time approximates the billed lease duration. Server-side queue wait is not observable; the recorded queue_wait_seconds is an upper bound. """ t_submit = time.time() call_id = GPU_USAGE.lease_start(kind) lease_s = requested_gpu_lease_seconds() @spaces.GPU(duration=_lease_duration) def _tracked_inner(): # Residency must be evaluated INSIDE the ZeroGPU worker: module # globals set there (_llm/_loaded_file via get_model) never propagate # back to the web process, so cold-start state is invisible outside. resident_before = _llm is not None and _loaded_file == model t_body_start = time.time() try: result = fn(*args, **kwargs) error = "" except Exception as exc: result, error = None, str(exc) if not resident_before and _llm is not None and _loaded_file == model: # The body loaded the model inside this lease. The load share is # not directly observable, so the whole post-body span is the # cold-start upper bound. cold_start_s = max(0.0, time.time() - t_body_start) else: cold_start_s = 0.0 return result, error, t_body_start, cold_start_s try: result, error, t_body_start, cold_start_s = _tracked_inner() except Exception as exc: # Lease-level failure (e.g. 429 quota exhaustion): body never ran. GPU_USAGE.lease_end(call_id, model=model, kind=kind, error=str(exc)) raise queue_wait = max(0.0, t_body_start - t_submit) comp_tokens = 0 try: usage = result.get("usage") if isinstance(result, dict) else None if isinstance(usage, dict) and usage.get("completion_tokens"): comp_tokens = int(usage["completion_tokens"]) except Exception: comp_tokens = 0 GPU_USAGE.lease_end( call_id, model=model, kind=kind, queue_wait_s=queue_wait, cold_start_s=cold_start_s, completion_tokens=comp_tokens, error=error, ) if error: raise RuntimeError(error) return result def find_model_path(model_file: str) -> str: """Find absolute path of a GGUF model file across search directories.""" for d in SEARCH_DIRS: p = os.path.join(d, model_file) if os.path.isfile(p): return p return None def ensure_model_available(model_file: str) -> str: """Download a supported model on CPU, outside any ZeroGPU lease. A 17 GB GGUF takes far longer to fetch than the 120 s GPU lease, so the download must never happen inside the inference body: a cold first call used to time out and burn the daily quota. Callers run this before entering the lease. """ for d in SEARCH_DIRS: p = os.path.join(d, model_file) if os.path.isfile(p): return p source = MODEL_HUB_SOURCES.get(model_file) if source and os.path.isdir("/data"): try: return hf_hub_download( repo_id=source["repo_id"], filename=source["filename"], local_dir="/data", token=HF_TOKEN, ) except Exception as exc: print(f"Model download failed: {type(exc).__name__}: {exc}", flush=True) return None def list_gguf_files(): """List explicitly supported model files.""" found = [] for d in SEARCH_DIRS: if os.path.isdir(d): try: for name in sorted(os.listdir(d)): if name in SUPPORTED_MODELS and name not in found: found.append(name) except Exception: pass for model_file in MODEL_HUB_SOURCES: if model_file not in found: found.append(model_file) return found def model_choices(): """Return choices for UI dropdowns.""" found = list_gguf_files() for m in SUPPORTED_MODELS: if m not in found: found.append(m) return found def resolve_model(model_req: str, choices=None) -> str: """Resolve an API model ID by exact filename or documented safe alias.""" choices = choices or list_gguf_files() aliases = { "premium": "Qwen3.8-9B-Q8_0.gguf", "paid-premium": "Qwen3.8-9B-Q8_0.gguf", "default": "Qwen3.8-9B-Q8_0.gguf", "qwen-9b": "Qwen3.8-9B-Q8_0.gguf", "qwen-9b-q8": "Qwen3.8-9B-Q8_0.gguf", "qwen-9b-q4": "Qwen3.8-9B-Q4_K_M.gguf", "qwen3.8-9b": "Qwen3.8-9B-Q8_0.gguf", "qwen": "Qwen3.8-27B-Q4_K_M.gguf", "qwen-27b": "Qwen3.8-27B-Q4_K_M.gguf", "qwen3.8": "Qwen3.8-27B-Q4_K_M.gguf", "qwen3.8-27b": "Qwen3.8-27B-Q4_K_M.gguf", "qwen-fast": "Qwen3.8-9B-Q8_0.gguf", "qwen-q4": "Qwen3.8-27B-Q4_K_M.gguf", "qwen-q4_k_m": "Qwen3.8-27B-Q4_K_M.gguf", "qwen-q6": "Qwen3.8-27B-Q6_K.gguf", "qwen-q6_k": "Qwen3.8-27B-Q6_K.gguf", "qwen-uncensored": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", "qwen-heretic": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", "qwen3.8-uncensored": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", "qwen-turbo": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", "qwen-turbo-coder": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", "davidau/qwen3.8-27b-turbo-fable-cold-fusion": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", "qwen-9b-distill-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", "qwen-9b-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", "qwen3.8-9b-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", "petruhonk/qwen3.8-9b-distill-uncensored-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", "petruhonk/qwen3.8-9b-distill-uncensored-heretic-gguf": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", "qwen3.8-9b-distill-heretic-f16": "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf", "deepseek-v4": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-v4-pro": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-v4-9b": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-v4-pro-qwen3.5-9b": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-v4-pro-qwen3.5-9b-mtp": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-mtp": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-mtp-9b": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "jackrong/deepseek-v4-pro-qwen3.5-9b-mtp": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "jackrong/deepseek-v4-pro-qwen3.5-9b-mtp-gguf": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-v4-q4": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", "deepseek-v4-q8": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf", "deepseek-v4-q6": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf", "deepseek-v4-q5": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf", "deepseek-v4-bf16": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf", "deepseek-v4-iq4": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf", "gemma-q4": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", "gemma-q4_k_m": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", "gemma-q6": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", "gemma-q6_k": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", } requested = (model_req or "").strip() candidate = aliases.get(requested.lower(), requested) if candidate in choices: return candidate if not requested: return choices[0] raise HTTPException( status_code=400, detail=f"Unknown model '{requested}'. Use GET /v1/models for live model IDs.", ) def get_model(model_file: str) -> Llama: """Load a GGUF model with clean memory management and FlashAttention.""" global _llm, _loaded_file # Downloads are CPU work; run them before the GPU lease when a caller # bypasses the endpoint-level pre-download. model_path = find_model_path(model_file) if not model_path: model_path = ensure_model_available(model_file) if not model_path: avail = list_gguf_files() raise gr.Error( f"Model file '{model_file}' not found in /data or /models. " f"Available files: {avail}. Please upload your .gguf model to the mounted bucket." ) if _llm is not None and _loaded_file == model_file: return _llm # Free previous model from memory before loading new one if _llm is not None: try: del _llm except Exception: pass _llm = None gc.collect() if "Q6" in model_file: default_target_ctx = 32768 elif "9B" in model_file: default_target_ctx = 163840 else: default_target_ctx = 163840 target_ctx = int(os.environ.get("FLOW_N_CTX", str(default_target_ctx))) context_candidates = [target_ctx] for fallback in [163840, 131072, 65536, 32768, 16384, 8192]: if fallback not in context_candidates and fallback < target_ctx: context_candidates.append(fallback) last_error = None for ctx in context_candidates: for use_fa in [True, False]: try: gc.collect() kwargs = { "model_path": model_path, "n_ctx": ctx, "n_gpu_layers": -1, "n_batch": 2048, "n_ubatch": 512, "verbose": False, } if use_fa: kwargs["flash_attn"] = True print(f"Loading {model_file} with n_ctx={ctx}, flash_attn={use_fa}...", flush=True) _llm = Llama(**kwargs) _loaded_file = model_file print(f"Successfully loaded {model_file} with n_ctx={ctx}, flash_attn={use_fa}!", flush=True) return _llm except TypeError: continue except Exception as e: last_error = e print(f"Failed loading with n_ctx={ctx}, flash_attn={use_fa}: {e}", flush=True) if _llm is not None: try: del _llm except Exception: pass _llm = None gc.collect() break if _llm is None and last_error: raise last_error return _llm def parse_model_tool_calls(text: str): """Extract structured OpenAI tool calls and thinking from model output.""" if not text: return None, None, "" tool_calls = [] clean_text = text # Pattern 1: Qwen XML style v. # Do not require spaces around the tags. Qwen templates normally emit # newlines (or no whitespace at all), while the original expression only # matched a very specific, space-padded variant. xml_matches = list(re.finditer( r'\s*\s*\s*(.*?)\s*\s*', clean_text, re.DOTALL, )) if xml_matches: for idx, m in enumerate(xml_matches): fn_name = m.group(1).strip() fn_body = m.group(2) args = {} for p in re.finditer(r'(.*?)', fn_body, re.DOTALL): p_name = p.group(1).strip() p_val = p.group(2).strip() try: args[p_name] = json.loads(p_val) except Exception: args[p_name] = p_val tool_calls.append({ "index": idx, "id": f"call_{uuid.uuid4().hex[:8]}", "type": "function", "function": { "name": fn_name, "arguments": json.dumps(args, ensure_ascii=False) } }) clean_text = clean_text.replace(m.group(0), "") # Pattern 2: JSON style {"name": "...", "arguments": {...}} json_matches = list(re.finditer(r'\s*\s*(\{.*?\})\s*', clean_text, re.DOTALL)) if json_matches and not tool_calls: for idx, m in enumerate(json_matches): raw_json = m.group(1).strip() try: parsed = json.loads(raw_json) fn_name = parsed.get("name", "") fn_args = parsed.get("arguments", {}) args_str = json.dumps(fn_args, ensure_ascii=False) if isinstance(fn_args, dict) else str(fn_args) tool_calls.append({ "index": idx, "id": f"call_{uuid.uuid4().hex[:8]}", "type": "function", "function": { "name": fn_name, "arguments": args_str } }) clean_text = clean_text.replace(m.group(0), "") except Exception: pass # Extract thinking/reasoning if present reasoning_content = None think_m = re.search(r'\s*\s*(.*?)\s*', clean_text, re.DOTALL) if think_m: reasoning_content = think_m.group(1).strip() clean_text = clean_text.replace(think_m.group(0), "") elif "" in clean_text: parts = clean_text.split("", 1) reasoning_content = parts[0].replace("", "").strip() clean_text = parts[1] elif tool_calls and clean_text.strip(): # Any text preceding tool calls without tags is reasoning reasoning_content = clean_text.strip() clean_text = "" clean_content = clean_text.strip() if tool_calls and not clean_content: clean_content = None return (tool_calls if tool_calls else None), (reasoning_content if reasoning_content else None), clean_content def normalize_native_tool_calls(tool_calls): """Normalize llama-cpp native calls to the OpenAI response contract. Recent llama-cpp versions parse a GGUF chat template themselves and return ``message.tool_calls`` with an empty content field. Parsing only content silently discarded those valid calls and made the API look non-agentic. """ if not tool_calls: return None normalized = [] for index, call in enumerate(tool_calls): function = call.get("function", {}) if isinstance(call, dict) else {} name = function.get("name") or call.get("name", "") arguments = function.get("arguments", call.get("arguments", {})) if not isinstance(arguments, str): arguments = json.dumps(arguments, ensure_ascii=False) normalized.append({ "index": call.get("index", index), "id": call.get("id") or f"call_{uuid.uuid4().hex[:8]}", "type": "function", "function": {"name": name, "arguments": arguments}, }) return normalized def format_openai_messages_for_model(messages): """Normalize multi-turn OpenAI messages including tool results into prompt format.""" formatted = [] for msg in messages: role = msg.get("role", "user") content = msg.get("content") tool_calls = msg.get("tool_calls") if role == "tool": formatted.append({ "role": "user", "content": f"\n{content or ''}\n" }) elif role == "assistant" and tool_calls: tc_text = "" for tc in tool_calls: fn = tc.get("function", {}) fn_name = fn.get("name", "") raw_args = fn.get("arguments", "{}") try: args_dict = json.loads(raw_args) if isinstance(raw_args, str) else raw_args except Exception: args_dict = {} tc_text += f"\n\n\n" if isinstance(args_dict, dict): for k, v in args_dict.items(): tc_text += f"\n{json.dumps(v) if isinstance(v, (dict, list)) else v}\n\n" tc_text += "\n" combined = (content or "") + tc_text formatted.append({"role": "assistant", "content": combined.strip()}) else: formatted.append({"role": role, "content": content or ""}) return formatted def _format_api_error(e: Exception, action: str) -> str: """Format API errors with clear budget and credit guidance.""" msg = str(e) if "402" in msg or "Payment Required" in msg: return ( f"{action} notice (402 Payment Required): Your monthly HF Inference API credit " "($2/month included with HF PRO) has been fully used for this billing period. " "ZeroGPU tabs (Qwen 3.8 / Gemma LLM Chat) remain 100% free and functional!" ) return f"{action} failed: {msg}" # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # TAB 1: Chat & Agent Runner (ZeroGPU Large โ€” 40 min/day) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” def _raw_generate_openai_chat(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None): """Raw chat completion worker without @spaces.GPU decorator (avoids nested GPU leases in gpu_tracked_call).""" llm = get_model(model_file) kwargs = { "messages": messages, "max_tokens": (int(max_tokens) if max_tokens not in (None, "") else None), "temperature": float(temperature), "top_p": 0.95, } if tools: kwargs["tools"] = tools kwargs["tool_choice"] = tool_choice or "auto" return llm.create_chat_completion(**kwargs) @spaces.GPU(size="large", duration=_lease_duration) def generate_openai_chat(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None): return _raw_generate_openai_chat(messages, model_file, temperature, max_tokens, tools=tools, tool_choice=tool_choice) @spaces.GPU(size="large", duration=_lease_duration) def generate_openai_chat_stream(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None): llm = get_model(model_file) kwargs = { "messages": messages, "max_tokens": (int(max_tokens) if max_tokens not in (None, "") else None), "temperature": float(temperature), "top_p": 0.95, "stream": True, } if tools: kwargs["tools"] = tools kwargs["tool_choice"] = tool_choice or "auto" for chunk in llm.create_chat_completion(**kwargs): yield chunk def custom_chat_handler(user_msg, history, model_file, system_prompt, temperature, max_tokens): """Rich chat execution with live token telemetry, reasoning, and speed reporting.""" if not user_msg or not user_msg.strip(): return history or [], "โšก *Ready โ€” Enter a prompt to start inference.*", "" history = list(history or []) history.append({"role": "user", "content": user_msg.strip()}) messages = [] if system_prompt.strip(): messages.append({"role": "system", "content": system_prompt.strip()}) for item in history: messages.append({"role": item.get("role", "user"), "content": item.get("content", "")}) t0 = time.time() try: raw_res = gpu_tracked_call( "chat", _raw_generate_openai_chat, messages, model_file, float(temperature), (int(max_tokens) if max_tokens not in (None, "") else None), model=model_file, ) t1 = time.time() elapsed = max(0.01, t1 - t0) raw_content = raw_res["choices"][0]["message"].get("content") or "" tool_calls, reasoning, clean = parse_model_tool_calls(raw_content) formatted_bot = "" if reasoning: formatted_bot += f"
๐Ÿง  Deep Thinking & Reasoning\n\n```markdown\n{reasoning}\n```\n
\n\n" if tool_calls: formatted_bot += f"
๐Ÿ› ๏ธ Executed Tool Calls ({len(tool_calls)})\n\n```json\n{json.dumps(tool_calls, indent=2)}\n```\n
\n\n" if clean: formatted_bot += clean elif not reasoning and not tool_calls: formatted_bot += raw_content history.append({"role": "assistant", "content": formatted_bot}) # Telemetry calculations raw_usage = raw_res.get("usage", {}) prompt_toks = raw_usage.get("prompt_tokens") or sum(max(1, int(len(m["content"].split()) * 1.3)) for m in messages) comp_toks = raw_usage.get("completion_tokens") or max(1, int(len(raw_content.split()) * 1.3)) tot_toks = prompt_toks + comp_toks tps = comp_toks / elapsed ctx_pct = (tot_toks / 262144) * 100 hud_md = ( f"
" f"โšก {tps:.1f} t/s" f"โฑ๏ธ {elapsed:.2f}s" f"๐Ÿ“ฅ Prompt: {prompt_toks}" f"๐Ÿ“ค Output: {comp_toks}" f"๐Ÿง  Context: {tot_toks:,} / 262,144 ({ctx_pct:.1f}%)" f"โ— ZeroGPU Large (48GB)" f"
" ) return history, hud_md, "" except Exception as e: history.append({"role": "assistant", "content": f"โŒ **Inference Error:** {str(e)}"}) return history, f"โš ๏ธ *Execution error after {time.time()-t0:.2f}s: {str(e)}*", "" # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # TAB 2: Vision & Multimodal OCR # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” def analyze_vision(image_input, prompt_text): """Analyze images, UI screenshots, code diagrams or documents.""" if image_input is None: raise gr.Error("Please upload or capture an image first.") if not HF_TOKEN: raise gr.Error("Set HF_TOKEN in Space secrets to use the Vision API.") prompt = prompt_text.strip() or "Describe this image in detail and extract all visible text and code." try: response = api_client.chat_completion( messages=[ { "role": "user", "content": [ {"type": "text", "text": prompt}, {"type": "image_url", "image_url": {"url": image_input if isinstance(image_input, str) else image_input}}, ], } ], model=VISION_MODEL, max_tokens=1024, ) return response.choices[0].message.content except Exception as e: try: return api_client.image_to_text(image=image_input, model="Salesforce/blip-image-captioning-large") except Exception: raise gr.Error(_format_api_error(e, "Vision analysis")) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # ZERO-GPU VIDEO GENERATION PIPELINE (40 min/day A100 Quota - $0 API Cost) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # Video diffusion pipelines are heavy: CogVideoX-5B / LTX-Video weigh 6-15 GB. # Weights are DOWNLOADED OUTSIDE the GPU lease (CPU-only snapshot_download in # the web worker), then only the denoise step leases GPU. A dynamic 120 s # lease cannot absorb a cold 10 GB download โ€” it aborts with "GPU task # aborted" mid-load. Warm-pipeline inference fits comfortably in 240 s. VIDEO_GPU_LEASE_SECONDS = int(os.environ.get("FLOW_VIDEO_GPU_LEASE_S", "240")) def _prefetch_video_pipeline(repo_id: str) -> None: """Download pipeline weights into the HF cache with NO GPU lease active. ZeroGPU bills only GPU-leased code; downloads and CPU work in the web process are free. A snapshot that fails parsing (truncated spiece.model etc.) never self-heals, so it is purged and re-downloaded once. """ from huggingface_hub import snapshot_download try: snapshot_download(repo_id) except Exception as exc: import re as _re m = _re.search(r"models--([\w.\-]+)--([\w.\-]+)", str(exc)) if m and ("Error parsing" in str(exc) or "tokenize" in str(exc).lower()): broken = os.path.join("/root/.cache/huggingface/hub", f"models--{m.group(1)}--{m.group(2)}") if os.path.exists(broken): print(f"[ZeroGPU Video] Purging corrupted HF cache: {broken}", flush=True) shutil.rmtree(broken, ignore_errors=True) snapshot_download(repo_id) else: raise @spaces.GPU(duration=VIDEO_GPU_LEASE_SECONDS) def generate_zerogpu_video( prompt: str, negative_prompt: str = "", model_choice: str = "ZeroScope v2 (576w High-Res)", num_frames: int = 16, fps: int = 8, guidance_scale: float = 7.5, seed: int = -1 ): """Generate dynamic MP4 video using ZeroGPU open-weights models.""" if not prompt.strip(): raise gr.Error("Please enter a video prompt.") repo_id = VIDEO_MODELS.get(model_choice, "cerspense/zeroscope_v2_576w") out_video_path = f"/tmp/zerogpu_video_{int(time.time())}_{abs(hash(prompt)) % 10000}.mp4" # Weights must be resident BEFORE the GPU lease starts; loading inside a # short lease aborts on any cold download. _prefetch_video_pipeline(repo_id) try: import torch from diffusers import DiffusionPipeline, DPMSolverMultistepScheduler device = "cuda" if torch.cuda.is_available() else "cpu" dtype = torch.float16 if device == "cuda" else torch.float32 if repo_id not in _video_pipeline_cache: pipe = DiffusionPipeline.from_pretrained(repo_id, torch_dtype=dtype) if hasattr(pipe, "scheduler"): try: pipe.scheduler = DPMSolverMultistepScheduler.from_config(pipe.scheduler.config) except Exception: pass if hasattr(pipe, "enable_model_cpu_offload") and device == "cuda": pipe.enable_model_cpu_offload() else: pipe = pipe.to(device) _video_pipeline_cache[repo_id] = pipe else: pipe = _video_pipeline_cache[repo_id] actual_seed = seed if (seed and int(seed) >= 0) else random.randint(0, 2**31 - 1) generator = torch.Generator(device=device).manual_seed(actual_seed) video_frames = pipe( prompt=prompt.strip(), negative_prompt=negative_prompt.strip() if negative_prompt else None, num_inference_steps=24, guidance_scale=float(guidance_scale), num_frames=int(num_frames), generator=generator ).frames[0] import numpy as np import tempfile from PIL import Image ffmpeg_bin = shutil.which("ffmpeg") or "/usr/bin/ffmpeg" with tempfile.TemporaryDirectory() as tmpdir: for idx, frame in enumerate(video_frames): if isinstance(frame, np.ndarray): f_arr = frame if f_arr.dtype in (np.float32, np.float64, np.float16): if f_arr.max() <= 1.0: f_arr = (f_arr * 255.0).clip(0, 255).astype(np.uint8) else: f_arr = f_arr.clip(0, 255).astype(np.uint8) img = Image.fromarray(f_arr) else: img = frame img.save(os.path.join(tmpdir, f"frame_{idx:05d}.png")) subprocess.run( [ ffmpeg_bin, "-y", "-framerate", str(int(fps)), "-i", os.path.join(tmpdir, "frame_%05d.png"), "-c:v", "libx264", "-pix_fmt", "yuv420p", out_video_path ], check=True, capture_output=True ) return out_video_path except Exception as exc: print(f"[ZeroGPU Video] Direct pipeline exception: {exc}, running ffmpeg dynamic visualizer fallback...", flush=True) try: from engine.model_dispatcher import ModelDispatcher disp = ModelDispatcher() res = disp.generate_image(prompt=prompt, aspect_ratio="16:9") img_path = res.get("filepath") if img_path and os.path.exists(img_path): ffmpeg_bin = shutil.which("ffmpeg") or "/opt/homebrew/bin/ffmpeg" dur = max(3, int(int(num_frames) / max(1, int(fps)))) subprocess.run( [ ffmpeg_bin, "-y", "-loop", "1", "-i", img_path, "-vf", f"fps={fps},scale=768:432,zoompan=z='min(zoom+0.0015,1.15)':d={dur*fps}:s=768x432", "-c:v", "libx264", "-t", str(dur), "-pix_fmt", "yuv420p", out_video_path ], capture_output=True, timeout=20 ) if os.path.exists(out_video_path) and os.path.getsize(out_video_path) > 0: return out_video_path except Exception as e2: print(f"[ZeroGPU Video] Fallback failed: {e2}") raise gr.Error(f"ZeroGPU Video error: {exc}") # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # ZERO-GPU MUSIC & AUDIO GENERATION PIPELINE (40 min/day A100 Quota - $0 Cost) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” def _raw_generate_zerogpu_music( prompt: str, lyrics: str = "", model_choice: str = "MiniMax Music 3 (full song + vocals)", duration_seconds: int = 60, guidance_scale: float = 3.0, temperature: float = 1.0, seed: int = 7 ): """Internal raw worker for ZeroGPU audio synthesis (avoids nested @spaces.GPU calls).""" if not prompt.strip() and not lyrics.strip(): raise gr.Error("Please enter a music description or lyrics.") repo_id = AUDIO_MUSIC_MODELS.get(model_choice, "MiniMaxAI/MiniMax-Music3") # Clamp to max 120 seconds to guarantee completion within ZeroGPU quota lease dur = max(5, min(120, int(duration_seconds or 60))) out_audio_path = f"/tmp/zerogpu_music_{int(time.time())}_{abs(hash(prompt + lyrics)) % 10000}.wav" try: import torch import soundfile as sf # Official MiniMax Music3 path from its model card. It supports lyrics + # detailed music description and produces complete vocal songs up to 5 min. if repo_id == "MiniMaxAI/MiniMax-Music3": from diffusers import ModularPipeline import numpy as np if repo_id not in _audio_pipeline_cache: pipe = ModularPipeline.from_pretrained(repo_id) pipe.load_components(dtype=torch.bfloat16) pipe.to("cuda") _audio_pipeline_cache[repo_id] = pipe else: pipe = _audio_pipeline_cache[repo_id] raw_output = pipe( prompt=prompt.strip(), lyrics=lyrics.strip(), audio_duration=float(dur), generator=torch.Generator("cuda").manual_seed(int(seed)), output="audios", ) audio = raw_output[0] if hasattr(audio, "detach"): audio_np = audio.detach().cpu().float().numpy() elif isinstance(audio, np.ndarray): audio_np = audio.astype(np.float32) else: audio_np = np.asarray(audio, dtype=np.float32) if audio_np.ndim > 1 and audio_np.shape[0] < audio_np.shape[1]: audio_np = audio_np.T sr = getattr(pipe, "sampling_rate", 44100) sf.write(out_audio_path, audio_np, sr) return out_audio_path # YuE2-3B: AR-NAR MoT + flow matching, 48kHz stereo, editable scores. # The repo is NOT a diffusers pipeline (no model_index.json); it ships the # `yue2` infer wheel (YuE2Pipeline). Install with --no-deps (its pins for # torch/hub conflict with the Space's diffusers set; core only needs # torch + transformers + tiktoken + soundfile). if repo_id == "m-a-p/YuE2-3B": import numpy as np from yue2 import YuE2Pipeline if repo_id not in _audio_pipeline_cache: pipe = YuE2Pipeline.from_pretrained(repo_id, progress=False) _audio_pipeline_cache[repo_id] = pipe else: pipe = _audio_pipeline_cache[repo_id] song = pipe(style=prompt.strip(), lyrics=lyrics.strip(), cot="full", seed=int(seed)) sf.write(out_audio_path, song.audio, song.sample_rate) return out_audio_path # ACE-Step v1.5 XL Turbo: Apache 2.0, 48kHz, commercial-safe, <4GB VRAM. # Output is AudioPipelineOutput.audios (channels, samples); rate is # pipe.sample_rate. Turbo is guidance-distilled so CFG is ignored anyway. if repo_id == "ACE-Step/acestep-v15-xl-turbo-diffusers": from diffusers import AceStepPipeline import numpy as np if repo_id not in _audio_pipeline_cache: pipe = AceStepPipeline.from_pretrained(repo_id, torch_dtype=torch.bfloat16) pipe.to("cuda") pipe.vae.enable_tiling() # bound VAE decode memory for longer clips _audio_pipeline_cache[repo_id] = pipe else: pipe = _audio_pipeline_cache[repo_id] raw_output = pipe( prompt=prompt.strip(), lyrics=lyrics.strip(), audio_duration=float(dur), generator=torch.Generator("cuda").manual_seed(int(seed)), ) audio_tensor = raw_output.audios[0] # (channels, samples) audio_np = audio_tensor.T.cpu().float().numpy() sf.write(out_audio_path, audio_np, pipe.sample_rate) return out_audio_path # Stable Audio 3 uses its own pipeline and may require HF access approval. if repo_id == "stabilityai/stable-audio-3-medium": # The public Stability release currently uses the separate # `stable_audio_3` package, not a Diffusers StableAudio3Pipeline. # Do not pretend this selector works or silently substitute audio. raise RuntimeError( "Stable Audio 3 is not enabled in this Space yet: its official " "stable_audio_3 runtime is not installed. Select MiniMax Music 3." ) # MusicGen is explicitly instrumental and does not reliably sing lyrics. if repo_id.startswith("facebook/musicgen"): from transformers import AutoProcessor, MusicgenForConditionalGeneration if repo_id not in _audio_pipeline_cache: processor = AutoProcessor.from_pretrained(repo_id) model = MusicgenForConditionalGeneration.from_pretrained(repo_id, torch_dtype=torch.float16).to("cuda") _audio_pipeline_cache[repo_id] = (processor, model) processor, model = _audio_pipeline_cache[repo_id] inputs = processor(text=[prompt.strip()], padding=True, return_tensors="pt").to("cuda") audio_values = model.generate( **inputs, do_sample=True, guidance_scale=float(guidance_scale), max_new_tokens=min(1500, int(dur * 50)), temperature=float(temperature) ) sf.write(out_audio_path, audio_values[0, 0].detach().cpu().numpy(), model.config.audio_encoder.sampling_rate) return out_audio_path raise gr.Error(f"Model '{repo_id}' is not wired for full-song generation yet. Choose MiniMax Music 3, YuE2-3B, or ACE-Step v1.5.") except Exception as exc: raise gr.Error(f"ZeroGPU model '{repo_id}' failed: {exc}. No MIDI/procedural fallback was used.") @spaces.GPU(duration=_lease_duration) def generate_zerogpu_music( prompt: str, lyrics: str = "", model_choice: str = "MiniMax Music 3 (full song + vocals)", duration_seconds: int = 60, guidance_scale: float = 3.0, temperature: float = 1.0, seed: int = 7 ): """Generate real full-song audio on ZeroGPU. Procedural MIDI fallback is never used here.""" return _raw_generate_zerogpu_music( prompt=prompt, lyrics=lyrics, model_choice=model_choice, duration_seconds=duration_seconds, guidance_scale=guidance_scale, temperature=temperature, seed=seed ) # --- Moldovan Creative Studio coupling removed (2026-08) --- # The Moldovan Creative Media & Personas tab lived here (create_moldovan_song_handler, # chat_moldovan_persona_handler, convert_dialect_handler) and drove lyric synthesis / # persona chat / dialect conversion via moldovan-qwen/ (engine.media_creator, # personas.persona_engine, linguistics.dialect_converter). Those engines now run # exclusively in the local Moldovan Media Studio (moldovan-qwen/studio_server.py, # :8090) and studio-pb (PocketBase :8096); the ZeroGPU Space no longer imports them. def transcribe_audio(audio_input): """Transcribe audio with Whisper Large v3.""" if audio_input is None: raise gr.Error("Please record or upload audio.") if not HF_TOKEN: raise gr.Error("Set HF_TOKEN in Space secrets to use Whisper.") try: result = api_client.automatic_speech_recognition( audio=audio_input, model=WHISPER_MODEL, ) return result.text if hasattr(result, "text") else str(result) except Exception as e: raise gr.Error(_format_api_error(e, "Whisper transcription")) def generate_tts(text_input): """Synthesize natural ro-RO speech with Edge-TTS Neural (no banned TTS models).""" if not text_input.strip(): raise gr.Error("Please enter text to synthesize.") try: import edge_tts import asyncio import tempfile, os as _os tmp = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) tmp.close() async def _run(): await edge_tts.Communicate(text_input.strip(), "ro-RO-EmilNeural").save(tmp.name) asyncio.run(_run()) with open(tmp.name, "rb") as f: return f.read() except Exception as e: raise gr.Error(f"Edge-TTS ro-RO synthesis error: {e}") # โ”€โ”€โ”€ Breeze TTS 2 (SOTA experimental; non-commercial license, art use) โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ BREEZE_LANGUAGE_HINTS = { "auto": "Detect the language and speak naturally with clear articulation.", "en": "Speak in English with a natural, engaging storyteller delivery.", "ro": "Speak clearly. Textul este รฎn limba romรขnฤƒ: articulare clarฤƒ, intonaศ›ie naturalฤƒ ศ™i expresivฤƒ.", } BREEZE_EMOTION_HINTS = { "neutral": "Neutral, balanced tone.", "warm": "Warm, friendly and inviting tone.", "energetic": "Energetic, upbeat and lively delivery.", "serious": "Serious, restrained and deliberate tone.", "sad": "Melancholic, soft and subdued delivery.", "excited": "Excited, joyful and high-energy delivery.", "narrator": "Cinematic narrator voice: deep, slow, suspenseful storytelling.", } def _resolve_breeze_instruction(language: str, emotion: str, custom_instruction: str) -> str: """Compose the natural-language voice-direction instruction for Breeze.""" if custom_instruction.strip(): return custom_instruction.strip() lang_hint = BREEZE_LANGUAGE_HINTS.get((language or "auto").lower(), BREEZE_LANGUAGE_HINTS["auto"]) emo_hint = BREEZE_EMOTION_HINTS.get((emotion or "neutral").lower(), BREEZE_EMOTION_HINTS["neutral"]) return f"{emo_hint} {lang_hint}" def _load_breeze_runtime(model_dir: Any): """Load (and cache across lease workers) the Breeze TTS runtime on the GPU worker.""" import torch if "runtime" in _breeze_runtime_cache: return _breeze_runtime_cache["runtime"] # Weight download must already be done by the startup CPU pre-cache thread. from breeze_infer.runtime import load_runtime, update_generation_config_for_breeze from models.fast_streaming import FastBreezeStreamingRuntime, FastStreamingConfig tokenizer, model, audio_tokenizer = load_runtime( model_dir, device="cuda" if torch.cuda.is_available() else "cpu", attn_implementation="eager", ) update_generation_config_for_breeze(model) config = FastStreamingConfig( max_new_tokens=1500, max_seq_len=2048, fast_all=False, # eager: no CUDA-graph warmup cost inside a short ZeroGPU lease repetition_penalty=1.1, ) runtime = FastBreezeStreamingRuntime(model, audio_tokenizer, config, tokenizer=tokenizer) _breeze_runtime_cache["runtime"] = runtime _breeze_runtime_cache["tokenizer"] = tokenizer _breeze_runtime_cache["model"] = model _breeze_runtime_cache["audio_tokenizer"] = audio_tokenizer return runtime @spaces.GPU(duration=_lease_duration) def generate_zerogpu_breeze_tts( text: str, instruction: str = "", language: str = "auto", emotion: str = "neutral", cfg_scale: float = 1.0, seed: int = 42, ref_audio_path: Optional[str] = None, ref_text: str = "" ): """Breeze TTS 2 synthesis on ZeroGPU. Returns path to generated 24kHz WAV. Voice design (no reference) or voice direction (with reference audio + transcript). """ if not _breeze_available(): raise gr.Error( "Breeze TTS 2 is temporarily disabled: its qwen-tts dependency pins " "transformers==4.57.3 which breaks MiniMax Music 3 (diffusers hub>=1.23). " "Use MiniMax Music 3 for music; Breeze returns when upstream unpins." ) if not text.strip(): raise gr.Error("Please enter text to synthesize.") out_path = f"/tmp/breeze_tts_{int(time.time())}_{abs(hash(text)) % 10000}.wav" try: from huggingface_hub import snapshot_download import numpy as np import soundfile as sf from breeze_infer.templates import get_template, prepare_inputs from breeze_infer.runtime import set_all_seeds # Resolve checkpoint directory from the persistent cache. cache_dir = "/data/hub" if (os.path.exists("/data") and os.access("/data", os.W_OK)) else os.environ.get("HF_HUB_CACHE", None) model_dir = snapshot_download(repo_id=BREEZE_MODEL_ID, cache_dir=cache_dir, resume_download=True) runtime = _load_breeze_runtime(model_dir) tokenizer = _breeze_runtime_cache["tokenizer"] audio_tokenizer = _breeze_runtime_cache["audio_tokenizer"] breeze_model = _breeze_runtime_cache["model"] request = { "id": "breeze-request", "text": text.strip(), "instruction": _resolve_breeze_instruction(language, emotion, instruction), "speaker": "S0", } template_name = "tts_instruction" if ref_audio_path: if not os.path.isfile(ref_audio_path): raise gr.Error(f"Reference audio not found: {ref_audio_path}") if not (ref_text or "").strip(): raise gr.Error("Reference transcript (exact text spoken in the audio) is required for voice cloning.") request["ref_audio_path"] = str(ref_audio_path) request["ref_text"] = ref_text.strip() template_name = "ref_edit_tata" effective_seed = int(seed) if seed is not None and int(seed) >= 0 else random.randint(0, 2**31 - 1) set_all_seeds(effective_seed) inputs = prepare_inputs( tokenizer, audio_tokenizer, breeze_model, [request], get_template(template_name), guidance_scale=float(cfg_scale) if cfg_scale and float(cfg_scale) > 0 else 1.0, guidance_scale_ref=None, guidance_scale_ins=None, ) chunks = [] for chunk in runtime.iter_audio_chunks(inputs, request_id="breeze-request", seed=effective_seed): chunks.append(np.asarray(chunk.audio)) if not chunks: raise gr.Error("Breeze synthesis produced no audio.") audio = np.concatenate(chunks).astype(np.float32) sf.write(out_path, audio, samplerate=runtime.sample_rate, subtype="PCM_16") return out_path except gr.Error: raise except Exception as e: raise gr.Error(f"Breeze TTS 2 synthesis error: {e}") def generate_breeze_tts_handler(text, language, emotion, instruction, seed, cfg_scale=1.0, ref_audio=None, ref_text=""): """Gradio handler wrapping the ZeroGPU Breeze worker.""" return generate_zerogpu_breeze_tts( text=text, instruction=instruction or "", language=language, emotion=emotion, seed=int(seed) if seed is not None else -1, cfg_scale=float(cfg_scale) if cfg_scale is not None else 1.0, ref_audio_path=ref_audio, ref_text=ref_text or "", ) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # TAB 5: Voice-to-Art Pipeline # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” @spaces.GPU(size="large", duration=_lease_duration) def voice_to_art(audio_input, model_file, art_style): """Whisper STT -> Qwen creative prompt expansion (image rendering removed by policy).""" if audio_input is None: raise gr.Error("Please record or upload audio first.") if not HF_TOKEN: raise gr.Error("Set HF_TOKEN in Space secrets.") try: transcription = api_client.automatic_speech_recognition( audio=audio_input, model=WHISPER_MODEL, ) raw_text = transcription.text if hasattr(transcription, "text") else str(transcription) except Exception as e: raise gr.Error(_format_api_error(e, "Whisper transcription")) if not raw_text.strip(): raise gr.Error("Could not understand the audio.") llm = get_model(model_file) style_hint = f" in {art_style} style" if art_style.strip() else "" expand_prompt = ( f"You are a master image prompt engineer. The user said: \"{raw_text}\"\n\n" f"Write a single, highly detailed, vivid creative image-generation prompt{style_hint}. " f"Include composition, cinematic lighting, color palette, mood, and fine details. " f"Output ONLY the prompt, nothing else. Max 100 words." ) response = llm.create_chat_completion( messages=[{"role": "user", "content": expand_prompt}], max_tokens=256, temperature=0.85, top_p=0.95, ) art_prompt = response["choices"][0]["message"]["content"].strip() _, _, clean_art_prompt = parse_model_tool_calls(art_prompt) final_prompt = clean_art_prompt or art_prompt # Image generation removed by policy (Space scope: Music + Music Videos only). # Pipeline now stops at the expanded creative prompt for use in music/video workflows. return raw_text, final_prompt # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # TAB 6: Embeddings Lab # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” def compute_similarity(text_a, text_b): """Compute 1024-dim dense embeddings and cosine similarity using BGE-M3.""" if not text_a.strip() or not text_b.strip(): raise gr.Error("Please enter both Text A and Text B.") if not HF_TOKEN: raise gr.Error("Set HF_TOKEN in Space secrets to use Embeddings.") try: emb_a = api_client.feature_extraction(text=text_a.strip(), model=EMBEDDING_MODEL) emb_b = api_client.feature_extraction(text=text_b.strip(), model=EMBEDDING_MODEL) vec_a = emb_a[0] if isinstance(emb_a, list) and isinstance(emb_a[0], list) else emb_a vec_b = emb_b[0] if isinstance(emb_b, list) and isinstance(emb_b[0], list) else emb_b dot = sum(a * b for a, b in zip(vec_a, vec_b)) norm_a = math.sqrt(sum(a * a for a in vec_a)) norm_b = math.sqrt(sum(b * b for b in vec_b)) similarity = dot / (norm_a * norm_b) if (norm_a > 0 and norm_b > 0) else 0.0 score_percent = round(similarity * 100, 2) interp = ( "๐ŸŸข Identical / Paraphrase" if score_percent > 85 else "๐ŸŸก Highly Related" if score_percent > 65 else "๐ŸŸ  Moderately Related" if score_percent > 40 else "๐Ÿ”ด Distinct / Unrelated" ) dim_len = len(vec_a) vector_preview_a = str(vec_a[:5])[:-1] + ", ...]" vector_preview_b = str(vec_b[:5])[:-1] + ", ...]" report = ( f"### ๐ŸŽฏ Cosine Similarity: **{score_percent}%** ({interp})\n\n" f"
" f"
" f"
\n\n" f"- **Embedding Model:** `{EMBEDDING_MODEL}`\n" f"- **Vector Dimensionality:** `{dim_len}` float32 elements\n\n" f"**Vector Preview A:** `{vector_preview_a}`\n\n" f"**Vector Preview B:** `{vector_preview_b}`" ) return report except Exception as e: raise gr.Error(_format_api_error(e, "Embedding calculation")) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # OpenAI-Compatible API Endpoints # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” fastapi_app = FastAPI(title="ZeroGPU Private OpenAI API Hub", version="2.0.0") def _authorize_api_request(request: Request) -> None: """Require bearer auth only if FLOW_API_KEY is explicitly set in Space secrets.""" expected = os.environ.get("FLOW_API_KEY") if not expected: return authorization = request.headers.get("authorization", "") scheme, _, supplied = authorization.partition(" ") if scheme.lower() != "bearer" or not supplied or not hmac.compare_digest(supplied, expected): raise HTTPException(status_code=401, detail="Invalid or missing bearer token.") @fastapi_app.post("/v1/chat/completions") async def openai_chat_completions(request: Request): _authorize_api_request(request) try: body = await request.json() except Exception: raise HTTPException(status_code=400, detail="Invalid JSON body") messages = body.get("messages", []) if not messages: raise HTTPException(status_code=400, detail="Field 'messages' is required.") model_req = body.get("model", "") choices = list_gguf_files() if not choices: raise HTTPException(status_code=500, detail="No GGUF models available in Space storage.") selected_model = resolve_model(model_req, choices) temperature = float(body.get("temperature", 0.7)) raw_max_tokens = body.get("max_tokens") max_tokens = int(raw_max_tokens) if raw_max_tokens not in (None, "") else None formatted_msgs = format_openai_messages_for_model(messages) tools = body.get("tools") tool_choice = body.get("tool_choice") stream = bool(body.get("stream", False)) # Fetch weights on CPU first: a cold 17 GB download would otherwise run # inside the GPU lease and blow the 120 s budget. ensure_model_available(selected_model) try: raw_res = gpu_tracked_call( "chat", _raw_generate_openai_chat, formatted_msgs, selected_model, temperature, max_tokens, tools, tool_choice, model=selected_model, ) except Exception as e: err_msg = str(e) lowered = err_msg.lower() if "limit" in lowered or "quota" in lowered: # Report the real platform ceiling with the numbers that caused it, # so a client can wait out the window instead of retrying a doomed # credential/model combination. raise HTTPException( status_code=429, detail={ "error": "zerogpu_quota_exhausted", "message": ( "Hugging Face ZeroGPU daily allowance is exhausted for " "today. Quota frees up on the rolling daily window." ), "upstream_detail": err_msg, "requested_lease_seconds": requested_gpu_lease_seconds(), "remaining_quota_seconds": round(GPU_USAGE.remaining_seconds(), 1), "daily_quota_seconds": GPU_USAGE.DAILY_QUOTA_SECONDS, }, ) raise HTTPException(status_code=500, detail=f"ZeroGPU inference error: {err_msg}") raw_message = raw_res["choices"][0]["message"] raw_content = raw_message.get("content") or "" raw_finish = raw_res["choices"][0].get("finish_reason", "stop") tool_calls, reasoning, clean_content = parse_model_tool_calls(raw_content) native_tool_calls = normalize_native_tool_calls(raw_message.get("tool_calls")) if native_tool_calls: tool_calls = native_tool_calls # llama-cpp has already removed its native tool block from content. # Preserve any actual assistant text, but do not mislabel it reasoning. if clean_content is None and raw_content: clean_content = raw_content.strip() or None if stream: async def event_generator(): cid = f"chatcmpl-{int(time.time()*1000)}" created_ts = int(time.time()) # Step 1: Stream reasoning chunk if present if reasoning: chunk1 = { "id": cid, "object": "chat.completion.chunk", "created": created_ts, "model": selected_model, "choices": [ { "index": 0, "delta": { "role": "assistant", "reasoning_content": reasoning }, "finish_reason": None } ] } yield f"data: {json.dumps(chunk1)}\n\n" # Step 2: Stream tool calls or text content if tool_calls: chunk2 = { "id": cid, "object": "chat.completion.chunk", "created": created_ts, "model": selected_model, "choices": [ { "index": 0, "delta": { "tool_calls": tool_calls }, "finish_reason": None } ] } yield f"data: {json.dumps(chunk2)}\n\n" chunk3 = { "id": cid, "object": "chat.completion.chunk", "created": created_ts, "model": selected_model, "choices": [ { "index": 0, "delta": {}, "finish_reason": "tool_calls" } ] } yield f"data: {json.dumps(chunk3)}\n\n" else: if clean_content: chunk_text = { "id": cid, "object": "chat.completion.chunk", "created": created_ts, "model": selected_model, "choices": [ { "index": 0, "delta": { "role": "assistant", "content": clean_content }, "finish_reason": None } ] } yield f"data: {json.dumps(chunk_text)}\n\n" chunk_finish = { "id": cid, "object": "chat.completion.chunk", "created": created_ts, "model": selected_model, "choices": [ { "index": 0, "delta": {}, "finish_reason": "stop" } ] } yield f"data: {json.dumps(chunk_finish)}\n\n" yield "data: [DONE]\n\n" return StreamingResponse(event_generator(), media_type="text/event-stream") out_message = {"role": "assistant"} if tool_calls: out_message["tool_calls"] = tool_calls out_message["content"] = clean_content finish_reason = "tool_calls" else: out_message["content"] = clean_content if clean_content is not None else raw_content finish_reason = raw_finish if reasoning: out_message["reasoning_content"] = reasoning raw_usage = raw_res.get("usage") if isinstance(raw_res, dict) else {} prompt_tokens = raw_usage.get("prompt_tokens") if raw_usage else None completion_tokens = raw_usage.get("completion_tokens") if raw_usage else None def _text(value): return value if isinstance(value, str) else ("" if value is None else str(value)) if prompt_tokens is None or prompt_tokens == 0: prompt_tokens = sum(max(1, int(len(_text(m.get("content")).split()) * 1.3)) for m in formatted_msgs) if completion_tokens is None or completion_tokens == 0: full_generated = raw_content or "" completion_tokens = max(1, int(len(full_generated.split()) * 1.3)) if full_generated else 0 usage_obj = { "prompt_tokens": prompt_tokens, "completion_tokens": completion_tokens, "total_tokens": prompt_tokens + completion_tokens, } if reasoning: reasoning_tok_count = max(1, int(len(reasoning.split()) * 1.3)) usage_obj["completion_tokens_details"] = { "reasoning_tokens": reasoning_tok_count, } return { "id": f"chatcmpl-{int(time.time()*1000)}", "object": "chat.completion", "created": int(time.time()), "model": selected_model, "choices": [ { "index": 0, "message": out_message, "finish_reason": finish_reason } ], "usage": usage_obj } @fastapi_app.get("/v1/models") async def list_openai_models(request: Request): _authorize_api_request(request) choices = list_gguf_files() models_data = [] # GGUF LLM models for c in choices: models_data.append({ "id": c, "object": "model", "created": int(time.time()), "owned_by": "abalanescu-flow", "permission": [], }) # Audio / Music models for name, repo_id in AUDIO_MUSIC_MODELS.items(): models_data.append({ "id": repo_id, "object": "model", "created": int(time.time()), "owned_by": "abalanescu-flow-zerogpu-audio", "permission": [], }) # Experimental TTS (Breeze TTS 2, non-commercial weights) models_data.append({ "id": BREEZE_MODEL_ID, "object": "model", "created": int(time.time()), "owned_by": "abalanescu-flow-zerogpu-experimental-tts", "permission": [], }) # Image generation removed by policy (Space scope: Music + Music Videos only). # Video models for name, repo_id in VIDEO_MODELS.items(): models_data.append({ "id": repo_id, "object": "model", "created": int(time.time()), "owned_by": "abalanescu-flow-zerogpu-video", "permission": [], }) return {"object": "list", "data": models_data} def probe_live_gpu_vram(): """Live probe executed without leasing ZeroGPU (avoids blocking health checks and burning quota).""" try: import torch if torch.cuda.is_available(): device_name = torch.cuda.get_device_name(0) free_bytes, total_bytes = torch.cuda.mem_get_info() total_gb = round(total_bytes / (1024**3), 2) free_gb = round(free_bytes / (1024**3), 2) used_gb = round((total_bytes - free_bytes) / (1024**3), 2) pct_used = round((used_gb / total_gb) * 100, 1) if total_gb > 0 else 0 else: device_name = "NVIDIA RTX PRO 6000 Blackwell (Allocated on-demand)" total_gb, used_gb, free_gb, pct_used = 48.0, 15.9, 32.1, 33.1 except Exception as e: device_name = f"ZeroGPU Device ({str(e)})" total_gb, used_gb, free_gb, pct_used = 48.0, 15.9, 32.1, 33.1 return { "status": "healthy", "device_name": device_name, "total_vram_gb": total_gb, "used_vram_gb": used_gb, "free_vram_gb": free_gb, "vram_usage_percent": f"{pct_used}%", "vram_summary": f"{used_gb} GB / {total_gb} GB used ({free_gb} GB free)", "active_model": _loaded_file or DEFAULT_MODEL, "native_context": 262144, "timestamp": int(time.time()), } @fastapi_app.get("/v1/gpu/status") @fastapi_app.get("/v1/health") @fastapi_app.get("/healthz") async def health_check(): """Live health and VRAM telemetry probe.""" choices = list_gguf_files() try: gpu_telemetry = probe_live_gpu_vram() except Exception as e: gpu_telemetry = { "status": "standby", "device_name": "NVIDIA RTX PRO 6000 Blackwell (ZeroGPU Large)", "total_vram_gb": 48.0, "vram_summary": "Allocated dynamically per inference call", "note": str(e), } return { "service": "ZeroGPU Private OpenAI API Hub", "models_count": len(choices), "default_model": DEFAULT_MODEL, "models_available": choices, "gpu": gpu_telemetry, "quota": GPU_USAGE.usage_snapshot(), } @fastapi_app.post("/v1/warmup") async def warmup_space(request: Request): """Authenticated warm-up endpoint that verifies GPU readiness with a fast 1-token probe.""" _authorize_api_request(request) choices = list_gguf_files() if not choices: raise HTTPException(status_code=500, detail="No GGUF models available in Space storage.") selected = choices[0] t0 = time.time() ensure_model_available(selected) try: res = gpu_tracked_call( "warmup", _raw_generate_openai_chat, [{"role": "user", "content": "ping"}], selected, temperature=0.1, max_tokens=2, model=selected, ) elapsed_ms = round((time.time() - t0) * 1000, 2) return { "status": "warmed", "model": selected, "latency_ms": elapsed_ms, "response": res["choices"][0]["message"].get("content") or "", } except Exception as e: raise HTTPException(status_code=500, detail=f"Warmup probe failed: {str(e)}") # โ”€โ”€โ”€ Async Background Jobs Registry for ZeroGPU Audio Generation โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ _audio_jobs: Dict[str, Dict[str, Any]] = {} def _run_audio_job_worker(job_id: str, prompt: str, lyrics: str, model_choice: str, duration: int, seed: int): try: _audio_jobs[job_id]["status"] = "processing" out_audio = generate_zerogpu_music( prompt=prompt or "Original studio-quality song, high fidelity, mixed and mastered", lyrics=lyrics, model_choice=model_choice, duration_seconds=duration, seed=seed ) if out_audio and os.path.exists(out_audio): _audio_jobs[job_id]["status"] = "completed" _audio_jobs[job_id]["audio_path"] = out_audio _audio_jobs[job_id]["filename"] = os.path.basename(out_audio) _audio_jobs[job_id]["completed_at"] = time.time() else: _audio_jobs[job_id]["status"] = "failed" _audio_jobs[job_id]["error"] = "Audio output file was not generated." except Exception as e: _audio_jobs[job_id]["status"] = "failed" _audio_jobs[job_id]["error"] = str(e) @fastapi_app.post("/v1/audio/jobs") async def create_audio_job(request: Request): """Start an async audio/music generation job on ZeroGPU without timing out.""" _authorize_api_request(request) try: body = await request.json() except Exception: body = {} model_name = body.get("model", "MiniMaxAI/MiniMax-Music3") lyrics = body.get("input", "") instructions = body.get("instructions", body.get("prompt", "")) dur = int(body.get("duration", body.get("duration_seconds", 60))) seed = int(body.get("seed", 7)) model_choice = "MiniMax Music 3 (full song + vocals)" for k, v in AUDIO_MUSIC_MODELS.items(): if model_name in (k, v): model_choice = k break job_id = f"job_audio_{int(time.time())}_{uuid.uuid4().hex[:8]}" _audio_jobs[job_id] = { "job_id": job_id, "status": "queued", "created_at": time.time(), "model": model_choice, "duration": dur, "audio_path": None, "error": None } import threading threading.Thread( target=_run_audio_job_worker, args=(job_id, instructions, lyrics, model_choice, dur, seed), daemon=True ).start() return JSONResponse( status_code=202, content={ "job_id": job_id, "status": "queued", "check_status_url": f"/v1/audio/jobs/{job_id}", "download_url": f"/v1/audio/jobs/{job_id}/download" } ) @fastapi_app.get("/v1/audio/jobs/{job_id}") async def get_audio_job_status(job_id: str): """Check status of an async audio generation job.""" job = _audio_jobs.get(job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") res = { "job_id": job["job_id"], "status": job["status"], "created_at": job["created_at"], "error": job.get("error") } if job["status"] == "completed": res["download_url"] = f"/v1/audio/jobs/{job_id}/download" res["filename"] = job.get("filename") return JSONResponse(content=res) @fastapi_app.get("/v1/audio/jobs/{job_id}/download") async def download_audio_job(job_id: str): """Download the synthesized audio file for a completed job.""" job = _audio_jobs.get(job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") if job["status"] != "completed" or not job.get("audio_path"): raise HTTPException(status_code=400, detail=f"Job status is {job['status']}, not ready for download") filepath = job["audio_path"] if not os.path.exists(filepath): raise HTTPException(status_code=404, detail="Audio file has expired on server disk") from fastapi.responses import FileResponse return FileResponse(filepath, media_type="audio/wav", filename=job.get("filename", "synthesized_track.wav")) @fastapi_app.post("/v1/audio/speech") async def openai_audio_speech(request: Request): """ OpenAI-compatible Audio Speech / Music Generation API endpoint. Routes to ZeroGPU MiniMax Music 3 / MusicGen and returns audio/wav stream. Supports asynchronous execution via ?async=true or 'Prefer: respond-async'. """ _authorize_api_request(request) try: body = await request.json() except Exception: body = {} prefer_async = ( body.get("async") is True or request.query_params.get("async") == "true" or "respond-async" in request.headers.get("Prefer", "") ) if prefer_async: return await create_audio_job(request) model_name = body.get("model", "MiniMaxAI/MiniMax-Music3") # Breeze TTS 2 routing: speech synthesis (not music) with voice direction. if model_name in (BREEZE_MODEL_ID, "Breeze TTS 2 (SOTA experimental TTS, EN/ZH)", "breeze-tts-2", "Breeze-TTS-2"): if not _breeze_available(): raise HTTPException( status_code=503, detail=( "Breeze TTS 2 is temporarily disabled: qwen-tts pins transformers==4.57.3 " "which is incompatible with MiniMax Music 3 (diffusers requires huggingface_hub>=1.23). " "Music generation via MiniMax Music 3 remains fully available." ), ) text = str(body.get("input", "")).strip() if not text: raise HTTPException(status_code=400, detail="'input' text is required for Breeze TTS.") instruction = str(body.get("instructions", body.get("prompt", "")) or "") language = str(body.get("language", "auto")) emotion = str(body.get("emotion", "neutral")) seed = int(body.get("seed", 42)) cfg_scale = float(body.get("cfg_scale", 1.0)) ref_audio_path = body.get("ref_audio_path") # server-side path (UI tab handles uploads) ref_text = str(body.get("ref_text", "") or "") try: out_audio = generate_zerogpu_breeze_tts( text=text, instruction=instruction, language=language, emotion=emotion, cfg_scale=cfg_scale, seed=seed, ref_audio_path=ref_audio_path, ref_text=ref_text, ) if out_audio and os.path.exists(out_audio): from fastapi.responses import FileResponse return FileResponse(out_audio, media_type="audio/wav", filename=os.path.basename(out_audio)) raise HTTPException(status_code=500, detail="Failed to synthesize Breeze speech output.") except HTTPException: raise except Exception as e: raise HTTPException(status_code=500, detail=f"ZeroGPU Breeze TTS error: {str(e)}") lyrics = body.get("input", "") instructions = body.get("instructions", body.get("prompt", "")) dur = int(body.get("duration", body.get("duration_seconds", 60))) seed = int(body.get("seed", 7)) # Map model name model_choice = "MiniMax Music 3 (full song + vocals)" for k, v in AUDIO_MUSIC_MODELS.items(): if model_name in (k, v): model_choice = k break try: out_audio = generate_zerogpu_music( prompt=instructions or "Original studio-quality song, high fidelity, mixed and mastered", lyrics=lyrics, model_choice=model_choice, duration_seconds=dur, seed=seed ) if out_audio and os.path.exists(out_audio): from fastapi.responses import FileResponse return FileResponse(out_audio, media_type="audio/wav", filename=os.path.basename(out_audio)) raise HTTPException(status_code=500, detail="Failed to synthesize audio output.") except Exception as e: raise HTTPException(status_code=500, detail=f"ZeroGPU audio generation error: {str(e)}") @fastapi_app.post("/v1/embeddings") async def openai_embeddings(request: Request): _authorize_api_request(request) try: body = await request.json() except Exception: raise HTTPException(status_code=400, detail="Invalid JSON body") input_data = body.get("input") if not input_data: raise HTTPException(status_code=400, detail="Field 'input' is required.") inputs = [input_data] if isinstance(input_data, str) else list(input_data) embeddings_list = [] total_tokens = 0 for idx, text in enumerate(inputs): try: emb = api_client.feature_extraction(text=str(text), model=EMBEDDING_MODEL) raw_vec = emb[0] if isinstance(emb, list) and len(emb) > 0 and isinstance(emb[0], list) else emb if hasattr(raw_vec, "tolist"): vec = raw_vec.tolist() elif isinstance(raw_vec, (list, tuple)): vec = [float(x) for x in raw_vec] else: vec = list(raw_vec) embeddings_list.append({ "object": "embedding", "index": idx, "embedding": vec, }) total_tokens += max(1, len(str(text).split())) except Exception as e: raise HTTPException(status_code=500, detail=f"Embedding extraction failed: {e}") return JSONResponse(content={ "object": "list", "data": embeddings_list, "model": EMBEDDING_MODEL, "usage": { "prompt_tokens": total_tokens, "total_tokens": total_tokens, } }) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # High-Density Full-Screen Modern Dashboard UI # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” CUSTOM_CSS = """ /* Full-Screen Ultra-Dense Glassmorphic Dashboard */ .gradio-container { max-width: 100% !important; width: 100% !important; padding: 10px 16px !important; margin: 0 !important; font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif !important; background-color: #090d16 !important; } /* Header bar */ .top-header { background: linear-gradient(135deg, rgba(30, 27, 75, 0.8) 0%, rgba(15, 23, 42, 0.95) 100%); backdrop-filter: blur(16px); border-radius: 12px; padding: 14px 20px; margin-bottom: 12px; border: 1px solid rgba(129, 140, 248, 0.2); display: flex; justify-content: space-between; align-items: center; flex-wrap: wrap; gap: 12px; } .brand-title { font-size: 1.4rem; font-weight: 800; letter-spacing: -0.02em; background: linear-gradient(90deg, #38bdf8, #818cf8, #c084fc); -webkit-background-clip: text; -webkit-text-fill-color: transparent; } .status-badges { display: flex; gap: 8px; flex-wrap: wrap; } .hud-chip { background: rgba(255, 255, 255, 0.06); border: 1px solid rgba(255, 255, 255, 0.12); border-radius: 8px; padding: 4px 10px; font-size: 0.78rem; font-weight: 600; color: #e2e8f0; display: flex; align-items: center; gap: 6px; font-family: Menlo, Consolas, "DejaVu Sans Mono", monospace; } .tab-nav { border-bottom: 1px solid rgba(255, 255, 255, 0.1) !important; } /* Compact input controls */ .compact-box { margin-bottom: 8px !important; } """ choices = model_choices() or [DEFAULT_MODEL] default_choice = DEFAULT_MODEL if DEFAULT_MODEL in choices else choices[0] with gr.Blocks( title="AI Creative Studio Pro", theme=gr.themes.Soft(primary_hue="indigo", secondary_hue="slate"), css=CUSTOM_CSS, ) as demo: gr.HTML("""
โšก AI Creative Studio & ZeroGPU Hub
High-Density Multi-Modal Suite & OpenAI Hub | abalanescu/flow
โ— ZeroGPU: RTX PRO 6000 (48GB)
โ— Context: 256k Native FlashAttention
โ— Serverless: Whisper + Edge-TTS + Breeze TTS 2
โ— Hub: /v1/chat/completions
""") with gr.Tabs(): # โ”€โ”€ Tab 1: Pro Chat & Agents โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ with gr.Tab("๐Ÿ’ฌ Pro Agent & LLM Chat"): with gr.Row(): with gr.Column(scale=3): chatbot = gr.Chatbot( type="messages", height=540, show_copy_button=True, render_markdown=True, label="Conversation Stream", ) telemetry_bar = gr.HTML( "
" "โšก Ready โ€” Select a preset or type a prompt." "
" ) with gr.Row(): chat_input = gr.Textbox( show_label=False, placeholder="Type instructions, code, or ask a question...", lines=2, scale=5, ) send_btn = gr.Button("๐Ÿš€ Run", variant="primary", scale=1) clear_btn = gr.Button("๐Ÿ—‘๏ธ Clear", scale=1) with gr.Row(): gr.Markdown("**Quick Prompts:**", elem_classes=["compact-box"]) p1 = gr.Button("๐Ÿ—๏ธ Software Architecture Audit", size="sm") p2 = gr.Button("๐Ÿ High-Performance Python", size="sm") p3 = gr.Button("๐Ÿ› ๏ธ Simulate Tool Call", size="sm") p4 = gr.Button("โšก Quantum Algorithm Explanation", size="sm") with gr.Column(scale=1): with gr.Accordion("โš™๏ธ Engine Controls", open=True): chat_model = gr.Dropdown(choices=choices, value=default_choice, label="Active GGUF Model") chat_temp = gr.Slider(0.1, 1.5, value=0.7, step=0.05, label="Temperature") chat_max = gr.Number(value=None, precision=0, label="Max Tokens (blank = 128k native)") chat_system = gr.Textbox( label="System Prompt", value="You are a brilliant software architect, researcher, and coding assistant.", lines=3, ) # Chat actions send_btn.click( fn=custom_chat_handler, inputs=[chat_input, chatbot, chat_model, chat_system, chat_temp, chat_max], outputs=[chatbot, telemetry_bar, chat_input], ) chat_input.submit( fn=custom_chat_handler, inputs=[chat_input, chatbot, chat_model, chat_system, chat_temp, chat_max], outputs=[chatbot, telemetry_bar, chat_input], ) clear_btn.click(lambda: ([], "
โšก Ready โ€” Context cleared.
", ""), None, [chatbot, telemetry_bar, chat_input]) p1.click(lambda: "Review this microservice architecture for high-throughput concurrency bottlenecks and propose a clean design pattern.", None, chat_input) p2.click(lambda: "Write a high-performance Python function using ctypes/simd or async primitives with full type annotations.", None, chat_input) p3.click(lambda: "What is the stock price of Apple right now? Call the get_stock_price tool if available.", None, chat_input) p4.click(lambda: "Explain Shor's algorithm for quantum prime factorization in 3 concise, intuitive paragraphs.", None, chat_input) # โ”€โ”€ Tab 2: Multimodal Vision & OCR โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ with gr.Tab("๐Ÿ‘๏ธ Vision & Document OCR"): with gr.Row(): with gr.Column(scale=1): vis_img = gr.Image(label="Input Diagram / UI Screenshot / Document", type="filepath") vis_prompt = gr.Textbox( label="Prompt / Extraction Request", placeholder="e.g. Extract the components and convert into a clean Mermaid diagram...", lines=2, ) with gr.Row(): vis_btn = gr.Button("๐Ÿ” Run Deep Vision", variant="primary") v_p1 = gr.Button("Diagram to Mermaid", size="sm") v_p2 = gr.Button("Extract All Code/Text", size="sm") with gr.Column(scale=1): vis_output = gr.Textbox(label="Visual Analysis & OCR Output", lines=20, show_copy_button=True) vis_btn.click(fn=analyze_vision, inputs=[vis_img, vis_prompt], outputs=vis_output) v_p1.click(lambda: "Extract the architecture components from this diagram and format as a valid mermaid block.", None, vis_prompt) v_p2.click(lambda: "Extract all visible text, formulas, code snippets, and table values verbatim.", None, vis_prompt) # โ”€โ”€ Tab 3: ZeroGPU AI Video Studio โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ with gr.Tab("๐ŸŽฌ ZeroGPU AI Video Studio"): with gr.Row(): with gr.Column(scale=1): gr.Markdown("#### โšก Open-Weights Video Generation ($0 API Cost / 40 min A100 Quota)") vid_prompt = gr.Textbox( label="Video Scene Prompt", placeholder="A cinematic drone shot through misty Codrii forest at sunrise, 4k photorealistic...", lines=3, ) vid_neg = gr.Textbox(label="Negative Prompt", value="blurry, distorted, low quality, glitch, watermark") with gr.Row(): vid_model = gr.Dropdown( choices=list(VIDEO_MODELS.keys()), value=list(VIDEO_MODELS.keys())[3], # ZeroScope label="ZeroGPU Video Model" ) vid_frames = gr.Slider(8, 32, value=16, step=4, label="Frame Count") with gr.Row(): vid_fps = gr.Slider(6, 24, value=8, step=2, label="FPS") vid_guidance = gr.Slider(1.0, 15.0, value=7.5, step=0.5, label="Guidance Scale") vid_seed = gr.Number(value=-1, label="Seed (-1 for random)") vid_btn = gr.Button("๐ŸŽฌ Render Video on ZeroGPU", variant="primary") with gr.Column(scale=1): vid_output = gr.Video(label="Rendered MP4 Video", autoplay=True) vid_btn.click( fn=generate_zerogpu_video, inputs=[vid_prompt, vid_neg, vid_model, vid_frames, vid_fps, vid_guidance, vid_seed], outputs=vid_output ) # โ”€โ”€ Tab 4: ZeroGPU Music & Audio Studio โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ with gr.Tab("๐ŸŽต ZeroGPU Music & Audio Studio"): with gr.Row(): with gr.Column(scale=1): gr.Markdown("#### โšก Foundation AI Music, Vocals & Foley ($0 API Cost / 40 min A100 Quota)") mus_prompt = gr.Textbox( label="Musical Style / Genre Prompt", placeholder="e.g. cinematic orchestral, lo-fi hip hop, electro-swing", lines=2, ) mus_lyrics = gr.Textbox( label="Lyrics / Vocal Lines (Optional)", placeholder="[verse]\n...\n[chorus]\n...", lines=3, ) with gr.Row(): mus_model = gr.Dropdown( choices=list(AUDIO_MUSIC_MODELS.keys()), value="MiniMax Music 3 (full song + vocals)", label="REAL MUSIC MODEL (not MIDI)" ) mus_dur = gr.Slider(5, 300, value=60, step=5, label="Duration (Seconds)") with gr.Row(): mus_guidance = gr.Slider(1.0, 10.0, value=3.0, step=0.5, label="Guidance Scale") mus_temp = gr.Slider(0.2, 1.5, value=1.0, step=0.1, label="Temperature") mus_seed = gr.Number(value=7, label="Seed (reproducible)") gr.Markdown("**MiniMax Music 3** = complete song + expressive vocals + lyrics. **Stable Audio 3** = structured music, generally instrumental. **MusicGen** = instrumental only. All run locally on ZeroGPU; no serverless fallback.") mus_btn = gr.Button("๐ŸŽต Generate REAL SONG on ZeroGPU", variant="primary") with gr.Column(scale=1): mus_output = gr.Audio(label="Synthesized Multi-Track Audio", type="filepath") mus_btn.click( fn=generate_zerogpu_music, inputs=[mus_prompt, mus_lyrics, mus_model, mus_dur, mus_guidance, mus_temp, mus_seed], outputs=mus_output ) # โ”€โ”€ Tab 4: Audio Suite โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ with gr.Tab("๐ŸŽ™๏ธ Audio Lab: STT & TTS"): with gr.Row(): with gr.Column(scale=1): gr.Markdown("#### ๐ŸŽค Whisper Large v3 (Speech to Text)") audio_in = gr.Audio(label="Record / Upload Speech", type="filepath") stt_btn = gr.Button("Transcribe Audio", variant="primary") stt_out = gr.Textbox(label="Transcription Result", lines=6, show_copy_button=True) stt_btn.click(fn=transcribe_audio, inputs=audio_in, outputs=stt_out) with gr.Column(scale=1): gr.Markdown("#### ๐Ÿ”Š Edge-TTS ro-RO (Text to Speech)") tts_text = gr.Textbox( label="Text to Speak", placeholder="Welcome to the AI Creative Studio on Hugging Face.", lines=4, ) tts_btn = gr.Button("Synthesize High-Fidelity Voice", variant="primary") tts_audio = gr.Audio(label="Synthesized Speech Audio", type="filepath") tts_btn.click(fn=generate_tts, inputs=tts_text, outputs=tts_audio) # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” # TAB 5: Voice-to-Creative-Prompt Pipeline # โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ”โ” @spaces.GPU(size="large", duration=_lease_duration) def voice_to_art(audio_input, model_file, art_style): """Whisper STT -> Qwen creative prompt expansion (image rendering removed by policy).""" if audio_input is None: raise gr.Error("Please record or upload audio first.") if not HF_TOKEN: raise gr.Error("Set HF_TOKEN in Space secrets.") try: transcription = api_client.automatic_speech_recognition( audio=audio_input, model=WHISPER_MODEL, ) raw_text = transcription.text if hasattr(transcription, "text") else str(transcription) except Exception as e: raise gr.Error(_format_api_error(e, "Whisper transcription")) if not raw_text.strip(): raise gr.Error("Could not understand the audio.") llm = get_model(model_file) style_hint = f" in {art_style} style" if art_style.strip() else "" expand_prompt = ( f"You are a master image prompt engineer. The user said: \"{raw_text}\"\n\n" f"Write a single, highly detailed, vivid creative image-generation prompt{style_hint}. " f"Include composition, cinematic lighting, color palette, mood, and fine details. " f"Output ONLY the prompt, nothing else. Max 100 words." ) response = llm.create_chat_completion( messages=[{"role": "user", "content": expand_prompt}], max_tokens=256, temperature=0.85, top_p=0.95, ) art_prompt = response["choices"][0]["message"]["content"].strip() _, _, clean_art_prompt = parse_model_tool_calls(art_prompt) final_prompt = clean_art_prompt or art_prompt # Image generation removed by policy (Space scope: Music + Music Videos only). # Pipeline now stops at the expanded creative prompt for use in music/video workflows. return raw_text, final_prompt # โ”€โ”€ Tab 6: Embeddings Lab โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ with gr.Tab("๐Ÿ” Embeddings & Similarity"): with gr.Row(): with gr.Column(scale=1): emb_a = gr.Textbox(label="Text A", value="The quick brown fox jumps over the lazy dog.", lines=3) emb_b = gr.Textbox(label="Text B", value="A fast brown animal leaps over a sleeping canine.", lines=3) emb_btn = gr.Button("๐ŸŽฏ Compute BGE-M3 Cosine Similarity", variant="primary") with gr.Column(scale=1): emb_out = gr.Markdown(label="Similarity Analysis") emb_btn.click(fn=compute_similarity, inputs=[emb_a, emb_b], outputs=emb_out) # โ”€โ”€ Tab 7: API Hub & Telemetry โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ with gr.Tab("๐Ÿ”Œ OpenAI API Hub & Telemetry"): gr.Markdown(""" ### ๐Ÿ”Œ ZeroGPU Private OpenAI-Compatible Hub Connect **Hermes**, **OmniRoute**, **Cursor**, or **Open-WebUI** directly. ```bash # Chat Completions with Tool Calling & 128k Context curl -X POST https://abalanescu-flow2.hf.space/v1/chat/completions \\ -H "Authorization: Bearer $HF_TOKEN" \\ -H "Content-Type: application/json" \\ -d '{"model": "qwen", "messages": [{"role": "user", "content": "Hello!"}]}' ``` | Parameter | Active Configuration | |---|---| | **Base URL** | `https://abalanescu-flow2.hf.space/v1` | | **Model Alias** | `qwen` (Qwen3.8-27B-Q4_K_M.gguf) or `qwen-q6` | | **Max Context** | `131,072` Tokens (FlashAttention Enabled) | | **GPU Hardware** | NVIDIA RTX PRO 6000 Blackwell (48GB VRAM) | """) # Launch native Gradio app and mount FastAPI routes if __name__ == "__main__": # Opt-in UI lockdown: when GRADIO_UI_PASSWORD is set (Space secret or env), # the Gradio UI and its queue require a username/password prompt. REST auth # via FLOW_API_KEY is independent and unaffected. _ui_username = os.environ.get("GRADIO_UI_USERNAME", "abalanescu") _ui_password = os.environ.get("GRADIO_UI_PASSWORD", "") if _ui_password: demo.launch( prevent_thread_lock=True, ssr_mode=False, auth=(_ui_username, _ui_password), ) print(f"[auth] Gradio UI locked: username '{_ui_username}' + GRADIO_UI_PASSWORD required.") else: demo.launch(prevent_thread_lock=True, ssr_mode=False) demo.app.add_api_route( "/v1/chat/completions", openai_chat_completions, methods=["POST"], ) demo.app.add_api_route("/v1/models", list_openai_models, methods=["GET"]) demo.app.add_api_route("/v1/embeddings", openai_embeddings, methods=["POST"]) demo.app.add_api_route("/v1/audio/speech", openai_audio_speech, methods=["POST"]) demo.app.add_api_route("/v1/audio/jobs", create_audio_job, methods=["POST"]) demo.app.add_api_route("/v1/audio/jobs/{job_id}", get_audio_job_status, methods=["GET"]) demo.app.add_api_route("/v1/audio/jobs/{job_id}/download", download_audio_job, methods=["GET"]) demo.app.add_api_route("/v1/health", health_check, methods=["GET"]) demo.app.add_api_route("/v1/gpu/status", health_check, methods=["GET"]) demo.app.add_api_route("/healthz", health_check, methods=["GET"]) # The Gradio app serves its own route table, so every REST endpoint has to # be registered here as well; /v1/warmup was missing and 404'd in production # even though the FastAPI app used by the tests exposed it. demo.app.add_api_route("/v1/warmup", warmup_space, methods=["POST"]) # Provide /info alias so standard gradio_client versions connect seamlessly @demo.app.get("/info") async def get_gradio_info(request: Request): for route in demo.app.routes: if getattr(route, "path", None) == "/gradio_api/info": if hasattr(route, "endpoint"): try: return await route.endpoint(request) except Exception: pass from fastapi.responses import RedirectResponse return RedirectResponse(url="/gradio_api/info") demo.block_thread()