Spaces:
Running on Zero
Running on Zero
Download app.py from abalanescu/flow2: direct link, hf CLI and curl.
- Browser
- Download file 126 kB
-
https://huggingface.co/spaces/abalanescu/flow2/resolve/main/app.py
- Command line
-
hf download hf://spaces/abalanescu/flow2/app.py
-
curl -L -o app.py https://huggingface.co/spaces/abalanescu/flow2/resolve/main/app.py
126 kB
| # @file app.py | |
| # @description AI Creative Studio & ZeroGPU LLM Hub — High-Density Pro Dashboard | |
| # Showcases both HF PRO subscription resources: | |
| # ⚡ ZeroGPU (48GB Blackwell RTX PRO 6000) for local 27B LLMs & Tool Calling | |
| # 🌐 Serverless HF Inference API for Whisper, BGE-M3 & Qwen-VL. | |
| # | |
| # @changes | |
| # - [2026-07-08] [Composer] - Initial Gemma Heretic bucket-backed chat app | |
| # - [2026-08-12] [Jcode] - AI Creative Studio: multi-tab app combining ZeroGPU + Inference API | |
| # - [2026-08-15] [Jcode] - Add Qwen3.8-27B, 128k context, FlashAttention, OpenAI tool calling | |
| # - [2026-08-16] [Jcode] - High-density full-screen professional dashboard with live telemetry HUD | |
| import os | |
| import sys | |
| import shutil | |
| import subprocess | |
| import tempfile | |
| import time | |
| import hmac | |
| import json | |
| import re | |
| import uuid | |
| import gc | |
| import math | |
| from typing import Dict, List, Any, Optional | |
| from fastapi import FastAPI, Request, HTTPException | |
| from fastapi.responses import JSONResponse, StreamingResponse | |
| import uvicorn | |
| # Preload CUDA 12 runtime libs via ctypes | |
| try: | |
| import glob as _glob | |
| import ctypes as _ctypes | |
| import site | |
| _libs = [] | |
| for sp in site.getsitepackages(): | |
| _libs += _glob.glob(os.path.join(sp, "nvidia", "*", "lib", "*.so*")) | |
| for _p in _libs: | |
| try: | |
| _ctypes.CDLL(_p) | |
| except Exception: | |
| pass | |
| except Exception: | |
| pass | |
| try: | |
| import gradio as gr | |
| except ImportError: | |
| # Graceful fallback (same pattern as spaces/llama_cpp): allows offline | |
| # unit tests to import app.py without gradio installed. Mocks cover UI | |
| # builders, .Error, .themes, and event-handler chaining methods. | |
| class _MockComponent: | |
| def __init__(self, *args, **kwargs): | |
| pass | |
| def __enter__(self): | |
| return self | |
| def __exit__(self, *args): | |
| return False | |
| def _chain(self, *args, **kwargs): | |
| return self | |
| change = click = submit = select = upload = blur = _chain | |
| launch = _chain | |
| class _MockThemes: | |
| def __getattr__(self, name): | |
| return lambda *a, **kw: None | |
| class _MockGradio: | |
| Error = type("Error", (Exception,), {}) | |
| themes = _MockThemes() | |
| def __getattr__(self, name): | |
| return _MockComponent | |
| gr = _MockGradio() | |
| try: | |
| import spaces | |
| except ImportError: | |
| class _MockSpaces: | |
| def GPU(fn=None, *args, **kwargs): | |
| if fn is not None and callable(fn): | |
| return fn | |
| def decorator(f): | |
| return f | |
| return decorator | |
| spaces = _MockSpaces() | |
| from huggingface_hub import InferenceClient, hf_hub_download | |
| try: | |
| from llama_cpp import Llama | |
| except ImportError: | |
| Llama = None | |
| # ─── Config ─────────────────────────────────────────────────────────────────── | |
| SEARCH_DIRS = ["/tmp", "/data", "/models", "."] | |
| SUPPORTED_MODELS = ( | |
| "Qwen3.8-9B-Q8_0.gguf", | |
| "Qwen3.8-9B-Q4_K_M.gguf", | |
| "Qwen3.8-27B-Q4_K_M.gguf", | |
| "Qwen3.8-27B-Q6_K.gguf", | |
| "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", | |
| "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", | |
| "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", | |
| "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf", | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf", | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf", | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf", | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf", | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf", | |
| "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", | |
| "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", | |
| ) | |
| DEFAULT_MODEL = SUPPORTED_MODELS[0] | |
| MODEL_HUB_SOURCES = { | |
| "Qwen3.8-9B-Q8_0.gguf": { | |
| "repo_id": "empero-ai/Qwen3.8-9B-GGUF", | |
| "filename": "Qwen3.8-9B-Q8_0.gguf", | |
| }, | |
| "Qwen3.8-9B-Q4_K_M.gguf": { | |
| "repo_id": "empero-ai/Qwen3.8-9B-GGUF", | |
| "filename": "Qwen3.8-9B-Q4_K_M.gguf", | |
| }, | |
| "Qwen3.8-27B-Q4_K_M.gguf": { | |
| "repo_id": "unsloth/Qwen3.8-27B-GGUF", | |
| "filename": "Qwen3.8-27B-Q4_K_M.gguf", | |
| }, | |
| "Qwen3.8-27B-Q6_K.gguf": { | |
| "repo_id": "unsloth/Qwen3.8-27B-GGUF", | |
| "filename": "Qwen3.8-27B-Q6_K.gguf", | |
| }, | |
| "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf": { | |
| "repo_id": "mradermacher/Qwen3.8-27B-Uncensored-i1-GGUF", | |
| "filename": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", | |
| }, | |
| # The regular quant is deliberately selected over the MTP build. The | |
| # current llama-cpp-python integration does not enable draft decoding, so | |
| # MTP heads add ~0.4 GiB without providing an inference benefit. | |
| "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf": { | |
| "repo_id": "DavidAU/Qwen3.8-27B-TURBO-Fable-Cold-Fusion-735-882-Heretic-Uncensored-NEO-CODER-MAX-MTP-GGUF", | |
| "filename": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", | |
| }, | |
| "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf": { | |
| "repo_id": "petruhonk/Qwen3.8-9B-Distill-uncensored-heretic-GGUF", | |
| "filename": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", | |
| }, | |
| "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf": { | |
| "repo_id": "petruhonk/Qwen3.8-9B-Distill-uncensored-heretic-GGUF", | |
| "filename": "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf", | |
| }, | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf": { | |
| "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", | |
| "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| }, | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf": { | |
| "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", | |
| "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf", | |
| }, | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf": { | |
| "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", | |
| "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf", | |
| }, | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf": { | |
| "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", | |
| "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf", | |
| }, | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf": { | |
| "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", | |
| "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf", | |
| }, | |
| "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf": { | |
| "repo_id": "Jackrong/DeepSeek-V4-Pro-Qwen3.5-9B-MTP-GGUF", | |
| "filename": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf", | |
| }, | |
| "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf": { | |
| "repo_id": "mradermacher/gemma-4-26B-A4B-it-ultra-uncensored-heretic-i1-GGUF", | |
| "filename": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", | |
| }, | |
| "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf": { | |
| "repo_id": "mradermacher/gemma-4-26B-A4B-it-ultra-uncensored-heretic-i1-GGUF", | |
| "filename": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", | |
| }, | |
| } | |
| import sys | |
| import shutil | |
| import random | |
| BASE_DIR = os.path.dirname(__file__) | |
| BREEZE_DIR = os.path.join(BASE_DIR, "zerogpu") # vendored breezeblue-ai/breeze-tts inference code | |
| if BREEZE_DIR not in sys.path: | |
| sys.path.insert(0, BREEZE_DIR) | |
| if BASE_DIR not in sys.path: | |
| sys.path.insert(0, BASE_DIR) | |
| BREEZE_MODEL_ID = "BreezeBlue/Breeze-TTS-2" | |
| _breeze_runtime_cache: Dict[str, Any] = {} | |
| def _breeze_available() -> bool: | |
| """Breeze TTS 2 needs the qwen-tts pin family (transformers==4.57.3), which | |
| is incompatible with diffusers Music3 ModularPipeline (huggingface_hub>=1.23). | |
| It is feature-gated: deps absent -> graceful 503 instead of build failure. | |
| breeze_infer is vendored under zerogpu/, so spec-checking it alone is not | |
| enough; qwen_tts is the pip-only package whose install implies the whole | |
| runtime dep chain (torchaudio, librosa, einops, onnxruntime) is present. | |
| Re-enable by adding qwen-tts + its deps back to requirements.txt once | |
| upstream unpins transformers.""" | |
| try: | |
| import importlib.util | |
| return ( | |
| importlib.util.find_spec("breeze_infer") is not None | |
| and importlib.util.find_spec("qwen_tts") is not None | |
| ) | |
| except Exception: | |
| return False | |
| # HF Serverless Inference API models | |
| # (FLUX_MODEL removed by policy: Space scope is Music + Music Videos only.) | |
| WHISPER_MODEL = "openai/whisper-large-v3" | |
| EMBEDDING_MODEL = "BAAI/bge-m3" | |
| VISION_MODEL = "Qwen/Qwen2.5-VL-72B-Instruct" | |
| # ─── Free Open-Weights ZeroGPU Model Registries ($0 Cost / 40 min A100 Quota) ── | |
| VIDEO_MODELS = { | |
| "Wan 2.1 (Alibaba T2V 1.3B)": "Wan-AI/Wan2.1-T2V-1.3B", | |
| "Wan 2.1 (Alibaba I2V 1.3B)": "Wan-AI/Wan2.1-I2V-1.3B-480P", | |
| "LTX-Video (Lightricks Fast 0.9.1)": "Lightricks/LTX-Video", | |
| "CogVideoX-5B (THUDM 5B SOTA)": "THUDM/CogVideoX-5b", | |
| "CogVideoX-2B (THUDM SOTA)": "THUDM/CogVideoX-2b", | |
| "ZeroScope v2 (576w High-Res)": "cerspense/zeroscope_v2_576w", | |
| "ZeroScope v2 XL (1024w Upscaler)": "cerspense/zeroscope_v2_XL", | |
| "ModelScope Text-to-Video": "damo-vilab/modelscope-damo-text-to-video-synthesis", | |
| "AnimateDiff (Motion Adapter)": "guoyww/animatediff-motion-adapter", | |
| "I2VGen-XL (Image to Video)": "ali-vilab/i2vgen-xl", | |
| } | |
| AUDIO_MUSIC_MODELS = { | |
| # Full-song vocal model. Official card: diffusers ModularPipeline, CUDA, 24GB+ or CPU offload. | |
| "MiniMax Music 3 (full song + vocals)": "MiniMaxAI/MiniMax-Music3", | |
| "YuE Music Full-Song (vocal + instrumental)": "m-a-p/YuE-s1-7B-dpo", | |
| # YuE2-3B: frontier quality (WildSongBench 6.96), editable scores, 48kHz. CC BY-NC 4.0. | |
| "YuE2-3B (editable scores, 48kHz)": "m-a-p/YuE2-3B", | |
| # ACE-Step v1.5 XL Turbo: Apache 2.0, commercial-safe, <4GB VRAM, 48kHz. | |
| "ACE-Step v1.5 XL Turbo (commercial-safe)": "ACE-Step/acestep-v15-xl-turbo-diffusers", | |
| # Full-song instrumental / audio model. Access may require accepting HF terms. | |
| "Stable Audio 3 Medium (full structured music)": "stabilityai/stable-audio-3-medium", | |
| "MusicGen Large (instrumental SOTA)": "facebook/musicgen-large", | |
| "MusicGen Medium (instrumental)": "facebook/musicgen-medium", | |
| "MusicGen Small (instrumental)": "facebook/musicgen-small", | |
| "MusicGen Melody (instrumental)": "facebook/musicgen-melody", | |
| "AudioGen Medium (sound effects)": "facebook/audiogen-medium", | |
| "Stable Audio Open 1.0 (audio / loops)": "stabilityai/stable-audio-open-1.0", | |
| "Bark (speech / singing experiments)": "suno/bark", | |
| "F5-TTS (Voice Cloning SOTA)": "SWivid/F5-TTS", | |
| "Parler-TTS Mini (TTS expressive)": "parler-tts/parler-tts-mini-v1", | |
| "Whisper Large v3 Turbo (STT SOTA)": "openai/whisper-large-v3-turbo", | |
| # Experimental quality TTS (voice design / clone / direction). Weights are | |
| # RESEARCH AND NON-COMMERCIAL licensed; accepted for art experiments only. | |
| } | |
| # Breeze TTS 2 feature-gated: see _breeze_available() (qwen-tts pin conflict). | |
| if _breeze_available(): | |
| AUDIO_MUSIC_MODELS["Breeze TTS 2 (SOTA experimental TTS, EN/ZH)"] = "BreezeBlue/Breeze-TTS-2" | |
| # IMAGE_MODELS removed by policy (image generation out of Space scope). | |
| EMBEDDING_MODELS = { | |
| "BGE-M3 (Multilingual 100+ langs, 8k ctx)": "BAAI/bge-m3", | |
| "Nomic Embed Text v1.5 (8k Matryoshka)": "nomic-ai/nomic-embed-text-v1.5", | |
| "BGE Reranker Large (Cross-Encoder SOTA)": "BAAI/bge-reranker-large", | |
| } | |
| # Pipeline caches for ZeroGPU execution | |
| _video_pipeline_cache = {} | |
| _audio_pipeline_cache = {} | |
| # Keep model downloads in the persistent Space cache. Loading remains lazy: a | |
| # model is downloaded only when the user selects it, never during app startup. | |
| _HF_AUDIO_CACHE = os.environ.get("HF_HOME", "/data") | |
| os.environ.setdefault("HF_HOME", _HF_AUDIO_CACHE) | |
| os.environ.setdefault("HF_HUB_CACHE", os.path.join(_HF_AUDIO_CACHE, "hub")) | |
| _image_pipeline_cache = {} | |
| # ─── Startup Background Model Pre-caching on CPU ───────────────────────────── | |
| def _preload_heavy_models_on_cpu(): | |
| """Download weights to /data or standard cache on CPU so ZeroGPU lease is not eaten by network download.""" | |
| try: | |
| from huggingface_hub import snapshot_download | |
| cache_dir = "/data/hub" if (os.path.exists("/data") and os.access("/data", os.W_OK)) else os.environ.get("HF_HUB_CACHE", None) | |
| print("[Startup Pre-Cache] Pre-caching MiniMax Music 3 on CPU...", flush=True) | |
| snapshot_download( | |
| repo_id="MiniMaxAI/MiniMax-Music3", | |
| cache_dir=cache_dir, | |
| resume_download=True, | |
| ) | |
| print("[Startup Pre-Cache] MiniMax Music 3 weights cached successfully!", flush=True) | |
| except Exception as e: | |
| print(f"[Startup Pre-Cache] Note: Background preload skipped or failed: {e}", flush=True) | |
| # YuE2-3B (4.96GB) + ACE-Step v1.5 (9.73GB) pre-cache on CPU: first music | |
| # requests on a fresh container must not burn ZeroGPU lease on downloads. | |
| _music_precache_repos = ["m-a-p/YuE2-3B", "ACE-Step/acestep-v15-xl-turbo-diffusers", "m-a-p/YuE2-Vae"] | |
| for _repo in _music_precache_repos: | |
| try: | |
| from huggingface_hub import snapshot_download | |
| print(f"[Startup Pre-Cache] Pre-caching {_repo} on CPU...", flush=True) | |
| snapshot_download(repo_id=_repo, cache_dir=cache_dir, resume_download=True) | |
| print(f"[Startup Pre-Cache] {_repo} cached successfully!", flush=True) | |
| except Exception as e: | |
| print(f"[Startup Pre-Cache] Note: {_repo} preload skipped: {e}", flush=True) | |
| # Breeze TTS 2 (7.3GB shards + 651MB audio tokenizer) pre-cache on CPU so the | |
| # ZeroGPU lease is never spent on network downloads. | |
| if _breeze_available(): | |
| try: | |
| from huggingface_hub import snapshot_download | |
| cache_dir = "/data/hub" if (os.path.exists("/data") and os.access("/data", os.W_OK)) else os.environ.get("HF_HUB_CACHE", None) | |
| print("[Startup Pre-Cache] Pre-caching Breeze TTS 2 on CPU...", flush=True) | |
| snapshot_download( | |
| repo_id="BreezeBlue/Breeze-TTS-2", | |
| cache_dir=cache_dir, | |
| resume_download=True, | |
| ) | |
| print("[Startup Pre-Cache] Breeze TTS 2 weights cached successfully!", flush=True) | |
| except Exception as e: | |
| print(f"[Startup Pre-Cache] Note: Breeze preload skipped or failed: {e}", flush=True) | |
| else: | |
| print("[Startup Pre-Cache] Breeze TTS 2 disabled (qwen-tts deps absent; see _breeze_available).", flush=True) | |
| if ( | |
| os.environ.get("HF_SKIP_PRELOAD") != "1" | |
| and os.environ.get("PYTEST_CURRENT_TEST") is None | |
| and os.environ.get("CI") is None | |
| ): | |
| import threading | |
| threading.Thread(target=_preload_heavy_models_on_cpu, daemon=True).start() | |
| # HF token from Space secrets | |
| HF_TOKEN = os.environ.get("HF_TOKEN", None) | |
| # Inference API client | |
| api_client = InferenceClient(token=HF_TOKEN) | |
| # ZeroGPU model state | |
| _llm = None | |
| _loaded_file = None | |
| # ZeroGPU billable lease budget. Hugging Face denies any lease larger than the | |
| # remaining daily allowance, so the requested duration is clamped to whatever | |
| # headroom is actually left instead of always asking for the full window. | |
| GPU_LEASE_DEFAULT_SECONDS = int(os.environ.get("FLOW_GPU_LEASE_DEFAULT_S", "120")) | |
| GPU_LEASE_MIN_SECONDS = int(os.environ.get("FLOW_GPU_LEASE_MIN_S", "20")) | |
| GPU_LEASE_SAFETY_SECONDS = int(os.environ.get("FLOW_GPU_LEASE_SAFETY_S", "5")) | |
| def quota_ledger_path() -> "str | None": | |
| """Path of the persisted ZeroGPU quota ledger. | |
| The allowance belongs to the account and survives Space restarts, so the | |
| ledger must too: an in-memory counter silently reported a full ``40 min`` | |
| after every reload while Hugging Face was already refusing leases. | |
| Returns ``None`` under pytest or when no writable location exists, which | |
| keeps unit tests hermetic and the tracker usable without persistent disk. | |
| """ | |
| if os.environ.get("PYTEST_CURRENT_TEST"): | |
| return None | |
| override = os.environ.get("FLOW_QUOTA_LEDGER_PATH") | |
| if override: | |
| return override | |
| for candidate in ("/data", BASE_DIR): | |
| try: | |
| if candidate and os.path.isdir(candidate) and os.access(candidate, os.W_OK): | |
| return os.path.join(candidate, "zerogpu_quota_ledger.json") | |
| except OSError: | |
| continue | |
| return None | |
| class GPUUsageTracker: | |
| """In-memory ZeroGPU quota telemetry (GOAL.md Milestone 3, line 54). | |
| Tracks wall-clock GPU-lease seconds per call, queue wait, cold-start | |
| (model load inside a lease), daily rollups, and quota-exhaustion | |
| (HTTP 429) state. ZeroGPU bills the whole lease duration, so elapsed | |
| wall time of each @spaces.GPU call is the honest accounting unit. | |
| The ledger is persisted through ``state_path`` because the allowance is an | |
| account-level, 24-hour window that outlives any single Space process: an | |
| in-memory-only counter reported a full 40 minutes after every reload while | |
| the platform was already refusing leases. ``state_path=None`` keeps the | |
| tracker purely in-memory (unit tests, read-only filesystems). | |
| """ | |
| DAILY_QUOTA_SECONDS = 40 * 60 # 40 min/day PRO budget (ZeroGPU) | |
| # A quota error sticks only for this long; after the TTL the flag expires | |
| # because a request is always allowed to re-probe the platform. HF documents | |
| # the allowance as "daily" and it resets 24h after the first GPU usage, so a | |
| # refusal is evidence about the present moment, not about the whole window. | |
| QUOTA_FLAG_TTL_SECONDS = int( | |
| os.environ.get("FLOW_QUOTA_FLAG_TTL_S", "2700") | |
| ) | |
| WINDOW_SECONDS = 24 * 60 * 60 # HF: quota resets 24h after first GPU usage | |
| def __init__(self, clock=time.time, state_path: "str | None" = None): | |
| self._clock = clock | |
| self._calls: List[Dict[str, Any]] = [] | |
| self._active: Dict[str, float] = {} | |
| self._last_quota_error: Dict[str, Any] = {} | |
| self._state_path = state_path | |
| self._window_started_at: "float | None" = None | |
| self._load_state() | |
| # -- persistence --------------------------------------------------- | |
| def _load_state(self) -> None: | |
| """Restore the window anchor and today's billed seconds from disk.""" | |
| if not self._state_path: | |
| return | |
| try: | |
| with open(self._state_path, "r", encoding="utf-8") as fh: | |
| state = json.load(fh) | |
| except (OSError, ValueError): | |
| return | |
| if not isinstance(state, dict): | |
| return | |
| started = state.get("window_started_at") | |
| if isinstance(started, (int, float)) and started > 0: | |
| self._window_started_at = float(started) | |
| restored = state.get("calls") | |
| if isinstance(restored, list): | |
| for record in restored: | |
| if isinstance(record, dict) and "lease_seconds" in record: | |
| self._calls.append(record) | |
| last_error = state.get("last_quota_error") | |
| if isinstance(last_error, dict) and last_error.get("ts"): | |
| self._last_quota_error = last_error | |
| self._roll_window_if_elapsed() | |
| def _save_state(self) -> None: | |
| """Persist the ledger. Best-effort: telemetry must never break a call.""" | |
| if not self._state_path: | |
| return | |
| payload = { | |
| "updated_at": int(self._clock()), | |
| "window_started_at": self._window_started_at, | |
| "window_resets_at": self.window_resets_at(), | |
| "calls": self._calls[-500:], | |
| "last_quota_error": self._last_quota_error or None, | |
| } | |
| try: | |
| tmp = f"{self._state_path}.tmp" | |
| with open(tmp, "w", encoding="utf-8") as fh: | |
| json.dump(payload, fh) | |
| os.replace(tmp, self._state_path) | |
| except OSError: | |
| pass | |
| # -- quota window -------------------------------------------------- | |
| def _roll_window_if_elapsed(self) -> None: | |
| """Drop records from an elapsed 24h window so the counter stays honest.""" | |
| resets_at = self.window_resets_at() | |
| if resets_at is None or self._clock() < resets_at: | |
| return | |
| self._calls = [] | |
| self._last_quota_error = {} | |
| self._window_started_at = None | |
| def window_resets_at(self) -> "float | None": | |
| if self._window_started_at is None: | |
| return None | |
| return self._window_started_at + self.WINDOW_SECONDS | |
| # -- lease lifecycle --------------------------------------------- | |
| def lease_start(self, label: str) -> str: | |
| self._roll_window_if_elapsed() | |
| if self._window_started_at is None: | |
| # HF anchors the 24h window on the first GPU usage, not midnight. | |
| self._window_started_at = self._clock() | |
| call_id = f"{label}-{int(self._clock() * 1000)}-{len(self._calls)}" | |
| self._active[call_id] = self._clock() | |
| return call_id | |
| def lease_end(self, call_id: str, *, model: str = "", kind: str = "chat", | |
| queue_wait_s: float = 0.0, cold_start_s: float = 0.0, | |
| completion_tokens: int = 0, error: str = "") -> Dict[str, Any]: | |
| started = self._active.pop(call_id, None) | |
| if started is None: | |
| return {} | |
| lease_s = round(max(0.0, self._clock() - started), 3) | |
| record = { | |
| "kind": kind, | |
| "model": model, | |
| "lease_seconds": lease_s, | |
| "queue_wait_seconds": round(queue_wait_s, 3), | |
| "cold_start_seconds": round(cold_start_s, 3), | |
| "completion_tokens": int(completion_tokens), | |
| "error": error[:200], | |
| "ts": int(self._clock()), | |
| } | |
| self._calls.append(record) | |
| if error and ("limit" in error.lower() or "quota" in error.lower()): | |
| self._last_quota_error = {"ts": record["ts"], "detail": error[:200]} | |
| self._save_state() | |
| return record | |
| # -- rollup -------------------------------------------------------- | |
| def _day_key(self) -> str: | |
| return time.strftime("%Y-%m-%d", time.gmtime(self._clock())) | |
| def daily_usage(self) -> Dict[str, Any]: | |
| self._roll_window_if_elapsed() | |
| today = self._day_key() | |
| day_calls = [ | |
| c for c in self._calls | |
| if time.strftime("%Y-%m-%d", time.gmtime(c["ts"])) == today | |
| ] | |
| total_s = round(sum(c["lease_seconds"] for c in day_calls), 1) | |
| ok_calls = [c for c in day_calls if not c["error"]] | |
| quota_error_fresh = ( | |
| self._last_quota_error | |
| and self._clock() - self._last_quota_error["ts"] | |
| < self.QUOTA_FLAG_TTL_SECONDS | |
| ) | |
| # An observed refusal from the platform outranks this ledger: the ledger | |
| # only knows leases this process watched, so it can claim minutes are | |
| # left while Hugging Face is already denying every lease. Reporting the | |
| # refusal as zero headroom is what stops a client from showing | |
| # "31/40 min" next to a 429. | |
| ledger_remaining = max( | |
| 0.0, float(self.DAILY_QUOTA_SECONDS) - total_s | |
| ) | |
| reported_remaining = ( | |
| 0.0 if quota_error_fresh else round(ledger_remaining, 1) | |
| ) | |
| return { | |
| "date": today, | |
| "calls_total": len(day_calls), | |
| "calls_ok": len(ok_calls), | |
| "calls_failed": len(day_calls) - len(ok_calls), | |
| "gpu_seconds_used": total_s, | |
| "gpu_minutes_used": round(total_s / 60, 2), | |
| "daily_quota_minutes": self.DAILY_QUOTA_SECONDS // 60, | |
| "quota_consumed_percent": round( | |
| total_s / self.DAILY_QUOTA_SECONDS * 100, 1 | |
| ), | |
| "quota_remaining_seconds": reported_remaining, | |
| "quota_remaining_minutes": round(reported_remaining / 60, 2), | |
| "quota_headroom_known": not quota_error_fresh, | |
| "quota_exhausted_observed": bool(quota_error_fresh), | |
| "quota_window_started_at": ( | |
| int(self._window_started_at) if self._window_started_at else None | |
| ), | |
| "quota_window_resets_at": ( | |
| int(resets_at) if (resets_at := self.window_resets_at()) else None | |
| ), | |
| "quota_window_seconds_remaining": ( | |
| max(0, int(resets_at - self._clock())) | |
| if (resets_at := self.window_resets_at()) | |
| else self.WINDOW_SECONDS | |
| ), | |
| "ledger_persisted": bool(self._state_path), | |
| "last_quota_error": self._last_quota_error or None, | |
| } | |
| def summary(self) -> Dict[str, Any]: | |
| """Aggregate latency/throughput profile across recorded calls.""" | |
| ok = [c for c in self._calls if not c["error"]] | |
| chat_calls = [c for c in ok if c["kind"] == "chat"] | |
| total_comp_tokens = sum(c.get("completion_tokens", 0) for c in chat_calls) | |
| total_lease = sum(c["lease_seconds"] for c in chat_calls) or 1e-9 | |
| cold_starts = [c["cold_start_seconds"] for c in ok if c["cold_start_seconds"] > 0] | |
| queue_waits = [c["queue_wait_seconds"] for c in ok if c["queue_wait_seconds"] > 0] | |
| return { | |
| "total_calls": len(self._calls), | |
| "total_lease_seconds": round(sum(c["lease_seconds"] for c in self._calls), 1), | |
| "avg_queue_wait_seconds": ( | |
| round(sum(queue_waits) / len(queue_waits), 3) if queue_waits else 0.0 | |
| ), | |
| "cold_start_seconds": round(max(cold_starts), 3) if cold_starts else 0.0, | |
| "chat_throughput_tokens_per_gpu_second": ( | |
| round(total_comp_tokens / total_lease, 2) if chat_calls else 0.0 | |
| ), | |
| } | |
| def used_seconds_today(self) -> float: | |
| """Billed lease seconds recorded so far in the current UTC quota day.""" | |
| today = self._day_key() | |
| return round( | |
| sum( | |
| c["lease_seconds"] | |
| for c in self._calls | |
| if time.strftime("%Y-%m-%d", time.gmtime(c["ts"])) == today | |
| ), | |
| 3, | |
| ) | |
| def remaining_seconds(self) -> float: | |
| """Daily quota headroom this tracker can still afford to lease. | |
| ZeroGPU bills the whole requested lease, and Hugging Face refuses a | |
| lease that exceeds the remaining allowance. Tracking actual lease | |
| seconds (not attempts) is what makes the number comparable to the | |
| platform ceiling. | |
| """ | |
| self._roll_window_if_elapsed() | |
| return max(0.0, float(self.DAILY_QUOTA_SECONDS) - self.used_seconds_today()) | |
| def usage_snapshot(self) -> Dict[str, Any]: | |
| return {"daily": self.daily_usage(), "profile": self.summary()} | |
| GPU_USAGE = GPUUsageTracker(state_path=quota_ledger_path()) | |
| class ZeroGpuQuotaError(RuntimeError): | |
| """Raised when the Space cannot lease enough GPU time for the request.""" | |
| def __init__(self, message: str, *, requested_seconds: int = 0, | |
| remaining_seconds: float = 0.0): | |
| super().__init__(message) | |
| self.requested_seconds = int(requested_seconds) | |
| self.remaining_seconds = float(remaining_seconds) | |
| def requested_gpu_lease_seconds() -> int: | |
| """Largest ZeroGPU lease the remaining daily quota can accommodate. | |
| Hugging Face bills the whole requested lease and denies any request that | |
| does not fit the remaining allowance, so a fixed 120 s request is refused | |
| outright once the day is nearly spent — even though a short completion | |
| would comfortably fit. Clamping to the remaining headroom lets a genuine | |
| request keep working instead of failing on a number the model never used. | |
| A floor is kept so a near-exhausted day still attempts one real lease: | |
| the platform is the source of truth about the rolling window, and a | |
| denied lease does not consume quota. | |
| """ | |
| remaining = GPU_USAGE.remaining_seconds() | |
| affordable = int(remaining - GPU_LEASE_SAFETY_SECONDS) | |
| return max(GPU_LEASE_MIN_SECONDS, min(GPU_LEASE_DEFAULT_SECONDS, affordable)) | |
| def _lease_duration(*_args: Any, **_kwargs: Any) -> int: | |
| """Dynamic @spaces.GPU duration: recomputed for every invocation.""" | |
| return requested_gpu_lease_seconds() | |
| def gpu_tracked_call(kind: str, fn, *args, model: str = "", **kwargs): | |
| """Run a ZeroGPU-leased callable with lease + queue accounting. | |
| The work runs inside the @spaces.GPU lease body, so submit->done wall | |
| time approximates the billed lease duration. Server-side queue wait | |
| is not observable; the recorded queue_wait_seconds is an upper bound. | |
| """ | |
| t_submit = time.time() | |
| call_id = GPU_USAGE.lease_start(kind) | |
| lease_s = requested_gpu_lease_seconds() | |
| def _tracked_inner(): | |
| # Residency must be evaluated INSIDE the ZeroGPU worker: module | |
| # globals set there (_llm/_loaded_file via get_model) never propagate | |
| # back to the web process, so cold-start state is invisible outside. | |
| resident_before = _llm is not None and _loaded_file == model | |
| t_body_start = time.time() | |
| try: | |
| result = fn(*args, **kwargs) | |
| error = "" | |
| except Exception as exc: | |
| result, error = None, str(exc) | |
| if not resident_before and _llm is not None and _loaded_file == model: | |
| # The body loaded the model inside this lease. The load share is | |
| # not directly observable, so the whole post-body span is the | |
| # cold-start upper bound. | |
| cold_start_s = max(0.0, time.time() - t_body_start) | |
| else: | |
| cold_start_s = 0.0 | |
| return result, error, t_body_start, cold_start_s | |
| try: | |
| result, error, t_body_start, cold_start_s = _tracked_inner() | |
| except Exception as exc: | |
| # Lease-level failure (e.g. 429 quota exhaustion): body never ran. | |
| GPU_USAGE.lease_end(call_id, model=model, kind=kind, error=str(exc)) | |
| raise | |
| queue_wait = max(0.0, t_body_start - t_submit) | |
| comp_tokens = 0 | |
| try: | |
| usage = result.get("usage") if isinstance(result, dict) else None | |
| if isinstance(usage, dict) and usage.get("completion_tokens"): | |
| comp_tokens = int(usage["completion_tokens"]) | |
| except Exception: | |
| comp_tokens = 0 | |
| GPU_USAGE.lease_end( | |
| call_id, model=model, kind=kind, | |
| queue_wait_s=queue_wait, cold_start_s=cold_start_s, | |
| completion_tokens=comp_tokens, error=error, | |
| ) | |
| if error: | |
| raise RuntimeError(error) | |
| return result | |
| def find_model_path(model_file: str) -> str: | |
| """Find absolute path of a GGUF model file across search directories.""" | |
| for d in SEARCH_DIRS: | |
| p = os.path.join(d, model_file) | |
| if os.path.isfile(p): | |
| return p | |
| return None | |
| def ensure_model_available(model_file: str) -> str: | |
| """Download a supported model on CPU, outside any ZeroGPU lease. | |
| A 17 GB GGUF takes far longer to fetch than the 120 s GPU lease, so the | |
| download must never happen inside the inference body: a cold first call | |
| used to time out and burn the daily quota. Callers run this before | |
| entering the lease. | |
| """ | |
| for d in SEARCH_DIRS: | |
| p = os.path.join(d, model_file) | |
| if os.path.isfile(p): | |
| return p | |
| source = MODEL_HUB_SOURCES.get(model_file) | |
| if source and os.path.isdir("/data"): | |
| try: | |
| return hf_hub_download( | |
| repo_id=source["repo_id"], | |
| filename=source["filename"], | |
| local_dir="/data", | |
| token=HF_TOKEN, | |
| ) | |
| except Exception as exc: | |
| print(f"Model download failed: {type(exc).__name__}: {exc}", flush=True) | |
| return None | |
| def list_gguf_files(): | |
| """List explicitly supported model files.""" | |
| found = [] | |
| for d in SEARCH_DIRS: | |
| if os.path.isdir(d): | |
| try: | |
| for name in sorted(os.listdir(d)): | |
| if name in SUPPORTED_MODELS and name not in found: | |
| found.append(name) | |
| except Exception: | |
| pass | |
| for model_file in MODEL_HUB_SOURCES: | |
| if model_file not in found: | |
| found.append(model_file) | |
| return found | |
| def model_choices(): | |
| """Return choices for UI dropdowns.""" | |
| found = list_gguf_files() | |
| for m in SUPPORTED_MODELS: | |
| if m not in found: | |
| found.append(m) | |
| return found | |
| def resolve_model(model_req: str, choices=None) -> str: | |
| """Resolve an API model ID by exact filename or documented safe alias.""" | |
| choices = choices or list_gguf_files() | |
| aliases = { | |
| "premium": "Qwen3.8-9B-Q8_0.gguf", | |
| "paid-premium": "Qwen3.8-9B-Q8_0.gguf", | |
| "default": "Qwen3.8-9B-Q8_0.gguf", | |
| "qwen-9b": "Qwen3.8-9B-Q8_0.gguf", | |
| "qwen-9b-q8": "Qwen3.8-9B-Q8_0.gguf", | |
| "qwen-9b-q4": "Qwen3.8-9B-Q4_K_M.gguf", | |
| "qwen3.8-9b": "Qwen3.8-9B-Q8_0.gguf", | |
| "qwen": "Qwen3.8-27B-Q4_K_M.gguf", | |
| "qwen-27b": "Qwen3.8-27B-Q4_K_M.gguf", | |
| "qwen3.8": "Qwen3.8-27B-Q4_K_M.gguf", | |
| "qwen3.8-27b": "Qwen3.8-27B-Q4_K_M.gguf", | |
| "qwen-fast": "Qwen3.8-9B-Q8_0.gguf", | |
| "qwen-q4": "Qwen3.8-27B-Q4_K_M.gguf", | |
| "qwen-q4_k_m": "Qwen3.8-27B-Q4_K_M.gguf", | |
| "qwen-q6": "Qwen3.8-27B-Q6_K.gguf", | |
| "qwen-q6_k": "Qwen3.8-27B-Q6_K.gguf", | |
| "qwen-uncensored": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", | |
| "qwen-heretic": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", | |
| "qwen3.8-uncensored": "Qwen3.8-27B-Uncensored.i1-Q4_K_M.gguf", | |
| "qwen-turbo": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", | |
| "qwen-turbo-coder": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", | |
| "davidau/qwen3.8-27b-turbo-fable-cold-fusion": "Qwen3.8-27B-TurboFCFusion-735-882-Here-Uncen-NEO-CODER-MAX-Q4_K_M.gguf", | |
| "qwen-9b-distill-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", | |
| "qwen-9b-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", | |
| "qwen3.8-9b-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", | |
| "petruhonk/qwen3.8-9b-distill-uncensored-heretic": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", | |
| "petruhonk/qwen3.8-9b-distill-uncensored-heretic-gguf": "Qwen3.8-9B-Distill-Heretic-Uncensored-Q8_0.gguf", | |
| "qwen3.8-9b-distill-heretic-f16": "Qwen3.8-9B-Distill-Heretic-Uncensored--F16.gguf", | |
| "deepseek-v4": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-v4-pro": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-v4-9b": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-v4-pro-qwen3.5-9b": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-v4-pro-qwen3.5-9b-mtp": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-mtp": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-mtp-9b": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "jackrong/deepseek-v4-pro-qwen3.5-9b-mtp": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "jackrong/deepseek-v4-pro-qwen3.5-9b-mtp-gguf": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-v4-q4": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q4_K_M.gguf", | |
| "deepseek-v4-q8": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q8_0.gguf", | |
| "deepseek-v4-q6": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q6_K.gguf", | |
| "deepseek-v4-q5": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-Q5_K_M.gguf", | |
| "deepseek-v4-bf16": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-BF16.gguf", | |
| "deepseek-v4-iq4": "DeepSeek-V4-Pro-Qwen3.5-9B-MTP-IQ4_XS.gguf", | |
| "gemma-q4": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", | |
| "gemma-q4_k_m": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q4_K_M.gguf", | |
| "gemma-q6": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", | |
| "gemma-q6_k": "gemma-4-26B-A4B-it-ultra-uncensored-heretic.i1-Q6_K.gguf", | |
| } | |
| requested = (model_req or "").strip() | |
| candidate = aliases.get(requested.lower(), requested) | |
| if candidate in choices: | |
| return candidate | |
| if not requested: | |
| return choices[0] | |
| raise HTTPException( | |
| status_code=400, | |
| detail=f"Unknown model '{requested}'. Use GET /v1/models for live model IDs.", | |
| ) | |
| def get_model(model_file: str) -> Llama: | |
| """Load a GGUF model with clean memory management and FlashAttention.""" | |
| global _llm, _loaded_file | |
| # Downloads are CPU work; run them before the GPU lease when a caller | |
| # bypasses the endpoint-level pre-download. | |
| model_path = find_model_path(model_file) | |
| if not model_path: | |
| model_path = ensure_model_available(model_file) | |
| if not model_path: | |
| avail = list_gguf_files() | |
| raise gr.Error( | |
| f"Model file '{model_file}' not found in /data or /models. " | |
| f"Available files: {avail}. Please upload your .gguf model to the mounted bucket." | |
| ) | |
| if _llm is not None and _loaded_file == model_file: | |
| return _llm | |
| # Free previous model from memory before loading new one | |
| if _llm is not None: | |
| try: | |
| del _llm | |
| except Exception: | |
| pass | |
| _llm = None | |
| gc.collect() | |
| if "Q6" in model_file: | |
| default_target_ctx = 32768 | |
| elif "9B" in model_file: | |
| default_target_ctx = 163840 | |
| else: | |
| default_target_ctx = 163840 | |
| target_ctx = int(os.environ.get("FLOW_N_CTX", str(default_target_ctx))) | |
| context_candidates = [target_ctx] | |
| for fallback in [163840, 131072, 65536, 32768, 16384, 8192]: | |
| if fallback not in context_candidates and fallback < target_ctx: | |
| context_candidates.append(fallback) | |
| last_error = None | |
| for ctx in context_candidates: | |
| for use_fa in [True, False]: | |
| try: | |
| gc.collect() | |
| kwargs = { | |
| "model_path": model_path, | |
| "n_ctx": ctx, | |
| "n_gpu_layers": -1, | |
| "n_batch": 2048, | |
| "n_ubatch": 512, | |
| "verbose": False, | |
| } | |
| if use_fa: | |
| kwargs["flash_attn"] = True | |
| print(f"Loading {model_file} with n_ctx={ctx}, flash_attn={use_fa}...", flush=True) | |
| _llm = Llama(**kwargs) | |
| _loaded_file = model_file | |
| print(f"Successfully loaded {model_file} with n_ctx={ctx}, flash_attn={use_fa}!", flush=True) | |
| return _llm | |
| except TypeError: | |
| continue | |
| except Exception as e: | |
| last_error = e | |
| print(f"Failed loading with n_ctx={ctx}, flash_attn={use_fa}: {e}", flush=True) | |
| if _llm is not None: | |
| try: | |
| del _llm | |
| except Exception: | |
| pass | |
| _llm = None | |
| gc.collect() | |
| break | |
| if _llm is None and last_error: | |
| raise last_error | |
| return _llm | |
| def parse_model_tool_calls(text: str): | |
| """Extract structured OpenAI tool calls and thinking from model output.""" | |
| if not text: | |
| return None, None, "" | |
| tool_calls = [] | |
| clean_text = text | |
| # Pattern 1: Qwen XML style <tool_call><function=name><parameter=k>v</parameter></function></tool_call>. | |
| # Do not require spaces around the tags. Qwen templates normally emit | |
| # newlines (or no whitespace at all), while the original expression only | |
| # matched a very specific, space-padded variant. | |
| xml_matches = list(re.finditer( | |
| r'\s*<tool_call>\s*<function=([a-zA-Z0-9_\-\.\:\/]+)>\s*(.*?)\s*</function>\s*</tool_call>', | |
| clean_text, | |
| re.DOTALL, | |
| )) | |
| if xml_matches: | |
| for idx, m in enumerate(xml_matches): | |
| fn_name = m.group(1).strip() | |
| fn_body = m.group(2) | |
| args = {} | |
| for p in re.finditer(r'<parameter=([a-zA-Z0-9_\-]+)>(.*?)</parameter>', fn_body, re.DOTALL): | |
| p_name = p.group(1).strip() | |
| p_val = p.group(2).strip() | |
| try: | |
| args[p_name] = json.loads(p_val) | |
| except Exception: | |
| args[p_name] = p_val | |
| tool_calls.append({ | |
| "index": idx, | |
| "id": f"call_{uuid.uuid4().hex[:8]}", | |
| "type": "function", | |
| "function": { | |
| "name": fn_name, | |
| "arguments": json.dumps(args, ensure_ascii=False) | |
| } | |
| }) | |
| clean_text = clean_text.replace(m.group(0), "") | |
| # Pattern 2: JSON style <tool_call>{"name": "...", "arguments": {...}}</tool_call> | |
| json_matches = list(re.finditer(r'\s*<tool_call>\s*(\{.*?\})\s*</tool_call>', clean_text, re.DOTALL)) | |
| if json_matches and not tool_calls: | |
| for idx, m in enumerate(json_matches): | |
| raw_json = m.group(1).strip() | |
| try: | |
| parsed = json.loads(raw_json) | |
| fn_name = parsed.get("name", "") | |
| fn_args = parsed.get("arguments", {}) | |
| args_str = json.dumps(fn_args, ensure_ascii=False) if isinstance(fn_args, dict) else str(fn_args) | |
| tool_calls.append({ | |
| "index": idx, | |
| "id": f"call_{uuid.uuid4().hex[:8]}", | |
| "type": "function", | |
| "function": { | |
| "name": fn_name, | |
| "arguments": args_str | |
| } | |
| }) | |
| clean_text = clean_text.replace(m.group(0), "") | |
| except Exception: | |
| pass | |
| # Extract thinking/reasoning if present | |
| reasoning_content = None | |
| think_m = re.search(r'\s*<think>\s*(.*?)\s*</think>', clean_text, re.DOTALL) | |
| if think_m: | |
| reasoning_content = think_m.group(1).strip() | |
| clean_text = clean_text.replace(think_m.group(0), "") | |
| elif "</think>" in clean_text: | |
| parts = clean_text.split("</think>", 1) | |
| reasoning_content = parts[0].replace("<think>", "").strip() | |
| clean_text = parts[1] | |
| elif tool_calls and clean_text.strip(): | |
| # Any text preceding tool calls without <think> tags is reasoning | |
| reasoning_content = clean_text.strip() | |
| clean_text = "" | |
| clean_content = clean_text.strip() | |
| if tool_calls and not clean_content: | |
| clean_content = None | |
| return (tool_calls if tool_calls else None), (reasoning_content if reasoning_content else None), clean_content | |
| def normalize_native_tool_calls(tool_calls): | |
| """Normalize llama-cpp native calls to the OpenAI response contract. | |
| Recent llama-cpp versions parse a GGUF chat template themselves and return | |
| ``message.tool_calls`` with an empty content field. Parsing only content | |
| silently discarded those valid calls and made the API look non-agentic. | |
| """ | |
| if not tool_calls: | |
| return None | |
| normalized = [] | |
| for index, call in enumerate(tool_calls): | |
| function = call.get("function", {}) if isinstance(call, dict) else {} | |
| name = function.get("name") or call.get("name", "") | |
| arguments = function.get("arguments", call.get("arguments", {})) | |
| if not isinstance(arguments, str): | |
| arguments = json.dumps(arguments, ensure_ascii=False) | |
| normalized.append({ | |
| "index": call.get("index", index), | |
| "id": call.get("id") or f"call_{uuid.uuid4().hex[:8]}", | |
| "type": "function", | |
| "function": {"name": name, "arguments": arguments}, | |
| }) | |
| return normalized | |
| def format_openai_messages_for_model(messages): | |
| """Normalize multi-turn OpenAI messages including tool results into prompt format.""" | |
| formatted = [] | |
| for msg in messages: | |
| role = msg.get("role", "user") | |
| content = msg.get("content") | |
| tool_calls = msg.get("tool_calls") | |
| if role == "tool": | |
| formatted.append({ | |
| "role": "user", | |
| "content": f"<tool_response>\n{content or ''}\n</tool_response>" | |
| }) | |
| elif role == "assistant" and tool_calls: | |
| tc_text = "" | |
| for tc in tool_calls: | |
| fn = tc.get("function", {}) | |
| fn_name = fn.get("name", "") | |
| raw_args = fn.get("arguments", "{}") | |
| try: | |
| args_dict = json.loads(raw_args) if isinstance(raw_args, str) else raw_args | |
| except Exception: | |
| args_dict = {} | |
| tc_text += f"\n<tool_call>\n<function={fn_name}>\n" | |
| if isinstance(args_dict, dict): | |
| for k, v in args_dict.items(): | |
| tc_text += f"<parameter={k}>\n{json.dumps(v) if isinstance(v, (dict, list)) else v}\n</parameter>\n" | |
| tc_text += "</function>\n</tool_call>" | |
| combined = (content or "") + tc_text | |
| formatted.append({"role": "assistant", "content": combined.strip()}) | |
| else: | |
| formatted.append({"role": role, "content": content or ""}) | |
| return formatted | |
| def _format_api_error(e: Exception, action: str) -> str: | |
| """Format API errors with clear budget and credit guidance.""" | |
| msg = str(e) | |
| if "402" in msg or "Payment Required" in msg: | |
| return ( | |
| f"{action} notice (402 Payment Required): Your monthly HF Inference API credit " | |
| "($2/month included with HF PRO) has been fully used for this billing period. " | |
| "ZeroGPU tabs (Qwen 3.8 / Gemma LLM Chat) remain 100% free and functional!" | |
| ) | |
| return f"{action} failed: {msg}" | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # TAB 1: Chat & Agent Runner (ZeroGPU Large — 40 min/day) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| def _raw_generate_openai_chat(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None): | |
| """Raw chat completion worker without @spaces.GPU decorator (avoids nested GPU leases in gpu_tracked_call).""" | |
| llm = get_model(model_file) | |
| kwargs = { | |
| "messages": messages, | |
| "max_tokens": (int(max_tokens) if max_tokens not in (None, "") else None), | |
| "temperature": float(temperature), | |
| "top_p": 0.95, | |
| } | |
| if tools: | |
| kwargs["tools"] = tools | |
| kwargs["tool_choice"] = tool_choice or "auto" | |
| return llm.create_chat_completion(**kwargs) | |
| def generate_openai_chat(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None): | |
| return _raw_generate_openai_chat(messages, model_file, temperature, max_tokens, tools=tools, tool_choice=tool_choice) | |
| def generate_openai_chat_stream(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None): | |
| llm = get_model(model_file) | |
| kwargs = { | |
| "messages": messages, | |
| "max_tokens": (int(max_tokens) if max_tokens not in (None, "") else None), | |
| "temperature": float(temperature), | |
| "top_p": 0.95, | |
| "stream": True, | |
| } | |
| if tools: | |
| kwargs["tools"] = tools | |
| kwargs["tool_choice"] = tool_choice or "auto" | |
| for chunk in llm.create_chat_completion(**kwargs): | |
| yield chunk | |
| def custom_chat_handler(user_msg, history, model_file, system_prompt, temperature, max_tokens): | |
| """Rich chat execution with live token telemetry, reasoning, and speed reporting.""" | |
| if not user_msg or not user_msg.strip(): | |
| return history or [], "⚡ *Ready — Enter a prompt to start inference.*", "" | |
| history = list(history or []) | |
| history.append({"role": "user", "content": user_msg.strip()}) | |
| messages = [] | |
| if system_prompt.strip(): | |
| messages.append({"role": "system", "content": system_prompt.strip()}) | |
| for item in history: | |
| messages.append({"role": item.get("role", "user"), "content": item.get("content", "")}) | |
| t0 = time.time() | |
| try: | |
| raw_res = gpu_tracked_call( | |
| "chat", | |
| _raw_generate_openai_chat, | |
| messages, | |
| model_file, | |
| float(temperature), | |
| (int(max_tokens) if max_tokens not in (None, "") else None), | |
| model=model_file, | |
| ) | |
| t1 = time.time() | |
| elapsed = max(0.01, t1 - t0) | |
| raw_content = raw_res["choices"][0]["message"].get("content") or "" | |
| tool_calls, reasoning, clean = parse_model_tool_calls(raw_content) | |
| formatted_bot = "" | |
| if reasoning: | |
| formatted_bot += f"<details open><summary>🧠 <b>Deep Thinking & Reasoning</b></summary>\n\n```markdown\n{reasoning}\n```\n</details>\n\n" | |
| if tool_calls: | |
| formatted_bot += f"<details open><summary>🛠️ <b>Executed Tool Calls ({len(tool_calls)})</b></summary>\n\n```json\n{json.dumps(tool_calls, indent=2)}\n```\n</details>\n\n" | |
| if clean: | |
| formatted_bot += clean | |
| elif not reasoning and not tool_calls: | |
| formatted_bot += raw_content | |
| history.append({"role": "assistant", "content": formatted_bot}) | |
| # Telemetry calculations | |
| raw_usage = raw_res.get("usage", {}) | |
| prompt_toks = raw_usage.get("prompt_tokens") or sum(max(1, int(len(m["content"].split()) * 1.3)) for m in messages) | |
| comp_toks = raw_usage.get("completion_tokens") or max(1, int(len(raw_content.split()) * 1.3)) | |
| tot_toks = prompt_toks + comp_toks | |
| tps = comp_toks / elapsed | |
| ctx_pct = (tot_toks / 262144) * 100 | |
| hud_md = ( | |
| f"<div style='display: flex; flex-wrap: wrap; gap: 12px; font-size: 0.85rem; padding: 8px 12px; " | |
| f"background: rgba(30, 41, 59, 0.7); border-radius: 8px; border: 1px solid rgba(255,255,255,0.1); font-family: monospace;'>" | |
| f"<span>⚡ <b>{tps:.1f} t/s</b></span>" | |
| f"<span>⏱️ <b>{elapsed:.2f}s</b></span>" | |
| f"<span>📥 Prompt: <b>{prompt_toks}</b></span>" | |
| f"<span>📤 Output: <b>{comp_toks}</b></span>" | |
| f"<span>🧠 Context: <b>{tot_toks:,} / 262,144 ({ctx_pct:.1f}%)</b></span>" | |
| f"<span style='color: #4ade80;'>● ZeroGPU Large (48GB)</span>" | |
| f"</div>" | |
| ) | |
| return history, hud_md, "" | |
| except Exception as e: | |
| history.append({"role": "assistant", "content": f"❌ **Inference Error:** {str(e)}"}) | |
| return history, f"⚠️ *Execution error after {time.time()-t0:.2f}s: {str(e)}*", "" | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # TAB 2: Vision & Multimodal OCR | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| def analyze_vision(image_input, prompt_text): | |
| """Analyze images, UI screenshots, code diagrams or documents.""" | |
| if image_input is None: | |
| raise gr.Error("Please upload or capture an image first.") | |
| if not HF_TOKEN: | |
| raise gr.Error("Set HF_TOKEN in Space secrets to use the Vision API.") | |
| prompt = prompt_text.strip() or "Describe this image in detail and extract all visible text and code." | |
| try: | |
| response = api_client.chat_completion( | |
| messages=[ | |
| { | |
| "role": "user", | |
| "content": [ | |
| {"type": "text", "text": prompt}, | |
| {"type": "image_url", "image_url": {"url": image_input if isinstance(image_input, str) else image_input}}, | |
| ], | |
| } | |
| ], | |
| model=VISION_MODEL, | |
| max_tokens=1024, | |
| ) | |
| return response.choices[0].message.content | |
| except Exception as e: | |
| try: | |
| return api_client.image_to_text(image=image_input, model="Salesforce/blip-image-captioning-large") | |
| except Exception: | |
| raise gr.Error(_format_api_error(e, "Vision analysis")) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # ZERO-GPU VIDEO GENERATION PIPELINE (40 min/day A100 Quota - $0 API Cost) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # Video diffusion pipelines are heavy: CogVideoX-5B / LTX-Video weigh 6-15 GB. | |
| # Weights are DOWNLOADED OUTSIDE the GPU lease (CPU-only snapshot_download in | |
| # the web worker), then only the denoise step leases GPU. A dynamic 120 s | |
| # lease cannot absorb a cold 10 GB download — it aborts with "GPU task | |
| # aborted" mid-load. Warm-pipeline inference fits comfortably in 240 s. | |
| VIDEO_GPU_LEASE_SECONDS = int(os.environ.get("FLOW_VIDEO_GPU_LEASE_S", "240")) | |
| def _prefetch_video_pipeline(repo_id: str) -> None: | |
| """Download pipeline weights into the HF cache with NO GPU lease active. | |
| ZeroGPU bills only GPU-leased code; downloads and CPU work in the web | |
| process are free. A snapshot that fails parsing (truncated spiece.model | |
| etc.) never self-heals, so it is purged and re-downloaded once. | |
| """ | |
| from huggingface_hub import snapshot_download | |
| try: | |
| snapshot_download(repo_id) | |
| except Exception as exc: | |
| import re as _re | |
| m = _re.search(r"models--([\w.\-]+)--([\w.\-]+)", str(exc)) | |
| if m and ("Error parsing" in str(exc) or "tokenize" in str(exc).lower()): | |
| broken = os.path.join("/root/.cache/huggingface/hub", f"models--{m.group(1)}--{m.group(2)}") | |
| if os.path.exists(broken): | |
| print(f"[ZeroGPU Video] Purging corrupted HF cache: {broken}", flush=True) | |
| shutil.rmtree(broken, ignore_errors=True) | |
| snapshot_download(repo_id) | |
| else: | |
| raise | |
| def generate_zerogpu_video( | |
| prompt: str, | |
| negative_prompt: str = "", | |
| model_choice: str = "ZeroScope v2 (576w High-Res)", | |
| num_frames: int = 16, | |
| fps: int = 8, | |
| guidance_scale: float = 7.5, | |
| seed: int = -1 | |
| ): | |
| """Generate dynamic MP4 video using ZeroGPU open-weights models.""" | |
| if not prompt.strip(): | |
| raise gr.Error("Please enter a video prompt.") | |
| repo_id = VIDEO_MODELS.get(model_choice, "cerspense/zeroscope_v2_576w") | |
| out_video_path = f"/tmp/zerogpu_video_{int(time.time())}_{abs(hash(prompt)) % 10000}.mp4" | |
| # Weights must be resident BEFORE the GPU lease starts; loading inside a | |
| # short lease aborts on any cold download. | |
| _prefetch_video_pipeline(repo_id) | |
| try: | |
| import torch | |
| from diffusers import DiffusionPipeline, DPMSolverMultistepScheduler | |
| device = "cuda" if torch.cuda.is_available() else "cpu" | |
| dtype = torch.float16 if device == "cuda" else torch.float32 | |
| if repo_id not in _video_pipeline_cache: | |
| pipe = DiffusionPipeline.from_pretrained(repo_id, torch_dtype=dtype) | |
| if hasattr(pipe, "scheduler"): | |
| try: | |
| pipe.scheduler = DPMSolverMultistepScheduler.from_config(pipe.scheduler.config) | |
| except Exception: | |
| pass | |
| if hasattr(pipe, "enable_model_cpu_offload") and device == "cuda": | |
| pipe.enable_model_cpu_offload() | |
| else: | |
| pipe = pipe.to(device) | |
| _video_pipeline_cache[repo_id] = pipe | |
| else: | |
| pipe = _video_pipeline_cache[repo_id] | |
| actual_seed = seed if (seed and int(seed) >= 0) else random.randint(0, 2**31 - 1) | |
| generator = torch.Generator(device=device).manual_seed(actual_seed) | |
| video_frames = pipe( | |
| prompt=prompt.strip(), | |
| negative_prompt=negative_prompt.strip() if negative_prompt else None, | |
| num_inference_steps=24, | |
| guidance_scale=float(guidance_scale), | |
| num_frames=int(num_frames), | |
| generator=generator | |
| ).frames[0] | |
| import numpy as np | |
| import tempfile | |
| from PIL import Image | |
| ffmpeg_bin = shutil.which("ffmpeg") or "/usr/bin/ffmpeg" | |
| with tempfile.TemporaryDirectory() as tmpdir: | |
| for idx, frame in enumerate(video_frames): | |
| if isinstance(frame, np.ndarray): | |
| f_arr = frame | |
| if f_arr.dtype in (np.float32, np.float64, np.float16): | |
| if f_arr.max() <= 1.0: | |
| f_arr = (f_arr * 255.0).clip(0, 255).astype(np.uint8) | |
| else: | |
| f_arr = f_arr.clip(0, 255).astype(np.uint8) | |
| img = Image.fromarray(f_arr) | |
| else: | |
| img = frame | |
| img.save(os.path.join(tmpdir, f"frame_{idx:05d}.png")) | |
| subprocess.run( | |
| [ | |
| ffmpeg_bin, "-y", "-framerate", str(int(fps)), | |
| "-i", os.path.join(tmpdir, "frame_%05d.png"), | |
| "-c:v", "libx264", "-pix_fmt", "yuv420p", | |
| out_video_path | |
| ], | |
| check=True, | |
| capture_output=True | |
| ) | |
| return out_video_path | |
| except Exception as exc: | |
| print(f"[ZeroGPU Video] Direct pipeline exception: {exc}, running ffmpeg dynamic visualizer fallback...", flush=True) | |
| try: | |
| from engine.model_dispatcher import ModelDispatcher | |
| disp = ModelDispatcher() | |
| res = disp.generate_image(prompt=prompt, aspect_ratio="16:9") | |
| img_path = res.get("filepath") | |
| if img_path and os.path.exists(img_path): | |
| ffmpeg_bin = shutil.which("ffmpeg") or "/opt/homebrew/bin/ffmpeg" | |
| dur = max(3, int(int(num_frames) / max(1, int(fps)))) | |
| subprocess.run( | |
| [ | |
| ffmpeg_bin, "-y", "-loop", "1", "-i", img_path, | |
| "-vf", f"fps={fps},scale=768:432,zoompan=z='min(zoom+0.0015,1.15)':d={dur*fps}:s=768x432", | |
| "-c:v", "libx264", "-t", str(dur), "-pix_fmt", "yuv420p", | |
| out_video_path | |
| ], | |
| capture_output=True, | |
| timeout=20 | |
| ) | |
| if os.path.exists(out_video_path) and os.path.getsize(out_video_path) > 0: | |
| return out_video_path | |
| except Exception as e2: | |
| print(f"[ZeroGPU Video] Fallback failed: {e2}") | |
| raise gr.Error(f"ZeroGPU Video error: {exc}") | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # ZERO-GPU MUSIC & AUDIO GENERATION PIPELINE (40 min/day A100 Quota - $0 Cost) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| def _raw_generate_zerogpu_music( | |
| prompt: str, | |
| lyrics: str = "", | |
| model_choice: str = "MiniMax Music 3 (full song + vocals)", | |
| duration_seconds: int = 60, | |
| guidance_scale: float = 3.0, | |
| temperature: float = 1.0, | |
| seed: int = 7 | |
| ): | |
| """Internal raw worker for ZeroGPU audio synthesis (avoids nested @spaces.GPU calls).""" | |
| if not prompt.strip() and not lyrics.strip(): | |
| raise gr.Error("Please enter a music description or lyrics.") | |
| repo_id = AUDIO_MUSIC_MODELS.get(model_choice, "MiniMaxAI/MiniMax-Music3") | |
| # Clamp to max 120 seconds to guarantee completion within ZeroGPU quota lease | |
| dur = max(5, min(120, int(duration_seconds or 60))) | |
| out_audio_path = f"/tmp/zerogpu_music_{int(time.time())}_{abs(hash(prompt + lyrics)) % 10000}.wav" | |
| try: | |
| import torch | |
| import soundfile as sf | |
| # Official MiniMax Music3 path from its model card. It supports lyrics + | |
| # detailed music description and produces complete vocal songs up to 5 min. | |
| if repo_id == "MiniMaxAI/MiniMax-Music3": | |
| from diffusers import ModularPipeline | |
| import numpy as np | |
| if repo_id not in _audio_pipeline_cache: | |
| pipe = ModularPipeline.from_pretrained(repo_id) | |
| pipe.load_components(dtype=torch.bfloat16) | |
| pipe.to("cuda") | |
| _audio_pipeline_cache[repo_id] = pipe | |
| else: | |
| pipe = _audio_pipeline_cache[repo_id] | |
| raw_output = pipe( | |
| prompt=prompt.strip(), | |
| lyrics=lyrics.strip(), | |
| audio_duration=float(dur), | |
| generator=torch.Generator("cuda").manual_seed(int(seed)), | |
| output="audios", | |
| ) | |
| audio = raw_output[0] | |
| if hasattr(audio, "detach"): | |
| audio_np = audio.detach().cpu().float().numpy() | |
| elif isinstance(audio, np.ndarray): | |
| audio_np = audio.astype(np.float32) | |
| else: | |
| audio_np = np.asarray(audio, dtype=np.float32) | |
| if audio_np.ndim > 1 and audio_np.shape[0] < audio_np.shape[1]: | |
| audio_np = audio_np.T | |
| sr = getattr(pipe, "sampling_rate", 44100) | |
| sf.write(out_audio_path, audio_np, sr) | |
| return out_audio_path | |
| # YuE2-3B: AR-NAR MoT + flow matching, 48kHz stereo, editable scores. | |
| # The repo is NOT a diffusers pipeline (no model_index.json); it ships the | |
| # `yue2` infer wheel (YuE2Pipeline). Install with --no-deps (its pins for | |
| # torch/hub conflict with the Space's diffusers set; core only needs | |
| # torch + transformers + tiktoken + soundfile). | |
| if repo_id == "m-a-p/YuE2-3B": | |
| import numpy as np | |
| from yue2 import YuE2Pipeline | |
| if repo_id not in _audio_pipeline_cache: | |
| pipe = YuE2Pipeline.from_pretrained(repo_id, progress=False) | |
| _audio_pipeline_cache[repo_id] = pipe | |
| else: | |
| pipe = _audio_pipeline_cache[repo_id] | |
| song = pipe(style=prompt.strip(), lyrics=lyrics.strip(), cot="full", seed=int(seed)) | |
| sf.write(out_audio_path, song.audio, song.sample_rate) | |
| return out_audio_path | |
| # ACE-Step v1.5 XL Turbo: Apache 2.0, 48kHz, commercial-safe, <4GB VRAM. | |
| # Output is AudioPipelineOutput.audios (channels, samples); rate is | |
| # pipe.sample_rate. Turbo is guidance-distilled so CFG is ignored anyway. | |
| if repo_id == "ACE-Step/acestep-v15-xl-turbo-diffusers": | |
| from diffusers import AceStepPipeline | |
| import numpy as np | |
| if repo_id not in _audio_pipeline_cache: | |
| pipe = AceStepPipeline.from_pretrained(repo_id, torch_dtype=torch.bfloat16) | |
| pipe.to("cuda") | |
| pipe.vae.enable_tiling() # bound VAE decode memory for longer clips | |
| _audio_pipeline_cache[repo_id] = pipe | |
| else: | |
| pipe = _audio_pipeline_cache[repo_id] | |
| raw_output = pipe( | |
| prompt=prompt.strip(), | |
| lyrics=lyrics.strip(), | |
| audio_duration=float(dur), | |
| generator=torch.Generator("cuda").manual_seed(int(seed)), | |
| ) | |
| audio_tensor = raw_output.audios[0] # (channels, samples) | |
| audio_np = audio_tensor.T.cpu().float().numpy() | |
| sf.write(out_audio_path, audio_np, pipe.sample_rate) | |
| return out_audio_path | |
| # Stable Audio 3 uses its own pipeline and may require HF access approval. | |
| if repo_id == "stabilityai/stable-audio-3-medium": | |
| # The public Stability release currently uses the separate | |
| # `stable_audio_3` package, not a Diffusers StableAudio3Pipeline. | |
| # Do not pretend this selector works or silently substitute audio. | |
| raise RuntimeError( | |
| "Stable Audio 3 is not enabled in this Space yet: its official " | |
| "stable_audio_3 runtime is not installed. Select MiniMax Music 3." | |
| ) | |
| # MusicGen is explicitly instrumental and does not reliably sing lyrics. | |
| if repo_id.startswith("facebook/musicgen"): | |
| from transformers import AutoProcessor, MusicgenForConditionalGeneration | |
| if repo_id not in _audio_pipeline_cache: | |
| processor = AutoProcessor.from_pretrained(repo_id) | |
| model = MusicgenForConditionalGeneration.from_pretrained(repo_id, torch_dtype=torch.float16).to("cuda") | |
| _audio_pipeline_cache[repo_id] = (processor, model) | |
| processor, model = _audio_pipeline_cache[repo_id] | |
| inputs = processor(text=[prompt.strip()], padding=True, return_tensors="pt").to("cuda") | |
| audio_values = model.generate( | |
| **inputs, do_sample=True, guidance_scale=float(guidance_scale), | |
| max_new_tokens=min(1500, int(dur * 50)), temperature=float(temperature) | |
| ) | |
| sf.write(out_audio_path, audio_values[0, 0].detach().cpu().numpy(), model.config.audio_encoder.sampling_rate) | |
| return out_audio_path | |
| raise gr.Error(f"Model '{repo_id}' is not wired for full-song generation yet. Choose MiniMax Music 3, YuE2-3B, or ACE-Step v1.5.") | |
| except Exception as exc: | |
| raise gr.Error(f"ZeroGPU model '{repo_id}' failed: {exc}. No MIDI/procedural fallback was used.") | |
| def generate_zerogpu_music( | |
| prompt: str, | |
| lyrics: str = "", | |
| model_choice: str = "MiniMax Music 3 (full song + vocals)", | |
| duration_seconds: int = 60, | |
| guidance_scale: float = 3.0, | |
| temperature: float = 1.0, | |
| seed: int = 7 | |
| ): | |
| """Generate real full-song audio on ZeroGPU. Procedural MIDI fallback is never used here.""" | |
| return _raw_generate_zerogpu_music( | |
| prompt=prompt, | |
| lyrics=lyrics, | |
| model_choice=model_choice, | |
| duration_seconds=duration_seconds, | |
| guidance_scale=guidance_scale, | |
| temperature=temperature, | |
| seed=seed | |
| ) | |
| # --- Moldovan Creative Studio coupling removed (2026-08) --- | |
| # The Moldovan Creative Media & Personas tab lived here (create_moldovan_song_handler, | |
| # chat_moldovan_persona_handler, convert_dialect_handler) and drove lyric synthesis / | |
| # persona chat / dialect conversion via moldovan-qwen/ (engine.media_creator, | |
| # personas.persona_engine, linguistics.dialect_converter). Those engines now run | |
| # exclusively in the local Moldovan Media Studio (moldovan-qwen/studio_server.py, | |
| # :8090) and studio-pb (PocketBase :8096); the ZeroGPU Space no longer imports them. | |
| def transcribe_audio(audio_input): | |
| """Transcribe audio with Whisper Large v3.""" | |
| if audio_input is None: | |
| raise gr.Error("Please record or upload audio.") | |
| if not HF_TOKEN: | |
| raise gr.Error("Set HF_TOKEN in Space secrets to use Whisper.") | |
| try: | |
| result = api_client.automatic_speech_recognition( | |
| audio=audio_input, | |
| model=WHISPER_MODEL, | |
| ) | |
| return result.text if hasattr(result, "text") else str(result) | |
| except Exception as e: | |
| raise gr.Error(_format_api_error(e, "Whisper transcription")) | |
| def generate_tts(text_input): | |
| """Synthesize natural ro-RO speech with Edge-TTS Neural (no banned TTS models).""" | |
| if not text_input.strip(): | |
| raise gr.Error("Please enter text to synthesize.") | |
| try: | |
| import edge_tts | |
| import asyncio | |
| import tempfile, os as _os | |
| tmp = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) | |
| tmp.close() | |
| async def _run(): | |
| await edge_tts.Communicate(text_input.strip(), "ro-RO-EmilNeural").save(tmp.name) | |
| asyncio.run(_run()) | |
| with open(tmp.name, "rb") as f: | |
| return f.read() | |
| except Exception as e: | |
| raise gr.Error(f"Edge-TTS ro-RO synthesis error: {e}") | |
| # ─── Breeze TTS 2 (SOTA experimental; non-commercial license, art use) ──────── | |
| BREEZE_LANGUAGE_HINTS = { | |
| "auto": "Detect the language and speak naturally with clear articulation.", | |
| "en": "Speak in English with a natural, engaging storyteller delivery.", | |
| "ro": "Speak clearly. Textul este în limba română: articulare clară, intonație naturală și expresivă.", | |
| } | |
| BREEZE_EMOTION_HINTS = { | |
| "neutral": "Neutral, balanced tone.", | |
| "warm": "Warm, friendly and inviting tone.", | |
| "energetic": "Energetic, upbeat and lively delivery.", | |
| "serious": "Serious, restrained and deliberate tone.", | |
| "sad": "Melancholic, soft and subdued delivery.", | |
| "excited": "Excited, joyful and high-energy delivery.", | |
| "narrator": "Cinematic narrator voice: deep, slow, suspenseful storytelling.", | |
| } | |
| def _resolve_breeze_instruction(language: str, emotion: str, custom_instruction: str) -> str: | |
| """Compose the natural-language voice-direction instruction for Breeze.""" | |
| if custom_instruction.strip(): | |
| return custom_instruction.strip() | |
| lang_hint = BREEZE_LANGUAGE_HINTS.get((language or "auto").lower(), BREEZE_LANGUAGE_HINTS["auto"]) | |
| emo_hint = BREEZE_EMOTION_HINTS.get((emotion or "neutral").lower(), BREEZE_EMOTION_HINTS["neutral"]) | |
| return f"{emo_hint} {lang_hint}" | |
| def _load_breeze_runtime(model_dir: Any): | |
| """Load (and cache across lease workers) the Breeze TTS runtime on the GPU worker.""" | |
| import torch | |
| if "runtime" in _breeze_runtime_cache: | |
| return _breeze_runtime_cache["runtime"] | |
| # Weight download must already be done by the startup CPU pre-cache thread. | |
| from breeze_infer.runtime import load_runtime, update_generation_config_for_breeze | |
| from models.fast_streaming import FastBreezeStreamingRuntime, FastStreamingConfig | |
| tokenizer, model, audio_tokenizer = load_runtime( | |
| model_dir, | |
| device="cuda" if torch.cuda.is_available() else "cpu", | |
| attn_implementation="eager", | |
| ) | |
| update_generation_config_for_breeze(model) | |
| config = FastStreamingConfig( | |
| max_new_tokens=1500, | |
| max_seq_len=2048, | |
| fast_all=False, # eager: no CUDA-graph warmup cost inside a short ZeroGPU lease | |
| repetition_penalty=1.1, | |
| ) | |
| runtime = FastBreezeStreamingRuntime(model, audio_tokenizer, config, tokenizer=tokenizer) | |
| _breeze_runtime_cache["runtime"] = runtime | |
| _breeze_runtime_cache["tokenizer"] = tokenizer | |
| _breeze_runtime_cache["model"] = model | |
| _breeze_runtime_cache["audio_tokenizer"] = audio_tokenizer | |
| return runtime | |
| def generate_zerogpu_breeze_tts( | |
| text: str, | |
| instruction: str = "", | |
| language: str = "auto", | |
| emotion: str = "neutral", | |
| cfg_scale: float = 1.0, | |
| seed: int = 42, | |
| ref_audio_path: Optional[str] = None, | |
| ref_text: str = "" | |
| ): | |
| """Breeze TTS 2 synthesis on ZeroGPU. Returns path to generated 24kHz WAV. | |
| Voice design (no reference) or voice direction (with reference audio + transcript). | |
| """ | |
| if not _breeze_available(): | |
| raise gr.Error( | |
| "Breeze TTS 2 is temporarily disabled: its qwen-tts dependency pins " | |
| "transformers==4.57.3 which breaks MiniMax Music 3 (diffusers hub>=1.23). " | |
| "Use MiniMax Music 3 for music; Breeze returns when upstream unpins." | |
| ) | |
| if not text.strip(): | |
| raise gr.Error("Please enter text to synthesize.") | |
| out_path = f"/tmp/breeze_tts_{int(time.time())}_{abs(hash(text)) % 10000}.wav" | |
| try: | |
| from huggingface_hub import snapshot_download | |
| import numpy as np | |
| import soundfile as sf | |
| from breeze_infer.templates import get_template, prepare_inputs | |
| from breeze_infer.runtime import set_all_seeds | |
| # Resolve checkpoint directory from the persistent cache. | |
| cache_dir = "/data/hub" if (os.path.exists("/data") and os.access("/data", os.W_OK)) else os.environ.get("HF_HUB_CACHE", None) | |
| model_dir = snapshot_download(repo_id=BREEZE_MODEL_ID, cache_dir=cache_dir, resume_download=True) | |
| runtime = _load_breeze_runtime(model_dir) | |
| tokenizer = _breeze_runtime_cache["tokenizer"] | |
| audio_tokenizer = _breeze_runtime_cache["audio_tokenizer"] | |
| breeze_model = _breeze_runtime_cache["model"] | |
| request = { | |
| "id": "breeze-request", | |
| "text": text.strip(), | |
| "instruction": _resolve_breeze_instruction(language, emotion, instruction), | |
| "speaker": "S0", | |
| } | |
| template_name = "tts_instruction" | |
| if ref_audio_path: | |
| if not os.path.isfile(ref_audio_path): | |
| raise gr.Error(f"Reference audio not found: {ref_audio_path}") | |
| if not (ref_text or "").strip(): | |
| raise gr.Error("Reference transcript (exact text spoken in the audio) is required for voice cloning.") | |
| request["ref_audio_path"] = str(ref_audio_path) | |
| request["ref_text"] = ref_text.strip() | |
| template_name = "ref_edit_tata" | |
| effective_seed = int(seed) if seed is not None and int(seed) >= 0 else random.randint(0, 2**31 - 1) | |
| set_all_seeds(effective_seed) | |
| inputs = prepare_inputs( | |
| tokenizer, | |
| audio_tokenizer, | |
| breeze_model, | |
| [request], | |
| get_template(template_name), | |
| guidance_scale=float(cfg_scale) if cfg_scale and float(cfg_scale) > 0 else 1.0, | |
| guidance_scale_ref=None, | |
| guidance_scale_ins=None, | |
| ) | |
| chunks = [] | |
| for chunk in runtime.iter_audio_chunks(inputs, request_id="breeze-request", seed=effective_seed): | |
| chunks.append(np.asarray(chunk.audio)) | |
| if not chunks: | |
| raise gr.Error("Breeze synthesis produced no audio.") | |
| audio = np.concatenate(chunks).astype(np.float32) | |
| sf.write(out_path, audio, samplerate=runtime.sample_rate, subtype="PCM_16") | |
| return out_path | |
| except gr.Error: | |
| raise | |
| except Exception as e: | |
| raise gr.Error(f"Breeze TTS 2 synthesis error: {e}") | |
| def generate_breeze_tts_handler(text, language, emotion, instruction, seed, cfg_scale=1.0, ref_audio=None, ref_text=""): | |
| """Gradio handler wrapping the ZeroGPU Breeze worker.""" | |
| return generate_zerogpu_breeze_tts( | |
| text=text, | |
| instruction=instruction or "", | |
| language=language, | |
| emotion=emotion, | |
| seed=int(seed) if seed is not None else -1, | |
| cfg_scale=float(cfg_scale) if cfg_scale is not None else 1.0, | |
| ref_audio_path=ref_audio, | |
| ref_text=ref_text or "", | |
| ) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # TAB 5: Voice-to-Art Pipeline | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| def voice_to_art(audio_input, model_file, art_style): | |
| """Whisper STT -> Qwen creative prompt expansion (image rendering removed by policy).""" | |
| if audio_input is None: | |
| raise gr.Error("Please record or upload audio first.") | |
| if not HF_TOKEN: | |
| raise gr.Error("Set HF_TOKEN in Space secrets.") | |
| try: | |
| transcription = api_client.automatic_speech_recognition( | |
| audio=audio_input, | |
| model=WHISPER_MODEL, | |
| ) | |
| raw_text = transcription.text if hasattr(transcription, "text") else str(transcription) | |
| except Exception as e: | |
| raise gr.Error(_format_api_error(e, "Whisper transcription")) | |
| if not raw_text.strip(): | |
| raise gr.Error("Could not understand the audio.") | |
| llm = get_model(model_file) | |
| style_hint = f" in {art_style} style" if art_style.strip() else "" | |
| expand_prompt = ( | |
| f"You are a master image prompt engineer. The user said: \"{raw_text}\"\n\n" | |
| f"Write a single, highly detailed, vivid creative image-generation prompt{style_hint}. " | |
| f"Include composition, cinematic lighting, color palette, mood, and fine details. " | |
| f"Output ONLY the prompt, nothing else. Max 100 words." | |
| ) | |
| response = llm.create_chat_completion( | |
| messages=[{"role": "user", "content": expand_prompt}], | |
| max_tokens=256, | |
| temperature=0.85, | |
| top_p=0.95, | |
| ) | |
| art_prompt = response["choices"][0]["message"]["content"].strip() | |
| _, _, clean_art_prompt = parse_model_tool_calls(art_prompt) | |
| final_prompt = clean_art_prompt or art_prompt | |
| # Image generation removed by policy (Space scope: Music + Music Videos only). | |
| # Pipeline now stops at the expanded creative prompt for use in music/video workflows. | |
| return raw_text, final_prompt | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # TAB 6: Embeddings Lab | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| def compute_similarity(text_a, text_b): | |
| """Compute 1024-dim dense embeddings and cosine similarity using BGE-M3.""" | |
| if not text_a.strip() or not text_b.strip(): | |
| raise gr.Error("Please enter both Text A and Text B.") | |
| if not HF_TOKEN: | |
| raise gr.Error("Set HF_TOKEN in Space secrets to use Embeddings.") | |
| try: | |
| emb_a = api_client.feature_extraction(text=text_a.strip(), model=EMBEDDING_MODEL) | |
| emb_b = api_client.feature_extraction(text=text_b.strip(), model=EMBEDDING_MODEL) | |
| vec_a = emb_a[0] if isinstance(emb_a, list) and isinstance(emb_a[0], list) else emb_a | |
| vec_b = emb_b[0] if isinstance(emb_b, list) and isinstance(emb_b[0], list) else emb_b | |
| dot = sum(a * b for a, b in zip(vec_a, vec_b)) | |
| norm_a = math.sqrt(sum(a * a for a in vec_a)) | |
| norm_b = math.sqrt(sum(b * b for b in vec_b)) | |
| similarity = dot / (norm_a * norm_b) if (norm_a > 0 and norm_b > 0) else 0.0 | |
| score_percent = round(similarity * 100, 2) | |
| interp = ( | |
| "🟢 Identical / Paraphrase" if score_percent > 85 else | |
| "🟡 Highly Related" if score_percent > 65 else | |
| "🟠 Moderately Related" if score_percent > 40 else | |
| "🔴 Distinct / Unrelated" | |
| ) | |
| dim_len = len(vec_a) | |
| vector_preview_a = str(vec_a[:5])[:-1] + ", ...]" | |
| vector_preview_b = str(vec_b[:5])[:-1] + ", ...]" | |
| report = ( | |
| f"### 🎯 Cosine Similarity: **{score_percent}%** ({interp})\n\n" | |
| f"<div style='height: 8px; width: 100%; background: #334155; border-radius: 4px; overflow: hidden; margin-bottom: 16px;'>" | |
| f"<div style='height: 100%; width: {score_percent}%; background: linear-gradient(90deg, #38bdf8, #818cf8);'></div>" | |
| f"</div>\n\n" | |
| f"- **Embedding Model:** `{EMBEDDING_MODEL}`\n" | |
| f"- **Vector Dimensionality:** `{dim_len}` float32 elements\n\n" | |
| f"**Vector Preview A:** `{vector_preview_a}`\n\n" | |
| f"**Vector Preview B:** `{vector_preview_b}`" | |
| ) | |
| return report | |
| except Exception as e: | |
| raise gr.Error(_format_api_error(e, "Embedding calculation")) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # OpenAI-Compatible API Endpoints | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| fastapi_app = FastAPI(title="ZeroGPU Private OpenAI API Hub", version="2.0.0") | |
| def _authorize_api_request(request: Request) -> None: | |
| """Require bearer auth only if FLOW_API_KEY is explicitly set in Space secrets.""" | |
| expected = os.environ.get("FLOW_API_KEY") | |
| if not expected: | |
| return | |
| authorization = request.headers.get("authorization", "") | |
| scheme, _, supplied = authorization.partition(" ") | |
| if scheme.lower() != "bearer" or not supplied or not hmac.compare_digest(supplied, expected): | |
| raise HTTPException(status_code=401, detail="Invalid or missing bearer token.") | |
| async def openai_chat_completions(request: Request): | |
| _authorize_api_request(request) | |
| try: | |
| body = await request.json() | |
| except Exception: | |
| raise HTTPException(status_code=400, detail="Invalid JSON body") | |
| messages = body.get("messages", []) | |
| if not messages: | |
| raise HTTPException(status_code=400, detail="Field 'messages' is required.") | |
| model_req = body.get("model", "") | |
| choices = list_gguf_files() | |
| if not choices: | |
| raise HTTPException(status_code=500, detail="No GGUF models available in Space storage.") | |
| selected_model = resolve_model(model_req, choices) | |
| temperature = float(body.get("temperature", 0.7)) | |
| raw_max_tokens = body.get("max_tokens") | |
| max_tokens = int(raw_max_tokens) if raw_max_tokens not in (None, "") else None | |
| formatted_msgs = format_openai_messages_for_model(messages) | |
| tools = body.get("tools") | |
| tool_choice = body.get("tool_choice") | |
| stream = bool(body.get("stream", False)) | |
| # Fetch weights on CPU first: a cold 17 GB download would otherwise run | |
| # inside the GPU lease and blow the 120 s budget. | |
| ensure_model_available(selected_model) | |
| try: | |
| raw_res = gpu_tracked_call( | |
| "chat", | |
| _raw_generate_openai_chat, | |
| formatted_msgs, selected_model, temperature, max_tokens, tools, tool_choice, | |
| model=selected_model, | |
| ) | |
| except Exception as e: | |
| err_msg = str(e) | |
| lowered = err_msg.lower() | |
| if "limit" in lowered or "quota" in lowered: | |
| # Report the real platform ceiling with the numbers that caused it, | |
| # so a client can wait out the window instead of retrying a doomed | |
| # credential/model combination. | |
| raise HTTPException( | |
| status_code=429, | |
| detail={ | |
| "error": "zerogpu_quota_exhausted", | |
| "message": ( | |
| "Hugging Face ZeroGPU daily allowance is exhausted for " | |
| "today. Quota frees up on the rolling daily window." | |
| ), | |
| "upstream_detail": err_msg, | |
| "requested_lease_seconds": requested_gpu_lease_seconds(), | |
| "remaining_quota_seconds": round(GPU_USAGE.remaining_seconds(), 1), | |
| "daily_quota_seconds": GPU_USAGE.DAILY_QUOTA_SECONDS, | |
| }, | |
| ) | |
| raise HTTPException(status_code=500, detail=f"ZeroGPU inference error: {err_msg}") | |
| raw_message = raw_res["choices"][0]["message"] | |
| raw_content = raw_message.get("content") or "" | |
| raw_finish = raw_res["choices"][0].get("finish_reason", "stop") | |
| tool_calls, reasoning, clean_content = parse_model_tool_calls(raw_content) | |
| native_tool_calls = normalize_native_tool_calls(raw_message.get("tool_calls")) | |
| if native_tool_calls: | |
| tool_calls = native_tool_calls | |
| # llama-cpp has already removed its native tool block from content. | |
| # Preserve any actual assistant text, but do not mislabel it reasoning. | |
| if clean_content is None and raw_content: | |
| clean_content = raw_content.strip() or None | |
| if stream: | |
| async def event_generator(): | |
| cid = f"chatcmpl-{int(time.time()*1000)}" | |
| created_ts = int(time.time()) | |
| # Step 1: Stream reasoning chunk if present | |
| if reasoning: | |
| chunk1 = { | |
| "id": cid, | |
| "object": "chat.completion.chunk", | |
| "created": created_ts, | |
| "model": selected_model, | |
| "choices": [ | |
| { | |
| "index": 0, | |
| "delta": { | |
| "role": "assistant", | |
| "reasoning_content": reasoning | |
| }, | |
| "finish_reason": None | |
| } | |
| ] | |
| } | |
| yield f"data: {json.dumps(chunk1)}\n\n" | |
| # Step 2: Stream tool calls or text content | |
| if tool_calls: | |
| chunk2 = { | |
| "id": cid, | |
| "object": "chat.completion.chunk", | |
| "created": created_ts, | |
| "model": selected_model, | |
| "choices": [ | |
| { | |
| "index": 0, | |
| "delta": { | |
| "tool_calls": tool_calls | |
| }, | |
| "finish_reason": None | |
| } | |
| ] | |
| } | |
| yield f"data: {json.dumps(chunk2)}\n\n" | |
| chunk3 = { | |
| "id": cid, | |
| "object": "chat.completion.chunk", | |
| "created": created_ts, | |
| "model": selected_model, | |
| "choices": [ | |
| { | |
| "index": 0, | |
| "delta": {}, | |
| "finish_reason": "tool_calls" | |
| } | |
| ] | |
| } | |
| yield f"data: {json.dumps(chunk3)}\n\n" | |
| else: | |
| if clean_content: | |
| chunk_text = { | |
| "id": cid, | |
| "object": "chat.completion.chunk", | |
| "created": created_ts, | |
| "model": selected_model, | |
| "choices": [ | |
| { | |
| "index": 0, | |
| "delta": { | |
| "role": "assistant", | |
| "content": clean_content | |
| }, | |
| "finish_reason": None | |
| } | |
| ] | |
| } | |
| yield f"data: {json.dumps(chunk_text)}\n\n" | |
| chunk_finish = { | |
| "id": cid, | |
| "object": "chat.completion.chunk", | |
| "created": created_ts, | |
| "model": selected_model, | |
| "choices": [ | |
| { | |
| "index": 0, | |
| "delta": {}, | |
| "finish_reason": "stop" | |
| } | |
| ] | |
| } | |
| yield f"data: {json.dumps(chunk_finish)}\n\n" | |
| yield "data: [DONE]\n\n" | |
| return StreamingResponse(event_generator(), media_type="text/event-stream") | |
| out_message = {"role": "assistant"} | |
| if tool_calls: | |
| out_message["tool_calls"] = tool_calls | |
| out_message["content"] = clean_content | |
| finish_reason = "tool_calls" | |
| else: | |
| out_message["content"] = clean_content if clean_content is not None else raw_content | |
| finish_reason = raw_finish | |
| if reasoning: | |
| out_message["reasoning_content"] = reasoning | |
| raw_usage = raw_res.get("usage") if isinstance(raw_res, dict) else {} | |
| prompt_tokens = raw_usage.get("prompt_tokens") if raw_usage else None | |
| completion_tokens = raw_usage.get("completion_tokens") if raw_usage else None | |
| def _text(value): | |
| return value if isinstance(value, str) else ("" if value is None else str(value)) | |
| if prompt_tokens is None or prompt_tokens == 0: | |
| prompt_tokens = sum(max(1, int(len(_text(m.get("content")).split()) * 1.3)) for m in formatted_msgs) | |
| if completion_tokens is None or completion_tokens == 0: | |
| full_generated = raw_content or "" | |
| completion_tokens = max(1, int(len(full_generated.split()) * 1.3)) if full_generated else 0 | |
| usage_obj = { | |
| "prompt_tokens": prompt_tokens, | |
| "completion_tokens": completion_tokens, | |
| "total_tokens": prompt_tokens + completion_tokens, | |
| } | |
| if reasoning: | |
| reasoning_tok_count = max(1, int(len(reasoning.split()) * 1.3)) | |
| usage_obj["completion_tokens_details"] = { | |
| "reasoning_tokens": reasoning_tok_count, | |
| } | |
| return { | |
| "id": f"chatcmpl-{int(time.time()*1000)}", | |
| "object": "chat.completion", | |
| "created": int(time.time()), | |
| "model": selected_model, | |
| "choices": [ | |
| { | |
| "index": 0, | |
| "message": out_message, | |
| "finish_reason": finish_reason | |
| } | |
| ], | |
| "usage": usage_obj | |
| } | |
| async def list_openai_models(request: Request): | |
| _authorize_api_request(request) | |
| choices = list_gguf_files() | |
| models_data = [] | |
| # GGUF LLM models | |
| for c in choices: | |
| models_data.append({ | |
| "id": c, | |
| "object": "model", | |
| "created": int(time.time()), | |
| "owned_by": "abalanescu-flow", | |
| "permission": [], | |
| }) | |
| # Audio / Music models | |
| for name, repo_id in AUDIO_MUSIC_MODELS.items(): | |
| models_data.append({ | |
| "id": repo_id, | |
| "object": "model", | |
| "created": int(time.time()), | |
| "owned_by": "abalanescu-flow-zerogpu-audio", | |
| "permission": [], | |
| }) | |
| # Experimental TTS (Breeze TTS 2, non-commercial weights) | |
| models_data.append({ | |
| "id": BREEZE_MODEL_ID, | |
| "object": "model", | |
| "created": int(time.time()), | |
| "owned_by": "abalanescu-flow-zerogpu-experimental-tts", | |
| "permission": [], | |
| }) | |
| # Image generation removed by policy (Space scope: Music + Music Videos only). | |
| # Video models | |
| for name, repo_id in VIDEO_MODELS.items(): | |
| models_data.append({ | |
| "id": repo_id, | |
| "object": "model", | |
| "created": int(time.time()), | |
| "owned_by": "abalanescu-flow-zerogpu-video", | |
| "permission": [], | |
| }) | |
| return {"object": "list", "data": models_data} | |
| def probe_live_gpu_vram(): | |
| """Live probe executed without leasing ZeroGPU (avoids blocking health checks and burning quota).""" | |
| try: | |
| import torch | |
| if torch.cuda.is_available(): | |
| device_name = torch.cuda.get_device_name(0) | |
| free_bytes, total_bytes = torch.cuda.mem_get_info() | |
| total_gb = round(total_bytes / (1024**3), 2) | |
| free_gb = round(free_bytes / (1024**3), 2) | |
| used_gb = round((total_bytes - free_bytes) / (1024**3), 2) | |
| pct_used = round((used_gb / total_gb) * 100, 1) if total_gb > 0 else 0 | |
| else: | |
| device_name = "NVIDIA RTX PRO 6000 Blackwell (Allocated on-demand)" | |
| total_gb, used_gb, free_gb, pct_used = 48.0, 15.9, 32.1, 33.1 | |
| except Exception as e: | |
| device_name = f"ZeroGPU Device ({str(e)})" | |
| total_gb, used_gb, free_gb, pct_used = 48.0, 15.9, 32.1, 33.1 | |
| return { | |
| "status": "healthy", | |
| "device_name": device_name, | |
| "total_vram_gb": total_gb, | |
| "used_vram_gb": used_gb, | |
| "free_vram_gb": free_gb, | |
| "vram_usage_percent": f"{pct_used}%", | |
| "vram_summary": f"{used_gb} GB / {total_gb} GB used ({free_gb} GB free)", | |
| "active_model": _loaded_file or DEFAULT_MODEL, | |
| "native_context": 262144, | |
| "timestamp": int(time.time()), | |
| } | |
| async def health_check(): | |
| """Live health and VRAM telemetry probe.""" | |
| choices = list_gguf_files() | |
| try: | |
| gpu_telemetry = probe_live_gpu_vram() | |
| except Exception as e: | |
| gpu_telemetry = { | |
| "status": "standby", | |
| "device_name": "NVIDIA RTX PRO 6000 Blackwell (ZeroGPU Large)", | |
| "total_vram_gb": 48.0, | |
| "vram_summary": "Allocated dynamically per inference call", | |
| "note": str(e), | |
| } | |
| return { | |
| "service": "ZeroGPU Private OpenAI API Hub", | |
| "models_count": len(choices), | |
| "default_model": DEFAULT_MODEL, | |
| "models_available": choices, | |
| "gpu": gpu_telemetry, | |
| "quota": GPU_USAGE.usage_snapshot(), | |
| } | |
| async def warmup_space(request: Request): | |
| """Authenticated warm-up endpoint that verifies GPU readiness with a fast 1-token probe.""" | |
| _authorize_api_request(request) | |
| choices = list_gguf_files() | |
| if not choices: | |
| raise HTTPException(status_code=500, detail="No GGUF models available in Space storage.") | |
| selected = choices[0] | |
| t0 = time.time() | |
| ensure_model_available(selected) | |
| try: | |
| res = gpu_tracked_call( | |
| "warmup", | |
| _raw_generate_openai_chat, | |
| [{"role": "user", "content": "ping"}], | |
| selected, | |
| temperature=0.1, | |
| max_tokens=2, | |
| model=selected, | |
| ) | |
| elapsed_ms = round((time.time() - t0) * 1000, 2) | |
| return { | |
| "status": "warmed", | |
| "model": selected, | |
| "latency_ms": elapsed_ms, | |
| "response": res["choices"][0]["message"].get("content") or "", | |
| } | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=f"Warmup probe failed: {str(e)}") | |
| # ─── Async Background Jobs Registry for ZeroGPU Audio Generation ────────── | |
| _audio_jobs: Dict[str, Dict[str, Any]] = {} | |
| def _run_audio_job_worker(job_id: str, prompt: str, lyrics: str, model_choice: str, duration: int, seed: int): | |
| try: | |
| _audio_jobs[job_id]["status"] = "processing" | |
| out_audio = generate_zerogpu_music( | |
| prompt=prompt or "Original studio-quality song, high fidelity, mixed and mastered", | |
| lyrics=lyrics, | |
| model_choice=model_choice, | |
| duration_seconds=duration, | |
| seed=seed | |
| ) | |
| if out_audio and os.path.exists(out_audio): | |
| _audio_jobs[job_id]["status"] = "completed" | |
| _audio_jobs[job_id]["audio_path"] = out_audio | |
| _audio_jobs[job_id]["filename"] = os.path.basename(out_audio) | |
| _audio_jobs[job_id]["completed_at"] = time.time() | |
| else: | |
| _audio_jobs[job_id]["status"] = "failed" | |
| _audio_jobs[job_id]["error"] = "Audio output file was not generated." | |
| except Exception as e: | |
| _audio_jobs[job_id]["status"] = "failed" | |
| _audio_jobs[job_id]["error"] = str(e) | |
| async def create_audio_job(request: Request): | |
| """Start an async audio/music generation job on ZeroGPU without timing out.""" | |
| _authorize_api_request(request) | |
| try: | |
| body = await request.json() | |
| except Exception: | |
| body = {} | |
| model_name = body.get("model", "MiniMaxAI/MiniMax-Music3") | |
| lyrics = body.get("input", "") | |
| instructions = body.get("instructions", body.get("prompt", "")) | |
| dur = int(body.get("duration", body.get("duration_seconds", 60))) | |
| seed = int(body.get("seed", 7)) | |
| model_choice = "MiniMax Music 3 (full song + vocals)" | |
| for k, v in AUDIO_MUSIC_MODELS.items(): | |
| if model_name in (k, v): | |
| model_choice = k | |
| break | |
| job_id = f"job_audio_{int(time.time())}_{uuid.uuid4().hex[:8]}" | |
| _audio_jobs[job_id] = { | |
| "job_id": job_id, | |
| "status": "queued", | |
| "created_at": time.time(), | |
| "model": model_choice, | |
| "duration": dur, | |
| "audio_path": None, | |
| "error": None | |
| } | |
| import threading | |
| threading.Thread( | |
| target=_run_audio_job_worker, | |
| args=(job_id, instructions, lyrics, model_choice, dur, seed), | |
| daemon=True | |
| ).start() | |
| return JSONResponse( | |
| status_code=202, | |
| content={ | |
| "job_id": job_id, | |
| "status": "queued", | |
| "check_status_url": f"/v1/audio/jobs/{job_id}", | |
| "download_url": f"/v1/audio/jobs/{job_id}/download" | |
| } | |
| ) | |
| async def get_audio_job_status(job_id: str): | |
| """Check status of an async audio generation job.""" | |
| job = _audio_jobs.get(job_id) | |
| if not job: | |
| raise HTTPException(status_code=404, detail="Job not found") | |
| res = { | |
| "job_id": job["job_id"], | |
| "status": job["status"], | |
| "created_at": job["created_at"], | |
| "error": job.get("error") | |
| } | |
| if job["status"] == "completed": | |
| res["download_url"] = f"/v1/audio/jobs/{job_id}/download" | |
| res["filename"] = job.get("filename") | |
| return JSONResponse(content=res) | |
| async def download_audio_job(job_id: str): | |
| """Download the synthesized audio file for a completed job.""" | |
| job = _audio_jobs.get(job_id) | |
| if not job: | |
| raise HTTPException(status_code=404, detail="Job not found") | |
| if job["status"] != "completed" or not job.get("audio_path"): | |
| raise HTTPException(status_code=400, detail=f"Job status is {job['status']}, not ready for download") | |
| filepath = job["audio_path"] | |
| if not os.path.exists(filepath): | |
| raise HTTPException(status_code=404, detail="Audio file has expired on server disk") | |
| from fastapi.responses import FileResponse | |
| return FileResponse(filepath, media_type="audio/wav", filename=job.get("filename", "synthesized_track.wav")) | |
| async def openai_audio_speech(request: Request): | |
| """ | |
| OpenAI-compatible Audio Speech / Music Generation API endpoint. | |
| Routes to ZeroGPU MiniMax Music 3 / MusicGen and returns audio/wav stream. | |
| Supports asynchronous execution via ?async=true or 'Prefer: respond-async'. | |
| """ | |
| _authorize_api_request(request) | |
| try: | |
| body = await request.json() | |
| except Exception: | |
| body = {} | |
| prefer_async = ( | |
| body.get("async") is True | |
| or request.query_params.get("async") == "true" | |
| or "respond-async" in request.headers.get("Prefer", "") | |
| ) | |
| if prefer_async: | |
| return await create_audio_job(request) | |
| model_name = body.get("model", "MiniMaxAI/MiniMax-Music3") | |
| # Breeze TTS 2 routing: speech synthesis (not music) with voice direction. | |
| if model_name in (BREEZE_MODEL_ID, "Breeze TTS 2 (SOTA experimental TTS, EN/ZH)", "breeze-tts-2", "Breeze-TTS-2"): | |
| if not _breeze_available(): | |
| raise HTTPException( | |
| status_code=503, | |
| detail=( | |
| "Breeze TTS 2 is temporarily disabled: qwen-tts pins transformers==4.57.3 " | |
| "which is incompatible with MiniMax Music 3 (diffusers requires huggingface_hub>=1.23). " | |
| "Music generation via MiniMax Music 3 remains fully available." | |
| ), | |
| ) | |
| text = str(body.get("input", "")).strip() | |
| if not text: | |
| raise HTTPException(status_code=400, detail="'input' text is required for Breeze TTS.") | |
| instruction = str(body.get("instructions", body.get("prompt", "")) or "") | |
| language = str(body.get("language", "auto")) | |
| emotion = str(body.get("emotion", "neutral")) | |
| seed = int(body.get("seed", 42)) | |
| cfg_scale = float(body.get("cfg_scale", 1.0)) | |
| ref_audio_path = body.get("ref_audio_path") # server-side path (UI tab handles uploads) | |
| ref_text = str(body.get("ref_text", "") or "") | |
| try: | |
| out_audio = generate_zerogpu_breeze_tts( | |
| text=text, | |
| instruction=instruction, | |
| language=language, | |
| emotion=emotion, | |
| cfg_scale=cfg_scale, | |
| seed=seed, | |
| ref_audio_path=ref_audio_path, | |
| ref_text=ref_text, | |
| ) | |
| if out_audio and os.path.exists(out_audio): | |
| from fastapi.responses import FileResponse | |
| return FileResponse(out_audio, media_type="audio/wav", filename=os.path.basename(out_audio)) | |
| raise HTTPException(status_code=500, detail="Failed to synthesize Breeze speech output.") | |
| except HTTPException: | |
| raise | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=f"ZeroGPU Breeze TTS error: {str(e)}") | |
| lyrics = body.get("input", "") | |
| instructions = body.get("instructions", body.get("prompt", "")) | |
| dur = int(body.get("duration", body.get("duration_seconds", 60))) | |
| seed = int(body.get("seed", 7)) | |
| # Map model name | |
| model_choice = "MiniMax Music 3 (full song + vocals)" | |
| for k, v in AUDIO_MUSIC_MODELS.items(): | |
| if model_name in (k, v): | |
| model_choice = k | |
| break | |
| try: | |
| out_audio = generate_zerogpu_music( | |
| prompt=instructions or "Original studio-quality song, high fidelity, mixed and mastered", | |
| lyrics=lyrics, | |
| model_choice=model_choice, | |
| duration_seconds=dur, | |
| seed=seed | |
| ) | |
| if out_audio and os.path.exists(out_audio): | |
| from fastapi.responses import FileResponse | |
| return FileResponse(out_audio, media_type="audio/wav", filename=os.path.basename(out_audio)) | |
| raise HTTPException(status_code=500, detail="Failed to synthesize audio output.") | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=f"ZeroGPU audio generation error: {str(e)}") | |
| async def openai_embeddings(request: Request): | |
| _authorize_api_request(request) | |
| try: | |
| body = await request.json() | |
| except Exception: | |
| raise HTTPException(status_code=400, detail="Invalid JSON body") | |
| input_data = body.get("input") | |
| if not input_data: | |
| raise HTTPException(status_code=400, detail="Field 'input' is required.") | |
| inputs = [input_data] if isinstance(input_data, str) else list(input_data) | |
| embeddings_list = [] | |
| total_tokens = 0 | |
| for idx, text in enumerate(inputs): | |
| try: | |
| emb = api_client.feature_extraction(text=str(text), model=EMBEDDING_MODEL) | |
| raw_vec = emb[0] if isinstance(emb, list) and len(emb) > 0 and isinstance(emb[0], list) else emb | |
| if hasattr(raw_vec, "tolist"): | |
| vec = raw_vec.tolist() | |
| elif isinstance(raw_vec, (list, tuple)): | |
| vec = [float(x) for x in raw_vec] | |
| else: | |
| vec = list(raw_vec) | |
| embeddings_list.append({ | |
| "object": "embedding", | |
| "index": idx, | |
| "embedding": vec, | |
| }) | |
| total_tokens += max(1, len(str(text).split())) | |
| except Exception as e: | |
| raise HTTPException(status_code=500, detail=f"Embedding extraction failed: {e}") | |
| return JSONResponse(content={ | |
| "object": "list", | |
| "data": embeddings_list, | |
| "model": EMBEDDING_MODEL, | |
| "usage": { | |
| "prompt_tokens": total_tokens, | |
| "total_tokens": total_tokens, | |
| } | |
| }) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # High-Density Full-Screen Modern Dashboard UI | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| CUSTOM_CSS = """ | |
| /* Full-Screen Ultra-Dense Glassmorphic Dashboard */ | |
| .gradio-container { | |
| max-width: 100% !important; | |
| width: 100% !important; | |
| padding: 10px 16px !important; | |
| margin: 0 !important; | |
| font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif !important; | |
| background-color: #090d16 !important; | |
| } | |
| /* Header bar */ | |
| .top-header { | |
| background: linear-gradient(135deg, rgba(30, 27, 75, 0.8) 0%, rgba(15, 23, 42, 0.95) 100%); | |
| backdrop-filter: blur(16px); | |
| border-radius: 12px; | |
| padding: 14px 20px; | |
| margin-bottom: 12px; | |
| border: 1px solid rgba(129, 140, 248, 0.2); | |
| display: flex; | |
| justify-content: space-between; | |
| align-items: center; | |
| flex-wrap: wrap; | |
| gap: 12px; | |
| } | |
| .brand-title { | |
| font-size: 1.4rem; | |
| font-weight: 800; | |
| letter-spacing: -0.02em; | |
| background: linear-gradient(90deg, #38bdf8, #818cf8, #c084fc); | |
| -webkit-background-clip: text; | |
| -webkit-text-fill-color: transparent; | |
| } | |
| .status-badges { | |
| display: flex; | |
| gap: 8px; | |
| flex-wrap: wrap; | |
| } | |
| .hud-chip { | |
| background: rgba(255, 255, 255, 0.06); | |
| border: 1px solid rgba(255, 255, 255, 0.12); | |
| border-radius: 8px; | |
| padding: 4px 10px; | |
| font-size: 0.78rem; | |
| font-weight: 600; | |
| color: #e2e8f0; | |
| display: flex; | |
| align-items: center; | |
| gap: 6px; | |
| font-family: Menlo, Consolas, "DejaVu Sans Mono", monospace; | |
| } | |
| .tab-nav { | |
| border-bottom: 1px solid rgba(255, 255, 255, 0.1) !important; | |
| } | |
| /* Compact input controls */ | |
| .compact-box { | |
| margin-bottom: 8px !important; | |
| } | |
| """ | |
| choices = model_choices() or [DEFAULT_MODEL] | |
| default_choice = DEFAULT_MODEL if DEFAULT_MODEL in choices else choices[0] | |
| with gr.Blocks( | |
| title="AI Creative Studio Pro", | |
| theme=gr.themes.Soft(primary_hue="indigo", secondary_hue="slate"), | |
| css=CUSTOM_CSS, | |
| ) as demo: | |
| gr.HTML(""" | |
| <div class="top-header"> | |
| <div> | |
| <div class="brand-title">⚡ AI Creative Studio & ZeroGPU Hub</div> | |
| <div style="font-size: 0.85rem; color: #94a3b8;">High-Density Multi-Modal Suite & OpenAI Hub | abalanescu/flow</div> | |
| </div> | |
| <div class="status-badges"> | |
| <div class="hud-chip"><span style="color:#38bdf8;">●</span> ZeroGPU: RTX PRO 6000 (48GB)</div> | |
| <div class="hud-chip"><span style="color:#a855f7;">●</span> Context: 256k Native FlashAttention</div> | |
| <div class="hud-chip"><span style="color:#34d399;">●</span> Serverless: Whisper + Edge-TTS + Breeze TTS 2</div> | |
| <div class="hud-chip"><span style="color:#fbbf24;">●</span> Hub: /v1/chat/completions</div> | |
| </div> | |
| </div> | |
| """) | |
| with gr.Tabs(): | |
| # ── Tab 1: Pro Chat & Agents ───────────────────────────────────── | |
| with gr.Tab("💬 Pro Agent & LLM Chat"): | |
| with gr.Row(): | |
| with gr.Column(scale=3): | |
| chatbot = gr.Chatbot( | |
| type="messages", | |
| height=540, | |
| show_copy_button=True, | |
| render_markdown=True, | |
| label="Conversation Stream", | |
| ) | |
| telemetry_bar = gr.HTML( | |
| "<div style='font-family: monospace; font-size: 0.85rem; padding: 6px 10px; background: rgba(30,41,59,0.5); border-radius: 6px; color: #94a3b8;'>" | |
| "⚡ Ready — Select a preset or type a prompt." | |
| "</div>" | |
| ) | |
| with gr.Row(): | |
| chat_input = gr.Textbox( | |
| show_label=False, | |
| placeholder="Type instructions, code, or ask a question...", | |
| lines=2, | |
| scale=5, | |
| ) | |
| send_btn = gr.Button("🚀 Run", variant="primary", scale=1) | |
| clear_btn = gr.Button("🗑️ Clear", scale=1) | |
| with gr.Row(): | |
| gr.Markdown("**Quick Prompts:**", elem_classes=["compact-box"]) | |
| p1 = gr.Button("🏗️ Software Architecture Audit", size="sm") | |
| p2 = gr.Button("🐍 High-Performance Python", size="sm") | |
| p3 = gr.Button("🛠️ Simulate Tool Call", size="sm") | |
| p4 = gr.Button("⚡ Quantum Algorithm Explanation", size="sm") | |
| with gr.Column(scale=1): | |
| with gr.Accordion("⚙️ Engine Controls", open=True): | |
| chat_model = gr.Dropdown(choices=choices, value=default_choice, label="Active GGUF Model") | |
| chat_temp = gr.Slider(0.1, 1.5, value=0.7, step=0.05, label="Temperature") | |
| chat_max = gr.Number(value=None, precision=0, label="Max Tokens (blank = 128k native)") | |
| chat_system = gr.Textbox( | |
| label="System Prompt", | |
| value="You are a brilliant software architect, researcher, and coding assistant.", | |
| lines=3, | |
| ) | |
| # Chat actions | |
| send_btn.click( | |
| fn=custom_chat_handler, | |
| inputs=[chat_input, chatbot, chat_model, chat_system, chat_temp, chat_max], | |
| outputs=[chatbot, telemetry_bar, chat_input], | |
| ) | |
| chat_input.submit( | |
| fn=custom_chat_handler, | |
| inputs=[chat_input, chatbot, chat_model, chat_system, chat_temp, chat_max], | |
| outputs=[chatbot, telemetry_bar, chat_input], | |
| ) | |
| clear_btn.click(lambda: ([], "<div style='font-family: monospace; font-size: 0.85rem; padding: 6px 10px; background: rgba(30,41,59,0.5); border-radius: 6px; color: #94a3b8;'>⚡ Ready — Context cleared.</div>", ""), None, [chatbot, telemetry_bar, chat_input]) | |
| p1.click(lambda: "Review this microservice architecture for high-throughput concurrency bottlenecks and propose a clean design pattern.", None, chat_input) | |
| p2.click(lambda: "Write a high-performance Python function using ctypes/simd or async primitives with full type annotations.", None, chat_input) | |
| p3.click(lambda: "What is the stock price of Apple right now? Call the get_stock_price tool if available.", None, chat_input) | |
| p4.click(lambda: "Explain Shor's algorithm for quantum prime factorization in 3 concise, intuitive paragraphs.", None, chat_input) | |
| # ── Tab 2: Multimodal Vision & OCR ──────────────────────────────── | |
| with gr.Tab("👁️ Vision & Document OCR"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| vis_img = gr.Image(label="Input Diagram / UI Screenshot / Document", type="filepath") | |
| vis_prompt = gr.Textbox( | |
| label="Prompt / Extraction Request", | |
| placeholder="e.g. Extract the components and convert into a clean Mermaid diagram...", | |
| lines=2, | |
| ) | |
| with gr.Row(): | |
| vis_btn = gr.Button("🔍 Run Deep Vision", variant="primary") | |
| v_p1 = gr.Button("Diagram to Mermaid", size="sm") | |
| v_p2 = gr.Button("Extract All Code/Text", size="sm") | |
| with gr.Column(scale=1): | |
| vis_output = gr.Textbox(label="Visual Analysis & OCR Output", lines=20, show_copy_button=True) | |
| vis_btn.click(fn=analyze_vision, inputs=[vis_img, vis_prompt], outputs=vis_output) | |
| v_p1.click(lambda: "Extract the architecture components from this diagram and format as a valid mermaid block.", None, vis_prompt) | |
| v_p2.click(lambda: "Extract all visible text, formulas, code snippets, and table values verbatim.", None, vis_prompt) | |
| # ── Tab 3: ZeroGPU AI Video Studio ────────────────────────────── | |
| with gr.Tab("🎬 ZeroGPU AI Video Studio"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| gr.Markdown("#### ⚡ Open-Weights Video Generation ($0 API Cost / 40 min A100 Quota)") | |
| vid_prompt = gr.Textbox( | |
| label="Video Scene Prompt", | |
| placeholder="A cinematic drone shot through misty Codrii forest at sunrise, 4k photorealistic...", | |
| lines=3, | |
| ) | |
| vid_neg = gr.Textbox(label="Negative Prompt", value="blurry, distorted, low quality, glitch, watermark") | |
| with gr.Row(): | |
| vid_model = gr.Dropdown( | |
| choices=list(VIDEO_MODELS.keys()), | |
| value=list(VIDEO_MODELS.keys())[3], # ZeroScope | |
| label="ZeroGPU Video Model" | |
| ) | |
| vid_frames = gr.Slider(8, 32, value=16, step=4, label="Frame Count") | |
| with gr.Row(): | |
| vid_fps = gr.Slider(6, 24, value=8, step=2, label="FPS") | |
| vid_guidance = gr.Slider(1.0, 15.0, value=7.5, step=0.5, label="Guidance Scale") | |
| vid_seed = gr.Number(value=-1, label="Seed (-1 for random)") | |
| vid_btn = gr.Button("🎬 Render Video on ZeroGPU", variant="primary") | |
| with gr.Column(scale=1): | |
| vid_output = gr.Video(label="Rendered MP4 Video", autoplay=True) | |
| vid_btn.click( | |
| fn=generate_zerogpu_video, | |
| inputs=[vid_prompt, vid_neg, vid_model, vid_frames, vid_fps, vid_guidance, vid_seed], | |
| outputs=vid_output | |
| ) | |
| # ── Tab 4: ZeroGPU Music & Audio Studio ────────────────────────── | |
| with gr.Tab("🎵 ZeroGPU Music & Audio Studio"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| gr.Markdown("#### ⚡ Foundation AI Music, Vocals & Foley ($0 API Cost / 40 min A100 Quota)") | |
| mus_prompt = gr.Textbox( | |
| label="Musical Style / Genre Prompt", | |
| placeholder="e.g. cinematic orchestral, lo-fi hip hop, electro-swing", | |
| lines=2, | |
| ) | |
| mus_lyrics = gr.Textbox( | |
| label="Lyrics / Vocal Lines (Optional)", | |
| placeholder="[verse]\n...\n[chorus]\n...", | |
| lines=3, | |
| ) | |
| with gr.Row(): | |
| mus_model = gr.Dropdown( | |
| choices=list(AUDIO_MUSIC_MODELS.keys()), | |
| value="MiniMax Music 3 (full song + vocals)", | |
| label="REAL MUSIC MODEL (not MIDI)" | |
| ) | |
| mus_dur = gr.Slider(5, 300, value=60, step=5, label="Duration (Seconds)") | |
| with gr.Row(): | |
| mus_guidance = gr.Slider(1.0, 10.0, value=3.0, step=0.5, label="Guidance Scale") | |
| mus_temp = gr.Slider(0.2, 1.5, value=1.0, step=0.1, label="Temperature") | |
| mus_seed = gr.Number(value=7, label="Seed (reproducible)") | |
| gr.Markdown("**MiniMax Music 3** = complete song + expressive vocals + lyrics. **Stable Audio 3** = structured music, generally instrumental. **MusicGen** = instrumental only. All run locally on ZeroGPU; no serverless fallback.") | |
| mus_btn = gr.Button("🎵 Generate REAL SONG on ZeroGPU", variant="primary") | |
| with gr.Column(scale=1): | |
| mus_output = gr.Audio(label="Synthesized Multi-Track Audio", type="filepath") | |
| mus_btn.click( | |
| fn=generate_zerogpu_music, | |
| inputs=[mus_prompt, mus_lyrics, mus_model, mus_dur, mus_guidance, mus_temp, mus_seed], | |
| outputs=mus_output | |
| ) | |
| # ── Tab 4: Audio Suite ─────────────────────────────────────────── | |
| with gr.Tab("🎙️ Audio Lab: STT & TTS"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| gr.Markdown("#### 🎤 Whisper Large v3 (Speech to Text)") | |
| audio_in = gr.Audio(label="Record / Upload Speech", type="filepath") | |
| stt_btn = gr.Button("Transcribe Audio", variant="primary") | |
| stt_out = gr.Textbox(label="Transcription Result", lines=6, show_copy_button=True) | |
| stt_btn.click(fn=transcribe_audio, inputs=audio_in, outputs=stt_out) | |
| with gr.Column(scale=1): | |
| gr.Markdown("#### 🔊 Edge-TTS ro-RO (Text to Speech)") | |
| tts_text = gr.Textbox( | |
| label="Text to Speak", | |
| placeholder="Welcome to the AI Creative Studio on Hugging Face.", | |
| lines=4, | |
| ) | |
| tts_btn = gr.Button("Synthesize High-Fidelity Voice", variant="primary") | |
| tts_audio = gr.Audio(label="Synthesized Speech Audio", type="filepath") | |
| tts_btn.click(fn=generate_tts, inputs=tts_text, outputs=tts_audio) | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| # TAB 5: Voice-to-Creative-Prompt Pipeline | |
| # ━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━ | |
| def voice_to_art(audio_input, model_file, art_style): | |
| """Whisper STT -> Qwen creative prompt expansion (image rendering removed by policy).""" | |
| if audio_input is None: | |
| raise gr.Error("Please record or upload audio first.") | |
| if not HF_TOKEN: | |
| raise gr.Error("Set HF_TOKEN in Space secrets.") | |
| try: | |
| transcription = api_client.automatic_speech_recognition( | |
| audio=audio_input, | |
| model=WHISPER_MODEL, | |
| ) | |
| raw_text = transcription.text if hasattr(transcription, "text") else str(transcription) | |
| except Exception as e: | |
| raise gr.Error(_format_api_error(e, "Whisper transcription")) | |
| if not raw_text.strip(): | |
| raise gr.Error("Could not understand the audio.") | |
| llm = get_model(model_file) | |
| style_hint = f" in {art_style} style" if art_style.strip() else "" | |
| expand_prompt = ( | |
| f"You are a master image prompt engineer. The user said: \"{raw_text}\"\n\n" | |
| f"Write a single, highly detailed, vivid creative image-generation prompt{style_hint}. " | |
| f"Include composition, cinematic lighting, color palette, mood, and fine details. " | |
| f"Output ONLY the prompt, nothing else. Max 100 words." | |
| ) | |
| response = llm.create_chat_completion( | |
| messages=[{"role": "user", "content": expand_prompt}], | |
| max_tokens=256, | |
| temperature=0.85, | |
| top_p=0.95, | |
| ) | |
| art_prompt = response["choices"][0]["message"]["content"].strip() | |
| _, _, clean_art_prompt = parse_model_tool_calls(art_prompt) | |
| final_prompt = clean_art_prompt or art_prompt | |
| # Image generation removed by policy (Space scope: Music + Music Videos only). | |
| # Pipeline now stops at the expanded creative prompt for use in music/video workflows. | |
| return raw_text, final_prompt | |
| # ── Tab 6: Embeddings Lab ──────────────────────────────────────── | |
| with gr.Tab("🔍 Embeddings & Similarity"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| emb_a = gr.Textbox(label="Text A", value="The quick brown fox jumps over the lazy dog.", lines=3) | |
| emb_b = gr.Textbox(label="Text B", value="A fast brown animal leaps over a sleeping canine.", lines=3) | |
| emb_btn = gr.Button("🎯 Compute BGE-M3 Cosine Similarity", variant="primary") | |
| with gr.Column(scale=1): | |
| emb_out = gr.Markdown(label="Similarity Analysis") | |
| emb_btn.click(fn=compute_similarity, inputs=[emb_a, emb_b], outputs=emb_out) | |
| # ── Tab 7: API Hub & Telemetry ─────────────────────────────────── | |
| with gr.Tab("🔌 OpenAI API Hub & Telemetry"): | |
| gr.Markdown(""" | |
| ### 🔌 ZeroGPU Private OpenAI-Compatible Hub | |
| Connect **Hermes**, **OmniRoute**, **Cursor**, or **Open-WebUI** directly. | |
| ```bash | |
| # Chat Completions with Tool Calling & 128k Context | |
| curl -X POST https://abalanescu-flow2.hf.space/v1/chat/completions \\ | |
| -H "Authorization: Bearer $HF_TOKEN" \\ | |
| -H "Content-Type: application/json" \\ | |
| -d '{"model": "qwen", "messages": [{"role": "user", "content": "Hello!"}]}' | |
| ``` | |
| | Parameter | Active Configuration | | |
| |---|---| | |
| | **Base URL** | `https://abalanescu-flow2.hf.space/v1` | | |
| | **Model Alias** | `qwen` (Qwen3.8-27B-Q4_K_M.gguf) or `qwen-q6` | | |
| | **Max Context** | `131,072` Tokens (FlashAttention Enabled) | | |
| | **GPU Hardware** | NVIDIA RTX PRO 6000 Blackwell (48GB VRAM) | | |
| """) | |
| # Launch native Gradio app and mount FastAPI routes | |
| if __name__ == "__main__": | |
| # Opt-in UI lockdown: when GRADIO_UI_PASSWORD is set (Space secret or env), | |
| # the Gradio UI and its queue require a username/password prompt. REST auth | |
| # via FLOW_API_KEY is independent and unaffected. | |
| _ui_username = os.environ.get("GRADIO_UI_USERNAME", "abalanescu") | |
| _ui_password = os.environ.get("GRADIO_UI_PASSWORD", "") | |
| if _ui_password: | |
| demo.launch( | |
| prevent_thread_lock=True, | |
| ssr_mode=False, | |
| auth=(_ui_username, _ui_password), | |
| ) | |
| print(f"[auth] Gradio UI locked: username '{_ui_username}' + GRADIO_UI_PASSWORD required.") | |
| else: | |
| demo.launch(prevent_thread_lock=True, ssr_mode=False) | |
| demo.app.add_api_route( | |
| "/v1/chat/completions", | |
| openai_chat_completions, | |
| methods=["POST"], | |
| ) | |
| demo.app.add_api_route("/v1/models", list_openai_models, methods=["GET"]) | |
| demo.app.add_api_route("/v1/embeddings", openai_embeddings, methods=["POST"]) | |
| demo.app.add_api_route("/v1/audio/speech", openai_audio_speech, methods=["POST"]) | |
| demo.app.add_api_route("/v1/audio/jobs", create_audio_job, methods=["POST"]) | |
| demo.app.add_api_route("/v1/audio/jobs/{job_id}", get_audio_job_status, methods=["GET"]) | |
| demo.app.add_api_route("/v1/audio/jobs/{job_id}/download", download_audio_job, methods=["GET"]) | |
| demo.app.add_api_route("/v1/health", health_check, methods=["GET"]) | |
| demo.app.add_api_route("/v1/gpu/status", health_check, methods=["GET"]) | |
| demo.app.add_api_route("/healthz", health_check, methods=["GET"]) | |
| # The Gradio app serves its own route table, so every REST endpoint has to | |
| # be registered here as well; /v1/warmup was missing and 404'd in production | |
| # even though the FastAPI app used by the tests exposed it. | |
| demo.app.add_api_route("/v1/warmup", warmup_space, methods=["POST"]) | |
| # Provide /info alias so standard gradio_client versions connect seamlessly | |
| async def get_gradio_info(request: Request): | |
| for route in demo.app.routes: | |
| if getattr(route, "path", None) == "/gradio_api/info": | |
| if hasattr(route, "endpoint"): | |
| try: | |
| return await route.endpoint(request) | |
| except Exception: | |
| pass | |
| from fastapi.responses import RedirectResponse | |
| return RedirectResponse(url="/gradio_api/info") | |
| demo.block_thread() | |