tags is reasoning
reasoning_content = clean_text.strip()
clean_text = ""
clean_content = clean_text.strip()
if tool_calls and not clean_content:
clean_content = None
return (tool_calls if tool_calls else None), (reasoning_content if reasoning_content else None), clean_content
def normalize_native_tool_calls(tool_calls):
"""Normalize llama-cpp native calls to the OpenAI response contract.
Recent llama-cpp versions parse a GGUF chat template themselves and return
``message.tool_calls`` with an empty content field. Parsing only content
silently discarded those valid calls and made the API look non-agentic.
"""
if not tool_calls:
return None
normalized = []
for index, call in enumerate(tool_calls):
function = call.get("function", {}) if isinstance(call, dict) else {}
name = function.get("name") or call.get("name", "")
arguments = function.get("arguments", call.get("arguments", {}))
if not isinstance(arguments, str):
arguments = json.dumps(arguments, ensure_ascii=False)
normalized.append({
"index": call.get("index", index),
"id": call.get("id") or f"call_{uuid.uuid4().hex[:8]}",
"type": "function",
"function": {"name": name, "arguments": arguments},
})
return normalized
def format_openai_messages_for_model(messages):
"""Normalize multi-turn OpenAI messages including tool results into prompt format."""
formatted = []
for msg in messages:
role = msg.get("role", "user")
content = msg.get("content")
tool_calls = msg.get("tool_calls")
if role == "tool":
formatted.append({
"role": "user",
"content": f"\n{content or ''}\n"
})
elif role == "assistant" and tool_calls:
tc_text = ""
for tc in tool_calls:
fn = tc.get("function", {})
fn_name = fn.get("name", "")
raw_args = fn.get("arguments", "{}")
try:
args_dict = json.loads(raw_args) if isinstance(raw_args, str) else raw_args
except Exception:
args_dict = {}
tc_text += f"\n\n\n"
if isinstance(args_dict, dict):
for k, v in args_dict.items():
tc_text += f"\n{json.dumps(v) if isinstance(v, (dict, list)) else v}\n\n"
tc_text += "\n"
combined = (content or "") + tc_text
formatted.append({"role": "assistant", "content": combined.strip()})
else:
formatted.append({"role": role, "content": content or ""})
return formatted
def _format_api_error(e: Exception, action: str) -> str:
"""Format API errors with clear budget and credit guidance."""
msg = str(e)
if "402" in msg or "Payment Required" in msg:
return (
f"{action} notice (402 Payment Required): Your monthly HF Inference API credit "
"($2/month included with HF PRO) has been fully used for this billing period. "
"ZeroGPU tabs (Qwen 3.8 / Gemma LLM Chat) remain 100% free and functional!"
)
return f"{action} failed: {msg}"
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# TAB 1: Chat & Agent Runner (ZeroGPU Large โ 40 min/day)
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
def _raw_generate_openai_chat(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None):
"""Raw chat completion worker without @spaces.GPU decorator (avoids nested GPU leases in gpu_tracked_call)."""
llm = get_model(model_file)
kwargs = {
"messages": messages,
"max_tokens": (int(max_tokens) if max_tokens not in (None, "") else None),
"temperature": float(temperature),
"top_p": 0.95,
}
if tools:
kwargs["tools"] = tools
kwargs["tool_choice"] = tool_choice or "auto"
return llm.create_chat_completion(**kwargs)
@spaces.GPU(size="large", duration=_lease_duration)
def generate_openai_chat(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None):
return _raw_generate_openai_chat(messages, model_file, temperature, max_tokens, tools=tools, tool_choice=tool_choice)
@spaces.GPU(size="large", duration=_lease_duration)
def generate_openai_chat_stream(messages, model_file, temperature, max_tokens, tools=None, tool_choice=None):
llm = get_model(model_file)
kwargs = {
"messages": messages,
"max_tokens": (int(max_tokens) if max_tokens not in (None, "") else None),
"temperature": float(temperature),
"top_p": 0.95,
"stream": True,
}
if tools:
kwargs["tools"] = tools
kwargs["tool_choice"] = tool_choice or "auto"
for chunk in llm.create_chat_completion(**kwargs):
yield chunk
def custom_chat_handler(user_msg, history, model_file, system_prompt, temperature, max_tokens):
"""Rich chat execution with live token telemetry, reasoning, and speed reporting."""
if not user_msg or not user_msg.strip():
return history or [], "โก *Ready โ Enter a prompt to start inference.*", ""
history = list(history or [])
history.append({"role": "user", "content": user_msg.strip()})
messages = []
if system_prompt.strip():
messages.append({"role": "system", "content": system_prompt.strip()})
for item in history:
messages.append({"role": item.get("role", "user"), "content": item.get("content", "")})
t0 = time.time()
try:
raw_res = gpu_tracked_call(
"chat",
_raw_generate_openai_chat,
messages,
model_file,
float(temperature),
(int(max_tokens) if max_tokens not in (None, "") else None),
model=model_file,
)
t1 = time.time()
elapsed = max(0.01, t1 - t0)
raw_content = raw_res["choices"][0]["message"].get("content") or ""
tool_calls, reasoning, clean = parse_model_tool_calls(raw_content)
formatted_bot = ""
if reasoning:
formatted_bot += f"๐ง Deep Thinking & Reasoning
\n\n```markdown\n{reasoning}\n```\n \n\n"
if tool_calls:
formatted_bot += f"๐ ๏ธ Executed Tool Calls ({len(tool_calls)})
\n\n```json\n{json.dumps(tool_calls, indent=2)}\n```\n \n\n"
if clean:
formatted_bot += clean
elif not reasoning and not tool_calls:
formatted_bot += raw_content
history.append({"role": "assistant", "content": formatted_bot})
# Telemetry calculations
raw_usage = raw_res.get("usage", {})
prompt_toks = raw_usage.get("prompt_tokens") or sum(max(1, int(len(m["content"].split()) * 1.3)) for m in messages)
comp_toks = raw_usage.get("completion_tokens") or max(1, int(len(raw_content.split()) * 1.3))
tot_toks = prompt_toks + comp_toks
tps = comp_toks / elapsed
ctx_pct = (tot_toks / 262144) * 100
hud_md = (
f""
f"โก {tps:.1f} t/s"
f"โฑ๏ธ {elapsed:.2f}s"
f"๐ฅ Prompt: {prompt_toks}"
f"๐ค Output: {comp_toks}"
f"๐ง Context: {tot_toks:,} / 262,144 ({ctx_pct:.1f}%)"
f"โ ZeroGPU Large (48GB)"
f"
"
)
return history, hud_md, ""
except Exception as e:
history.append({"role": "assistant", "content": f"โ **Inference Error:** {str(e)}"})
return history, f"โ ๏ธ *Execution error after {time.time()-t0:.2f}s: {str(e)}*", ""
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# TAB 2: Vision & Multimodal OCR
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
def analyze_vision(image_input, prompt_text):
"""Analyze images, UI screenshots, code diagrams or documents."""
if image_input is None:
raise gr.Error("Please upload or capture an image first.")
if not HF_TOKEN:
raise gr.Error("Set HF_TOKEN in Space secrets to use the Vision API.")
prompt = prompt_text.strip() or "Describe this image in detail and extract all visible text and code."
try:
response = api_client.chat_completion(
messages=[
{
"role": "user",
"content": [
{"type": "text", "text": prompt},
{"type": "image_url", "image_url": {"url": image_input if isinstance(image_input, str) else image_input}},
],
}
],
model=VISION_MODEL,
max_tokens=1024,
)
return response.choices[0].message.content
except Exception as e:
try:
return api_client.image_to_text(image=image_input, model="Salesforce/blip-image-captioning-large")
except Exception:
raise gr.Error(_format_api_error(e, "Vision analysis"))
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# ZERO-GPU VIDEO GENERATION PIPELINE (40 min/day A100 Quota - $0 API Cost)
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# Video diffusion pipelines are heavy: CogVideoX-5B / LTX-Video weigh 6-15 GB.
# Weights are DOWNLOADED OUTSIDE the GPU lease (CPU-only snapshot_download in
# the web worker), then only the denoise step leases GPU. A dynamic 120 s
# lease cannot absorb a cold 10 GB download โ it aborts with "GPU task
# aborted" mid-load. Warm-pipeline inference fits comfortably in 240 s.
VIDEO_GPU_LEASE_SECONDS = int(os.environ.get("FLOW_VIDEO_GPU_LEASE_S", "240"))
def _prefetch_video_pipeline(repo_id: str) -> None:
"""Download pipeline weights into the HF cache with NO GPU lease active.
ZeroGPU bills only GPU-leased code; downloads and CPU work in the web
process are free. A snapshot that fails parsing (truncated spiece.model
etc.) never self-heals, so it is purged and re-downloaded once.
"""
from huggingface_hub import snapshot_download
try:
snapshot_download(repo_id)
except Exception as exc:
import re as _re
m = _re.search(r"models--([\w.\-]+)--([\w.\-]+)", str(exc))
if m and ("Error parsing" in str(exc) or "tokenize" in str(exc).lower()):
broken = os.path.join("/root/.cache/huggingface/hub", f"models--{m.group(1)}--{m.group(2)}")
if os.path.exists(broken):
print(f"[ZeroGPU Video] Purging corrupted HF cache: {broken}", flush=True)
shutil.rmtree(broken, ignore_errors=True)
snapshot_download(repo_id)
else:
raise
@spaces.GPU(duration=VIDEO_GPU_LEASE_SECONDS)
def generate_zerogpu_video(
prompt: str,
negative_prompt: str = "",
model_choice: str = "ZeroScope v2 (576w High-Res)",
num_frames: int = 16,
fps: int = 8,
guidance_scale: float = 7.5,
seed: int = -1
):
"""Generate dynamic MP4 video using ZeroGPU open-weights models."""
if not prompt.strip():
raise gr.Error("Please enter a video prompt.")
repo_id = VIDEO_MODELS.get(model_choice, "cerspense/zeroscope_v2_576w")
out_video_path = f"/tmp/zerogpu_video_{int(time.time())}_{abs(hash(prompt)) % 10000}.mp4"
# Weights must be resident BEFORE the GPU lease starts; loading inside a
# short lease aborts on any cold download.
_prefetch_video_pipeline(repo_id)
try:
import torch
from diffusers import DiffusionPipeline, DPMSolverMultistepScheduler
device = "cuda" if torch.cuda.is_available() else "cpu"
dtype = torch.float16 if device == "cuda" else torch.float32
if repo_id not in _video_pipeline_cache:
pipe = DiffusionPipeline.from_pretrained(repo_id, torch_dtype=dtype)
if hasattr(pipe, "scheduler"):
try:
pipe.scheduler = DPMSolverMultistepScheduler.from_config(pipe.scheduler.config)
except Exception:
pass
if hasattr(pipe, "enable_model_cpu_offload") and device == "cuda":
pipe.enable_model_cpu_offload()
else:
pipe = pipe.to(device)
_video_pipeline_cache[repo_id] = pipe
else:
pipe = _video_pipeline_cache[repo_id]
actual_seed = seed if (seed and int(seed) >= 0) else random.randint(0, 2**31 - 1)
generator = torch.Generator(device=device).manual_seed(actual_seed)
video_frames = pipe(
prompt=prompt.strip(),
negative_prompt=negative_prompt.strip() if negative_prompt else None,
num_inference_steps=24,
guidance_scale=float(guidance_scale),
num_frames=int(num_frames),
generator=generator
).frames[0]
import numpy as np
import tempfile
from PIL import Image
ffmpeg_bin = shutil.which("ffmpeg") or "/usr/bin/ffmpeg"
with tempfile.TemporaryDirectory() as tmpdir:
for idx, frame in enumerate(video_frames):
if isinstance(frame, np.ndarray):
f_arr = frame
if f_arr.dtype in (np.float32, np.float64, np.float16):
if f_arr.max() <= 1.0:
f_arr = (f_arr * 255.0).clip(0, 255).astype(np.uint8)
else:
f_arr = f_arr.clip(0, 255).astype(np.uint8)
img = Image.fromarray(f_arr)
else:
img = frame
img.save(os.path.join(tmpdir, f"frame_{idx:05d}.png"))
subprocess.run(
[
ffmpeg_bin, "-y", "-framerate", str(int(fps)),
"-i", os.path.join(tmpdir, "frame_%05d.png"),
"-c:v", "libx264", "-pix_fmt", "yuv420p",
out_video_path
],
check=True,
capture_output=True
)
return out_video_path
except Exception as exc:
print(f"[ZeroGPU Video] Direct pipeline exception: {exc}, running ffmpeg dynamic visualizer fallback...", flush=True)
try:
from engine.model_dispatcher import ModelDispatcher
disp = ModelDispatcher()
res = disp.generate_image(prompt=prompt, aspect_ratio="16:9")
img_path = res.get("filepath")
if img_path and os.path.exists(img_path):
ffmpeg_bin = shutil.which("ffmpeg") or "/opt/homebrew/bin/ffmpeg"
dur = max(3, int(int(num_frames) / max(1, int(fps))))
subprocess.run(
[
ffmpeg_bin, "-y", "-loop", "1", "-i", img_path,
"-vf", f"fps={fps},scale=768:432,zoompan=z='min(zoom+0.0015,1.15)':d={dur*fps}:s=768x432",
"-c:v", "libx264", "-t", str(dur), "-pix_fmt", "yuv420p",
out_video_path
],
capture_output=True,
timeout=20
)
if os.path.exists(out_video_path) and os.path.getsize(out_video_path) > 0:
return out_video_path
except Exception as e2:
print(f"[ZeroGPU Video] Fallback failed: {e2}")
raise gr.Error(f"ZeroGPU Video error: {exc}")
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# ZERO-GPU MUSIC & AUDIO GENERATION PIPELINE (40 min/day A100 Quota - $0 Cost)
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
def _raw_generate_zerogpu_music(
prompt: str,
lyrics: str = "",
model_choice: str = "MiniMax Music 3 (full song + vocals)",
duration_seconds: int = 60,
guidance_scale: float = 3.0,
temperature: float = 1.0,
seed: int = 7
):
"""Internal raw worker for ZeroGPU audio synthesis (avoids nested @spaces.GPU calls)."""
if not prompt.strip() and not lyrics.strip():
raise gr.Error("Please enter a music description or lyrics.")
repo_id = AUDIO_MUSIC_MODELS.get(model_choice, "MiniMaxAI/MiniMax-Music3")
# Clamp to max 120 seconds to guarantee completion within ZeroGPU quota lease
dur = max(5, min(120, int(duration_seconds or 60)))
out_audio_path = f"/tmp/zerogpu_music_{int(time.time())}_{abs(hash(prompt + lyrics)) % 10000}.wav"
try:
import torch
import soundfile as sf
# Official MiniMax Music3 path from its model card. It supports lyrics +
# detailed music description and produces complete vocal songs up to 5 min.
if repo_id == "MiniMaxAI/MiniMax-Music3":
from diffusers import ModularPipeline
import numpy as np
if repo_id not in _audio_pipeline_cache:
pipe = ModularPipeline.from_pretrained(repo_id)
pipe.load_components(dtype=torch.bfloat16)
pipe.to("cuda")
_audio_pipeline_cache[repo_id] = pipe
else:
pipe = _audio_pipeline_cache[repo_id]
raw_output = pipe(
prompt=prompt.strip(),
lyrics=lyrics.strip(),
audio_duration=float(dur),
generator=torch.Generator("cuda").manual_seed(int(seed)),
output="audios",
)
audio = raw_output[0]
if hasattr(audio, "detach"):
audio_np = audio.detach().cpu().float().numpy()
elif isinstance(audio, np.ndarray):
audio_np = audio.astype(np.float32)
else:
audio_np = np.asarray(audio, dtype=np.float32)
if audio_np.ndim > 1 and audio_np.shape[0] < audio_np.shape[1]:
audio_np = audio_np.T
sr = getattr(pipe, "sampling_rate", 44100)
sf.write(out_audio_path, audio_np, sr)
return out_audio_path
# YuE2-3B: AR-NAR MoT + flow matching, 48kHz stereo, editable scores.
# The repo is NOT a diffusers pipeline (no model_index.json); it ships the
# `yue2` infer wheel (YuE2Pipeline). Install with --no-deps (its pins for
# torch/hub conflict with the Space's diffusers set; core only needs
# torch + transformers + tiktoken + soundfile).
if repo_id == "m-a-p/YuE2-3B":
import numpy as np
from yue2 import YuE2Pipeline
if repo_id not in _audio_pipeline_cache:
pipe = YuE2Pipeline.from_pretrained(repo_id, progress=False)
_audio_pipeline_cache[repo_id] = pipe
else:
pipe = _audio_pipeline_cache[repo_id]
song = pipe(style=prompt.strip(), lyrics=lyrics.strip(), cot="full", seed=int(seed))
sf.write(out_audio_path, song.audio, song.sample_rate)
return out_audio_path
# ACE-Step v1.5 XL Turbo: Apache 2.0, 48kHz, commercial-safe, <4GB VRAM.
# Output is AudioPipelineOutput.audios (channels, samples); rate is
# pipe.sample_rate. Turbo is guidance-distilled so CFG is ignored anyway.
if repo_id == "ACE-Step/acestep-v15-xl-turbo-diffusers":
from diffusers import AceStepPipeline
import numpy as np
if repo_id not in _audio_pipeline_cache:
pipe = AceStepPipeline.from_pretrained(repo_id, torch_dtype=torch.bfloat16)
pipe.to("cuda")
pipe.vae.enable_tiling() # bound VAE decode memory for longer clips
_audio_pipeline_cache[repo_id] = pipe
else:
pipe = _audio_pipeline_cache[repo_id]
raw_output = pipe(
prompt=prompt.strip(),
lyrics=lyrics.strip(),
audio_duration=float(dur),
generator=torch.Generator("cuda").manual_seed(int(seed)),
)
audio_tensor = raw_output.audios[0] # (channels, samples)
audio_np = audio_tensor.T.cpu().float().numpy()
sf.write(out_audio_path, audio_np, pipe.sample_rate)
return out_audio_path
# Stable Audio 3 uses its own pipeline and may require HF access approval.
if repo_id == "stabilityai/stable-audio-3-medium":
# The public Stability release currently uses the separate
# `stable_audio_3` package, not a Diffusers StableAudio3Pipeline.
# Do not pretend this selector works or silently substitute audio.
raise RuntimeError(
"Stable Audio 3 is not enabled in this Space yet: its official "
"stable_audio_3 runtime is not installed. Select MiniMax Music 3."
)
# MusicGen is explicitly instrumental and does not reliably sing lyrics.
if repo_id.startswith("facebook/musicgen"):
from transformers import AutoProcessor, MusicgenForConditionalGeneration
if repo_id not in _audio_pipeline_cache:
processor = AutoProcessor.from_pretrained(repo_id)
model = MusicgenForConditionalGeneration.from_pretrained(repo_id, torch_dtype=torch.float16).to("cuda")
_audio_pipeline_cache[repo_id] = (processor, model)
processor, model = _audio_pipeline_cache[repo_id]
inputs = processor(text=[prompt.strip()], padding=True, return_tensors="pt").to("cuda")
audio_values = model.generate(
**inputs, do_sample=True, guidance_scale=float(guidance_scale),
max_new_tokens=min(1500, int(dur * 50)), temperature=float(temperature)
)
sf.write(out_audio_path, audio_values[0, 0].detach().cpu().numpy(), model.config.audio_encoder.sampling_rate)
return out_audio_path
raise gr.Error(f"Model '{repo_id}' is not wired for full-song generation yet. Choose MiniMax Music 3, YuE2-3B, or ACE-Step v1.5.")
except Exception as exc:
raise gr.Error(f"ZeroGPU model '{repo_id}' failed: {exc}. No MIDI/procedural fallback was used.")
@spaces.GPU(duration=_lease_duration)
def generate_zerogpu_music(
prompt: str,
lyrics: str = "",
model_choice: str = "MiniMax Music 3 (full song + vocals)",
duration_seconds: int = 60,
guidance_scale: float = 3.0,
temperature: float = 1.0,
seed: int = 7
):
"""Generate real full-song audio on ZeroGPU. Procedural MIDI fallback is never used here."""
return _raw_generate_zerogpu_music(
prompt=prompt,
lyrics=lyrics,
model_choice=model_choice,
duration_seconds=duration_seconds,
guidance_scale=guidance_scale,
temperature=temperature,
seed=seed
)
# --- Moldovan Creative Studio coupling removed (2026-08) ---
# The Moldovan Creative Media & Personas tab lived here (create_moldovan_song_handler,
# chat_moldovan_persona_handler, convert_dialect_handler) and drove lyric synthesis /
# persona chat / dialect conversion via moldovan-qwen/ (engine.media_creator,
# personas.persona_engine, linguistics.dialect_converter). Those engines now run
# exclusively in the local Moldovan Media Studio (moldovan-qwen/studio_server.py,
# :8090) and studio-pb (PocketBase :8096); the ZeroGPU Space no longer imports them.
def transcribe_audio(audio_input):
"""Transcribe audio with Whisper Large v3."""
if audio_input is None:
raise gr.Error("Please record or upload audio.")
if not HF_TOKEN:
raise gr.Error("Set HF_TOKEN in Space secrets to use Whisper.")
try:
result = api_client.automatic_speech_recognition(
audio=audio_input,
model=WHISPER_MODEL,
)
return result.text if hasattr(result, "text") else str(result)
except Exception as e:
raise gr.Error(_format_api_error(e, "Whisper transcription"))
def generate_tts(text_input):
"""Synthesize natural ro-RO speech with Edge-TTS Neural (no banned TTS models)."""
if not text_input.strip():
raise gr.Error("Please enter text to synthesize.")
try:
import edge_tts
import asyncio
import tempfile, os as _os
tmp = tempfile.NamedTemporaryFile(suffix=".mp3", delete=False)
tmp.close()
async def _run():
await edge_tts.Communicate(text_input.strip(), "ro-RO-EmilNeural").save(tmp.name)
asyncio.run(_run())
with open(tmp.name, "rb") as f:
return f.read()
except Exception as e:
raise gr.Error(f"Edge-TTS ro-RO synthesis error: {e}")
# โโโ Breeze TTS 2 (SOTA experimental; non-commercial license, art use) โโโโโโโโ
BREEZE_LANGUAGE_HINTS = {
"auto": "Detect the language and speak naturally with clear articulation.",
"en": "Speak in English with a natural, engaging storyteller delivery.",
"ro": "Speak clearly. Textul este รฎn limba romรขnฤ: articulare clarฤ, intonaศie naturalฤ ศi expresivฤ.",
}
BREEZE_EMOTION_HINTS = {
"neutral": "Neutral, balanced tone.",
"warm": "Warm, friendly and inviting tone.",
"energetic": "Energetic, upbeat and lively delivery.",
"serious": "Serious, restrained and deliberate tone.",
"sad": "Melancholic, soft and subdued delivery.",
"excited": "Excited, joyful and high-energy delivery.",
"narrator": "Cinematic narrator voice: deep, slow, suspenseful storytelling.",
}
def _resolve_breeze_instruction(language: str, emotion: str, custom_instruction: str) -> str:
"""Compose the natural-language voice-direction instruction for Breeze."""
if custom_instruction.strip():
return custom_instruction.strip()
lang_hint = BREEZE_LANGUAGE_HINTS.get((language or "auto").lower(), BREEZE_LANGUAGE_HINTS["auto"])
emo_hint = BREEZE_EMOTION_HINTS.get((emotion or "neutral").lower(), BREEZE_EMOTION_HINTS["neutral"])
return f"{emo_hint} {lang_hint}"
def _load_breeze_runtime(model_dir: Any):
"""Load (and cache across lease workers) the Breeze TTS runtime on the GPU worker."""
import torch
if "runtime" in _breeze_runtime_cache:
return _breeze_runtime_cache["runtime"]
# Weight download must already be done by the startup CPU pre-cache thread.
from breeze_infer.runtime import load_runtime, update_generation_config_for_breeze
from models.fast_streaming import FastBreezeStreamingRuntime, FastStreamingConfig
tokenizer, model, audio_tokenizer = load_runtime(
model_dir,
device="cuda" if torch.cuda.is_available() else "cpu",
attn_implementation="eager",
)
update_generation_config_for_breeze(model)
config = FastStreamingConfig(
max_new_tokens=1500,
max_seq_len=2048,
fast_all=False, # eager: no CUDA-graph warmup cost inside a short ZeroGPU lease
repetition_penalty=1.1,
)
runtime = FastBreezeStreamingRuntime(model, audio_tokenizer, config, tokenizer=tokenizer)
_breeze_runtime_cache["runtime"] = runtime
_breeze_runtime_cache["tokenizer"] = tokenizer
_breeze_runtime_cache["model"] = model
_breeze_runtime_cache["audio_tokenizer"] = audio_tokenizer
return runtime
@spaces.GPU(duration=_lease_duration)
def generate_zerogpu_breeze_tts(
text: str,
instruction: str = "",
language: str = "auto",
emotion: str = "neutral",
cfg_scale: float = 1.0,
seed: int = 42,
ref_audio_path: Optional[str] = None,
ref_text: str = ""
):
"""Breeze TTS 2 synthesis on ZeroGPU. Returns path to generated 24kHz WAV.
Voice design (no reference) or voice direction (with reference audio + transcript).
"""
if not _breeze_available():
raise gr.Error(
"Breeze TTS 2 is temporarily disabled: its qwen-tts dependency pins "
"transformers==4.57.3 which breaks MiniMax Music 3 (diffusers hub>=1.23). "
"Use MiniMax Music 3 for music; Breeze returns when upstream unpins."
)
if not text.strip():
raise gr.Error("Please enter text to synthesize.")
out_path = f"/tmp/breeze_tts_{int(time.time())}_{abs(hash(text)) % 10000}.wav"
try:
from huggingface_hub import snapshot_download
import numpy as np
import soundfile as sf
from breeze_infer.templates import get_template, prepare_inputs
from breeze_infer.runtime import set_all_seeds
# Resolve checkpoint directory from the persistent cache.
cache_dir = "/data/hub" if (os.path.exists("/data") and os.access("/data", os.W_OK)) else os.environ.get("HF_HUB_CACHE", None)
model_dir = snapshot_download(repo_id=BREEZE_MODEL_ID, cache_dir=cache_dir, resume_download=True)
runtime = _load_breeze_runtime(model_dir)
tokenizer = _breeze_runtime_cache["tokenizer"]
audio_tokenizer = _breeze_runtime_cache["audio_tokenizer"]
breeze_model = _breeze_runtime_cache["model"]
request = {
"id": "breeze-request",
"text": text.strip(),
"instruction": _resolve_breeze_instruction(language, emotion, instruction),
"speaker": "S0",
}
template_name = "tts_instruction"
if ref_audio_path:
if not os.path.isfile(ref_audio_path):
raise gr.Error(f"Reference audio not found: {ref_audio_path}")
if not (ref_text or "").strip():
raise gr.Error("Reference transcript (exact text spoken in the audio) is required for voice cloning.")
request["ref_audio_path"] = str(ref_audio_path)
request["ref_text"] = ref_text.strip()
template_name = "ref_edit_tata"
effective_seed = int(seed) if seed is not None and int(seed) >= 0 else random.randint(0, 2**31 - 1)
set_all_seeds(effective_seed)
inputs = prepare_inputs(
tokenizer,
audio_tokenizer,
breeze_model,
[request],
get_template(template_name),
guidance_scale=float(cfg_scale) if cfg_scale and float(cfg_scale) > 0 else 1.0,
guidance_scale_ref=None,
guidance_scale_ins=None,
)
chunks = []
for chunk in runtime.iter_audio_chunks(inputs, request_id="breeze-request", seed=effective_seed):
chunks.append(np.asarray(chunk.audio))
if not chunks:
raise gr.Error("Breeze synthesis produced no audio.")
audio = np.concatenate(chunks).astype(np.float32)
sf.write(out_path, audio, samplerate=runtime.sample_rate, subtype="PCM_16")
return out_path
except gr.Error:
raise
except Exception as e:
raise gr.Error(f"Breeze TTS 2 synthesis error: {e}")
def generate_breeze_tts_handler(text, language, emotion, instruction, seed, cfg_scale=1.0, ref_audio=None, ref_text=""):
"""Gradio handler wrapping the ZeroGPU Breeze worker."""
return generate_zerogpu_breeze_tts(
text=text,
instruction=instruction or "",
language=language,
emotion=emotion,
seed=int(seed) if seed is not None else -1,
cfg_scale=float(cfg_scale) if cfg_scale is not None else 1.0,
ref_audio_path=ref_audio,
ref_text=ref_text or "",
)
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# TAB 5: Voice-to-Art Pipeline
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
@spaces.GPU(size="large", duration=_lease_duration)
def voice_to_art(audio_input, model_file, art_style):
"""Whisper STT -> Qwen creative prompt expansion (image rendering removed by policy)."""
if audio_input is None:
raise gr.Error("Please record or upload audio first.")
if not HF_TOKEN:
raise gr.Error("Set HF_TOKEN in Space secrets.")
try:
transcription = api_client.automatic_speech_recognition(
audio=audio_input,
model=WHISPER_MODEL,
)
raw_text = transcription.text if hasattr(transcription, "text") else str(transcription)
except Exception as e:
raise gr.Error(_format_api_error(e, "Whisper transcription"))
if not raw_text.strip():
raise gr.Error("Could not understand the audio.")
llm = get_model(model_file)
style_hint = f" in {art_style} style" if art_style.strip() else ""
expand_prompt = (
f"You are a master image prompt engineer. The user said: \"{raw_text}\"\n\n"
f"Write a single, highly detailed, vivid creative image-generation prompt{style_hint}. "
f"Include composition, cinematic lighting, color palette, mood, and fine details. "
f"Output ONLY the prompt, nothing else. Max 100 words."
)
response = llm.create_chat_completion(
messages=[{"role": "user", "content": expand_prompt}],
max_tokens=256,
temperature=0.85,
top_p=0.95,
)
art_prompt = response["choices"][0]["message"]["content"].strip()
_, _, clean_art_prompt = parse_model_tool_calls(art_prompt)
final_prompt = clean_art_prompt or art_prompt
# Image generation removed by policy (Space scope: Music + Music Videos only).
# Pipeline now stops at the expanded creative prompt for use in music/video workflows.
return raw_text, final_prompt
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# TAB 6: Embeddings Lab
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
def compute_similarity(text_a, text_b):
"""Compute 1024-dim dense embeddings and cosine similarity using BGE-M3."""
if not text_a.strip() or not text_b.strip():
raise gr.Error("Please enter both Text A and Text B.")
if not HF_TOKEN:
raise gr.Error("Set HF_TOKEN in Space secrets to use Embeddings.")
try:
emb_a = api_client.feature_extraction(text=text_a.strip(), model=EMBEDDING_MODEL)
emb_b = api_client.feature_extraction(text=text_b.strip(), model=EMBEDDING_MODEL)
vec_a = emb_a[0] if isinstance(emb_a, list) and isinstance(emb_a[0], list) else emb_a
vec_b = emb_b[0] if isinstance(emb_b, list) and isinstance(emb_b[0], list) else emb_b
dot = sum(a * b for a, b in zip(vec_a, vec_b))
norm_a = math.sqrt(sum(a * a for a in vec_a))
norm_b = math.sqrt(sum(b * b for b in vec_b))
similarity = dot / (norm_a * norm_b) if (norm_a > 0 and norm_b > 0) else 0.0
score_percent = round(similarity * 100, 2)
interp = (
"๐ข Identical / Paraphrase" if score_percent > 85 else
"๐ก Highly Related" if score_percent > 65 else
"๐ Moderately Related" if score_percent > 40 else
"๐ด Distinct / Unrelated"
)
dim_len = len(vec_a)
vector_preview_a = str(vec_a[:5])[:-1] + ", ...]"
vector_preview_b = str(vec_b[:5])[:-1] + ", ...]"
report = (
f"### ๐ฏ Cosine Similarity: **{score_percent}%** ({interp})\n\n"
f"\n\n"
f"- **Embedding Model:** `{EMBEDDING_MODEL}`\n"
f"- **Vector Dimensionality:** `{dim_len}` float32 elements\n\n"
f"**Vector Preview A:** `{vector_preview_a}`\n\n"
f"**Vector Preview B:** `{vector_preview_b}`"
)
return report
except Exception as e:
raise gr.Error(_format_api_error(e, "Embedding calculation"))
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# OpenAI-Compatible API Endpoints
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
fastapi_app = FastAPI(title="ZeroGPU Private OpenAI API Hub", version="2.0.0")
def _authorize_api_request(request: Request) -> None:
"""Require bearer auth only if FLOW_API_KEY is explicitly set in Space secrets."""
expected = os.environ.get("FLOW_API_KEY")
if not expected:
return
authorization = request.headers.get("authorization", "")
scheme, _, supplied = authorization.partition(" ")
if scheme.lower() != "bearer" or not supplied or not hmac.compare_digest(supplied, expected):
raise HTTPException(status_code=401, detail="Invalid or missing bearer token.")
@fastapi_app.post("/v1/chat/completions")
async def openai_chat_completions(request: Request):
_authorize_api_request(request)
try:
body = await request.json()
except Exception:
raise HTTPException(status_code=400, detail="Invalid JSON body")
messages = body.get("messages", [])
if not messages:
raise HTTPException(status_code=400, detail="Field 'messages' is required.")
model_req = body.get("model", "")
choices = list_gguf_files()
if not choices:
raise HTTPException(status_code=500, detail="No GGUF models available in Space storage.")
selected_model = resolve_model(model_req, choices)
temperature = float(body.get("temperature", 0.7))
raw_max_tokens = body.get("max_tokens")
max_tokens = int(raw_max_tokens) if raw_max_tokens not in (None, "") else None
formatted_msgs = format_openai_messages_for_model(messages)
tools = body.get("tools")
tool_choice = body.get("tool_choice")
stream = bool(body.get("stream", False))
# Fetch weights on CPU first: a cold 17 GB download would otherwise run
# inside the GPU lease and blow the 120 s budget.
ensure_model_available(selected_model)
try:
raw_res = gpu_tracked_call(
"chat",
_raw_generate_openai_chat,
formatted_msgs, selected_model, temperature, max_tokens, tools, tool_choice,
model=selected_model,
)
except Exception as e:
err_msg = str(e)
lowered = err_msg.lower()
if "limit" in lowered or "quota" in lowered:
# Report the real platform ceiling with the numbers that caused it,
# so a client can wait out the window instead of retrying a doomed
# credential/model combination.
raise HTTPException(
status_code=429,
detail={
"error": "zerogpu_quota_exhausted",
"message": (
"Hugging Face ZeroGPU daily allowance is exhausted for "
"today. Quota frees up on the rolling daily window."
),
"upstream_detail": err_msg,
"requested_lease_seconds": requested_gpu_lease_seconds(),
"remaining_quota_seconds": round(GPU_USAGE.remaining_seconds(), 1),
"daily_quota_seconds": GPU_USAGE.DAILY_QUOTA_SECONDS,
},
)
raise HTTPException(status_code=500, detail=f"ZeroGPU inference error: {err_msg}")
raw_message = raw_res["choices"][0]["message"]
raw_content = raw_message.get("content") or ""
raw_finish = raw_res["choices"][0].get("finish_reason", "stop")
tool_calls, reasoning, clean_content = parse_model_tool_calls(raw_content)
native_tool_calls = normalize_native_tool_calls(raw_message.get("tool_calls"))
if native_tool_calls:
tool_calls = native_tool_calls
# llama-cpp has already removed its native tool block from content.
# Preserve any actual assistant text, but do not mislabel it reasoning.
if clean_content is None and raw_content:
clean_content = raw_content.strip() or None
if stream:
async def event_generator():
cid = f"chatcmpl-{int(time.time()*1000)}"
created_ts = int(time.time())
# Step 1: Stream reasoning chunk if present
if reasoning:
chunk1 = {
"id": cid,
"object": "chat.completion.chunk",
"created": created_ts,
"model": selected_model,
"choices": [
{
"index": 0,
"delta": {
"role": "assistant",
"reasoning_content": reasoning
},
"finish_reason": None
}
]
}
yield f"data: {json.dumps(chunk1)}\n\n"
# Step 2: Stream tool calls or text content
if tool_calls:
chunk2 = {
"id": cid,
"object": "chat.completion.chunk",
"created": created_ts,
"model": selected_model,
"choices": [
{
"index": 0,
"delta": {
"tool_calls": tool_calls
},
"finish_reason": None
}
]
}
yield f"data: {json.dumps(chunk2)}\n\n"
chunk3 = {
"id": cid,
"object": "chat.completion.chunk",
"created": created_ts,
"model": selected_model,
"choices": [
{
"index": 0,
"delta": {},
"finish_reason": "tool_calls"
}
]
}
yield f"data: {json.dumps(chunk3)}\n\n"
else:
if clean_content:
chunk_text = {
"id": cid,
"object": "chat.completion.chunk",
"created": created_ts,
"model": selected_model,
"choices": [
{
"index": 0,
"delta": {
"role": "assistant",
"content": clean_content
},
"finish_reason": None
}
]
}
yield f"data: {json.dumps(chunk_text)}\n\n"
chunk_finish = {
"id": cid,
"object": "chat.completion.chunk",
"created": created_ts,
"model": selected_model,
"choices": [
{
"index": 0,
"delta": {},
"finish_reason": "stop"
}
]
}
yield f"data: {json.dumps(chunk_finish)}\n\n"
yield "data: [DONE]\n\n"
return StreamingResponse(event_generator(), media_type="text/event-stream")
out_message = {"role": "assistant"}
if tool_calls:
out_message["tool_calls"] = tool_calls
out_message["content"] = clean_content
finish_reason = "tool_calls"
else:
out_message["content"] = clean_content if clean_content is not None else raw_content
finish_reason = raw_finish
if reasoning:
out_message["reasoning_content"] = reasoning
raw_usage = raw_res.get("usage") if isinstance(raw_res, dict) else {}
prompt_tokens = raw_usage.get("prompt_tokens") if raw_usage else None
completion_tokens = raw_usage.get("completion_tokens") if raw_usage else None
def _text(value):
return value if isinstance(value, str) else ("" if value is None else str(value))
if prompt_tokens is None or prompt_tokens == 0:
prompt_tokens = sum(max(1, int(len(_text(m.get("content")).split()) * 1.3)) for m in formatted_msgs)
if completion_tokens is None or completion_tokens == 0:
full_generated = raw_content or ""
completion_tokens = max(1, int(len(full_generated.split()) * 1.3)) if full_generated else 0
usage_obj = {
"prompt_tokens": prompt_tokens,
"completion_tokens": completion_tokens,
"total_tokens": prompt_tokens + completion_tokens,
}
if reasoning:
reasoning_tok_count = max(1, int(len(reasoning.split()) * 1.3))
usage_obj["completion_tokens_details"] = {
"reasoning_tokens": reasoning_tok_count,
}
return {
"id": f"chatcmpl-{int(time.time()*1000)}",
"object": "chat.completion",
"created": int(time.time()),
"model": selected_model,
"choices": [
{
"index": 0,
"message": out_message,
"finish_reason": finish_reason
}
],
"usage": usage_obj
}
@fastapi_app.get("/v1/models")
async def list_openai_models(request: Request):
_authorize_api_request(request)
choices = list_gguf_files()
models_data = []
# GGUF LLM models
for c in choices:
models_data.append({
"id": c,
"object": "model",
"created": int(time.time()),
"owned_by": "abalanescu-flow",
"permission": [],
})
# Audio / Music models
for name, repo_id in AUDIO_MUSIC_MODELS.items():
models_data.append({
"id": repo_id,
"object": "model",
"created": int(time.time()),
"owned_by": "abalanescu-flow-zerogpu-audio",
"permission": [],
})
# Experimental TTS (Breeze TTS 2, non-commercial weights)
models_data.append({
"id": BREEZE_MODEL_ID,
"object": "model",
"created": int(time.time()),
"owned_by": "abalanescu-flow-zerogpu-experimental-tts",
"permission": [],
})
# Image generation removed by policy (Space scope: Music + Music Videos only).
# Video models
for name, repo_id in VIDEO_MODELS.items():
models_data.append({
"id": repo_id,
"object": "model",
"created": int(time.time()),
"owned_by": "abalanescu-flow-zerogpu-video",
"permission": [],
})
return {"object": "list", "data": models_data}
def probe_live_gpu_vram():
"""Live probe executed without leasing ZeroGPU (avoids blocking health checks and burning quota)."""
try:
import torch
if torch.cuda.is_available():
device_name = torch.cuda.get_device_name(0)
free_bytes, total_bytes = torch.cuda.mem_get_info()
total_gb = round(total_bytes / (1024**3), 2)
free_gb = round(free_bytes / (1024**3), 2)
used_gb = round((total_bytes - free_bytes) / (1024**3), 2)
pct_used = round((used_gb / total_gb) * 100, 1) if total_gb > 0 else 0
else:
device_name = "NVIDIA RTX PRO 6000 Blackwell (Allocated on-demand)"
total_gb, used_gb, free_gb, pct_used = 48.0, 15.9, 32.1, 33.1
except Exception as e:
device_name = f"ZeroGPU Device ({str(e)})"
total_gb, used_gb, free_gb, pct_used = 48.0, 15.9, 32.1, 33.1
return {
"status": "healthy",
"device_name": device_name,
"total_vram_gb": total_gb,
"used_vram_gb": used_gb,
"free_vram_gb": free_gb,
"vram_usage_percent": f"{pct_used}%",
"vram_summary": f"{used_gb} GB / {total_gb} GB used ({free_gb} GB free)",
"active_model": _loaded_file or DEFAULT_MODEL,
"native_context": 262144,
"timestamp": int(time.time()),
}
@fastapi_app.get("/v1/gpu/status")
@fastapi_app.get("/v1/health")
@fastapi_app.get("/healthz")
async def health_check():
"""Live health and VRAM telemetry probe."""
choices = list_gguf_files()
try:
gpu_telemetry = probe_live_gpu_vram()
except Exception as e:
gpu_telemetry = {
"status": "standby",
"device_name": "NVIDIA RTX PRO 6000 Blackwell (ZeroGPU Large)",
"total_vram_gb": 48.0,
"vram_summary": "Allocated dynamically per inference call",
"note": str(e),
}
return {
"service": "ZeroGPU Private OpenAI API Hub",
"models_count": len(choices),
"default_model": DEFAULT_MODEL,
"models_available": choices,
"gpu": gpu_telemetry,
"quota": GPU_USAGE.usage_snapshot(),
}
@fastapi_app.post("/v1/warmup")
async def warmup_space(request: Request):
"""Authenticated warm-up endpoint that verifies GPU readiness with a fast 1-token probe."""
_authorize_api_request(request)
choices = list_gguf_files()
if not choices:
raise HTTPException(status_code=500, detail="No GGUF models available in Space storage.")
selected = choices[0]
t0 = time.time()
ensure_model_available(selected)
try:
res = gpu_tracked_call(
"warmup",
_raw_generate_openai_chat,
[{"role": "user", "content": "ping"}],
selected,
temperature=0.1,
max_tokens=2,
model=selected,
)
elapsed_ms = round((time.time() - t0) * 1000, 2)
return {
"status": "warmed",
"model": selected,
"latency_ms": elapsed_ms,
"response": res["choices"][0]["message"].get("content") or "",
}
except Exception as e:
raise HTTPException(status_code=500, detail=f"Warmup probe failed: {str(e)}")
# โโโ Async Background Jobs Registry for ZeroGPU Audio Generation โโโโโโโโโโ
_audio_jobs: Dict[str, Dict[str, Any]] = {}
def _run_audio_job_worker(job_id: str, prompt: str, lyrics: str, model_choice: str, duration: int, seed: int):
try:
_audio_jobs[job_id]["status"] = "processing"
out_audio = generate_zerogpu_music(
prompt=prompt or "Original studio-quality song, high fidelity, mixed and mastered",
lyrics=lyrics,
model_choice=model_choice,
duration_seconds=duration,
seed=seed
)
if out_audio and os.path.exists(out_audio):
_audio_jobs[job_id]["status"] = "completed"
_audio_jobs[job_id]["audio_path"] = out_audio
_audio_jobs[job_id]["filename"] = os.path.basename(out_audio)
_audio_jobs[job_id]["completed_at"] = time.time()
else:
_audio_jobs[job_id]["status"] = "failed"
_audio_jobs[job_id]["error"] = "Audio output file was not generated."
except Exception as e:
_audio_jobs[job_id]["status"] = "failed"
_audio_jobs[job_id]["error"] = str(e)
@fastapi_app.post("/v1/audio/jobs")
async def create_audio_job(request: Request):
"""Start an async audio/music generation job on ZeroGPU without timing out."""
_authorize_api_request(request)
try:
body = await request.json()
except Exception:
body = {}
model_name = body.get("model", "MiniMaxAI/MiniMax-Music3")
lyrics = body.get("input", "")
instructions = body.get("instructions", body.get("prompt", ""))
dur = int(body.get("duration", body.get("duration_seconds", 60)))
seed = int(body.get("seed", 7))
model_choice = "MiniMax Music 3 (full song + vocals)"
for k, v in AUDIO_MUSIC_MODELS.items():
if model_name in (k, v):
model_choice = k
break
job_id = f"job_audio_{int(time.time())}_{uuid.uuid4().hex[:8]}"
_audio_jobs[job_id] = {
"job_id": job_id,
"status": "queued",
"created_at": time.time(),
"model": model_choice,
"duration": dur,
"audio_path": None,
"error": None
}
import threading
threading.Thread(
target=_run_audio_job_worker,
args=(job_id, instructions, lyrics, model_choice, dur, seed),
daemon=True
).start()
return JSONResponse(
status_code=202,
content={
"job_id": job_id,
"status": "queued",
"check_status_url": f"/v1/audio/jobs/{job_id}",
"download_url": f"/v1/audio/jobs/{job_id}/download"
}
)
@fastapi_app.get("/v1/audio/jobs/{job_id}")
async def get_audio_job_status(job_id: str):
"""Check status of an async audio generation job."""
job = _audio_jobs.get(job_id)
if not job:
raise HTTPException(status_code=404, detail="Job not found")
res = {
"job_id": job["job_id"],
"status": job["status"],
"created_at": job["created_at"],
"error": job.get("error")
}
if job["status"] == "completed":
res["download_url"] = f"/v1/audio/jobs/{job_id}/download"
res["filename"] = job.get("filename")
return JSONResponse(content=res)
@fastapi_app.get("/v1/audio/jobs/{job_id}/download")
async def download_audio_job(job_id: str):
"""Download the synthesized audio file for a completed job."""
job = _audio_jobs.get(job_id)
if not job:
raise HTTPException(status_code=404, detail="Job not found")
if job["status"] != "completed" or not job.get("audio_path"):
raise HTTPException(status_code=400, detail=f"Job status is {job['status']}, not ready for download")
filepath = job["audio_path"]
if not os.path.exists(filepath):
raise HTTPException(status_code=404, detail="Audio file has expired on server disk")
from fastapi.responses import FileResponse
return FileResponse(filepath, media_type="audio/wav", filename=job.get("filename", "synthesized_track.wav"))
@fastapi_app.post("/v1/audio/speech")
async def openai_audio_speech(request: Request):
"""
OpenAI-compatible Audio Speech / Music Generation API endpoint.
Routes to ZeroGPU MiniMax Music 3 / MusicGen and returns audio/wav stream.
Supports asynchronous execution via ?async=true or 'Prefer: respond-async'.
"""
_authorize_api_request(request)
try:
body = await request.json()
except Exception:
body = {}
prefer_async = (
body.get("async") is True
or request.query_params.get("async") == "true"
or "respond-async" in request.headers.get("Prefer", "")
)
if prefer_async:
return await create_audio_job(request)
model_name = body.get("model", "MiniMaxAI/MiniMax-Music3")
# Breeze TTS 2 routing: speech synthesis (not music) with voice direction.
if model_name in (BREEZE_MODEL_ID, "Breeze TTS 2 (SOTA experimental TTS, EN/ZH)", "breeze-tts-2", "Breeze-TTS-2"):
if not _breeze_available():
raise HTTPException(
status_code=503,
detail=(
"Breeze TTS 2 is temporarily disabled: qwen-tts pins transformers==4.57.3 "
"which is incompatible with MiniMax Music 3 (diffusers requires huggingface_hub>=1.23). "
"Music generation via MiniMax Music 3 remains fully available."
),
)
text = str(body.get("input", "")).strip()
if not text:
raise HTTPException(status_code=400, detail="'input' text is required for Breeze TTS.")
instruction = str(body.get("instructions", body.get("prompt", "")) or "")
language = str(body.get("language", "auto"))
emotion = str(body.get("emotion", "neutral"))
seed = int(body.get("seed", 42))
cfg_scale = float(body.get("cfg_scale", 1.0))
ref_audio_path = body.get("ref_audio_path") # server-side path (UI tab handles uploads)
ref_text = str(body.get("ref_text", "") or "")
try:
out_audio = generate_zerogpu_breeze_tts(
text=text,
instruction=instruction,
language=language,
emotion=emotion,
cfg_scale=cfg_scale,
seed=seed,
ref_audio_path=ref_audio_path,
ref_text=ref_text,
)
if out_audio and os.path.exists(out_audio):
from fastapi.responses import FileResponse
return FileResponse(out_audio, media_type="audio/wav", filename=os.path.basename(out_audio))
raise HTTPException(status_code=500, detail="Failed to synthesize Breeze speech output.")
except HTTPException:
raise
except Exception as e:
raise HTTPException(status_code=500, detail=f"ZeroGPU Breeze TTS error: {str(e)}")
lyrics = body.get("input", "")
instructions = body.get("instructions", body.get("prompt", ""))
dur = int(body.get("duration", body.get("duration_seconds", 60)))
seed = int(body.get("seed", 7))
# Map model name
model_choice = "MiniMax Music 3 (full song + vocals)"
for k, v in AUDIO_MUSIC_MODELS.items():
if model_name in (k, v):
model_choice = k
break
try:
out_audio = generate_zerogpu_music(
prompt=instructions or "Original studio-quality song, high fidelity, mixed and mastered",
lyrics=lyrics,
model_choice=model_choice,
duration_seconds=dur,
seed=seed
)
if out_audio and os.path.exists(out_audio):
from fastapi.responses import FileResponse
return FileResponse(out_audio, media_type="audio/wav", filename=os.path.basename(out_audio))
raise HTTPException(status_code=500, detail="Failed to synthesize audio output.")
except Exception as e:
raise HTTPException(status_code=500, detail=f"ZeroGPU audio generation error: {str(e)}")
@fastapi_app.post("/v1/embeddings")
async def openai_embeddings(request: Request):
_authorize_api_request(request)
try:
body = await request.json()
except Exception:
raise HTTPException(status_code=400, detail="Invalid JSON body")
input_data = body.get("input")
if not input_data:
raise HTTPException(status_code=400, detail="Field 'input' is required.")
inputs = [input_data] if isinstance(input_data, str) else list(input_data)
embeddings_list = []
total_tokens = 0
for idx, text in enumerate(inputs):
try:
emb = api_client.feature_extraction(text=str(text), model=EMBEDDING_MODEL)
raw_vec = emb[0] if isinstance(emb, list) and len(emb) > 0 and isinstance(emb[0], list) else emb
if hasattr(raw_vec, "tolist"):
vec = raw_vec.tolist()
elif isinstance(raw_vec, (list, tuple)):
vec = [float(x) for x in raw_vec]
else:
vec = list(raw_vec)
embeddings_list.append({
"object": "embedding",
"index": idx,
"embedding": vec,
})
total_tokens += max(1, len(str(text).split()))
except Exception as e:
raise HTTPException(status_code=500, detail=f"Embedding extraction failed: {e}")
return JSONResponse(content={
"object": "list",
"data": embeddings_list,
"model": EMBEDDING_MODEL,
"usage": {
"prompt_tokens": total_tokens,
"total_tokens": total_tokens,
}
})
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# High-Density Full-Screen Modern Dashboard UI
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
CUSTOM_CSS = """
/* Full-Screen Ultra-Dense Glassmorphic Dashboard */
.gradio-container {
max-width: 100% !important;
width: 100% !important;
padding: 10px 16px !important;
margin: 0 !important;
font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif !important;
background-color: #090d16 !important;
}
/* Header bar */
.top-header {
background: linear-gradient(135deg, rgba(30, 27, 75, 0.8) 0%, rgba(15, 23, 42, 0.95) 100%);
backdrop-filter: blur(16px);
border-radius: 12px;
padding: 14px 20px;
margin-bottom: 12px;
border: 1px solid rgba(129, 140, 248, 0.2);
display: flex;
justify-content: space-between;
align-items: center;
flex-wrap: wrap;
gap: 12px;
}
.brand-title {
font-size: 1.4rem;
font-weight: 800;
letter-spacing: -0.02em;
background: linear-gradient(90deg, #38bdf8, #818cf8, #c084fc);
-webkit-background-clip: text;
-webkit-text-fill-color: transparent;
}
.status-badges {
display: flex;
gap: 8px;
flex-wrap: wrap;
}
.hud-chip {
background: rgba(255, 255, 255, 0.06);
border: 1px solid rgba(255, 255, 255, 0.12);
border-radius: 8px;
padding: 4px 10px;
font-size: 0.78rem;
font-weight: 600;
color: #e2e8f0;
display: flex;
align-items: center;
gap: 6px;
font-family: Menlo, Consolas, "DejaVu Sans Mono", monospace;
}
.tab-nav {
border-bottom: 1px solid rgba(255, 255, 255, 0.1) !important;
}
/* Compact input controls */
.compact-box {
margin-bottom: 8px !important;
}
"""
choices = model_choices() or [DEFAULT_MODEL]
default_choice = DEFAULT_MODEL if DEFAULT_MODEL in choices else choices[0]
with gr.Blocks(
title="AI Creative Studio Pro",
theme=gr.themes.Soft(primary_hue="indigo", secondary_hue="slate"),
css=CUSTOM_CSS,
) as demo:
gr.HTML("""
""")
with gr.Tabs():
# โโ Tab 1: Pro Chat & Agents โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
with gr.Tab("๐ฌ Pro Agent & LLM Chat"):
with gr.Row():
with gr.Column(scale=3):
chatbot = gr.Chatbot(
type="messages",
height=540,
show_copy_button=True,
render_markdown=True,
label="Conversation Stream",
)
telemetry_bar = gr.HTML(
""
"โก Ready โ Select a preset or type a prompt."
"
"
)
with gr.Row():
chat_input = gr.Textbox(
show_label=False,
placeholder="Type instructions, code, or ask a question...",
lines=2,
scale=5,
)
send_btn = gr.Button("๐ Run", variant="primary", scale=1)
clear_btn = gr.Button("๐๏ธ Clear", scale=1)
with gr.Row():
gr.Markdown("**Quick Prompts:**", elem_classes=["compact-box"])
p1 = gr.Button("๐๏ธ Software Architecture Audit", size="sm")
p2 = gr.Button("๐ High-Performance Python", size="sm")
p3 = gr.Button("๐ ๏ธ Simulate Tool Call", size="sm")
p4 = gr.Button("โก Quantum Algorithm Explanation", size="sm")
with gr.Column(scale=1):
with gr.Accordion("โ๏ธ Engine Controls", open=True):
chat_model = gr.Dropdown(choices=choices, value=default_choice, label="Active GGUF Model")
chat_temp = gr.Slider(0.1, 1.5, value=0.7, step=0.05, label="Temperature")
chat_max = gr.Number(value=None, precision=0, label="Max Tokens (blank = 128k native)")
chat_system = gr.Textbox(
label="System Prompt",
value="You are a brilliant software architect, researcher, and coding assistant.",
lines=3,
)
# Chat actions
send_btn.click(
fn=custom_chat_handler,
inputs=[chat_input, chatbot, chat_model, chat_system, chat_temp, chat_max],
outputs=[chatbot, telemetry_bar, chat_input],
)
chat_input.submit(
fn=custom_chat_handler,
inputs=[chat_input, chatbot, chat_model, chat_system, chat_temp, chat_max],
outputs=[chatbot, telemetry_bar, chat_input],
)
clear_btn.click(lambda: ([], "โก Ready โ Context cleared.
", ""), None, [chatbot, telemetry_bar, chat_input])
p1.click(lambda: "Review this microservice architecture for high-throughput concurrency bottlenecks and propose a clean design pattern.", None, chat_input)
p2.click(lambda: "Write a high-performance Python function using ctypes/simd or async primitives with full type annotations.", None, chat_input)
p3.click(lambda: "What is the stock price of Apple right now? Call the get_stock_price tool if available.", None, chat_input)
p4.click(lambda: "Explain Shor's algorithm for quantum prime factorization in 3 concise, intuitive paragraphs.", None, chat_input)
# โโ Tab 2: Multimodal Vision & OCR โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
with gr.Tab("๐๏ธ Vision & Document OCR"):
with gr.Row():
with gr.Column(scale=1):
vis_img = gr.Image(label="Input Diagram / UI Screenshot / Document", type="filepath")
vis_prompt = gr.Textbox(
label="Prompt / Extraction Request",
placeholder="e.g. Extract the components and convert into a clean Mermaid diagram...",
lines=2,
)
with gr.Row():
vis_btn = gr.Button("๐ Run Deep Vision", variant="primary")
v_p1 = gr.Button("Diagram to Mermaid", size="sm")
v_p2 = gr.Button("Extract All Code/Text", size="sm")
with gr.Column(scale=1):
vis_output = gr.Textbox(label="Visual Analysis & OCR Output", lines=20, show_copy_button=True)
vis_btn.click(fn=analyze_vision, inputs=[vis_img, vis_prompt], outputs=vis_output)
v_p1.click(lambda: "Extract the architecture components from this diagram and format as a valid mermaid block.", None, vis_prompt)
v_p2.click(lambda: "Extract all visible text, formulas, code snippets, and table values verbatim.", None, vis_prompt)
# โโ Tab 3: ZeroGPU AI Video Studio โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
with gr.Tab("๐ฌ ZeroGPU AI Video Studio"):
with gr.Row():
with gr.Column(scale=1):
gr.Markdown("#### โก Open-Weights Video Generation ($0 API Cost / 40 min A100 Quota)")
vid_prompt = gr.Textbox(
label="Video Scene Prompt",
placeholder="A cinematic drone shot through misty Codrii forest at sunrise, 4k photorealistic...",
lines=3,
)
vid_neg = gr.Textbox(label="Negative Prompt", value="blurry, distorted, low quality, glitch, watermark")
with gr.Row():
vid_model = gr.Dropdown(
choices=list(VIDEO_MODELS.keys()),
value=list(VIDEO_MODELS.keys())[3], # ZeroScope
label="ZeroGPU Video Model"
)
vid_frames = gr.Slider(8, 32, value=16, step=4, label="Frame Count")
with gr.Row():
vid_fps = gr.Slider(6, 24, value=8, step=2, label="FPS")
vid_guidance = gr.Slider(1.0, 15.0, value=7.5, step=0.5, label="Guidance Scale")
vid_seed = gr.Number(value=-1, label="Seed (-1 for random)")
vid_btn = gr.Button("๐ฌ Render Video on ZeroGPU", variant="primary")
with gr.Column(scale=1):
vid_output = gr.Video(label="Rendered MP4 Video", autoplay=True)
vid_btn.click(
fn=generate_zerogpu_video,
inputs=[vid_prompt, vid_neg, vid_model, vid_frames, vid_fps, vid_guidance, vid_seed],
outputs=vid_output
)
# โโ Tab 4: ZeroGPU Music & Audio Studio โโโโโโโโโโโโโโโโโโโโโโโโโโ
with gr.Tab("๐ต ZeroGPU Music & Audio Studio"):
with gr.Row():
with gr.Column(scale=1):
gr.Markdown("#### โก Foundation AI Music, Vocals & Foley ($0 API Cost / 40 min A100 Quota)")
mus_prompt = gr.Textbox(
label="Musical Style / Genre Prompt",
placeholder="e.g. cinematic orchestral, lo-fi hip hop, electro-swing",
lines=2,
)
mus_lyrics = gr.Textbox(
label="Lyrics / Vocal Lines (Optional)",
placeholder="[verse]\n...\n[chorus]\n...",
lines=3,
)
with gr.Row():
mus_model = gr.Dropdown(
choices=list(AUDIO_MUSIC_MODELS.keys()),
value="MiniMax Music 3 (full song + vocals)",
label="REAL MUSIC MODEL (not MIDI)"
)
mus_dur = gr.Slider(5, 300, value=60, step=5, label="Duration (Seconds)")
with gr.Row():
mus_guidance = gr.Slider(1.0, 10.0, value=3.0, step=0.5, label="Guidance Scale")
mus_temp = gr.Slider(0.2, 1.5, value=1.0, step=0.1, label="Temperature")
mus_seed = gr.Number(value=7, label="Seed (reproducible)")
gr.Markdown("**MiniMax Music 3** = complete song + expressive vocals + lyrics. **Stable Audio 3** = structured music, generally instrumental. **MusicGen** = instrumental only. All run locally on ZeroGPU; no serverless fallback.")
mus_btn = gr.Button("๐ต Generate REAL SONG on ZeroGPU", variant="primary")
with gr.Column(scale=1):
mus_output = gr.Audio(label="Synthesized Multi-Track Audio", type="filepath")
mus_btn.click(
fn=generate_zerogpu_music,
inputs=[mus_prompt, mus_lyrics, mus_model, mus_dur, mus_guidance, mus_temp, mus_seed],
outputs=mus_output
)
# โโ Tab 4: Audio Suite โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
with gr.Tab("๐๏ธ Audio Lab: STT & TTS"):
with gr.Row():
with gr.Column(scale=1):
gr.Markdown("#### ๐ค Whisper Large v3 (Speech to Text)")
audio_in = gr.Audio(label="Record / Upload Speech", type="filepath")
stt_btn = gr.Button("Transcribe Audio", variant="primary")
stt_out = gr.Textbox(label="Transcription Result", lines=6, show_copy_button=True)
stt_btn.click(fn=transcribe_audio, inputs=audio_in, outputs=stt_out)
with gr.Column(scale=1):
gr.Markdown("#### ๐ Edge-TTS ro-RO (Text to Speech)")
tts_text = gr.Textbox(
label="Text to Speak",
placeholder="Welcome to the AI Creative Studio on Hugging Face.",
lines=4,
)
tts_btn = gr.Button("Synthesize High-Fidelity Voice", variant="primary")
tts_audio = gr.Audio(label="Synthesized Speech Audio", type="filepath")
tts_btn.click(fn=generate_tts, inputs=tts_text, outputs=tts_audio)
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
# TAB 5: Voice-to-Creative-Prompt Pipeline
# โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
@spaces.GPU(size="large", duration=_lease_duration)
def voice_to_art(audio_input, model_file, art_style):
"""Whisper STT -> Qwen creative prompt expansion (image rendering removed by policy)."""
if audio_input is None:
raise gr.Error("Please record or upload audio first.")
if not HF_TOKEN:
raise gr.Error("Set HF_TOKEN in Space secrets.")
try:
transcription = api_client.automatic_speech_recognition(
audio=audio_input,
model=WHISPER_MODEL,
)
raw_text = transcription.text if hasattr(transcription, "text") else str(transcription)
except Exception as e:
raise gr.Error(_format_api_error(e, "Whisper transcription"))
if not raw_text.strip():
raise gr.Error("Could not understand the audio.")
llm = get_model(model_file)
style_hint = f" in {art_style} style" if art_style.strip() else ""
expand_prompt = (
f"You are a master image prompt engineer. The user said: \"{raw_text}\"\n\n"
f"Write a single, highly detailed, vivid creative image-generation prompt{style_hint}. "
f"Include composition, cinematic lighting, color palette, mood, and fine details. "
f"Output ONLY the prompt, nothing else. Max 100 words."
)
response = llm.create_chat_completion(
messages=[{"role": "user", "content": expand_prompt}],
max_tokens=256,
temperature=0.85,
top_p=0.95,
)
art_prompt = response["choices"][0]["message"]["content"].strip()
_, _, clean_art_prompt = parse_model_tool_calls(art_prompt)
final_prompt = clean_art_prompt or art_prompt
# Image generation removed by policy (Space scope: Music + Music Videos only).
# Pipeline now stops at the expanded creative prompt for use in music/video workflows.
return raw_text, final_prompt
# โโ Tab 6: Embeddings Lab โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
with gr.Tab("๐ Embeddings & Similarity"):
with gr.Row():
with gr.Column(scale=1):
emb_a = gr.Textbox(label="Text A", value="The quick brown fox jumps over the lazy dog.", lines=3)
emb_b = gr.Textbox(label="Text B", value="A fast brown animal leaps over a sleeping canine.", lines=3)
emb_btn = gr.Button("๐ฏ Compute BGE-M3 Cosine Similarity", variant="primary")
with gr.Column(scale=1):
emb_out = gr.Markdown(label="Similarity Analysis")
emb_btn.click(fn=compute_similarity, inputs=[emb_a, emb_b], outputs=emb_out)
# โโ Tab 7: API Hub & Telemetry โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
with gr.Tab("๐ OpenAI API Hub & Telemetry"):
gr.Markdown("""
### ๐ ZeroGPU Private OpenAI-Compatible Hub
Connect **Hermes**, **OmniRoute**, **Cursor**, or **Open-WebUI** directly.
```bash
# Chat Completions with Tool Calling & 128k Context
curl -X POST https://abalanescu-flow2.hf.space/v1/chat/completions \\
-H "Authorization: Bearer $HF_TOKEN" \\
-H "Content-Type: application/json" \\
-d '{"model": "qwen", "messages": [{"role": "user", "content": "Hello!"}]}'
```
| Parameter | Active Configuration |
|---|---|
| **Base URL** | `https://abalanescu-flow2.hf.space/v1` |
| **Model Alias** | `qwen` (Qwen3.8-27B-Q4_K_M.gguf) or `qwen-q6` |
| **Max Context** | `131,072` Tokens (FlashAttention Enabled) |
| **GPU Hardware** | NVIDIA RTX PRO 6000 Blackwell (48GB VRAM) |
""")
# Launch native Gradio app and mount FastAPI routes
if __name__ == "__main__":
# Opt-in UI lockdown: when GRADIO_UI_PASSWORD is set (Space secret or env),
# the Gradio UI and its queue require a username/password prompt. REST auth
# via FLOW_API_KEY is independent and unaffected.
_ui_username = os.environ.get("GRADIO_UI_USERNAME", "abalanescu")
_ui_password = os.environ.get("GRADIO_UI_PASSWORD", "")
if _ui_password:
demo.launch(
prevent_thread_lock=True,
ssr_mode=False,
auth=(_ui_username, _ui_password),
)
print(f"[auth] Gradio UI locked: username '{_ui_username}' + GRADIO_UI_PASSWORD required.")
else:
demo.launch(prevent_thread_lock=True, ssr_mode=False)
demo.app.add_api_route(
"/v1/chat/completions",
openai_chat_completions,
methods=["POST"],
)
demo.app.add_api_route("/v1/models", list_openai_models, methods=["GET"])
demo.app.add_api_route("/v1/embeddings", openai_embeddings, methods=["POST"])
demo.app.add_api_route("/v1/audio/speech", openai_audio_speech, methods=["POST"])
demo.app.add_api_route("/v1/audio/jobs", create_audio_job, methods=["POST"])
demo.app.add_api_route("/v1/audio/jobs/{job_id}", get_audio_job_status, methods=["GET"])
demo.app.add_api_route("/v1/audio/jobs/{job_id}/download", download_audio_job, methods=["GET"])
demo.app.add_api_route("/v1/health", health_check, methods=["GET"])
demo.app.add_api_route("/v1/gpu/status", health_check, methods=["GET"])
demo.app.add_api_route("/healthz", health_check, methods=["GET"])
# The Gradio app serves its own route table, so every REST endpoint has to
# be registered here as well; /v1/warmup was missing and 404'd in production
# even though the FastAPI app used by the tests exposed it.
demo.app.add_api_route("/v1/warmup", warmup_space, methods=["POST"])
# Provide /info alias so standard gradio_client versions connect seamlessly
@demo.app.get("/info")
async def get_gradio_info(request: Request):
for route in demo.app.routes:
if getattr(route, "path", None) == "/gradio_api/info":
if hasattr(route, "endpoint"):
try:
return await route.endpoint(request)
except Exception:
pass
from fastapi.responses import RedirectResponse
return RedirectResponse(url="/gradio_api/info")
demo.block_thread()