import json import os import urllib.request import gradio as gr from openai import OpenAI BASE_URL = "https://api.unorouter.com/v1" PRICING_URL = "https://unorouter.com/api/models/pricing" DEMO_KEY = os.environ.get("UNOROUTER_API_KEY", "") DEFAULT_MODEL = "step-3.7-flash:free" FALLBACK_MODELS = [ "step-3.7-flash:free", "qwen3.5-397b-a17b:free", "mistral-medium-3.5:free", "gemma-4-31b-it:free", "llama-3.3-70b:free", "kimi-k2.6:free", "glm-4.5-flash:free", "deepseek-v3.2:free", ] def load_free_models(): try: req = urllib.request.Request( PRICING_URL, headers={"User-Agent": "unorouter-hf-space"} ) with urllib.request.urlopen(req, timeout=15) as r: data = json.load(r) models = sorted( m["name"] for m in data.get("models", []) if m.get("isFree") and m.get("type") == "text" and "openai" in (m.get("endpointTypes") or []) ) return models or FALLBACK_MODELS except Exception: return FALLBACK_MODELS MODELS = load_free_models() if DEFAULT_MODEL not in MODELS: DEFAULT_MODEL = MODELS[0] def respond(message, history, model, api_key): key = (api_key or "").strip() or DEMO_KEY if not key: yield ( "No API key configured. Get a free one at https://unorouter.com " "(Discord or GitHub signup) and paste it below." ) return client = OpenAI(base_url=BASE_URL, api_key=key) messages = [ {"role": t["role"], "content": t["content"]} for t in history if t.get("role") in ("user", "assistant") and isinstance(t.get("content"), str) ] messages.append({"role": "user", "content": message}) try: stream = client.chat.completions.create( model=model, messages=messages, stream=True, max_tokens=2048 ) text = "" for chunk in stream: if chunk.choices and chunk.choices[0].delta.content: text += chunk.choices[0].delta.content yield text if not text: yield "(the model returned no visible text, try again or switch models)" except Exception as e: msg = str(e) if "busy" in msg or "rate limit" in msg.lower(): yield ( f"All free providers for {model} are at capacity right now. " "Free pools drain under load and recover automatically. " "Pick another model, or grab your own key at https://unorouter.com " "for full limits." ) else: yield f"Error: {msg}" with gr.Blocks(title="UnoRouter Free Chat") as demo: gr.Markdown( f"""# UnoRouter Free Chat Chat with **{len(MODELS)} free models** through [UnoRouter](https://unorouter.com), one OpenAI-compatible key for 200+ AI models. The same key also powers coding agents (Claude Code, Cline, OpenCode) and drops into SillyTavern or Janitor. [Get your own free key](https://unorouter.com) | [Discord](https://discord.com/invite/eRAeFd9aqy) | [GitHub (fully open source)](https://github.com/unorouter)""" ) model = gr.Dropdown(MODELS, value=DEFAULT_MODEL, label="Model (all free)") api_key = gr.Textbox( label="Your own UnoRouter API key (optional)", type="password", placeholder="sk-... leave empty to use the shared demo key", ) gr.ChatInterface( respond, type="messages", additional_inputs=[model, api_key], examples=[ ["Explain what an LLM gateway does in two sentences.", DEFAULT_MODEL, ""], ["Write a haiku about free AI models.", DEFAULT_MODEL, ""], ], ) gr.Markdown( "Shared demo key is limited to free models. If a model reports at capacity, " "its free pool is drained and will recover automatically." ) demo.queue(default_concurrency_limit=4).launch()