import gradio as gr from fastapi import FastAPI, Request, HTTPException from fastapi.responses import StreamingResponse import random import asyncio import aiohttp import uvicorn from src.config import get_api_keys from src.model_tester import ModelTester from src.scheduler import Scheduler model_tester = ModelTester() scheduler = Scheduler(task_callback=lambda: model_tester.scan_all_models()) fastapi_app = FastAPI(title="OpenRouter Free API") @fastapi_app.on_event("startup") async def startup_event(): scheduler.start() @fastapi_app.get("/v1/models") async def list_models(): available = model_tester.get_available_models(free_only=False) available_free = model_tester.get_available_models(free_only=True) models = [] for model_id in available: is_free = model_id in available_free models.append({ "id": model_id, "object": "model", "created": 1677610602, "owned_by": "openrouter", "free": is_free }) return {"object": "list", "data": models} async def _select_candidates(model_hint: str) -> list[str]: """获取候选模型列表(带排序:free 优先)""" await model_tester.refresh_model_list_async() all_free = model_tester.get_all_free_models() all_models = model_tester._all_models if not model_hint: return all_free[:10] hint = model_hint.lower() # free 候选(精确优先) free_exact = [] free_fuzzy = [] for m in all_free: name = m.replace(":free", "").split("/")[-1].lower() if name == hint: free_exact.append(m) elif name.startswith(hint) or hint in name: free_fuzzy.append(m) # 付费候选 free_set = set(all_free) paid_fuzzy = [] for m in all_models: if m in free_set: continue name = m.replace(":free", "").split("/")[-1].lower() if name == hint or hint in name or hint in m.lower(): paid_fuzzy.append(m) return (free_exact + free_fuzzy + paid_fuzzy)[:15] def _is_definitive_error(error: str) -> bool: """不可重试的错误(模型不存在/无权访问)""" keywords = ("HTTP 400", "HTTP 403", "HTTP 404", "model_not_found", "not found", "not_found") return any(k in error for k in keywords) async def _proxy_with_retry(body: dict, candidates: list[str]) -> dict: """ 带重试的代理请求。 遍历候选模型,对每个模型: - 遇到限流/超时/5xx → 换 key + 退避重试 - 遇到 400/403/404 → 跳到下一个候选 - 所有候选全失败 → 返回 400 """ api_keys = get_api_keys() max_retries = 3 # 每个模型最多重试次数 retry_delay = 5 # 退避基数(秒) for candidate in candidates: failed_keys: set = set() for attempt in range(max_retries): # 从可用 key 中选一个 available = [k for k in api_keys if k not in failed_keys] if not available: print(f"[proxy] All keys exhausted for {candidate}") break api_key = random.choice(available) body["model"] = candidate url = "https://openrouter.ai/api/v1/chat/completions" headers = { "Authorization": f"Bearer {api_key}", "Content-Type": "application/json" } try: async with aiohttp.ClientSession() as session: async with session.post(url, json=body, headers=headers, timeout=aiohttp.ClientTimeout(total=120)) as response: data = await response.json() if "error" in data: error_detail = str(data["error"]) print(f"[proxy] {candidate} → {error_detail[:100]}") # 不可重试 → 跳下一个候选 if _is_definitive_error(error_detail): break # 401 → 换 key if "401" in error_detail or "unauthorized" in error_detail.lower(): failed_keys.add(api_key) continue # 429 / 5xx / 其他 → 退避重试 sleep_s = retry_delay * (attempt + 1) print(f"[proxy] {candidate} retry {attempt + 1}/{max_retries} in {sleep_s}s") await asyncio.sleep(sleep_s) continue return data except (asyncio.TimeoutError, aiohttp.ClientError) as e: print(f"[proxy] {candidate} error: {e}") sleep_s = retry_delay * (attempt + 1) await asyncio.sleep(sleep_s) continue except Exception as e: print(f"[proxy] {candidate} unexpected error: {e}") sleep_s = retry_delay * (attempt + 1) await asyncio.sleep(sleep_s) continue # 当前候选所有重试用完,换下一个 raise HTTPException(status_code=400, detail="No available model responded") async def _proxy_streaming(body: dict, candidates: list[str]): """流式代理 — 逐个尝试候选模型""" api_keys = get_api_keys() url = "https://openrouter.ai/api/v1/chat/completions" body["stream"] = True for candidate in candidates: api_key = random.choice(api_keys) body["model"] = candidate headers = { "Authorization": f"Bearer {api_key}", "Content-Type": "application/json" } try: async with aiohttp.ClientSession() as session: async with session.post(url, json=body, headers=headers, timeout=aiohttp.ClientTimeout(total=120)) as response: if response.status != 200: print(f"[proxy:stream] {candidate} → HTTP {response.status}, trying next") continue async for chunk in response.content: yield chunk return except Exception as e: print(f"[proxy:stream] {candidate} error: {e}, trying next") continue raise HTTPException(status_code=400, detail="No available model for streaming") @fastapi_app.post("/v1/chat/completions") async def chat_completions(request: Request): body = await request.json() model_hint = body.get("model") candidates = await _select_candidates(model_hint) if not candidates: raise HTTPException(status_code=400, detail="No available model") if body.get("stream"): return StreamingResponse( _proxy_streaming(body, candidates), media_type="text/event-stream" ) return await _proxy_with_retry(body, candidates) @fastapi_app.get("/health") async def health(): return {"status": "ok"} def get_scan_status(): scan_result = model_tester.scan_result total = scan_result.get("total_available", 0) free = scan_result.get("free_available", 0) return f"Free: {free} | Total: {total}" def format_model_list(models): return "\n".join(models) if models else "No models available" with gr.Blocks(title="OpenRouter Free API") as demo: gr.Markdown("# OpenRouter Free API") gr.Markdown("Standard OpenAI-compatible API with free model support") gr.Markdown(f"**Status: {get_scan_status()}**") gr.Markdown("## Available Free Models") gr.Textbox(value=format_model_list(model_tester.get_available_models(free_only=True)), lines=15, interactive=False) app = gr.mount_gradio_app(fastapi_app, demo, path="/") if __name__ == "__main__": uvicorn.run(app, host="0.0.0.0", port=7860)