import os import spaces import gradio as gr import torch from fastapi.staticfiles import StaticFiles from starlette.responses import FileResponse from starlette.routing import Route FRONTEND = os.path.join(os.path.dirname(os.path.abspath(__file__)), "frontend") @spaces.GPU(duration=30) def infer(x): return f"cuda available: {torch.cuda.is_available()}, got: {x}" with gr.Blocks() as demo: inp, out = gr.Textbox(), gr.Textbox() gr.Button("Run").click(infer, inp, out, api_name="infer") demo.queue() # demo.launch() is the officially documented ZeroGPU entrypoint — its # startup hook is what registers @spaces.GPU functions. We keep it (rather # than gr.mount_gradio_app on our own FastAPI instance) and instead attach # custom routes/static files onto the live app it hands back. app, _local_url, _share_url = demo.launch( server_name="0.0.0.0", server_port=7860, prevent_thread_lock=True, ) @app.get("/api/health") def health(): return {"status": "ok"} app.mount("/static", StaticFiles(directory=FRONTEND), name="static") async def index(request): return FileResponse(os.path.join(FRONTEND, "index.html")) # Starlette returns the first matching route and Gradio registered its own "/" # during launch(), so inserting ahead of it hands the root URL to our UI. # Gradio's API routes stay intact — it just becomes an invisible GPU engine. app.router.routes.insert(0, Route("/", index)) demo.block_thread()