Spaces:
Running on Zero
Running on Zero
File size: 1,733 Bytes
c002fa3 896aa83 c002fa3 896aa83 c002fa3 896aa83 c002fa3 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 | """
Embedding API for a Hugging Face Space (Gradio SDK, ZeroGPU hardware slot,
but runs on CPU only — the actual embed() function never calls the GPU
decorator, so it consumes zero ZeroGPU quota. This is how a free personal
account can run a real Python backend now that CPU Basic requires PRO.
A no-op @spaces.GPU-decorated function is defined below solely because
HF's ZeroGPU runtime refuses to start a Space unless at least one such
function is present. It is never called, so it never claims GPU time.
"""
import json
import gradio as gr
import spaces
from sentence_transformers import SentenceTransformer
MODEL_NAME = "sentence-transformers-testing/stsb-bert-tiny-safetensors"
API_KEY = "hellonumbers77@" # change this to a strong random token
print(f"Loading model: {MODEL_NAME} ...")
model = SentenceTransformer(MODEL_NAME)
print("Model loaded. Embedding dim:", model.get_sentence_embedding_dimension())
@spaces.GPU
def _zerogpu_presence_stub():
"""Never called — exists only to satisfy ZeroGPU's startup check."""
pass
def embed(texts_json: str, api_key: str) -> str:
if api_key != API_KEY:
return json.dumps({"error": "unauthorized"})
try:
texts = json.loads(texts_json)
except json.JSONDecodeError:
return json.dumps({"error": "texts_json must be a JSON array of strings"})
vectors = model.encode(texts).tolist()
return json.dumps({"embeddings": vectors})
demo = gr.Interface(
fn=embed,
inputs=[
gr.Textbox(label="texts (JSON array)", value='["hello world"]'),
gr.Textbox(label="api_key", type="password"),
],
outputs=gr.Textbox(label="result (JSON)"),
api_name="embed",
)
if __name__ == "__main__":
demo.launch()
|