Blueberryaman's picture
Upload folder using huggingface_hub
896aa83 verified
Raw History Blame Contribute Delete
1.73 kB
"""
Embedding API for a Hugging Face Space (Gradio SDK, ZeroGPU hardware slot,
but runs on CPU only — the actual embed() function never calls the GPU
decorator, so it consumes zero ZeroGPU quota. This is how a free personal
account can run a real Python backend now that CPU Basic requires PRO.
A no-op @spaces.GPU-decorated function is defined below solely because
HF's ZeroGPU runtime refuses to start a Space unless at least one such
function is present. It is never called, so it never claims GPU time.
"""
import json
import gradio as gr
import spaces
from sentence_transformers import SentenceTransformer
MODEL_NAME = "sentence-transformers-testing/stsb-bert-tiny-safetensors"
API_KEY = "hellonumbers77@" # change this to a strong random token
print(f"Loading model: {MODEL_NAME} ...")
model = SentenceTransformer(MODEL_NAME)
print("Model loaded. Embedding dim:", model.get_sentence_embedding_dimension())
@spaces.GPU
def _zerogpu_presence_stub():
"""Never called — exists only to satisfy ZeroGPU's startup check."""
pass
def embed(texts_json: str, api_key: str) -> str:
if api_key != API_KEY:
return json.dumps({"error": "unauthorized"})
try:
texts = json.loads(texts_json)
except json.JSONDecodeError:
return json.dumps({"error": "texts_json must be a JSON array of strings"})
vectors = model.encode(texts).tolist()
return json.dumps({"embeddings": vectors})
demo = gr.Interface(
fn=embed,
inputs=[
gr.Textbox(label="texts (JSON array)", value='["hello world"]'),
gr.Textbox(label="api_key", type="password"),
],
outputs=gr.Textbox(label="result (JSON)"),
api_name="embed",
)
if __name__ == "__main__":
demo.launch()