Download app.py from 640510702phithak/embed: direct link, hf CLI and curl.
- Browser
- Download file 1.02 kB
-
https://huggingface.co/640510702phithak/embed/resolve/main/app.py
- Command line
-
hf download hf://640510702phithak/embed/app.py
-
curl -L -o app.py https://huggingface.co/640510702phithak/embed/resolve/main/app.py
1.02 kB
| from fastapi import FastAPI | |
| from transformers import AutoTokenizer, AutoModel | |
| import torch | |
| # โหลดโมเดล Pre-trained เช่น Sentence-BERT | |
| MODEL_NAME = "sentence-transformers/all-MiniLM-L6-v2" | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME) | |
| model = AutoModel.from_pretrained(MODEL_NAME) | |
| # สร้าง FastAPI | |
| app = FastAPI() | |
| # ฟังก์ชันแปลงข้อความเป็นเวกเตอร์ | |
| def get_embedding(text): | |
| inputs = tokenizer(text, return_tensors="pt", padding=True, truncation=True) | |
| with torch.no_grad(): | |
| outputs = model(**inputs) | |
| embedding = outputs.last_hidden_state.mean(dim=1) # ใช้ค่าเฉลี่ยของ hidden states | |
| return embedding.squeeze().tolist() | |
| # API Endpoint | |
| async def embed_text(data: dict): | |
| text = data.get("text", "") | |
| if not text: | |
| return {"error": "No text provided"} | |
| vector = get_embedding(text) | |
| return {"text": text, "embedding": vector} | |