Instructions to use rbanfield/clip-vit-large-patch14 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use rbanfield/clip-vit-large-patch14 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("zero-shot-image-classification", model="rbanfield/clip-vit-large-patch14") pipe( "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/hub/parrots.png", candidate_labels=["animals", "humans", "landscape"], )# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForZeroShotImageClassification processor = AutoProcessor.from_pretrained("rbanfield/clip-vit-large-patch14") model = AutoModelForZeroShotImageClassification.from_pretrained("rbanfield/clip-vit-large-patch14", device_map="auto") - Notebooks
- Google Colab
- Kaggle
File size: 1,419 Bytes
b926327 8182cb1 b926327 e62633b b926327 8182cb1 b926327 418414b b926327 418414b b926327 418414b b926327 418414b b926327 418414b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 | from io import BytesIO
import base64
from PIL import Image
import torch
from transformers import CLIPProcessor, CLIPModel
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
class EndpointHandler():
def __init__(self, path=""):
self.model = CLIPModel.from_pretrained("rbanfield/clip-vit-large-patch14").to("cpu")
self.processor = CLIPProcessor.from_pretrained("rbanfield/clip-vit-large-patch14")
def __call__(self, data):
text_input = None
if isinstance(data, dict):
inputs = data.pop("inputs", None)
text_input = inputs.get('text',None)
image_data = BytesIO(base64.b64decode(inputs['image'])) if 'image' in inputs else None
else:
# assuming its an image sent via binary
image_data = BytesIO(data)
if text_input:
processor = self.processor(text=text_input, return_tensors="pt", padding=True).to(device)
with torch.no_grad():
return {"embeddings": self.model.get_text_features(**processor).tolist()}
elif image_data:
image = Image.open(image_data)
processor = self.processor(images=image, return_tensors="pt").to(device)
with torch.no_grad():
return {"embeddings": self.model.get_image_features(**processor).tolist()}
else:
return {"embeddings": None}
|