Instructions to use spellingdragon/whisper-large-v3-handler with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use spellingdragon/whisper-large-v3-handler with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("automatic-speech-recognition", model="spellingdragon/whisper-large-v3-handler")# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForSpeechSeq2Seq processor = AutoProcessor.from_pretrained("spellingdragon/whisper-large-v3-handler") model = AutoModelForSpeechSeq2Seq.from_pretrained("spellingdragon/whisper-large-v3-handler", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download handler.py from spellingdragon/whisper-large-v3-handler: direct link, hf CLI and curl.
- Browser
- Download file 2.15 kB
-
https://huggingface.co/spellingdragon/whisper-large-v3-handler/resolve/main/handler.py
- Command line
-
hf download hf://spellingdragon/whisper-large-v3-handler/handler.py
-
curl -L -o handler.py https://huggingface.co/spellingdragon/whisper-large-v3-handler/resolve/main/handler.py
2.15 kB
| from typing import Dict, List, Any | |
| import torch | |
| from transformers.pipelines.audio_utils import ffmpeg_read | |
| from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, AutoTokenizer, pipeline | |
| class EndpointHandler(): | |
| def __init__(self, path=""): | |
| device = "cuda:0" if torch.cuda.is_available() else "cpu" | |
| torch_dtype = torch.float16 if torch.cuda.is_available() else torch.float32 | |
| model_id = "openai/whisper-large-v3-turbo" | |
| model = AutoModelForSpeechSeq2Seq.from_pretrained( | |
| model_id, torch_dtype=torch_dtype, low_cpu_mem_usage=True, use_safetensors=True | |
| ) | |
| model.to(device) | |
| processor = AutoProcessor.from_pretrained(model_id) | |
| self.pipeline = pipeline( | |
| "automatic-speech-recognition", | |
| model=model, | |
| tokenizer=processor.tokenizer, | |
| feature_extractor=processor.feature_extractor, | |
| max_new_tokens=128, | |
| chunk_length_s=30, | |
| batch_size=16, | |
| return_timestamps=True, | |
| torch_dtype=torch_dtype, | |
| device=device, | |
| ) | |
| self.model = model | |
| def __call__(self, data: Dict[str, bytes]) -> Dict[str, str]: | |
| """ | |
| Args: | |
| data (:obj:): | |
| includes the input data and the parameters for the inference. | |
| Return: | |
| A :obj:`list`:. The object returned should be a list of one list like [[{"label": 0.9939950108528137}]] containing : | |
| - "label": A string representing what the label/class is. There can be multiple labels. | |
| - "score": A score between 0 and 1 describing how confident the model is for this label/class. | |
| """ | |
| inputs = data.pop("inputs", data) | |
| parameters = data.pop("parameters", None) | |
| # pass inputs with all kwargs in data | |
| if parameters is not None: | |
| result = self.pipeline(inputs, return_timestamps=True, **parameters) | |
| else: | |
| result = self.pipeline(inputs, return_timestamps=True, generate_kwargs={"task": "translate"}) | |
| # postprocess the prediction | |
| return result |