Instructions to use immanuelpeter/MiniMax-M3-Vision with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use immanuelpeter/MiniMax-M3-Vision with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-feature-extraction", model="immanuelpeter/MiniMax-M3-Vision")# Load model directly from transformers import AutoTokenizer, AutoModel tokenizer = AutoTokenizer.from_pretrained("immanuelpeter/MiniMax-M3-Vision") model = AutoModel.from_pretrained("immanuelpeter/MiniMax-M3-Vision", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download examples/inference.py from immanuelpeter/MiniMax-M3-Vision: direct link, hf CLI and curl.
- Browser
- Download file 1.47 kB
-
https://huggingface.co/immanuelpeter/MiniMax-M3-Vision/resolve/main/examples/inference.py
- Command line
-
hf download hf://immanuelpeter/MiniMax-M3-Vision/examples/inference.py
-
curl -L -o inference.py https://huggingface.co/immanuelpeter/MiniMax-M3-Vision/resolve/main/examples/inference.py
1.47 kB
| import argparse | |
| from pathlib import Path | |
| import sys | |
| from PIL import Image | |
| import torch | |
| from transformers import MiniMaxM3VLImageProcessor, MiniMaxM3VLVisionModel | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[1])) | |
| from projector import load_projector | |
| MODEL_ID = "immanuelpeter/MiniMax-M3-Vision" | |
| def main() -> None: | |
| parser = argparse.ArgumentParser(description="Extract MiniMax-M3 visual features.") | |
| parser.add_argument("image", type=Path) | |
| parser.add_argument("--model", default=MODEL_ID) | |
| args = parser.parse_args() | |
| device = torch.device("cuda" if torch.cuda.is_available() else "cpu") | |
| dtype = torch.bfloat16 if device.type == "cuda" else torch.float32 | |
| tower = MiniMaxM3VLVisionModel.from_pretrained(args.model, dtype=dtype).to(device).eval() | |
| processor = MiniMaxM3VLImageProcessor.from_pretrained(args.model) | |
| projector = load_projector(args.model) | |
| projector.to(device=device, dtype=dtype).eval() | |
| inputs = processor(images=Image.open(args.image).convert("RGB"), return_tensors="pt") | |
| with torch.inference_mode(): | |
| merged = tower( | |
| pixel_values=inputs["pixel_values"].to(device=device, dtype=dtype), | |
| grid_thw=inputs["image_grid_thw"].to(device), | |
| ).last_hidden_state | |
| tokens = merged.reshape(-1, merged.shape[-1]) | |
| projected = projector(tokens) | |
| print("tower", merged.shape) | |
| print("projected", projected.shape) | |
| if __name__ == "__main__": | |
| main() | |