Instructions to use calmresearch-ai/Model3 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use calmresearch-ai/Model3 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="calmresearch-ai/Model3")# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("calmresearch-ai/Model3", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use calmresearch-ai/Model3 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "calmresearch-ai/Model3" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "calmresearch-ai/Model3", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/calmresearch-ai/Model3
- SGLang
How to use calmresearch-ai/Model3 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "calmresearch-ai/Model3" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "calmresearch-ai/Model3", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "calmresearch-ai/Model3" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "calmresearch-ai/Model3", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use calmresearch-ai/Model3 with Docker Model Runner:
docker model run hf.co/calmresearch-ai/Model3
File size: 7,699 Bytes
33039e1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 | """Image preprocessing.
An image becomes a `n_vit_h x n_vit_w` patch grid for the ViT and a `n_llm_h x n_llm_w` token grid
after the 3x3 aligner downsample, which the LLM sees as
[IMAGE_START] + ([IMAGE] * n_llm_w + [IMAGE_NEW_LINE]) * n_llm_h + [IMAGE_END]
Every one of those positions carries `image_token_id` in `input_ids`; only the token type tells them
apart. The IMAGE slots are filled with aligner rows in reading order.
"""
import base64
import io
import math
from dataclasses import dataclass
from urllib.request import urlopen
import numpy as np
import torch
from PIL import Image, ImageOps
TEXT = -1
IMAGE_START, IMAGE, IMAGE_NEW_LINE, IMAGE_END = range(4)
@dataclass
class ImageInput:
start: int
patches: torch.Tensor
n_vit_h: int
n_vit_w: int
types: torch.Tensor
def num_image_tokens(n_llm_h: int, n_llm_w: int) -> int:
return n_llm_h * (n_llm_w + 1) + 2
def llm_grid(best_height: int, best_width: int, patch_size: int, downsample_ratio: int):
"""Token grid the aligner produces from a patch grid of this pixel size."""
return math.ceil((best_height // patch_size) / downsample_ratio), math.ceil(
(best_width // patch_size) / downsample_ratio
)
def solve_resize_ratio(height, width, patch_size, downsample_ratio, max_n_token):
"""Largest aspect-preserving pixel size whose token grid still fits in max_n_token."""
r = height / width
max_w_float = math.sqrt((max_n_token - 2) / r + 0.25) - 0.5
max_h_float = max_w_float * r
cell = patch_size * downsample_ratio
if max_w_float < 1.0: # very tall: collapse to a single column
return (max_n_token - 2) // 2 * cell, cell
if max_h_float < 1.0: # very wide: collapse to a single row
return cell, (max_n_token - 3) * cell
beta = min(math.floor(max_w_float) * cell / width, math.floor(max_h_float) * cell / height)
return math.floor(height * beta / patch_size) * patch_size, math.floor(width * beta / patch_size) * patch_size
def safe_resize(height, width, best_height, best_width, patch_size, downsample_ratio, max_n_token):
"""Shrink the pixel size until the image costs at most max_n_token LLM tokens."""
n_llm_h, n_llm_w = llm_grid(best_height, best_width, patch_size, downsample_ratio)
if num_image_tokens(n_llm_h, n_llm_w) > max_n_token:
best_height, best_width = solve_resize_ratio(height, width, patch_size, downsample_ratio, max_n_token)
n_llm_h, n_llm_w = llm_grid(best_height, best_width, patch_size, downsample_ratio)
assert num_image_tokens(n_llm_h, n_llm_w) <= max_n_token
return n_llm_h, n_llm_w, best_height, best_width
def load_image_bytes(record) -> bytes:
"""Load image bytes from raw/base64 data, an Anthropic source, URL, or path."""
data = record.get("data")
if isinstance(data, bytes):
return data
if isinstance(data, str):
return base64.b64decode(data)
source = record.get("source")
if isinstance(source, dict):
if source.get("data") is not None:
return base64.b64decode(source["data"])
if source.get("url"):
return load_image_bytes({"url": source["url"]})
url = record.get("url")
if isinstance(url, str) and url:
if url.startswith("data:"):
header, _, payload = url.partition(",")
if ";base64" not in header:
raise ValueError(f"Unsupported data URL encoding: {header}")
return base64.b64decode(payload)
if url.startswith(("http://", "https://")):
with urlopen(url, timeout=30) as response:
return response.read()
with open(url, "rb") as file:
return file.read()
raise ValueError(f"Cannot load image from record: {list(record.keys())}")
def plan_image_grid(width: int, height: int, args):
"""Resize plan for an image of the given original size; a pure function of its arguments."""
p = args.vision_patch_size
if args.vision_max_wh_ratio is not None and width > height * args.vision_max_wh_ratio:
width = height * args.vision_max_wh_ratio
if 0 < width * height < args.vision_min_pixels:
ratio = (args.vision_min_pixels / (width * height)) ** 0.5
width = int(width * ratio)
height = int(height * ratio)
best_width = math.ceil(width / p) * p
best_height = math.ceil(height / p) * p
return safe_resize(height, width, best_height, best_width, p, args.vision_downsample_ratio, args.vision_max_n_token)
def load_image(record, args):
"""Load and transform one image record into ViT patches."""
p = args.vision_patch_size
with Image.open(io.BytesIO(load_image_bytes(record))) as source:
image = source.convert("RGB")
n_llm_h, n_llm_w, best_height, best_width = plan_image_grid(image.width, image.height, args)
n_vit_h, n_vit_w = best_height // p, best_width // p
if args.vision_max_wh_ratio is not None and image.width >= args.vision_max_wh_ratio * image.height:
image = image.resize((best_width, best_height))
else:
image = ImageOps.pad(image, (best_width, best_height), color=(127, 127, 127))
x = torch.from_numpy(np.asarray(image, dtype=np.float32)).permute(2, 0, 1) / 255
x = ((x - 0.5) / 0.5).to(torch.bfloat16)
patches = x.reshape(3, n_vit_h, p, n_vit_w, p).permute(1, 3, 0, 2, 4).reshape(n_vit_h * n_vit_w, 3, p, p)
return patches, n_vit_h, n_vit_w, n_llm_h, n_llm_w
def image_token_types(n_llm_h: int, n_llm_w: int) -> torch.Tensor:
"""Default layout: the aligner grid in reading order, one IMAGE_NEW_LINE per row."""
types = [IMAGE_START]
types += ([IMAGE] * n_llm_w + [IMAGE_NEW_LINE]) * n_llm_h
types.append(IMAGE_END)
return torch.tensor(types, dtype=torch.int64)
def prepare_vl_inputs(prompt, images, tokenizer, args):
"""Tokenize `prompt`, expanding each image placeholder token into its image span.
Returns (tokens, token_types, image_inputs). Image-span positions carry `args.image_token_id` in
`tokens` and are distinguished only by `token_types` (TEXT elsewhere). `image_inputs` is None when
the prompt has no images."""
from encoding import IMAGE_PLACEHOLDER
# The placeholder is spelled differently across tokenizer revisions, so the id comes from the
# config; only cross-check it when this tokenizer does know the training-time spelling.
image_token_id = args.image_token_id
placeholder_id = tokenizer.convert_tokens_to_ids(IMAGE_PLACEHOLDER)
if placeholder_id is not None and placeholder_id != tokenizer.unk_token_id:
assert placeholder_id == image_token_id, (placeholder_id, image_token_id)
prompt_tokens = tokenizer.encode(prompt)
num_placeholders = sum(token == image_token_id for token in prompt_tokens)
if num_placeholders != len(images):
raise ValueError(f"Found {num_placeholders} image tokens but got {len(images)} images")
if num_placeholders and not args.vision_enabled:
raise ValueError("The model config has no vision tower (vision_n_layers == 0) but the prompt contains images")
tokens, token_types, image_inputs = [], [], []
image_iter = iter(images)
for tok in prompt_tokens:
if tok != image_token_id:
tokens.append(tok)
token_types.append(TEXT)
continue
patches, n_vit_h, n_vit_w, n_llm_h, n_llm_w = load_image(next(image_iter), args)
types = image_token_types(n_llm_h, n_llm_w)
image_inputs.append(ImageInput(len(tokens), patches, n_vit_h, n_vit_w, types))
tokens += [image_token_id] * types.numel()
token_types += types.tolist()
return tokens, token_types, image_inputs or None
|