Instructions to use MTEnt/dot with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use MTEnt/dot with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="MTEnt/dot") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("MTEnt/dot") model = AutoModelForCausalLM.from_pretrained("MTEnt/dot", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use MTEnt/dot with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "MTEnt/dot" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "MTEnt/dot", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/MTEnt/dot
- SGLang
How to use MTEnt/dot with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "MTEnt/dot" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "MTEnt/dot", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "MTEnt/dot" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "MTEnt/dot", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use MTEnt/dot with Docker Model Runner:
docker model run hf.co/MTEnt/dot
Download dot_rd/export.py from MTEnt/dot: direct link, hf CLI and curl.
- Browser
- Download file 5.58 kB
-
https://huggingface.co/MTEnt/dot/resolve/main/dot_rd/export.py
- Command line
-
hf download hf://MTEnt/dot/dot_rd/export.py
-
curl -L -o export.py https://huggingface.co/MTEnt/dot/resolve/main/dot_rd/export.py
5.58 kB
| from __future__ import annotations | |
| import hashlib | |
| import json | |
| import os | |
| from pathlib import Path | |
| from typing import Any | |
| from safetensors import safe_open | |
| from safetensors.torch import load_file, save_file | |
| import torch | |
| from .model import DotRecurrentDepthModel | |
| def _sha256(path: Path) -> str: | |
| digest = hashlib.sha256() | |
| with path.open("rb") as handle: | |
| for chunk in iter(lambda: handle.read(8 * 1024 * 1024), b""): | |
| digest.update(chunk) | |
| return digest.hexdigest() | |
| def export_reasoning_core(checkpoint_path: str | Path, output_dir: str | Path) -> dict[str, Any]: | |
| checkpoint = Path(checkpoint_path) | |
| payload = torch.load(checkpoint, map_location="cpu", weights_only=False) | |
| if payload.get("format_version") != 1: | |
| raise ValueError(f"unsupported checkpoint format: {payload.get('format_version')}") | |
| state = payload.get("reasoning_core") | |
| if not isinstance(state, dict) or not state: | |
| raise ValueError("checkpoint has no reasoning_core state") | |
| tensors = {name: tensor.detach().contiguous().cpu() for name, tensor in state.items()} | |
| root = Path(output_dir) | |
| root.mkdir(parents=True, exist_ok=True) | |
| final_weights = root / "reasoning_core.safetensors" | |
| temporary_weights = root / "reasoning_core.safetensors.tmp" | |
| save_file(tensors, temporary_weights) | |
| os.replace(temporary_weights, final_weights) | |
| manifest = { | |
| "format_version": 1, | |
| "artifact": "Dot recurrent-depth reasoning core", | |
| "source_checkpoint": str(checkpoint), | |
| "source_step": int(payload["step"]), | |
| "weights": final_weights.name, | |
| "weights_sha256": _sha256(final_weights), | |
| "runtime": { | |
| "base_model": ".", | |
| "loader": "dot_rd.export.load_exported_core", | |
| "use_cache": False, | |
| }, | |
| "architecture": payload["architecture"], | |
| "training_metrics": payload.get("metrics") or {}, | |
| } | |
| final_manifest = root / "dot_recurrent_manifest.json" | |
| temporary_manifest = root / "dot_recurrent_manifest.json.tmp" | |
| temporary_manifest.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8") | |
| os.replace(temporary_manifest, final_manifest) | |
| return manifest | |
| def load_exported_core(path: str | Path, model: DotRecurrentDepthModel) -> dict[str, Any]: | |
| root = Path(path) | |
| manifest = json.loads((root / "dot_recurrent_manifest.json").read_text(encoding="utf-8")) | |
| weights_path = root / manifest["weights"] | |
| actual_hash = _sha256(weights_path) | |
| if actual_hash != manifest["weights_sha256"]: | |
| raise ValueError( | |
| f"reasoning core hash mismatch: expected {manifest['weights_sha256']}, got {actual_hash}" | |
| ) | |
| expected = model.architecture_manifest() | |
| actual = manifest.get("architecture") or {} | |
| for field in ("architecture", "base_parameter_count", "total_parameter_count", "new_parameter_count"): | |
| if actual.get(field) != expected.get(field): | |
| raise ValueError(f"exported core architecture mismatch for {field}") | |
| model.reasoning_core.load_state_dict(load_file(weights_path, device="cpu"), strict=True) | |
| return manifest | |
| def load_inference_checkpoint( | |
| path: str | Path, model: DotRecurrentDepthModel | |
| ) -> dict[str, Any]: | |
| """Load a verified Dot core plus its explicitly saved backbone repair delta.""" | |
| root = Path(path) | |
| manifest = json.loads((root / "manifest.json").read_text(encoding="utf-8")) | |
| weights_path = root / manifest["weights"] | |
| actual_hash = _sha256(weights_path) | |
| if actual_hash != manifest["weights_sha256"]: | |
| raise ValueError( | |
| f"inference checkpoint hash mismatch: expected {manifest['weights_sha256']}, " | |
| f"got {actual_hash}" | |
| ) | |
| expected = model.architecture_manifest() | |
| actual = manifest.get("architecture") or {} | |
| for field in ("architecture", "base_parameter_count", "total_parameter_count", "new_parameter_count"): | |
| if actual.get(field) != expected.get(field): | |
| raise ValueError(f"inference checkpoint architecture mismatch for {field}") | |
| core_state = model.reasoning_core.state_dict() | |
| backbone_state = model.backbone.state_dict() | |
| seen_core: set[str] = set() | |
| seen_backbone: set[str] = set() | |
| with safe_open(weights_path, framework="pt", device="cpu") as handle: | |
| for key in handle.keys(): | |
| if key.startswith("reasoning_core."): | |
| name = key.removeprefix("reasoning_core.") | |
| target = core_state.get(name) | |
| seen_core.add(name) | |
| elif key.startswith("backbone_delta."): | |
| name = key.removeprefix("backbone_delta.") | |
| target = backbone_state.get(name) | |
| seen_backbone.add(name) | |
| else: | |
| raise ValueError(f"unsupported inference checkpoint tensor: {key}") | |
| if target is None: | |
| raise ValueError(f"checkpoint tensor does not exist in Dot: {key}") | |
| tensor = handle.get_tensor(key) | |
| if tensor.shape != target.shape: | |
| raise ValueError(f"checkpoint shape mismatch for {key}: {tensor.shape} != {target.shape}") | |
| with torch.no_grad(): | |
| target.copy_(tensor.to(device=target.device, dtype=target.dtype)) | |
| missing_core = sorted(set(core_state).difference(seen_core)) | |
| if missing_core: | |
| raise ValueError(f"inference checkpoint is missing core tensors: {missing_core[:5]}") | |
| if not seen_backbone: | |
| raise ValueError("inference checkpoint has no backbone repair delta") | |
| return manifest | |