Text Generation
MLX
Safetensors
modilify_mk2
diffusion
mixture-of-experts
custom-code
modilify-mk2
conversational
Instructions to use modilify/Modilify-Mk2-preview-mlx with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use modilify/Modilify-Mk2-preview-mlx with MLX:
# Make sure mlx-lm is installed # pip install --upgrade mlx-lm # Generate text with mlx-lm from mlx_lm import load, generate model, tokenizer = load("modilify/Modilify-Mk2-preview-mlx") prompt = "Write a story about Einstein" messages = [{"role": "user", "content": prompt}] prompt = tokenizer.apply_chat_template( messages, add_generation_prompt=True ) text = generate(model, tokenizer, prompt=prompt, verbose=True) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Pi
How to use modilify/Modilify-Mk2-preview-mlx with Pi:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx"
Configure the model in Pi
# Install Pi: npm install -g @earendil-works/pi-coding-agent # Add to ~/.pi/agent/models.json: { "providers": { "mlx-lm": { "baseUrl": "http://localhost:8080/v1", "api": "openai-completions", "apiKey": "none", "models": [ { "id": "modilify/Modilify-Mk2-preview-mlx" } ] } } }Run Pi
# Start Pi in your project directory: pi
- MLX LM
How to use modilify/Modilify-Mk2-preview-mlx with MLX LM:
Generate or start a chat session
# Install MLX LM uv tool install mlx-lm # Interactive chat REPL mlx_lm.chat --model "modilify/Modilify-Mk2-preview-mlx"
Run an OpenAI-compatible server
# Install MLX LM uv tool install mlx-lm # Start the server mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx" # Calling the OpenAI-compatible server with curl curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "modilify/Modilify-Mk2-preview-mlx", "messages": [ {"role": "user", "content": "Hello"} ] }' - Hermes Agent
How to use modilify/Modilify-Mk2-preview-mlx with Hermes Agent:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx"
Configure Hermes
# Install Hermes: curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash hermes setup # Point Hermes at the local server: hermes config set model.provider custom hermes config set model.base_url http://127.0.0.1:8080/v1 hermes config set model.default modilify/Modilify-Mk2-preview-mlx
Run Hermes
hermes
- Atomic Chat
- OpenClaw
How to use modilify/Modilify-Mk2-preview-mlx with OpenClaw:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx"
Configure OpenClaw
# Install OpenClaw: npm install -g openclaw@latest # Register the local server and set it as the default model: openclaw onboard --non-interactive --mode local \ --auth-choice custom-api-key \ --custom-base-url http://127.0.0.1:8080/v1 \ --custom-model-id "modilify/Modilify-Mk2-preview-mlx" \ --custom-provider-id mlx-lm \ --custom-compatibility openai \ --custom-text-input \ --accept-risk \ --skip-health
Run OpenClaw
openclaw agent --local --agent main --message "Hello from Hugging Face"
Download modilify_mk2/mlx_model.py from modilify/Modilify-Mk2-preview-mlx: direct link, hf CLI and curl.
- Browser
- Download file 5.53 kB
-
https://huggingface.co/modilify/Modilify-Mk2-preview-mlx/resolve/main/modilify_mk2/mlx_model.py
- Command line
-
hf download hf://modilify/Modilify-Mk2-preview-mlx/modilify_mk2/mlx_model.py
-
curl -L -o mlx_model.py https://huggingface.co/modilify/Modilify-Mk2-preview-mlx/resolve/main/modilify_mk2/mlx_model.py
5.53 kB
| """Schema25 native MLX text trunk with GDN2 trajectory memory and dual readers.""" | |
| from __future__ import annotations | |
| from dataclasses import dataclass | |
| from typing import Any | |
| import mlx.core as mx | |
| from mlx import nn | |
| from .mlx_latent import create_mlx_latent | |
| from .mlx_state import MLXLatentState | |
| class MLXCanvasOutput: | |
| heavy_hidden: mx.array | |
| working_state: mx.array | |
| next_latent_state: MLXLatentState | |
| token_embeddings: mx.array | |
| class MLXModilifyMk2(nn.Module): | |
| """Shared DiffusionGemma trunk plus native commit-only trajectory module.""" | |
| def __init__(self, backbone: Any, config: Any) -> None: | |
| super().__init__() | |
| self.model = backbone.model | |
| self.latent_deliberation = create_mlx_latent(config) | |
| self.config = config | |
| def _merge_context(self, token_embeddings: mx.array, | |
| context: mx.array) -> mx.array: | |
| """Frozen self-conditioning bridge with the schema25 residual cap.""" | |
| mapper = self.model.decoder.self_conditioning | |
| normed = mapper.pre_norm(context.astype(token_embeddings.dtype)) | |
| mapped = mapper.down_proj( | |
| nn.gelu_approx(mapper.gate_proj(normed)) * mapper.up_proj(normed) | |
| ) | |
| mapped_fp32 = mapped.astype(mx.float32) | |
| energy = mx.mean(mx.square(mapped_fp32), axis=-1, keepdims=True) | |
| token_rms = mx.sqrt(mx.mean(mx.square(token_embeddings.astype(mx.float32)), | |
| axis=-1, keepdims=True)) | |
| cap = 0.5 * token_rms | |
| scale = cap / mx.sqrt(energy + mx.square(cap) + 1.0e-12) | |
| combined = token_embeddings + (mapped_fp32 * scale).astype(mapped.dtype) | |
| return mapper.post_norm(combined) | |
| class LoRAConfig: | |
| r: int | |
| alpha: int | |
| dropout: float | |
| target_modules: tuple[str, ...] | |
| expert_r: int = 8 | |
| expert_alpha: int = 8 | |
| def _make_inference_router(base: Any): | |
| """Preserve the checkpoint's exact routing and expert weighting.""" | |
| import mlx.core as mx | |
| from mlx_vlm.models.diffusion_gemma.language import Router | |
| class InferenceRouter(Router): | |
| def __call__(self, x): | |
| x = mx.fast.rms_norm(x, None, self.eps) | |
| x = x * self.scale * self._root_size | |
| scores = self.proj(x) | |
| k = self.config.top_k_experts | |
| indices = mx.stop_gradient(mx.argpartition(scores, kth=-k, axis=-1)[..., -k:]) | |
| weights = mx.take_along_axis(scores, indices, axis=-1) | |
| weights = mx.softmax(weights, axis=-1, precise=True) | |
| return indices, weights * self.per_expert_scale[indices] | |
| router = InferenceRouter(base.config) | |
| router.proj = base.proj | |
| router.scale = base.scale | |
| router.per_expert_scale = base.per_expert_scale | |
| return router | |
| def mlx_text_config(config: Any): | |
| """Translate the validated schema25 text config to mlx-vlm's MLX model.""" | |
| from mlx_vlm.models.diffusion_gemma.config import ModelConfig | |
| payload = config.to_dict() | |
| payload["model_type"] = "diffusion_gemma" | |
| payload["text_config"]["model_type"] = "diffusion_gemma_text" | |
| payload["vision_config"] = None | |
| return ModelConfig.from_dict(payload) | |
| def create_mlx_text_backbone(config: Any): | |
| """Create a text-only MLX trunk with the official DiffusionGemma topology.""" | |
| from mlx_vlm.models.diffusion_gemma.diffusion_gemma import Model | |
| return Model(mlx_text_config(config)) | |
| def inject_mlx_lora(model: Any, config: Any) -> int: | |
| """Attach mlx-lm adapters to the shared text trunk and MoE experts.""" | |
| from mlx_lm.tuner.lora import LoRALinear, LoRASwitchLinear | |
| if min(config.r, config.alpha, config.expert_r, config.expert_alpha) <= 0: | |
| raise ValueError("MLX LoRA ranks and alphas must be positive.") | |
| if not 0 <= config.dropout < 1: | |
| raise ValueError("MLX LoRA dropout must be in [0, 1).") | |
| model.freeze() | |
| targets = set(config.target_modules) | |
| injected = 0 | |
| for layer in model.model.decoder.layers: | |
| layer.router = _make_inference_router(layer.router) | |
| layer.router.freeze() | |
| for parent in (layer.self_attn, layer.mlp): | |
| for name, module in list(parent.named_modules()): | |
| if "." in name or name not in targets: | |
| continue | |
| adapter = LoRALinear.from_base( | |
| module, | |
| r=config.r, | |
| dropout=config.dropout, | |
| scale=config.alpha / config.r, | |
| ) | |
| adapter.lora_a = adapter.lora_a.astype(module.weight.dtype) | |
| adapter.lora_b = adapter.lora_b.astype(module.weight.dtype) | |
| setattr( | |
| parent, | |
| name, | |
| adapter, | |
| ) | |
| injected += 1 | |
| for name in ("gate_up_proj", "down_proj"): | |
| module = getattr(layer.experts, name) | |
| adapter = LoRASwitchLinear.from_base( | |
| module, | |
| r=config.expert_r, | |
| dropout=config.dropout, | |
| scale=config.expert_alpha / config.expert_r, | |
| ) | |
| adapter.lora_a = adapter.lora_a.astype(module.weight.dtype) | |
| adapter.lora_b = adapter.lora_b.astype(module.weight.dtype) | |
| setattr( | |
| layer.experts, | |
| name, | |
| adapter, | |
| ) | |
| injected += 1 | |
| if not injected: | |
| raise RuntimeError("No MLX LoRA target modules were found.") | |
| return injected | |