Text Generation
MLX
Safetensors
modilify_mk2
diffusion
mixture-of-experts
custom-code
modilify-mk2
conversational
Instructions to use modilify/Modilify-Mk2-preview-mlx with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use modilify/Modilify-Mk2-preview-mlx with MLX:
# Make sure mlx-lm is installed # pip install --upgrade mlx-lm # Generate text with mlx-lm from mlx_lm import load, generate model, tokenizer = load("modilify/Modilify-Mk2-preview-mlx") prompt = "Write a story about Einstein" messages = [{"role": "user", "content": prompt}] prompt = tokenizer.apply_chat_template( messages, add_generation_prompt=True ) text = generate(model, tokenizer, prompt=prompt, verbose=True) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Pi
How to use modilify/Modilify-Mk2-preview-mlx with Pi:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx"
Configure the model in Pi
# Install Pi: npm install -g @earendil-works/pi-coding-agent # Add to ~/.pi/agent/models.json: { "providers": { "mlx-lm": { "baseUrl": "http://localhost:8080/v1", "api": "openai-completions", "apiKey": "none", "models": [ { "id": "modilify/Modilify-Mk2-preview-mlx" } ] } } }Run Pi
# Start Pi in your project directory: pi
- MLX LM
How to use modilify/Modilify-Mk2-preview-mlx with MLX LM:
Generate or start a chat session
# Install MLX LM uv tool install mlx-lm # Interactive chat REPL mlx_lm.chat --model "modilify/Modilify-Mk2-preview-mlx"
Run an OpenAI-compatible server
# Install MLX LM uv tool install mlx-lm # Start the server mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx" # Calling the OpenAI-compatible server with curl curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "modilify/Modilify-Mk2-preview-mlx", "messages": [ {"role": "user", "content": "Hello"} ] }' - Hermes Agent
How to use modilify/Modilify-Mk2-preview-mlx with Hermes Agent:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx"
Configure Hermes
# Install Hermes: curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash hermes setup # Point Hermes at the local server: hermes config set model.provider custom hermes config set model.base_url http://127.0.0.1:8080/v1 hermes config set model.default modilify/Modilify-Mk2-preview-mlx
Run Hermes
hermes
- Atomic Chat
- OpenClaw
How to use modilify/Modilify-Mk2-preview-mlx with OpenClaw:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "modilify/Modilify-Mk2-preview-mlx"
Configure OpenClaw
# Install OpenClaw: npm install -g openclaw@latest # Register the local server and set it as the default model: openclaw onboard --non-interactive --mode local \ --auth-choice custom-api-key \ --custom-base-url http://127.0.0.1:8080/v1 \ --custom-model-id "modilify/Modilify-Mk2-preview-mlx" \ --custom-provider-id mlx-lm \ --custom-compatibility openai \ --custom-text-input \ --accept-risk \ --skip-health
Run OpenClaw
openclaw agent --local --agent main --message "Hello from Hugging Face"
Download modilify_mk2/mlx_gdn2_memory.py from modilify/Modilify-Mk2-preview-mlx: direct link, hf CLI and curl.
- Browser
- Download file 6.35 kB
-
https://huggingface.co/modilify/Modilify-Mk2-preview-mlx/resolve/main/modilify_mk2/mlx_gdn2_memory.py
- Command line
-
hf download hf://modilify/Modilify-Mk2-preview-mlx/modilify_mk2/mlx_gdn2_memory.py
-
curl -L -o mlx_gdn2_memory.py https://huggingface.co/modilify/Modilify-Mk2-preview-mlx/resolve/main/modilify_mk2/mlx_gdn2_memory.py
6.35 kB
| """MLX counterpart of the FP32 GDN2 trajectory recurrence.""" | |
| from __future__ import annotations | |
| import mlx.core as mx | |
| from mlx import nn | |
| def _l2(value: mx.array) -> mx.array: | |
| return value * mx.rsqrt(mx.maximum(mx.sum(mx.square(value), axis=-1, keepdims=True), 1e-12)) | |
| class _GateProjection(nn.Module): | |
| def __init__(self, source: int, target: int, rank: int) -> None: | |
| super().__init__() | |
| self.down = nn.Linear(source, rank, bias=False) | |
| self.up = nn.Linear(rank, target, bias=False) | |
| def __call__(self, source: mx.array) -> mx.array: | |
| return self.up(self.down(source)) | |
| class GDN2Memory(nn.Module): | |
| def __init__(self, input_dim: int, heads: int, key_dim: int, value_dim: int, | |
| *, observation_dim: int | None = None) -> None: | |
| super().__init__() | |
| if min(input_dim, heads, key_dim, value_dim) <= 0: | |
| raise ValueError("GDN2 dimensions must be positive.") | |
| self.heads, self.key_dim, self.value_dim = heads, key_dim, value_dim | |
| observation_dim = input_dim if observation_dim is None else observation_dim | |
| if observation_dim <= 0: | |
| raise ValueError("GDN2 observation width must be positive.") | |
| self.q_proj = nn.Linear(input_dim, heads * key_dim, bias=False) | |
| self.k_proj = nn.Linear(observation_dim, heads * key_dim, bias=False) | |
| self.v_proj = nn.Linear(observation_dim, heads * value_dim, bias=False) | |
| self.f_proj = _GateProjection(observation_dim, heads * key_dim, min(observation_dim, key_dim)) | |
| self.b_proj = nn.Linear(observation_dim, heads * key_dim, bias=False) | |
| self.w_proj = nn.Linear(observation_dim, heads * value_dim, bias=False) | |
| self.g_proj = _GateProjection(input_dim, heads * value_dim, min(input_dim, value_dim)) | |
| self.o_proj = nn.Linear(heads * value_dim, input_dim, bias=False) | |
| self.a_log = mx.zeros((heads,), mx.float32) | |
| self.dt_bias = mx.full((heads, key_dim), -6.906255, mx.float32) | |
| def empty(self, *leading: int) -> mx.array: | |
| return mx.zeros((*leading, self.heads, self.key_dim, self.value_dim), mx.float32) | |
| def _projections(self, source: mx.array): | |
| normalized = source.astype(mx.float32) | |
| source = (normalized * mx.rsqrt(mx.mean(mx.square(normalized), axis=-1, keepdims=True) + 1e-6)).astype(source.dtype) | |
| shape = source.shape[:-1] | |
| key_shape = (*shape, self.heads, self.key_dim) | |
| value_shape = (*shape, self.heads, self.value_dim) | |
| k = _l2(nn.silu(self.k_proj(source).astype(mx.float32)).reshape(key_shape)) | |
| v = nn.silu(self.v_proj(source).astype(mx.float32)).reshape(value_shape) | |
| head_rate = mx.exp(self.a_log.astype(mx.float32)).reshape( | |
| *((1,) * (source.ndim - 1)), self.heads, 1) | |
| decay = mx.exp(-head_rate * nn.softplus( | |
| self.f_proj(source).astype(mx.float32).reshape(key_shape) + self.dt_bias.astype(mx.float32))) | |
| erase = mx.sigmoid(self.b_proj(source).astype(mx.float32).reshape(key_shape)) | |
| write = mx.sigmoid(self.w_proj(source).astype(mx.float32).reshape(value_shape)) | |
| return k, v, decay, erase, write | |
| def _query(self, source: mx.array) -> mx.array: | |
| return _l2(nn.silu(self.q_proj(source).astype(mx.float32)).reshape( | |
| *source.shape[:-1], self.heads, self.key_dim)) | |
| def _output(self, value: mx.array, source: mx.array) -> mx.array: | |
| gate = nn.silu(self.g_proj(source).astype(mx.float32).reshape(value.shape)) | |
| value = value * mx.rsqrt(mx.mean(mx.square(value), axis=-1, keepdims=True) + 1e-6) * gate | |
| return self.o_proj(value.reshape(*source.shape[:-1], -1).astype(source.dtype)) | |
| def read(self, state: mx.array, source: mx.array) -> mx.array: | |
| if state.shape != (*source.shape[:-1], self.heads, self.key_dim, self.value_dim): | |
| raise ValueError("GDN2 state and query leading dimensions differ.") | |
| value = mx.sum(self._query(source)[..., None] * state.astype(mx.float32), axis=-2) | |
| return self._output(value, source) | |
| def read_shared(self, state: mx.array, source: mx.array) -> mx.array: | |
| if source.ndim != 3 or state.shape != (source.shape[0], self.heads, self.key_dim, self.value_dim): | |
| raise ValueError("Shared GDN2 read requires [batch, canvas, width] queries.") | |
| query = self._query(source).transpose(0, 2, 1, 3) | |
| value = (query @ state.astype(mx.float32)).transpose(0, 2, 1, 3) | |
| return self._output(value, source) | |
| def _transition(state, k, v, decay, erase, write, valid): | |
| decayed = state.astype(mx.float32) * decay[..., None] | |
| old = mx.sum((erase * k)[..., None] * decayed, axis=-2) | |
| candidate = decayed + k[..., None] * (write * v - old)[..., None, :] | |
| return candidate if valid is None else mx.where( | |
| valid[..., None, None, None], candidate, state.astype(mx.float32)) | |
| def transition(self, state: mx.array, source: mx.array, | |
| valid: mx.array | None = None) -> mx.array: | |
| # Do not override nn.Module.update: it installs parameters for optimizers, | |
| # dtype conversion, and autodiff. State evolution is a separate operation. | |
| if state.shape != (*source.shape[:-1], self.heads, self.key_dim, self.value_dim): | |
| raise ValueError("GDN2 state and observation leading dimensions differ.") | |
| k, v, decay, erase, write = self._projections(source) | |
| if valid is not None: | |
| if valid.shape != source.shape[:-1]: | |
| raise ValueError("GDN2 valid mask must match observation rows.") | |
| return self._transition(state, k, v, decay, erase, write, valid) | |
| def write_sequence(self, state: mx.array, source: mx.array, | |
| valid: mx.array) -> mx.array: | |
| if source.ndim != 3 or valid.shape != source.shape[:2]: | |
| raise ValueError("GDN2 sequence and mask must share [batch, length].") | |
| if state.shape != (source.shape[0], self.heads, self.key_dim, self.value_dim): | |
| raise ValueError("GDN2 sequence state shape differs.") | |
| projected = self._projections(source) | |
| for index in range(source.shape[1]): | |
| state = self._transition(state, *(part[:, index] for part in projected), valid[:, index]) | |
| return state | |
| __all__ = ["GDN2Memory"] | |