Text Generation
Transformers
Safetensors
modilify_mk2
diffusion
mixture-of-experts
trust-remote-code
conversational
custom_code
Instructions to use modilify/Modilify-Mk2-preview with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use modilify/Modilify-Mk2-preview with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="modilify/Modilify-Mk2-preview", trust_remote_code=True) messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("modilify/Modilify-Mk2-preview", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use modilify/Modilify-Mk2-preview with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "modilify/Modilify-Mk2-preview" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "modilify/Modilify-Mk2-preview", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/modilify/Modilify-Mk2-preview
- SGLang
How to use modilify/Modilify-Mk2-preview with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "modilify/Modilify-Mk2-preview" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "modilify/Modilify-Mk2-preview", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "modilify/Modilify-Mk2-preview" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "modilify/Modilify-Mk2-preview", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use modilify/Modilify-Mk2-preview with Docker Model Runner:
docker model run hf.co/modilify/Modilify-Mk2-preview
Download commit_policy.py from modilify/Modilify-Mk2-preview: direct link, hf CLI and curl.
- Browser
- Download file 11.1 kB
-
https://huggingface.co/modilify/Modilify-Mk2-preview/resolve/main/commit_policy.py
- Command line
-
hf download hf://modilify/Modilify-Mk2-preview/commit_policy.py
-
curl -L -o commit_policy.py https://huggingface.co/modilify/Modilify-Mk2-preview/resolve/main/commit_policy.py
11.1 kB
| """Confidence-and-entropy prefix commitment.""" | |
| from __future__ import annotations | |
| from collections.abc import Sequence | |
| from dataclasses import dataclass | |
| import math | |
| import torch | |
| from .latent_deliberation import ( | |
| advance_trajectory_clocks, | |
| should_force_trajectory_jump, | |
| ) | |
| JUMP_FAILURE_BUDGET = 2.0 | |
| FUSED_EPS = 1e-6 | |
| def commit_target_confidence_bias( | |
| target_confidence: float | None, | |
| failure_budget: float = 0.2, | |
| budget_safety_ratio: float = 0.85, | |
| ) -> float: | |
| if target_confidence is None or target_confidence <= 0.0 or target_confidence >= 1.0: | |
| return 0.0 | |
| target_failure = min(budget_safety_ratio * failure_budget, 1.0 - target_confidence) | |
| target_failure = max(target_failure, 1.0e-4) | |
| target_conf = 1.0 - target_failure | |
| logit_c = math.log(target_conf / (1.0 - target_conf)) | |
| logit_p = math.log(target_confidence / (1.0 - target_confidence)) | |
| return max(logit_c - logit_p, 0.0) | |
| def fused_commit_confidence( | |
| proposal_confidence: torch.Tensor, | |
| token_entropy: torch.Tensor, | |
| *, | |
| eps: float = FUSED_EPS, | |
| entropy_weight: float = 1.0, | |
| confidence_power: float = 1.0, | |
| top_k: int | None = None, | |
| min_p: float | None = None, | |
| target_confidence: float | None = None, | |
| failure_budget: float = 0.2, | |
| ) -> torch.Tensor: | |
| """Fuse proposal confidence with token entropy into effective commit confidence. | |
| Effective confidence is defined using an excess-entropy sigmoid: | |
| p = clamp(proposal_confidence, eps, 1 - eps) | |
| h2 = -p * log(p) - (1 - p) * log(1 - p) # binary entropy of p | |
| excess = max(token_entropy - h2, 0) | |
| base_fused = sigmoid(logit(p) + bias - entropy_weight * excess) | |
| fused = base_fused ** confidence_power | |
| """ | |
| p = proposal_confidence.float().nan_to_num(0.5).clamp(min=eps, max=1.0 - eps) | |
| H = token_entropy.float().nan_to_num(0.0).clamp(min=0.0) | |
| # Binary entropy of p, using log1p for the (1-p) term. | |
| h2 = -p * torch.log(p) - (1.0 - p) * torch.log1p(-p) | |
| excess = (H - h2).clamp(min=0.0) | |
| k_eff = None | |
| if top_k is not None and top_k > 0: | |
| k_eff = torch.full_like(p, float(top_k)) | |
| if min_p is not None and min_p > 0: | |
| thresh = torch.clamp_min(float(min_p) * p, 1.0e-6) | |
| k_min_p = torch.clamp_min((1.0 - p) / thresh, 1.0) | |
| k_eff = k_min_p if k_eff is None else torch.minimum(k_eff, k_min_p) | |
| if k_eff is not None: | |
| max_excess = (1.0 - p) * torch.log(k_eff) | |
| excess = torch.minimum(excess, max_excess) | |
| # logit(p) = log(p / (1-p)) = log(p) - log1p(-p) | |
| logit_p = torch.log(p) - torch.log1p(-p) | |
| bias = commit_target_confidence_bias(target_confidence, failure_budget=failure_budget) | |
| base_fused = torch.sigmoid(logit_p + bias - float(entropy_weight) * excess) | |
| fused = base_fused.pow(float(confidence_power)) | |
| return fused.clamp(min=eps, max=1.0 - eps) | |
| def fused_commit_failure_rate( | |
| proposal_confidence: torch.Tensor, | |
| token_entropy: torch.Tensor, | |
| **kwargs: object, | |
| ) -> torch.Tensor: | |
| """Return ``1 - effective_confidence`` from the shared fusion helper.""" | |
| return 1.0 - fused_commit_confidence( | |
| proposal_confidence, token_entropy, **kwargs | |
| ) | |
| class CommitPolicyDecision: | |
| """Proposal-to-commit transition.""" | |
| normal_lengths: torch.LongTensor | |
| commit_lengths: torch.LongTensor | |
| commit_token_ids: torch.LongTensor | |
| jump_rows: torch.BoolTensor | |
| ponder_steps: torch.IntTensor | |
| stagnation_steps: torch.IntTensor | |
| def prefix_failure_commit_lengths( | |
| failure_rate: torch.Tensor, | |
| *, | |
| failure_budget: float, | |
| valid_mask: torch.BoolTensor | None = None, | |
| ) -> torch.LongTensor: | |
| """Return the longest valid prefix satisfying ``cumsum(failure_rate) < budget``.""" | |
| if failure_rate.ndim != 2: | |
| raise ValueError("Failure rate must have shape [batch, canvas].") | |
| if failure_budget <= 0: | |
| raise ValueError("Commit failure budget must be positive.") | |
| if valid_mask is None: | |
| valid_mask = torch.ones_like(failure_rate, dtype=torch.bool) | |
| if valid_mask.shape != failure_rate.shape: | |
| raise ValueError("Commit validity mask must match failure rate.") | |
| risk = failure_rate.float().clamp(0.0, 1.0) * valid_mask.to(torch.float32) | |
| cumulative_risk = risk.cumsum(dim=-1) | |
| contiguous_valid = valid_mask.long().cumprod(dim=-1).bool() | |
| allowed = cumulative_risk.lt(float(failure_budget)) & contiguous_valid | |
| return allowed.long().cumprod(dim=-1).sum(dim=-1) | |
| def first_committed_token_lengths( | |
| proposal: torch.LongTensor, | |
| commit_lengths: torch.LongTensor, | |
| token_id: int | Sequence[int], | |
| *, | |
| positions: torch.LongTensor | None = None, | |
| ) -> torch.LongTensor: | |
| """Clip each committed prefix immediately after its first matching stop token.""" | |
| if proposal.ndim != 2 or commit_lengths.shape != proposal.shape[:1]: | |
| raise ValueError("Proposal and commit lengths must share a batch dimension.") | |
| if positions is None: | |
| positions = torch.arange(proposal.shape[1], device=proposal.device).unsqueeze(0) | |
| elif positions.shape != (1, proposal.shape[1]): | |
| raise ValueError("Commit positions must have shape [1, canvas].") | |
| committed = positions.lt(commit_lengths[:, None]) | |
| stop_token_ids = ( | |
| (int(token_id),) | |
| if isinstance(token_id, int) | |
| else tuple(dict.fromkeys(int(value) for value in token_id)) | |
| ) | |
| if not stop_token_ids: | |
| raise ValueError("At least one stop token ID is required.") | |
| matches = proposal.eq(stop_token_ids[0]) | |
| for value in stop_token_ids[1:]: | |
| matches |= proposal.eq(value) | |
| matches &= committed | |
| sentinel = torch.full_like(positions, proposal.shape[1]) | |
| first = torch.where(matches, positions, sentinel).min(dim=-1).values | |
| clipped = torch.where(first.lt(proposal.shape[1]), first + 1, commit_lengths) | |
| return torch.minimum(clipped, commit_lengths) | |
| def bounded_prefix_failure_commit_lengths( | |
| committed_token_ids: torch.LongTensor, | |
| failure_rate: torch.Tensor, | |
| *, | |
| failure_budget: float, | |
| remaining_lengths: torch.LongTensor, | |
| stop_token_id: int | Sequence[int], | |
| valid_mask: torch.BoolTensor | None = None, | |
| positions: torch.LongTensor | None = None, | |
| ) -> torch.LongTensor: | |
| """Apply length and stop-token bounds to the shared failure-rate policy.""" | |
| if committed_token_ids.shape != failure_rate.shape: | |
| raise ValueError("Committed token IDs and failure rate must share [batch, canvas].") | |
| if remaining_lengths.shape != committed_token_ids.shape[:1]: | |
| raise ValueError("Remaining lengths must have shape [batch].") | |
| commit_lengths = prefix_failure_commit_lengths( | |
| failure_rate, | |
| failure_budget=failure_budget, | |
| valid_mask=valid_mask, | |
| ) | |
| commit_lengths = torch.minimum(commit_lengths, remaining_lengths.clamp_min(0)) | |
| return first_committed_token_lengths( | |
| committed_token_ids, | |
| commit_lengths, | |
| stop_token_id, | |
| positions=positions, | |
| ) | |
| def select_commit_lengths( | |
| sampled_token_ids: torch.LongTensor, | |
| normal_failure_rate: torch.Tensor, | |
| previous_failure_rate: torch.Tensor, | |
| greedy_token_ids: torch.LongTensor, | |
| jump_failure_rate: torch.Tensor, | |
| *, | |
| ponder_steps: torch.Tensor, | |
| stagnation_steps: torch.Tensor, | |
| active_rows: torch.BoolTensor, | |
| remaining_lengths: torch.LongTensor, | |
| failure_budget: float, | |
| stop_token_id: int | Sequence[int], | |
| stagnation_threshold: int, | |
| min_progress: float, | |
| max_ponder_steps: int | None = None, | |
| valid_mask: torch.BoolTensor | None = None, | |
| ) -> CommitPolicyDecision: | |
| """Use normal sampled commits and a fixed-budget greedy JUMP. | |
| Progress is measured from the signed change in fused failure rate over the | |
| frontier region (the union of the previous and current commit prefixes plus | |
| one blocking position), not from raw confidence/entropy deltas. | |
| """ | |
| if not ( | |
| sampled_token_ids.shape | |
| == normal_failure_rate.shape | |
| == previous_failure_rate.shape | |
| == greedy_token_ids.shape | |
| == jump_failure_rate.shape | |
| ): | |
| raise ValueError("Sampled and greedy statistics must share [batch, canvas].") | |
| canvas_length = normal_failure_rate.shape[1] | |
| positions = torch.arange(canvas_length, device=normal_failure_rate.device)[None, :] | |
| normal = bounded_prefix_failure_commit_lengths( | |
| sampled_token_ids, | |
| normal_failure_rate, | |
| failure_budget=failure_budget, | |
| remaining_lengths=remaining_lengths, | |
| stop_token_id=stop_token_id, | |
| valid_mask=valid_mask, | |
| positions=positions, | |
| ) | |
| previous_prefix_length = prefix_failure_commit_lengths( | |
| previous_failure_rate, | |
| failure_budget=failure_budget, | |
| valid_mask=valid_mask, | |
| ) | |
| frontier_length = torch.maximum(previous_prefix_length, normal) + 1 | |
| valid_lengths = ( | |
| valid_mask.long().sum(dim=-1) | |
| if valid_mask is not None | |
| else torch.full_like(frontier_length, canvas_length) | |
| ) | |
| frontier_length = torch.minimum(frontier_length, valid_lengths) | |
| progress_mask = positions < frontier_length[:, None] | |
| if valid_mask is not None: | |
| progress_mask &= valid_mask | |
| progress_mask &= active_rows[:, None] | |
| signed_improvement = ( | |
| previous_failure_rate.float() - normal_failure_rate.float() | |
| ) | |
| weights = progress_mask.float() | |
| progress = ( | |
| signed_improvement * weights | |
| ).sum(dim=-1) / weights.sum(dim=-1).clamp_min(1.0) | |
| next_ponder, next_stagnation = advance_trajectory_clocks( | |
| ponder_steps, | |
| stagnation_steps, | |
| commit_lengths=normal, | |
| active_rows=active_rows, | |
| ) | |
| jump_rows = normal.eq(0) & active_rows & should_force_trajectory_jump( | |
| next_stagnation, | |
| progress_scores=progress, | |
| min_progress=min_progress, | |
| stagnation_threshold=stagnation_threshold, | |
| ponder_steps=next_ponder, | |
| max_ponder_steps=max_ponder_steps, | |
| ) | |
| jump_commit = bounded_prefix_failure_commit_lengths( | |
| greedy_token_ids, | |
| jump_failure_rate, | |
| failure_budget=JUMP_FAILURE_BUDGET, | |
| remaining_lengths=remaining_lengths, | |
| stop_token_id=stop_token_id, | |
| valid_mask=valid_mask, | |
| positions=positions, | |
| ) | |
| committed = torch.where(jump_rows, jump_commit, normal) | |
| commit_token_ids = torch.where( | |
| jump_rows[:, None], | |
| greedy_token_ids, | |
| sampled_token_ids, | |
| ) | |
| committed = torch.where(active_rows, committed, 0) | |
| committed_rows = committed.gt(0) | |
| jump_rows &= committed_rows | |
| next_ponder = torch.where( | |
| committed_rows, | |
| 0, | |
| next_ponder, | |
| ).to(torch.int32) | |
| next_stagnation = torch.where( | |
| committed_rows, | |
| 0, | |
| next_stagnation, | |
| ).to(torch.int32) | |
| return CommitPolicyDecision( | |
| normal_lengths=normal, | |
| commit_lengths=committed, | |
| commit_token_ids=commit_token_ids, | |
| jump_rows=jump_rows, | |
| ponder_steps=next_ponder, | |
| stagnation_steps=next_stagnation, | |
| ) | |