Text Generation
MLX
English
structured-generation
parallel-decoding
constrained-decoding
apple-silicon
classification
json
Instructions to use botp/Qwen-2.5-1B-RLCD with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use botp/Qwen-2.5-1B-RLCD with MLX:
# Make sure mlx-lm is installed # pip install --upgrade mlx-lm # if on a CUDA device, also pip install mlx[cuda] # Generate text with mlx-lm from mlx_lm import load, generate model, tokenizer = load("botp/Qwen-2.5-1B-RLCD") prompt = "Once upon a time in" text = generate(model, tokenizer, prompt=prompt, verbose=True) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- MLX LM
How to use botp/Qwen-2.5-1B-RLCD with MLX LM:
Generate or start a chat session
# Install MLX LM uv tool install mlx-lm # Generate some text mlx_lm.generate --model "botp/Qwen-2.5-1B-RLCD" --prompt "Once upon a time"
- Atomic Chat
File size: 3,127 Bytes
84f0c1f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 | """
Benchmark runner comparing Autoregressive Generation vs.
Parallel Constrained Decision Engine on Apple Silicon.
"""
import time
import json
import argparse
from typing import Dict, Any, List
from core.schema import StructuredSchema
from core.engine import run_naive_generation, run_parallel_generation, get_engine
def compare_single(context: str, schema_dict: Dict[str, Any]) -> Dict[str, Any]:
"""Runs both engines on the exact same problem prompt and returns side-by-side metrics."""
schema = StructuredSchema(schema_dict)
# 1. Run Autoregressive Baseline
naive_res = run_naive_generation(context, schema)
# 2. Run Parallel Constrained Engine
parallel_res = run_parallel_generation(context, schema)
speedup = naive_res["elapsed_ms"] / max(parallel_res["elapsed_ms"], 1.0)
steps_speedup = naive_res["sequential_forward_passes"] / max(parallel_res["sequential_forward_passes"], 1.0)
return {
"speedup_multiplier": round(speedup, 1),
"steps_reduction": round(steps_speedup, 1),
"naive": naive_res,
"parallel": parallel_res,
# Backward compatibility
"rlcd": parallel_res
}
def run_benchmark_suite(preset_paths: List[str], warmup: bool = True) -> List[Dict[str, Any]]:
print("=" * 70)
print("Parallel Constrained vs. Autoregressive Generation Benchmark")
print("=" * 70)
get_engine()
if warmup:
print("\n[+] Warming up GPU compute graphs...")
with open(preset_paths[0]) as f:
p = json.load(f)
compare_single(p["context"], p["schema"])
print("[+] Warmup complete.\n")
results = []
for path in preset_paths:
with open(path) as f:
preset = json.load(f)
print(f"--> Running preset: {preset['title']} ({len(preset['schema'])} fields)...")
comp = compare_single(preset["context"], preset["schema"])
comp["preset_id"] = preset["id"]
comp["preset_title"] = preset["title"]
results.append(comp)
n = comp["naive"]
r = comp["parallel"]
print(f" Autoregressive Baseline : {n['elapsed_ms']:>8.1f} ms | {n['total_tokens']:>3} tokens ({n['tokens_per_second']} tok/s) | Passes: {n['sequential_forward_passes']}")
print(f" Parallel Constrained : {r['elapsed_ms']:>8.1f} ms | 0 tokens (O(1)) | Passes: {r['sequential_forward_passes']}")
print(f" >> SPEEDUP: {comp['speedup_multiplier']}x faster (Step reduction: {comp['steps_reduction']}x)")
print(f" >> Schema match: Naive={n['schema_match']} | Parallel={r['schema_match']} (100% guaranteed)")
print("-" * 70)
return results
if __name__ == "__main__":
parser = argparse.ArgumentParser(description="Run Parallel vs Autoregressive LLM JSON benchmark")
parser.add_argument("--presets", nargs="+", default=[
"presets/fintech_fraud.json",
"presets/support_triage.json",
"presets/high_cardinality_255.json"
])
args = parser.parse_args()
run_benchmark_suite(args.presets)
|