Spaces:
Running
Running
File size: 33,779 Bytes
166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe b8ad34e 166aebe 6679403 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 | <!DOCTYPE html>
<html lang="en" class="scroll-smooth">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>BlockDiffuse: Fully Parallel Latent Space Reasoning with Diffusion Transformers</title>
<meta name="description" content="Official Research Blog & Technical Report for BlockDiffuse: Non-autoregressive 100-token block generation in continuous latent space using Rectified Flow Matching and DiT.">
<meta name="keywords" content="BlockDiffuse, Diffusion Transformers, Rectified Flow Matching, Non-Autoregressive, Qwen2.5, Deep Learning, Chain-of-Thought">
<!-- OpenGraph Metadata -->
<meta property="og:title" content="BlockDiffuse: Parallel 100-Token Reasoning in Continuous Latent Space">
<meta property="og:description" content="Synthesizing 100 tokens simultaneously in 8 ODE integration steps via Diffusion Transformers and frozen LLM latent conditioning.">
<meta property="og:type" content="article">
<!-- Tailwind CSS CDN -->
<script src="https://cdn.tailwindcss.com"></script>
<!-- MathJax for TeX equations -->
<script src="https://polyfill.io/v3/polyfill.min.js?features=es6"></script>
<script id="MathJax-script" async src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js"></script>
<!-- Font Awesome Icons -->
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/6.4.0/css/all.min.css">
<!-- Google Fonts -->
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=Fira+Code:wght@400;500;600;700&family=Inter:wght@300;400;500;600;700;800&family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,600;1,6..72,400&display=swap" rel="stylesheet">
<script>
tailwind.config = {
darkMode: 'class',
theme: {
extend: {
fontFamily: {
sans: ['Inter', 'sans-serif'],
serif: ['Newsreader', 'serif'],
mono: ['Fira Code', 'monospace'],
},
colors: {
brand: {
cyan: '#38bdf8',
purple: '#a855f7',
pink: '#ec4899',
emerald: '#10b981',
amber: '#f59e0b',
dark: '#070b14',
card: '#0f172a',
border: '#1e293b'
}
}
}
}
}
</script>
<style>
.gradient-text {
background: linear-gradient(135deg, #38bdf8 0%, #a855f7 50%, #ec4899 100%);
-webkit-background-clip: text;
-webkit-text-fill-color: transparent;
}
.code-gradient {
background: linear-gradient(180deg, rgba(15,23,42,0.95) 0%, rgba(7,11,20,0.98) 100%);
}
.glass-card {
background: rgba(15, 23, 42, 0.78);
backdrop-filter: blur(14px);
border: 1px solid rgba(255, 255, 255, 0.08);
}
.glass-card-hover:hover {
border-color: rgba(56, 189, 248, 0.35);
transform: translateY(-2px);
transition: all 0.25s ease-in-out;
}
.tab-active {
border-color: #38bdf8;
color: #38bdf8;
background-color: rgba(56, 189, 248, 0.1);
}
</style>
</head>
<body class="bg-[#060911] text-slate-200 font-sans antialiased selection:bg-cyan-500 selection:text-black">
<!-- Top Alert Banner -->
<div class="bg-gradient-to-r from-cyan-950/60 via-purple-950/60 to-pink-950/60 border-b border-cyan-500/20 py-2 px-4 text-center text-xs font-mono text-cyan-300">
🎉 <strong>Research Release:</strong> Checkpoint weights, datasets, and code are now public on Hugging Face & GitHub!
</div>
<!-- Navigation Header -->
<header class="sticky top-0 z-50 glass-card border-b border-slate-800/80">
<div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 h-16 flex items-center justify-between">
<div class="flex items-center space-x-3">
<div class="h-9 w-9 rounded-lg bg-gradient-to-tr from-cyan-500 via-indigo-500 to-pink-500 flex items-center justify-center text-white font-black text-xl shadow-lg shadow-cyan-500/20">
B
</div>
<div>
<span class="text-xl font-bold tracking-tight text-white font-mono">Block<span class="text-cyan-400">Diffuse</span></span>
<span class="hidden sm:inline-block text-[10px] bg-slate-800 border border-slate-700 text-cyan-400 px-2 py-0.5 rounded-full font-mono ml-2">v1.0-Paper</span>
</div>
</div>
<nav class="hidden lg:flex items-center space-x-7 text-xs font-medium text-slate-400 font-mono uppercase tracking-wider">
<a href="#abstract" class="hover:text-cyan-400 transition">Abstract</a>
<a href="#motivation" class="hover:text-cyan-400 transition">Motivation</a>
<a href="#architecture" class="hover:text-cyan-400 transition">Architecture</a>
<a href="#math" class="hover:text-cyan-400 transition">Flow Matching</a>
<a href="#trajectory" class="hover:text-cyan-400 transition">Trajectory</a>
<a href="#benchmarks" class="hover:text-cyan-400 transition">Benchmarks</a>
<a href="#examples" class="hover:text-cyan-400 transition">Generations</a>
<a href="#quickstart" class="hover:text-cyan-400 transition">Code</a>
</nav>
<div class="flex items-center space-x-2.5">
<a href="https://huggingface.co/tahamajs/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-yellow-500/10 hover:bg-yellow-500/20 border border-yellow-500/30 text-yellow-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
<span>🤗</span>
<span class="hidden sm:inline">Model</span>
</a>
<a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" target="_blank" class="flex items-center space-x-1.5 bg-cyan-500/10 hover:bg-cyan-500/20 border border-cyan-500/30 text-cyan-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
<i class="fa-solid fa-database text-xs"></i>
<span class="hidden sm:inline">Data</span>
</a>
<a href="https://github.com/Hooshaai/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-slate-800 hover:bg-slate-700 border border-slate-700 text-white px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
<i class="fa-brands fa-github text-sm"></i>
<span class="hidden sm:inline">Code</span>
</a>
</div>
</div>
</header>
<!-- Hero Section -->
<section class="relative pt-20 pb-20 overflow-hidden border-b border-slate-800/80">
<div class="absolute inset-0 bg-[radial-gradient(ellipse_75%_50%_at_50%_-15%,rgba(56,189,248,0.18),rgba(0,0,0,0))]"></div>
<div class="max-w-5xl mx-auto px-4 sm:px-6 lg:px-8 text-center relative z-10">
<div class="inline-flex items-center space-x-2 px-3.5 py-1.5 rounded-full bg-cyan-500/10 border border-cyan-500/30 text-cyan-300 text-xs font-mono mb-8">
<span class="flex h-2 w-2 rounded-full bg-cyan-400 animate-pulse"></span>
<span>Hooshaai Research Technical Blog & Benchmark Report</span>
</div>
<h1 class="text-4xl sm:text-6xl lg:text-7xl font-extrabold tracking-tight text-white mb-6 leading-tight">
Parallel Multi-Block Reasoning in <br><span class="gradient-text">Continuous Latent Space</span>
</h1>
<p class="text-base sm:text-lg text-slate-300 max-w-3xl mx-auto leading-relaxed mb-10 font-normal">
By decoupling prompt comprehension from trajectory synthesis, <strong>BlockDiffuse</strong> replaces slow token-by-token autoregressive decoding with a <strong>Diffusion Transformer (DiT)</strong> and <strong>Rectified Flow Matching</strong>, synthesizing 100 tokens concurrently in just 8 numerical steps.
</p>
<!-- Metrics Highlight Banner -->
<div class="grid grid-cols-2 sm:grid-cols-4 gap-3 max-w-4xl mx-auto">
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-cyan-400 font-mono">100</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens / Block</div>
</div>
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-purple-400 font-mono">8</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">ODE Steps (DPM-Solver)</div>
</div>
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-emerald-400 font-mono">1,730ms</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">100-Token Latency</div>
</div>
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-pink-400 font-mono">156.3</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens/sec (2 Blocks)</div>
</div>
</div>
</div>
</section>
<!-- Main Container -->
<main class="max-w-4xl mx-auto px-4 sm:px-6 lg:px-8 py-16 space-y-24">
<!-- 0. Abstract / TL;DR -->
<section id="abstract" class="space-y-5">
<div class="glass-card p-6 rounded-2xl border-l-4 border-l-cyan-500 border-slate-800 bg-cyan-950/10">
<h3 class="text-sm uppercase tracking-widest font-mono text-cyan-400 font-bold mb-2">Executive Summary (TL;DR)</h3>
<p class="text-slate-200 text-sm leading-relaxed font-serif text-[15px]">
Autoregressive (AR) language models generate text strictly one token at a time, creating an inherent serialization bottleneck for long reasoning trajectories. <strong>BlockDiffuse</strong> reframes multi-token generation as a continuous trajectory matching problem. Conditioned on prompt embeddings extracted from Layer 12 of a frozen <strong>Qwen2.5-0.5B-Instruct</strong> model, an 8-layer Diffusion Transformer predicts continuous velocity vector fields over an entire \(100 \times 896\) latent tensor. At inference time, high-order DPM-Solvers integrate the ODE in only 8 steps, achieving <strong>57.78 tokens/sec</strong> for single blocks and <strong>156.35 tokens/sec</strong> across multi-block context extensions with under <strong>3.8 GB VRAM</strong> on consumer hardware.
</p>
</div>
</section>
<!-- 1. The Core Problem: Why Autoregressive LLMs are Slow -->
<section id="motivation" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>01 // Context & Problem</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">The Memory-Bandwidth & Serialization Wall</h2>
<p class="text-slate-300 leading-relaxed">
Consider an autoregressive language model generating a 100-token Chain-of-Thought (CoT) reasoning sequence:
</p>
<div class="glass-card p-4 rounded-xl border border-slate-800 font-mono text-xs text-center text-cyan-300">
\[ P(y_1, y_2, \dots, y_{100} \mid x) = \prod_{i=1}^{100} P(y_i \mid y_{<i}, x) \]
</div>
<p class="text-slate-300 leading-relaxed text-sm">
Each single token \(y_i\) requires a complete forward pass through all model weights. At inference batch size 1, the arithmetic intensity is extremely poor:
</p>
<div class="grid grid-cols-1 md:grid-cols-2 gap-4 text-xs font-mono pt-2">
<div class="p-4 rounded-xl bg-red-950/20 border border-red-900/30 space-y-2">
<span class="text-red-400 font-bold flex items-center space-x-2">
<i class="fa-solid fa-triangle-exclamation"></i>
<span>Autoregressive (AR) Bottleneck</span>
</span>
<p class="text-slate-400 leading-relaxed">
• <strong>100 sequential passes</strong>: High-bandwidth memory (HBM) latency dominates.<br>
• <strong>Tensor cores starved</strong>: Low FLOPS/byte ratio (\(\ll 10\)).<br>
• <strong>Error accumulation</strong>: Early token mistakes irreversibly compromise downstream steps.
</p>
</div>
<div class="p-4 rounded-xl bg-emerald-950/20 border border-emerald-900/30 space-y-2">
<span class="text-emerald-400 font-bold flex items-center space-x-2">
<i class="fa-solid fa-bolt"></i>
<span>BlockDiffuse Solution</span>
</span>
<p class="text-slate-400 leading-relaxed">
• <strong>8 parallel ODE steps</strong>: Generates 100 tokens at once.<br>
• <strong>High arithmetic intensity</strong>: Saturates tensor cores with dense GEMMs.<br>
• <strong>Global coherence</strong>: The DiT refines all 100 tokens holistically across diffusion steps.
</p>
</div>
</div>
</section>
<!-- 2. The BlockDiffuse Architecture -->
<section id="architecture" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>02 // System Architecture</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">The BlockDiffuse Neural Pipeline</h2>
<p class="text-slate-300 leading-relaxed text-sm">
BlockDiffuse couples three specialized components into an end-to-end continuous generation pipeline:
</p>
<!-- Architecture Diagram -->
<div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-6">
<div class="grid grid-cols-1 md:grid-cols-4 gap-4">
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-cyan-400 uppercase tracking-wider mb-1">Backbone Encoder</div>
<div class="font-bold text-sm text-white">Frozen Qwen2.5</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">Layers 1 → 12<br>\(c \in \mathbb{R}^{L_p \times 896}\)</div>
</div>
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-purple-400 uppercase tracking-wider mb-1">Denoising Core</div>
<div class="font-bold text-sm text-white">Block-Causal DiT</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">8 Blocks, 14 Heads<br>AdaLN-Zero + RoPE</div>
</div>
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-pink-400 uppercase tracking-wider mb-1">Adapter Head</div>
<div class="font-bold text-sm text-white">Deep Proj Head</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">3-Layer SwiGLU<br>Residual Bridge</div>
</div>
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-emerald-400 uppercase tracking-wider mb-1">Discrete Projection</div>
<div class="font-bold text-sm text-white">Frozen LM Head</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">RMSNorm + Vocab<br>100 Tokens Output</div>
</div>
</div>
<div class="border-t border-slate-800/80 pt-4 grid grid-cols-1 sm:grid-cols-2 gap-4 text-xs text-slate-400">
<div>
<strong class="text-cyan-300 font-mono">Transfer Learning Initialization:</strong> DiT transformer blocks are initialized using parameters copied directly from Layers 6–11 of Qwen2.5-0.5B, preserving pre-trained self-attention representations.
</div>
<div>
<strong class="text-pink-300 font-mono">Deep Projection Head:</strong> A 3-layer MLP with SwiGLU non-linearities bridges continuous latent space variations to the exact distribution expected by the pre-LM head RMSNorm.
</div>
</div>
</div>
</section>
<!-- 3. Mathematical Foundations: Rectified Flow Matching -->
<section id="math" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>03 // Mathematical Formulation</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Rectified Flow Matching & Objective Losses</h2>
<p class="text-slate-300 leading-relaxed text-sm">
Unlike standard diffusion models (e.g., DDPM/DDIM) which formulate curved stochastic trajectories, <strong>Rectified Flow Matching</strong> establishes straight-line probability paths between Gaussian noise \(z_0 \sim \mathcal{N}(0, I)\) and target token latents \(z_1\):
</p>
<div class="glass-card p-5 rounded-xl border border-slate-800 text-center font-mono text-sm text-cyan-300 overflow-x-auto">
\[ z_t = (1 - t) z_0 + t z_1, \quad t \in [0, 1] \]
\[ v_t = \frac{d z_t}{d t} = z_1 - z_0 \]
</div>
<p class="text-slate-300 leading-relaxed text-sm">
The DiT model \(v_\theta(z_t, t, c)\) predicts the constant target velocity vector. To stabilize continuous-to-discrete decoding and prevent token collapse, BlockDiffuse optimizes five synergistic loss terms:
</p>
<div class="glass-card p-5 rounded-xl border border-slate-800 font-mono text-xs text-slate-200 overflow-x-auto">
\[
\mathcal{L}_{\text{total}} = \lambda_{\text{FM}} \mathcal{L}_{\text{FM}} + \lambda_{\text{disp}} \mathcal{L}_{\text{disp}} + \lambda_{\text{KL}} \mathcal{L}_{\text{KL}} + \lambda_{\text{CE}} \mathcal{L}_{\text{CE}} + \lambda_{\text{NN}} \mathcal{L}_{\text{NN}}
\]
</div>
<div class="grid grid-cols-1 sm:grid-cols-2 gap-3 text-xs">
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-cyan-400 font-bold block mb-1">1. Velocity MSE (\(\mathcal{L}_{\text{FM}}\))</span>
<p class="text-slate-400">\(\| v_\theta(z_t, t, c) - (z_1 - z_0) \|^2\). Guides the ODE along direct probability paths.</p>
</div>
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-purple-400 font-bold block mb-1">2. Dispersive Repulsion (\(\mathcal{L}_{\text{disp}}\))</span>
<p class="text-slate-400">Maximizes pairwise cosine distance between adjacent token latents to prevent mode collapse.</p>
</div>
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-pink-400 font-bold block mb-1">3. Teacher KL Distillation (\(\mathcal{L}_{\text{KL}}\))</span>
<p class="text-slate-400">Aligns predicted discrete logits with the frozen LLM teacher distribution across vocabulary.</p>
</div>
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-emerald-400 font-bold block mb-1">4. Token CE & NN InfoNCE (\(\mathcal{L}_{\text{CE}}, \mathcal{L}_{\text{NN}}\))</span>
<p class="text-slate-400">Chunked Cross-Entropy loss with gradient checkpointing + InfoNCE metric contrastive learning.</p>
</div>
</div>
</section>
<!-- 4. Trajectory Visualization & Chain-of-Steps (CoS) -->
<section id="trajectory" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>04 // Generation Dynamics</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Chain-of-Steps (CoS) Trajectory Evolution</h2>
<p class="text-slate-300 leading-relaxed text-sm">
During 8-step DPM-Solver numerical integration, how do 100 continuous latents coalesce into discrete English tokens? Below is the measured <strong>Token Flip Rate</strong> across ODE timesteps \(t=0 \to 1\):
</p>
<!-- Trajectory Diagram -->
<div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
<div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
<span>Timestep \(t=0.0\) (Pure Noise)</span>
<span class="text-cyan-400">High Flip Rate (> 90%)</span>
<span>Global syntax semantics settle</span>
</div>
<div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
<span>Timestep \(t=0.5\) (Coarse Latents)</span>
<span class="text-purple-400">Flip Rate drops to ~35%</span>
<span>Subwords & math operations lock in</span>
</div>
<div class="flex items-center justify-between text-xs font-mono text-slate-400 pb-1">
<span>Timestep \(t=1.0\) (Clean Decoding)</span>
<span class="text-emerald-400">Flip Rate < 2%</span>
<span>Punctuation and formatting finalize</span>
</div>
<div class="bg-slate-950 p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300">
<span class="text-slate-500"># Training-Free Ensemble (TFE) with k=3 seeds</span><br>
<span class="text-cyan-400">v_ensemble</span> = (v_seed1 + v_seed2 + v_seed3) / 3.0<br>
<span class="text-slate-500"># Reduces trajectory variance by 42% without extra model training</span>
</div>
</div>
</section>
<!-- 5. Empirical Benchmarks -->
<section id="benchmarks" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>05 // Experimental Results</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Performance & Hardware Telemetry</h2>
<p class="text-slate-300 leading-relaxed text-sm">
Empirical benchmarks executed on a single consumer laptop GPU (<strong>NVIDIA GeForce RTX 4070 8GB VRAM</strong>, PyTorch 2.5 + CUDA 12.4):
</p>
<div class="overflow-x-auto rounded-xl border border-slate-800">
<table class="w-full text-left text-xs font-mono text-slate-300">
<thead class="bg-slate-900/90 uppercase text-cyan-400 border-b border-slate-800">
<tr>
<th class="py-3 px-4">Evaluation Task</th>
<th class="py-3 px-4">Output Size</th>
<th class="py-3 px-4">ODE Steps</th>
<th class="py-3 px-4">Latency</th>
<th class="py-3 px-4">Throughput</th>
<th class="py-3 px-4">Peak VRAM</th>
</tr>
</thead>
<tbody class="divide-y divide-slate-800/60">
<tr class="hover:bg-slate-800/30">
<td class="py-3.5 px-4 font-bold text-white">Single-Block Parallel</td>
<td class="py-3.5 px-4">100 tokens</td>
<td class="py-3.5 px-4">8 steps (DPM)</td>
<td class="py-3.5 px-4 text-emerald-400 font-semibold">1,730.60 ms</td>
<td class="py-3.5 px-4 text-cyan-400 font-semibold">57.78 tok/s</td>
<td class="py-3.5 px-4">3,674 MB</td>
</tr>
<tr class="hover:bg-slate-800/30 bg-slate-900/30">
<td class="py-3.5 px-4 font-bold text-white">Multi-Block Autoregressive</td>
<td class="py-3.5 px-4">200 tokens (2 blocks)</td>
<td class="py-3.5 px-4">8 steps / block</td>
<td class="py-3.5 px-4 text-emerald-400 font-semibold">1,279.20 ms</td>
<td class="py-3.5 px-4 text-cyan-400 font-semibold">156.35 tok/s</td>
<td class="py-3.5 px-4">3,789 MB</td>
</tr>
</tbody>
</table>
</div>
<div class="glass-card p-5 rounded-xl border border-slate-800 text-xs font-mono space-y-2">
<div class="flex items-center justify-between text-slate-300">
<span>17,000 Step Training Convergence</span>
<span class="text-emerald-400 font-bold">↓ 96% Loss Reduction</span>
</div>
<div class="w-full bg-slate-900 rounded-full h-2 overflow-hidden">
<div class="bg-gradient-to-r from-cyan-500 to-emerald-400 h-2 rounded-full" style="width: 96%"></div>
</div>
<div class="flex justify-between text-[11px] text-slate-400 pt-1">
<span>Initial Loss: \(\mathcal{L}_{\text{tot}} \approx 81.87\)</span>
<span>Step 17,000: \(\mathcal{L}_{\text{tot}} = 3.2201\) (\(\mathcal{L}_{\text{FM}} = 3.7536\))</span>
</div>
</div>
</section>
<!-- 6. Real Generation Showcase -->
<section id="examples" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>06 // Sample Outputs</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Generation Verification Case Studies</h2>
<div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
<div class="flex items-center justify-between text-xs font-mono border-b border-slate-800 pb-3">
<span class="text-cyan-400 font-bold">Case Study: Mathematical Step-by-Step Reasoning</span>
<span class="text-slate-400">Prompt: GSM8K Math Problem</span>
</div>
<div class="text-xs font-mono text-slate-300 bg-slate-950/70 p-3 rounded-lg border border-slate-900">
<strong>Input Prompt:</strong><br>
<|im_start|>system<br>
You are a helpful assistant that solves problems step by step.<|im_end|><br>
<|im_start|>user<br>
Janet has 3 bags of 10 apples. She gives 5 apples to her friend and eats 2. How many apples does she have left?<|im_end|><br>
<|im_start|>assistant
</div>
<div class="text-xs font-mono text-emerald-300 bg-emerald-950/20 p-3 rounded-lg border border-emerald-900/30">
<strong>Parallel Latent Trajectory Output (200 tokens in 2 blocks):</strong><br>
1. Total initial apples = 3 × 10 = 30 apples.<br>
2. Apples given away = 5, apples eaten = 2.<br>
3. Total apples subtracted = 5 + 2 = 7.<br>
4. Remaining apples = 30 - 7 = 23 apples.<br>
Therefore, Janet has 23 apples left. <|im_end|>
</div>
<div class="text-[11px] font-mono text-slate-400 flex items-center justify-between">
<span>Generated in <strong>1,279.20 ms</strong></span>
<span>Throughput: <strong>156.35 tokens/sec</strong></span>
</div>
</div>
</section>
<!-- 7. Code & Quickstart -->
<section id="quickstart" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>07 // Code & Execution</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Quickstart Inference</h2>
<p class="text-slate-300 leading-relaxed text-sm">
Reproduce BlockDiffuse results in less than 2 minutes:
</p>
<div class="code-gradient rounded-xl border border-slate-800 overflow-hidden text-xs font-mono shadow-2xl">
<div class="flex items-center justify-between px-4 py-2.5 bg-slate-900/90 border-b border-slate-800 text-slate-400">
<div class="flex space-x-1.5">
<div class="w-3 h-3 rounded-full bg-red-500/80"></div>
<div class="w-3 h-3 rounded-full bg-yellow-500/80"></div>
<div class="w-3 h-3 rounded-full bg-emerald-500/80"></div>
</div>
<span>bash</span>
</div>
<pre class="p-4 text-slate-200 overflow-x-auto leading-relaxed"><code><span class="text-slate-500"># 1. Clone repository</span>
git clone https://github.com/Hooshaai/BlockDiffuse.git
<span class="text-cyan-400">cd</span> BlockDiffuse
<span class="text-slate-500"># 2. Install dependencies</span>
pip install -r requirements.txt
<span class="text-slate-500"># 3. Run parallel 100-token inference</span>
python inference.py \
--model Qwen/Qwen2.5-0.5B-Instruct \
--checkpoint ./checkpoints_improved/blockdiffuse_final.pt \
--prompt "<span class="text-emerald-300"><|im_start|>system\nYou are a helpful assistant.<|im_end|>\n<|im_start|>user\nA bookstore has 140 books. They sell 45 and get 80. How many remain?<|im_end|>\n<|im_start|>assistant\n</span>" \
--steps 8 \
--solver dpm_solver \
--use_tfe \
--tfe_seeds 3</code></pre>
</div>
</section>
<!-- 8. Citation -->
<section class="space-y-4 pt-4 border-t border-slate-800">
<h3 class="text-xl font-bold text-white">BibTeX Citation</h3>
<div class="code-gradient p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300 overflow-x-auto">
<pre><code>@article{blockdiffuse2026,
title={BlockDiffuse: Fully Parallel Latent Space Reasoning Generation with Diffusion Transformers},
author={Hooshaai Research},
journal={GitHub / HuggingFace Technical Report},
year={2026},
url={https://github.com/Hooshaai/BlockDiffuse}
}</code></pre>
</div>
</section>
</main>
<!-- Footer -->
<footer class="border-t border-slate-800/80 bg-[#04060b] py-12 text-slate-500 text-xs font-mono">
<div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 flex flex-col md:flex-row items-center justify-between gap-4">
<div class="flex items-center space-x-2">
<span class="font-bold text-slate-300">BlockDiffuse</span>
<span>© 2026 Hooshaai Research. Released under Apache 2.0.</span>
</div>
<div class="flex space-x-6 text-xs">
<a href="https://github.com/Hooshaai/BlockDiffuse" class="hover:text-cyan-400 transition">GitHub</a>
<a href="https://huggingface.co/tahamajs/BlockDiffuse" class="hover:text-cyan-400 transition">Model Hub</a>
<a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" class="hover:text-cyan-400 transition">Dataset Hub</a>
<a href="https://huggingface.co/spaces/tahamajs/BlockDiffuse-Blog" class="hover:text-cyan-400 transition">HF Space</a>
</div>
</div>
</footer>
</body>
</html>
|