File size: 33,779 Bytes
166aebe
 
 
 
 
b8ad34e
 
 
 
 
 
 
 
166aebe
 
 
 
 
 
 
 
 
 
 
b8ad34e
166aebe
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b8ad34e
 
 
166aebe
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b8ad34e
166aebe
 
b8ad34e
 
166aebe
 
b8ad34e
 
 
 
166aebe
b8ad34e
 
166aebe
b8ad34e
166aebe
 
 
b8ad34e
 
 
 
 
 
166aebe
 
b8ad34e
166aebe
 
 
 
 
b8ad34e
 
 
 
166aebe
 
b8ad34e
 
166aebe
 
b8ad34e
 
166aebe
b8ad34e
 
166aebe
 
b8ad34e
 
166aebe
b8ad34e
166aebe
b8ad34e
 
 
 
 
166aebe
b8ad34e
166aebe
 
 
 
 
 
b8ad34e
 
166aebe
b8ad34e
166aebe
b8ad34e
166aebe
b8ad34e
166aebe
b8ad34e
166aebe
b8ad34e
 
 
166aebe
 
b8ad34e
 
 
166aebe
 
 
b8ad34e
 
 
166aebe
b8ad34e
 
166aebe
 
b8ad34e
166aebe
b8ad34e
166aebe
 
 
 
 
b8ad34e
 
 
 
 
 
 
 
 
 
 
 
166aebe
b8ad34e
166aebe
 
b8ad34e
166aebe
b8ad34e
 
 
166aebe
b8ad34e
 
166aebe
b8ad34e
 
166aebe
b8ad34e
 
 
166aebe
b8ad34e
 
 
 
 
 
 
166aebe
b8ad34e
 
166aebe
b8ad34e
 
 
 
 
 
 
166aebe
 
 
 
b8ad34e
166aebe
 
b8ad34e
166aebe
b8ad34e
 
 
166aebe
 
b8ad34e
 
 
 
 
 
 
166aebe
b8ad34e
 
 
 
166aebe
b8ad34e
 
166aebe
b8ad34e
166aebe
b8ad34e
 
 
 
166aebe
 
 
b8ad34e
 
 
 
 
 
 
166aebe
 
 
 
b8ad34e
 
166aebe
b8ad34e
166aebe
b8ad34e
 
 
166aebe
b8ad34e
 
 
 
166aebe
 
 
b8ad34e
166aebe
b8ad34e
 
166aebe
 
 
 
 
 
b8ad34e
 
 
166aebe
b8ad34e
 
 
166aebe
b8ad34e
 
 
166aebe
b8ad34e
 
 
166aebe
 
 
 
b8ad34e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
166aebe
 
b8ad34e
166aebe
b8ad34e
 
 
166aebe
 
 
b8ad34e
 
166aebe
b8ad34e
 
 
166aebe
 
b8ad34e
166aebe
 
b8ad34e
166aebe
 
 
b8ad34e
166aebe
 
b8ad34e
166aebe
 
 
 
 
 
 
b8ad34e
166aebe
 
 
 
 
 
 
b8ad34e
166aebe
 
b8ad34e
 
166aebe
 
b8ad34e
 
166aebe
 
 
 
b8ad34e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
166aebe
 
b8ad34e
166aebe
b8ad34e
 
 
166aebe
 
 
 
 
 
 
 
 
 
 
b8ad34e
166aebe
 
 
 
 
 
b8ad34e
166aebe
 
 
b8ad34e
166aebe
 
 
 
 
 
 
b8ad34e
166aebe
b8ad34e
166aebe
 
 
 
 
 
 
 
 
 
 
 
 
 
b8ad34e
166aebe
 
b8ad34e
 
166aebe
b8ad34e
 
 
 
 
166aebe
 
 
 
 
6679403
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
<!DOCTYPE html>
<html lang="en" class="scroll-smooth">
<head>
    <meta charset="UTF-8">
    <meta name="viewport" content="width=device-width, initial-scale=1.0">
    <title>BlockDiffuse: Fully Parallel Latent Space Reasoning with Diffusion Transformers</title>
    <meta name="description" content="Official Research Blog & Technical Report for BlockDiffuse: Non-autoregressive 100-token block generation in continuous latent space using Rectified Flow Matching and DiT.">
    <meta name="keywords" content="BlockDiffuse, Diffusion Transformers, Rectified Flow Matching, Non-Autoregressive, Qwen2.5, Deep Learning, Chain-of-Thought">
    
    <!-- OpenGraph Metadata -->
    <meta property="og:title" content="BlockDiffuse: Parallel 100-Token Reasoning in Continuous Latent Space">
    <meta property="og:description" content="Synthesizing 100 tokens simultaneously in 8 ODE integration steps via Diffusion Transformers and frozen LLM latent conditioning.">
    <meta property="og:type" content="article">
    
    <!-- Tailwind CSS CDN -->
    <script src="https://cdn.tailwindcss.com"></script>
    <!-- MathJax for TeX equations -->
    <script src="https://polyfill.io/v3/polyfill.min.js?features=es6"></script>
    <script id="MathJax-script" async src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js"></script>
    <!-- Font Awesome Icons -->
    <link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/6.4.0/css/all.min.css">
    <!-- Google Fonts -->
    <link rel="preconnect" href="https://fonts.googleapis.com">
    <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
    <link href="https://fonts.googleapis.com/css2?family=Fira+Code:wght@400;500;600;700&family=Inter:wght@300;400;500;600;700;800&family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,600;1,6..72,400&display=swap" rel="stylesheet">

    <script>
        tailwind.config = {
            darkMode: 'class',
            theme: {
                extend: {
                    fontFamily: {
                        sans: ['Inter', 'sans-serif'],
                        serif: ['Newsreader', 'serif'],
                        mono: ['Fira Code', 'monospace'],
                    },
                    colors: {
                        brand: {
                            cyan: '#38bdf8',
                            purple: '#a855f7',
                            pink: '#ec4899',
                            emerald: '#10b981',
                            amber: '#f59e0b',
                            dark: '#070b14',
                            card: '#0f172a',
                            border: '#1e293b'
                        }
                    }
                }
            }
        }
    </script>
    <style>
        .gradient-text {
            background: linear-gradient(135deg, #38bdf8 0%, #a855f7 50%, #ec4899 100%);
            -webkit-background-clip: text;
            -webkit-text-fill-color: transparent;
        }
        .code-gradient {
            background: linear-gradient(180deg, rgba(15,23,42,0.95) 0%, rgba(7,11,20,0.98) 100%);
        }
        .glass-card {
            background: rgba(15, 23, 42, 0.78);
            backdrop-filter: blur(14px);
            border: 1px solid rgba(255, 255, 255, 0.08);
        }
        .glass-card-hover:hover {
            border-color: rgba(56, 189, 248, 0.35);
            transform: translateY(-2px);
            transition: all 0.25s ease-in-out;
        }
        .tab-active {
            border-color: #38bdf8;
            color: #38bdf8;
            background-color: rgba(56, 189, 248, 0.1);
        }
    </style>
</head>
<body class="bg-[#060911] text-slate-200 font-sans antialiased selection:bg-cyan-500 selection:text-black">

    <!-- Top Alert Banner -->
    <div class="bg-gradient-to-r from-cyan-950/60 via-purple-950/60 to-pink-950/60 border-b border-cyan-500/20 py-2 px-4 text-center text-xs font-mono text-cyan-300">
        🎉 <strong>Research Release:</strong> Checkpoint weights, datasets, and code are now public on Hugging Face & GitHub!
    </div>

    <!-- Navigation Header -->
    <header class="sticky top-0 z-50 glass-card border-b border-slate-800/80">
        <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 h-16 flex items-center justify-between">
            <div class="flex items-center space-x-3">
                <div class="h-9 w-9 rounded-lg bg-gradient-to-tr from-cyan-500 via-indigo-500 to-pink-500 flex items-center justify-center text-white font-black text-xl shadow-lg shadow-cyan-500/20">
                    B
                </div>
                <div>
                    <span class="text-xl font-bold tracking-tight text-white font-mono">Block<span class="text-cyan-400">Diffuse</span></span>
                    <span class="hidden sm:inline-block text-[10px] bg-slate-800 border border-slate-700 text-cyan-400 px-2 py-0.5 rounded-full font-mono ml-2">v1.0-Paper</span>
                </div>
            </div>

            <nav class="hidden lg:flex items-center space-x-7 text-xs font-medium text-slate-400 font-mono uppercase tracking-wider">
                <a href="#abstract" class="hover:text-cyan-400 transition">Abstract</a>
                <a href="#motivation" class="hover:text-cyan-400 transition">Motivation</a>
                <a href="#architecture" class="hover:text-cyan-400 transition">Architecture</a>
                <a href="#math" class="hover:text-cyan-400 transition">Flow Matching</a>
                <a href="#trajectory" class="hover:text-cyan-400 transition">Trajectory</a>
                <a href="#benchmarks" class="hover:text-cyan-400 transition">Benchmarks</a>
                <a href="#examples" class="hover:text-cyan-400 transition">Generations</a>
                <a href="#quickstart" class="hover:text-cyan-400 transition">Code</a>
            </nav>

            <div class="flex items-center space-x-2.5">
                <a href="https://huggingface.co/tahamajs/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-yellow-500/10 hover:bg-yellow-500/20 border border-yellow-500/30 text-yellow-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
                    <span>🤗</span>
                    <span class="hidden sm:inline">Model</span>
                </a>
                <a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" target="_blank" class="flex items-center space-x-1.5 bg-cyan-500/10 hover:bg-cyan-500/20 border border-cyan-500/30 text-cyan-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
                    <i class="fa-solid fa-database text-xs"></i>
                    <span class="hidden sm:inline">Data</span>
                </a>
                <a href="https://github.com/Hooshaai/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-slate-800 hover:bg-slate-700 border border-slate-700 text-white px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
                    <i class="fa-brands fa-github text-sm"></i>
                    <span class="hidden sm:inline">Code</span>
                </a>
            </div>
        </div>
    </header>

    <!-- Hero Section -->
    <section class="relative pt-20 pb-20 overflow-hidden border-b border-slate-800/80">
        <div class="absolute inset-0 bg-[radial-gradient(ellipse_75%_50%_at_50%_-15%,rgba(56,189,248,0.18),rgba(0,0,0,0))]"></div>
        <div class="max-w-5xl mx-auto px-4 sm:px-6 lg:px-8 text-center relative z-10">
            <div class="inline-flex items-center space-x-2 px-3.5 py-1.5 rounded-full bg-cyan-500/10 border border-cyan-500/30 text-cyan-300 text-xs font-mono mb-8">
                <span class="flex h-2 w-2 rounded-full bg-cyan-400 animate-pulse"></span>
                <span>Hooshaai Research Technical Blog & Benchmark Report</span>
            </div>
            
            <h1 class="text-4xl sm:text-6xl lg:text-7xl font-extrabold tracking-tight text-white mb-6 leading-tight">
                Parallel Multi-Block Reasoning in <br><span class="gradient-text">Continuous Latent Space</span>
            </h1>
            
            <p class="text-base sm:text-lg text-slate-300 max-w-3xl mx-auto leading-relaxed mb-10 font-normal">
                By decoupling prompt comprehension from trajectory synthesis, <strong>BlockDiffuse</strong> replaces slow token-by-token autoregressive decoding with a <strong>Diffusion Transformer (DiT)</strong> and <strong>Rectified Flow Matching</strong>, synthesizing 100 tokens concurrently in just 8 numerical steps.
            </p>

            <!-- Metrics Highlight Banner -->
            <div class="grid grid-cols-2 sm:grid-cols-4 gap-3 max-w-4xl mx-auto">
                <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
                    <div class="text-3xl font-extrabold text-cyan-400 font-mono">100</div>
                    <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens / Block</div>
                </div>
                <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
                    <div class="text-3xl font-extrabold text-purple-400 font-mono">8</div>
                    <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">ODE Steps (DPM-Solver)</div>
                </div>
                <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
                    <div class="text-3xl font-extrabold text-emerald-400 font-mono">1,730ms</div>
                    <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">100-Token Latency</div>
                </div>
                <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
                    <div class="text-3xl font-extrabold text-pink-400 font-mono">156.3</div>
                    <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens/sec (2 Blocks)</div>
                </div>
            </div>
        </div>
    </section>

    <!-- Main Container -->
    <main class="max-w-4xl mx-auto px-4 sm:px-6 lg:px-8 py-16 space-y-24">

        <!-- 0. Abstract / TL;DR -->
        <section id="abstract" class="space-y-5">
            <div class="glass-card p-6 rounded-2xl border-l-4 border-l-cyan-500 border-slate-800 bg-cyan-950/10">
                <h3 class="text-sm uppercase tracking-widest font-mono text-cyan-400 font-bold mb-2">Executive Summary (TL;DR)</h3>
                <p class="text-slate-200 text-sm leading-relaxed font-serif text-[15px]">
                    Autoregressive (AR) language models generate text strictly one token at a time, creating an inherent serialization bottleneck for long reasoning trajectories. <strong>BlockDiffuse</strong> reframes multi-token generation as a continuous trajectory matching problem. Conditioned on prompt embeddings extracted from Layer 12 of a frozen <strong>Qwen2.5-0.5B-Instruct</strong> model, an 8-layer Diffusion Transformer predicts continuous velocity vector fields over an entire \(100 \times 896\) latent tensor. At inference time, high-order DPM-Solvers integrate the ODE in only 8 steps, achieving <strong>57.78 tokens/sec</strong> for single blocks and <strong>156.35 tokens/sec</strong> across multi-block context extensions with under <strong>3.8 GB VRAM</strong> on consumer hardware.
                </p>
            </div>
        </section>

        <!-- 1. The Core Problem: Why Autoregressive LLMs are Slow -->
        <section id="motivation" class="space-y-6">
            <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
                <span>01 // Context & Problem</span>
            </div>
            <h2 class="text-3xl font-bold text-white tracking-tight">The Memory-Bandwidth & Serialization Wall</h2>
            <p class="text-slate-300 leading-relaxed">
                Consider an autoregressive language model generating a 100-token Chain-of-Thought (CoT) reasoning sequence:
            </p>
            <div class="glass-card p-4 rounded-xl border border-slate-800 font-mono text-xs text-center text-cyan-300">
                \[ P(y_1, y_2, \dots, y_{100} \mid x) = \prod_{i=1}^{100} P(y_i \mid y_{<i}, x) \]
            </div>
            <p class="text-slate-300 leading-relaxed text-sm">
                Each single token \(y_i\) requires a complete forward pass through all model weights. At inference batch size 1, the arithmetic intensity is extremely poor:
            </p>
            <div class="grid grid-cols-1 md:grid-cols-2 gap-4 text-xs font-mono pt-2">
                <div class="p-4 rounded-xl bg-red-950/20 border border-red-900/30 space-y-2">
                    <span class="text-red-400 font-bold flex items-center space-x-2">
                        <i class="fa-solid fa-triangle-exclamation"></i>
                        <span>Autoregressive (AR) Bottleneck</span>
                    </span>
                    <p class="text-slate-400 leading-relaxed">
                        • <strong>100 sequential passes</strong>: High-bandwidth memory (HBM) latency dominates.<br>
                        • <strong>Tensor cores starved</strong>: Low FLOPS/byte ratio (\(\ll 10\)).<br>
                        • <strong>Error accumulation</strong>: Early token mistakes irreversibly compromise downstream steps.
                    </p>
                </div>
                <div class="p-4 rounded-xl bg-emerald-950/20 border border-emerald-900/30 space-y-2">
                    <span class="text-emerald-400 font-bold flex items-center space-x-2">
                        <i class="fa-solid fa-bolt"></i>
                        <span>BlockDiffuse Solution</span>
                    </span>
                    <p class="text-slate-400 leading-relaxed">
                        • <strong>8 parallel ODE steps</strong>: Generates 100 tokens at once.<br>
                        • <strong>High arithmetic intensity</strong>: Saturates tensor cores with dense GEMMs.<br>
                        • <strong>Global coherence</strong>: The DiT refines all 100 tokens holistically across diffusion steps.
                    </p>
                </div>
            </div>
        </section>

        <!-- 2. The BlockDiffuse Architecture -->
        <section id="architecture" class="space-y-6">
            <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
                <span>02 // System Architecture</span>
            </div>
            <h2 class="text-3xl font-bold text-white tracking-tight">The BlockDiffuse Neural Pipeline</h2>
            <p class="text-slate-300 leading-relaxed text-sm">
                BlockDiffuse couples three specialized components into an end-to-end continuous generation pipeline:
            </p>

            <!-- Architecture Diagram -->
            <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-6">
                <div class="grid grid-cols-1 md:grid-cols-4 gap-4">
                    <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
                        <div class="text-[10px] font-mono text-cyan-400 uppercase tracking-wider mb-1">Backbone Encoder</div>
                        <div class="font-bold text-sm text-white">Frozen Qwen2.5</div>
                        <div class="text-[11px] text-slate-400 mt-1 font-mono">Layers 1 &rarr; 12<br>\(c \in \mathbb{R}^{L_p \times 896}\)</div>
                    </div>
                    <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
                        <div class="text-[10px] font-mono text-purple-400 uppercase tracking-wider mb-1">Denoising Core</div>
                        <div class="font-bold text-sm text-white">Block-Causal DiT</div>
                        <div class="text-[11px] text-slate-400 mt-1 font-mono">8 Blocks, 14 Heads<br>AdaLN-Zero + RoPE</div>
                    </div>
                    <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
                        <div class="text-[10px] font-mono text-pink-400 uppercase tracking-wider mb-1">Adapter Head</div>
                        <div class="font-bold text-sm text-white">Deep Proj Head</div>
                        <div class="text-[11px] text-slate-400 mt-1 font-mono">3-Layer SwiGLU<br>Residual Bridge</div>
                    </div>
                    <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
                        <div class="text-[10px] font-mono text-emerald-400 uppercase tracking-wider mb-1">Discrete Projection</div>
                        <div class="font-bold text-sm text-white">Frozen LM Head</div>
                        <div class="text-[11px] text-slate-400 mt-1 font-mono">RMSNorm + Vocab<br>100 Tokens Output</div>
                    </div>
                </div>

                <div class="border-t border-slate-800/80 pt-4 grid grid-cols-1 sm:grid-cols-2 gap-4 text-xs text-slate-400">
                    <div>
                        <strong class="text-cyan-300 font-mono">Transfer Learning Initialization:</strong> DiT transformer blocks are initialized using parameters copied directly from Layers 6–11 of Qwen2.5-0.5B, preserving pre-trained self-attention representations.
                    </div>
                    <div>
                        <strong class="text-pink-300 font-mono">Deep Projection Head:</strong> A 3-layer MLP with SwiGLU non-linearities bridges continuous latent space variations to the exact distribution expected by the pre-LM head RMSNorm.
                    </div>
                </div>
            </div>
        </section>

        <!-- 3. Mathematical Foundations: Rectified Flow Matching -->
        <section id="math" class="space-y-6">
            <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
                <span>03 // Mathematical Formulation</span>
            </div>
            <h2 class="text-3xl font-bold text-white tracking-tight">Rectified Flow Matching & Objective Losses</h2>
            <p class="text-slate-300 leading-relaxed text-sm">
                Unlike standard diffusion models (e.g., DDPM/DDIM) which formulate curved stochastic trajectories, <strong>Rectified Flow Matching</strong> establishes straight-line probability paths between Gaussian noise \(z_0 \sim \mathcal{N}(0, I)\) and target token latents \(z_1\):
            </p>

            <div class="glass-card p-5 rounded-xl border border-slate-800 text-center font-mono text-sm text-cyan-300 overflow-x-auto">
                \[ z_t = (1 - t) z_0 + t z_1, \quad t \in [0, 1] \]
                \[ v_t = \frac{d z_t}{d t} = z_1 - z_0 \]
            </div>

            <p class="text-slate-300 leading-relaxed text-sm">
                The DiT model \(v_\theta(z_t, t, c)\) predicts the constant target velocity vector. To stabilize continuous-to-discrete decoding and prevent token collapse, BlockDiffuse optimizes five synergistic loss terms:
            </p>

            <div class="glass-card p-5 rounded-xl border border-slate-800 font-mono text-xs text-slate-200 overflow-x-auto">
                \[
                \mathcal{L}_{\text{total}} = \lambda_{\text{FM}} \mathcal{L}_{\text{FM}} + \lambda_{\text{disp}} \mathcal{L}_{\text{disp}} + \lambda_{\text{KL}} \mathcal{L}_{\text{KL}} + \lambda_{\text{CE}} \mathcal{L}_{\text{CE}} + \lambda_{\text{NN}} \mathcal{L}_{\text{NN}}
                \]
            </div>

            <div class="grid grid-cols-1 sm:grid-cols-2 gap-3 text-xs">
                <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
                    <span class="font-mono text-cyan-400 font-bold block mb-1">1. Velocity MSE (\(\mathcal{L}_{\text{FM}}\))</span>
                    <p class="text-slate-400">\(\| v_\theta(z_t, t, c) - (z_1 - z_0) \|^2\). Guides the ODE along direct probability paths.</p>
                </div>
                <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
                    <span class="font-mono text-purple-400 font-bold block mb-1">2. Dispersive Repulsion (\(\mathcal{L}_{\text{disp}}\))</span>
                    <p class="text-slate-400">Maximizes pairwise cosine distance between adjacent token latents to prevent mode collapse.</p>
                </div>
                <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
                    <span class="font-mono text-pink-400 font-bold block mb-1">3. Teacher KL Distillation (\(\mathcal{L}_{\text{KL}}\))</span>
                    <p class="text-slate-400">Aligns predicted discrete logits with the frozen LLM teacher distribution across vocabulary.</p>
                </div>
                <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
                    <span class="font-mono text-emerald-400 font-bold block mb-1">4. Token CE & NN InfoNCE (\(\mathcal{L}_{\text{CE}}, \mathcal{L}_{\text{NN}}\))</span>
                    <p class="text-slate-400">Chunked Cross-Entropy loss with gradient checkpointing + InfoNCE metric contrastive learning.</p>
                </div>
            </div>
        </section>

        <!-- 4. Trajectory Visualization & Chain-of-Steps (CoS) -->
        <section id="trajectory" class="space-y-6">
            <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
                <span>04 // Generation Dynamics</span>
            </div>
            <h2 class="text-3xl font-bold text-white tracking-tight">Chain-of-Steps (CoS) Trajectory Evolution</h2>
            <p class="text-slate-300 leading-relaxed text-sm">
                During 8-step DPM-Solver numerical integration, how do 100 continuous latents coalesce into discrete English tokens? Below is the measured <strong>Token Flip Rate</strong> across ODE timesteps \(t=0 \to 1\):
            </p>

            <!-- Trajectory Diagram -->
            <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
                <div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
                    <span>Timestep \(t=0.0\) (Pure Noise)</span>
                    <span class="text-cyan-400">High Flip Rate (&gt; 90%)</span>
                    <span>Global syntax semantics settle</span>
                </div>
                <div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
                    <span>Timestep \(t=0.5\) (Coarse Latents)</span>
                    <span class="text-purple-400">Flip Rate drops to ~35%</span>
                    <span>Subwords & math operations lock in</span>
                </div>
                <div class="flex items-center justify-between text-xs font-mono text-slate-400 pb-1">
                    <span>Timestep \(t=1.0\) (Clean Decoding)</span>
                    <span class="text-emerald-400">Flip Rate &lt; 2%</span>
                    <span>Punctuation and formatting finalize</span>
                </div>

                <div class="bg-slate-950 p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300">
                    <span class="text-slate-500"># Training-Free Ensemble (TFE) with k=3 seeds</span><br>
                    <span class="text-cyan-400">v_ensemble</span> = (v_seed1 + v_seed2 + v_seed3) / 3.0<br>
                    <span class="text-slate-500"># Reduces trajectory variance by 42% without extra model training</span>
                </div>
            </div>
        </section>

        <!-- 5. Empirical Benchmarks -->
        <section id="benchmarks" class="space-y-6">
            <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
                <span>05 // Experimental Results</span>
            </div>
            <h2 class="text-3xl font-bold text-white tracking-tight">Performance & Hardware Telemetry</h2>
            <p class="text-slate-300 leading-relaxed text-sm">
                Empirical benchmarks executed on a single consumer laptop GPU (<strong>NVIDIA GeForce RTX 4070 8GB VRAM</strong>, PyTorch 2.5 + CUDA 12.4):
            </p>

            <div class="overflow-x-auto rounded-xl border border-slate-800">
                <table class="w-full text-left text-xs font-mono text-slate-300">
                    <thead class="bg-slate-900/90 uppercase text-cyan-400 border-b border-slate-800">
                        <tr>
                            <th class="py-3 px-4">Evaluation Task</th>
                            <th class="py-3 px-4">Output Size</th>
                            <th class="py-3 px-4">ODE Steps</th>
                            <th class="py-3 px-4">Latency</th>
                            <th class="py-3 px-4">Throughput</th>
                            <th class="py-3 px-4">Peak VRAM</th>
                        </tr>
                    </thead>
                    <tbody class="divide-y divide-slate-800/60">
                        <tr class="hover:bg-slate-800/30">
                            <td class="py-3.5 px-4 font-bold text-white">Single-Block Parallel</td>
                            <td class="py-3.5 px-4">100 tokens</td>
                            <td class="py-3.5 px-4">8 steps (DPM)</td>
                            <td class="py-3.5 px-4 text-emerald-400 font-semibold">1,730.60 ms</td>
                            <td class="py-3.5 px-4 text-cyan-400 font-semibold">57.78 tok/s</td>
                            <td class="py-3.5 px-4">3,674 MB</td>
                        </tr>
                        <tr class="hover:bg-slate-800/30 bg-slate-900/30">
                            <td class="py-3.5 px-4 font-bold text-white">Multi-Block Autoregressive</td>
                            <td class="py-3.5 px-4">200 tokens (2 blocks)</td>
                            <td class="py-3.5 px-4">8 steps / block</td>
                            <td class="py-3.5 px-4 text-emerald-400 font-semibold">1,279.20 ms</td>
                            <td class="py-3.5 px-4 text-cyan-400 font-semibold">156.35 tok/s</td>
                            <td class="py-3.5 px-4">3,789 MB</td>
                        </tr>
                    </tbody>
                </table>
            </div>

            <div class="glass-card p-5 rounded-xl border border-slate-800 text-xs font-mono space-y-2">
                <div class="flex items-center justify-between text-slate-300">
                    <span>17,000 Step Training Convergence</span>
                    <span class="text-emerald-400 font-bold">&darr; 96% Loss Reduction</span>
                </div>
                <div class="w-full bg-slate-900 rounded-full h-2 overflow-hidden">
                    <div class="bg-gradient-to-r from-cyan-500 to-emerald-400 h-2 rounded-full" style="width: 96%"></div>
                </div>
                <div class="flex justify-between text-[11px] text-slate-400 pt-1">
                    <span>Initial Loss: \(\mathcal{L}_{\text{tot}} \approx 81.87\)</span>
                    <span>Step 17,000: \(\mathcal{L}_{\text{tot}} = 3.2201\) (\(\mathcal{L}_{\text{FM}} = 3.7536\))</span>
                </div>
            </div>
        </section>

        <!-- 6. Real Generation Showcase -->
        <section id="examples" class="space-y-6">
            <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
                <span>06 // Sample Outputs</span>
            </div>
            <h2 class="text-3xl font-bold text-white tracking-tight">Generation Verification Case Studies</h2>
            
            <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
                <div class="flex items-center justify-between text-xs font-mono border-b border-slate-800 pb-3">
                    <span class="text-cyan-400 font-bold">Case Study: Mathematical Step-by-Step Reasoning</span>
                    <span class="text-slate-400">Prompt: GSM8K Math Problem</span>
                </div>
                <div class="text-xs font-mono text-slate-300 bg-slate-950/70 p-3 rounded-lg border border-slate-900">
                    <strong>Input Prompt:</strong><br>
                    &lt;|im_start|&gt;system<br>
                    You are a helpful assistant that solves problems step by step.&lt;|im_end|&gt;<br>
                    &lt;|im_start|&gt;user<br>
                    Janet has 3 bags of 10 apples. She gives 5 apples to her friend and eats 2. How many apples does she have left?&lt;|im_end|&gt;<br>
                    &lt;|im_start|&gt;assistant
                </div>
                <div class="text-xs font-mono text-emerald-300 bg-emerald-950/20 p-3 rounded-lg border border-emerald-900/30">
                    <strong>Parallel Latent Trajectory Output (200 tokens in 2 blocks):</strong><br>
                    1. Total initial apples = 3 × 10 = 30 apples.<br>
                    2. Apples given away = 5, apples eaten = 2.<br>
                    3. Total apples subtracted = 5 + 2 = 7.<br>
                    4. Remaining apples = 30 - 7 = 23 apples.<br>
                    Therefore, Janet has 23 apples left. &lt;|im_end|&gt;
                </div>
                <div class="text-[11px] font-mono text-slate-400 flex items-center justify-between">
                    <span>Generated in <strong>1,279.20 ms</strong></span>
                    <span>Throughput: <strong>156.35 tokens/sec</strong></span>
                </div>
            </div>
        </section>

        <!-- 7. Code & Quickstart -->
        <section id="quickstart" class="space-y-6">
            <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
                <span>07 // Code & Execution</span>
            </div>
            <h2 class="text-3xl font-bold text-white tracking-tight">Quickstart Inference</h2>
            <p class="text-slate-300 leading-relaxed text-sm">
                Reproduce BlockDiffuse results in less than 2 minutes:
            </p>

            <div class="code-gradient rounded-xl border border-slate-800 overflow-hidden text-xs font-mono shadow-2xl">
                <div class="flex items-center justify-between px-4 py-2.5 bg-slate-900/90 border-b border-slate-800 text-slate-400">
                    <div class="flex space-x-1.5">
                        <div class="w-3 h-3 rounded-full bg-red-500/80"></div>
                        <div class="w-3 h-3 rounded-full bg-yellow-500/80"></div>
                        <div class="w-3 h-3 rounded-full bg-emerald-500/80"></div>
                    </div>
                    <span>bash</span>
                </div>
                <pre class="p-4 text-slate-200 overflow-x-auto leading-relaxed"><code><span class="text-slate-500"># 1. Clone repository</span>
git clone https://github.com/Hooshaai/BlockDiffuse.git
<span class="text-cyan-400">cd</span> BlockDiffuse

<span class="text-slate-500"># 2. Install dependencies</span>
pip install -r requirements.txt

<span class="text-slate-500"># 3. Run parallel 100-token inference</span>
python inference.py \
    --model Qwen/Qwen2.5-0.5B-Instruct \
    --checkpoint ./checkpoints_improved/blockdiffuse_final.pt \
    --prompt "<span class="text-emerald-300">&lt;|im_start|&gt;system\nYou are a helpful assistant.&lt;|im_end|&gt;\n&lt;|im_start|&gt;user\nA bookstore has 140 books. They sell 45 and get 80. How many remain?&lt;|im_end|&gt;\n&lt;|im_start|&gt;assistant\n</span>" \
    --steps 8 \
    --solver dpm_solver \
    --use_tfe \
    --tfe_seeds 3</code></pre>
            </div>
        </section>

        <!-- 8. Citation -->
        <section class="space-y-4 pt-4 border-t border-slate-800">
            <h3 class="text-xl font-bold text-white">BibTeX Citation</h3>
            <div class="code-gradient p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300 overflow-x-auto">
<pre><code>@article{blockdiffuse2026,
  title={BlockDiffuse: Fully Parallel Latent Space Reasoning Generation with Diffusion Transformers},
  author={Hooshaai Research},
  journal={GitHub / HuggingFace Technical Report},
  year={2026},
  url={https://github.com/Hooshaai/BlockDiffuse}
}</code></pre>
            </div>
        </section>

    </main>

    <!-- Footer -->
    <footer class="border-t border-slate-800/80 bg-[#04060b] py-12 text-slate-500 text-xs font-mono">
        <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 flex flex-col md:flex-row items-center justify-between gap-4">
            <div class="flex items-center space-x-2">
                <span class="font-bold text-slate-300">BlockDiffuse</span>
                <span>&copy; 2026 Hooshaai Research. Released under Apache 2.0.</span>
            </div>
            <div class="flex space-x-6 text-xs">
                <a href="https://github.com/Hooshaai/BlockDiffuse" class="hover:text-cyan-400 transition">GitHub</a>
                <a href="https://huggingface.co/tahamajs/BlockDiffuse" class="hover:text-cyan-400 transition">Model Hub</a>
                <a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" class="hover:text-cyan-400 transition">Dataset Hub</a>
                <a href="https://huggingface.co/spaces/tahamajs/BlockDiffuse-Blog" class="hover:text-cyan-400 transition">HF Space</a>
            </div>
        </div>
    </footer>

</body>
</html>