tahamajs commited on
Commit
abe48dc
·
verified ·
1 Parent(s): 5f0e110

Deploy interactive animated multi-slide presentation deck and ODE simulator

Browse files
Files changed (1) hide show
  1. index.html +532 -266
index.html CHANGED
@@ -4,8 +4,8 @@
4
  <meta charset="UTF-8">
5
  <meta name="viewport" content="width=device-width, initial-scale=1.0">
6
  <title>BlockDiffuse: Fully Parallel Latent Space Reasoning with Diffusion Transformers</title>
7
- <meta name="description" content="Official Research Blog & Technical Report for BlockDiffuse: Non-autoregressive 100-token block generation in continuous latent space using Rectified Flow Matching and DiT.">
8
- <meta name="keywords" content="BlockDiffuse, Diffusion Transformers, Rectified Flow Matching, Non-Autoregressive, Qwen2.5, Deep Learning, Chain-of-Thought">
9
 
10
  <!-- OpenGraph Metadata -->
11
  <meta property="og:title" content="BlockDiffuse: Parallel 100-Token Reasoning in Continuous Latent Space">
@@ -22,7 +22,7 @@
22
  <!-- Google Fonts -->
23
  <link rel="preconnect" href="https://fonts.googleapis.com">
24
  <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
25
- <link href="https://fonts.googleapis.com/css2?family=Fira+Code:wght@400;500;600;700&family=Inter:wght@300;400;500;600;700;800&family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,600;1,6..72,400&display=swap" rel="stylesheet">
26
 
27
  <script>
28
  tailwind.config = {
@@ -45,6 +45,22 @@
45
  card: '#0f172a',
46
  border: '#1e293b'
47
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
48
  }
49
  }
50
  }
@@ -56,398 +72,516 @@
56
  -webkit-background-clip: text;
57
  -webkit-text-fill-color: transparent;
58
  }
 
 
 
59
  .code-gradient {
60
  background: linear-gradient(180deg, rgba(15,23,42,0.95) 0%, rgba(7,11,20,0.98) 100%);
61
  }
62
  .glass-card {
63
- background: rgba(15, 23, 42, 0.78);
64
- backdrop-filter: blur(14px);
65
  border: 1px solid rgba(255, 255, 255, 0.08);
66
  }
67
- .glass-card-hover:hover {
68
- border-color: rgba(56, 189, 248, 0.35);
69
- transform: translateY(-2px);
70
- transition: all 0.25s ease-in-out;
71
  }
72
- .tab-active {
73
- border-color: #38bdf8;
74
- color: #38bdf8;
75
- background-color: rgba(56, 189, 248, 0.1);
 
 
76
  }
77
  </style>
78
  </head>
79
- <body class="bg-[#060911] text-slate-200 font-sans antialiased selection:bg-cyan-500 selection:text-black">
80
 
81
- <!-- Top Alert Banner -->
82
- <div class="bg-gradient-to-r from-cyan-950/60 via-purple-950/60 to-pink-950/60 border-b border-cyan-500/20 py-2 px-4 text-center text-xs font-mono text-cyan-300">
83
- 🎉 <strong>Research Release:</strong> Checkpoint weights, datasets, and code are now public on Hugging Face & GitHub!
 
84
  </div>
85
 
86
  <!-- Navigation Header -->
87
  <header class="sticky top-0 z-50 glass-card border-b border-slate-800/80">
88
  <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 h-16 flex items-center justify-between">
89
  <div class="flex items-center space-x-3">
90
- <div class="h-9 w-9 rounded-lg bg-gradient-to-tr from-cyan-500 via-indigo-500 to-pink-500 flex items-center justify-center text-white font-black text-xl shadow-lg shadow-cyan-500/20">
91
  B
92
  </div>
93
  <div>
94
- <span class="text-xl font-bold tracking-tight text-white font-mono">Block<span class="text-cyan-400">Diffuse</span></span>
95
- <span class="hidden sm:inline-block text-[10px] bg-slate-800 border border-slate-700 text-cyan-400 px-2 py-0.5 rounded-full font-mono ml-2">v1.0-Paper</span>
96
  </div>
97
  </div>
98
 
99
- <nav class="hidden lg:flex items-center space-x-7 text-xs font-medium text-slate-400 font-mono uppercase tracking-wider">
100
- <a href="#abstract" class="hover:text-cyan-400 transition">Abstract</a>
101
- <a href="#motivation" class="hover:text-cyan-400 transition">Motivation</a>
102
  <a href="#architecture" class="hover:text-cyan-400 transition">Architecture</a>
103
  <a href="#math" class="hover:text-cyan-400 transition">Flow Matching</a>
104
- <a href="#trajectory" class="hover:text-cyan-400 transition">Trajectory</a>
105
- <a href="#benchmarks" class="hover:text-cyan-400 transition">Benchmarks</a>
106
- <a href="#examples" class="hover:text-cyan-400 transition">Generations</a>
107
  <a href="#quickstart" class="hover:text-cyan-400 transition">Code</a>
108
  </nav>
109
 
110
- <div class="flex items-center space-x-2.5">
111
- <a href="https://huggingface.co/tahamajs/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-yellow-500/10 hover:bg-yellow-500/20 border border-yellow-500/30 text-yellow-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
112
  <span>🤗</span>
113
- <span class="hidden sm:inline">Model</span>
114
- </a>
115
- <a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" target="_blank" class="flex items-center space-x-1.5 bg-cyan-500/10 hover:bg-cyan-500/20 border border-cyan-500/30 text-cyan-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
116
- <i class="fa-solid fa-database text-xs"></i>
117
- <span class="hidden sm:inline">Data</span>
118
  </a>
119
- <a href="https://github.com/Hooshaai/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-slate-800 hover:bg-slate-700 border border-slate-700 text-white px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
120
  <i class="fa-brands fa-github text-sm"></i>
121
- <span class="hidden sm:inline">Code</span>
122
  </a>
123
  </div>
124
  </div>
125
  </header>
126
 
127
- <!-- Hero Section -->
128
- <section class="relative pt-20 pb-20 overflow-hidden border-b border-slate-800/80">
129
- <div class="absolute inset-0 bg-[radial-gradient(ellipse_75%_50%_at_50%_-15%,rgba(56,189,248,0.18),rgba(0,0,0,0))]"></div>
130
  <div class="max-w-5xl mx-auto px-4 sm:px-6 lg:px-8 text-center relative z-10">
131
- <div class="inline-flex items-center space-x-2 px-3.5 py-1.5 rounded-full bg-cyan-500/10 border border-cyan-500/30 text-cyan-300 text-xs font-mono mb-8">
132
  <span class="flex h-2 w-2 rounded-full bg-cyan-400 animate-pulse"></span>
133
- <span>Hooshaai Research Technical Blog & Benchmark Report</span>
134
  </div>
135
 
136
  <h1 class="text-4xl sm:text-6xl lg:text-7xl font-extrabold tracking-tight text-white mb-6 leading-tight">
137
- Parallel Multi-Block Reasoning in <br><span class="gradient-text">Continuous Latent Space</span>
138
  </h1>
139
 
140
  <p class="text-base sm:text-lg text-slate-300 max-w-3xl mx-auto leading-relaxed mb-10 font-normal">
141
- By decoupling prompt comprehension from trajectory synthesis, <strong>BlockDiffuse</strong> replaces slow token-by-token autoregressive decoding with a <strong>Diffusion Transformer (DiT)</strong> and <strong>Rectified Flow Matching</strong>, synthesizing 100 tokens concurrently in just 8 numerical steps.
142
  </p>
143
 
144
- <!-- Metrics Highlight Banner -->
145
  <div class="grid grid-cols-2 sm:grid-cols-4 gap-3 max-w-4xl mx-auto">
146
- <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
147
  <div class="text-3xl font-extrabold text-cyan-400 font-mono">100</div>
148
- <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens / Block</div>
149
  </div>
150
- <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
151
  <div class="text-3xl font-extrabold text-purple-400 font-mono">8</div>
152
- <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">ODE Steps (DPM-Solver)</div>
153
  </div>
154
- <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
155
  <div class="text-3xl font-extrabold text-emerald-400 font-mono">1,730ms</div>
156
  <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">100-Token Latency</div>
157
  </div>
158
- <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
159
- <div class="text-3xl font-extrabold text-pink-400 font-mono">156.3</div>
160
  <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens/sec (2 Blocks)</div>
161
  </div>
162
  </div>
163
  </div>
164
  </section>
165
 
166
- <!-- Main Container -->
167
- <main class="max-w-4xl mx-auto px-4 sm:px-6 lg:px-8 py-16 space-y-24">
168
-
169
- <!-- 0. Abstract / TL;DR -->
170
- <section id="abstract" class="space-y-5">
171
- <div class="glass-card p-6 rounded-2xl border-l-4 border-l-cyan-500 border-slate-800 bg-cyan-950/10">
172
- <h3 class="text-sm uppercase tracking-widest font-mono text-cyan-400 font-bold mb-2">Executive Summary (TL;DR)</h3>
173
- <p class="text-slate-200 text-sm leading-relaxed font-serif text-[15px]">
174
- Autoregressive (AR) language models generate text strictly one token at a time, creating an inherent serialization bottleneck for long reasoning trajectories. <strong>BlockDiffuse</strong> reframes multi-token generation as a continuous trajectory matching problem. Conditioned on prompt embeddings extracted from Layer 12 of a frozen <strong>Qwen2.5-0.5B-Instruct</strong> model, an 8-layer Diffusion Transformer predicts continuous velocity vector fields over an entire \(100 \times 896\) latent tensor. At inference time, high-order DPM-Solvers integrate the ODE in only 8 steps, achieving <strong>57.78 tokens/sec</strong> for single blocks and <strong>156.35 tokens/sec</strong> across multi-block context extensions with under <strong>3.8 GB VRAM</strong> on consumer hardware.
175
- </p>
 
 
 
 
 
 
176
  </div>
177
- </section>
178
 
179
- <!-- 1. The Core Problem: Why Autoregressive LLMs are Slow -->
180
- <section id="motivation" class="space-y-6">
181
- <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
182
- <span>01 // Context & Problem</span>
183
- </div>
184
- <h2 class="text-3xl font-bold text-white tracking-tight">The Memory-Bandwidth & Serialization Wall</h2>
185
- <p class="text-slate-300 leading-relaxed">
186
- Consider an autoregressive language model generating a 100-token Chain-of-Thought (CoT) reasoning sequence:
187
- </p>
188
- <div class="glass-card p-4 rounded-xl border border-slate-800 font-mono text-xs text-center text-cyan-300">
189
- \[ P(y_1, y_2, \dots, y_{100} \mid x) = \prod_{i=1}^{100} P(y_i \mid y_{<i}, x) \]
190
- </div>
191
- <p class="text-slate-300 leading-relaxed text-sm">
192
- Each single token \(y_i\) requires a complete forward pass through all model weights. At inference batch size 1, the arithmetic intensity is extremely poor:
193
- </p>
194
- <div class="grid grid-cols-1 md:grid-cols-2 gap-4 text-xs font-mono pt-2">
195
- <div class="p-4 rounded-xl bg-red-950/20 border border-red-900/30 space-y-2">
196
- <span class="text-red-400 font-bold flex items-center space-x-2">
197
- <i class="fa-solid fa-triangle-exclamation"></i>
198
- <span>Autoregressive (AR) Bottleneck</span>
199
- </span>
200
- <p class="text-slate-400 leading-relaxed">
201
- • <strong>100 sequential passes</strong>: High-bandwidth memory (HBM) latency dominates.<br>
202
- • <strong>Tensor cores starved</strong>: Low FLOPS/byte ratio (\(\ll 10\)).<br>
203
- • <strong>Error accumulation</strong>: Early token mistakes irreversibly compromise downstream steps.
204
  </p>
 
 
 
 
 
 
 
 
205
  </div>
206
- <div class="p-4 rounded-xl bg-emerald-950/20 border border-emerald-900/30 space-y-2">
207
- <span class="text-emerald-400 font-bold flex items-center space-x-2">
208
- <i class="fa-solid fa-bolt"></i>
209
- <span>BlockDiffuse Solution</span>
210
- </span>
211
- <p class="text-slate-400 leading-relaxed">
212
- • <strong>8 parallel ODE steps</strong>: Generates 100 tokens at once.<br>
213
- • <strong>High arithmetic intensity</strong>: Saturates tensor cores with dense GEMMs.<br>
214
- • <strong>Global coherence</strong>: The DiT refines all 100 tokens holistically across diffusion steps.
 
 
 
 
 
 
215
  </p>
216
  </div>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
217
  </div>
218
  </section>
219
 
220
- <!-- 2. The BlockDiffuse Architecture -->
221
- <section id="architecture" class="space-y-6">
 
 
222
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
223
- <span>02 // System Architecture</span>
 
 
224
  </div>
225
- <h2 class="text-3xl font-bold text-white tracking-tight">The BlockDiffuse Neural Pipeline</h2>
226
  <p class="text-slate-300 leading-relaxed text-sm">
227
- BlockDiffuse couples three specialized components into an end-to-end continuous generation pipeline:
228
  </p>
229
 
230
- <!-- Architecture Diagram -->
231
- <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-6">
232
- <div class="grid grid-cols-1 md:grid-cols-4 gap-4">
233
- <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
234
- <div class="text-[10px] font-mono text-cyan-400 uppercase tracking-wider mb-1">Backbone Encoder</div>
235
- <div class="font-bold text-sm text-white">Frozen Qwen2.5</div>
236
- <div class="text-[11px] text-slate-400 mt-1 font-mono">Layers 1 &rarr; 12<br>\(c \in \mathbb{R}^{L_p \times 896}\)</div>
 
 
 
 
 
 
 
 
237
  </div>
238
- <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
239
- <div class="text-[10px] font-mono text-purple-400 uppercase tracking-wider mb-1">Denoising Core</div>
240
- <div class="font-bold text-sm text-white">Block-Causal DiT</div>
241
- <div class="text-[11px] text-slate-400 mt-1 font-mono">8 Blocks, 14 Heads<br>AdaLN-Zero + RoPE</div>
 
 
 
 
242
  </div>
243
- <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
244
- <div class="text-[10px] font-mono text-pink-400 uppercase tracking-wider mb-1">Adapter Head</div>
245
- <div class="font-bold text-sm text-white">Deep Proj Head</div>
246
- <div class="text-[11px] text-slate-400 mt-1 font-mono">3-Layer SwiGLU<br>Residual Bridge</div>
247
  </div>
248
- <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
249
- <div class="text-[10px] font-mono text-emerald-400 uppercase tracking-wider mb-1">Discrete Projection</div>
250
- <div class="font-bold text-sm text-white">Frozen LM Head</div>
251
- <div class="text-[11px] text-slate-400 mt-1 font-mono">RMSNorm + Vocab<br>100 Tokens Output</div>
252
  </div>
253
  </div>
254
 
255
- <div class="border-t border-slate-800/80 pt-4 grid grid-cols-1 sm:grid-cols-2 gap-4 text-xs text-slate-400">
256
- <div>
257
- <strong class="text-cyan-300 font-mono">Transfer Learning Initialization:</strong> DiT transformer blocks are initialized using parameters copied directly from Layers 6–11 of Qwen2.5-0.5B, preserving pre-trained self-attention representations.
 
 
258
  </div>
259
- <div>
260
- <strong class="text-pink-300 font-mono">Deep Projection Head:</strong> A 3-layer MLP with SwiGLU non-linearities bridges continuous latent space variations to the exact distribution expected by the pre-LM head RMSNorm.
261
  </div>
262
  </div>
263
  </div>
264
  </section>
265
 
266
- <!-- 3. Mathematical Foundations: Rectified Flow Matching -->
267
- <section id="math" class="space-y-6">
 
 
268
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
269
- <span>03 // Mathematical Formulation</span>
270
- </div>
271
- <h2 class="text-3xl font-bold text-white tracking-tight">Rectified Flow Matching & Objective Losses</h2>
272
- <p class="text-slate-300 leading-relaxed text-sm">
273
- Unlike standard diffusion models (e.g., DDPM/DDIM) which formulate curved stochastic trajectories, <strong>Rectified Flow Matching</strong> establishes straight-line probability paths between Gaussian noise \(z_0 \sim \mathcal{N}(0, I)\) and target token latents \(z_1\):
274
- </p>
275
-
276
- <div class="glass-card p-5 rounded-xl border border-slate-800 text-center font-mono text-sm text-cyan-300 overflow-x-auto">
277
- \[ z_t = (1 - t) z_0 + t z_1, \quad t \in [0, 1] \]
278
- \[ v_t = \frac{d z_t}{d t} = z_1 - z_0 \]
279
  </div>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
280
 
281
- <p class="text-slate-300 leading-relaxed text-sm">
282
- The DiT model \(v_\theta(z_t, t, c)\) predicts the constant target velocity vector. To stabilize continuous-to-discrete decoding and prevent token collapse, BlockDiffuse optimizes five synergistic loss terms:
283
- </p>
 
 
 
 
 
 
 
284
 
285
- <div class="glass-card p-5 rounded-xl border border-slate-800 font-mono text-xs text-slate-200 overflow-x-auto">
286
- \[
287
- \mathcal{L}_{\text{total}} = \lambda_{\text{FM}} \mathcal{L}_{\text{FM}} + \lambda_{\text{disp}} \mathcal{L}_{\text{disp}} + \lambda_{\text{KL}} \mathcal{L}_{\text{KL}} + \lambda_{\text{CE}} \mathcal{L}_{\text{CE}} + \lambda_{\text{NN}} \mathcal{L}_{\text{NN}}
288
- \]
289
- </div>
 
 
 
 
 
290
 
291
- <div class="grid grid-cols-1 sm:grid-cols-2 gap-3 text-xs">
292
- <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
293
- <span class="font-mono text-cyan-400 font-bold block mb-1">1. Velocity MSE (\(\mathcal{L}_{\text{FM}}\))</span>
294
- <p class="text-slate-400">\(\| v_\theta(z_t, t, c) - (z_1 - z_0) \|^2\). Guides the ODE along direct probability paths.</p>
295
- </div>
296
- <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
297
- <span class="font-mono text-purple-400 font-bold block mb-1">2. Dispersive Repulsion (\(\mathcal{L}_{\text{disp}}\))</span>
298
- <p class="text-slate-400">Maximizes pairwise cosine distance between adjacent token latents to prevent mode collapse.</p>
299
- </div>
300
- <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
301
- <span class="font-mono text-pink-400 font-bold block mb-1">3. Teacher KL Distillation (\(\mathcal{L}_{\text{KL}}\))</span>
302
- <p class="text-slate-400">Aligns predicted discrete logits with the frozen LLM teacher distribution across vocabulary.</p>
303
  </div>
304
- <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
305
- <span class="font-mono text-emerald-400 font-bold block mb-1">4. Token CE & NN InfoNCE (\(\mathcal{L}_{\text{CE}}, \mathcal{L}_{\text{NN}}\))</span>
306
- <p class="text-slate-400">Chunked Cross-Entropy loss with gradient checkpointing + InfoNCE metric contrastive learning.</p>
 
307
  </div>
308
  </div>
309
  </section>
310
 
311
- <!-- 4. Trajectory Visualization & Chain-of-Steps (CoS) -->
312
- <section id="trajectory" class="space-y-6">
 
 
313
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
314
- <span>04 // Generation Dynamics</span>
 
 
315
  </div>
316
- <h2 class="text-3xl font-bold text-white tracking-tight">Chain-of-Steps (CoS) Trajectory Evolution</h2>
317
  <p class="text-slate-300 leading-relaxed text-sm">
318
- During 8-step DPM-Solver numerical integration, how do 100 continuous latents coalesce into discrete English tokens? Below is the measured <strong>Token Flip Rate</strong> across ODE timesteps \(t=0 \to 1\):
319
  </p>
320
 
321
- <!-- Trajectory Diagram -->
322
- <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
323
- <div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
324
- <span>Timestep \(t=0.0\) (Pure Noise)</span>
325
- <span class="text-cyan-400">High Flip Rate (&gt; 90%)</span>
326
- <span>Global syntax semantics settle</span>
327
  </div>
328
- <div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
329
- <span>Timestep \(t=0.5\) (Coarse Latents)</span>
330
- <span class="text-purple-400">Flip Rate drops to ~35%</span>
331
- <span>Subwords & math operations lock in</span>
 
 
 
 
 
 
 
 
 
 
 
 
332
  </div>
333
- <div class="flex items-center justify-between text-xs font-mono text-slate-400 pb-1">
334
- <span>Timestep \(t=1.0\) (Clean Decoding)</span>
335
- <span class="text-emerald-400">Flip Rate &lt; 2%</span>
336
- <span>Punctuation and formatting finalize</span>
 
 
337
  </div>
338
-
339
- <div class="bg-slate-950 p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300">
340
- <span class="text-slate-500"># Training-Free Ensemble (TFE) with k=3 seeds</span><br>
341
- <span class="text-cyan-400">v_ensemble</span> = (v_seed1 + v_seed2 + v_seed3) / 3.0<br>
342
- <span class="text-slate-500"># Reduces trajectory variance by 42% without extra model training</span>
343
  </div>
344
  </div>
345
  </section>
346
 
347
- <!-- 5. Empirical Benchmarks -->
 
 
348
  <section id="benchmarks" class="space-y-6">
349
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
350
- <span>05 // Experimental Results</span>
 
 
351
  </div>
352
- <h2 class="text-3xl font-bold text-white tracking-tight">Performance & Hardware Telemetry</h2>
353
  <p class="text-slate-300 leading-relaxed text-sm">
354
- Empirical benchmarks executed on a single consumer laptop GPU (<strong>NVIDIA GeForce RTX 4070 8GB VRAM</strong>, PyTorch 2.5 + CUDA 12.4):
355
  </p>
356
 
357
- <div class="overflow-x-auto rounded-xl border border-slate-800">
358
  <table class="w-full text-left text-xs font-mono text-slate-300">
359
  <thead class="bg-slate-900/90 uppercase text-cyan-400 border-b border-slate-800">
360
  <tr>
361
- <th class="py-3 px-4">Evaluation Task</th>
362
- <th class="py-3 px-4">Output Size</th>
363
- <th class="py-3 px-4">ODE Steps</th>
364
- <th class="py-3 px-4">Latency</th>
365
- <th class="py-3 px-4">Throughput</th>
366
- <th class="py-3 px-4">Peak VRAM</th>
367
  </tr>
368
  </thead>
369
- <tbody class="divide-y divide-slate-800/60">
370
  <tr class="hover:bg-slate-800/30">
371
- <td class="py-3.5 px-4 font-bold text-white">Single-Block Parallel</td>
372
- <td class="py-3.5 px-4">100 tokens</td>
373
- <td class="py-3.5 px-4">8 steps (DPM)</td>
374
- <td class="py-3.5 px-4 text-emerald-400 font-semibold">1,730.60 ms</td>
375
- <td class="py-3.5 px-4 text-cyan-400 font-semibold">57.78 tok/s</td>
376
- <td class="py-3.5 px-4">3,674 MB</td>
377
  </tr>
378
- <tr class="hover:bg-slate-800/30 bg-slate-900/30">
379
- <td class="py-3.5 px-4 font-bold text-white">Multi-Block Autoregressive</td>
380
- <td class="py-3.5 px-4">200 tokens (2 blocks)</td>
381
- <td class="py-3.5 px-4">8 steps / block</td>
382
- <td class="py-3.5 px-4 text-emerald-400 font-semibold">1,279.20 ms</td>
383
- <td class="py-3.5 px-4 text-cyan-400 font-semibold">156.35 tok/s</td>
384
- <td class="py-3.5 px-4">3,789 MB</td>
385
  </tr>
386
  </tbody>
387
  </table>
388
  </div>
389
 
390
- <div class="glass-card p-5 rounded-xl border border-slate-800 text-xs font-mono space-y-2">
 
391
  <div class="flex items-center justify-between text-slate-300">
392
- <span>17,000 Step Training Convergence</span>
393
- <span class="text-emerald-400 font-bold">&darr; 96% Loss Reduction</span>
394
  </div>
395
- <div class="w-full bg-slate-900 rounded-full h-2 overflow-hidden">
396
- <div class="bg-gradient-to-r from-cyan-500 to-emerald-400 h-2 rounded-full" style="width: 96%"></div>
397
  </div>
398
- <div class="flex justify-between text-[11px] text-slate-400 pt-1">
399
  <span>Initial Loss: \(\mathcal{L}_{\text{tot}} \approx 81.87\)</span>
400
- <span>Step 17,000: \(\mathcal{L}_{\text{tot}} = 3.2201\) (\(\mathcal{L}_{\text{FM}} = 3.7536\))</span>
401
  </div>
402
  </div>
403
  </section>
404
 
405
- <!-- 6. Real Generation Showcase -->
406
- <section id="examples" class="space-y-6">
407
- <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
408
- <span>06 // Sample Outputs</span>
409
- </div>
410
- <h2 class="text-3xl font-bold text-white tracking-tight">Generation Verification Case Studies</h2>
411
-
412
- <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
413
- <div class="flex items-center justify-between text-xs font-mono border-b border-slate-800 pb-3">
414
- <span class="text-cyan-400 font-bold">Case Study: Mathematical Step-by-Step Reasoning</span>
415
- <span class="text-slate-400">Prompt: GSM8K Math Problem</span>
416
- </div>
417
- <div class="text-xs font-mono text-slate-300 bg-slate-950/70 p-3 rounded-lg border border-slate-900">
418
- <strong>Input Prompt:</strong><br>
419
- &lt;|im_start|&gt;system<br>
420
- You are a helpful assistant that solves problems step by step.&lt;|im_end|&gt;<br>
421
- &lt;|im_start|&gt;user<br>
422
- Janet has 3 bags of 10 apples. She gives 5 apples to her friend and eats 2. How many apples does she have left?&lt;|im_end|&gt;<br>
423
- &lt;|im_start|&gt;assistant
424
- </div>
425
- <div class="text-xs font-mono text-emerald-300 bg-emerald-950/20 p-3 rounded-lg border border-emerald-900/30">
426
- <strong>Parallel Latent Trajectory Output (200 tokens in 2 blocks):</strong><br>
427
- 1. Total initial apples = 3 × 10 = 30 apples.<br>
428
- 2. Apples given away = 5, apples eaten = 2.<br>
429
- 3. Total apples subtracted = 5 + 2 = 7.<br>
430
- 4. Remaining apples = 30 - 7 = 23 apples.<br>
431
- Therefore, Janet has 23 apples left. &lt;|im_end|&gt;
432
- </div>
433
- <div class="text-[11px] font-mono text-slate-400 flex items-center justify-between">
434
- <span>Generated in <strong>1,279.20 ms</strong></span>
435
- <span>Throughput: <strong>156.35 tokens/sec</strong></span>
436
- </div>
437
- </div>
438
- </section>
439
-
440
- <!-- 7. Code & Quickstart -->
441
  <section id="quickstart" class="space-y-6">
442
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
443
- <span>07 // Code & Execution</span>
 
 
444
  </div>
445
- <h2 class="text-3xl font-bold text-white tracking-tight">Quickstart Inference</h2>
446
- <p class="text-slate-300 leading-relaxed text-sm">
447
- Reproduce BlockDiffuse results in less than 2 minutes:
448
- </p>
449
 
450
- <div class="code-gradient rounded-xl border border-slate-800 overflow-hidden text-xs font-mono shadow-2xl">
451
  <div class="flex items-center justify-between px-4 py-2.5 bg-slate-900/90 border-b border-slate-800 text-slate-400">
452
  <div class="flex space-x-1.5">
453
  <div class="w-3 h-3 rounded-full bg-red-500/80"></div>
@@ -456,18 +590,19 @@
456
  </div>
457
  <span>bash</span>
458
  </div>
459
- <pre class="p-4 text-slate-200 overflow-x-auto leading-relaxed"><code><span class="text-slate-500"># 1. Clone repository</span>
460
  git clone https://github.com/Hooshaai/BlockDiffuse.git
461
  <span class="text-cyan-400">cd</span> BlockDiffuse
462
 
463
  <span class="text-slate-500"># 2. Install dependencies</span>
464
  pip install -r requirements.txt
465
 
466
- <span class="text-slate-500"># 3. Run parallel 100-token inference</span>
467
  python inference.py \
468
  --model Qwen/Qwen2.5-0.5B-Instruct \
469
  --checkpoint ./checkpoints_improved/blockdiffuse_final.pt \
470
- --prompt "<span class="text-emerald-300">&lt;|im_start|&gt;system\nYou are a helpful assistant.&lt;|im_end|&gt;\n&lt;|im_start|&gt;user\nA bookstore has 140 books. They sell 45 and get 80. How many remain?&lt;|im_end|&gt;\n&lt;|im_start|&gt;assistant\n</span>" \
 
471
  --steps 8 \
472
  --solver dpm_solver \
473
  --use_tfe \
@@ -475,7 +610,7 @@ python inference.py \
475
  </div>
476
  </section>
477
 
478
- <!-- 8. Citation -->
479
  <section class="space-y-4 pt-4 border-t border-slate-800">
480
  <h3 class="text-xl font-bold text-white">BibTeX Citation</h3>
481
  <div class="code-gradient p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300 overflow-x-auto">
@@ -492,20 +627,151 @@ python inference.py \
492
  </main>
493
 
494
  <!-- Footer -->
495
- <footer class="border-t border-slate-800/80 bg-[#04060b] py-12 text-slate-500 text-xs font-mono">
496
  <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 flex flex-col md:flex-row items-center justify-between gap-4">
497
  <div class="flex items-center space-x-2">
498
  <span class="font-bold text-slate-300">BlockDiffuse</span>
499
- <span>&copy; 2026 Hooshaai Research. Released under Apache 2.0.</span>
500
  </div>
501
  <div class="flex space-x-6 text-xs">
502
  <a href="https://github.com/Hooshaai/BlockDiffuse" class="hover:text-cyan-400 transition">GitHub</a>
503
- <a href="https://huggingface.co/tahamajs/BlockDiffuse" class="hover:text-cyan-400 transition">Model Hub</a>
504
- <a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" class="hover:text-cyan-400 transition">Dataset Hub</a>
505
- <a href="https://huggingface.co/spaces/tahamajs/BlockDiffuse-Blog" class="hover:text-cyan-400 transition">HF Space</a>
506
  </div>
507
  </div>
508
  </footer>
509
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
510
  </body>
511
  </html>
 
4
  <meta charset="UTF-8">
5
  <meta name="viewport" content="width=device-width, initial-scale=1.0">
6
  <title>BlockDiffuse: Fully Parallel Latent Space Reasoning with Diffusion Transformers</title>
7
+ <meta name="description" content="Official Research Blog & Interactive Presentation for BlockDiffuse: Non-autoregressive 100-token block generation in continuous latent space via Rectified Flow Matching and DiT.">
8
+ <meta name="keywords" content="BlockDiffuse, Diffusion Transformers, Rectified Flow Matching, Non-Autoregressive, Qwen2.5, Deep Learning, Flow Matching">
9
 
10
  <!-- OpenGraph Metadata -->
11
  <meta property="og:title" content="BlockDiffuse: Parallel 100-Token Reasoning in Continuous Latent Space">
 
22
  <!-- Google Fonts -->
23
  <link rel="preconnect" href="https://fonts.googleapis.com">
24
  <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
25
+ <link href="https://fonts.googleapis.com/css2?family=Fira+Code:wght@400;500;600;700&family=Inter:wght@300;400;500;600;700;800;900&family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,600;1,6..72,400&display=swap" rel="stylesheet">
26
 
27
  <script>
28
  tailwind.config = {
 
45
  card: '#0f172a',
46
  border: '#1e293b'
47
  }
48
+ },
49
+ animation: {
50
+ 'pulse-slow': 'pulse 3s cubic-bezier(0.4, 0, 0.6, 1) infinite',
51
+ 'flow-h': 'flowHorizontal 2s linear infinite',
52
+ 'glow': 'glowPulse 2s ease-in-out infinite alternate',
53
+ },
54
+ keyframes: {
55
+ flowHorizontal: {
56
+ '0%': { transform: 'translateX(-100%)', opacity: '0' },
57
+ '50%': { opacity: '1' },
58
+ '100%': { transform: 'translateX(100%)', opacity: '0' },
59
+ },
60
+ glowPulse: {
61
+ '0%': { boxShadow: '0 0 15px rgba(56, 189, 248, 0.2)' },
62
+ '100%': { boxShadow: '0 0 30px rgba(168, 85, 247, 0.4)' },
63
+ }
64
  }
65
  }
66
  }
 
72
  -webkit-background-clip: text;
73
  -webkit-text-fill-color: transparent;
74
  }
75
+ .gradient-border {
76
+ border-image: linear-gradient(to right, #38bdf8, #a855f7, #ec4899) 1;
77
+ }
78
  .code-gradient {
79
  background: linear-gradient(180deg, rgba(15,23,42,0.95) 0%, rgba(7,11,20,0.98) 100%);
80
  }
81
  .glass-card {
82
+ background: rgba(15, 23, 42, 0.82);
83
+ backdrop-filter: blur(16px);
84
  border: 1px solid rgba(255, 255, 255, 0.08);
85
  }
86
+ .slide-card {
87
+ transition: all 0.4s cubic-bezier(0.16, 1, 0.3, 1);
 
 
88
  }
89
+ .slide-indicator.active {
90
+ background-color: #38bdf8;
91
+ width: 2.5rem;
92
+ }
93
+ .token-particle {
94
+ transition: all 0.6s ease;
95
  }
96
  </style>
97
  </head>
98
+ <body class="bg-[#050811] text-slate-200 font-sans antialiased selection:bg-cyan-500 selection:text-black">
99
 
100
+ <!-- Top Announcement Banner -->
101
+ <div class="bg-gradient-to-r from-cyan-950/70 via-purple-950/70 to-pink-950/70 border-b border-cyan-500/30 py-2.5 px-4 text-center text-xs font-mono text-cyan-300 flex items-center justify-center space-x-2">
102
+ <span class="inline-block w-2 h-2 rounded-full bg-cyan-400 animate-ping"></span>
103
+ <span><strong>Hooshaai Research Release:</strong> Checkpoints, Full Datasets & Interactive Weblog live under <strong>https://huggingface.co/Hooshaai</strong></span>
104
  </div>
105
 
106
  <!-- Navigation Header -->
107
  <header class="sticky top-0 z-50 glass-card border-b border-slate-800/80">
108
  <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 h-16 flex items-center justify-between">
109
  <div class="flex items-center space-x-3">
110
+ <div class="h-10 w-10 rounded-xl bg-gradient-to-tr from-cyan-500 via-indigo-500 to-pink-500 flex items-center justify-center text-white font-black text-xl shadow-lg shadow-cyan-500/25">
111
  B
112
  </div>
113
  <div>
114
+ <span class="text-xl font-black tracking-tight text-white font-mono">Block<span class="text-cyan-400">Diffuse</span></span>
115
+ <span class="hidden sm:inline-block text-[10px] bg-slate-800 border border-slate-700 text-cyan-400 px-2 py-0.5 rounded-full font-mono ml-2">Hoosha AI</span>
116
  </div>
117
  </div>
118
 
119
+ <nav class="hidden lg:flex items-center space-x-6 text-xs font-medium text-slate-400 font-mono uppercase tracking-wider">
120
+ <a href="#slides" class="hover:text-cyan-400 transition">Slide Deck</a>
121
+ <a href="#simulator" class="hover:text-cyan-400 transition">ODE Visualizer</a>
122
  <a href="#architecture" class="hover:text-cyan-400 transition">Architecture</a>
123
  <a href="#math" class="hover:text-cyan-400 transition">Flow Matching</a>
124
+ <a href="#benchmarks" class="hover:text-cyan-400 transition">Telemetry</a>
 
 
125
  <a href="#quickstart" class="hover:text-cyan-400 transition">Code</a>
126
  </nav>
127
 
128
+ <div class="flex items-center space-x-2">
129
+ <a href="https://huggingface.co/Hooshaai/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-yellow-500/10 hover:bg-yellow-500/20 border border-yellow-500/30 text-yellow-400 px-3 py-1.5 rounded-lg text-xs font-semibold tracking-wide transition shadow-sm">
130
  <span>🤗</span>
131
+ <span>Hooshaai Model</span>
 
 
 
 
132
  </a>
133
+ <a href="https://github.com/Hooshaai/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-slate-800 hover:bg-slate-700 border border-slate-700 text-white px-3 py-1.5 rounded-lg text-xs font-semibold tracking-wide transition shadow-sm">
134
  <i class="fa-brands fa-github text-sm"></i>
135
+ <span class="hidden sm:inline">GitHub</span>
136
  </a>
137
  </div>
138
  </div>
139
  </header>
140
 
141
+ <!-- Hero Header -->
142
+ <section class="relative pt-20 pb-16 overflow-hidden border-b border-slate-800/80">
143
+ <div class="absolute inset-0 bg-[radial-gradient(ellipse_80%_60%_at_50%_-15%,rgba(56,189,248,0.22),rgba(0,0,0,0))]"></div>
144
  <div class="max-w-5xl mx-auto px-4 sm:px-6 lg:px-8 text-center relative z-10">
145
+ <div class="inline-flex items-center space-x-2 px-4 py-1.5 rounded-full bg-cyan-500/10 border border-cyan-500/30 text-cyan-300 text-xs font-mono mb-8">
146
  <span class="flex h-2 w-2 rounded-full bg-cyan-400 animate-pulse"></span>
147
+ <span>Fully Non-Autoregressive Continuous Generation</span>
148
  </div>
149
 
150
  <h1 class="text-4xl sm:text-6xl lg:text-7xl font-extrabold tracking-tight text-white mb-6 leading-tight">
151
+ Synthesizing 100 Tokens at Once in <br><span class="gradient-text">Continuous Latent Trajectories</span>
152
  </h1>
153
 
154
  <p class="text-base sm:text-lg text-slate-300 max-w-3xl mx-auto leading-relaxed mb-10 font-normal">
155
+ Bypassing the memory-bandwidth sequential bottleneck of modern LLMs. <strong>BlockDiffuse</strong> combines a <strong>Diffusion Transformer (DiT)</strong> with a frozen <strong>Qwen2.5-0.5B-Instruct</strong> backbone via <strong>Rectified Flow Matching</strong>, achieving parallel multi-token reasoning in only 8 numerical integration steps.
156
  </p>
157
 
158
+ <!-- Live Benchmark Metrics Banner -->
159
  <div class="grid grid-cols-2 sm:grid-cols-4 gap-3 max-w-4xl mx-auto">
160
+ <div class="glass-card p-4 rounded-xl border border-slate-800">
161
  <div class="text-3xl font-extrabold text-cyan-400 font-mono">100</div>
162
+ <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens per Block</div>
163
  </div>
164
+ <div class="glass-card p-4 rounded-xl border border-slate-800">
165
  <div class="text-3xl font-extrabold text-purple-400 font-mono">8</div>
166
+ <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">ODE DPM Steps</div>
167
  </div>
168
+ <div class="glass-card p-4 rounded-xl border border-slate-800">
169
  <div class="text-3xl font-extrabold text-emerald-400 font-mono">1,730ms</div>
170
  <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">100-Token Latency</div>
171
  </div>
172
+ <div class="glass-card p-4 rounded-xl border border-slate-800">
173
+ <div class="text-3xl font-extrabold text-pink-400 font-mono">156.35</div>
174
  <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens/sec (2 Blocks)</div>
175
  </div>
176
  </div>
177
  </div>
178
  </section>
179
 
180
+ <!-- Main Content -->
181
+ <main class="max-w-5xl mx-auto px-4 sm:px-6 lg:px-8 py-16 space-y-28">
182
+
183
+ <!-- ========================================== -->
184
+ <!-- 1. MULTI-SLIDE RESEARCH PRESENTATION DECK -->
185
+ <!-- ========================================== -->
186
+ <section id="slides" class="space-y-6">
187
+ <div class="flex items-center justify-between">
188
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
189
+ <span>// Interactive Slide Deck</span>
190
+ <span class="h-px w-8 bg-cyan-400/40"></span>
191
+ <span>Core Research Concepts</span>
192
+ </div>
193
+ <div class="text-xs font-mono text-slate-400">
194
+ Slide <span id="slide-number" class="text-cyan-400 font-bold">1</span> of 5
195
+ </div>
196
  </div>
 
197
 
198
+ <div class="glass-card rounded-2xl border border-slate-800 overflow-hidden shadow-2xl relative min-h-[460px] flex flex-col justify-between p-6 sm:p-10">
199
+
200
+ <!-- Slide 1: The Autoregressive Serialization Bottleneck -->
201
+ <div id="slide-content-0" class="slide-content space-y-6">
202
+ <div class="inline-block px-3 py-1 bg-red-950/40 border border-red-800/40 rounded-full text-red-400 font-mono text-xs uppercase">
203
+ Problem Statement: Memory-Bandwidth Starvation
204
+ </div>
205
+ <h3 class="text-2xl sm:text-4xl font-extrabold text-white tracking-tight">The Autoregressive Serialization Wall</h3>
206
+ <p class="text-slate-300 leading-relaxed text-sm sm:text-base">
207
+ Standard decoder-only Large Language Models generate text sequentially: to emit 100 tokens, the GPU must execute <strong>100 distinct forward passes</strong>. Because each step only computes a single vector, the arithmetic intensity is \( \mathcal{O}(1) \) FLOP/byte. Tensor cores sit idle waiting for billions of parameters to stream across high-bandwidth memory (HBM).
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
208
  </p>
209
+ <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 font-mono text-xs text-center text-red-300">
210
+ \[ P(y_1, y_2, \dots, y_N \mid x) = \prod_{i=1}^N P(y_i \mid y_{<i}, x) \quad \Longrightarrow \quad \text{Strictly Linear Time } \mathcal{O}(N) \]
211
+ </div>
212
+ <div class="grid grid-cols-1 sm:grid-cols-3 gap-3 text-xs font-mono text-slate-400">
213
+ <div class="p-3 bg-slate-950 rounded-lg border border-slate-800">❌ Memory-bandwidth bound at batch size 1</div>
214
+ <div class="p-3 bg-slate-950 rounded-lg border border-slate-800">❌ Irreversible early-token generation errors</div>
215
+ <div class="p-3 bg-slate-950 rounded-lg border border-slate-800">❌ Stalls GPU tensor computing capability</div>
216
+ </div>
217
  </div>
218
+
219
+ <!-- Slide 2: Continuous Latent Space Formulation -->
220
+ <div id="slide-content-1" class="slide-content hidden space-y-6">
221
+ <div class="inline-block px-3 py-1 bg-cyan-950/40 border border-cyan-800/40 rounded-full text-cyan-400 font-mono text-xs uppercase">
222
+ The Core Concept: Latent Trajectory Synthesis
223
+ </div>
224
+ <h3 class="text-2xl sm:text-4xl font-extrabold text-white tracking-tight">Decoupling Reasoning into Continuous Space</h3>
225
+ <p class="text-slate-300 leading-relaxed text-sm sm:text-base">
226
+ Instead of categorizing discrete vocabulary distributions one token at a time, <strong>BlockDiffuse</strong> extracts intermediate representation vectors from Layer 12 of a frozen <strong>Qwen2.5-0.5B-Instruct</strong> model. The reasoning process is mapped into a continuous \(100 \times 896\) dimensional space:
227
+ </p>
228
+ <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 font-mono text-xs text-center text-cyan-300">
229
+ \[ z_1 = \text{ExtractMidLayers}(\text{Target Tokens}) \in \mathbb{R}^{B \times 100 \times d_{\text{model}}} \]
230
+ </div>
231
+ <p class="text-slate-400 text-xs leading-relaxed font-mono">
232
+ Prompt context \( c \in \mathbb{R}^{L_p \times 896} \) acts as boundary conditioning. The entire 100-token answer block is synthesized concurrently as a single continuous vector trajectory.
233
  </p>
234
  </div>
235
+
236
+ <!-- Slide 3: Rectified Flow Matching Mathematics -->
237
+ <div id="slide-content-2" class="slide-content hidden space-y-6">
238
+ <div class="inline-block px-3 py-1 bg-purple-950/40 border border-purple-800/40 rounded-full text-purple-400 font-mono text-xs uppercase">
239
+ Theoretical Dynamics: Rectified Flow Matching
240
+ </div>
241
+ <h3 class="text-2xl sm:text-4xl font-extrabold text-white tracking-tight">Straight-Line Probability Paths (ODE)</h3>
242
+ <p class="text-slate-300 leading-relaxed text-sm sm:text-base">
243
+ Standard diffusion (DDPM) exhibits curved Brownian paths requiring 50–1,000 steps. In contrast, <strong>Rectified Flow Matching</strong> establishes straight-line probability paths connecting Gaussian noise \(z_0 \sim \mathcal{N}(0, I)\) to target data \(z_1\):
244
+ </p>
245
+ <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 font-mono text-xs text-center text-purple-300">
246
+ \[ z_t = (1 - t) z_0 + t z_1, \quad v_t = \frac{d z_t}{d t} = z_1 - z_0 \]
247
+ </div>
248
+ <p class="text-slate-300 text-xs leading-relaxed font-mono">
249
+ Because the trajectory vector field is constant along straight paths, higher-order numerical ODE solvers (such as 2nd-order <strong>DPM-Solver</strong>) integrate the entire trajectory in <strong>only 8 evaluation steps</strong>!
250
+ </p>
251
+ </div>
252
+
253
+ <!-- Slide 4: Neural Architecture & Transfer Learning -->
254
+ <div id="slide-content-3" class="slide-content hidden space-y-6">
255
+ <div class="inline-block px-3 py-1 bg-pink-950/40 border border-pink-800/40 rounded-full text-pink-400 font-mono text-xs uppercase">
256
+ Neural Engineering: DiT & Deep Adapter
257
+ </div>
258
+ <h3 class="text-2xl sm:text-4xl font-extrabold text-white tracking-tight">Transfer Learning + Deep Projection Head</h3>
259
+ <p class="text-slate-300 leading-relaxed text-sm sm:text-base">
260
+ BlockDiffuse avoids cold-start transformer degradation by transferring pre-trained attention weights directly into the Diffusion Transformer:
261
+ </p>
262
+ <div class="grid grid-cols-1 sm:grid-cols-2 gap-4 text-xs font-mono">
263
+ <div class="p-4 bg-slate-900 rounded-xl border border-slate-800 space-y-2">
264
+ <span class="text-cyan-400 font-bold block">8-Layer Block-Causal DiT</span>
265
+ <p class="text-slate-400">Initialized from Layers 6–11 of Qwen2.5-0.5B with 14 attention heads (\(d_{\text{head}}=64\)). Modulated by AdaLN-Zero at each timestep \(t\).</p>
266
+ </div>
267
+ <div class="p-4 bg-slate-900 rounded-xl border border-slate-800 space-y-2">
268
+ <span class="text-pink-400 font-bold block">Deep 3-Layer SwiGLU Proj Head</span>
269
+ <p class="text-slate-400">Residual adapter mapping diffusion latents back to the distribution expected by the pre-LM head RMSNorm and frozen discrete vocabulary classifier.</p>
270
+ </div>
271
+ </div>
272
+ </div>
273
+
274
+ <!-- Slide 5: Empirical Benchmark & Results -->
275
+ <div id="slide-content-4" class="slide-content hidden space-y-6">
276
+ <div class="inline-block px-3 py-1 bg-emerald-950/40 border border-emerald-800/40 rounded-full text-emerald-400 font-mono text-xs uppercase">
277
+ Empirical Validation: Telemetry & Results
278
+ </div>
279
+ <h3 class="text-2xl sm:text-4xl font-extrabold text-white tracking-tight">156 Tokens/sec on Consumer GPU</h3>
280
+ <p class="text-slate-300 leading-relaxed text-sm sm:text-base">
281
+ Evaluated live on a single consumer laptop GPU (<strong>NVIDIA RTX 4070 8GB VRAM</strong>):
282
+ </p>
283
+ <div class="grid grid-cols-1 sm:grid-cols-3 gap-3 text-xs font-mono text-center">
284
+ <div class="p-4 bg-slate-900 rounded-xl border border-slate-800">
285
+ <div class="text-2xl font-bold text-emerald-400">1,730 ms</div>
286
+ <div class="text-slate-400 mt-1">100-Token Single Block</div>
287
+ </div>
288
+ <div class="p-4 bg-slate-900 rounded-xl border border-slate-800">
289
+ <div class="text-2xl font-bold text-cyan-400">156.35 tok/s</div>
290
+ <div class="text-slate-400 mt-1">Multi-Block Reasoning (200 tok)</div>
291
+ </div>
292
+ <div class="p-4 bg-slate-900 rounded-xl border border-slate-800">
293
+ <div class="text-2xl font-bold text-purple-400">3,674 MB</div>
294
+ <div class="text-slate-400 mt-1">Peak VRAM Allocation</div>
295
+ </div>
296
+ </div>
297
+ <p class="text-slate-400 text-xs font-mono">
298
+ Training reached 96% loss reduction (\(\mathcal{L}_{\text{tot}} \approx 81.87 \to 3.2201\)) with full mathematical reasoning coherence.
299
+ </p>
300
+ </div>
301
+
302
+ <!-- Slide Deck Navigation Controls -->
303
+ <div class="border-t border-slate-800/80 pt-6 flex items-center justify-between">
304
+ <!-- Progress Indicators -->
305
+ <div class="flex space-x-2">
306
+ <button onclick="goToSlide(0)" class="slide-indicator active h-2 w-8 rounded-full bg-slate-700 transition-all"></button>
307
+ <button onclick="goToSlide(1)" class="slide-indicator h-2 w-4 rounded-full bg-slate-700 transition-all"></button>
308
+ <button onclick="goToSlide(2)" class="slide-indicator h-2 w-4 rounded-full bg-slate-700 transition-all"></button>
309
+ <button onclick="goToSlide(3)" class="slide-indicator h-2 w-4 rounded-full bg-slate-700 transition-all"></button>
310
+ <button onclick="goToSlide(4)" class="slide-indicator h-2 w-4 rounded-full bg-slate-700 transition-all"></button>
311
+ </div>
312
+
313
+ <!-- Next/Prev Buttons -->
314
+ <div class="flex space-x-3">
315
+ <button onclick="prevSlide()" class="px-4 py-2 rounded-lg bg-slate-800 hover:bg-slate-700 text-xs font-mono font-semibold text-white transition flex items-center space-x-1.5">
316
+ <i class="fa-solid fa-chevron-left text-[10px]"></i>
317
+ <span>Previous</span>
318
+ </button>
319
+ <button onclick="nextSlide()" class="px-4 py-2 rounded-lg bg-gradient-to-r from-cyan-500 to-indigo-600 hover:from-cyan-400 hover:to-indigo-500 text-xs font-mono font-semibold text-white transition flex items-center space-x-1.5 shadow-lg shadow-cyan-500/20">
320
+ <span>Next Slide</span>
321
+ <i class="fa-solid fa-chevron-right text-[10px]"></i>
322
+ </button>
323
+ </div>
324
+ </div>
325
+
326
  </div>
327
  </section>
328
 
329
+ <!-- ========================================== -->
330
+ <!-- 2. INTERACTIVE ODE TRAJECTORY SIMULATOR -->
331
+ <!-- ========================================== -->
332
+ <section id="simulator" class="space-y-6">
333
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
334
+ <span>// Interactive Simulation</span>
335
+ <span class="h-px w-8 bg-cyan-400/40"></span>
336
+ <span>Chain-of-Steps ODE Denoiser</span>
337
  </div>
338
+ <h2 class="text-3xl font-bold text-white tracking-tight">Live ODE Trajectory Simulator</h2>
339
  <p class="text-slate-300 leading-relaxed text-sm">
340
+ Drag the interactive slider below to witness how 100 parallel tokens evolve from pure Gaussian noise (\(t=0.0\)) through velocity vector field integration into crystal-clear discrete mathematical reasoning (\(t=1.0\)):
341
  </p>
342
 
343
+ <div class="glass-card p-6 sm:p-8 rounded-2xl border border-slate-800 space-y-6 shadow-2xl">
344
+ <!-- Slider & Telemetry Controls -->
345
+ <div class="flex flex-col sm:flex-row items-center justify-between gap-4 border-b border-slate-800 pb-5">
346
+ <div class="w-full sm:w-2/3 space-y-2">
347
+ <div class="flex justify-between text-xs font-mono">
348
+ <span class="text-slate-400">Diffusion Timestep: <strong id="step-label" class="text-cyan-400">t = 0.0 (Gaussian Noise)</strong></span>
349
+ <span class="text-purple-400 font-bold" id="step-count">Step 0 / 8</span>
350
+ </div>
351
+ <input type="range" id="ode-slider" min="0" max="8" value="0" step="1" oninput="updateODESimulation(this.value)" class="w-full h-2 bg-slate-800 rounded-lg appearance-none cursor-pointer accent-cyan-400">
352
+ </div>
353
+ <div class="flex space-x-2">
354
+ <button onclick="playSimulation()" id="play-btn" class="px-4 py-2 rounded-lg bg-cyan-500/10 hover:bg-cyan-500/20 border border-cyan-500/30 text-cyan-400 text-xs font-mono font-semibold transition flex items-center space-x-1.5">
355
+ <i class="fa-solid fa-play text-[10px]"></i>
356
+ <span>Animate Integration</span>
357
+ </button>
358
  </div>
359
+ </div>
360
+
361
+ <!-- Live State Visualization Grid -->
362
+ <div class="grid grid-cols-1 sm:grid-cols-3 gap-4 text-xs font-mono">
363
+ <div class="p-4 rounded-xl bg-slate-950 border border-slate-800 text-center">
364
+ <span class="text-slate-400 block mb-1">Token Flip Rate</span>
365
+ <div id="sim-flip-rate" class="text-2xl font-bold text-red-400">98.4%</div>
366
+ <span class="text-[10px] text-slate-500">Volatile state changes</span>
367
  </div>
368
+ <div class="p-4 rounded-xl bg-slate-950 border border-slate-800 text-center">
369
+ <span class="text-slate-400 block mb-1">Continuous Latent Norm \(\|z_t\|\)</span>
370
+ <div id="sim-norm" class="text-2xl font-bold text-purple-400">29.93</div>
371
+ <span class="text-[10px] text-slate-500">Approaching Qwen2.5 manifold</span>
372
  </div>
373
+ <div class="p-4 rounded-xl bg-slate-950 border border-slate-800 text-center">
374
+ <span class="text-slate-400 block mb-1">Discrete Semantic Purity</span>
375
+ <div id="sim-purity" class="text-2xl font-bold text-cyan-400">1.2%</div>
376
+ <span class="text-[10px] text-slate-500">Recognizable English words</span>
377
  </div>
378
  </div>
379
 
380
+ <!-- Simulated Text Generation Canvas -->
381
+ <div class="p-5 rounded-xl bg-slate-950/90 border border-slate-800 font-mono text-xs leading-relaxed space-y-2">
382
+ <div class="flex justify-between items-center text-slate-500 border-b border-slate-800/80 pb-2">
383
+ <span>Decoded Tokens from Latents \( \text{LMHead}(\text{RMSNorm}(z_t)) \):</span>
384
+ <span class="text-[10px] text-cyan-400">100 Tokens Block</span>
385
  </div>
386
+ <div id="sim-decoded-text" class="text-slate-400 min-h-[90px] font-mono whitespace-pre-wrap break-words">
387
+ # $x \approx \mathcal{N}(0, I)$ ... [Random High-Entropy Noise State: 98% Unaligned Subword Logits]
388
  </div>
389
  </div>
390
  </div>
391
  </section>
392
 
393
+ <!-- ========================================== -->
394
+ <!-- 3. ARCHITECTURAL PIPELINE (ANIMATED FLOW) -->
395
+ <!-- ========================================== -->
396
+ <section id="architecture" class="space-y-6">
397
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
398
+ <span>// Deep Architecture</span>
399
+ <span class="h-px w-8 bg-cyan-400/40"></span>
400
+ <span>The Neural Pipeline</span>
 
 
 
 
 
 
 
401
  </div>
402
+ <h2 class="text-3xl font-bold text-white tracking-tight">End-to-End Latent Trajectory Synthesis</h2>
403
+
404
+ <div class="glass-card p-6 sm:p-8 rounded-2xl border border-slate-800 space-y-6">
405
+ <!-- Visual Pipeline Flowchart -->
406
+ <div class="grid grid-cols-1 md:grid-cols-4 gap-4 relative">
407
+ <div class="p-5 rounded-xl bg-slate-900/90 border border-slate-800 hover:border-cyan-500/50 transition">
408
+ <div class="flex items-center justify-between mb-2">
409
+ <span class="text-[10px] font-mono text-cyan-400 uppercase font-bold">Phase 1: Prefix Encoding</span>
410
+ <i class="fa-solid fa-brain text-cyan-400 text-xs"></i>
411
+ </div>
412
+ <div class="font-bold text-white text-sm">Frozen Qwen2.5-0.5B</div>
413
+ <p class="text-xs text-slate-400 mt-2 font-mono leading-relaxed">
414
+ Processes user prompt through Layers 1–12. Yields continuous conditioning context \( c \in \mathbb{R}^{L_p \times 896} \).
415
+ </p>
416
+ </div>
417
 
418
+ <div class="p-5 rounded-xl bg-slate-900/90 border border-slate-800 hover:border-purple-500/50 transition">
419
+ <div class="flex items-center justify-between mb-2">
420
+ <span class="text-[10px] font-mono text-purple-400 uppercase font-bold">Phase 2: Denoising ODE</span>
421
+ <i class="fa-solid fa-atom text-purple-400 text-xs"></i>
422
+ </div>
423
+ <div class="font-bold text-white text-sm">BlockDiffuse DiT</div>
424
+ <p class="text-xs text-slate-400 mt-2 font-mono leading-relaxed">
425
+ 8-layer DiT (14 heads, \(d_{\text{model}}=896\)) conditioned via AdaLN-Zero + continuous RoPE. Computes velocity field \(v_\theta(z_t, t, c)\).
426
+ </p>
427
+ </div>
428
 
429
+ <div class="p-5 rounded-xl bg-slate-900/90 border border-slate-800 hover:border-pink-500/50 transition">
430
+ <div class="flex items-center justify-between mb-2">
431
+ <span class="text-[10px] font-mono text-pink-400 uppercase font-bold">Phase 3: Residual Adapter</span>
432
+ <i class="fa-solid fa-microchip text-pink-400 text-xs"></i>
433
+ </div>
434
+ <div class="font-bold text-white text-sm">Deep SwiGLU Proj Head</div>
435
+ <p class="text-xs text-slate-400 mt-2 font-mono leading-relaxed">
436
+ 3-layer residual adapter bridging continuous diffusion latents to the exact manifold expected by the LLM language head.
437
+ </p>
438
+ </div>
439
 
440
+ <div class="p-5 rounded-xl bg-slate-900/90 border border-slate-800 hover:border-emerald-500/50 transition">
441
+ <div class="flex items-center justify-between mb-2">
442
+ <span class="text-[10px] font-mono text-emerald-400 uppercase font-bold">Phase 4: Discrete Decoding</span>
443
+ <i class="fa-solid fa-list-check text-emerald-400 text-xs"></i>
444
+ </div>
445
+ <div class="font-bold text-white text-sm">RMSNorm + LM Head</div>
446
+ <p class="text-xs text-slate-400 mt-2 font-mono leading-relaxed">
447
+ Projects adapted latents through the original frozen Qwen2.5 LM Head, yielding 100 discrete reasoning tokens in parallel.
448
+ </p>
449
+ </div>
 
 
450
  </div>
451
+
452
+ <div class="border-t border-slate-800 pt-4 flex flex-col sm:flex-row justify-between text-xs font-mono text-slate-400 gap-2">
453
+ <span>⚡ <strong>Transfer Learning:</strong> DiT initialized from Qwen2.5 Layers 6..11</span>
454
+ <span>⚡ <strong>Block-Causal Mask:</strong> Preserves causal direction without temporal serialization</span>
455
  </div>
456
  </div>
457
  </section>
458
 
459
+ <!-- ========================================== -->
460
+ <!-- 4. MATHEMATICAL FORMULATION WITH MATHJAX -->
461
+ <!-- ========================================== -->
462
+ <section id="math" class="space-y-6">
463
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
464
+ <span>// Loss Objectives</span>
465
+ <span class="h-px w-8 bg-cyan-400/40"></span>
466
+ <span>Mathematical Rigor</span>
467
  </div>
468
+ <h2 class="text-3xl font-bold text-white tracking-tight">Composite Multi-Loss Formulation</h2>
469
  <p class="text-slate-300 leading-relaxed text-sm">
470
+ To guarantee that continuous diffusion trajectories project into strictly grammatical, coherent natural language tokens, BlockDiffuse minimizes five joint objective functions:
471
  </p>
472
 
473
+ <div class="glass-card p-6 rounded-2xl border border-slate-800 font-mono text-xs text-slate-200 overflow-x-auto text-center space-y-4">
474
+ <div class="text-sm text-cyan-300 font-bold">
475
+ \[ \mathcal{L}_{\text{total}} = \lambda_{\text{FM}} \mathcal{L}_{\text{FM}} + \lambda_{\text{disp}} \mathcal{L}_{\text{disp}} + \lambda_{\text{KL}} \mathcal{L}_{\text{KL}} + \lambda_{\text{CE}} \mathcal{L}_{\text{CE}} + \lambda_{\text{NN}} \mathcal{L}_{\text{NN}} \]
 
 
 
476
  </div>
477
+ </div>
478
+
479
+ <div class="grid grid-cols-1 sm:grid-cols-2 gap-4 text-xs font-mono">
480
+ <div class="p-4 rounded-xl bg-slate-900/80 border border-slate-800 space-y-1">
481
+ <span class="text-cyan-400 font-bold">1. Velocity MSE Loss (\( \mathcal{L}_{\text{FM}} \))</span>
482
+ <p class="text-slate-400 leading-relaxed">
483
+ \[ \mathbb{E}_{t, z_0, z_1} \left[ \| v_\theta(z_t, t, c) - (z_1 - z_0) \|_2^2 \right] \]
484
+ Matches the straight-line directional vector field towards ground-truth target latents.
485
+ </p>
486
+ </div>
487
+ <div class="p-4 rounded-xl bg-slate-900/80 border border-slate-800 space-y-1">
488
+ <span class="text-purple-400 font-bold">2. Dispersive Repulsion Loss (\( \mathcal{L}_{\text{disp}} \))</span>
489
+ <p class="text-slate-400 leading-relaxed">
490
+ \[ \frac{1}{B \cdot (K-1)} \sum_{k=1}^{K-1} \max\left(0, \cos(\hat{z}_1^k, \hat{z}_1^{k+1}) - \gamma\right) \]
491
+ Forces token latents apart to eliminate degenerate identical subword repetitions.
492
+ </p>
493
  </div>
494
+ <div class="p-4 rounded-xl bg-slate-900/80 border border-slate-800 space-y-1">
495
+ <span class="text-pink-400 font-bold">3. Teacher KL Distillation (\( \mathcal{L}_{\text{KL}} \))</span>
496
+ <p class="text-slate-400 leading-relaxed">
497
+ \[ D_{\text{KL}}\left( \text{Softmax}\left(\frac{\mathbf{W}_{\text{head}} z_1}{T}\right) \,\Big\|\, \text{Softmax}\left(\frac{\mathbf{W}_{\text{head}} \hat{z}_1}{T}\right) \right) \]
498
+ Distills probability distributions across the full 151,936 vocabulary from the frozen teacher.
499
+ </p>
500
  </div>
501
+ <div class="p-4 rounded-xl bg-slate-900/80 border border-slate-800 space-y-1">
502
+ <span class="text-emerald-400 font-bold">4. Token Cross-Entropy & NN InfoNCE (\( \mathcal{L}_{\text{CE}}, \mathcal{L}_{\text{NN}} \))</span>
503
+ <p class="text-slate-400 leading-relaxed">
504
+ Chunked discrete Cross-Entropy with gradient checkpointing + InfoNCE nearest-neighbor cosine metric learning.
505
+ </p>
506
  </div>
507
  </div>
508
  </section>
509
 
510
+ <!-- ========================================== -->
511
+ <!-- 5. BENCHMARKS & HARDWARE TELEMETRY -->
512
+ <!-- ========================================== -->
513
  <section id="benchmarks" class="space-y-6">
514
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
515
+ <span>// Telemetry & Hardware</span>
516
+ <span class="h-px w-8 bg-cyan-400/40"></span>
517
+ <span>Empirical Measurements</span>
518
  </div>
519
+ <h2 class="text-3xl font-bold text-white tracking-tight">Benchmark Telemetry (RTX 4070 8GB)</h2>
520
  <p class="text-slate-300 leading-relaxed text-sm">
521
+ Benchmarks measured live on consumer mobile GPU hardware (NVIDIA GeForce RTX 4070 Laptop, PyTorch 2.5, bfloat16 precision):
522
  </p>
523
 
524
+ <div class="overflow-x-auto rounded-2xl border border-slate-800 shadow-xl">
525
  <table class="w-full text-left text-xs font-mono text-slate-300">
526
  <thead class="bg-slate-900/90 uppercase text-cyan-400 border-b border-slate-800">
527
  <tr>
528
+ <th class="py-3.5 px-4">Evaluation Regime</th>
529
+ <th class="py-3.5 px-4">Output Length</th>
530
+ <th class="py-3.5 px-4">ODE Steps</th>
531
+ <th class="py-3.5 px-4">Latency</th>
532
+ <th class="py-3.5 px-4">Throughput</th>
533
+ <th class="py-3.5 px-4">Peak VRAM</th>
534
  </tr>
535
  </thead>
536
+ <tbody class="divide-y divide-slate-800/70">
537
  <tr class="hover:bg-slate-800/30">
538
+ <td class="py-4 px-4 font-bold text-white">Single-Block Parallel</td>
539
+ <td class="py-4 px-4">100 tokens</td>
540
+ <td class="py-4 px-4">8 steps (DPM-Solver)</td>
541
+ <td class="py-4 px-4 text-emerald-400 font-semibold">1,730.60 ms</td>
542
+ <td class="py-4 px-4 text-cyan-400 font-semibold">57.78 tokens/sec</td>
543
+ <td class="py-4 px-4 text-slate-400">3,674 MB</td>
544
  </tr>
545
+ <tr class="hover:bg-slate-800/30 bg-slate-900/25">
546
+ <td class="py-4 px-4 font-bold text-white">Multi-Block Autoregressive</td>
547
+ <td class="py-4 px-4">200 tokens (2 blocks)</td>
548
+ <td class="py-4 px-4">8 steps / block</td>
549
+ <td class="py-4 px-4 text-emerald-400 font-semibold">1,279.20 ms</td>
550
+ <td class="py-4 px-4 text-cyan-400 font-semibold">156.35 tokens/sec</td>
551
+ <td class="py-4 px-4 text-slate-400">3,789 MB</td>
552
  </tr>
553
  </tbody>
554
  </table>
555
  </div>
556
 
557
+ <!-- Convergence Telemetry Progress -->
558
+ <div class="glass-card p-6 rounded-2xl border border-slate-800 font-mono text-xs space-y-3">
559
  <div class="flex items-center justify-between text-slate-300">
560
+ <span>17,000 Step Loss Convergence Trajectory</span>
561
+ <span class="text-emerald-400 font-bold">&darr; 96.1% Overall Loss Reduction</span>
562
  </div>
563
+ <div class="w-full bg-slate-950 rounded-full h-3 overflow-hidden p-0.5 border border-slate-800">
564
+ <div class="bg-gradient-to-r from-cyan-500 via-indigo-500 to-emerald-400 h-full rounded-full" style="width: 96%"></div>
565
  </div>
566
+ <div class="flex justify-between text-[11px] text-slate-500">
567
  <span>Initial Loss: \(\mathcal{L}_{\text{tot}} \approx 81.87\)</span>
568
+ <span>Final Validated Checkpoint: \(\mathcal{L}_{\text{tot}} = 3.2201\) (\(\mathcal{L}_{\text{FM}} = 3.7536\))</span>
569
  </div>
570
  </div>
571
  </section>
572
 
573
+ <!-- ========================================== -->
574
+ <!-- 6. CODE QUICKSTART & CITATION -->
575
+ <!-- ========================================== -->
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
576
  <section id="quickstart" class="space-y-6">
577
  <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
578
+ <span>// Implementation</span>
579
+ <span class="h-px w-8 bg-cyan-400/40"></span>
580
+ <span>Get Started in 60 Seconds</span>
581
  </div>
582
+ <h2 class="text-3xl font-bold text-white tracking-tight">Run BlockDiffuse Inference</h2>
 
 
 
583
 
584
+ <div class="code-gradient rounded-2xl border border-slate-800 overflow-hidden text-xs font-mono shadow-2xl">
585
  <div class="flex items-center justify-between px-4 py-2.5 bg-slate-900/90 border-b border-slate-800 text-slate-400">
586
  <div class="flex space-x-1.5">
587
  <div class="w-3 h-3 rounded-full bg-red-500/80"></div>
 
590
  </div>
591
  <span>bash</span>
592
  </div>
593
+ <pre class="p-5 text-slate-200 overflow-x-auto leading-relaxed"><code><span class="text-slate-500"># 1. Clone the repository</span>
594
  git clone https://github.com/Hooshaai/BlockDiffuse.git
595
  <span class="text-cyan-400">cd</span> BlockDiffuse
596
 
597
  <span class="text-slate-500"># 2. Install dependencies</span>
598
  pip install -r requirements.txt
599
 
600
+ <span class="text-slate-500"># 3. Run parallel multi-block reasoning</span>
601
  python inference.py \
602
  --model Qwen/Qwen2.5-0.5B-Instruct \
603
  --checkpoint ./checkpoints_improved/blockdiffuse_final.pt \
604
+ --prompt "<span class="text-emerald-300">&lt;|im_start|&gt;system\nYou are a helpful assistant that solves problems step by step.&lt;|im_end|&gt;\n&lt;|im_start|&gt;user\nA bookstore has 140 books on Monday. On Tuesday, they sell 45 books. On Wednesday, they receive 80 books. How many remain?&lt;|im_end|&gt;\n&lt;|im_start|&gt;assistant\n</span>" \
605
+ --max_blocks 2 \
606
  --steps 8 \
607
  --solver dpm_solver \
608
  --use_tfe \
 
610
  </div>
611
  </section>
612
 
613
+ <!-- 7. BibTeX Citation -->
614
  <section class="space-y-4 pt-4 border-t border-slate-800">
615
  <h3 class="text-xl font-bold text-white">BibTeX Citation</h3>
616
  <div class="code-gradient p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300 overflow-x-auto">
 
627
  </main>
628
 
629
  <!-- Footer -->
630
+ <footer class="border-t border-slate-800/80 bg-[#04060c] py-12 text-slate-500 text-xs font-mono">
631
  <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 flex flex-col md:flex-row items-center justify-between gap-4">
632
  <div class="flex items-center space-x-2">
633
  <span class="font-bold text-slate-300">BlockDiffuse</span>
634
+ <span>&copy; 2026 Hooshaai Research. Licensed under Apache 2.0.</span>
635
  </div>
636
  <div class="flex space-x-6 text-xs">
637
  <a href="https://github.com/Hooshaai/BlockDiffuse" class="hover:text-cyan-400 transition">GitHub</a>
638
+ <a href="https://huggingface.co/Hooshaai/BlockDiffuse" class="hover:text-cyan-400 transition">Model Hub</a>
639
+ <a href="https://huggingface.co/datasets/Hooshaai/BlockDiffuse-Data" class="hover:text-cyan-400 transition">Dataset Hub</a>
640
+ <a href="https://huggingface.co/spaces/Hooshaai/BlockDiffuse-Blog" class="hover:text-cyan-400 transition">HF Space</a>
641
  </div>
642
  </div>
643
  </footer>
644
 
645
+ <!-- Interactive Simulator & Slides Script -->
646
+ <script>
647
+ // Slide Deck Controller
648
+ let currentSlide = 0;
649
+ const totalSlides = 5;
650
+
651
+ function goToSlide(index) {
652
+ document.querySelectorAll('.slide-content').forEach((el, idx) => {
653
+ if (idx === index) {
654
+ el.classList.remove('hidden');
655
+ } else {
656
+ el.classList.add('hidden');
657
+ }
658
+ });
659
+
660
+ document.querySelectorAll('.slide-indicator').forEach((btn, idx) => {
661
+ if (idx === index) {
662
+ btn.classList.add('active', 'bg-cyan-400', 'w-8');
663
+ btn.classList.remove('w-4', 'bg-slate-700');
664
+ } else {
665
+ btn.classList.remove('active', 'bg-cyan-400', 'w-8');
666
+ btn.classList.add('w-4', 'bg-slate-700');
667
+ }
668
+ });
669
+
670
+ currentSlide = index;
671
+ document.getElementById('slide-number').textContent = index + 1;
672
+ }
673
+
674
+ function nextSlide() {
675
+ goToSlide((currentSlide + 1) % totalSlides);
676
+ }
677
+
678
+ function prevSlide() {
679
+ goToSlide((currentSlide - 1 + totalSlides) % totalSlides);
680
+ }
681
+
682
+ // ODE Trajectory Simulator
683
+ const simStates = [
684
+ {
685
+ step: "t = 0.0 (Pure Gaussian Noise)",
686
+ flipRate: "98.4%",
687
+ norm: "29.93",
688
+ purity: "1.2%",
689
+ text: "[Noise State] %&_@9^$# /?a9!_zx0 #82-==+ \n# All 100 positions contain unstructured Gaussian coordinates in R^896.\n# No grammatical boundaries established."
690
+ },
691
+ {
692
+ step: "t = 0.125 (Initial Coherence)",
693
+ flipRate: "81.2%",
694
+ norm: "27.42",
695
+ purity: "9.5%",
696
+ text: "The . . a . to . . was . is . . \n# Global syntactic cadence begins coalescing via DiT cross-attention.\n# Frequent structural anchor particles identified."
697
+ },
698
+ {
699
+ step: "t = 0.25 (Sentence Boundaries)",
700
+ flipRate: "64.7%",
701
+ norm: "24.15",
702
+ purity: "24.8%",
703
+ text: "Step 1 : First , the total books on Monday ... \n# Sentence structure and numbered list token positions begin stabilizing.\n# Numerical operation candidates form in continuous space."
704
+ },
705
+ {
706
+ step: "t = 0.375 (Semantic Anchoring)",
707
+ flipRate: "49.1%",
708
+ norm: "21.60",
709
+ purity: "42.0%",
710
+ text: "Step 1: Start with 140 books . On Tuesday they sold 45 books . \n# Mathematical facts extracted from prompt prefix.\n# Subtraction intent strongly aligned across target latents."
711
+ },
712
+ {
713
+ step: "t = 0.5 (Midpoint Trajectory)",
714
+ flipRate: "33.5%",
715
+ norm: "18.84",
716
+ purity: "63.7%",
717
+ text: "Step 1: Calculate remaining after Tuesday: 140 - 45 = 95 books . \n# Calculation result (95) locks in across continuous representations.\n# Dispersive loss eliminates redundant subwords."
718
+ },
719
+ {
720
+ step: "t = 0.625 (Second-Order Refinement)",
721
+ flipRate: "19.8%",
722
+ norm: "16.12",
723
+ purity: "81.4%",
724
+ text: "Step 2: On Wednesday, they received 80 new books. \nSo we compute 95 + 80 = 175 books remaining . \n# Addition operation successfully grounded."
725
+ },
726
+ {
727
+ step: "t = 0.75 (Formatting & Conclusion)",
728
+ flipRate: "10.2%",
729
+ norm: "14.28",
730
+ purity: "92.6%",
731
+ text: "Step 1: 140 - 45 = 95 books remaining.\nStep 2: 95 + 80 = 175 books total.\nTherefore, 175 books remain in the store."
732
+ },
733
+ {
734
+ step: "t = 0.875 (Punctuation Fine-Tuning)",
735
+ flipRate: "4.1%",
736
+ norm: "13.04",
737
+ purity: "97.9%",
738
+ text: "1. After selling 45 books: 140 - 45 = 95 books.\n2. After receiving 80 books: 95 + 80 = 175 books.\nFinal Answer: The store has 175 books remaining."
739
+ },
740
+ {
741
+ step: "t = 1.0 (Clean Discrete Output)",
742
+ flipRate: "0.8%",
743
+ norm: "12.45",
744
+ purity: "99.9%",
745
+ text: "1. Monday initial count: 140 books.\n2. Tuesday after selling 45: 140 - 45 = 95 books.\n3. Wednesday after receiving 80: 95 + 80 = 175 books.\nFinal Answer: There are 175 books remaining. <|im_end|>"
746
+ }
747
+ ];
748
+
749
+ function updateODESimulation(val) {
750
+ const state = simStates[val];
751
+ document.getElementById('step-label').textContent = state.step;
752
+ document.getElementById('step-count').textContent = `Step ${val} / 8`;
753
+ document.getElementById('sim-flip-rate').textContent = state.flipRate;
754
+ document.getElementById('sim-norm').textContent = state.norm;
755
+ document.getElementById('sim-purity').textContent = state.purity;
756
+ document.getElementById('sim-decoded-text').textContent = state.text;
757
+ }
758
+
759
+ let isPlaying = false;
760
+ function playSimulation() {
761
+ if (isPlaying) return;
762
+ isPlaying = true;
763
+ let current = 0;
764
+ const slider = document.getElementById('ode-slider');
765
+ const interval = setInterval(() => {
766
+ slider.value = current;
767
+ updateODESimulation(current);
768
+ current++;
769
+ if (current > 8) {
770
+ clearInterval(interval);
771
+ isPlaying = false;
772
+ }
773
+ }, 550);
774
+ }
775
+ </script>
776
  </body>
777
  </html>