BlockDiffuse-Blog / index.html
tahamajs's picture
Deploy fully comprehensive research blog to HF Space
b8ad34e verified
Raw
History Blame Contribute Delete
33.8 kB
<!DOCTYPE html>
<html lang="en" class="scroll-smooth">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>BlockDiffuse: Fully Parallel Latent Space Reasoning with Diffusion Transformers</title>
<meta name="description" content="Official Research Blog & Technical Report for BlockDiffuse: Non-autoregressive 100-token block generation in continuous latent space using Rectified Flow Matching and DiT.">
<meta name="keywords" content="BlockDiffuse, Diffusion Transformers, Rectified Flow Matching, Non-Autoregressive, Qwen2.5, Deep Learning, Chain-of-Thought">
<!-- OpenGraph Metadata -->
<meta property="og:title" content="BlockDiffuse: Parallel 100-Token Reasoning in Continuous Latent Space">
<meta property="og:description" content="Synthesizing 100 tokens simultaneously in 8 ODE integration steps via Diffusion Transformers and frozen LLM latent conditioning.">
<meta property="og:type" content="article">
<!-- Tailwind CSS CDN -->
<script src="https://cdn.tailwindcss.com"></script>
<!-- MathJax for TeX equations -->
<script src="https://polyfill.io/v3/polyfill.min.js?features=es6"></script>
<script id="MathJax-script" async src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js"></script>
<!-- Font Awesome Icons -->
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/6.4.0/css/all.min.css">
<!-- Google Fonts -->
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=Fira+Code:wght@400;500;600;700&family=Inter:wght@300;400;500;600;700;800&family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,600;1,6..72,400&display=swap" rel="stylesheet">
<script>
tailwind.config = {
darkMode: 'class',
theme: {
extend: {
fontFamily: {
sans: ['Inter', 'sans-serif'],
serif: ['Newsreader', 'serif'],
mono: ['Fira Code', 'monospace'],
},
colors: {
brand: {
cyan: '#38bdf8',
purple: '#a855f7',
pink: '#ec4899',
emerald: '#10b981',
amber: '#f59e0b',
dark: '#070b14',
card: '#0f172a',
border: '#1e293b'
}
}
}
}
}
</script>
<style>
.gradient-text {
background: linear-gradient(135deg, #38bdf8 0%, #a855f7 50%, #ec4899 100%);
-webkit-background-clip: text;
-webkit-text-fill-color: transparent;
}
.code-gradient {
background: linear-gradient(180deg, rgba(15,23,42,0.95) 0%, rgba(7,11,20,0.98) 100%);
}
.glass-card {
background: rgba(15, 23, 42, 0.78);
backdrop-filter: blur(14px);
border: 1px solid rgba(255, 255, 255, 0.08);
}
.glass-card-hover:hover {
border-color: rgba(56, 189, 248, 0.35);
transform: translateY(-2px);
transition: all 0.25s ease-in-out;
}
.tab-active {
border-color: #38bdf8;
color: #38bdf8;
background-color: rgba(56, 189, 248, 0.1);
}
</style>
</head>
<body class="bg-[#060911] text-slate-200 font-sans antialiased selection:bg-cyan-500 selection:text-black">
<!-- Top Alert Banner -->
<div class="bg-gradient-to-r from-cyan-950/60 via-purple-950/60 to-pink-950/60 border-b border-cyan-500/20 py-2 px-4 text-center text-xs font-mono text-cyan-300">
🎉 <strong>Research Release:</strong> Checkpoint weights, datasets, and code are now public on Hugging Face & GitHub!
</div>
<!-- Navigation Header -->
<header class="sticky top-0 z-50 glass-card border-b border-slate-800/80">
<div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 h-16 flex items-center justify-between">
<div class="flex items-center space-x-3">
<div class="h-9 w-9 rounded-lg bg-gradient-to-tr from-cyan-500 via-indigo-500 to-pink-500 flex items-center justify-center text-white font-black text-xl shadow-lg shadow-cyan-500/20">
B
</div>
<div>
<span class="text-xl font-bold tracking-tight text-white font-mono">Block<span class="text-cyan-400">Diffuse</span></span>
<span class="hidden sm:inline-block text-[10px] bg-slate-800 border border-slate-700 text-cyan-400 px-2 py-0.5 rounded-full font-mono ml-2">v1.0-Paper</span>
</div>
</div>
<nav class="hidden lg:flex items-center space-x-7 text-xs font-medium text-slate-400 font-mono uppercase tracking-wider">
<a href="#abstract" class="hover:text-cyan-400 transition">Abstract</a>
<a href="#motivation" class="hover:text-cyan-400 transition">Motivation</a>
<a href="#architecture" class="hover:text-cyan-400 transition">Architecture</a>
<a href="#math" class="hover:text-cyan-400 transition">Flow Matching</a>
<a href="#trajectory" class="hover:text-cyan-400 transition">Trajectory</a>
<a href="#benchmarks" class="hover:text-cyan-400 transition">Benchmarks</a>
<a href="#examples" class="hover:text-cyan-400 transition">Generations</a>
<a href="#quickstart" class="hover:text-cyan-400 transition">Code</a>
</nav>
<div class="flex items-center space-x-2.5">
<a href="https://huggingface.co/tahamajs/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-yellow-500/10 hover:bg-yellow-500/20 border border-yellow-500/30 text-yellow-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
<span>🤗</span>
<span class="hidden sm:inline">Model</span>
</a>
<a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" target="_blank" class="flex items-center space-x-1.5 bg-cyan-500/10 hover:bg-cyan-500/20 border border-cyan-500/30 text-cyan-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
<i class="fa-solid fa-database text-xs"></i>
<span class="hidden sm:inline">Data</span>
</a>
<a href="https://github.com/Hooshaai/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-slate-800 hover:bg-slate-700 border border-slate-700 text-white px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
<i class="fa-brands fa-github text-sm"></i>
<span class="hidden sm:inline">Code</span>
</a>
</div>
</div>
</header>
<!-- Hero Section -->
<section class="relative pt-20 pb-20 overflow-hidden border-b border-slate-800/80">
<div class="absolute inset-0 bg-[radial-gradient(ellipse_75%_50%_at_50%_-15%,rgba(56,189,248,0.18),rgba(0,0,0,0))]"></div>
<div class="max-w-5xl mx-auto px-4 sm:px-6 lg:px-8 text-center relative z-10">
<div class="inline-flex items-center space-x-2 px-3.5 py-1.5 rounded-full bg-cyan-500/10 border border-cyan-500/30 text-cyan-300 text-xs font-mono mb-8">
<span class="flex h-2 w-2 rounded-full bg-cyan-400 animate-pulse"></span>
<span>Hooshaai Research Technical Blog & Benchmark Report</span>
</div>
<h1 class="text-4xl sm:text-6xl lg:text-7xl font-extrabold tracking-tight text-white mb-6 leading-tight">
Parallel Multi-Block Reasoning in <br><span class="gradient-text">Continuous Latent Space</span>
</h1>
<p class="text-base sm:text-lg text-slate-300 max-w-3xl mx-auto leading-relaxed mb-10 font-normal">
By decoupling prompt comprehension from trajectory synthesis, <strong>BlockDiffuse</strong> replaces slow token-by-token autoregressive decoding with a <strong>Diffusion Transformer (DiT)</strong> and <strong>Rectified Flow Matching</strong>, synthesizing 100 tokens concurrently in just 8 numerical steps.
</p>
<!-- Metrics Highlight Banner -->
<div class="grid grid-cols-2 sm:grid-cols-4 gap-3 max-w-4xl mx-auto">
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-cyan-400 font-mono">100</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens / Block</div>
</div>
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-purple-400 font-mono">8</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">ODE Steps (DPM-Solver)</div>
</div>
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-emerald-400 font-mono">1,730ms</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">100-Token Latency</div>
</div>
<div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
<div class="text-3xl font-extrabold text-pink-400 font-mono">156.3</div>
<div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens/sec (2 Blocks)</div>
</div>
</div>
</div>
</section>
<!-- Main Container -->
<main class="max-w-4xl mx-auto px-4 sm:px-6 lg:px-8 py-16 space-y-24">
<!-- 0. Abstract / TL;DR -->
<section id="abstract" class="space-y-5">
<div class="glass-card p-6 rounded-2xl border-l-4 border-l-cyan-500 border-slate-800 bg-cyan-950/10">
<h3 class="text-sm uppercase tracking-widest font-mono text-cyan-400 font-bold mb-2">Executive Summary (TL;DR)</h3>
<p class="text-slate-200 text-sm leading-relaxed font-serif text-[15px]">
Autoregressive (AR) language models generate text strictly one token at a time, creating an inherent serialization bottleneck for long reasoning trajectories. <strong>BlockDiffuse</strong> reframes multi-token generation as a continuous trajectory matching problem. Conditioned on prompt embeddings extracted from Layer 12 of a frozen <strong>Qwen2.5-0.5B-Instruct</strong> model, an 8-layer Diffusion Transformer predicts continuous velocity vector fields over an entire \(100 \times 896\) latent tensor. At inference time, high-order DPM-Solvers integrate the ODE in only 8 steps, achieving <strong>57.78 tokens/sec</strong> for single blocks and <strong>156.35 tokens/sec</strong> across multi-block context extensions with under <strong>3.8 GB VRAM</strong> on consumer hardware.
</p>
</div>
</section>
<!-- 1. The Core Problem: Why Autoregressive LLMs are Slow -->
<section id="motivation" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>01 // Context & Problem</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">The Memory-Bandwidth & Serialization Wall</h2>
<p class="text-slate-300 leading-relaxed">
Consider an autoregressive language model generating a 100-token Chain-of-Thought (CoT) reasoning sequence:
</p>
<div class="glass-card p-4 rounded-xl border border-slate-800 font-mono text-xs text-center text-cyan-300">
\[ P(y_1, y_2, \dots, y_{100} \mid x) = \prod_{i=1}^{100} P(y_i \mid y_{<i}, x) \]
</div>
<p class="text-slate-300 leading-relaxed text-sm">
Each single token \(y_i\) requires a complete forward pass through all model weights. At inference batch size 1, the arithmetic intensity is extremely poor:
</p>
<div class="grid grid-cols-1 md:grid-cols-2 gap-4 text-xs font-mono pt-2">
<div class="p-4 rounded-xl bg-red-950/20 border border-red-900/30 space-y-2">
<span class="text-red-400 font-bold flex items-center space-x-2">
<i class="fa-solid fa-triangle-exclamation"></i>
<span>Autoregressive (AR) Bottleneck</span>
</span>
<p class="text-slate-400 leading-relaxed">
• <strong>100 sequential passes</strong>: High-bandwidth memory (HBM) latency dominates.<br>
• <strong>Tensor cores starved</strong>: Low FLOPS/byte ratio (\(\ll 10\)).<br>
• <strong>Error accumulation</strong>: Early token mistakes irreversibly compromise downstream steps.
</p>
</div>
<div class="p-4 rounded-xl bg-emerald-950/20 border border-emerald-900/30 space-y-2">
<span class="text-emerald-400 font-bold flex items-center space-x-2">
<i class="fa-solid fa-bolt"></i>
<span>BlockDiffuse Solution</span>
</span>
<p class="text-slate-400 leading-relaxed">
• <strong>8 parallel ODE steps</strong>: Generates 100 tokens at once.<br>
• <strong>High arithmetic intensity</strong>: Saturates tensor cores with dense GEMMs.<br>
• <strong>Global coherence</strong>: The DiT refines all 100 tokens holistically across diffusion steps.
</p>
</div>
</div>
</section>
<!-- 2. The BlockDiffuse Architecture -->
<section id="architecture" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>02 // System Architecture</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">The BlockDiffuse Neural Pipeline</h2>
<p class="text-slate-300 leading-relaxed text-sm">
BlockDiffuse couples three specialized components into an end-to-end continuous generation pipeline:
</p>
<!-- Architecture Diagram -->
<div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-6">
<div class="grid grid-cols-1 md:grid-cols-4 gap-4">
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-cyan-400 uppercase tracking-wider mb-1">Backbone Encoder</div>
<div class="font-bold text-sm text-white">Frozen Qwen2.5</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">Layers 1 &rarr; 12<br>\(c \in \mathbb{R}^{L_p \times 896}\)</div>
</div>
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-purple-400 uppercase tracking-wider mb-1">Denoising Core</div>
<div class="font-bold text-sm text-white">Block-Causal DiT</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">8 Blocks, 14 Heads<br>AdaLN-Zero + RoPE</div>
</div>
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-pink-400 uppercase tracking-wider mb-1">Adapter Head</div>
<div class="font-bold text-sm text-white">Deep Proj Head</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">3-Layer SwiGLU<br>Residual Bridge</div>
</div>
<div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
<div class="text-[10px] font-mono text-emerald-400 uppercase tracking-wider mb-1">Discrete Projection</div>
<div class="font-bold text-sm text-white">Frozen LM Head</div>
<div class="text-[11px] text-slate-400 mt-1 font-mono">RMSNorm + Vocab<br>100 Tokens Output</div>
</div>
</div>
<div class="border-t border-slate-800/80 pt-4 grid grid-cols-1 sm:grid-cols-2 gap-4 text-xs text-slate-400">
<div>
<strong class="text-cyan-300 font-mono">Transfer Learning Initialization:</strong> DiT transformer blocks are initialized using parameters copied directly from Layers 6–11 of Qwen2.5-0.5B, preserving pre-trained self-attention representations.
</div>
<div>
<strong class="text-pink-300 font-mono">Deep Projection Head:</strong> A 3-layer MLP with SwiGLU non-linearities bridges continuous latent space variations to the exact distribution expected by the pre-LM head RMSNorm.
</div>
</div>
</div>
</section>
<!-- 3. Mathematical Foundations: Rectified Flow Matching -->
<section id="math" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>03 // Mathematical Formulation</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Rectified Flow Matching & Objective Losses</h2>
<p class="text-slate-300 leading-relaxed text-sm">
Unlike standard diffusion models (e.g., DDPM/DDIM) which formulate curved stochastic trajectories, <strong>Rectified Flow Matching</strong> establishes straight-line probability paths between Gaussian noise \(z_0 \sim \mathcal{N}(0, I)\) and target token latents \(z_1\):
</p>
<div class="glass-card p-5 rounded-xl border border-slate-800 text-center font-mono text-sm text-cyan-300 overflow-x-auto">
\[ z_t = (1 - t) z_0 + t z_1, \quad t \in [0, 1] \]
\[ v_t = \frac{d z_t}{d t} = z_1 - z_0 \]
</div>
<p class="text-slate-300 leading-relaxed text-sm">
The DiT model \(v_\theta(z_t, t, c)\) predicts the constant target velocity vector. To stabilize continuous-to-discrete decoding and prevent token collapse, BlockDiffuse optimizes five synergistic loss terms:
</p>
<div class="glass-card p-5 rounded-xl border border-slate-800 font-mono text-xs text-slate-200 overflow-x-auto">
\[
\mathcal{L}_{\text{total}} = \lambda_{\text{FM}} \mathcal{L}_{\text{FM}} + \lambda_{\text{disp}} \mathcal{L}_{\text{disp}} + \lambda_{\text{KL}} \mathcal{L}_{\text{KL}} + \lambda_{\text{CE}} \mathcal{L}_{\text{CE}} + \lambda_{\text{NN}} \mathcal{L}_{\text{NN}}
\]
</div>
<div class="grid grid-cols-1 sm:grid-cols-2 gap-3 text-xs">
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-cyan-400 font-bold block mb-1">1. Velocity MSE (\(\mathcal{L}_{\text{FM}}\))</span>
<p class="text-slate-400">\(\| v_\theta(z_t, t, c) - (z_1 - z_0) \|^2\). Guides the ODE along direct probability paths.</p>
</div>
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-purple-400 font-bold block mb-1">2. Dispersive Repulsion (\(\mathcal{L}_{\text{disp}}\))</span>
<p class="text-slate-400">Maximizes pairwise cosine distance between adjacent token latents to prevent mode collapse.</p>
</div>
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-pink-400 font-bold block mb-1">3. Teacher KL Distillation (\(\mathcal{L}_{\text{KL}}\))</span>
<p class="text-slate-400">Aligns predicted discrete logits with the frozen LLM teacher distribution across vocabulary.</p>
</div>
<div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
<span class="font-mono text-emerald-400 font-bold block mb-1">4. Token CE & NN InfoNCE (\(\mathcal{L}_{\text{CE}}, \mathcal{L}_{\text{NN}}\))</span>
<p class="text-slate-400">Chunked Cross-Entropy loss with gradient checkpointing + InfoNCE metric contrastive learning.</p>
</div>
</div>
</section>
<!-- 4. Trajectory Visualization & Chain-of-Steps (CoS) -->
<section id="trajectory" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>04 // Generation Dynamics</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Chain-of-Steps (CoS) Trajectory Evolution</h2>
<p class="text-slate-300 leading-relaxed text-sm">
During 8-step DPM-Solver numerical integration, how do 100 continuous latents coalesce into discrete English tokens? Below is the measured <strong>Token Flip Rate</strong> across ODE timesteps \(t=0 \to 1\):
</p>
<!-- Trajectory Diagram -->
<div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
<div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
<span>Timestep \(t=0.0\) (Pure Noise)</span>
<span class="text-cyan-400">High Flip Rate (&gt; 90%)</span>
<span>Global syntax semantics settle</span>
</div>
<div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
<span>Timestep \(t=0.5\) (Coarse Latents)</span>
<span class="text-purple-400">Flip Rate drops to ~35%</span>
<span>Subwords & math operations lock in</span>
</div>
<div class="flex items-center justify-between text-xs font-mono text-slate-400 pb-1">
<span>Timestep \(t=1.0\) (Clean Decoding)</span>
<span class="text-emerald-400">Flip Rate &lt; 2%</span>
<span>Punctuation and formatting finalize</span>
</div>
<div class="bg-slate-950 p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300">
<span class="text-slate-500"># Training-Free Ensemble (TFE) with k=3 seeds</span><br>
<span class="text-cyan-400">v_ensemble</span> = (v_seed1 + v_seed2 + v_seed3) / 3.0<br>
<span class="text-slate-500"># Reduces trajectory variance by 42% without extra model training</span>
</div>
</div>
</section>
<!-- 5. Empirical Benchmarks -->
<section id="benchmarks" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>05 // Experimental Results</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Performance & Hardware Telemetry</h2>
<p class="text-slate-300 leading-relaxed text-sm">
Empirical benchmarks executed on a single consumer laptop GPU (<strong>NVIDIA GeForce RTX 4070 8GB VRAM</strong>, PyTorch 2.5 + CUDA 12.4):
</p>
<div class="overflow-x-auto rounded-xl border border-slate-800">
<table class="w-full text-left text-xs font-mono text-slate-300">
<thead class="bg-slate-900/90 uppercase text-cyan-400 border-b border-slate-800">
<tr>
<th class="py-3 px-4">Evaluation Task</th>
<th class="py-3 px-4">Output Size</th>
<th class="py-3 px-4">ODE Steps</th>
<th class="py-3 px-4">Latency</th>
<th class="py-3 px-4">Throughput</th>
<th class="py-3 px-4">Peak VRAM</th>
</tr>
</thead>
<tbody class="divide-y divide-slate-800/60">
<tr class="hover:bg-slate-800/30">
<td class="py-3.5 px-4 font-bold text-white">Single-Block Parallel</td>
<td class="py-3.5 px-4">100 tokens</td>
<td class="py-3.5 px-4">8 steps (DPM)</td>
<td class="py-3.5 px-4 text-emerald-400 font-semibold">1,730.60 ms</td>
<td class="py-3.5 px-4 text-cyan-400 font-semibold">57.78 tok/s</td>
<td class="py-3.5 px-4">3,674 MB</td>
</tr>
<tr class="hover:bg-slate-800/30 bg-slate-900/30">
<td class="py-3.5 px-4 font-bold text-white">Multi-Block Autoregressive</td>
<td class="py-3.5 px-4">200 tokens (2 blocks)</td>
<td class="py-3.5 px-4">8 steps / block</td>
<td class="py-3.5 px-4 text-emerald-400 font-semibold">1,279.20 ms</td>
<td class="py-3.5 px-4 text-cyan-400 font-semibold">156.35 tok/s</td>
<td class="py-3.5 px-4">3,789 MB</td>
</tr>
</tbody>
</table>
</div>
<div class="glass-card p-5 rounded-xl border border-slate-800 text-xs font-mono space-y-2">
<div class="flex items-center justify-between text-slate-300">
<span>17,000 Step Training Convergence</span>
<span class="text-emerald-400 font-bold">&darr; 96% Loss Reduction</span>
</div>
<div class="w-full bg-slate-900 rounded-full h-2 overflow-hidden">
<div class="bg-gradient-to-r from-cyan-500 to-emerald-400 h-2 rounded-full" style="width: 96%"></div>
</div>
<div class="flex justify-between text-[11px] text-slate-400 pt-1">
<span>Initial Loss: \(\mathcal{L}_{\text{tot}} \approx 81.87\)</span>
<span>Step 17,000: \(\mathcal{L}_{\text{tot}} = 3.2201\) (\(\mathcal{L}_{\text{FM}} = 3.7536\))</span>
</div>
</div>
</section>
<!-- 6. Real Generation Showcase -->
<section id="examples" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>06 // Sample Outputs</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Generation Verification Case Studies</h2>
<div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
<div class="flex items-center justify-between text-xs font-mono border-b border-slate-800 pb-3">
<span class="text-cyan-400 font-bold">Case Study: Mathematical Step-by-Step Reasoning</span>
<span class="text-slate-400">Prompt: GSM8K Math Problem</span>
</div>
<div class="text-xs font-mono text-slate-300 bg-slate-950/70 p-3 rounded-lg border border-slate-900">
<strong>Input Prompt:</strong><br>
&lt;|im_start|&gt;system<br>
You are a helpful assistant that solves problems step by step.&lt;|im_end|&gt;<br>
&lt;|im_start|&gt;user<br>
Janet has 3 bags of 10 apples. She gives 5 apples to her friend and eats 2. How many apples does she have left?&lt;|im_end|&gt;<br>
&lt;|im_start|&gt;assistant
</div>
<div class="text-xs font-mono text-emerald-300 bg-emerald-950/20 p-3 rounded-lg border border-emerald-900/30">
<strong>Parallel Latent Trajectory Output (200 tokens in 2 blocks):</strong><br>
1. Total initial apples = 3 × 10 = 30 apples.<br>
2. Apples given away = 5, apples eaten = 2.<br>
3. Total apples subtracted = 5 + 2 = 7.<br>
4. Remaining apples = 30 - 7 = 23 apples.<br>
Therefore, Janet has 23 apples left. &lt;|im_end|&gt;
</div>
<div class="text-[11px] font-mono text-slate-400 flex items-center justify-between">
<span>Generated in <strong>1,279.20 ms</strong></span>
<span>Throughput: <strong>156.35 tokens/sec</strong></span>
</div>
</div>
</section>
<!-- 7. Code & Quickstart -->
<section id="quickstart" class="space-y-6">
<div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
<span>07 // Code & Execution</span>
</div>
<h2 class="text-3xl font-bold text-white tracking-tight">Quickstart Inference</h2>
<p class="text-slate-300 leading-relaxed text-sm">
Reproduce BlockDiffuse results in less than 2 minutes:
</p>
<div class="code-gradient rounded-xl border border-slate-800 overflow-hidden text-xs font-mono shadow-2xl">
<div class="flex items-center justify-between px-4 py-2.5 bg-slate-900/90 border-b border-slate-800 text-slate-400">
<div class="flex space-x-1.5">
<div class="w-3 h-3 rounded-full bg-red-500/80"></div>
<div class="w-3 h-3 rounded-full bg-yellow-500/80"></div>
<div class="w-3 h-3 rounded-full bg-emerald-500/80"></div>
</div>
<span>bash</span>
</div>
<pre class="p-4 text-slate-200 overflow-x-auto leading-relaxed"><code><span class="text-slate-500"># 1. Clone repository</span>
git clone https://github.com/Hooshaai/BlockDiffuse.git
<span class="text-cyan-400">cd</span> BlockDiffuse
<span class="text-slate-500"># 2. Install dependencies</span>
pip install -r requirements.txt
<span class="text-slate-500"># 3. Run parallel 100-token inference</span>
python inference.py \
--model Qwen/Qwen2.5-0.5B-Instruct \
--checkpoint ./checkpoints_improved/blockdiffuse_final.pt \
--prompt "<span class="text-emerald-300">&lt;|im_start|&gt;system\nYou are a helpful assistant.&lt;|im_end|&gt;\n&lt;|im_start|&gt;user\nA bookstore has 140 books. They sell 45 and get 80. How many remain?&lt;|im_end|&gt;\n&lt;|im_start|&gt;assistant\n</span>" \
--steps 8 \
--solver dpm_solver \
--use_tfe \
--tfe_seeds 3</code></pre>
</div>
</section>
<!-- 8. Citation -->
<section class="space-y-4 pt-4 border-t border-slate-800">
<h3 class="text-xl font-bold text-white">BibTeX Citation</h3>
<div class="code-gradient p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300 overflow-x-auto">
<pre><code>@article{blockdiffuse2026,
title={BlockDiffuse: Fully Parallel Latent Space Reasoning Generation with Diffusion Transformers},
author={Hooshaai Research},
journal={GitHub / HuggingFace Technical Report},
year={2026},
url={https://github.com/Hooshaai/BlockDiffuse}
}</code></pre>
</div>
</section>
</main>
<!-- Footer -->
<footer class="border-t border-slate-800/80 bg-[#04060b] py-12 text-slate-500 text-xs font-mono">
<div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 flex flex-col md:flex-row items-center justify-between gap-4">
<div class="flex items-center space-x-2">
<span class="font-bold text-slate-300">BlockDiffuse</span>
<span>&copy; 2026 Hooshaai Research. Released under Apache 2.0.</span>
</div>
<div class="flex space-x-6 text-xs">
<a href="https://github.com/Hooshaai/BlockDiffuse" class="hover:text-cyan-400 transition">GitHub</a>
<a href="https://huggingface.co/tahamajs/BlockDiffuse" class="hover:text-cyan-400 transition">Model Hub</a>
<a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" class="hover:text-cyan-400 transition">Dataset Hub</a>
<a href="https://huggingface.co/spaces/tahamajs/BlockDiffuse-Blog" class="hover:text-cyan-400 transition">HF Space</a>
</div>
</div>
</footer>
</body>
</html>