tahamajs commited on
Commit
5f0e110
·
verified ·
1 Parent(s): 09d4bcb

Deploy full research blog to Hooshaai Space

Browse files
Files changed (1) hide show
  1. index.html +510 -18
index.html CHANGED
@@ -1,19 +1,511 @@
1
- <!doctype html>
2
- <html>
3
- <head>
4
- <meta charset="utf-8" />
5
- <meta name="viewport" content="width=device-width" />
6
- <title>My static Space</title>
7
- <link rel="stylesheet" href="style.css" />
8
- </head>
9
- <body>
10
- <div class="card">
11
- <h1>Welcome to your static Space!</h1>
12
- <p>You can modify this app directly by editing <i>index.html</i> in the Files and versions tab.</p>
13
- <p>
14
- Also don't forget to check the
15
- <a href="https://huggingface.co/docs/hub/spaces" target="_blank">Spaces documentation</a>.
16
- </p>
17
- </div>
18
- </body>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
19
  </html>
 
1
+ <!DOCTYPE html>
2
+ <html lang="en" class="scroll-smooth">
3
+ <head>
4
+ <meta charset="UTF-8">
5
+ <meta name="viewport" content="width=device-width, initial-scale=1.0">
6
+ <title>BlockDiffuse: Fully Parallel Latent Space Reasoning with Diffusion Transformers</title>
7
+ <meta name="description" content="Official Research Blog & Technical Report for BlockDiffuse: Non-autoregressive 100-token block generation in continuous latent space using Rectified Flow Matching and DiT.">
8
+ <meta name="keywords" content="BlockDiffuse, Diffusion Transformers, Rectified Flow Matching, Non-Autoregressive, Qwen2.5, Deep Learning, Chain-of-Thought">
9
+
10
+ <!-- OpenGraph Metadata -->
11
+ <meta property="og:title" content="BlockDiffuse: Parallel 100-Token Reasoning in Continuous Latent Space">
12
+ <meta property="og:description" content="Synthesizing 100 tokens simultaneously in 8 ODE integration steps via Diffusion Transformers and frozen LLM latent conditioning.">
13
+ <meta property="og:type" content="article">
14
+
15
+ <!-- Tailwind CSS CDN -->
16
+ <script src="https://cdn.tailwindcss.com"></script>
17
+ <!-- MathJax for TeX equations -->
18
+ <script src="https://polyfill.io/v3/polyfill.min.js?features=es6"></script>
19
+ <script id="MathJax-script" async src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-mml-chtml.js"></script>
20
+ <!-- Font Awesome Icons -->
21
+ <link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/font-awesome/6.4.0/css/all.min.css">
22
+ <!-- Google Fonts -->
23
+ <link rel="preconnect" href="https://fonts.googleapis.com">
24
+ <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
25
+ <link href="https://fonts.googleapis.com/css2?family=Fira+Code:wght@400;500;600;700&family=Inter:wght@300;400;500;600;700;800&family=Newsreader:ital,opsz,wght@0,6..72,400;0,6..72,600;1,6..72,400&display=swap" rel="stylesheet">
26
+
27
+ <script>
28
+ tailwind.config = {
29
+ darkMode: 'class',
30
+ theme: {
31
+ extend: {
32
+ fontFamily: {
33
+ sans: ['Inter', 'sans-serif'],
34
+ serif: ['Newsreader', 'serif'],
35
+ mono: ['Fira Code', 'monospace'],
36
+ },
37
+ colors: {
38
+ brand: {
39
+ cyan: '#38bdf8',
40
+ purple: '#a855f7',
41
+ pink: '#ec4899',
42
+ emerald: '#10b981',
43
+ amber: '#f59e0b',
44
+ dark: '#070b14',
45
+ card: '#0f172a',
46
+ border: '#1e293b'
47
+ }
48
+ }
49
+ }
50
+ }
51
+ }
52
+ </script>
53
+ <style>
54
+ .gradient-text {
55
+ background: linear-gradient(135deg, #38bdf8 0%, #a855f7 50%, #ec4899 100%);
56
+ -webkit-background-clip: text;
57
+ -webkit-text-fill-color: transparent;
58
+ }
59
+ .code-gradient {
60
+ background: linear-gradient(180deg, rgba(15,23,42,0.95) 0%, rgba(7,11,20,0.98) 100%);
61
+ }
62
+ .glass-card {
63
+ background: rgba(15, 23, 42, 0.78);
64
+ backdrop-filter: blur(14px);
65
+ border: 1px solid rgba(255, 255, 255, 0.08);
66
+ }
67
+ .glass-card-hover:hover {
68
+ border-color: rgba(56, 189, 248, 0.35);
69
+ transform: translateY(-2px);
70
+ transition: all 0.25s ease-in-out;
71
+ }
72
+ .tab-active {
73
+ border-color: #38bdf8;
74
+ color: #38bdf8;
75
+ background-color: rgba(56, 189, 248, 0.1);
76
+ }
77
+ </style>
78
+ </head>
79
+ <body class="bg-[#060911] text-slate-200 font-sans antialiased selection:bg-cyan-500 selection:text-black">
80
+
81
+ <!-- Top Alert Banner -->
82
+ <div class="bg-gradient-to-r from-cyan-950/60 via-purple-950/60 to-pink-950/60 border-b border-cyan-500/20 py-2 px-4 text-center text-xs font-mono text-cyan-300">
83
+ 🎉 <strong>Research Release:</strong> Checkpoint weights, datasets, and code are now public on Hugging Face & GitHub!
84
+ </div>
85
+
86
+ <!-- Navigation Header -->
87
+ <header class="sticky top-0 z-50 glass-card border-b border-slate-800/80">
88
+ <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 h-16 flex items-center justify-between">
89
+ <div class="flex items-center space-x-3">
90
+ <div class="h-9 w-9 rounded-lg bg-gradient-to-tr from-cyan-500 via-indigo-500 to-pink-500 flex items-center justify-center text-white font-black text-xl shadow-lg shadow-cyan-500/20">
91
+ B
92
+ </div>
93
+ <div>
94
+ <span class="text-xl font-bold tracking-tight text-white font-mono">Block<span class="text-cyan-400">Diffuse</span></span>
95
+ <span class="hidden sm:inline-block text-[10px] bg-slate-800 border border-slate-700 text-cyan-400 px-2 py-0.5 rounded-full font-mono ml-2">v1.0-Paper</span>
96
+ </div>
97
+ </div>
98
+
99
+ <nav class="hidden lg:flex items-center space-x-7 text-xs font-medium text-slate-400 font-mono uppercase tracking-wider">
100
+ <a href="#abstract" class="hover:text-cyan-400 transition">Abstract</a>
101
+ <a href="#motivation" class="hover:text-cyan-400 transition">Motivation</a>
102
+ <a href="#architecture" class="hover:text-cyan-400 transition">Architecture</a>
103
+ <a href="#math" class="hover:text-cyan-400 transition">Flow Matching</a>
104
+ <a href="#trajectory" class="hover:text-cyan-400 transition">Trajectory</a>
105
+ <a href="#benchmarks" class="hover:text-cyan-400 transition">Benchmarks</a>
106
+ <a href="#examples" class="hover:text-cyan-400 transition">Generations</a>
107
+ <a href="#quickstart" class="hover:text-cyan-400 transition">Code</a>
108
+ </nav>
109
+
110
+ <div class="flex items-center space-x-2.5">
111
+ <a href="https://huggingface.co/tahamajs/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-yellow-500/10 hover:bg-yellow-500/20 border border-yellow-500/30 text-yellow-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
112
+ <span>🤗</span>
113
+ <span class="hidden sm:inline">Model</span>
114
+ </a>
115
+ <a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" target="_blank" class="flex items-center space-x-1.5 bg-cyan-500/10 hover:bg-cyan-500/20 border border-cyan-500/30 text-cyan-400 px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
116
+ <i class="fa-solid fa-database text-xs"></i>
117
+ <span class="hidden sm:inline">Data</span>
118
+ </a>
119
+ <a href="https://github.com/Hooshaai/BlockDiffuse" target="_blank" class="flex items-center space-x-1.5 bg-slate-800 hover:bg-slate-700 border border-slate-700 text-white px-3 py-1.5 rounded-md text-xs font-semibold tracking-wide transition shadow-sm">
120
+ <i class="fa-brands fa-github text-sm"></i>
121
+ <span class="hidden sm:inline">Code</span>
122
+ </a>
123
+ </div>
124
+ </div>
125
+ </header>
126
+
127
+ <!-- Hero Section -->
128
+ <section class="relative pt-20 pb-20 overflow-hidden border-b border-slate-800/80">
129
+ <div class="absolute inset-0 bg-[radial-gradient(ellipse_75%_50%_at_50%_-15%,rgba(56,189,248,0.18),rgba(0,0,0,0))]"></div>
130
+ <div class="max-w-5xl mx-auto px-4 sm:px-6 lg:px-8 text-center relative z-10">
131
+ <div class="inline-flex items-center space-x-2 px-3.5 py-1.5 rounded-full bg-cyan-500/10 border border-cyan-500/30 text-cyan-300 text-xs font-mono mb-8">
132
+ <span class="flex h-2 w-2 rounded-full bg-cyan-400 animate-pulse"></span>
133
+ <span>Hooshaai Research Technical Blog & Benchmark Report</span>
134
+ </div>
135
+
136
+ <h1 class="text-4xl sm:text-6xl lg:text-7xl font-extrabold tracking-tight text-white mb-6 leading-tight">
137
+ Parallel Multi-Block Reasoning in <br><span class="gradient-text">Continuous Latent Space</span>
138
+ </h1>
139
+
140
+ <p class="text-base sm:text-lg text-slate-300 max-w-3xl mx-auto leading-relaxed mb-10 font-normal">
141
+ By decoupling prompt comprehension from trajectory synthesis, <strong>BlockDiffuse</strong> replaces slow token-by-token autoregressive decoding with a <strong>Diffusion Transformer (DiT)</strong> and <strong>Rectified Flow Matching</strong>, synthesizing 100 tokens concurrently in just 8 numerical steps.
142
+ </p>
143
+
144
+ <!-- Metrics Highlight Banner -->
145
+ <div class="grid grid-cols-2 sm:grid-cols-4 gap-3 max-w-4xl mx-auto">
146
+ <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
147
+ <div class="text-3xl font-extrabold text-cyan-400 font-mono">100</div>
148
+ <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens / Block</div>
149
+ </div>
150
+ <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
151
+ <div class="text-3xl font-extrabold text-purple-400 font-mono">8</div>
152
+ <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">ODE Steps (DPM-Solver)</div>
153
+ </div>
154
+ <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
155
+ <div class="text-3xl font-extrabold text-emerald-400 font-mono">1,730ms</div>
156
+ <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">100-Token Latency</div>
157
+ </div>
158
+ <div class="glass-card glass-card-hover p-4 rounded-xl border border-slate-800">
159
+ <div class="text-3xl font-extrabold text-pink-400 font-mono">156.3</div>
160
+ <div class="text-xs text-slate-400 mt-1 uppercase tracking-wider font-semibold">Tokens/sec (2 Blocks)</div>
161
+ </div>
162
+ </div>
163
+ </div>
164
+ </section>
165
+
166
+ <!-- Main Container -->
167
+ <main class="max-w-4xl mx-auto px-4 sm:px-6 lg:px-8 py-16 space-y-24">
168
+
169
+ <!-- 0. Abstract / TL;DR -->
170
+ <section id="abstract" class="space-y-5">
171
+ <div class="glass-card p-6 rounded-2xl border-l-4 border-l-cyan-500 border-slate-800 bg-cyan-950/10">
172
+ <h3 class="text-sm uppercase tracking-widest font-mono text-cyan-400 font-bold mb-2">Executive Summary (TL;DR)</h3>
173
+ <p class="text-slate-200 text-sm leading-relaxed font-serif text-[15px]">
174
+ Autoregressive (AR) language models generate text strictly one token at a time, creating an inherent serialization bottleneck for long reasoning trajectories. <strong>BlockDiffuse</strong> reframes multi-token generation as a continuous trajectory matching problem. Conditioned on prompt embeddings extracted from Layer 12 of a frozen <strong>Qwen2.5-0.5B-Instruct</strong> model, an 8-layer Diffusion Transformer predicts continuous velocity vector fields over an entire \(100 \times 896\) latent tensor. At inference time, high-order DPM-Solvers integrate the ODE in only 8 steps, achieving <strong>57.78 tokens/sec</strong> for single blocks and <strong>156.35 tokens/sec</strong> across multi-block context extensions with under <strong>3.8 GB VRAM</strong> on consumer hardware.
175
+ </p>
176
+ </div>
177
+ </section>
178
+
179
+ <!-- 1. The Core Problem: Why Autoregressive LLMs are Slow -->
180
+ <section id="motivation" class="space-y-6">
181
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
182
+ <span>01 // Context & Problem</span>
183
+ </div>
184
+ <h2 class="text-3xl font-bold text-white tracking-tight">The Memory-Bandwidth & Serialization Wall</h2>
185
+ <p class="text-slate-300 leading-relaxed">
186
+ Consider an autoregressive language model generating a 100-token Chain-of-Thought (CoT) reasoning sequence:
187
+ </p>
188
+ <div class="glass-card p-4 rounded-xl border border-slate-800 font-mono text-xs text-center text-cyan-300">
189
+ \[ P(y_1, y_2, \dots, y_{100} \mid x) = \prod_{i=1}^{100} P(y_i \mid y_{<i}, x) \]
190
+ </div>
191
+ <p class="text-slate-300 leading-relaxed text-sm">
192
+ Each single token \(y_i\) requires a complete forward pass through all model weights. At inference batch size 1, the arithmetic intensity is extremely poor:
193
+ </p>
194
+ <div class="grid grid-cols-1 md:grid-cols-2 gap-4 text-xs font-mono pt-2">
195
+ <div class="p-4 rounded-xl bg-red-950/20 border border-red-900/30 space-y-2">
196
+ <span class="text-red-400 font-bold flex items-center space-x-2">
197
+ <i class="fa-solid fa-triangle-exclamation"></i>
198
+ <span>Autoregressive (AR) Bottleneck</span>
199
+ </span>
200
+ <p class="text-slate-400 leading-relaxed">
201
+ • <strong>100 sequential passes</strong>: High-bandwidth memory (HBM) latency dominates.<br>
202
+ • <strong>Tensor cores starved</strong>: Low FLOPS/byte ratio (\(\ll 10\)).<br>
203
+ • <strong>Error accumulation</strong>: Early token mistakes irreversibly compromise downstream steps.
204
+ </p>
205
+ </div>
206
+ <div class="p-4 rounded-xl bg-emerald-950/20 border border-emerald-900/30 space-y-2">
207
+ <span class="text-emerald-400 font-bold flex items-center space-x-2">
208
+ <i class="fa-solid fa-bolt"></i>
209
+ <span>BlockDiffuse Solution</span>
210
+ </span>
211
+ <p class="text-slate-400 leading-relaxed">
212
+ • <strong>8 parallel ODE steps</strong>: Generates 100 tokens at once.<br>
213
+ • <strong>High arithmetic intensity</strong>: Saturates tensor cores with dense GEMMs.<br>
214
+ • <strong>Global coherence</strong>: The DiT refines all 100 tokens holistically across diffusion steps.
215
+ </p>
216
+ </div>
217
+ </div>
218
+ </section>
219
+
220
+ <!-- 2. The BlockDiffuse Architecture -->
221
+ <section id="architecture" class="space-y-6">
222
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
223
+ <span>02 // System Architecture</span>
224
+ </div>
225
+ <h2 class="text-3xl font-bold text-white tracking-tight">The BlockDiffuse Neural Pipeline</h2>
226
+ <p class="text-slate-300 leading-relaxed text-sm">
227
+ BlockDiffuse couples three specialized components into an end-to-end continuous generation pipeline:
228
+ </p>
229
+
230
+ <!-- Architecture Diagram -->
231
+ <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-6">
232
+ <div class="grid grid-cols-1 md:grid-cols-4 gap-4">
233
+ <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
234
+ <div class="text-[10px] font-mono text-cyan-400 uppercase tracking-wider mb-1">Backbone Encoder</div>
235
+ <div class="font-bold text-sm text-white">Frozen Qwen2.5</div>
236
+ <div class="text-[11px] text-slate-400 mt-1 font-mono">Layers 1 &rarr; 12<br>\(c \in \mathbb{R}^{L_p \times 896}\)</div>
237
+ </div>
238
+ <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
239
+ <div class="text-[10px] font-mono text-purple-400 uppercase tracking-wider mb-1">Denoising Core</div>
240
+ <div class="font-bold text-sm text-white">Block-Causal DiT</div>
241
+ <div class="text-[11px] text-slate-400 mt-1 font-mono">8 Blocks, 14 Heads<br>AdaLN-Zero + RoPE</div>
242
+ </div>
243
+ <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
244
+ <div class="text-[10px] font-mono text-pink-400 uppercase tracking-wider mb-1">Adapter Head</div>
245
+ <div class="font-bold text-sm text-white">Deep Proj Head</div>
246
+ <div class="text-[11px] text-slate-400 mt-1 font-mono">3-Layer SwiGLU<br>Residual Bridge</div>
247
+ </div>
248
+ <div class="p-4 rounded-xl bg-slate-900/90 border border-slate-800 text-center">
249
+ <div class="text-[10px] font-mono text-emerald-400 uppercase tracking-wider mb-1">Discrete Projection</div>
250
+ <div class="font-bold text-sm text-white">Frozen LM Head</div>
251
+ <div class="text-[11px] text-slate-400 mt-1 font-mono">RMSNorm + Vocab<br>100 Tokens Output</div>
252
+ </div>
253
+ </div>
254
+
255
+ <div class="border-t border-slate-800/80 pt-4 grid grid-cols-1 sm:grid-cols-2 gap-4 text-xs text-slate-400">
256
+ <div>
257
+ <strong class="text-cyan-300 font-mono">Transfer Learning Initialization:</strong> DiT transformer blocks are initialized using parameters copied directly from Layers 6–11 of Qwen2.5-0.5B, preserving pre-trained self-attention representations.
258
+ </div>
259
+ <div>
260
+ <strong class="text-pink-300 font-mono">Deep Projection Head:</strong> A 3-layer MLP with SwiGLU non-linearities bridges continuous latent space variations to the exact distribution expected by the pre-LM head RMSNorm.
261
+ </div>
262
+ </div>
263
+ </div>
264
+ </section>
265
+
266
+ <!-- 3. Mathematical Foundations: Rectified Flow Matching -->
267
+ <section id="math" class="space-y-6">
268
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
269
+ <span>03 // Mathematical Formulation</span>
270
+ </div>
271
+ <h2 class="text-3xl font-bold text-white tracking-tight">Rectified Flow Matching & Objective Losses</h2>
272
+ <p class="text-slate-300 leading-relaxed text-sm">
273
+ Unlike standard diffusion models (e.g., DDPM/DDIM) which formulate curved stochastic trajectories, <strong>Rectified Flow Matching</strong> establishes straight-line probability paths between Gaussian noise \(z_0 \sim \mathcal{N}(0, I)\) and target token latents \(z_1\):
274
+ </p>
275
+
276
+ <div class="glass-card p-5 rounded-xl border border-slate-800 text-center font-mono text-sm text-cyan-300 overflow-x-auto">
277
+ \[ z_t = (1 - t) z_0 + t z_1, \quad t \in [0, 1] \]
278
+ \[ v_t = \frac{d z_t}{d t} = z_1 - z_0 \]
279
+ </div>
280
+
281
+ <p class="text-slate-300 leading-relaxed text-sm">
282
+ The DiT model \(v_\theta(z_t, t, c)\) predicts the constant target velocity vector. To stabilize continuous-to-discrete decoding and prevent token collapse, BlockDiffuse optimizes five synergistic loss terms:
283
+ </p>
284
+
285
+ <div class="glass-card p-5 rounded-xl border border-slate-800 font-mono text-xs text-slate-200 overflow-x-auto">
286
+ \[
287
+ \mathcal{L}_{\text{total}} = \lambda_{\text{FM}} \mathcal{L}_{\text{FM}} + \lambda_{\text{disp}} \mathcal{L}_{\text{disp}} + \lambda_{\text{KL}} \mathcal{L}_{\text{KL}} + \lambda_{\text{CE}} \mathcal{L}_{\text{CE}} + \lambda_{\text{NN}} \mathcal{L}_{\text{NN}}
288
+ \]
289
+ </div>
290
+
291
+ <div class="grid grid-cols-1 sm:grid-cols-2 gap-3 text-xs">
292
+ <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
293
+ <span class="font-mono text-cyan-400 font-bold block mb-1">1. Velocity MSE (\(\mathcal{L}_{\text{FM}}\))</span>
294
+ <p class="text-slate-400">\(\| v_\theta(z_t, t, c) - (z_1 - z_0) \|^2\). Guides the ODE along direct probability paths.</p>
295
+ </div>
296
+ <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
297
+ <span class="font-mono text-purple-400 font-bold block mb-1">2. Dispersive Repulsion (\(\mathcal{L}_{\text{disp}}\))</span>
298
+ <p class="text-slate-400">Maximizes pairwise cosine distance between adjacent token latents to prevent mode collapse.</p>
299
+ </div>
300
+ <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
301
+ <span class="font-mono text-pink-400 font-bold block mb-1">3. Teacher KL Distillation (\(\mathcal{L}_{\text{KL}}\))</span>
302
+ <p class="text-slate-400">Aligns predicted discrete logits with the frozen LLM teacher distribution across vocabulary.</p>
303
+ </div>
304
+ <div class="p-3.5 rounded-lg bg-slate-900/70 border border-slate-800">
305
+ <span class="font-mono text-emerald-400 font-bold block mb-1">4. Token CE & NN InfoNCE (\(\mathcal{L}_{\text{CE}}, \mathcal{L}_{\text{NN}}\))</span>
306
+ <p class="text-slate-400">Chunked Cross-Entropy loss with gradient checkpointing + InfoNCE metric contrastive learning.</p>
307
+ </div>
308
+ </div>
309
+ </section>
310
+
311
+ <!-- 4. Trajectory Visualization & Chain-of-Steps (CoS) -->
312
+ <section id="trajectory" class="space-y-6">
313
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
314
+ <span>04 // Generation Dynamics</span>
315
+ </div>
316
+ <h2 class="text-3xl font-bold text-white tracking-tight">Chain-of-Steps (CoS) Trajectory Evolution</h2>
317
+ <p class="text-slate-300 leading-relaxed text-sm">
318
+ During 8-step DPM-Solver numerical integration, how do 100 continuous latents coalesce into discrete English tokens? Below is the measured <strong>Token Flip Rate</strong> across ODE timesteps \(t=0 \to 1\):
319
+ </p>
320
+
321
+ <!-- Trajectory Diagram -->
322
+ <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
323
+ <div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
324
+ <span>Timestep \(t=0.0\) (Pure Noise)</span>
325
+ <span class="text-cyan-400">High Flip Rate (&gt; 90%)</span>
326
+ <span>Global syntax semantics settle</span>
327
+ </div>
328
+ <div class="flex items-center justify-between text-xs font-mono text-slate-400 border-b border-slate-800 pb-3">
329
+ <span>Timestep \(t=0.5\) (Coarse Latents)</span>
330
+ <span class="text-purple-400">Flip Rate drops to ~35%</span>
331
+ <span>Subwords & math operations lock in</span>
332
+ </div>
333
+ <div class="flex items-center justify-between text-xs font-mono text-slate-400 pb-1">
334
+ <span>Timestep \(t=1.0\) (Clean Decoding)</span>
335
+ <span class="text-emerald-400">Flip Rate &lt; 2%</span>
336
+ <span>Punctuation and formatting finalize</span>
337
+ </div>
338
+
339
+ <div class="bg-slate-950 p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300">
340
+ <span class="text-slate-500"># Training-Free Ensemble (TFE) with k=3 seeds</span><br>
341
+ <span class="text-cyan-400">v_ensemble</span> = (v_seed1 + v_seed2 + v_seed3) / 3.0<br>
342
+ <span class="text-slate-500"># Reduces trajectory variance by 42% without extra model training</span>
343
+ </div>
344
+ </div>
345
+ </section>
346
+
347
+ <!-- 5. Empirical Benchmarks -->
348
+ <section id="benchmarks" class="space-y-6">
349
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
350
+ <span>05 // Experimental Results</span>
351
+ </div>
352
+ <h2 class="text-3xl font-bold text-white tracking-tight">Performance & Hardware Telemetry</h2>
353
+ <p class="text-slate-300 leading-relaxed text-sm">
354
+ Empirical benchmarks executed on a single consumer laptop GPU (<strong>NVIDIA GeForce RTX 4070 8GB VRAM</strong>, PyTorch 2.5 + CUDA 12.4):
355
+ </p>
356
+
357
+ <div class="overflow-x-auto rounded-xl border border-slate-800">
358
+ <table class="w-full text-left text-xs font-mono text-slate-300">
359
+ <thead class="bg-slate-900/90 uppercase text-cyan-400 border-b border-slate-800">
360
+ <tr>
361
+ <th class="py-3 px-4">Evaluation Task</th>
362
+ <th class="py-3 px-4">Output Size</th>
363
+ <th class="py-3 px-4">ODE Steps</th>
364
+ <th class="py-3 px-4">Latency</th>
365
+ <th class="py-3 px-4">Throughput</th>
366
+ <th class="py-3 px-4">Peak VRAM</th>
367
+ </tr>
368
+ </thead>
369
+ <tbody class="divide-y divide-slate-800/60">
370
+ <tr class="hover:bg-slate-800/30">
371
+ <td class="py-3.5 px-4 font-bold text-white">Single-Block Parallel</td>
372
+ <td class="py-3.5 px-4">100 tokens</td>
373
+ <td class="py-3.5 px-4">8 steps (DPM)</td>
374
+ <td class="py-3.5 px-4 text-emerald-400 font-semibold">1,730.60 ms</td>
375
+ <td class="py-3.5 px-4 text-cyan-400 font-semibold">57.78 tok/s</td>
376
+ <td class="py-3.5 px-4">3,674 MB</td>
377
+ </tr>
378
+ <tr class="hover:bg-slate-800/30 bg-slate-900/30">
379
+ <td class="py-3.5 px-4 font-bold text-white">Multi-Block Autoregressive</td>
380
+ <td class="py-3.5 px-4">200 tokens (2 blocks)</td>
381
+ <td class="py-3.5 px-4">8 steps / block</td>
382
+ <td class="py-3.5 px-4 text-emerald-400 font-semibold">1,279.20 ms</td>
383
+ <td class="py-3.5 px-4 text-cyan-400 font-semibold">156.35 tok/s</td>
384
+ <td class="py-3.5 px-4">3,789 MB</td>
385
+ </tr>
386
+ </tbody>
387
+ </table>
388
+ </div>
389
+
390
+ <div class="glass-card p-5 rounded-xl border border-slate-800 text-xs font-mono space-y-2">
391
+ <div class="flex items-center justify-between text-slate-300">
392
+ <span>17,000 Step Training Convergence</span>
393
+ <span class="text-emerald-400 font-bold">&darr; 96% Loss Reduction</span>
394
+ </div>
395
+ <div class="w-full bg-slate-900 rounded-full h-2 overflow-hidden">
396
+ <div class="bg-gradient-to-r from-cyan-500 to-emerald-400 h-2 rounded-full" style="width: 96%"></div>
397
+ </div>
398
+ <div class="flex justify-between text-[11px] text-slate-400 pt-1">
399
+ <span>Initial Loss: \(\mathcal{L}_{\text{tot}} \approx 81.87\)</span>
400
+ <span>Step 17,000: \(\mathcal{L}_{\text{tot}} = 3.2201\) (\(\mathcal{L}_{\text{FM}} = 3.7536\))</span>
401
+ </div>
402
+ </div>
403
+ </section>
404
+
405
+ <!-- 6. Real Generation Showcase -->
406
+ <section id="examples" class="space-y-6">
407
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
408
+ <span>06 // Sample Outputs</span>
409
+ </div>
410
+ <h2 class="text-3xl font-bold text-white tracking-tight">Generation Verification Case Studies</h2>
411
+
412
+ <div class="glass-card p-6 rounded-2xl border border-slate-800 space-y-4">
413
+ <div class="flex items-center justify-between text-xs font-mono border-b border-slate-800 pb-3">
414
+ <span class="text-cyan-400 font-bold">Case Study: Mathematical Step-by-Step Reasoning</span>
415
+ <span class="text-slate-400">Prompt: GSM8K Math Problem</span>
416
+ </div>
417
+ <div class="text-xs font-mono text-slate-300 bg-slate-950/70 p-3 rounded-lg border border-slate-900">
418
+ <strong>Input Prompt:</strong><br>
419
+ &lt;|im_start|&gt;system<br>
420
+ You are a helpful assistant that solves problems step by step.&lt;|im_end|&gt;<br>
421
+ &lt;|im_start|&gt;user<br>
422
+ Janet has 3 bags of 10 apples. She gives 5 apples to her friend and eats 2. How many apples does she have left?&lt;|im_end|&gt;<br>
423
+ &lt;|im_start|&gt;assistant
424
+ </div>
425
+ <div class="text-xs font-mono text-emerald-300 bg-emerald-950/20 p-3 rounded-lg border border-emerald-900/30">
426
+ <strong>Parallel Latent Trajectory Output (200 tokens in 2 blocks):</strong><br>
427
+ 1. Total initial apples = 3 × 10 = 30 apples.<br>
428
+ 2. Apples given away = 5, apples eaten = 2.<br>
429
+ 3. Total apples subtracted = 5 + 2 = 7.<br>
430
+ 4. Remaining apples = 30 - 7 = 23 apples.<br>
431
+ Therefore, Janet has 23 apples left. &lt;|im_end|&gt;
432
+ </div>
433
+ <div class="text-[11px] font-mono text-slate-400 flex items-center justify-between">
434
+ <span>Generated in <strong>1,279.20 ms</strong></span>
435
+ <span>Throughput: <strong>156.35 tokens/sec</strong></span>
436
+ </div>
437
+ </div>
438
+ </section>
439
+
440
+ <!-- 7. Code & Quickstart -->
441
+ <section id="quickstart" class="space-y-6">
442
+ <div class="flex items-center space-x-3 text-cyan-400 font-mono text-xs uppercase tracking-widest">
443
+ <span>07 // Code & Execution</span>
444
+ </div>
445
+ <h2 class="text-3xl font-bold text-white tracking-tight">Quickstart Inference</h2>
446
+ <p class="text-slate-300 leading-relaxed text-sm">
447
+ Reproduce BlockDiffuse results in less than 2 minutes:
448
+ </p>
449
+
450
+ <div class="code-gradient rounded-xl border border-slate-800 overflow-hidden text-xs font-mono shadow-2xl">
451
+ <div class="flex items-center justify-between px-4 py-2.5 bg-slate-900/90 border-b border-slate-800 text-slate-400">
452
+ <div class="flex space-x-1.5">
453
+ <div class="w-3 h-3 rounded-full bg-red-500/80"></div>
454
+ <div class="w-3 h-3 rounded-full bg-yellow-500/80"></div>
455
+ <div class="w-3 h-3 rounded-full bg-emerald-500/80"></div>
456
+ </div>
457
+ <span>bash</span>
458
+ </div>
459
+ <pre class="p-4 text-slate-200 overflow-x-auto leading-relaxed"><code><span class="text-slate-500"># 1. Clone repository</span>
460
+ git clone https://github.com/Hooshaai/BlockDiffuse.git
461
+ <span class="text-cyan-400">cd</span> BlockDiffuse
462
+
463
+ <span class="text-slate-500"># 2. Install dependencies</span>
464
+ pip install -r requirements.txt
465
+
466
+ <span class="text-slate-500"># 3. Run parallel 100-token inference</span>
467
+ python inference.py \
468
+ --model Qwen/Qwen2.5-0.5B-Instruct \
469
+ --checkpoint ./checkpoints_improved/blockdiffuse_final.pt \
470
+ --prompt "<span class="text-emerald-300">&lt;|im_start|&gt;system\nYou are a helpful assistant.&lt;|im_end|&gt;\n&lt;|im_start|&gt;user\nA bookstore has 140 books. They sell 45 and get 80. How many remain?&lt;|im_end|&gt;\n&lt;|im_start|&gt;assistant\n</span>" \
471
+ --steps 8 \
472
+ --solver dpm_solver \
473
+ --use_tfe \
474
+ --tfe_seeds 3</code></pre>
475
+ </div>
476
+ </section>
477
+
478
+ <!-- 8. Citation -->
479
+ <section class="space-y-4 pt-4 border-t border-slate-800">
480
+ <h3 class="text-xl font-bold text-white">BibTeX Citation</h3>
481
+ <div class="code-gradient p-4 rounded-xl border border-slate-800 font-mono text-xs text-slate-300 overflow-x-auto">
482
+ <pre><code>@article{blockdiffuse2026,
483
+ title={BlockDiffuse: Fully Parallel Latent Space Reasoning Generation with Diffusion Transformers},
484
+ author={Hooshaai Research},
485
+ journal={GitHub / HuggingFace Technical Report},
486
+ year={2026},
487
+ url={https://github.com/Hooshaai/BlockDiffuse}
488
+ }</code></pre>
489
+ </div>
490
+ </section>
491
+
492
+ </main>
493
+
494
+ <!-- Footer -->
495
+ <footer class="border-t border-slate-800/80 bg-[#04060b] py-12 text-slate-500 text-xs font-mono">
496
+ <div class="max-w-7xl mx-auto px-4 sm:px-6 lg:px-8 flex flex-col md:flex-row items-center justify-between gap-4">
497
+ <div class="flex items-center space-x-2">
498
+ <span class="font-bold text-slate-300">BlockDiffuse</span>
499
+ <span>&copy; 2026 Hooshaai Research. Released under Apache 2.0.</span>
500
+ </div>
501
+ <div class="flex space-x-6 text-xs">
502
+ <a href="https://github.com/Hooshaai/BlockDiffuse" class="hover:text-cyan-400 transition">GitHub</a>
503
+ <a href="https://huggingface.co/tahamajs/BlockDiffuse" class="hover:text-cyan-400 transition">Model Hub</a>
504
+ <a href="https://huggingface.co/datasets/tahamajs/BlockDiffuse-Data" class="hover:text-cyan-400 transition">Dataset Hub</a>
505
+ <a href="https://huggingface.co/spaces/tahamajs/BlockDiffuse-Blog" class="hover:text-cyan-400 transition">HF Space</a>
506
+ </div>
507
+ </div>
508
+ </footer>
509
+
510
+ </body>
511
  </html>