The other approach treats a picture as a sequence. A tokenizer (a VQ-VAE or VQGAN) maps each image patch to a learned codebook entry, a transformer predicts image tokens one by one after the prompt's text tokens, and a decoder turns the grid into pixels. OpenAI 86 's first DALL-E 86 (January 2021) wrote 1,024 tokens (a 32x32 grid, codebook of 8,192) with a 12-billion-parameter transformer; Google's Parti (2022) scaled it to 20 billion.

With image and text in one token stream, a model can reason about layout, so token-based models spell words, follow counts and keep edits consistent far better than early diffusion models. OpenAI says GPT Image 86 models before gpt-image-2 "generate images by first producing specialized image tokens," and bills every model by image output tokens (272 for a low-quality 1024x1024 gpt-image-1 image, 4,160 at high). It has not published how gpt-image-2 and 2.5 work, so read their quality setting as a token budget (Token Pricing).
<!doctype html>
<style>
body { margin: 0; padding: 8px; background: #fafaf7; font: 12px system-ui, sans-serif; color: #263238; }
canvas { display: block; max-width: 100%; }
</style>
<canvas id="c" width="600" height="300"></canvas>
<p><button id="again">Replay</button> Left: one codebook token at a time, raster order (1,024 tokens, as in the first DALL-E). Right: all positions refined together.</p>
<script>
const ctx = document.getElementById('c').getContext('2d'), G = 32, cell = 8;
// The target picture: a cover with six spines, as a 32×32 grid of palette indices (the "codebook")
const palette = ['#fff3e0', '#1f5f8b', '#5b3f99', '#e09a10', '#3f7d3a', '#b5452f', '#2a9d8f', '#6b4f3a'];
const target = [];
for (let y = 0; y < G; y++) for (let x = 0; x < G; x++) {
let k = 0;
const book = Math.floor((x - 3) / 4.4);
if (y >= 26 && y < 28) k = 7; // shelf
else if (book >= 0 && book < 6 && (x - 3) % 4.4 < 3.6 && y > 6 + (book % 3) * 2 && y < 26) k = book + 1;
target.push(k);
}
let seed = 3;
const rand = () => (seed = (seed * 16807) % 2147483647) / 2147483647;
const noise = target.map(() => [rand() * 255, rand() * 255, rand() * 255]);
const rgb = hex => [1, 3, 5].map(i => parseInt(hex.slice(i, i + 2), 16));
let frame = 0;
function draw() {
ctx.clearRect(0, 0, 600, 300);
ctx.font = 'bold 13px system-ui'; ctx.fillStyle = '#263238';
ctx.fillText('autoregressive (tokens)', 20, 18); ctx.fillText('diffusion (denoising)', 330, 18);
const written = Math.min(G * G, frame * 12); // tokens emitted so far
const t = Math.min(1, frame / 86); // diffusion progress 0 → 1
for (let i = 0; i < G * G; i++) {
const x = i % G, y = Math.floor(i / G);
// Left: a token is either written (final value) or not yet generated
ctx.fillStyle = i < written ? palette[target[i]] : '#e3dfd4';
ctx.fillRect(20 + x * cell, 28 + y * cell, cell - 1, cell - 1);
// Right: every pixel moves from noise toward the image at once
const c = rgb(palette[target[i]]), n = noise[i], k = t * t * (3 - 2 * t);
ctx.fillStyle = `rgb(${c.map((v, j) => n[j] + (v - n[j]) * k).join(',')})`;
ctx.fillRect(330 + x * cell, 28 + y * cell, cell, cell);
}
// The cursor of the autoregressive model
if (written < G * G) {
ctx.strokeStyle = '#b5452f'; ctx.lineWidth = 2;
ctx.strokeRect(20 + (written % G) * cell, 28 + Math.floor(written / G) * cell, cell, cell);
}
ctx.font = '12px system-ui'; ctx.fillStyle = '#263238';
ctx.fillText(`${written} / 1024 tokens`, 20, 292);
ctx.fillText(`step ${Math.round(t * 30)} / 30`, 330, 292);
if (written < G * G || t < 1) { frame++; requestAnimationFrame(draw); }
}
document.getElementById('again').onclick = () => { frame = 0; draw(); };
draw();
</script>