The denoiser never reads words. A text encoder turns the prompt into a grid of vectors, and cross-attention layers let each image position pull in the words relevant to it. Stable Diffusion 1.5 18,251 uses the text half of OpenAI 86 's CLIP 34,379 (2021, trained on 400 million image-caption pairs), which reads at most 77 tokens and silently cuts the rest. Google's Imagen (2022) found that a bigger language-model encoder, T5-XXL, helped more than a bigger image model; Stable Diffusion 3 and FLUX.1 57,408 pair T5-XXL with CLIP and follow long prompts well.
Diffusion models also apply classifier-free guidance (Ho and Salimans, 2021): each step runs the denoiser with the prompt and with an empty prompt, then pushes the result away from the unconditioned prediction, eps = eps_uncond + w * (eps_cond - eps_uncond). Stable Diffusion's default w is 7.5; higher values look harsh, and the second pass doubles the work per step. Since encoders learned from captions, describe the scene in sentences, not keyword lists; hosted APIs even rewrite short prompts into long captions (Responses API Tool).
<!doctype html>
<style>
body { margin: 0; padding: 8px; background: #fafaf7; font: 12px system-ui, sans-serif; color: #263238; max-width: 600px; }
textarea { width: 100%; height: 60px; font: 12px system-ui; box-sizing: border-box; }
#tokens { display: flex; flex-wrap: wrap; gap: 2px; margin: 6px 0; }
#tokens span { padding: 1px 3px; border-radius: 3px; background: #d8eaf5; font-size: 11px; }
#tokens span.cut { background: #eee; color: #aaa; text-decoration: line-through; }
svg { width: 100%; display: block; background: #fff; }
</style>
<textarea id="prompt">A lighthouse keeper's daughter stands on a stone harbor wall at dusk, looking out to sea while a small red boat drifts home past the lighthouse, gulls circling overhead, warm saffron light fading into a deep blue sky, painterly gouache book-cover art with soft edges and visible brush texture, calm lower third left empty for the title, no text, no letters, no border, cream accents on the waves and the girl's scarf</textarea>
<div id="count"></div>
<div id="tokens"></div>
<p>Guidance weight w <input type="range" id="w" min="0" max="15" step="0.5" value="7.5"> <b id="wv"></b></p>
<svg viewBox="0 0 600 190" font-size="12"></svg>
<script>
// A rough CLIP-like tokenizer: words and punctuation (real BPE splits rare words further)
const tokenize = text => ['<start>', ...(text.toLowerCase().match(/[a-z0-9']+|[^\sa-z0-9]/g) || []), '<end>'];
function showTokens() {
const tokens = tokenize(document.getElementById('prompt').value);
document.getElementById('tokens').innerHTML = tokens.map((t, i) =>
`<span class="${i >= 77 ? 'cut' : ''}">${t}</span>`).join('');
document.getElementById('count').textContent =
`${tokens.length} tokens: CLIP reads the first 77 and silently drops ${Math.max(0, tokens.length - 77)}; T5-XXL reads them all.`;
}
document.getElementById('prompt').addEventListener('input', showTokens);
showTokens();
// Classifier-free guidance: eps = eps_uncond + w * (eps_cond - eps_uncond)
const svg = document.querySelector('svg'), O = [60, 150];
const uncond = [140, -20], cond = [60, -80]; // two noise predictions (2-D stand-ins)
function arrow(id, [x1, y1], [x2, y2], color, label, dash = '') {
let p = svg.querySelector('#' + id);
if (!p) { svg.insertAdjacentHTML('beforeend', `<g id="${id}"><line/><circle r="4"/><text/></g>`); p = svg.querySelector('#' + id); }
p.querySelector('line').setAttribute('x1', x1); p.querySelector('line').setAttribute('y1', y1);
p.querySelector('line').setAttribute('x2', x2); p.querySelector('line').setAttribute('y2', y2);
p.querySelector('line').setAttribute('stroke', color); p.querySelector('line').setAttribute('stroke-width', 3);
p.querySelector('line').setAttribute('stroke-dasharray', dash);
p.querySelector('circle').setAttribute('cx', x2); p.querySelector('circle').setAttribute('cy', y2);
p.querySelector('circle').setAttribute('fill', color);
const t = p.querySelector('text');
t.setAttribute('x', x2 + 8); t.setAttribute('y', y2 + 4); t.setAttribute('fill', color); t.textContent = label;
}
function guide() {
const w = +document.getElementById('w').value;
document.getElementById('wv').textContent = `${w}${w === 7.5 ? ' (Stable Diffusion default)' : w > 10 ? ' (harsh)' : ''}`;
const u = [O[0] + uncond[0], O[1] + uncond[1]], c = [O[0] + cond[0], O[1] + cond[1]];
const g = [u[0] + w * (c[0] - u[0]) / 4, u[1] + w * (c[1] - u[1]) / 4]; // drawn at 1/4 scale
arrow('u', O, u, '#90a4ae', 'eps_uncond (empty prompt)');
arrow('c', O, c, '#1f5f8b', 'eps_cond (your prompt)');
arrow('g', u, g, '#e09a10', `guided: pushed away from uncond`, '6 4');
svg.querySelector('#note')?.remove();
svg.insertAdjacentHTML('beforeend', `<text id="note" x="330" y="30">Two denoiser passes per step:</text>`);
svg.querySelector('#note').insertAdjacentHTML('afterend', '');
}
document.getElementById('w').addEventListener('input', guide);
guide();
svg.insertAdjacentHTML('beforeend', '<text x="330" y="48" fill="#546e7a">with the prompt and with an empty one,</text>' +
'<text x="330" y="66" fill="#546e7a">so guidance doubles the work per step.</text>');
</script>