Diffusion Models

Diffusion Models and the Denoising Process

A diffusion model learns to undo a simple forward process that blends a training image x0 with Gaussian noise eps over T steps (T = 1000 in Ho, Jain and Abbeel's 2020 DDPM paper). Any step can be reached in one jump: x_t = sqrt(alphaBar_t) * x0 + sqrt(1 - alphaBar_t) * eps, where alphaBar_t (ᾱ) falls from 1 to 0 on a fixed noise schedule. The figure applies it with the cosine schedule to six book spines, drawn with ImageData's ImageData.

The forward process at five steps: by t = 600 the spines are barely visible, and at t = 1000 only noise is left
The forward process at five steps: by t = 600 the spines are barely visible, and at t = 1000 only noise is left

Training asks the network to predict the noise added to an image at step t. Generation runs backward: from pure noise, predict the noise, remove part of it, repeat. Modern samplers (DDIM, DPM-Solver) need 20-50 steps, not 1,000. Early steps fix layout and color and late ones add texture, as Streaming Partial Images's partial images show.

The diffusion forward process on six book spines, x_t = sqrt(ᾱ)·x0 + sqrt(1 − ᾱ)·ε, with a cosine scheduleHTMLLive
<!doctype html>
<style>
  body { margin: 0; padding: 8px; background: #fafaf7; font: 12px system-ui, sans-serif; color: #263238; }
  canvas { display: block; max-width: 100%; }
  .ui { margin: 6px 0; }
</style>
<canvas id="strip" width="600" height="150"></canvas>
<div class="ui">t = <input type="range" id="t" min="0" max="1000" value="400" style="width:300px"> <b id="tv"></b>
  <button id="reverse">Run the reverse process</button></div>
<canvas id="big" width="600" height="200"></canvas>
<script>
  const S = 110;                                         // each tile is S × S pixels
  // x0: six book spines on a shelf, drawn once into ImageData
  const src = document.createElement('canvas');
  src.width = src.height = S;
  const g = src.getContext('2d');
  g.fillStyle = '#fff3e0';  g.fillRect(0, 0, S, S);
  ['#1f5f8b', '#5b3f99', '#e09a10', '#3f7d3a', '#b5452f', '#2a9d8f'].forEach((c, i) => {
    g.fillStyle = c;  g.fillRect(8 + i * 16, 22 + (i % 3) * 8, 13, 70 - (i % 3) * 8);
  });
  g.fillStyle = '#6b4f3a';  g.fillRect(0, 92, S, 8);
  const x0 = g.getImageData(0, 0, S, S).data;

  // A fixed noise image eps ~ N(0, 1) per channel (Box-Muller with a seeded generator)
  let seed = 7;
  const rand = () => (seed = (seed * 16807) % 2147483647) / 2147483647;
  const eps = Float32Array.from({ length: x0.length }, () => Math.sqrt(-2 * Math.log(rand() || 1e-9)) * Math.cos(2 * Math.PI * rand()));

  // Cosine schedule (Nichol and Dhariwal): alphaBar falls from 1 to 0 as t goes 0 → T
  const T = 1000, alphaBar = t => Math.cos(((t / T + 0.008) / 1.008) * Math.PI / 2) ** 2;

  // Jump straight to step t: x_t = sqrt(ab) * x0 + sqrt(1 - ab) * eps, in [-1, 1] pixel units
  function noisy(t, target, x0data = x0) {
    const ab = alphaBar(t), a = Math.sqrt(ab), b = Math.sqrt(1 - ab);
    const out = new ImageData(S, S);
    for (let i = 0; i < x0data.length; i += 4) {
      for (let c = 0; c < 3; c++) {
        const v = a * (x0data[i + c] / 127.5 - 1) + b * eps[i + c];
        out.data[i + c] = (v + 1) * 127.5;
      }
      out.data[i + 3] = 255;
    }
    return out;
  }

  // The strip: five steps of the forward process
  const strip = document.getElementById('strip').getContext('2d');
  strip.font = '12px system-ui';
  [0, 200, 400, 600, 1000].forEach((t, k) => {
    strip.putImageData(noisy(t), 8 + k * 118, 8);
    strip.fillStyle = '#263238';
    strip.fillText(`t = ${t}, ᾱ = ${alphaBar(t).toFixed(2)}`, 8 + k * 118, 136);
  });

  // The big view: pick any t, or watch a (cheating) reverse process that walks t back to 0
  const big = document.getElementById('big').getContext('2d');
  const tile = document.createElement('canvas');
  tile.width = tile.height = S;
  function show(t) {
    tile.getContext('2d').putImageData(noisy(t), 0, 0);
    big.imageSmoothingEnabled = false;
    big.clearRect(0, 0, 600, 200);
    big.drawImage(tile, 0, 0, 180, 180);
    big.fillStyle = '#263238';  big.font = '13px system-ui';
    const ab = alphaBar(t);
    big.fillText(`signal weight sqrt(ᾱ) = ${Math.sqrt(ab).toFixed(3)}`, 200, 30);
    big.fillText(`noise weight sqrt(1 − ᾱ) = ${Math.sqrt(1 - ab).toFixed(3)}`, 200, 52);
    // the schedule curve with the current step marked
    big.strokeStyle = '#1f5f8b';  big.beginPath();
    for (let s = 0; s <= T; s += 10) big.lineTo(200 + s * 0.38, 180 - alphaBar(s) * 100);
    big.stroke();
    big.fillStyle = '#e09a10';  big.beginPath();  big.arc(200 + t * 0.38, 180 - ab * 100, 5, 0, 7);  big.fill();
    big.fillStyle = '#263238';  big.fillText('ᾱ(t), cosine schedule', 420, 90);
    document.getElementById('tv').textContent = t;
  }
  const slider = document.getElementById('t');
  slider.addEventListener('input', () => show(+slider.value));
  document.getElementById('reverse').onclick = () => {
    // A trained network would predict eps at each step; here we know it, so 25 sampler steps suffice
    let t = T;
    const timer = setInterval(() => {
      t = Math.max(0, t - 40);
      slider.value = t;  show(t);
      if (t === 0) clearInterval(timer);
    }, 80);
  };
  show(400);
</script>