SD on the CPU

Generating BookNest Art with Stable Diffusion on the CPU

Hugging Face 1,113 's diffusers 34,620 (github.com/huggingface/diffusers (https://github.com/huggingface/diffusers 34,620 ), Apache 2.0) wraps a model's text encoder, U-Net, scheduler and VAE in one pipeline. Install the CPU build of PyTorch 11,702 (pip 21,050 install torch --index-url https://download.pytorch.org/whl/cpu 11,702 ), then pip install diffusers transformers accelerate. This script paints the subject of Cover Prompts's first cover:

sd-cover.py: SD 1.5 on four CPU threads, timedPython
# sd-cover.py: Stable Diffusion 1.5 on the CPU with diffusers (Section 8.11.3)
import resource, time
import diffusers, torch
from diffusers import DPMSolverMultistepScheduler, StableDiffusionPipeline
torch.set_num_threads(4)                          # one thread per logical CPU
t0 = time.time()
pipe = StableDiffusionPipeline.from_pretrained(
    'stable-diffusion-v1-5/stable-diffusion-v1-5', variant='fp16',   # 2.1 GB download
    torch_dtype=torch.float32, safety_checker=None)                   # CPUs compute in fp32
pipe.scheduler = DPMSolverMultistepScheduler.from_config(pipe.scheduler.config)
t1 = time.time()
image = pipe('A lighthouse keeper\'s daughter on a harbor wall at dusk, gouache painting, '
             'deep blue and cream, soft light, book cover art',
             negative_prompt='text, letters, blurry, deformed', num_inference_steps=20,
             guidance_scale=7.0, width=512, height=512,
             generator=torch.Generator().manual_seed(7)).images[0]
t2 = time.time()
image.save('sd-harbor.png')
peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss / 1024          # Linux: KB
print(f'torch {torch.__version__}, diffusers {diffusers.__version__}')
print(f'load {t1 - t0:.1f} s, 20 steps {t2 - t1:.1f} s ({(t2 - t1) / 20:.1f} s/step)')
print(f'peak RAM {peak:,.0f} MB, image {image.size}')
Output
torch 2.14.0+cpu, diffusers 0.40.0
load 16.8 s, 20 steps 611.4 s (30.6 s/step)
peak RAM 6,129 MB, image (512, 512)
Stable Diffusion 1.5 on the CPU (left, 512 x 512, 10 minutes, free) and gpt-image-2.5-flare at low quality (right, 1024 x 1536, 11 seconds, $0.0051) for the same subject
Stable Diffusion 1.5 18,251 on the CPU (left, 512 x 512, 10 minutes, free) and gpt-image-2.5-flare at low quality (right, 1024 x 1536, 11 seconds, $0.0051) for the same subject

With cached weights, loading took 17 seconds (the first run downloads 2.1 GB). The 20 steps took 10 minutes because other work shared this 4-CPU machine: on the idle machine the same settings took 185 seconds, so contention made it 3.3 times slower. Each step runs the U-Net twice for classifier-free guidance (Text Conditioning), and the fixed seed makes the run repeatable. The painting is coherent, but the lighthouse is missing and the coat is red: CLIP 34,379 reads at most 77 tokens and weighs them loosely. diffusers 0.40 also warned that torch_dtype is being renamed dtype.

Twenty sampler steps from a fixed seed, replayed: noise resolves into the harbor painting, with CPU timingsHTMLLive
<!doctype html>
<style>
  body { margin: 0; padding: 8px; background: #fafaf7; font: 12px system-ui, sans-serif; color: #263238; }
  canvas { display: block; max-width: 100%; }
</style>
<p>seed <input type="number" id="seed" value="7" style="width:60px"> machine <select id="m"><option value="185">idle 4-CPU (185 s)</option><option value="600">shared, contended (10 min)</option></select>
  <button id="run">pipe(…, num_inference_steps=20)</button></p>
<canvas id="c" width="600" height="250"></canvas>
<script>
  const ctx = document.getElementById('c').getContext('2d'), S = 200;
  // The target the "model" converges to: a gouache-style harbor at dusk (no lighthouse, a red coat,
  // as CLIP's loose reading of the prompt produced)
  const target = document.createElement('canvas');
  target.width = target.height = S;
  const g = target.getContext('2d');
  const sky = g.createLinearGradient(0, 0, 0, 120);
  sky.addColorStop(0, '#1f3f66');  sky.addColorStop(1, '#e8b26a');
  g.fillStyle = sky;  g.fillRect(0, 0, S, 120);
  g.fillStyle = '#16435f';  g.fillRect(0, 120, S, 80);
  g.fillStyle = '#7a6a5a';  g.fillRect(0, 140, 120, 24);                         // harbor wall
  g.fillStyle = '#b5452f';  g.fillRect(60, 104, 14, 36);                        // the girl's red coat
  g.fillStyle = '#f2d2b0';  g.beginPath();  g.arc(67, 98, 7, 0, 7);  g.fill();
  const tpx = g.getImageData(0, 0, S, S).data;
  const frame = ctx.createImageData(S, S);
  let noise, raf;
  function makeNoise(seed) {                                                     // the fixed seed makes runs repeatable
    let s = seed || 1;
    const r = () => (s = (s * 16807) % 2147483647) / 2147483647;
    return Float32Array.from({ length: tpx.length }, () => r() * 255);
  }
  function render(step) {
    // A DPM-style schedule: the share of signal rises quickly in early steps (layout, color) and late steps add detail
    const k = Math.min(1, (step / 20) ** 0.7);
    for (let i = 0; i < tpx.length; i += 4) {
      for (let c = 0; c < 3; c++) frame.data[i + c] = noise[i + c] * (1 - k) + tpx[i + c] * k;
      frame.data[i + 3] = 255;
    }
    ctx.putImageData(frame, 10, 10);
  }
  function panel(step, seconds, total) {
    ctx.clearRect(220, 0, 380, 250);
    ctx.fillStyle = '#263238';  ctx.font = '12px system-ui';
    ctx.fillText(`step ${step} / 20`, 230, 30);
    ctx.fillText(`elapsed ${seconds.toFixed(0)} s of ${total} s (${(total / 20).toFixed(1)} s/step)`, 230, 50);
    ctx.fillText('each step runs the U-Net twice (guidance_scale 7.0)', 230, 70);
    ctx.fillStyle = '#e0d8c8';  ctx.fillRect(230, 84, 340, 14);
    ctx.fillStyle = '#1f5f8b';  ctx.fillRect(230, 84, 340 * step / 20, 14);
    ctx.fillStyle = '#263238';
    ['Stable Diffusion 1.5 on the CPU: 512 × 512, free', 'gpt-image-2.5-flare, low: 1024 × 1536, 11 s, $0.0051',
     'load 17 s with cached weights (first run: 2.1 GB download)'].forEach((t, i) => ctx.fillText(t, 230, 130 + i * 20));
  }
  function run() {
    cancelAnimationFrame(raf);
    noise = makeNoise(+document.getElementById('seed').value);
    const total = +document.getElementById('m').value, start = performance.now();
    const tick = () => {
      const step = Math.min(20, Math.floor((performance.now() - start) / 110));   // replayed fast
      render(step);  panel(step, step * total / 20, total);
      if (step < 20) raf = requestAnimationFrame(tick);
    };
    tick();
  }
  document.getElementById('run').onclick = run;
  run();
</script>