Hugging Face 1,113 's diffusers 34,620 (github.com/huggingface/diffusers (https://github.com/huggingface/diffusers 34,620 ), Apache 2.0) wraps a model's text encoder, U-Net, scheduler and VAE in one pipeline. Install the CPU build of PyTorch 11,702 (pip 21,050 install torch --index-url https://download.pytorch.org/whl/cpu 11,702 ), then pip install diffusers transformers accelerate. This script paints the subject of Cover Prompts's first cover:
# sd-cover.py: Stable Diffusion 1.5 on the CPU with diffusers (Section 8.11.3)
import resource, time
import diffusers, torch
from diffusers import DPMSolverMultistepScheduler, StableDiffusionPipeline
torch.set_num_threads(4) # one thread per logical CPU
t0 = time.time()
pipe = StableDiffusionPipeline.from_pretrained(
'stable-diffusion-v1-5/stable-diffusion-v1-5', variant='fp16', # 2.1 GB download
torch_dtype=torch.float32, safety_checker=None) # CPUs compute in fp32
pipe.scheduler = DPMSolverMultistepScheduler.from_config(pipe.scheduler.config)
t1 = time.time()
image = pipe('A lighthouse keeper\'s daughter on a harbor wall at dusk, gouache painting, '
'deep blue and cream, soft light, book cover art',
negative_prompt='text, letters, blurry, deformed', num_inference_steps=20,
guidance_scale=7.0, width=512, height=512,
generator=torch.Generator().manual_seed(7)).images[0]
t2 = time.time()
image.save('sd-harbor.png')
peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss / 1024 # Linux: KB
print(f'torch {torch.__version__}, diffusers {diffusers.__version__}')
print(f'load {t1 - t0:.1f} s, 20 steps {t2 - t1:.1f} s ({(t2 - t1) / 20:.1f} s/step)')
print(f'peak RAM {peak:,.0f} MB, image {image.size}')torch 2.14.0+cpu, diffusers 0.40.0 load 16.8 s, 20 steps 611.4 s (30.6 s/step) peak RAM 6,129 MB, image (512, 512)

With cached weights, loading took 17 seconds (the first run downloads 2.1 GB). The 20 steps took 10 minutes because other work shared this 4-CPU machine: on the idle machine the same settings took 185 seconds, so contention made it 3.3 times slower. Each step runs the U-Net twice for classifier-free guidance (Text Conditioning), and the fixed seed makes the run repeatable. The painting is coherent, but the lighthouse is missing and the coat is red: CLIP 34,379 reads at most 77 tokens and weighs them loosely. diffusers 0.40 also warned that torch_dtype is being renamed dtype.
<!doctype html>
<style>
body { margin: 0; padding: 8px; background: #fafaf7; font: 12px system-ui, sans-serif; color: #263238; }
canvas { display: block; max-width: 100%; }
</style>
<p>seed <input type="number" id="seed" value="7" style="width:60px"> machine <select id="m"><option value="185">idle 4-CPU (185 s)</option><option value="600">shared, contended (10 min)</option></select>
<button id="run">pipe(…, num_inference_steps=20)</button></p>
<canvas id="c" width="600" height="250"></canvas>
<script>
const ctx = document.getElementById('c').getContext('2d'), S = 200;
// The target the "model" converges to: a gouache-style harbor at dusk (no lighthouse, a red coat,
// as CLIP's loose reading of the prompt produced)
const target = document.createElement('canvas');
target.width = target.height = S;
const g = target.getContext('2d');
const sky = g.createLinearGradient(0, 0, 0, 120);
sky.addColorStop(0, '#1f3f66'); sky.addColorStop(1, '#e8b26a');
g.fillStyle = sky; g.fillRect(0, 0, S, 120);
g.fillStyle = '#16435f'; g.fillRect(0, 120, S, 80);
g.fillStyle = '#7a6a5a'; g.fillRect(0, 140, 120, 24); // harbor wall
g.fillStyle = '#b5452f'; g.fillRect(60, 104, 14, 36); // the girl's red coat
g.fillStyle = '#f2d2b0'; g.beginPath(); g.arc(67, 98, 7, 0, 7); g.fill();
const tpx = g.getImageData(0, 0, S, S).data;
const frame = ctx.createImageData(S, S);
let noise, raf;
function makeNoise(seed) { // the fixed seed makes runs repeatable
let s = seed || 1;
const r = () => (s = (s * 16807) % 2147483647) / 2147483647;
return Float32Array.from({ length: tpx.length }, () => r() * 255);
}
function render(step) {
// A DPM-style schedule: the share of signal rises quickly in early steps (layout, color) and late steps add detail
const k = Math.min(1, (step / 20) ** 0.7);
for (let i = 0; i < tpx.length; i += 4) {
for (let c = 0; c < 3; c++) frame.data[i + c] = noise[i + c] * (1 - k) + tpx[i + c] * k;
frame.data[i + 3] = 255;
}
ctx.putImageData(frame, 10, 10);
}
function panel(step, seconds, total) {
ctx.clearRect(220, 0, 380, 250);
ctx.fillStyle = '#263238'; ctx.font = '12px system-ui';
ctx.fillText(`step ${step} / 20`, 230, 30);
ctx.fillText(`elapsed ${seconds.toFixed(0)} s of ${total} s (${(total / 20).toFixed(1)} s/step)`, 230, 50);
ctx.fillText('each step runs the U-Net twice (guidance_scale 7.0)', 230, 70);
ctx.fillStyle = '#e0d8c8'; ctx.fillRect(230, 84, 340, 14);
ctx.fillStyle = '#1f5f8b'; ctx.fillRect(230, 84, 340 * step / 20, 14);
ctx.fillStyle = '#263238';
['Stable Diffusion 1.5 on the CPU: 512 × 512, free', 'gpt-image-2.5-flare, low: 1024 × 1536, 11 s, $0.0051',
'load 17 s with cached weights (first run: 2.1 GB download)'].forEach((t, i) => ctx.fillText(t, 230, 130 + i * 20));
}
function run() {
cancelAnimationFrame(raf);
noise = makeNoise(+document.getElementById('seed').value);
const total = +document.getElementById('m').value, start = performance.now();
const tick = () => {
const step = Math.min(20, Math.floor((performance.now() - start) / 110)); // replayed fast
render(step); panel(step, step * total / 20, total);
if (step < 20) raf = requestAnimationFrame(tick);
};
tick();
}
document.getElementById('run').onclick = run;
run();
</script>