Pipeline Caching

Pipeline Caching and Compilation Hints

Pipeline creation, which compiles shaders, is WebGPU's slowest call. Chrome 1 caches compiled pipelines in memory, and can keep compiled shaders in its GPU disk cache, so an identical descriptor is nearly free the second time. Timing createRenderPipelineAsync() on this shared machine, with a shader of two small loops: the first pipeline took 111.7 ms, an identical descriptor 1.1 ms, the same module with a new override value 27.8 ms, and that descriptor again 1.0 ms.

An identical descriptor came back about 100 times faster; a new override value compiled a new variant, but faster than the first, since the module was already parsed. A second run of the page, in a fresh browser profile, started at 37.7 ms, probably helped by the GPU driver's own shader cache. createShaderModule() also accepts compilationHints: [{ entryPoint, layout }] to let a browser compile early; the specification defines no observable effect, so treat it as optional. Create pipelines at startup with createRenderPipelineAsync(), keep variants few, and never create one per frame.

Timing createRenderPipelineAsync(): a first compile, an identical descriptor, a new override value, and that variant againHTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="340"></canvas>
  <canvas id="labels" width="600" height="340"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

// A shader with two small loops and an override; a random seed in a comment makes the
// first compile genuinely new on every page load (no disk-cache hit from an earlier visit).
const code = /* wgsl */ `// seed ${Math.random()}
override stripes: f32 = 6;
struct Out { @builtin(position) pos: vec4f, @location(0) uv: vec2f }
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> Out {
  let q = vec2f(f32(v & 1), f32(v >> 1));
  return Out(vec4f(-0.95 + f32(i) * 0.5 + q.x * 0.42, 0.15 + q.y * 0.65, 0, 1), q);
}
@fragment fn fs(in: Out) -> @location(0) vec4f {
  var c = vec3f(0);
  for (var k = 0; k < 8; k++) { c += vec3f(0.08, 0.40, 0.75) * (0.1 + 0.02 * f32(k)); }
  for (var k = 0; k < 4; k++) { c *= 0.97 + 0.03 * step(0.5, fract(in.uv.y * stripes)); }
  return vec4f(c, 1);
}`;
const bars = /* wgsl */ `
@group(0) @binding(0) var<storage> ms: array<f32>;
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> @builtin(position) vec4f {
  let q = vec2f(f32(v & 1), f32(v >> 1));
  let w = min(ms[i] / max(max(ms[0], ms[2]), 0.001), 1.0) * 1.1;
  return vec4f(-0.35 + q.x * w, -0.12 - f32(i) * 0.2 - q.y * 0.12, 0, 1);
}
@fragment fn fs() -> @location(0) vec4f { return vec4f(0.85, 0.45, 0.15, 1); }`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const device = await adapter.requestDevice();
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });
  const module = device.createShaderModule({ code });
  const descriptor = (stripes) => ({ layout: 'auto', primitive: { topology: 'triangle-strip' },
    vertex: { module }, fragment: { module, targets: [{ format }], constants: { stripes } } });
  async function timed(stripes) {
    const t0 = performance.now();
    const pipeline = await device.createRenderPipelineAsync(descriptor(stripes));
    return [pipeline, performance.now() - t0];
  }
  const runs = [['first pipeline', 6], ['identical descriptor', 6], ['new override value', 20], ['that variant again', 20]];
  const results = [];
  for (const [name, stripes] of runs) results.push([name, ...(await timed(stripes))]);

  const times = device.createBuffer({ size: 16, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST });
  device.queue.writeBuffer(times, 0, new Float32Array(results.map((r) => r[2])));
  const barModule = device.createShaderModule({ code: bars });
  const barPipeline = device.createRenderPipeline({ layout: 'auto', primitive: { topology: 'triangle-strip' },
    vertex: { module: barModule }, fragment: { module: barModule, targets: [{ format }] } });

  const encoder = device.createCommandEncoder();
  const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
    clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
  results.forEach(([, pipeline], i) => { pass.setPipeline(pipeline); pass.draw(4, 1, 0, i); });
  pass.setPipeline(barPipeline);
  pass.setBindGroup(0, device.createBindGroup({ layout: barPipeline.getBindGroupLayout(0), entries: [{ binding: 0, resource: { buffer: times } }] }));
  pass.draw(4, 4);
  pass.end();
  device.queue.submit([encoder.finish()]);

  ink.font = '12px system-ui, sans-serif'; ink.fillStyle = '#222';
  results.forEach(([name, , ms], i) => {
    ink.textAlign = 'center'; ink.fillText(`${i + 1}. ${name}`, 63 + i * 150, 18);
    ink.textAlign = 'right'; ink.fillText(`${name}: ${ms.toFixed(1)} ms`, 190, 198 + i * 34);
  });
  ink.textAlign = 'left'; ink.fillStyle = '#444';
  ink.fillText('Identical descriptors come back from the cache; a new override value compiles a new variant.', 12, 326);
}
main();
</script>