Performance Best Practices

WebGPU moves cost from every draw to object creation, so most advice is about doing expensive things once. The evidence comes from this chapter's own measurements on the shared GTX 1650:

Performance habits and what they saved in this chapter
Practice Measured here
Create bind groups once, never per draw 1.3 vs 17.3 ms for 10,000 draws (Bind Group Trade-offs)
Instance, or draw from a storage buffer Under 0.1 ms of JavaScript (Bind Group Trade-offs)
Replay static draws as render bundles 0.4-1.2 down to 0.1 ms (When Bundles Help)
Keep data on the GPU 9-14 ms upload vs 0.13 ms reduction (Two-Pass Reduction)
Use workgroups of 64 or more 0.26 vs 16.4 ms at size 1 (Compute and Workgroups)

Also build pipelines at startup with the async calls, pool buffers, use storeOp: 'discard' for attachments you never read (Depth and Multisampling), and submit once per frame. Then measure again: once JavaScript is cheap, fragment work and memory bandwidth usually dominate.

Five performance habits and what each saved in the chapter's measurements, as before-and-after bars on a log scaleHTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="370"></canvas>
  <canvas id="labels" width="600" height="370"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

// practice, slow way (ms), fast way (ms), what was compared (the chapter's GTX 1650 measurements)
const practices = [
  ['Create bind groups once, never per draw', 17.3, 1.3, '10,000 draws: new group per draw vs pre-built'],
  ['Instance, or draw from a storage buffer', 1.3, 0.1, 'JavaScript per frame: 10,000 draws vs one instanced draw'],
  ['Replay static draws as render bundles', 1.2, 0.1, 'encode + submit, 10,000 draws with group changes'],
  ['Keep data on the GPU', 14, 0.13, 'uploading 16 MiB vs reducing it where it already lives'],
  ['Use workgroups of 64 or more', 16.4, 0.26, 'one pass over 4M floats: size 1 vs size 64']];
const code = /* wgsl */ `
struct Bar { rect: vec4f, color: vec4f }
@group(0) @binding(0) var<storage> bars: array<Bar>;
struct Out { @builtin(position) pos: vec4f, @location(0) color: vec3f, @location(1) u: f32 }
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> Out {
  let q = vec2f(f32(v & 1), f32(v >> 1));
  let px = bars[i].rect.xy + q * bars[i].rect.zw;
  return Out(vec4f(px.x / 300 - 1, 1 - px.y / 185, 0, 1), bars[i].color.rgb, q.x);
}
@fragment fn fs(in: Out) -> @location(0) vec4f { return vec4f(in.color * (0.85 + 0.15 * in.u), 1); }`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const device = await adapter.requestDevice();
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });
  const scale = (ms) => (Math.log10(ms) + 1.5) / 3 * 400;   // 0.03 ms .. 30 ms across 400 px
  const bars = [];
  practices.forEach(([, slow, fast], i) => {
    const y = 58 + i * 60;
    bars.push([150, y, scale(slow), 14, 0.85, 0.40, 0.25, 1], [150, y + 17, scale(fast), 14, 0.16, 0.56, 0.30, 1]);
  });
  const module = device.createShaderModule({ code });
  const pipeline = device.createRenderPipeline({ layout: 'auto', primitive: { topology: 'triangle-strip' },
    vertex: { module }, fragment: { module, targets: [{ format }] } });
  const data = new Float32Array(bars.flat());
  const buffer = device.createBuffer({ size: data.byteLength, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST });
  device.queue.writeBuffer(buffer, 0, data);
  const encoder = device.createCommandEncoder();
  const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
    clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
  pass.setPipeline(pipeline);
  pass.setBindGroup(0, device.createBindGroup({ layout: pipeline.getBindGroupLayout(0), entries: [{ binding: 0, resource: { buffer } }] }));
  pass.draw(4, bars.length);
  pass.end();
  device.queue.submit([encoder.finish()]);

  ink.font = 'bold 13px system-ui, sans-serif'; ink.fillStyle = '#222';
  ink.fillText('What each habit saved in this chapter (ms, log scale)', 12, 24);
  practices.forEach(([name, slow, fast, what], i) => {
    const y = 58 + i * 60;
    ink.font = 'bold 12px system-ui, sans-serif'; ink.fillStyle = '#222'; ink.fillText(name, 150, y - 5);
    ink.font = '11px system-ui, sans-serif'; ink.fillStyle = '#555';
    ink.fillText(`${slow}`, 156 + scale(slow), y + 11); ink.fillText(`${fast}`, 156 + scale(fast), y + 28);
    ink.textAlign = 'right'; ink.fillText(`${Math.round(slow / fast)}x`, 140, y + 21); ink.textAlign = 'left';
  });
  ink.font = '11.5px system-ui, sans-serif'; ink.fillStyle = '#444';
  ink.fillText('Also: async pipelines at startup, pooled buffers, storeOp "discard", one submit per frame, then measure again.', 12, 362);
}
main();
</script>