Bind Group Trade-offs

One Bind Group Versus Many: a Performance Trade-off

Ten thousand small triangles, each with its own 32-byte uniform, were drawn five ways in one render pass, timed to submit() (JavaScript cost) and to onSubmittedWorkDone(); medians of 20 frames on the shared machine:

CPU and total cost of per-object data (Chrome 154 1 , GTX 1650)
10,000 objects Encode and submit To completion
Pre-built bind group per object 1.3 ms 6.5 ms
One group, dynamic offsets as [offset] 2.8 ms 6.8 ms
One group, offsets from a Uint32Array 1.5 ms 5.8 ms
One draw, 10,000 instances from a storage buffer under 0.1 ms 3.1 ms
A new bind group created per draw 17.3 ms 32.0 ms

Pre-built groups and dynamic offsets cost about the same once the offsets array is not allocated per call, but the dynamic version keeps one group instead of 10,000. Creating groups in the loop was 13 times slower, and instancing removed the loop altogether. Organize groups by update frequency (group 0 per frame, 1 per material, 2 per object) so a draw loop changes only the last.

The book's five ways to feed 10,000 per-object uniforms as bars, over the fastest one running live: one instanced drawHTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="360"></canvas>
  <canvas id="labels" width="600" height="360"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

// Medians of 20 frames, 10,000 objects (Chrome 154, GTX 1650): encode and submit, to completion.
const methods = [['Pre-built bind group per object', 1.3, 6.5], ['One group, offsets as [offset]', 2.8, 6.8],
  ['One group, offsets from a Uint32Array', 1.5, 5.8], ['One draw, 10,000 instances', 0.08, 3.1], ['A new bind group per draw', 17.3, 32.0]];
const N = 10000;

// Live: 10,000 small triangles, each with its own 32-byte record, in one instanced draw.
const code = /* wgsl */ `
struct Object { offset: vec2f, spin: f32, hue: f32, pad: vec4f }     // 32 bytes each
@group(0) @binding(0) var<storage> objects: array<Object>;
@group(0) @binding(1) var<uniform> time: f32;
struct Out { @builtin(position) pos: vec4f, @location(0) color: vec3f }
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> Out {
  let o = objects[i];
  let a = time * o.spin + f32(v) * 2.094;
  let p = o.offset + vec2f(cos(a), sin(a)) * vec2f(0.012, 0.02);
  let c = vec3f(0.08, 0.40, 0.75) * (1 - o.hue) + vec3f(0.85, 0.55, 0.20) * o.hue;
  return Out(vec4f(p, 0, 1), mix(c, vec3f(0.97, 0.96, 0.93), 0.72));   // pale, as a backdrop
}
@fragment fn fs(in: Out) -> @location(0) vec4f { return vec4f(in.color, 1); }`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const device = await adapter.requestDevice();
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });

  const records = new Float32Array(N * 8);
  for (let i = 0; i < N; i++) records.set([Math.random() * 2 - 1, Math.random() * 2 - 1, Math.random() * 4 - 2, Math.random()], i * 8);
  const objects = device.createBuffer({ size: records.byteLength, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST });
  device.queue.writeBuffer(objects, 0, records);
  const time = device.createBuffer({ size: 4, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST });
  const module = device.createShaderModule({ code });
  const pipeline = device.createRenderPipeline({ layout: 'auto', vertex: { module }, fragment: { module, targets: [{ format }] } });
  const group = device.createBindGroup({ layout: pipeline.getBindGroupLayout(0), entries: [
    { binding: 0, resource: { buffer: objects } }, { binding: 1, resource: { buffer: time } }] });

  const maxMs = 32;
  const bar = (x, y, ms, color) => { ink.fillStyle = color; ink.fillRect(x, y, Math.max(ms / maxMs * 330, 2), 12); };
  function drawChart(encodeMs) {
    ink.clearRect(0, 0, 600, 360);
    ink.font = 'bold 13px system-ui, sans-serif'; ink.fillStyle = '#222';
    ink.fillText('10,000 objects with their own uniforms: ms per frame', 12, 22);
    methods.forEach(([name, encode, total], i) => {
      const y = 48 + i * 50;
      ink.font = '12px system-ui, sans-serif'; ink.fillStyle = '#222'; ink.fillText(name, 12, y);
      bar(12, y + 6, encode, '#1f5f8b'); bar(12, y + 20, total, '#e0990f');
      ink.fillStyle = '#333'; ink.fillText(`${encode < 0.1 ? '< 0.1' : encode} / ${total}`, 20 + Math.max(encode, total) / maxMs * 330, y + 26);
    });
    ink.fillStyle = '#1f5f8b'; ink.fillRect(12, 312, 12, 12); ink.fillStyle = '#333'; ink.fillText('encode and submit (JavaScript)', 30, 322);
    ink.fillStyle = '#e0990f'; ink.fillRect(230, 312, 12, 12); ink.fillStyle = '#333'; ink.fillText('to completion', 248, 322);
    ink.fillStyle = '#1e6b3a'; ink.fillText(`Behind the chart, live: all 10,000 in one draw(3, 10000), encoded in ${encodeMs.toFixed(2)} ms here.`, 12, 348);
  }

  let last = 0;
  function frame(now) {
    device.queue.writeBuffer(time, 0, new Float32Array([now / 1000]));
    const t0 = performance.now();
    const encoder = device.createCommandEncoder();
    const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
      clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
    pass.setPipeline(pipeline);
    pass.setBindGroup(0, group);
    pass.draw(3, N);                                   // instancing removes the loop altogether
    pass.end();
    device.queue.submit([encoder.finish()]);
    if (now - last > 500) { drawChart(performance.now() - t0); last = now; }
    requestAnimationFrame(frame);
  }
  requestAnimationFrame(frame);
}
main();
</script>