When GPU-Driven Pays

When GPU-Driven Rendering Is Worth the Complexity

GPU-driven rendering moves decisions from JavaScript into shaders: fast and parallel, but harder to debug, since a wrong count shows up as missing objects rather than an exception. It pays when there are thousands of objects or more, the per-object decision is simple and data-parallel, and the result feeds the GPU rather than the page. BookNest's six covers never needed it; its 1,200-book aisle and its particles are where it starts to earn its keep. Tens of objects that change rarely are better culled in JavaScript, and deep per-object logic belongs there too, with only the results uploaded; when the page needs a count or a pick, add a small read-back a frame late.

Adopt it in steps: cull on the CPU first, move the test into a compute shader and compare through a read-back, then switch to drawIndirect() and delete the read-back. Render bundles (Indirect Draws and Bundles) remove most of the remaining JavaScript cost, and Profiling and Beyond shows how to measure whether the GPU keeps up.

When GPU-driven rendering pays: an illustrative cost curve by object count, and the three-step path to adopt itHTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="360"></canvas>
  <canvas id="labels" width="600" height="360"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

// Two illustrative cost models over log10(objects) from 1 to 100,000, drawn per pixel by a fragment shader:
// CPU culling grows linearly with the count; the GPU path has a fixed setup cost, then grows slowly.
const code = /* wgsl */ `
@vertex fn vs(@builtin(vertex_index) v: u32) -> @builtin(position) vec4f {
  let p = array(vec2f(-1, -1), vec2f(3, -1), vec2f(-1, 3))[v];
  return vec4f(p, 0, 1);
}
fn cpuCost(n: f32) -> f32 { return 0.02 + n * 0.00004; }               // ms, per frame
fn gpuCost(n: f32) -> f32 { return 0.6 + n * 0.0000015; }
@fragment fn fs(@builtin(position) pos: vec4f) -> @location(0) vec4f {
  let x = (pos.x - 50) / 330;                  // 0..1 -> 10^0 .. 10^5 objects
  let y = 1 - (pos.y - 40) / 190;              // 0..1 -> 0 .. 4 ms
  let inside = x >= 0 && x <= 1 && y >= 0 && y <= 1;   // decided at the end: fwidth needs uniform flow
  let n = pow(10, clamp(x, 0, 1) * 5);
  let cpu = cpuCost(n) / 4;  let gpu = gpuCost(n) / 4;
  let dc = y - cpu;  let dg = y - gpu;
  let lineC = 1 - clamp(abs(dc) / fwidth(dc) - 1, 0, 1);
  let lineG = 1 - clamp(abs(dg) / fwidth(dg) - 1, 0, 1);
  var color = select(vec3f(1), vec3f(0.92, 0.96, 0.92), gpu < cpu);    // shade where GPU-driven wins
  let grid = step(0.985, fract(x * 5)) + step(0.985, fract(y * 4));
  color = mix(color, vec3f(0.85), clamp(grid, 0, 1));
  color = mix(color, vec3f(0.85, 0.45, 0.15), lineC);
  color = mix(color, vec3f(0.08, 0.40, 0.75), lineG);
  return vec4f(select(vec3f(0.97, 0.96, 0.93), color, inside), 1);
}`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const device = await adapter.requestDevice();
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });
  const module = device.createShaderModule({ code });
  const pipeline = device.createRenderPipeline({ layout: 'auto', vertex: { module }, fragment: { module, targets: [{ format }] } });
  const encoder = device.createCommandEncoder();
  const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
    clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
  pass.setPipeline(pipeline);
  pass.draw(3);
  pass.end();
  device.queue.submit([encoder.finish()]);

  ink.font = 'bold 12.5px system-ui, sans-serif'; ink.fillStyle = '#222';
  ink.fillText('CPU time per frame for culling (illustrative)', 50, 28);
  ink.font = '11px system-ui, sans-serif'; ink.fillStyle = '#555'; ink.textAlign = 'center';
  ['1', '10', '100', '1k', '10k', '100k'].forEach((t, i) => ink.fillText(t, 50 + i * 66, 246));
  ink.fillText('objects (log scale)', 215, 262);
  ink.fillStyle = '#b06015'; ink.fillText('CPU culling in JavaScript', 150, 110);
  ink.fillStyle = '#1f4f8a'; ink.fillText('GPU-driven: fixed setup, then flat', 300, 190);
  ink.fillStyle = '#1e6b3a'; ink.fillText('green: GPU-driven wins', 330, 60);
  ink.textAlign = 'left'; ink.font = '12px system-ui, sans-serif'; ink.fillStyle = '#222';
  const notes = ['Worth it when:', '- thousands of objects or more', '- a simple, data-parallel test', '- the result feeds the GPU', '',
    'Not worth it for:', '- tens of rarely changing objects', '- deep per-object logic', '',
    'Adopt in steps:', '1. cull on the CPU', '2. same test in compute,', '   compare via read-back', '3. drawIndirect(), drop', '   the read-back'];
  notes.forEach((l, i) => ink.fillText(l, 400, 42 + i * 19));
  ink.fillStyle = '#555'; ink.font = '11px system-ui, sans-serif';
  ink.fillText('The curves are a model, not measurements: GPU-driven bugs show up as missing objects, not exceptions.', 12, 350);
}
main();
</script>