Indirect Plus Bundles

Combining Indirect Drawing with Render Bundles

The limitation of bundles is that their draw list is frozen; the limitation of indirect draws is that someone must write the arguments. Put a drawIndirect() inside a bundle and each covers the other: the bundle is recorded once, and a compute pass before it rewrites the arguments and the instance list every frame. The CPU then submits the same two passes forever, whatever the scene shows:

A GPU-driven frame: the CPU's calls never change, the GPU decides what is drawn
A GPU-driven frame: the CPU's calls never change, the GPU decides what is drawn

This is the core of GPU-driven rendering, where culling, level-of-detail selection and even sorting run on the GPU (Shelf Culling extends it to a frustum-culled shelf).

A GPU-driven frame: a compute pass rewrites the draw arguments, a bundle recorded once holds a single drawIndirect()HTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="340"></canvas>
  <canvas id="labels" width="600" height="340"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

const COUNT = 400;
const code = /* wgsl */ `
struct DrawArgs { vertexCount: u32, instanceCount: atomic<u32>, firstVertex: u32, firstInstance: u32 }
@group(0) @binding(0) var<storage, read_write> args: DrawArgs;
@group(0) @binding(1) var<storage, read_write> visible: array<u32>;
@group(0) @binding(2) var<uniform> time: f32;
fn spot(i: u32) -> vec2f {                                  // 400 books scattered on a floor plan
  let h = i * 2654435761u;
  return vec2f(f32(h & 1023u) / 1023, f32((h >> 10) & 1023u) / 1023) * 2 - 1;
}
@compute @workgroup_size(64) fn cull(@builtin(global_invocation_id) id: vec3u) {
  if (id.x >= ${COUNT}) { return; }
  let lamp = vec2f(cos(time), sin(time * 1.3)) * 0.55;       // the "camera": a moving circle of view
  if (distance(spot(id.x), lamp) < 0.45) {
    visible[atomicAdd(&args.instanceCount, 1)] = id.x;
  }
}
@group(0) @binding(1) var<storage> kept: array<u32>;
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> @builtin(position) vec4f {
  let q = vec2f(f32(v & 1), f32(v >> 1)) - 0.5;
  return vec4f(spot(kept[i]) * vec2f(0.95, 0.8) + q * vec2f(0.03, 0.05), 0, 1);
}
@fragment fn fs() -> @location(0) vec4f { return vec4f(0.08, 0.40, 0.75, 1); }`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const device = await adapter.requestDevice();
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });
  const B = GPUBufferUsage;
  const args = device.createBuffer({ size: 16, usage: B.INDIRECT | B.STORAGE | B.COPY_DST });
  const visible = device.createBuffer({ size: COUNT * 4, usage: B.STORAGE });
  const time = device.createBuffer({ size: 4, usage: B.UNIFORM | B.COPY_DST });
  const module = device.createShaderModule({ code });
  const cull = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint: 'cull' } });
  const draw = device.createRenderPipeline({ layout: 'auto', primitive: { topology: 'triangle-strip' },
    vertex: { module, entryPoint: 'vs' }, fragment: { module, entryPoint: 'fs', targets: [{ format }] } });
  const cullGroup = device.createBindGroup({ layout: cull.getBindGroupLayout(0), entries: [
    { binding: 0, resource: { buffer: args } }, { binding: 1, resource: { buffer: visible } }, { binding: 2, resource: { buffer: time } }] });

  // Recorded once: the bundle's draw list is frozen, but its counts come from 'args'.
  const recorder = device.createRenderBundleEncoder({ colorFormats: [format] });
  recorder.setPipeline(draw);
  recorder.setBindGroup(0, device.createBindGroup({ layout: draw.getBindGroupLayout(0), entries: [{ binding: 1, resource: { buffer: visible } }] }));
  recorder.drawIndirect(args, 0);
  const bundle = recorder.finish();

  ink.font = '12px ui-monospace, monospace'; ink.fillStyle = '#222';
  const calls = ['every frame, the same CPU calls:', '  writeBuffer(time)  writeBuffer(args reset)',
                 '  compute: dispatchWorkgroups(7)', '  render:  executeBundles([bundle])'];
  calls.forEach((l, i) => { ink.fillStyle = 'rgba(247,244,238,0.85)'; ink.fillRect(8, 6 + i * 16, 340, 16); ink.fillStyle = '#222'; ink.fillText(l, 12, 18 + i * 16); });
  ink.fillStyle = '#1f4f8a'; ink.fillText('the GPU decides what is drawn', 12, 330);

  function frame(now) {
    device.queue.writeBuffer(time, 0, new Float32Array([now / 1500]));
    device.queue.writeBuffer(args, 0, new Uint32Array([4, 0, 0, 0]));     // reset the count before culling
    const encoder = device.createCommandEncoder();
    const cp = encoder.beginComputePass();
    cp.setPipeline(cull); cp.setBindGroup(0, cullGroup); cp.dispatchWorkgroups(Math.ceil(COUNT / 64)); cp.end();
    const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
      clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
    pass.executeBundles([bundle]);
    pass.end();
    device.queue.submit([encoder.finish()]);
    requestAnimationFrame(frame);
  }
  requestAnimationFrame(frame);
}
main();
</script>