A compute pass can size the next one. When a filter keeps an unknown number of items, a small shader writes ceil(count / 64) into a buffer, and the next pass processes exactly that many workgroups:
const dispatchArgs = device.createBuffer({ size: 12,
usage: GPUBufferUsage.INDIRECT | GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST });
device.queue.writeBuffer(dispatchArgs, 0, new Uint32Array([3, 2, 1])); // or a shader
const pass = encoder.beginComputePass();
pass.setPipeline(pipeline), pass.setBindGroup(0, group);
pass.dispatchWorkgroupsIndirect(dispatchArgs, 0); // x, y, z read on the GPU
pass.end();With @workgroup_size(64) and a counter incremented by every invocation, the arguments (3, 2, 1) ran 384 invocations. A count above maxComputeWorkgroupsPerDimension (65,535 by default) makes the GPU skip the dispatch silently, so clamp counts in the shader that writes them. GPGPU Patterns and Particles and Culling use this for prefix sums and particle systems whose live count changes every frame.
<!doctype html>
<style>
body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
.stage { position: relative; width: 100%; max-width: 600px; }
.stage canvas { display: block; width: 100%; }
.stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
<canvas id="view" width="600" height="330"></canvas>
<canvas id="labels" width="600" height="330"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');
function showMessage(text) { // 2D fallback when WebGPU is missing
const ctx = canvas.getContext('2d');
ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}
const CELLS = 1024; // room for 16 workgroups of 64
const code = /* wgsl */ `
@group(0) @binding(0) var<storage, read_write> dispatchArgs: array<u32, 3>; // INDIRECT | STORAGE
@group(0) @binding(1) var<storage, read_write> cells: array<u32>;
@group(0) @binding(2) var<uniform> count: u32; // items a filter kept
@compute @workgroup_size(1) fn size() {
dispatchArgs = array(min((count + 63) / 64, 65535u), 1u, 1u); // clamp: over the limit the GPU skips silently
}
@compute @workgroup_size(64) fn work(@builtin(global_invocation_id) g: vec3u, @builtin(workgroup_id) w: vec3u) {
cells[g.x] = select(w.x + 1, 999u, g.x >= count); // 999: an idle lane of the last workgroup
}
@compute @workgroup_size(64) fn clear(@builtin(global_invocation_id) g: vec3u) { cells[g.x] = 0; }
@group(0) @binding(1) var<storage> shown: array<u32>;
struct Out { @builtin(position) pos: vec4f, @location(0) color: vec3f }
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> Out {
let q = vec2f(f32(v & 1), f32(v >> 1));
let px = vec2f(12 + f32(i % 64) * 9, 60 + f32(i / 64) * 14) + q * vec2f(8, 12);
let c = shown[i];
var color = vec3f(0.88, 0.87, 0.84); // not launched
if (c == 999u) { color = vec3f(0.98, 0.80, 0.60); } // launched, past the end
else if (c > 0u) { let h = f32(c % 2); color = mix(vec3f(0.08, 0.40, 0.75), vec3f(0.16, 0.56, 0.30), h); }
return Out(vec4f(px.x / 300 - 1, 1 - px.y / 165, 0, 1), color);
}
@fragment fn fs(in: Out) -> @location(0) vec4f { return vec4f(in.color, 1); }`;
async function main() {
const adapter = await navigator.gpu?.requestAdapter();
if (!adapter) return showMessage('WebGPU is not available in this browser');
const device = await adapter.requestDevice();
const context = canvas.getContext('webgpu');
const format = navigator.gpu.getPreferredCanvasFormat();
context.configure({ device, format });
const B = GPUBufferUsage;
const dispatchArgs = device.createBuffer({ size: 12, usage: B.INDIRECT | B.STORAGE | B.COPY_DST });
const cells = device.createBuffer({ size: CELLS * 4, usage: B.STORAGE });
const count = device.createBuffer({ size: 4, usage: B.UNIFORM | B.COPY_DST });
const module = device.createShaderModule({ code });
// Separate groups per pipeline: 'work' must not bind dispatchArgs as storage while the same
// dispatch reads it as INDIRECT (a writable usage and another usage in one scope is an error).
const pipelineFor = (entryPoint) => device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint } });
const [size, work, clear] = ['size', 'work', 'clear'].map(pipelineFor);
const resources = [{ buffer: dispatchArgs }, { buffer: cells }, { buffer: count }];
const groupFor = (pipeline, bindings) => device.createBindGroup({ layout: pipeline.getBindGroupLayout(0),
entries: bindings.map((binding) => ({ binding, resource: resources[binding] })) });
const sizeGroup = groupFor(size, [0, 2]), workGroup = groupFor(work, [1, 2]), clearGroup = groupFor(clear, [1]);
const render = device.createRenderPipeline({ layout: 'auto', primitive: { topology: 'triangle-strip' },
vertex: { module, entryPoint: 'vs' }, fragment: { module, entryPoint: 'fs', targets: [{ format }] } });
const renderGroup = device.createBindGroup({ layout: render.getBindGroupLayout(0), entries: [{ binding: 1, resource: { buffer: cells } }] });
let tick = -1, n = 0;
function frame(now) {
const t = Math.floor(now / 900);
if (t !== tick) { tick = t; n = 40 + ((t * 277) % 960); device.queue.writeBuffer(count, 0, new Uint32Array([n])); }
const encoder = device.createCommandEncoder();
const cp = encoder.beginComputePass();
cp.setPipeline(clear); cp.setBindGroup(0, clearGroup); cp.dispatchWorkgroups(CELLS / 64);
cp.setPipeline(size); cp.setBindGroup(0, sizeGroup); cp.dispatchWorkgroups(1); // the GPU sizes the next dispatch...
cp.setPipeline(work); cp.setBindGroup(0, workGroup); cp.dispatchWorkgroupsIndirect(dispatchArgs, 0); // ...read on the GPU
cp.end();
const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
pass.setPipeline(render); pass.setBindGroup(0, renderGroup); pass.draw(4, CELLS);
pass.end();
device.queue.submit([encoder.finish()]);
ink.clearRect(0, 0, 600, 330);
ink.font = '12.5px ui-monospace, monospace'; ink.fillStyle = '#222';
ink.fillText(`count = ${n} kept items -> dispatchArgs = [${Math.ceil(n / 64)}, 1, 1] (${Math.ceil(n / 64) * 64} invocations)`, 12, 24);
ink.font = '11.5px system-ui, sans-serif'; ink.fillStyle = '#555';
ink.fillText('One cell per invocation; blue and green alternate by workgroup. Orange: idle lanes of the last workgroup.', 12, 44);
ink.fillText('Grey: never launched. The JavaScript is identical every frame; only the GPU-written count changes.', 12, 300);
requestAnimationFrame(frame);
}
requestAnimationFrame(frame);
}
main();
</script>