GPU-driven rendering moves decisions from JavaScript into shaders: fast and parallel, but harder to debug, since a wrong count shows up as missing objects rather than an exception. It pays when there are thousands of objects or more, the per-object decision is simple and data-parallel, and the result feeds the GPU rather than the page. BookNest's six covers never needed it; its 1,200-book aisle and its particles are where it starts to earn its keep. Tens of objects that change rarely are better culled in JavaScript, and deep per-object logic belongs there too, with only the results uploaded; when the page needs a count or a pick, add a small read-back a frame late.
Adopt it in steps: cull on the CPU first, move the test into a compute shader and compare through a read-back, then switch to drawIndirect() and delete the read-back. Render bundles (Indirect Draws and Bundles) remove most of the remaining JavaScript cost, and Profiling and Beyond shows how to measure whether the GPU keeps up.
<!doctype html>
<style>
body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
.stage { position: relative; width: 100%; max-width: 600px; }
.stage canvas { display: block; width: 100%; }
.stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
<canvas id="view" width="600" height="360"></canvas>
<canvas id="labels" width="600" height="360"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');
function showMessage(text) { // 2D fallback when WebGPU is missing
const ctx = canvas.getContext('2d');
ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}
// Two illustrative cost models over log10(objects) from 1 to 100,000, drawn per pixel by a fragment shader:
// CPU culling grows linearly with the count; the GPU path has a fixed setup cost, then grows slowly.
const code = /* wgsl */ `
@vertex fn vs(@builtin(vertex_index) v: u32) -> @builtin(position) vec4f {
let p = array(vec2f(-1, -1), vec2f(3, -1), vec2f(-1, 3))[v];
return vec4f(p, 0, 1);
}
fn cpuCost(n: f32) -> f32 { return 0.02 + n * 0.00004; } // ms, per frame
fn gpuCost(n: f32) -> f32 { return 0.6 + n * 0.0000015; }
@fragment fn fs(@builtin(position) pos: vec4f) -> @location(0) vec4f {
let x = (pos.x - 50) / 330; // 0..1 -> 10^0 .. 10^5 objects
let y = 1 - (pos.y - 40) / 190; // 0..1 -> 0 .. 4 ms
let inside = x >= 0 && x <= 1 && y >= 0 && y <= 1; // decided at the end: fwidth needs uniform flow
let n = pow(10, clamp(x, 0, 1) * 5);
let cpu = cpuCost(n) / 4; let gpu = gpuCost(n) / 4;
let dc = y - cpu; let dg = y - gpu;
let lineC = 1 - clamp(abs(dc) / fwidth(dc) - 1, 0, 1);
let lineG = 1 - clamp(abs(dg) / fwidth(dg) - 1, 0, 1);
var color = select(vec3f(1), vec3f(0.92, 0.96, 0.92), gpu < cpu); // shade where GPU-driven wins
let grid = step(0.985, fract(x * 5)) + step(0.985, fract(y * 4));
color = mix(color, vec3f(0.85), clamp(grid, 0, 1));
color = mix(color, vec3f(0.85, 0.45, 0.15), lineC);
color = mix(color, vec3f(0.08, 0.40, 0.75), lineG);
return vec4f(select(vec3f(0.97, 0.96, 0.93), color, inside), 1);
}`;
async function main() {
const adapter = await navigator.gpu?.requestAdapter();
if (!adapter) return showMessage('WebGPU is not available in this browser');
const device = await adapter.requestDevice();
const context = canvas.getContext('webgpu');
const format = navigator.gpu.getPreferredCanvasFormat();
context.configure({ device, format });
const module = device.createShaderModule({ code });
const pipeline = device.createRenderPipeline({ layout: 'auto', vertex: { module }, fragment: { module, targets: [{ format }] } });
const encoder = device.createCommandEncoder();
const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
pass.setPipeline(pipeline);
pass.draw(3);
pass.end();
device.queue.submit([encoder.finish()]);
ink.font = 'bold 12.5px system-ui, sans-serif'; ink.fillStyle = '#222';
ink.fillText('CPU time per frame for culling (illustrative)', 50, 28);
ink.font = '11px system-ui, sans-serif'; ink.fillStyle = '#555'; ink.textAlign = 'center';
['1', '10', '100', '1k', '10k', '100k'].forEach((t, i) => ink.fillText(t, 50 + i * 66, 246));
ink.fillText('objects (log scale)', 215, 262);
ink.fillStyle = '#b06015'; ink.fillText('CPU culling in JavaScript', 150, 110);
ink.fillStyle = '#1f4f8a'; ink.fillText('GPU-driven: fixed setup, then flat', 300, 190);
ink.fillStyle = '#1e6b3a'; ink.fillText('green: GPU-driven wins', 330, 60);
ink.textAlign = 'left'; ink.font = '12px system-ui, sans-serif'; ink.fillStyle = '#222';
const notes = ['Worth it when:', '- thousands of objects or more', '- a simple, data-parallel test', '- the result feeds the GPU', '',
'Not worth it for:', '- tens of rarely changing objects', '- deep per-object logic', '',
'Adopt in steps:', '1. cull on the CPU', '2. same test in compute,', ' compare via read-back', '3. drawIndirect(), drop', ' the read-back'];
notes.forEach((l, i) => ink.fillText(l, 400, 42 + i * 19));
ink.fillStyle = '#555'; ink.font = '11px system-ui, sans-serif';
ink.fillText('The curves are a model, not measurements: GPU-driven bugs show up as missing objects, not exceptions.', 12, 350);
}
main();
</script>