Uniform Ring Buffers

A Uniform Buffer Ring for Double Buffering

In OpenGL, updating a buffer the GPU is still reading stalls, so engines rotate two or three buffers. WebGPU's writeBuffer() is ordered on the queue, so one buffer is always correct; a ring pays off only for large data, where writing into mapped memory saves the browser's extra copy. MAP_WRITE buffers can only be copy sources, so the ring holds staging buffers, remapped with mapAsync() once the GPU has used them:

A ring of mapped staging buffers feeding a uniform bufferJavaScript
const device = await (await navigator.gpu.requestAdapter()).requestDevice();
const B = GPUBufferUsage, SIZE = 64 * 1024;          // 64 KiB of per-object data a frame
const uniforms = device.createBuffer({ size: SIZE, usage: B.UNIFORM | B.COPY_DST });
const ring = [];                                     // mapped staging buffers, ready to fill
let created = 0;
function upload(fill) {
  const staging = ring.pop() ?? (created++, device.createBuffer({ size: SIZE,
    usage: B.MAP_WRITE | B.COPY_SRC, mappedAtCreation: true }));
  fill(new Float32Array(staging.getMappedRange()));  // write straight into GPU-visible memory
  staging.unmap();
  const encoder = device.createCommandEncoder();
  encoder.copyBufferToBuffer(staging, uniforms);     // whole buffer: 64 KiB
  device.queue.submit([encoder.finish()]);
  staging.mapAsync(GPUMapMode.WRITE).then(() => ring.push(staging));  // back when idle
}
for (let frame = 0; frame < 120; frame++) {
  upload((floats) => floats.fill(frame));            // new data every frame
  await new Promise(requestAnimationFrame);
}
console.log(`120 frames, ${created} staging buffers created, ${ring.length} idle now`);
Output
120 frames, 2 staging buffers created, 2 idle now

Over 120 frames at 64 KiB each, the ring never needed more than two buffers: each came back mapped before the next frame. Under GPU load it grows to three or four, the classic triple buffering, bounded by how fast the GPU finishes. For a few kilobytes per frame, writeBuffer() is simpler and as fast.

A ring of mapped staging buffers feeding a 64 KiB uniform every frame, with the ring's size and idle count shown liveHTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="340"></canvas>
  <canvas id="labels" width="600" height="340"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

const SIZE = 64 * 1024, CELLS = SIZE / 16;         // 64 KiB = 4,096 vec4f of per-object data
const code = /* wgsl */ `
@group(0) @binding(0) var<uniform> cells: array<vec4f, ${CELLS}>;   // the whole 64 KiB uniform
struct Out { @builtin(position) pos: vec4f, @location(0) color: vec3f }
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> Out {
  let q = vec2f(f32(v & 1), f32(v >> 1));
  let cell = vec2f(f32(i % 64), f32(i / 64));                        // a 64 x 64 grid
  let c = cells[i];
  let px = vec2f(80, 60) + cell * 4 + q * 3.4;
  return Out(vec4f(px.x / 300 - 1, 1 - px.y / 170, 0, 1), c.rgb);
}
@fragment fn fs(in: Out) -> @location(0) vec4f { return vec4f(in.color, 1); }`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const device = await adapter.requestDevice();
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });
  const B = GPUBufferUsage;
  const uniforms = device.createBuffer({ size: SIZE, usage: B.UNIFORM | B.COPY_DST });
  const ring = [];                                   // mapped staging buffers, ready to fill
  let created = 0;
  function upload(fill, encoder) {
    const staging = ring.pop() ?? (created++, device.createBuffer({ size: SIZE,
      usage: B.MAP_WRITE | B.COPY_SRC, mappedAtCreation: true }));
    fill(new Float32Array(staging.getMappedRange()));  // write straight into GPU-visible memory
    staging.unmap();
    encoder.copyBufferToBuffer(staging, 0, uniforms, 0, SIZE);
    return staging;                                   // remapped after the submit
  }

  const module = device.createShaderModule({ code });
  const pipeline = device.createRenderPipeline({ layout: 'auto', primitive: { topology: 'triangle-strip' },
    vertex: { module }, fragment: { module, targets: [{ format }] } });
  const group = device.createBindGroup({ layout: pipeline.getBindGroupLayout(0), entries: [{ binding: 0, resource: { buffer: uniforms } }] });

  let frames = 0;
  function frame(now) {
    const t = now / 1000;
    const encoder = device.createCommandEncoder();
    const used = upload((floats) => {                 // new data every frame: a ripple of colours
      for (let i = 0; i < CELLS; i++) {
        const x = (i % 64) - 32, y = Math.floor(i / 64) - 32;
        const w = 0.5 + 0.5 * Math.sin(Math.hypot(x, y) * 0.35 - t * 3);
        floats[i * 4] = 0.08 + 0.8 * w; floats[i * 4 + 1] = 0.40 + 0.2 * w; floats[i * 4 + 2] = 0.75 - 0.5 * w;
      }
    }, encoder);
    const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
      clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
    pass.setPipeline(pipeline);
    pass.setBindGroup(0, group);
    pass.draw(4, CELLS);
    pass.end();
    device.queue.submit([encoder.finish()]);
    used.mapAsync(GPUMapMode.WRITE).then(() => ring.push(used));   // back in the ring when the GPU is done
    if (++frames % 10 === 0) {
      ink.clearRect(340, 0, 260, 340);
      ink.font = '12.5px system-ui, sans-serif'; ink.fillStyle = '#222';
      ['64 KiB of per-object data a frame', 'written into mapped staging memory,', 'copied to one uniform buffer', '',
       `frames: ${frames}`, `staging buffers created: ${created}`, `idle in the ring now: ${ring.length}`, '',
       'Two or three buffers suffice: each', 'comes back mapped before it is', 'needed again. For a few KB,', 'writeBuffer() is simpler.']
        .forEach((line, k) => ink.fillText(line, 356, 70 + k * 19));
      ink.fillStyle = '#1f4f8a';
      for (let k = 0; k < created; k++) { ink.fillRect(356 + k * 30, 310, 24, 16); }
      ink.fillStyle = '#9fe0a8';
      for (let k = 0; k < ring.length; k++) { ink.fillRect(360 + k * 30, 314, 16, 8); }
    }
    requestAnimationFrame(frame);
  }
  requestAnimationFrame(frame);
}
main();
</script>