Barriers

Synchronizing with workgroupBarrier() and storageBarrier()

A barrier makes every invocation in the workgroup wait until all reach it, and makes earlier writes visible to later reads: workgroupBarrier() for workgroup memory, storageBarrier() for storage buffers, textureBarrier() for read-write storage textures, and workgroupUniformLoad(&v), a workgroup barrier that also returns v as a uniform value. All are scoped to one workgroup: storageBarrier() does not wait for other workgroups, so a grid-wide step needs a new dispatch. And since every invocation must reach the same barrier, barriers are allowed only in uniform control flow, checked by the uniformity analysis met in Built-in Functions:

A barrier that only half the workgroup would reachJavaScript
// out: 1 x u32
var<workgroup> t: array<u32, 64>;
@group(0) @binding(0) var<storage, read_write> out: array<u32>;
@compute @workgroup_size(64) fn main(@builtin(local_invocation_index) i: u32) {
  if (i < 32) { t[i] = i; workgroupBarrier(); }
}
Output
5:27 error: 'workgroupBarrier' must only be called from uniform control flow
5:3 info: control flow depends on possibly non-uniform value
5:7 info: builtin 'i' of 'main' may be non-uniform

On a real GPU the subgroups that never arrive would hang, so WGSL rejects the shader; move the barrier out of the if. When a loop bound comes from workgroup memory, let n = workgroupUniformLoad(&count); supplies the proof: a loop over n with barriers inside compiled, while reading count directly did not.

Barriers must be reached by the whole workgroup: two rejected shaders, then a barrier-driven smoothing loop sized by workgroupUniformLoad()HTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="360"></canvas>
  <canvas id="labels" width="600" height="360"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

// Rejected: only half the workgroup would ever reach this barrier.
const halfBarrier = /* wgsl */ `
var<workgroup> t: array<u32, 64>;
@compute @workgroup_size(64) fn main(@builtin(local_invocation_index) i: u32) {
  if (i < 32) { t[i] = i; workgroupBarrier(); }
}`;
// The smoothing loop: its step count lives in workgroup memory.
const smoothing = (loadCount) => /* wgsl */ `
@group(0) @binding(0) var<storage, read_write> data: array<f32>;   // 64 values in, 64 smoothed out
@group(0) @binding(1) var<uniform> requested: u32;
var<workgroup> v: array<f32, 64>;
var<workgroup> steps: u32;
@compute @workgroup_size(64) fn main(@builtin(local_invocation_index) i: u32) {
  v[i] = data[i];
  if (i == 0) { steps = requested; }          // one invocation decides
  let n = ${loadCount};
  for (var s = 0u; s < n; s++) {
    let left = v[max(i, 1u) - 1];  let mid = v[i];  let right = v[min(i + 1, 63u)];
    workgroupBarrier();                       // everyone has read before anyone writes
    v[i] = (left + mid + right) / 3;
    workgroupBarrier();                       // everyone has written before the next read
  }
  data[64 + i] = v[i];
}`;
const view = /* wgsl */ `
@group(0) @binding(0) var<storage> data: array<f32>;
struct Out { @builtin(position) pos: vec4f, @location(0) color: vec3f }
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> Out {
  let q = vec2f(f32(v & 1), f32(v >> 1));
  let smoothed = i >= 64u;
  let k = f32(i % 64);
  let base = select(-0.05, -0.9, smoothed);
  let p = vec2f(-0.95 + k * 0.0297 + q.x * 0.024, base + q.y * data[i] * 0.7);
  return Out(vec4f(p, 0, 1), select(vec3f(0.62, 0.65, 0.70), vec3f(0.08, 0.40, 0.75), smoothed));
}
@fragment fn fs(in: Out) -> @location(0) vec4f { return vec4f(in.color, 1); }`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const device = await adapter.requestDevice();
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });
  async function firstError(code) {              // compile, keep the expected error out of the console
    device.pushErrorScope('validation');
    const { messages } = await device.createShaderModule({ code }).getCompilationInfo();
    await device.popErrorScope();
    return messages.find((m) => m.type === 'error')?.message ?? 'compiled';
  }
  const e1 = await firstError(halfBarrier);
  const e2 = await firstError(smoothing('steps'));  // reading 'steps' directly: not provably uniform

  const B = GPUBufferUsage;
  const input = new Float32Array(128);
  for (let i = 0; i < 64; i++) input[i] = 0.35 + 0.25 * Math.sin(i / 7) + 0.3 * Math.random();   // noisy daily sales
  const data = device.createBuffer({ size: 512, usage: B.STORAGE | B.COPY_DST });
  device.queue.writeBuffer(data, 0, input);
  const requested = device.createBuffer({ size: 4, usage: B.UNIFORM | B.COPY_DST });
  device.queue.writeBuffer(requested, 0, new Uint32Array([6]));
  const compute = device.createComputePipeline({ layout: 'auto',
    compute: { module: device.createShaderModule({ code: smoothing('workgroupUniformLoad(&steps)') }) } });
  const module = device.createShaderModule({ code: view });
  const render = device.createRenderPipeline({ layout: 'auto', primitive: { topology: 'triangle-strip' },
    vertex: { module }, fragment: { module, targets: [{ format }] } });

  const encoder = device.createCommandEncoder();
  const cp = encoder.beginComputePass();
  cp.setPipeline(compute);
  cp.setBindGroup(0, device.createBindGroup({ layout: compute.getBindGroupLayout(0), entries: [
    { binding: 0, resource: { buffer: data } }, { binding: 1, resource: { buffer: requested } }] }));
  cp.dispatchWorkgroups(1);
  cp.end();
  const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
    clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
  pass.setPipeline(render);
  pass.setBindGroup(0, device.createBindGroup({ layout: render.getBindGroupLayout(0), entries: [{ binding: 0, resource: { buffer: data } }] }));
  pass.draw(4, 128);
  pass.end();
  device.queue.submit([encoder.finish()]);

  ink.font = '11.5px ui-monospace, monospace'; ink.fillStyle = '#8a2b2b';
  ink.fillText(`barrier inside if (i < 32): ${e1}`.slice(0, 92), 10, 18);
  ink.fillText(`loop bound read from var<workgroup>: ${e2}`.slice(0, 92), 10, 36);
  ink.font = '12px system-ui, sans-serif'; ink.fillStyle = '#444';
  ink.fillText('input: 64 noisy values in one workgroup', 10, 60);
  ink.fillStyle = '#1f4f8a';
  ink.fillText('after 6 smoothing steps: let n = workgroupUniformLoad(&steps), two barriers per step', 10, 250);
}
main();
</script>