Subgroup Size

Querying Subgroup Size Portably

The specification allows any power of two from 4 to 128, possibly varying by shader. Khronos's subgroup tutorial cites 32 for NVIDIA and 64 for AMD. Read the range from adapter.info.subgroupMinSize and subgroupMaxSize (32 and 32 here, adapter.info), the size a shader got from the subgroup_size built-in, and demand one with the @subgroup_size attribute of the subgroup-size-control feature:

Fixing the subgroup size (requires the subgroup-size-control feature)JavaScript
enable subgroup_size_control;                   // also enables subgroups
@group(0) @binding(0) var<storage, read_write> out: array<u32>;
@compute @workgroup_size(64) @subgroup_size(32)  // 64 must be a multiple of 32
fn main(@builtin(subgroup_size) size: u32) { out[0] = size; }

This compiled and ran here; @subgroup_size(16) compiled but failed pipeline creation with "The subgroup_size attribute (16) is not in the allowed range ([32, 32])". Portable code sizes its workgroup-memory scratch for the smallest subgroup (workgroupSize / subgroupMinSize partial results), loops over lanes with subgroup_size rather than a constant 32, and keeps a workgroup-memory path for devices without subgroups at all.

The subgroup size range from adapter.info, and a 256-invocation workgroup drawn lane by lane as the GPU split itHTMLLive
<!doctype html>
<style>
  body { margin: 0; background: #f7f4ee; font: 14px system-ui, sans-serif; }
  .stage { position: relative; width: 100%; max-width: 600px; }
  .stage canvas { display: block; width: 100%; }
  .stage canvas + canvas { position: absolute; inset: 0; pointer-events: none; }
</style>
<div class="stage">
  <canvas id="view" width="600" height="340"></canvas>
  <canvas id="labels" width="600" height="340"></canvas>
</div>
<script>
const canvas = document.getElementById('view');
const ink = document.getElementById('labels').getContext('2d');

function showMessage(text) {                     // 2D fallback when WebGPU is missing
  const ctx = canvas.getContext('2d');
  ctx.fillStyle = '#fbeaea'; ctx.fillRect(0, 0, canvas.width, canvas.height);
  ctx.fillStyle = '#8a2b2b'; ctx.font = '18px system-ui, sans-serif'; ctx.textAlign = 'center';
  ctx.fillText(text, canvas.width / 2, canvas.height / 2);
}

const WG = 256;
// Every invocation records the size it got and its lane; the loop uses subgroup_size, not 32.
const code = /* wgsl */ `
enable subgroups;
@group(0) @binding(0) var<storage, read_write> out: array<vec2u>;
@compute @workgroup_size(${WG}) fn main(@builtin(local_invocation_index) i: u32,
    @builtin(subgroup_invocation_id) lane: u32, @builtin(subgroup_size) size: u32) {
  out[i] = vec2u(size, lane);
}`;
const view = /* wgsl */ `
@group(0) @binding(0) var<storage> out: array<vec2u>;
struct Out { @builtin(position) pos: vec4f, @location(0) color: vec3f }
@vertex fn vs(@builtin(vertex_index) v: u32, @builtin(instance_index) i: u32) -> Out {
  let q = vec2f(f32(v & 1), f32(v >> 1));
  let size = out[i].x;  let lane = out[i].y;
  let group = f32(i / size);                               // which subgroup this lane belongs to
  let px = vec2f(12 + f32(i % 32) * 18, 70 + f32(i / 32) * 22) + q * vec2f(16, 18);
  let hue = group * 0.9;
  let shade = 0.75 + 0.25 * f32(lane) / f32(size);         // lanes brighten along the subgroup
  return Out(vec4f(px.x / 300 - 1, 1 - px.y / 170, 0, 1), (0.5 + 0.35 * cos(vec3f(hue, hue + 2.1, hue + 4.2))) * shade);
}
@fragment fn fs(in: Out) -> @location(0) vec4f { return vec4f(in.color, 1); }`;

async function main() {
  const adapter = await navigator.gpu?.requestAdapter();
  if (!adapter) return showMessage('WebGPU is not available in this browser');
  const { subgroupMinSize, subgroupMaxSize } = adapter.info;       // e.g. 32 and 32 on NVIDIA
  if (!adapter.features.has('subgroups')) {
    return showMessage(`No 'subgroups' feature; size range reported: ${subgroupMinSize ?? '?'} to ${subgroupMaxSize ?? '?'}`);
  }
  const hasControl = adapter.features.has('subgroup-size-control');
  const device = await adapter.requestDevice({ requiredFeatures: ['subgroups'] });
  const context = canvas.getContext('webgpu');
  const format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });
  const B = GPUBufferUsage;
  const out = device.createBuffer({ size: WG * 8, usage: B.STORAGE | B.COPY_SRC });
  const read = device.createBuffer({ size: 8, usage: B.COPY_DST | B.MAP_READ });
  const compute = device.createComputePipeline({ layout: 'auto', compute: { module: device.createShaderModule({ code }) } });
  const module = device.createShaderModule({ code: view });
  const render = device.createRenderPipeline({ layout: 'auto', primitive: { topology: 'triangle-strip' },
    vertex: { module }, fragment: { module, targets: [{ format }] } });

  const encoder = device.createCommandEncoder();
  const cp = encoder.beginComputePass();
  cp.setPipeline(compute);
  cp.setBindGroup(0, device.createBindGroup({ layout: compute.getBindGroupLayout(0), entries: [{ binding: 0, resource: { buffer: out } }] }));
  cp.dispatchWorkgroups(1);
  cp.end();
  const pass = encoder.beginRenderPass({ colorAttachments: [{ view: context.getCurrentTexture().createView(),
    clearValue: [0.97, 0.96, 0.93, 1], loadOp: 'clear', storeOp: 'store' }] });
  pass.setPipeline(render);
  pass.setBindGroup(0, device.createBindGroup({ layout: render.getBindGroupLayout(0), entries: [{ binding: 0, resource: { buffer: out } }] }));
  pass.draw(4, WG);
  pass.end();
  encoder.copyBufferToBuffer(out, 0, read, 0, 8);
  device.queue.submit([encoder.finish()]);
  await read.mapAsync(GPUMapMode.READ);
  const size = new Uint32Array(read.getMappedRange())[0];
  read.unmap();

  ink.font = 'bold 13px system-ui, sans-serif'; ink.fillStyle = '#222';
  ink.fillText(`adapter.info: subgroupMinSize ${subgroupMinSize}, subgroupMaxSize ${subgroupMaxSize}; this shader got ${size}`, 12, 24);
  ink.font = '12px system-ui, sans-serif'; ink.fillStyle = '#555';
  ink.fillText(`@workgroup_size(${WG}): ${WG / size} subgroups, one colour each, 32 lanes per row`, 12, 46);
  ink.fillStyle = '#222';
  const lines = [`Portable scratch for per-subgroup partial results: ${WG} / subgroupMinSize = ${WG / subgroupMinSize} slots`,
    'Loop over lanes with the subgroup_size built-in, never a constant 32.',
    hasControl ? 'subgroup-size-control is offered: @subgroup_size(n) can demand a size in that range.'
               : 'No subgroup-size-control here, so @subgroup_size(n) is unavailable.'];
  lines.forEach((l, i) => ink.fillText(l, 12, 270 + i * 20));
}
main();
</script>