Bottlenecks

State Changes and Draw Calls as Common Bottlenecks

Every draw call and state change passes through WebGL's validation, the command buffer to the GPU process, ANGLE 2,592 and the driver, so many small objects cost by call count, not pixels. The test draws 2,400 small covers with one call and two uniform changes each, then as one instanced call (Attribute Divisors):

Timing 2,400 draw calls against one instanced callJavaScript
const canvas = Object.assign(document.createElement('canvas'), { width: 800, height: 400 });
const gl = canvas.getContext('webgl2'), wall = GLH.coverWall(), N = 2400;
const { attributes: a, uniforms: u } = GLH.load(gl, `#version 300 es
  in vec2 aPosition, aOffset; uniform vec2 uOffset; uniform vec3 uColor; out vec3 vColor;
  void main() { vColor = uColor;
                gl_Position = vec4(aPosition * 0.05 + uOffset + aOffset, 0, 1); }`,
`#version 300 es
  precision mediump float; in vec3 vColor; out vec4 c; void main() { c = vec4(vColor, 1); }`);
const offsets = new Float32Array(N * 2).map((_, i) => Math.sin(i * 12.9898) * 0.9);
gl.bindVertexArray(GLH.vao(gl, a, wall.vertices,
  [['aPosition', 2], ['aUV', 2], ['aBook', 1]], wall.indices));
GLH.buffer(gl, offsets);                             // per-instance offsets, disabled for now
gl.vertexAttribPointer(a.aOffset, 2, gl.FLOAT, false, 0, 0);
gl.vertexAttribDivisor(a.aOffset, 1);
const many = await GLH.time(gl, () => {
  for (let i = 0; i < N; i++) {                        // 2 uniform calls + 1 draw each
    u.uOffset(offsets.subarray(i * 2, i * 2 + 2)); u.uColor(GLH.rgb(GLH.covers[i % 6]));
    gl.drawElements(gl.TRIANGLES, 6, gl.UNSIGNED_SHORT, (i % 6) * 12);
  }
});
u.uOffset([0, 0]); gl.enableVertexAttribArray(a.aOffset);   // offsets now come per instance
const one = await GLH.time(gl, () =>
  gl.drawElementsInstanced(gl.TRIANGLES, 6, gl.UNSIGNED_SHORT, 0, N));
for (const [label, { cpu, gpu }] of [['2,400 calls ', many], ['1 instanced', one]]) {
  console.log(`${label}: CPU ${cpu.toFixed(1)} ms, GPU ${gpu.toFixed(3)} ms`);
}
Output
2,400 calls : CPU 2.6 ms, GPU 2.681 ms
1 instanced: CPU 0.0 ms, GPU 0.042 ms

Over 13 runs on the GTX 1650, the calls took 2.6 to 3.4 ms of CPU against under 0.1 ms (the clock step), and 0.55 to 3.3 ms of GPU time against a steady 0.042 ms (twice about 16.5 ms, outliers even a median kept): 13 to 79 times more for the same pixels. SwiftShader 2,546 's timer gave 25 to 34 ms for both, since its "GPU" is the CPU. A 60 Hz frame has 16.7 ms, so cut calls first (instancing, batching), then sort draws by program and texture.

2,400 small covers drawn with 2,400 calls (left) and with one instanced call (right), timed live on the CPU and GPUHTMLLive
<!doctype html>
<style>
  body { margin: 0; font: 11px system-ui, sans-serif; background: #f7f4ee; color: #333; }
  canvas { display: block; width: 100%; max-width: 600px; }
  .names { display: flex; max-width: 600px; text-align: center; font: 11px monospace; }
  .names div { flex: 1; padding: 4px 2px; }
</style>
<canvas id="c" width="1200" height="400"></canvas>
<div class="names"><div id="a">2,400 draw calls, 2 uniform calls each</div><div id="b">1 drawElementsInstanced call</div></div>
<script>
const gl = document.getElementById('c').getContext('webgl2');
const N = 2400;
const program = gl.createProgram();
for (const [type, src] of [[gl.VERTEX_SHADER, `#version 300 es
layout(location = 0) in vec2 aPosition; layout(location = 1) in vec2 aOffset; layout(location = 2) in vec3 aColor;
uniform vec2 uOffset; uniform vec3 uColor; uniform bool uInstanced; out vec3 vColor;
void main() { vColor = uInstanced ? aColor : uColor; gl_Position = vec4(aPosition * vec2(0.03, 0.06) + uOffset + aOffset, 0, 1); }`],
  [gl.FRAGMENT_SHADER, `#version 300 es
precision mediump float; in vec3 vColor; out vec4 c; void main() { c = vec4(vColor, 1); }`]]) {
  const s = gl.createShader(type); gl.shaderSource(s, src); gl.compileShader(s); gl.attachShader(program, s);
}
gl.linkProgram(program); gl.useProgram(program);
const palette = [[0.12, 0.37, 0.55], [0.36, 0.25, 0.6], [0.88, 0.6, 0.06], [0.25, 0.49, 0.23], [0.71, 0.27, 0.18], [0.16, 0.62, 0.56]];
const hash = (n) => { const s = Math.sin(n * 12.9898) * 43758.5453; return s - Math.floor(s); };
const offsets = new Float32Array(N * 2).map((_, i) => hash(i) * 1.84 - 0.92);   // scattered positions
const colors = new Float32Array(N * 3).map((_, i) => palette[Math.floor(i / 3) % 6][i % 3]);
gl.bindVertexArray(gl.createVertexArray());
gl.bindBuffer(gl.ARRAY_BUFFER, gl.createBuffer());
gl.bufferData(gl.ARRAY_BUFFER, new Float32Array([-1, -1, 1, -1, 1, 1, -1, 1]), gl.STATIC_DRAW);
gl.vertexAttribPointer(0, 2, gl.FLOAT, false, 0, 0); gl.enableVertexAttribArray(0);
gl.bindBuffer(gl.ELEMENT_ARRAY_BUFFER, gl.createBuffer());
gl.bufferData(gl.ELEMENT_ARRAY_BUFFER, new Uint16Array([0, 1, 2, 0, 2, 3]), gl.STATIC_DRAW);
[[1, offsets, 2], [2, colors, 3]].forEach(([loc, data, size]) => {          // per-instance data, divisor 1
  gl.bindBuffer(gl.ARRAY_BUFFER, gl.createBuffer());
  gl.bufferData(gl.ARRAY_BUFFER, data, gl.STATIC_DRAW);
  gl.vertexAttribPointer(loc, size, gl.FLOAT, false, 0, 0);
  gl.vertexAttribDivisor(loc, 1);
});
const u = (n) => gl.getUniformLocation(program, n);
const timer = gl.getExtension('EXT_disjoint_timer_query_webgl2');
const stats = [{ cpu: [], gpu: [], pending: [] }, { cpu: [], gpu: [], pending: [] }];
const median = (list) => [...list].sort((p, q) => p - q)[list.length >> 1];
function timed(k, draw) {
  const s = stats[k], query = timer && s.pending.length < 3 && gl.createQuery();
  if (query) gl.beginQuery(timer.TIME_ELAPSED_EXT, query);
  const t0 = performance.now(); draw(); s.cpu.push(performance.now() - t0);
  if (query) { gl.endQuery(timer.TIME_ELAPSED_EXT); s.pending.push(query); }
  while (s.pending.length && gl.getQueryParameter(s.pending[0], gl.QUERY_RESULT_AVAILABLE)) {
    const q = s.pending.shift(); s.gpu.push(gl.getQueryParameter(q, gl.QUERY_RESULT) / 1e6); gl.deleteQuery(q);
  }
  if (s.cpu.length > 60) s.cpu.shift(); if (s.gpu.length > 60) s.gpu.shift();
}
gl.enable(gl.SCISSOR_TEST);
function frame() {
  gl.clearColor(0.93, 0.91, 0.87, 1);
  gl.viewport(0, 0, 598, 400); gl.scissor(0, 0, 598, 400); gl.clear(gl.COLOR_BUFFER_BIT);
  timed(0, () => {                                              // many calls: validation and driver work per call
    gl.uniform1i(u('uInstanced'), 0);
    gl.disableVertexAttribArray(1); gl.disableVertexAttribArray(2);
    for (let i = 0; i < N; i++) {
      gl.uniform2f(u('uOffset'), offsets[i * 2], offsets[i * 2 + 1]);
      gl.uniform3fv(u('uColor'), palette[i % 6]);
      gl.drawElements(gl.TRIANGLES, 6, gl.UNSIGNED_SHORT, 0);
    }
  });
  gl.viewport(602, 0, 598, 400); gl.scissor(602, 0, 598, 400); gl.clear(gl.COLOR_BUFFER_BIT);
  timed(1, () => {                                              // one call: the same pixels
    gl.uniform1i(u('uInstanced'), 1); gl.uniform2f(u('uOffset'), 0, 0);
    gl.enableVertexAttribArray(1); gl.enableVertexAttribArray(2);
    gl.drawElementsInstanced(gl.TRIANGLES, 6, gl.UNSIGNED_SHORT, 0, N);
  });
  const fmt = (s) => `CPU ${median(s.cpu).toFixed(2)} ms` + (s.gpu.length ? `, GPU ${median(s.gpu).toFixed(3)} ms` : '');
  document.getElementById('a').textContent = `2,400 calls: ${fmt(stats[0])}`;
  document.getElementById('b').textContent = `1 instanced call: ${fmt(stats[1])}`;
  requestAnimationFrame(frame);
}
requestAnimationFrame(frame);
</script>