Robust Recovery

Designing a Robust Device-Loss Recovery Path

The specification's advice: most losses are transient, so unless you destroyed the device, get a new one and rebuild. That works when you keep a CPU-side source of truth (data, image sources and WGSL in JavaScript), put all GPU setup in one function so recovery is simply a second call, bound the retries (a null adapter or repeated losses mean the WebGL 2 path of WebGL), and let the render loop read the device from one variable, skipping frames while it is null. This shelf keeps its page counts in a Float32Array and destroys its first device to simulate a reset:

demos/ch04/device-loss.html: rebuilding everything after a lossHTMLLive
<canvas id="gpu" width="480" height="100"></canvas><pre id="log"></pre>
<script type="module">
const pages = new Float32Array([312, 428, 256, 198, 344, 176]);   // CPU copy survives a loss
const code = `@group(0) @binding(0) var<storage, read> pages: array<f32>;
  @vertex fn vs(@builtin(vertex_index) i: u32) -> @builtin(position) vec4f {
    let c = array(vec2f(0, 0), vec2f(1, 0), vec2f(0, 1), vec2f(1, 0), vec2f(1, 1),
                  vec2f(0, 1))[i % 6];
    return vec4f(-0.9 + f32(i / 6) * 0.3 + c.x * 0.24, -0.9 + c.y * pages[i / 6] / 240, 0, 1);
  }
  @fragment fn fs() -> @location(0) vec4f { return vec4f(0.08, 0.40, 0.75, 1); }`;
const say = (s) => { log.textContent += s + '\n'; };
let attempt = 0, simulated = false;
async function start() {                        // builds everything, every time
  const adapter = await navigator.gpu.requestAdapter();   // never the old adapter
  if (!adapter) return say('no adapter: switching to WebGL 2');
  const device = await adapter.requestDevice(), n = ++attempt;
  device.lost.then((info) => {
    say(`device #${n} lost (${info.reason}): ${info.message}`);
    if ((info.reason !== 'destroyed' || simulated) && attempt < 3) start();   // bounded
    simulated = false;
  });
  const context = gpu.getContext('webgpu'), format = navigator.gpu.getPreferredCanvasFormat();
  context.configure({ device, format });        // same context, new device
  const buffer = device.createBuffer({ size: 24, usage: GPUBufferUsage.STORAGE | 8 });
  device.queue.writeBuffer(buffer, 0, pages);   // re-upload from the CPU copy
  const module = device.createShaderModule({ code }), layout = 'auto';
  const pipeline = device.createRenderPipeline({ layout, vertex: { module },
    fragment: { module, targets: [{ format }] } });
  const bindGroup = device.createBindGroup({ layout: pipeline.getBindGroupLayout(0),
    entries: [{ binding: 0, resource: { buffer } }] });
  const encoder = device.createCommandEncoder();
  const view = context.getCurrentTexture().createView();
  const pass = encoder.beginRenderPass({ colorAttachments: [{ view, loadOp: 'clear',
    storeOp: 'store', clearValue: [0.96, 0.94, 0.9, 1] }] });
  pass.setPipeline(pipeline); pass.setBindGroup(0, bindGroup); pass.draw(36); pass.end();
  device.queue.submit([encoder.finish()]);
  await device.queue.onSubmittedWorkDone();
  say(`device #${n} drew the shelf`), window.__done = n === 2;
  return device;
}
simulated = true;
(await start()).destroy();                      // stand-in for a driver reset
</script>
Browser output of Listing 4.21
Browser output of 21

The second device drew the shelf (8 is GPUBufferUsage.COPY_DST). A real start() also restarts the render loop; retry after a short delay, and send each loss to telemetry, since losses cluster on particular drivers.