The Responses API puts image generation inside a conversation. A mainline model (here the inexpensive gpt-6-luna) reads your input, calls the built-in image_generation tool with a rewritten prompt, and returns the image. You gain multi-turn editing (Multi-Turn Editing) and file or URL inputs, and pay for the mainline model's tokens.
import fs from 'node:fs';
import OpenAI from 'openai';
const openai = new OpenAI();
const response = await openai.responses.create({
model: 'gpt-6-luna', // the mainline model that calls the tool
input: 'Draw a round shop sign for a bookshop called BookNest: an open book '
+ 'forming a bird\'s nest, deep blue on cream, flat style',
tools: [{ type: 'image_generation', model: 'gpt-image-2.5-flare',
size: '1024x1024', quality: 'low' }],
});
const call = response.output.find(item => item.type === 'image_generation_call');
fs.writeFileSync('booknest-sign.png', Buffer.from(call.result, 'base64'));
console.log(response.output.map(item => item.type).join(', '), '->', call.action);
const brief = call.revised_prompt.slice(0, 160).replace(/\s\S*$/, ' ...');
console.log(`revised_prompt: ${brief}`.replace(/(.{1,90})(\s|$)/g, '$1\n').trimEnd());
const { usage, tool_usage: { image_gen } } = response;
console.log(`${response.model}: ${usage.input_tokens} in, ${usage.output_tokens} out;`,
`image tool: ${image_gen.input_tokens} in, ${image_gen.output_tokens} out`);Output
reasoning, image_generation_call, message -> generate revised_prompt: Create a polished flat vector illustration of a ROUND SHOP SIGN for a bookshop named “BookNest”. Centered, perfectly circular sign with a warm cream background ... gpt-6-luna: 1842 in, 229 out; image tool: 171 in, 196 out
The tool expanded 20 words into a 105-word brief, and tool_usage bills it like the Images API: 171 text and 196 image tokens, $0.0067, plus $0.0003 for gpt-6-luna, $0.0070 in all. Set the tool's action to 'generate' or 'edit' to override the model's choice. Use the Images API for one-shot jobs and the Responses API when a user refines a picture over several turns.

<!doctype html>
<style>
body { margin: 0; padding: 8px; background: #fafaf7; font: 12px system-ui, sans-serif; color: #263238; }
svg { width: 100%; max-width: 600px; display: block; }
</style>
<script src="https://cdn.jsdelivr.net/npm/d3@7.9.0/dist/d3.min.js"></script>
<svg viewBox="0 0 600 330" font-size="11"></svg>
<p><button id="play">Replay the turn</button></p>
<script>
const svg = d3.select('svg');
const steps = [
{ x: 10, title: 'your input', lines: ['"Draw a round shop sign for a', 'bookshop called BookNest…"', '20 words'], color: '#1f5f8b' },
{ x: 160, title: 'gpt-6-luna', lines: ['reads the input,', 'decides to call the tool', 'action: generate | edit'], color: '#5b3f99' },
{ x: 310, title: 'image_generation', lines: ['revised_prompt:', '105 words', 'gpt-image-2.5-flare, low'], color: '#e09a10' },
{ x: 460, title: 'response.output', lines: ['image_generation_call', 'result: base64 PNG', 'message'], color: '#3f7d3a' },
];
const boxes = svg.selectAll('g.step').data(steps).join('g').attr('class', 'step').attr('transform', d => `translate(${d.x},20)`);
boxes.append('rect').attr('width', 130).attr('height', 90).attr('rx', 8).attr('fill', '#fff').attr('stroke', d => d.color).attr('stroke-width', 2);
boxes.append('text').attr('x', 8).attr('y', 18).attr('font-weight', 'bold').attr('fill', d => d.color).text(d => d.title);
boxes.selectAll('text.l').data(d => d.lines).join('text').attr('class', 'l').attr('x', 8).attr('y', (t, i) => 38 + i * 16).text(t => t);
svg.selectAll('path.arrow').data([0, 1, 2]).join('path').attr('class', 'arrow')
.attr('d', i => `M${140 + i * 150},65h18m-6,-5l6,5l-6,5`).attr('stroke', '#8d6e63').attr('fill', 'none');
const dot = svg.append('circle').attr('r', 7).attr('cy', 65).attr('fill', '#e09a10');
// Word counts: the tool expands a short request into a long caption-style brief
const words = svg.append('g').attr('transform', 'translate(10,140)');
words.append('text').attr('font-weight', 'bold').text('prompt length (words)');
const wx = d3.scaleLinear([0, 110], [120, 580]);
[['you wrote', 20, '#1f5f8b'], ['revised_prompt', 105, '#e09a10']].forEach(([name, n, color], i) => {
words.append('text').attr('y', 26 + i * 26).text(name);
words.append('rect').attr('class', 'w').attr('x', wx(0)).attr('y', 14 + i * 26).attr('height', 18).attr('fill', color).attr('data-w', wx(n) - wx(0));
words.append('text').attr('class', 'wl').attr('x', wx(n) + 4).attr('y', 27 + i * 26).text(n).attr('opacity', 0);
});
// The bill: tool usage priced like the Images API, plus the mainline model's tokens
const bill = svg.append('g').attr('transform', 'translate(10,230)');
bill.append('text').attr('font-weight', 'bold').text('cost of the turn');
const parts = [['image tool: 171 text + 196 image tokens', 0.0067, '#e09a10'], ['gpt-6-luna tokens', 0.0003, '#5b3f99']];
const bx = d3.scaleLinear([0, 0.0075], [0, 460]);
let acc = 0;
parts.forEach(([label, v, color], i) => {
bill.append('rect').attr('class', 'b').attr('x', 120 + bx(acc)).attr('y', 14).attr('height', 26).attr('fill', color).attr('data-w', bx(v));
bill.append('text').attr('x', 120).attr('y', 58 + i * 16).attr('fill', color).text(`■ ${label}: $${v.toFixed(4)}`);
acc += v;
});
bill.append('text').attr('y', 32).text('$0.0070 in all');
function play() {
svg.selectAll('rect.w, rect.b').attr('width', 0);
svg.selectAll('text.wl').attr('opacity', 0);
dot.attr('cx', 75).transition().duration(2400).ease(d3.easeLinear).attrTween('cx', () => t => 75 + t * 450);
svg.selectAll('rect.w').transition().delay((d, i) => 600 + i * 900).duration(700).attr('width', function () { return this.dataset.w; });
svg.selectAll('text.wl').transition().delay((d, i) => 1300 + i * 900).attr('opacity', 1);
svg.selectAll('rect.b').transition().delay(2400).duration(700).attr('width', function () { return this.dataset.w; });
}
d3.select('#play').on('click', play);
play();
</script>