Image generation
For the complete documentation index, see llms.txt. Markdown versions of documentation pages are available by appending
.mdto the page URL.
Overview
The OpenAI API lets you generate and edit images from text prompts using GPT Image models, including our latest, gpt-image-2. You can access image generation capabilities through two APIs:
Image API
Starting with gpt-image-1 and later models, the Image API provides two endpoints, each with distinct capabilities:
- Generations: Generate images from scratch based on a text prompt
- Edits: Modify existing images using a new prompt, either partially or entirely
Responses API
The Responses API allows you to generate images as part of conversations or multi-step flows. It supports image generation as a built-in tool, and accepts image inputs and outputs within context.
Compared to the Image API, it adds:
- Multi-turn editing: Iteratively make high fidelity edits to images with prompting
- Flexible inputs: Accept image File IDs as input images, not just bytes
The Responses API image generation tool uses its own GPT Image model selection. For details on mainline models that support calling this tool, refer to the supported models below.
Choosing the right API
- If you only need to generate or edit a single image from one prompt, the Image API is your best choice.
- If you want to build conversational, editable image experiences with GPT Image, go with the Responses API.
With the Image API, you choose a GPT Image model directly. With the Responses API, you choose a mainline model that supports the image generation tool; the tool handles GPT Image model selection. Responses API requests include the mainline model's token usage in addition to image generation costs.
Both APIs let you customize output by adjusting quality, size, format, and compression. Transparent backgrounds depend on model support.
This guide focuses on GPT Image.
To ensure these models are used responsibly, you may need to complete the API
Organization
Verification
from your developer
console before
using GPT Image models, including gpt-image-2, gpt-image-1.5,
gpt-image-1, and gpt-image-1-mini.
Generate Images
You can use the image generation endpoint to create images based on text prompts, or the image generation tool in the Responses API to generate images as part of a conversation.
To learn more about customizing the output (size, quality, format, compression), refer to the customize image output section below.
You can set the n parameter to generate multiple images at once in a single request (by default, the API returns a single image).
Image API
Generate an image
import OpenAI from "openai";
import fs from "fs";
const openai = new OpenAI();
const prompt = `
A children's book drawing of a veterinarian using a stethoscope to
listen to the heartbeat of a baby otter.
`;
const result = await openai.images.generate({
model: "gpt-image-2",
prompt,
});
// Save the image to a file
const image_base64 = result.data[0].b64_json;
const image_bytes = Buffer.from(image_base64, "base64");
fs.writeFileSync("otter.png", image_bytes);
from openai import OpenAI
import base64
client = OpenAI()
prompt = """
A children's book drawing of a veterinarian using a stethoscope to
listen to the heartbeat of a baby otter.
"""
result = client.images.generate(model="gpt-image-2", prompt=prompt)
image_base64 = result.data[0].b64_json
image_bytes = base64.b64decode(image_base64)
# Save the image to a file
with open("otter.png", "wb") as f:
f.write(image_bytes)
package main
import (
"context"
"encoding/base64"
"os"
"github.com/openai/openai-go/v3"
)
func main() {
client := openai.NewClient()
result, err := client.Images.Generate(context.Background(), openai.ImageGenerateParams{
Model: openai.ImageModel("gpt-image-2"),
Prompt: "A children's book drawing of a veterinarian using a stethoscope to " +
"listen to the heartbeat of a baby otter.",
})
if err != nil {
panic(err)
}
image, err := base64.StdEncoding.DecodeString(result.Data[0].B64JSON)
if err != nil {
panic(err)
}
if err := os.WriteFile("otter.png", image, 0o600); err != nil {
panic(err)
}
}
curl -X POST "https://api.openai.com/v1/images/generations" \
-H "Authorization: Bearer $OPENAI_API_KEY" \
-H "Content-type: application/json" \
-d '{
"model": "gpt-image-2",
"prompt": "A children'\''s book drawing of a veterinarian using a stethoscope to listen to the heartbeat of a baby otter."
}' | jq -r '.data[0].b64_json' | base64 --decode > otter.png
openai images generate \
--model gpt-image-2 \
--prompt "A children's book drawing of a veterinarian using a stethoscope to listen to the heartbeat of a baby otter." \
--raw-output \
--transform 'data.0.b64_json' | base64 --decode > otter.png
Responses API
Generate an image
import OpenAI from "openai";
const openai = new OpenAI();
const response = await openai.responses.create({
model: "gpt-5.6",
input:
"Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools: [{ type: "image_generation" }],
});
// Save the image to a file
const imageData = response.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData.length > 0) {
const imageBase64 = imageData[0];
const fs = await import("fs");
fs.writeFileSync("otter.png", Buffer.from(imageBase64, "base64"));
}
from openai import OpenAI
import base64
client = OpenAI()
response = client.responses.create(
model="gpt-5.6",
input="Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools=[{"type": "image_generation"}],
)
# Save the image to a file
image_data = [
output.result
for output in response.output
if output.type == "image_generation_call"
]
if image_data:
image_base64 = image_data[0]
with open("otter.png", "wb") as f:
f.write(base64.b64decode(image_base64))
package main
import (
"context"
"encoding/base64"
"os"
"github.com/openai/openai-go/v3"
"github.com/openai/openai-go/v3/responses"
)
func main() {
client := openai.NewClient()
response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{
OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
})
if err != nil {
panic(err)
}
saveFirstGeneratedImage(response, "otter.png")
}
func saveFirstGeneratedImage(response *responses.Response, filename string) {
for _, output := range response.Output {
if output.Type != "image_generation_call" {
continue
}
image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
return
}
panic("response did not include an image generation call")
}
Multi-turn image generation
With the Responses API, you can build multi-turn conversations involving image generation either by providing image generation calls outputs within context (you can also just use the image ID), or by using the previous_response_id parameter.
This lets you iterate on images across multiple turns—refining prompts, applying new instructions, and evolving the visual output as the conversation progresses.
With the Responses API image generation tool, supported tool models can choose whether to generate a new image or edit one already in the conversation. The optional action parameter controls this behavior: keep action: "auto" to let the model decide, set action: "generate" to always create a new image, or set action: "edit" to force editing when an image is in context.
Force image creation with action
import OpenAI from "openai";
const openai = new OpenAI();
const response = await openai.responses.create({
model: "gpt-5.6",
input:
"Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools: [{ type: "image_generation", action: "generate" }],
});
// Save the image to a file
const imageData = response.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData.length > 0) {
const imageBase64 = imageData[0];
const fs = await import("fs");
fs.writeFileSync("otter.png", Buffer.from(imageBase64, "base64"));
}
from openai import OpenAI
import base64
client = OpenAI()
response = client.responses.create(
model="gpt-5.6",
input="Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools=[{"type": "image_generation", "action": "generate"}],
)
# Save the image to a file
image_data = [
output.result
for output in response.output
if output.type == "image_generation_call"
]
if image_data:
image_base64 = image_data[0]
with open("otter.png", "wb") as f:
f.write(base64.b64decode(image_base64))
package main
import (
"context"
"encoding/base64"
"os"
"github.com/openai/openai-go/v3"
"github.com/openai/openai-go/v3/responses"
)
func main() {
client := openai.NewClient()
response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{
OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{Action: "generate"}}},
})
if err != nil {
panic(err)
}
for _, output := range response.Output {
if output.Type != "image_generation_call" {
continue
}
image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
if err != nil {
panic(err)
}
if err := os.WriteFile("otter.png", image, 0o600); err != nil {
panic(err)
}
return
}
panic("response did not include an image generation call")
}
If you force edit without providing an image in context, the call will return an error. Leave action at auto to have the model decide when to generate or edit.
Using previous response ID
Multi-turn image generation
import OpenAI from "openai";
const openai = new OpenAI();
const response = await openai.responses.create({
model: "gpt-5.6",
input:
"Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools: [{ type: "image_generation" }],
});
const imageData = response.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData.length > 0) {
const imageBase64 = imageData[0];
const fs = await import("fs");
fs.writeFileSync("cat_and_otter.png", Buffer.from(imageBase64, "base64"));
}
// Follow up
const response_fwup = await openai.responses.create({
model: "gpt-5.6",
previous_response_id: response.id,
input: "Now make it look realistic",
tools: [{ type: "image_generation" }],
});
const imageData_fwup = response_fwup.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData_fwup.length > 0) {
const imageBase64 = imageData_fwup[0];
const fs = await import("fs");
fs.writeFileSync(
"cat_and_otter_realistic.png",
Buffer.from(imageBase64, "base64")
);
}
from openai import OpenAI
import base64
client = OpenAI()
response = client.responses.create(
model="gpt-5.6",
input="Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools=[{"type": "image_generation"}],
)
image_data = [
output.result
for output in response.output
if output.type == "image_generation_call"
]
if image_data:
image_base64 = image_data[0]
with open("cat_and_otter.png", "wb") as f:
f.write(base64.b64decode(image_base64))
# Follow up
response_fwup = client.responses.create(
model="gpt-5.6",
previous_response_id=response.id,
input="Now make it look realistic",
tools=[{"type": "image_generation"}],
)
image_data_fwup = [
output.result
for output in response_fwup.output
if output.type == "image_generation_call"
]
if image_data_fwup:
image_base64 = image_data_fwup[0]
with open("cat_and_otter_realistic.png", "wb") as f:
f.write(base64.b64decode(image_base64))
package main
import (
"context"
"encoding/base64"
"os"
"github.com/openai/openai-go/v3"
"github.com/openai/openai-go/v3/responses"
)
func main() {
client := openai.NewClient()
first, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{
OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
})
if err != nil {
panic(err)
}
saveFirstGeneratedImage(first, "cat_and_otter.png")
followUp, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
PreviousResponseID: openai.String(first.ID),
Input: responses.ResponseNewParamsInputUnion{
OfString: openai.String("Now make it look realistic"),
},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
})
if err != nil {
panic(err)
}
saveFirstGeneratedImage(followUp, "cat_and_otter_realistic.png")
}
func saveFirstGeneratedImage(response *responses.Response, filename string) {
for _, output := range response.Output {
if output.Type != "image_generation_call" {
continue
}
image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
return
}
panic("response did not include an image generation call")
}
Using image ID
Multi-turn image generation
import OpenAI from "openai";
const openai = new OpenAI();
const response = await openai.responses.create({
model: "gpt-5.6",
input:
"Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools: [{ type: "image_generation" }],
});
const imageGenerationCalls = response.output.filter(
(output) => output.type === "image_generation_call"
);
const imageData = imageGenerationCalls.map((output) => output.result);
if (imageData.length > 0) {
const imageBase64 = imageData[0];
const fs = await import("fs");
fs.writeFileSync("cat_and_otter.png", Buffer.from(imageBase64, "base64"));
}
// Follow up
const response_fwup = await openai.responses.create({
model: "gpt-5.6",
input: [
{
role: "user",
content: [{ type: "input_text", text: "Now make it look realistic" }],
},
{
type: "image_generation_call",
id: imageGenerationCalls[0].id,
},
],
tools: [{ type: "image_generation" }],
});
const imageData_fwup = response_fwup.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData_fwup.length > 0) {
const imageBase64 = imageData_fwup[0];
const fs = await import("fs");
fs.writeFileSync(
"cat_and_otter_realistic.png",
Buffer.from(imageBase64, "base64")
);
}
import openai
import base64
response = openai.responses.create(
model="gpt-5.6",
input="Generate an image of gray tabby cat hugging an otter with an orange scarf",
tools=[{"type": "image_generation"}],
)
image_generation_calls = [
output for output in response.output if output.type == "image_generation_call"
]
image_data = [output.result for output in image_generation_calls]
if image_data:
image_base64 = image_data[0]
with open("cat_and_otter.png", "wb") as f:
f.write(base64.b64decode(image_base64))
# Follow up
response_fwup = openai.responses.create(
model="gpt-5.6",
input=[
{
"role": "user",
"content": [{"type": "input_text", "text": "Now make it look realistic"}],
},
{
"type": "image_generation_call",
"id": image_generation_calls[0].id,
},
],
tools=[{"type": "image_generation"}],
)
image_data_fwup = [
output.result
for output in response_fwup.output
if output.type == "image_generation_call"
]
if image_data_fwup:
image_base64 = image_data_fwup[0]
with open("cat_and_otter_realistic.png", "wb") as f:
f.write(base64.b64decode(image_base64))
package main
import (
"context"
"encoding/base64"
"encoding/json"
"os"
"github.com/openai/openai-go/v3"
"github.com/openai/openai-go/v3/responses"
)
func main() {
client := openai.NewClient()
first, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{
OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
})
if err != nil {
panic(err)
}
call := firstImageGenerationCall(first)
saveImage("cat_and_otter.png", call.Result)
input := outputAsInput(first.Output)
input = append(input, responses.ResponseInputItemParamOfMessage(
responses.ResponseInputMessageContentListParam{responses.ResponseInputContentParamOfInputText("Now make it look realistic")},
responses.EasyInputMessageRoleUser,
))
followUp, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{OfInputItemList: input},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
})
if err != nil {
panic(err)
}
saveImage("cat_and_otter_realistic.png", firstImageGenerationCall(followUp).Result)
}
func firstImageGenerationCall(response *responses.Response) responses.ResponseOutputItemImageGenerationCall {
for _, output := range response.Output {
if output.Type == "image_generation_call" {
return output.AsImageGenerationCall()
}
}
panic("response did not include an image generation call")
}
func outputAsInput(output []responses.ResponseOutputItemUnion) []responses.ResponseInputItemUnionParam {
input := make([]responses.ResponseInputItemUnionParam, 0, len(output))
for _, item := range output {
var converted responses.ResponseInputItemUnion
if err := json.Unmarshal([]byte(item.RawJSON()), &converted); err != nil {
panic(err)
}
input = append(input, converted.ToParam())
}
return input
}
func saveImage(filename, encoded string) {
image, err := base64.StdEncoding.DecodeString(encoded)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
}
Result
| "Generate an image of gray tabby cat hugging an otter with an orange scarf" |
|
| "Now make it look realistic" |
|
Streaming
The Responses API and Image API support streaming image generation. You can stream partial images as the APIs generate them, providing a more interactive experience.
You can adjust the partial_images parameter to receive 0-3 partial images.
- If you set
partial_imagesto 0, you will only receive the final image. - For values larger than zero, you may not receive the full number of partial images you requested if the full image is generated more quickly.
Responses API
Stream an image
import OpenAI from "openai";
import fs from "fs";
const openai = new OpenAI();
function saveBase64Image(filename, imageBase64) {
const imageBuffer = Buffer.from(imageBase64, "base64");
fs.writeFileSync(filename, imageBuffer);
}
const stream = await openai.responses.create({
model: "gpt-5.6",
input:
"Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape",
stream: true,
tools: [{ type: "image_generation", partial_images: 2 }],
});
for await (const event of stream) {
if (event.type === "response.image_generation_call.partial_image") {
const idx = event.partial_image_index;
saveBase64Image(`river-partial-${idx}.png`, event.partial_image_b64);
} else if (event.type === "response.completed") {
const imageData = event.response.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData.length > 0) {
saveBase64Image("river-final.png", imageData[0]);
}
}
}
from openai import OpenAI
import base64
client = OpenAI()
def save_base64_image(filename, image_base64):
image_bytes = base64.b64decode(image_base64)
with open(filename, "wb") as f:
f.write(image_bytes)
stream = client.responses.create(
model="gpt-5.6",
input="Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape",
stream=True,
tools=[{"type": "image_generation", "partial_images": 2}],
)
for event in stream:
if event.type == "response.image_generation_call.partial_image":
idx = event.partial_image_index
save_base64_image(f"river-partial-{idx}.png", event.partial_image_b64)
elif event.type == "response.completed":
image_data = [
output.result
for output in event.response.output
if output.type == "image_generation_call"
]
if image_data:
save_base64_image("river-final.png", image_data[0])
package main
import (
"context"
"encoding/base64"
"fmt"
"os"
"github.com/openai/openai-go/v3"
"github.com/openai/openai-go/v3/responses"
)
func main() {
client := openai.NewClient()
stream := client.Responses.NewStreaming(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{
OfString: openai.String("Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape"),
},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{PartialImages: openai.Int(2)}}},
})
for stream.Next() {
event := stream.Current()
if event.Type == "response.image_generation_call.partial_image" {
partial := event.AsResponseImageGenerationCallPartialImage()
saveImage(fmt.Sprintf("river-partial-%d.png", partial.PartialImageIndex), partial.PartialImageB64)
}
if event.Type == "response.completed" {
for _, output := range event.AsResponseCompleted().Response.Output {
if output.Type == "image_generation_call" {
saveImage("river-final.png", output.AsImageGenerationCall().Result)
}
}
}
}
if err := stream.Err(); err != nil {
panic(err)
}
}
func saveImage(filename, encoded string) {
image, err := base64.StdEncoding.DecodeString(encoded)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
}
Image API
Stream an image
import fs from "fs";
import OpenAI from "openai";
const openai = new OpenAI();
const prompt =
"Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape";
const stream = await openai.images.generate({
prompt: prompt,
model: "gpt-image-2",
stream: true,
partial_images: 2,
});
for await (const event of stream) {
if (event.type === "image_generation.partial_image") {
const idx = event.partial_image_index;
const imageBase64 = event.b64_json;
const imageBuffer = Buffer.from(imageBase64, "base64");
fs.writeFileSync(`river${idx}.png`, imageBuffer);
}
}
from openai import OpenAI
import base64
client = OpenAI()
stream = client.images.generate(
prompt="Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape",
model="gpt-image-2",
stream=True,
partial_images=2,
)
for event in stream:
if event.type == "image_generation.partial_image":
idx = event.partial_image_index
image_base64 = event.b64_json
image_bytes = base64.b64decode(image_base64)
with open(f"river{idx}.png", "wb") as f:
f.write(image_bytes)
package main
import (
"context"
"encoding/base64"
"fmt"
"os"
"github.com/openai/openai-go/v3"
)
func main() {
client := openai.NewClient()
stream := client.Images.GenerateStreaming(context.Background(), openai.ImageGenerateParams{
Model: openai.ImageModel("gpt-image-2"),
Prompt: "Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape",
PartialImages: openai.Int(2),
})
for stream.Next() {
event := stream.Current()
if event.Type != "image_generation.partial_image" {
continue
}
partial := event.AsImageGenerationPartialImage()
saveImage(fmt.Sprintf("river%d.png", partial.PartialImageIndex), partial.B64JSON)
}
if err := stream.Err(); err != nil {
panic(err)
}
}
func saveImage(filename, encoded string) {
image, err := base64.StdEncoding.DecodeString(encoded)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
}
Result
| Partial 1 | Partial 2 | Final image |
|---|---|---|
![]() |
![]() |
![]() |
Prompt: Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape
Revised prompt
When using the image generation tool in the Responses API, the mainline model (for example, gpt-5.5) will automatically revise your prompt for improved performance.
You can access the revised prompt in the revised_prompt field of the image generation call:
Revised prompt response
{
"id": "ig_123",
"type": "image_generation_call",
"status": "completed",
"revised_prompt": "A gray tabby cat hugging an otter. The otter is wearing an orange scarf. Both animals are cute and friendly, depicted in a warm, heartwarming style.",
"result": "..."
}
Edit Images
The image edits endpoint lets you:
- Edit existing images
- Generate new images using other images as a reference
- Edit parts of an image by uploading an image and mask that identifies the areas to replace
Create a new image using image references
You can use one or more images as a reference to generate a new image.
In this example, we'll use 4 input images to generate a new image of a gift basket containing the items in the reference images.
Responses API
With the Responses API, you can provide input images in 3 different ways:
- By providing a fully qualified URL
- By providing an image as a Base64-encoded data URL
- By providing a file ID (created with the Files API)
Create a File
Create a File
import fs from "fs";
import OpenAI from "openai";
const openai = new OpenAI();
async function createFile(filePath) {
const fileContent = fs.createReadStream(filePath);
const result = await openai.files.create({
file: fileContent,
purpose: "vision",
});
return result.id;
}
from openai import OpenAI
client = OpenAI()
def create_file(file_path):
with open(file_path, "rb") as file_content:
result = client.files.create(
file=file_content,
purpose="vision",
)
return result.id
package main
import (
"context"
"fmt"
"os"
"github.com/openai/openai-go/v3"
)
func main() {
client := openai.NewClient()
file, err := os.Open("image.png")
if err != nil {
panic(err)
}
defer file.Close()
uploaded, err := client.Files.New(context.Background(), openai.FileNewParams{
File: file,
Purpose: openai.FilePurposeVision,
})
if err != nil {
panic(err)
}
fmt.Println(uploaded.ID)
}
Create a base64 encoded image
Create a base64 encoded image
import fs from "fs";
function encodeImage(filePath) {
const base64Image = fs.readFileSync(filePath, "base64");
return base64Image;
}
import base64
def encode_image(file_path):
with open(file_path, "rb") as f:
base64_image = base64.b64encode(f.read()).decode("utf-8")
return base64_image
package main
import (
"encoding/base64"
"fmt"
"os"
)
func main() {
image, err := os.ReadFile("image.png")
if err != nil {
panic(err)
}
fmt.Println(base64.StdEncoding.EncodeToString(image))
}
Edit an image
import fs from "fs";
import OpenAI from "openai";
const openai = new OpenAI();
function encodeImage(filePath) {
return fs.readFileSync(filePath, "base64");
}
async function createFile(filePath) {
const result = await openai.files.create({
file: fs.createReadStream(filePath),
purpose: "vision",
});
return result.id;
}
const prompt = `Generate a photorealistic image of a gift basket on a white background
labeled 'Relax & Unwind' with a ribbon and handwriting-like font,
containing all the items in the reference pictures.`;
const base64Image1 = encodeImage("fixtures/body-lotion.png");
const base64Image2 = encodeImage("fixtures/soap.png");
const fileId1 = await createFile("fixtures/bath-bomb.png");
const fileId2 = await createFile("fixtures/incense-kit.png");
const response = await openai.responses.create({
model: "gpt-5.6",
input: [
{
role: "user",
content: [
{ type: "input_text", text: prompt },
{
type: "input_image",
image_url: `data:image/png;base64,${base64Image1}`,
detail: "auto",
},
{
type: "input_image",
image_url: `data:image/png;base64,${base64Image2}`,
detail: "auto",
},
{
type: "input_image",
file_id: fileId1,
detail: "auto",
},
{
type: "input_image",
file_id: fileId2,
detail: "auto",
},
],
},
],
tools: [{ type: "image_generation" }],
});
const imageData = response.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData.length > 0) {
const imageBase64 = imageData[0];
fs.writeFileSync("gift-basket.png", Buffer.from(imageBase64, "base64"));
} else {
console.log(response.output_text);
}
from openai import OpenAI
import base64
client = OpenAI()
def encode_image(file_path):
with open(file_path, "rb") as image_file:
return base64.b64encode(image_file.read()).decode("utf-8")
def create_file(file_path):
with open(file_path, "rb") as file_content:
result = client.files.create(file=file_content, purpose="vision")
return result.id
prompt = """Generate a photorealistic image of a gift basket on a white background
labeled 'Relax & Unwind' with a ribbon and handwriting-like font,
containing all the items in the reference pictures."""
base64_image1 = encode_image("body-lotion.png")
base64_image2 = encode_image("soap.png")
file_id1 = create_file("bath-bomb.png")
file_id2 = create_file("incense-kit.png")
response = client.responses.create(
model="gpt-5.6",
input=[
{
"role": "user",
"content": [
{"type": "input_text", "text": prompt},
{
"type": "input_image",
"image_url": f"data:image/png;base64,{base64_image1}",
},
{
"type": "input_image",
"image_url": f"data:image/png;base64,{base64_image2}",
},
{
"type": "input_image",
"file_id": file_id1,
},
{
"type": "input_image",
"file_id": file_id2,
},
],
}
],
tools=[{"type": "image_generation"}],
)
image_generation_calls = [
output for output in response.output if output.type == "image_generation_call"
]
image_data = [output.result for output in image_generation_calls]
if image_data:
image_base64 = image_data[0]
with open("gift-basket.png", "wb") as f:
f.write(base64.b64decode(image_base64))
else:
print(response.output_text)
package main
import (
"context"
"encoding/base64"
"os"
"github.com/openai/openai-go/v3"
"github.com/openai/openai-go/v3/responses"
)
func main() {
client := openai.NewClient()
bathBombID := uploadImage(client, "bath-bomb.png")
incenseKitID := uploadImage(client, "incense-kit.png")
response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{OfInputItemList: responses.ResponseInputParam{
responses.ResponseInputItemParamOfMessage(
responses.ResponseInputMessageContentListParam{
responses.ResponseInputContentParamOfInputText("Generate a photorealistic image of a gift basket on a white background labeled 'Relax & Unwind' with a ribbon and handwriting-like font, containing all the items in the reference pictures."),
{OfInputImage: &responses.ResponseInputImageParam{ImageURL: openai.String(dataURL("body-lotion.png")), Detail: responses.ResponseInputImageDetailAuto}},
{OfInputImage: &responses.ResponseInputImageParam{ImageURL: openai.String(dataURL("soap.png")), Detail: responses.ResponseInputImageDetailAuto}},
{OfInputImage: &responses.ResponseInputImageParam{FileID: openai.String(bathBombID), Detail: responses.ResponseInputImageDetailAuto}},
{OfInputImage: &responses.ResponseInputImageParam{FileID: openai.String(incenseKitID), Detail: responses.ResponseInputImageDetailAuto}},
},
responses.EasyInputMessageRoleUser,
),
}},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
})
if err != nil {
panic(err)
}
saveFirstGeneratedImage(response, "gift-basket.png")
}
func uploadImage(client openai.Client, filename string) string {
file, err := os.Open(filename)
if err != nil {
panic(err)
}
defer file.Close()
uploaded, err := client.Files.New(context.Background(), openai.FileNewParams{File: file, Purpose: openai.FilePurposeVision})
if err != nil {
panic(err)
}
return uploaded.ID
}
func dataURL(filename string) string {
image, err := os.ReadFile(filename)
if err != nil {
panic(err)
}
return "data:image/png;base64," + base64.StdEncoding.EncodeToString(image)
}
func saveFirstGeneratedImage(response *responses.Response, filename string) {
for _, output := range response.Output {
if output.Type != "image_generation_call" {
continue
}
image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
return
}
panic("response did not include an image generation call")
}
Image API
Edit an image
import fs from "fs";
import OpenAI, { toFile } from "openai";
const client = new OpenAI();
const prompt = `
Generate a photorealistic image of a gift basket on a white background
labeled 'Relax & Unwind' with a ribbon and handwriting-like font,
containing all the items in the reference pictures.
`;
const imageFiles = [
"fixtures/bath-bomb.png",
"fixtures/body-lotion.png",
"fixtures/incense-kit.png",
"fixtures/soap.png",
];
const images = await Promise.all(
imageFiles.map(
async (file) =>
await toFile(fs.createReadStream(file), null, {
type: "image/png",
})
)
);
const response = await client.images.edit({
model: "gpt-image-2",
image: images,
prompt,
});
// Save the image to a file
const image_base64 = response.data[0].b64_json;
const image_bytes = Buffer.from(image_base64, "base64");
fs.writeFileSync("basket.png", image_bytes);
import base64
from openai import OpenAI
client = OpenAI()
prompt = """
Generate a photorealistic image of a gift basket on a white background
labeled 'Relax & Unwind' with a ribbon and handwriting-like font,
containing all the items in the reference pictures.
"""
result = client.images.edit(
model="gpt-image-2",
image=[
open("body-lotion.png", "rb"),
open("bath-bomb.png", "rb"),
open("incense-kit.png", "rb"),
open("soap.png", "rb"),
],
prompt=prompt,
)
image_base64 = result.data[0].b64_json
image_bytes = base64.b64decode(image_base64)
# Save the image to a file
with open("gift-basket.png", "wb") as f:
f.write(image_bytes)
package main
import (
"context"
"encoding/base64"
"io"
"os"
"github.com/openai/openai-go/v3"
)
func main() {
client := openai.NewClient()
files, closeFiles := openImages(
"bath-bomb.png",
"body-lotion.png",
"incense-kit.png",
"soap.png",
)
defer closeFiles()
response, err := client.Images.Edit(context.Background(), openai.ImageEditParams{
Model: openai.ImageModel("gpt-image-2"),
Image: openai.ImageEditParamsImageUnion{OfFileArray: files},
Prompt: "Generate a photorealistic image of a gift basket on a white background " +
"labeled 'Relax & Unwind' with a ribbon and handwriting-like font, containing all the items in the reference pictures.",
})
if err != nil {
panic(err)
}
saveImage("basket.png", response.Data[0].B64JSON)
}
func openImages(names ...string) ([]io.Reader, func()) {
images := make([]io.Reader, 0, len(names))
files := make([]*os.File, 0, len(names))
for _, name := range names {
file, err := os.Open(name)
if err != nil {
closeFiles(files)
panic(err)
}
images = append(images, openai.File(file, name, "image/png"))
files = append(files, file)
}
return images, func() { closeFiles(files) }
}
func closeFiles(files []*os.File) {
for _, file := range files {
if err := file.Close(); err != nil {
panic(err)
}
}
}
func saveImage(filename, encoded string) {
image, err := base64.StdEncoding.DecodeString(encoded)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
}
curl -s -D >(grep -i x-request-id >&2) \
-o >(jq -r '.data[0].b64_json' | base64 --decode > gift-basket.png) \
-X POST "https://api.openai.com/v1/images/edits" \
-H "Authorization: Bearer $OPENAI_API_KEY" \
-F "model=gpt-image-2" \
-F "image[]=@body-lotion.png" \
-F "image[]=@bath-bomb.png" \
-F "image[]=@incense-kit.png" \
-F "image[]=@soap.png" \
-F 'prompt=Generate a photorealistic image of a gift basket on a white background labeled "Relax & Unwind" with a ribbon and handwriting-like font, containing all the items in the reference pictures'
openai images edit \
--model gpt-image-2 \
--image body-lotion.png \
--image bath-bomb.png \
--image incense-kit.png \
--image soap.png \
--prompt 'Generate a photorealistic image of a gift basket on a white background labeled "Relax & Unwind" with a ribbon and handwriting-like font, containing all the items in the reference pictures' \
--raw-output \
--transform 'data.0.b64_json' | base64 --decode > gift-basket.png
Edit an image using a mask
You can provide a mask to indicate which part of the image should be edited.
When using a mask with GPT Image, additional instructions are sent to the model to help guide the editing process accordingly.
Masking with GPT Image is entirely prompt-based. The model uses the mask as guidance, but may not follow its exact shape with complete precision.
If you provide multiple input images, the mask will be applied to the first image.
Responses API
Edit an image with a mask
import fs from "fs";
import OpenAI from "openai";
const openai = new OpenAI();
async function createFile(filePath) {
const result = await openai.files.create({
file: fs.createReadStream(filePath),
purpose: "vision",
});
return result.id;
}
const fileId = await createFile("fixtures/sunlit_lounge.png");
const maskId = await createFile("fixtures/mask.png");
const response = await openai.responses.create({
model: "gpt-5.6",
input: [
{
role: "user",
content: [
{
type: "input_text",
text: "generate an image of the same sunlit indoor lounge area with a pool but the pool should contain a flamingo",
},
{
type: "input_image",
file_id: fileId,
detail: "auto",
},
],
},
],
tools: [
{
type: "image_generation",
quality: "high",
input_image_mask: {
file_id: maskId,
},
},
],
});
const imageData = response.output
.filter((output) => output.type === "image_generation_call")
.map((output) => output.result);
if (imageData.length > 0) {
const imageBase64 = imageData[0];
fs.writeFileSync("lounge.png", Buffer.from(imageBase64, "base64"));
}
from openai import OpenAI
import base64
client = OpenAI()
def create_file(file_path):
with open(file_path, "rb") as file_content:
result = client.files.create(file=file_content, purpose="vision")
return result.id
fileId = create_file("sunlit_lounge.png")
maskId = create_file("mask.png")
response = client.responses.create(
model="gpt-5.6",
input=[
{
"role": "user",
"content": [
{
"type": "input_text",
"text": "generate an image of the same sunlit indoor lounge area with a pool but the pool should contain a flamingo",
},
{
"type": "input_image",
"file_id": fileId,
},
],
},
],
tools=[
{
"type": "image_generation",
"quality": "high",
"input_image_mask": {
"file_id": maskId,
},
},
],
)
image_data = [
output.result
for output in response.output
if output.type == "image_generation_call"
]
if image_data:
image_base64 = image_data[0]
with open("lounge.png", "wb") as f:
f.write(base64.b64decode(image_base64))
package main
import (
"context"
"encoding/base64"
"os"
"github.com/openai/openai-go/v3"
"github.com/openai/openai-go/v3/responses"
)
func main() {
client := openai.NewClient()
imageID := uploadImage(client, "sunlit_lounge.png")
maskID := uploadImage(client, "mask.png")
response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
Model: "gpt-5.6",
Input: responses.ResponseNewParamsInputUnion{OfInputItemList: responses.ResponseInputParam{
responses.ResponseInputItemParamOfMessage(
responses.ResponseInputMessageContentListParam{
responses.ResponseInputContentParamOfInputText("Generate an image of the same sunlit indoor lounge area with a pool, but the pool should contain a flamingo."),
{OfInputImage: &responses.ResponseInputImageParam{FileID: openai.String(imageID), Detail: responses.ResponseInputImageDetailAuto}},
},
responses.EasyInputMessageRoleUser,
),
}},
Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{
Quality: "high",
InputImageMask: responses.ToolImageGenerationInputImageMaskParam{FileID: openai.String(maskID)},
}}},
})
if err != nil {
panic(err)
}
saveFirstGeneratedImage(response, "lounge.png")
}
func uploadImage(client openai.Client, filename string) string {
file, err := os.Open(filename)
if err != nil {
panic(err)
}
defer file.Close()
uploaded, err := client.Files.New(context.Background(), openai.FileNewParams{File: file, Purpose: openai.FilePurposeVision})
if err != nil {
panic(err)
}
return uploaded.ID
}
func saveFirstGeneratedImage(response *responses.Response, filename string) {
for _, output := range response.Output {
if output.Type != "image_generation_call" {
continue
}
image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
if err != nil {
panic(err)
}
if err := os.WriteFile(filename, image, 0o600); err != nil {
panic(err)
}
return
}
panic("response did not include an image generation call")
}
Image API
Edit an image with a mask
import fs from "fs";
import OpenAI, { toFile } from "openai";
const client = new OpenAI();
const rsp = await client.images.edit({
model: "gpt-image-2",
image: await toFile(fs.createReadStream("fixtures/sunlit_lounge.png"), null, {
type: "image/png",
}),
mask: await toFile(fs.createReadStream("fixtures/mask.png"), null, {
type: "image/png",
}),
prompt: "A sunlit indoor lounge area with a pool containing a flamingo",
});
// Save the image to a file
const image_base64 = rsp.data[0].b64_json;
const image_bytes = Buffer.from(image_base64, "base64");
fs.writeFileSync("lounge.png", image_bytes);
from openai import OpenAI
import base64
client = OpenAI()
result = client.images.edit(
model="gpt-image-2",
image=open("sunlit_lounge.png", "rb"),
mask=open("mask.png", "rb"),
prompt="A sunlit indoor lounge area with a pool containing a flamingo",
)
image_base64 = result.data[0].b64_json
image_bytes = base64.b64decode(image_base64)
# Save the image to a file
with open("composition.png", "wb") as f:
f.write(image_bytes)
package main
import (
"context"
"encoding/base64"
"os"
"github.com/openai/openai-go/v3"
)
func main() {
client := openai.NewClient()
image, err := os.Open("sunlit_lounge.png")
if err != nil {
panic(err)
}
defer image.Close()
mask, err := os.Open("mask.png")
if err != nil {
panic(err)
}
defer mask.Close()
response, err := client.Images.Edit(context.Background(), openai.ImageEditParams{
Model: openai.ImageModel("gpt-image-2"),
Image: openai.ImageEditParamsImageUnion{OfFile: openai.File(image, "sunlit_lounge.png", "image/png")},
Mask: openai.File(mask, "mask.png", "image/png"),
Prompt: "A sunlit indoor lounge area with a pool containing a flamingo",
})
if err != nil {
panic(err)
}
result, err := base64.StdEncoding.DecodeString(response.Data[0].B64JSON)
if err != nil {
panic(err)
}
if err := os.WriteFile("lounge.png", result, 0o600); err != nil {
panic(err)
}
}
curl -s -D >(grep -i x-request-id >&2) \
-o >(jq -r '.data[0].b64_json' | base64 --decode > lounge.png) \
-X POST "https://api.openai.com/v1/images/edits" \
-H "Authorization: Bearer $OPENAI_API_KEY" \
-F "model=gpt-image-2" \
-F "mask=@mask.png" \
-F "image[]=@sunlit_lounge.png" \
-F 'prompt=A sunlit indoor lounge area with a pool containing a flamingo'
openai images edit \
--model gpt-image-2 \
--image sunlit_lounge.png \
--mask mask.png \
--prompt "A sunlit indoor lounge area with a pool containing a flamingo" \
--raw-output \
--transform 'data.0.b64_json' | base64 --decode > out.png
| Image | Mask | Output |
|---|---|---|
![]() |
![]() |
![]() |
Prompt: a sunlit indoor lounge area with a pool containing a flamingo
Mask requirements
The image to edit and mask must be of the same format and size (less than 50MB in size).
The mask image must also contain an alpha channel. If you're using an image editing tool to create the mask, make sure to save the mask with an alpha channel.
You can modify a black and white image programmatically to add an alpha channel.
Add an alpha channel to a black and white mask
from PIL import Image
from io import BytesIO
# 1. Load your black & white mask as a grayscale image
mask = Image.open("mask.png").convert("L")
# 2. Convert it to RGBA so it has space for an alpha channel
mask_rgba = mask.convert("RGBA")
# 3. Then use the mask itself to fill that alpha channel
mask_rgba.putalpha(mask)
# 4. Convert the mask into bytes
buf = BytesIO()
mask_rgba.save(buf, format="PNG")
mask_bytes = buf.getvalue()
# 5. Save the resulting file
img_path_mask_alpha = "mask_alpha.png"
with open(img_path_mask_alpha, "wb") as f:
f.write(mask_bytes)
package main
import (
"image"
"image/color"
"image/png"
"os"
)
func main() {
file, err := os.Open("mask.png")
if err != nil {
panic(err)
}
defer file.Close()
mask, _, err := image.Decode(file)
if err != nil {
panic(err)
}
bounds := mask.Bounds()
withAlpha := image.NewNRGBA(bounds)
for y := bounds.Min.Y; y < bounds.Max.Y; y++ {
for x := bounds.Min.X; x < bounds.Max.X; x++ {
gray := color.GrayModel.Convert(mask.At(x, y)).(color.Gray)
withAlpha.SetNRGBA(x, y, color.NRGBA{R: gray.Y, G: gray.Y, B: gray.Y, A: gray.Y})
}
}
output, err := os.Create("mask_alpha.png")
if err != nil {
panic(err)
}
if err := png.Encode(output, withAlpha); err != nil {
panic(err)
}
if err := output.Close(); err != nil {
panic(err)
}
}
Image input fidelity
The input_fidelity parameter controls how strongly a model preserves details from input images during edits and reference-image workflows. For gpt-image-2, omit this parameter; the API doesn't allow changing it because the model processes every image input at high fidelity automatically.
Because gpt-image-2 always processes image inputs at high fidelity, image
input tokens can be higher for edit requests that include reference images. To
understand the cost implications, refer to the vision
costs
section.
Customize Image Output
You can configure the following output options:
- Size: Image dimensions (for example,
1024x1024,1024x1536) - Quality: Rendering quality (for example,
low,medium,high) - Format: File output format
- Compression: Compression level (0-100%) for JPEG and WebP formats
- Background: Opaque or automatic
size, quality, and background support the auto option, where the model will automatically select the best option based on the prompt.
gpt-image-2 doesn't currently support transparent backgrounds. Requests with
background: "transparent" aren't supported for this model.
Size and quality options
gpt-image-2 accepts any resolution in the size parameter when it satisfies the constraints below. Square images are typically fastest to generate.
| Popular sizes |
|
| Size constraints |
|
| Quality options |
|
Use quality: "low" for fast drafts, thumbnails, and quick iterations. It is
the fastest option and works well for many common use cases before you move to
medium or high for final assets.
Outputs that contain more than 2560x1440 (3,686,400) total pixels,
typically referred to as 2K, are considered experimental.
Output format
The Image API returns base64-encoded image data.
The default format is png, but you can also request jpeg or webp.
If using jpeg or webp, you can also specify the output_compression parameter to control the compression level (0-100%). For example, output_compression=50 will compress the image by 50%.
Using jpeg is faster than png, so you should prioritize this format if
latency is a concern.
Limitations
GPT Image models (gpt-image-2, gpt-image-1.5, gpt-image-1, and gpt-image-1-mini) are powerful and versatile image generation models, but they still have some limitations to be aware of:
- Latency: Complex prompts may take up to 2 minutes to process.
- Text Rendering: Although significantly improved, the model can still struggle with precise text placement and clarity.
- Consistency: While capable of producing consistent imagery, the model may occasionally struggle to maintain visual consistency for recurring characters or brand elements across multiple generations.
- Composition Control: Despite improved instruction following, the model may have difficulty placing elements precisely in structured or layout-sensitive compositions.
Content Moderation
All prompts and generated images are filtered in accordance with our content policy.
For image generation using GPT Image models (gpt-image-2, gpt-image-1.5, gpt-image-1, and gpt-image-1-mini), you can control moderation strictness with the moderation parameter. This parameter supports two values:
auto(default): Standard filtering that seeks to limit creating certain categories of potentially age-inappropriate content.low: Less restrictive filtering.
Handling blocked requests and other errors
Handle image generation failures the same way you handle other API errors: check the HTTP status or SDK exception type, log the request ID, and refer to the error codes guide for authentication, quota, rate-limit, and server failures. Retries are appropriate for transient failures like 429 and 5xx, but not for image generation user errors that require changing the request.
Some image generation failures are user-correctable and may return error.type = "image_generation_user_error". Don't automatically retry these errors without modifying the prompt or input images. For programmatic handling, use error.code as the stable discriminator.
When error.code = "moderation_blocked", the error may also include an optional error.moderation_details object:
{
"error": {
"type": "image_generation_user_error",
"code": "moderation_blocked",
"moderation_details": {
"moderation_stage": "input",
"categories": ["harassment"]
}
}
}
The moderation_details object provides coarse debugging context without exposing internal classifier labels or scores.
moderation_stage can be:
input: The block came from the prompt or request inputs.output: The block came from a generated image or downstream output moderation stage.unknown: A rare fallback when provenance is hard to determine.
categories contains coarse public labels. For example, you might see values like harassment, self-harm, sexual, or violence.
For most apps, keep the primary end-user message generic. Use moderation_details for developer logs, support workflows, analytics, and light remediation hints.
For example, if harassment appears, suggest removing abusive or targeting language. If the block happened at the input stage, guide the user to revise the prompt. If it happened at the output stage, treat it as a generated result safety block and distinguish it in your logs. Always branch on error.code = "moderation_blocked" first, and treat moderation_details as optional extra context.
Handle moderation-blocked image generation errors
import OpenAI from "openai";
const openai = new OpenAI();
try {
// The same error handling pattern applies to image generation requests,
// image edits, and Responses API tool calls that generate images.
await openai.images.generate({
model: "gpt-image-2",
prompt: "Create a poster humiliating my coworker with insulting captions",
});
} catch (error) {
if (error?.code !== "moderation_blocked") {
throw error;
}
const moderationDetails = error?.moderation_details;
const categories = moderationDetails?.categories ?? [];
const stage = moderationDetails?.moderation_stage;
let hint =
"This request could not be completed because it did not meet safety requirements.";
if (categories.includes("harassment")) {
hint =
"Try removing abusive or targeting language and focus on neutral visual details instead.";
} else if (stage === "input") {
hint =
"Try revising the prompt or input images and submit the request again.";
} else if (stage === "output") {
hint =
"The generated result was blocked by a safety check. Try changing the prompt and generating again.";
}
console.error("Image generation blocked", {
request_id: error?.request_id,
code: error?.code,
moderation_details: moderationDetails,
});
console.log(hint);
}
import openai
from openai import OpenAI
client = OpenAI()
try:
# The same error handling pattern applies to image generation requests,
# image edits, and Responses API tool calls that generate images.
client.images.generate(
model="gpt-image-2",
prompt="Create a poster humiliating my coworker with insulting captions",
)
except openai.BadRequestError as error:
if error.code != "moderation_blocked":
raise
error_body = error.body if isinstance(error.body, dict) else {}
moderation_details = error_body.get("moderation_details") or {}
categories = moderation_details.get("categories") or []
stage = moderation_details.get("moderation_stage")
hint = "This request could not be completed because it did not meet safety requirements."
if "harassment" in categories:
hint = "Try removing abusive or targeting language and focus on neutral visual details instead."
elif stage == "input":
hint = "Try revising the prompt or input images and submit the request again."
elif stage == "output":
hint = "The generated result was blocked by a safety check. Try changing the prompt and generating again."
print(
"Image generation blocked",
{
"request_id": error.request_id,
"code": error.code,
"moderation_details": moderation_details,
},
)
print(hint)
package main
import (
"context"
"encoding/json"
"errors"
"fmt"
"slices"
"github.com/openai/openai-go/v3"
)
func main() {
client := openai.NewClient()
_, err := client.Images.Generate(context.Background(), openai.ImageGenerateParams{
Model: openai.ImageModel("gpt-image-2"),
Prompt: "Create a poster humiliating my coworker with insulting captions",
})
if err == nil {
return
}
var apiError *openai.Error
if !errors.As(err, &apiError) || apiError.Code != "moderation_blocked" {
panic(err)
}
var body struct {
ModerationDetails struct {
Categories []string `json:"categories"`
ModerationStage string `json:"moderation_stage"`
} `json:"moderation_details"`
}
if err := json.Unmarshal([]byte(apiError.RawJSON()), &body); err != nil {
panic(err)
}
hint := "This request could not be completed because it did not meet safety requirements."
if slices.Contains(body.ModerationDetails.Categories, "harassment") {
hint = "Try removing abusive or targeting language and focus on neutral visual details instead."
} else if body.ModerationDetails.ModerationStage == "input" {
hint = "Try revising the prompt or input images and submit the request again."
} else if body.ModerationDetails.ModerationStage == "output" {
hint = "The generated result was blocked by a safety check. Try changing the prompt and generating again."
}
fmt.Printf("Image generation blocked (%s): %s\n", apiError.Code, hint)
}
Supported models
When using image generation in the Responses API, gpt-5 and newer models should support the image generation tool. Check the model detail page for your model to confirm if your desired model can use the image generation tool.
Cost and latency
gpt-image-2 output tokens
For gpt-image-2, use the calculator to estimate output tokens from the requested quality and size:
Models prior to gpt-image-2
GPT Image models prior to gpt-image-2 generate images by first producing specialized image tokens. Both latency and eventual cost are proportional to the number of tokens required to render an image—larger image sizes and higher quality settings result in more tokens.
The number of tokens generated depends on image dimensions and quality:
| Quality | Square (1024×1024) | Portrait (1024×1536) | Landscape (1536×1024) |
|---|---|---|---|
| Low | 272 tokens | 408 tokens | 400 tokens |
| Medium | 1056 tokens | 1584 tokens | 1568 tokens |
| High | 4160 tokens | 6240 tokens | 6208 tokens |
Note that you will also need to account for input tokens: text tokens for the prompt and image tokens for the input images if editing images.
Because gpt-image-2 always processes image inputs at high fidelity, edit requests that include reference images can use more input tokens.
Refer to the pricing page for current text and image token prices, and use the Calculating costs section below to estimate request costs.
The final cost is the sum of:
- input text tokens
- input image tokens if using the edits endpoint
- image output tokens
Calculating costs
Use the pricing calculator below to estimate request costs for GPT Image models.
gpt-image-2 supports thousands of valid resolutions; the table below lists the
same sizes used for previous GPT Image models for comparison. For GPT Image 1.5,
GPT Image 1, and GPT Image 1 Mini, the legacy per-image output pricing table is
also listed below. You should still account for text and image input tokens when
estimating the total cost of a request.
A larger non-square resolution can sometimes produce fewer output tokens than a smaller or square resolution at the same quality setting.
| Model | Quality | 1024 x 1024 | 1024 x 1536 | 1536 x 1024 |
|---|---|---|---|---|
|
GPT Image 2
Additional sizes available
|
Partial images cost
If you want to stream image generation using the partial_images parameter, each partial image will incur an additional 100 image output tokens.
guides/image-generation.md +746 −0
110 f.write(image_bytes)110 f.write(image_bytes)
111```111```
112 112
113```go
114package main
115
116import (
117 "context"
118 "encoding/base64"
119 "os"
120
121 "github.com/openai/openai-go/v3"
122)
123
124func main() {
125 client := openai.NewClient()
126 result, err := client.Images.Generate(context.Background(), openai.ImageGenerateParams{
127 Model: openai.ImageModel("gpt-image-2"),
128 Prompt: "A children's book drawing of a veterinarian using a stethoscope to " +
129 "listen to the heartbeat of a baby otter.",
130 })
131 if err != nil {
132 panic(err)
133 }
134 image, err := base64.StdEncoding.DecodeString(result.Data[0].B64JSON)
135 if err != nil {
136 panic(err)
137 }
138 if err := os.WriteFile("otter.png", image, 0o600); err != nil {
139 panic(err)
140 }
141}
142```
143
113```bash144```bash
114curl -X POST "https://api.openai.com/v1/images/generations" \145curl -X POST "https://api.openai.com/v1/images/generations" \
115 -H "Authorization: Bearer $OPENAI_API_KEY" \146 -H "Authorization: Bearer $OPENAI_API_KEY" \
185 f.write(base64.b64decode(image_base64))216 f.write(base64.b64decode(image_base64))
186```217```
187 218
219```go
220package main
221
222import (
223 "context"
224 "encoding/base64"
225 "os"
226
227 "github.com/openai/openai-go/v3"
228 "github.com/openai/openai-go/v3/responses"
229)
230
231func main() {
232 client := openai.NewClient()
233 response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
234 Model: "gpt-5.6",
235 Input: responses.ResponseNewParamsInputUnion{
236 OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
237 },
238 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
239 })
240 if err != nil {
241 panic(err)
242 }
243 saveFirstGeneratedImage(response, "otter.png")
244}
245
246func saveFirstGeneratedImage(response *responses.Response, filename string) {
247 for _, output := range response.Output {
248 if output.Type != "image_generation_call" {
249 continue
250 }
251 image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
252 if err != nil {
253 panic(err)
254 }
255 if err := os.WriteFile(filename, image, 0o600); err != nil {
256 panic(err)
257 }
258 return
259 }
260 panic("response did not include an image generation call")
261}
262```
263
188 264
189 265
190### Multi-turn image generation266### Multi-turn image generation
244 f.write(base64.b64decode(image_base64))320 f.write(base64.b64decode(image_base64))
245```321```
246 322
323```go
324package main
325
326import (
327 "context"
328 "encoding/base64"
329 "os"
330
331 "github.com/openai/openai-go/v3"
332 "github.com/openai/openai-go/v3/responses"
333)
334
335func main() {
336 client := openai.NewClient()
337 response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
338 Model: "gpt-5.6",
339 Input: responses.ResponseNewParamsInputUnion{
340 OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
341 },
342 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{Action: "generate"}}},
343 })
344 if err != nil {
345 panic(err)
346 }
347 for _, output := range response.Output {
348 if output.Type != "image_generation_call" {
349 continue
350 }
351 image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
352 if err != nil {
353 panic(err)
354 }
355 if err := os.WriteFile("otter.png", image, 0o600); err != nil {
356 panic(err)
357 }
358 return
359 }
360 panic("response did not include an image generation call")
361}
362```
363
247 364
248If you force `edit` without providing an image in context, the call will return an error. Leave `action` at `auto` to have the model decide when to generate or edit.365If you force `edit` without providing an image in context, the call will return an error. Leave `action` at `auto` to have the model decide when to generate or edit.
249 366
343 f.write(base64.b64decode(image_base64))460 f.write(base64.b64decode(image_base64))
344```461```
345 462
463```go
464package main
465
466import (
467 "context"
468 "encoding/base64"
469 "os"
470
471 "github.com/openai/openai-go/v3"
472 "github.com/openai/openai-go/v3/responses"
473)
474
475func main() {
476 client := openai.NewClient()
477 first, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
478 Model: "gpt-5.6",
479 Input: responses.ResponseNewParamsInputUnion{
480 OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
481 },
482 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
483 })
484 if err != nil {
485 panic(err)
486 }
487 saveFirstGeneratedImage(first, "cat_and_otter.png")
488
489 followUp, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
490 Model: "gpt-5.6",
491 PreviousResponseID: openai.String(first.ID),
492 Input: responses.ResponseNewParamsInputUnion{
493 OfString: openai.String("Now make it look realistic"),
494 },
495 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
496 })
497 if err != nil {
498 panic(err)
499 }
500 saveFirstGeneratedImage(followUp, "cat_and_otter_realistic.png")
501}
502
503func saveFirstGeneratedImage(response *responses.Response, filename string) {
504 for _, output := range response.Output {
505 if output.Type != "image_generation_call" {
506 continue
507 }
508 image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
509 if err != nil {
510 panic(err)
511 }
512 if err := os.WriteFile(filename, image, 0o600); err != nil {
513 panic(err)
514 }
515 return
516 }
517 panic("response did not include an image generation call")
518}
519```
520
346 521
347 522
348 523
458 f.write(base64.b64decode(image_base64))633 f.write(base64.b64decode(image_base64))
459```634```
460 635
636```go
637package main
638
639import (
640 "context"
641 "encoding/base64"
642 "encoding/json"
643 "os"
644
645 "github.com/openai/openai-go/v3"
646 "github.com/openai/openai-go/v3/responses"
647)
648
649func main() {
650 client := openai.NewClient()
651 first, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
652 Model: "gpt-5.6",
653 Input: responses.ResponseNewParamsInputUnion{
654 OfString: openai.String("Generate an image of gray tabby cat hugging an otter with an orange scarf"),
655 },
656 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
657 })
658 if err != nil {
659 panic(err)
660 }
661 call := firstImageGenerationCall(first)
662 saveImage("cat_and_otter.png", call.Result)
663 input := outputAsInput(first.Output)
664 input = append(input, responses.ResponseInputItemParamOfMessage(
665 responses.ResponseInputMessageContentListParam{responses.ResponseInputContentParamOfInputText("Now make it look realistic")},
666 responses.EasyInputMessageRoleUser,
667 ))
668
669 followUp, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
670 Model: "gpt-5.6",
671 Input: responses.ResponseNewParamsInputUnion{OfInputItemList: input},
672 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
673 })
674 if err != nil {
675 panic(err)
676 }
677 saveImage("cat_and_otter_realistic.png", firstImageGenerationCall(followUp).Result)
678}
679
680func firstImageGenerationCall(response *responses.Response) responses.ResponseOutputItemImageGenerationCall {
681 for _, output := range response.Output {
682 if output.Type == "image_generation_call" {
683 return output.AsImageGenerationCall()
684 }
685 }
686 panic("response did not include an image generation call")
687}
688
689func outputAsInput(output []responses.ResponseOutputItemUnion) []responses.ResponseInputItemUnionParam {
690 input := make([]responses.ResponseInputItemUnionParam, 0, len(output))
691 for _, item := range output {
692 var converted responses.ResponseInputItemUnion
693 if err := json.Unmarshal([]byte(item.RawJSON()), &converted); err != nil {
694 panic(err)
695 }
696 input = append(input, converted.ToParam())
697 }
698 return input
699}
700
701func saveImage(filename, encoded string) {
702 image, err := base64.StdEncoding.DecodeString(encoded)
703 if err != nil {
704 panic(err)
705 }
706 if err := os.WriteFile(filename, image, 0o600); err != nil {
707 panic(err)
708 }
709}
710```
711
461 712
462 713
463#### Result714#### Result
584 save_base64_image("river-final.png", image_data[0])835 save_base64_image("river-final.png", image_data[0])
585```836```
586 837
838```go
839package main
840
841import (
842 "context"
843 "encoding/base64"
844 "fmt"
845 "os"
846
847 "github.com/openai/openai-go/v3"
848 "github.com/openai/openai-go/v3/responses"
849)
850
851func main() {
852 client := openai.NewClient()
853 stream := client.Responses.NewStreaming(context.Background(), responses.ResponseNewParams{
854 Model: "gpt-5.6",
855 Input: responses.ResponseNewParamsInputUnion{
856 OfString: openai.String("Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape"),
857 },
858 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{PartialImages: openai.Int(2)}}},
859 })
860 for stream.Next() {
861 event := stream.Current()
862 if event.Type == "response.image_generation_call.partial_image" {
863 partial := event.AsResponseImageGenerationCallPartialImage()
864 saveImage(fmt.Sprintf("river-partial-%d.png", partial.PartialImageIndex), partial.PartialImageB64)
865 }
866 if event.Type == "response.completed" {
867 for _, output := range event.AsResponseCompleted().Response.Output {
868 if output.Type == "image_generation_call" {
869 saveImage("river-final.png", output.AsImageGenerationCall().Result)
870 }
871 }
872 }
873 }
874 if err := stream.Err(); err != nil {
875 panic(err)
876 }
877}
878
879func saveImage(filename, encoded string) {
880 image, err := base64.StdEncoding.DecodeString(encoded)
881 if err != nil {
882 panic(err)
883 }
884 if err := os.WriteFile(filename, image, 0o600); err != nil {
885 panic(err)
886 }
887}
888```
889
587 890
588 891
589 892
640 f.write(image_bytes)943 f.write(image_bytes)
641```944```
642 945
946```go
947package main
948
949import (
950 "context"
951 "encoding/base64"
952 "fmt"
953 "os"
954
955 "github.com/openai/openai-go/v3"
956)
957
958func main() {
959 client := openai.NewClient()
960 stream := client.Images.GenerateStreaming(context.Background(), openai.ImageGenerateParams{
961 Model: openai.ImageModel("gpt-image-2"),
962 Prompt: "Draw a gorgeous image of a river made of white owl feathers, snaking its way through a serene winter landscape",
963 PartialImages: openai.Int(2),
964 })
965 for stream.Next() {
966 event := stream.Current()
967 if event.Type != "image_generation.partial_image" {
968 continue
969 }
970 partial := event.AsImageGenerationPartialImage()
971 saveImage(fmt.Sprintf("river%d.png", partial.PartialImageIndex), partial.B64JSON)
972 }
973 if err := stream.Err(); err != nil {
974 panic(err)
975 }
976}
977
978func saveImage(filename, encoded string) {
979 image, err := base64.StdEncoding.DecodeString(encoded)
980 if err != nil {
981 panic(err)
982 }
983 if err := os.WriteFile(filename, image, 0o600); err != nil {
984 panic(err)
985 }
986}
987```
988
643 989
644 990
645#### Result991#### Result
739 return result.id1085 return result.id
740```1086```
741 1087
1088```go
1089package main
1090
1091import (
1092 "context"
1093 "fmt"
1094 "os"
1095
1096 "github.com/openai/openai-go/v3"
1097)
1098
1099func main() {
1100 client := openai.NewClient()
1101 file, err := os.Open("image.png")
1102 if err != nil {
1103 panic(err)
1104 }
1105 defer file.Close()
1106
1107 uploaded, err := client.Files.New(context.Background(), openai.FileNewParams{
1108 File: file,
1109 Purpose: openai.FilePurposeVision,
1110 })
1111 if err != nil {
1112 panic(err)
1113 }
1114 fmt.Println(uploaded.ID)
1115}
1116```
1117
742 1118
743#### Create a base64 encoded image1119#### Create a base64 encoded image
744 1120
763 return base64_image1139 return base64_image
764```1140```
765 1141
1142```go
1143package main
1144
1145import (
1146 "encoding/base64"
1147 "fmt"
1148 "os"
1149)
1150
1151func main() {
1152 image, err := os.ReadFile("image.png")
1153 if err != nil {
1154 panic(err)
1155 }
1156 fmt.Println(base64.StdEncoding.EncodeToString(image))
1157}
1158```
1159
766 1160
767Edit an image1161Edit an image
768 1162
908 print(response.output_text)1302 print(response.output_text)
909```1303```
910 1304
1305```go
1306package main
1307
1308import (
1309 "context"
1310 "encoding/base64"
1311 "os"
1312
1313 "github.com/openai/openai-go/v3"
1314 "github.com/openai/openai-go/v3/responses"
1315)
1316
1317func main() {
1318 client := openai.NewClient()
1319 bathBombID := uploadImage(client, "bath-bomb.png")
1320 incenseKitID := uploadImage(client, "incense-kit.png")
1321
1322 response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
1323 Model: "gpt-5.6",
1324 Input: responses.ResponseNewParamsInputUnion{OfInputItemList: responses.ResponseInputParam{
1325 responses.ResponseInputItemParamOfMessage(
1326 responses.ResponseInputMessageContentListParam{
1327 responses.ResponseInputContentParamOfInputText("Generate a photorealistic image of a gift basket on a white background labeled 'Relax & Unwind' with a ribbon and handwriting-like font, containing all the items in the reference pictures."),
1328 {OfInputImage: &responses.ResponseInputImageParam{ImageURL: openai.String(dataURL("body-lotion.png")), Detail: responses.ResponseInputImageDetailAuto}},
1329 {OfInputImage: &responses.ResponseInputImageParam{ImageURL: openai.String(dataURL("soap.png")), Detail: responses.ResponseInputImageDetailAuto}},
1330 {OfInputImage: &responses.ResponseInputImageParam{FileID: openai.String(bathBombID), Detail: responses.ResponseInputImageDetailAuto}},
1331 {OfInputImage: &responses.ResponseInputImageParam{FileID: openai.String(incenseKitID), Detail: responses.ResponseInputImageDetailAuto}},
1332 },
1333 responses.EasyInputMessageRoleUser,
1334 ),
1335 }},
1336 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{}}},
1337 })
1338 if err != nil {
1339 panic(err)
1340 }
1341 saveFirstGeneratedImage(response, "gift-basket.png")
1342}
1343
1344func uploadImage(client openai.Client, filename string) string {
1345 file, err := os.Open(filename)
1346 if err != nil {
1347 panic(err)
1348 }
1349 defer file.Close()
1350 uploaded, err := client.Files.New(context.Background(), openai.FileNewParams{File: file, Purpose: openai.FilePurposeVision})
1351 if err != nil {
1352 panic(err)
1353 }
1354 return uploaded.ID
1355}
1356
1357func dataURL(filename string) string {
1358 image, err := os.ReadFile(filename)
1359 if err != nil {
1360 panic(err)
1361 }
1362 return "data:image/png;base64," + base64.StdEncoding.EncodeToString(image)
1363}
1364
1365func saveFirstGeneratedImage(response *responses.Response, filename string) {
1366 for _, output := range response.Output {
1367 if output.Type != "image_generation_call" {
1368 continue
1369 }
1370 image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
1371 if err != nil {
1372 panic(err)
1373 }
1374 if err := os.WriteFile(filename, image, 0o600); err != nil {
1375 panic(err)
1376 }
1377 return
1378 }
1379 panic("response did not include an image generation call")
1380}
1381```
1382
911 1383
912 1384
913 1385
989 f.write(image_bytes)1461 f.write(image_bytes)
990```1462```
991 1463
1464```go
1465package main
1466
1467import (
1468 "context"
1469 "encoding/base64"
1470 "io"
1471 "os"
1472
1473 "github.com/openai/openai-go/v3"
1474)
1475
1476func main() {
1477 client := openai.NewClient()
1478 files, closeFiles := openImages(
1479 "bath-bomb.png",
1480 "body-lotion.png",
1481 "incense-kit.png",
1482 "soap.png",
1483 )
1484 defer closeFiles()
1485
1486 response, err := client.Images.Edit(context.Background(), openai.ImageEditParams{
1487 Model: openai.ImageModel("gpt-image-2"),
1488 Image: openai.ImageEditParamsImageUnion{OfFileArray: files},
1489 Prompt: "Generate a photorealistic image of a gift basket on a white background " +
1490 "labeled 'Relax & Unwind' with a ribbon and handwriting-like font, containing all the items in the reference pictures.",
1491 })
1492 if err != nil {
1493 panic(err)
1494 }
1495 saveImage("basket.png", response.Data[0].B64JSON)
1496}
1497
1498func openImages(names ...string) ([]io.Reader, func()) {
1499 images := make([]io.Reader, 0, len(names))
1500 files := make([]*os.File, 0, len(names))
1501 for _, name := range names {
1502 file, err := os.Open(name)
1503 if err != nil {
1504 closeFiles(files)
1505 panic(err)
1506 }
1507 images = append(images, openai.File(file, name, "image/png"))
1508 files = append(files, file)
1509 }
1510 return images, func() { closeFiles(files) }
1511}
1512
1513func closeFiles(files []*os.File) {
1514 for _, file := range files {
1515 if err := file.Close(); err != nil {
1516 panic(err)
1517 }
1518 }
1519}
1520
1521func saveImage(filename, encoded string) {
1522 image, err := base64.StdEncoding.DecodeString(encoded)
1523 if err != nil {
1524 panic(err)
1525 }
1526 if err := os.WriteFile(filename, image, 0o600); err != nil {
1527 panic(err)
1528 }
1529}
1530```
1531
992```bash1532```bash
993curl -s -D >(grep -i x-request-id >&2) \1533curl -s -D >(grep -i x-request-id >&2) \
994 -o >(jq -r '.data[0].b64_json' | base64 --decode > gift-basket.png) \1534 -o >(jq -r '.data[0].b64_json' | base64 --decode > gift-basket.png) \
1145 f.write(base64.b64decode(image_base64))1685 f.write(base64.b64decode(image_base64))
1146```1686```
1147 1687
1688```go
1689package main
1690
1691import (
1692 "context"
1693 "encoding/base64"
1694 "os"
1695
1696 "github.com/openai/openai-go/v3"
1697 "github.com/openai/openai-go/v3/responses"
1698)
1699
1700func main() {
1701 client := openai.NewClient()
1702 imageID := uploadImage(client, "sunlit_lounge.png")
1703 maskID := uploadImage(client, "mask.png")
1704 response, err := client.Responses.New(context.Background(), responses.ResponseNewParams{
1705 Model: "gpt-5.6",
1706 Input: responses.ResponseNewParamsInputUnion{OfInputItemList: responses.ResponseInputParam{
1707 responses.ResponseInputItemParamOfMessage(
1708 responses.ResponseInputMessageContentListParam{
1709 responses.ResponseInputContentParamOfInputText("Generate an image of the same sunlit indoor lounge area with a pool, but the pool should contain a flamingo."),
1710 {OfInputImage: &responses.ResponseInputImageParam{FileID: openai.String(imageID), Detail: responses.ResponseInputImageDetailAuto}},
1711 },
1712 responses.EasyInputMessageRoleUser,
1713 ),
1714 }},
1715 Tools: []responses.ToolUnionParam{{OfImageGeneration: &responses.ToolImageGenerationParam{
1716 Quality: "high",
1717 InputImageMask: responses.ToolImageGenerationInputImageMaskParam{FileID: openai.String(maskID)},
1718 }}},
1719 })
1720 if err != nil {
1721 panic(err)
1722 }
1723 saveFirstGeneratedImage(response, "lounge.png")
1724}
1725
1726func uploadImage(client openai.Client, filename string) string {
1727 file, err := os.Open(filename)
1728 if err != nil {
1729 panic(err)
1730 }
1731 defer file.Close()
1732 uploaded, err := client.Files.New(context.Background(), openai.FileNewParams{File: file, Purpose: openai.FilePurposeVision})
1733 if err != nil {
1734 panic(err)
1735 }
1736 return uploaded.ID
1737}
1738
1739func saveFirstGeneratedImage(response *responses.Response, filename string) {
1740 for _, output := range response.Output {
1741 if output.Type != "image_generation_call" {
1742 continue
1743 }
1744 image, err := base64.StdEncoding.DecodeString(output.AsImageGenerationCall().Result)
1745 if err != nil {
1746 panic(err)
1747 }
1748 if err := os.WriteFile(filename, image, 0o600); err != nil {
1749 panic(err)
1750 }
1751 return
1752 }
1753 panic("response did not include an image generation call")
1754}
1755```
1756
1148 1757
1149 1758
1150 1759
1198 f.write(image_bytes)1807 f.write(image_bytes)
1199```1808```
1200 1809
1810```go
1811package main
1812
1813import (
1814 "context"
1815 "encoding/base64"
1816 "os"
1817
1818 "github.com/openai/openai-go/v3"
1819)
1820
1821func main() {
1822 client := openai.NewClient()
1823 image, err := os.Open("sunlit_lounge.png")
1824 if err != nil {
1825 panic(err)
1826 }
1827 defer image.Close()
1828 mask, err := os.Open("mask.png")
1829 if err != nil {
1830 panic(err)
1831 }
1832 defer mask.Close()
1833
1834 response, err := client.Images.Edit(context.Background(), openai.ImageEditParams{
1835 Model: openai.ImageModel("gpt-image-2"),
1836 Image: openai.ImageEditParamsImageUnion{OfFile: openai.File(image, "sunlit_lounge.png", "image/png")},
1837 Mask: openai.File(mask, "mask.png", "image/png"),
1838 Prompt: "A sunlit indoor lounge area with a pool containing a flamingo",
1839 })
1840 if err != nil {
1841 panic(err)
1842 }
1843 result, err := base64.StdEncoding.DecodeString(response.Data[0].B64JSON)
1844 if err != nil {
1845 panic(err)
1846 }
1847 if err := os.WriteFile("lounge.png", result, 0o600); err != nil {
1848 panic(err)
1849 }
1850}
1851```
1852
1201```bash1853```bash
1202curl -s -D >(grep -i x-request-id >&2) \1854curl -s -D >(grep -i x-request-id >&2) \
1203 -o >(jq -r '.data[0].b64_json' | base64 --decode > lounge.png) \1855 -o >(jq -r '.data[0].b64_json' | base64 --decode > lounge.png) \
1271 f.write(mask_bytes)1923 f.write(mask_bytes)
1272```1924```
1273 1925
1926```go
1927package main
1928
1929import (
1930 "image"
1931 "image/color"
1932 "image/png"
1933 "os"
1934)
1935
1936func main() {
1937 file, err := os.Open("mask.png")
1938 if err != nil {
1939 panic(err)
1940 }
1941 defer file.Close()
1942
1943 mask, _, err := image.Decode(file)
1944 if err != nil {
1945 panic(err)
1946 }
1947 bounds := mask.Bounds()
1948 withAlpha := image.NewNRGBA(bounds)
1949 for y := bounds.Min.Y; y < bounds.Max.Y; y++ {
1950 for x := bounds.Min.X; x < bounds.Max.X; x++ {
1951 gray := color.GrayModel.Convert(mask.At(x, y)).(color.Gray)
1952 withAlpha.SetNRGBA(x, y, color.NRGBA{R: gray.Y, G: gray.Y, B: gray.Y, A: gray.Y})
1953 }
1954 }
1955
1956 output, err := os.Create("mask_alpha.png")
1957 if err != nil {
1958 panic(err)
1959 }
1960 if err := png.Encode(output, withAlpha); err != nil {
1961 panic(err)
1962 }
1963 if err := output.Close(); err != nil {
1964 panic(err)
1965 }
1966}
1967```
1968
1274 1969
1275### Image input fidelity1970### Image input fidelity
1276 1971
1537 print(hint)2232 print(hint)
1538```2233```
1539 2234
2235```go
2236package main
2237
2238import (
2239 "context"
2240 "encoding/json"
2241 "errors"
2242 "fmt"
2243 "slices"
2244
2245 "github.com/openai/openai-go/v3"
2246)
2247
2248func main() {
2249 client := openai.NewClient()
2250 _, err := client.Images.Generate(context.Background(), openai.ImageGenerateParams{
2251 Model: openai.ImageModel("gpt-image-2"),
2252 Prompt: "Create a poster humiliating my coworker with insulting captions",
2253 })
2254 if err == nil {
2255 return
2256 }
2257
2258 var apiError *openai.Error
2259 if !errors.As(err, &apiError) || apiError.Code != "moderation_blocked" {
2260 panic(err)
2261 }
2262
2263 var body struct {
2264 ModerationDetails struct {
2265 Categories []string `json:"categories"`
2266 ModerationStage string `json:"moderation_stage"`
2267 } `json:"moderation_details"`
2268 }
2269 if err := json.Unmarshal([]byte(apiError.RawJSON()), &body); err != nil {
2270 panic(err)
2271 }
2272
2273 hint := "This request could not be completed because it did not meet safety requirements."
2274 if slices.Contains(body.ModerationDetails.Categories, "harassment") {
2275 hint = "Try removing abusive or targeting language and focus on neutral visual details instead."
2276 } else if body.ModerationDetails.ModerationStage == "input" {
2277 hint = "Try revising the prompt or input images and submit the request again."
2278 } else if body.ModerationDetails.ModerationStage == "output" {
2279 hint = "The generated result was blocked by a safety check. Try changing the prompt and generating again."
2280 }
2281
2282 fmt.Printf("Image generation blocked (%s): %s\n", apiError.Code, hint)
2283}
2284```
2285
1540 2286
1541### Supported models2287### Supported models
1542 2288





