pi-codex-image-gen 0.1.11 → 0.1.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "pi-codex-image-gen",
3
- "version": "0.1.11",
4
- "description": "Image generation for Pi using the ChatGPT Images 2.0 model.",
3
+ "version": "0.1.13",
4
+ "description": "Image generation and editing for Pi using your ChatGPT Codex login.",
5
5
  "type": "module",
6
6
  "license": "Apache-2.0",
7
7
  "author": "Jose Mocito",
@@ -41,6 +41,7 @@
41
41
  "scripts": {
42
42
  "check": "tsc --noEmit",
43
43
  "typecheck": "tsc --noEmit",
44
+ "test": "rm -rf .test-dist && tsc --noEmit false --outDir .test-dist && node --test tests/*.test.mjs; status=$?; rm -rf .test-dist; exit $status",
44
45
  "pack:dry-run": "npm pack --dry-run"
45
46
  },
46
47
  "pi": {
@@ -57,16 +58,19 @@
57
58
  "typebox": "*"
58
59
  },
59
60
  "devDependencies": {
60
- "@earendil-works/pi-ai": "^0.80.0",
61
- "@earendil-works/pi-coding-agent": "^0.80.0",
62
- "typebox": "^1.2.10",
63
- "@types/node": "^25.9.3",
64
- "typescript": "^6.0.3"
61
+ "@earendil-works/pi-ai": "^0.85.1",
62
+ "@earendil-works/pi-coding-agent": "^0.85.1",
63
+ "@types/node": "^26.2.0",
64
+ "typebox": "^1.3.10",
65
+ "typescript": "^7.0.2"
65
66
  },
66
67
  "publishConfig": {
67
68
  "access": "public"
68
69
  },
69
70
  "engines": {
70
71
  "node": ">=20.6.0"
72
+ },
73
+ "dependencies": {
74
+ "@mocito/install-telemetry": "0.1.1"
71
75
  }
72
76
  }
@@ -7,7 +7,7 @@ description: "Generate or edit raster images when the task benefits from AI-crea
7
7
 
8
8
  > Adapted from OpenAI Codex's `imagegen` skill for Pi.
9
9
  > Original source: https://github.com/openai/codex/tree/main/codex-rs/skills/src/assets/samples/imagegen
10
- > Required Pi modifications: use Pi's `codex_generate_image` tool name, Pi artifact paths, and the bundled helper path; direct image editing remains CLI fallback unless the Pi tool grows edit support.
10
+ > Required Pi modifications: use Pi's `codex_generate_image` tool name, Pi artifact paths, and the bundled helper path.
11
11
 
12
12
  Generates or edits images for the current project (for example website assets, game assets, UI mockups, product mockups, wireframes, logo design, photorealistic images, or infographics).
13
13
 
@@ -15,8 +15,8 @@ Generates or edits images for the current project (for example website assets, g
15
15
 
16
16
  This skill has exactly two top-level modes:
17
17
 
18
- - **Default Pi tool mode (preferred):** Pi `codex_generate_image` tool for normal image generation, editing, and simple transparent-image requests. Does not require `OPENAI_API_KEY`.
19
- - **Fallback CLI mode:** `scripts/image_gen.py` CLI. Use when the user explicitly asks for the CLI/API/model path, or after the user explicitly confirms a true model-native transparency fallback with `gpt-image-1.5`. Requires `OPENAI_API_KEY`.
18
+ - **Default Pi tool mode (preferred):** Pi `codex_generate_image` tool for new image generation, edits using up to five local or recent conversation images, reference variants, and simple transparent-image requests. Does not require `OPENAI_API_KEY`.
19
+ - **Fallback CLI mode:** `scripts/image_gen.py` CLI. Use when the user explicitly asks for the CLI/API/model path, or after the user explicitly confirms a native transparency API fallback. Requires `OPENAI_API_KEY` and separate API billing.
20
20
 
21
21
  Within CLI fallback, the CLI exposes three subcommands:
22
22
 
@@ -25,11 +25,14 @@ Within CLI fallback, the CLI exposes three subcommands:
25
25
  - `generate-batch`
26
26
 
27
27
  Rules:
28
- - Use the Pi `codex_generate_image` tool by default for normal image generation and editing requests.
29
- - Do not switch to CLI fallback for ordinary quality, size, or file-path control.
28
+ - Use the Pi `codex_generate_image` tool by default for new image generation requests.
29
+ - The tool's `model` parameter selects a Codex routing model, not Flare or Sunburst. Do not claim a specific served image model unless the response reports it. Read `details.reportedImage` for backend-reported output settings, then inspect the actual image; prompt requests for quality, dimensions, or transparency are not guarantees.
30
+ - Do not automatically repeat quota, connection, timeout, or incomplete-stream failures. The remote generation may already have consumed quota.
31
+ - Use `referencedImagePaths` for edits when every target has a local path. Use `numLastImagesToInclude` only when a target is available solely in recent conversation history. Never provide both selectors. Masks and advanced CLI-only controls still require confirmed CLI fallback.
32
+ - Do not switch to CLI fallback for ordinary generation quality, size, or output file-path control.
30
33
  - If the user explicitly asks for a transparent image/background, stay on Pi `codex_generate_image` first: prompt for a flat removable chroma-key background, then remove it locally with the installed helper at `scripts/remove_chroma_key.py`.
31
34
  - Never silently switch from Pi `codex_generate_image` or CLI `gpt-image-2` to CLI `gpt-image-1.5`. Treat this as a model/path downgrade and ask the user before doing it, unless the user has already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback.
32
- - If a transparent request appears too complex for clean chroma-key removal, asks for true/native transparency, or local removal fails validation, explain that true transparency requires CLI `gpt-image-1.5 --background transparent --output-format png` because `gpt-image-2` does not support `background=transparent`, then ask whether to proceed. Run the CLI fallback only after the user confirms.
35
+ - If a transparent request appears too complex for clean chroma-key removal, asks for true/native transparency, or local removal fails validation, offer CLI `gpt-image-2 --background transparent --output-format png` (native transparency preview). Run the CLI fallback only after the user confirms.
33
36
  - The word `batch` by itself does not mean CLI fallback. If the user asks for many assets or says to batch-generate assets without explicitly asking for CLI/API/model controls, stay on the Pi tool path and issue one Pi tool call per requested asset or variant.
34
37
  - If the Pi tool fails or is unavailable, tell the user the CLI fallback exists and that it requires `OPENAI_API_KEY`. Proceed only if the user explicitly asks for that fallback.
35
38
  - If the user explicitly asks for CLI mode, use the bundled `scripts/image_gen.py` workflow. Do not create one-off SDK runners.
@@ -38,12 +41,13 @@ Rules:
38
41
  Pi tool save-path policy:
39
42
  - In Pi tool mode, generated images are saved under Pi's agent directory by default: `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*`. The default Pi agent directory is `~/.pi/agent`, but it can be overridden with `PI_CODING_AGENT_DIR`; use Pi's configured agent directory, not a hardcoded home path.
40
43
  - Do not describe or rely on OS temp as the default Pi tool destination.
41
- - Do not describe or rely on a destination-path argument (if any) on the Pi `codex_generate_image` tool. If a specific location is needed, generate first and then move or copy the selected output from `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*`.
44
+ - Use the tool's `save` and `saveDir` controls to choose a save directory. Custom mode appends a session directory; it does not accept an exact output filename. If an exact asset path is needed, copy the generated image there and leave the original in place.
42
45
  - Save-path precedence in Pi tool mode:
43
- 1. If the user names a destination, move or copy the selected output there.
44
- 2. If the image is meant for the current project, move or copy the final selected image into the workspace before finishing.
46
+ 1. If the user names a destination, copy the selected output there and leave the original in place.
47
+ 2. If the image is meant for the current project, copy the final selected image into the workspace before finishing and leave the original in place.
45
48
  3. If the image is only for preview or brainstorming, render it inline; the underlying file can remain at the default `<pi-agent-dir>/generated-images/<pi-session-id>/*` path.
46
49
  - Never leave a project-referenced asset only at the default `<pi-agent-dir>/generated-images/<pi-session-id>/*` path.
50
+ - Move or delete a Pi-generated original only when the user explicitly requests it.
47
51
  - Do not overwrite an existing asset unless the user explicitly asked for replacement; otherwise create a sibling versioned filename such as `hero-v2.png` or `item-icon-edited.png`.
48
52
 
49
53
  Shared prompt guidance for both modes lives in `references/prompting.md` and `references/sample-prompts.md`.
@@ -82,9 +86,10 @@ Intent:
82
86
  - If the user provides no images, treat the request as **generate**.
83
87
 
84
88
  Pi edit semantics:
85
- - The current Pi `codex_generate_image` tool is for new image generation. Do not promise arbitrary filesystem-path editing through the Pi tool.
86
- - If the user wants to edit an existing image, use the explicit CLI fallback only when the user asks for it or confirms it.
87
- - If a local file needs direct file-path control, masks, or other explicit CLI-only parameters, use the explicit CLI fallback only after confirmation.
89
+ - Use `referencedImagePaths` when all edit targets have readable local paths, with at most five paths.
90
+ - Use `numLastImagesToInclude` for the smallest recent-conversation window containing all targets, from one to five images.
91
+ - Never provide both image selectors. If neither can include every target, ask the user to attach or provide the missing image.
92
+ - Use confirmed CLI fallback only for masks or other explicit CLI-only parameters.
88
93
  - For edits, preserve invariants aggressively and save non-destructively by default.
89
94
 
90
95
  Execution strategy:
@@ -95,7 +100,7 @@ Execution strategy:
95
100
  Assume the user wants a new image unless they clearly ask to change an existing one.
96
101
 
97
102
  ## Workflow
98
- 1. Decide the top-level mode: Pi tool by default, including simple transparent-output requests; fallback CLI only if explicitly requested or after the user explicitly confirms a transparent-output fallback.
103
+ 1. Decide the top-level mode: Pi tool by default for generation, supported edits, and simple transparent-output requests; fallback CLI only if explicitly requested or after the user confirms an unsupported edit control or transparent-output fallback.
99
104
  2. Decide the intent: `generate` or `edit`.
100
105
  3. Decide whether the output is preview-only or meant to be consumed by the current project.
101
106
  4. Decide the execution strategy: single asset vs repeated Pi tool calls vs CLI `generate-batch`.
@@ -104,17 +109,17 @@ Assume the user wants a new image unless they clearly ask to change an existing
104
109
  - reference image
105
110
  - edit target
106
111
  - supporting insert/style/compositing input
107
- 7. If the edit target is only on the local filesystem, use CLI fallback for direct edits only after the user asks for or confirms fallback mode.
112
+ 7. For local edit targets, pass up to five paths through `referencedImagePaths`. For pathless conversation images, use the smallest valid `numLastImagesToInclude`.
108
113
  8. If the user asked for a photo, illustration, sprite, product image, banner, or other explicitly raster-style asset, use `codex_generate_image` rather than substituting SVG/HTML/CSS placeholders. If the request is for an icon, logo, or UI graphic that should match existing repo-native SVG/vector/code assets, prefer editing those directly instead.
109
114
  9. Augment the prompt based on specificity:
110
115
  - If the user's prompt is already specific and detailed, normalize it into a clear spec without adding creative requirements.
111
116
  - If the user's prompt is generic, add tasteful augmentation only when it materially improves output quality.
112
- 10. Use the Pi `codex_generate_image` tool by default.
113
- 11. For transparent-output requests, follow the transparent image guidance below: generate with Pi `codex_generate_image` on a flat chroma-key background, copy the selected output into the workspace or `tmp/imagegen/`, run the installed `scripts/remove_chroma_key.py` helper, and validate the alpha result before using it. If this path looks unsuitable or fails, ask before switching to CLI `gpt-image-1.5`.
117
+ 10. Use the Pi `codex_generate_image` tool by default for generation and supported existing-image edits. Ask for CLI fallback confirmation only when the request requires unsupported controls such as masks.
118
+ 11. For transparent-output requests, follow the transparent image guidance below: generate with Pi `codex_generate_image` on a flat chroma-key background, copy the selected output into the workspace or `tmp/imagegen/`, run the installed `scripts/remove_chroma_key.py` helper, and validate the alpha result before using it. If this path looks unsuitable or fails, ask before switching to the API CLI.
114
119
  12. Inspect outputs and validate: subject, style, composition, text accuracy, and invariants/avoid items.
115
120
  13. Iterate with a single targeted change, then re-check.
116
121
  14. For preview-only work, render the image inline; the underlying file may remain at the default `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*` path.
117
- 15. For project-bound work, move or copy the selected artifact into the workspace and update any consuming code or references. Never leave a project-referenced asset only at the default `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*` path.
122
+ 15. For project-bound work, copy the selected artifact into the workspace, leave the original in place, and update any consuming code or references. Never leave a project-referenced asset only at the default `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*` path.
118
123
  16. For batches or multi-asset requests, persist every requested deliverable final in the workspace unless the user explicitly asked to keep outputs preview-only. Discarded variants do not need to be kept unless requested.
119
124
  17. If the user explicitly chooses or confirms the CLI fallback, then use the fallback-only docs for model, quality, size, `input_fidelity`, masks, output format, output paths, and network setup.
120
125
  18. Always report the final saved path(s) for any workspace-bound asset(s), plus the final prompt or prompt set and whether the Pi tool or fallback CLI mode was used.
@@ -126,7 +131,7 @@ Transparent-image requests still use Pi `codex_generate_image` first. Because th
126
131
  Default sequence:
127
132
  1. Use Pi `codex_generate_image` to generate the requested subject on a perfectly flat solid chroma-key background.
128
133
  2. Choose a key color that is unlikely to appear in the subject: default `#00ff00`, use `#ff00ff` for green subjects, and avoid `#0000ff` for blue subjects.
129
- 3. After generation, move or copy the selected source image from `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*` into the workspace or `tmp/imagegen/`.
134
+ 3. After generation, copy the selected source image from `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*` into the workspace or `tmp/imagegen/` and leave the original in place.
130
135
  4. Run the bundled helper from this skill directory:
131
136
  ```bash
132
137
  python "scripts/remove_chroma_key.py" \
@@ -151,12 +156,12 @@ Do not use #00ff00 anywhere in the subject.
151
156
  No cast shadow, no contact shadow, no reflection, no watermark, and no text unless explicitly requested.
152
157
  ```
153
158
 
154
- Do not automatically use CLI `gpt-image-1.5 --background transparent --output-format png` instead of chroma keying. Ask the user first when the user asks for true/native transparency, when local removal fails validation, or when the requested image is complex: hair, fur, feathers, smoke, glass, liquids, translucent materials, reflective objects, soft shadows, realistic product grounding, or subject colors that conflict with all practical key colors.
159
+ Do not automatically use CLI `gpt-image-2 --background transparent --output-format png` instead of chroma keying. Ask the user first when the user asks for true/native transparency, when local removal fails validation, or when the requested image is complex: hair, fur, feathers, smoke, glass, liquids, translucent materials, reflective objects, soft shadows, realistic product grounding, or subject colors that conflict with all practical key colors.
155
160
 
156
161
  Use a concise confirmation like:
157
162
 
158
163
  ```text
159
- This likely needs true native transparency. The default Pi tool path uses a chroma-key background plus local removal, but true transparency requires the CLI fallback with gpt-image-1.5 because gpt-image-2 does not support background=transparent. It also requires OPENAI_API_KEY. Should I proceed with that CLI fallback?
164
+ This likely needs native transparency. The Pi tool uses chroma-key removal. The API CLI supports native transparency in preview with gpt-image-2. It requires OPENAI_API_KEY and separate API billing. Should I use that fallback?
160
165
  ```
161
166
 
162
167
  ## Prompt augmentation
@@ -276,7 +281,7 @@ Constraints: change only the background; keep the product and its edges unchange
276
281
  - If the prompt is generic, add only the extra detail that will materially help.
277
282
  - If the prompt is already detailed, normalize it instead of expanding it.
278
283
  - For CLI fallback only, see `references/cli.md` and `references/image-api.md` for model, `quality`, `input_fidelity`, masks, output format, and output-path guidance.
279
- - For transparent images, use the built-in-first chroma-key workflow unless the request is complex enough to need true CLI transparency; ask before switching to CLI `gpt-image-1.5`.
284
+ - For transparent images, use the built-in-first chroma-key workflow unless the request needs native CLI transparency; ask before switching to the API CLI.
280
285
 
281
286
  More principles shared by both modes: `references/prompting.md`.
282
287
  Copy/paste specs shared by both modes: `references/sample-prompts.md`.
@@ -288,14 +293,14 @@ Asset-type templates (website assets, game assets, wireframes, logo) are consoli
288
293
 
289
294
  The fallback CLI defaults to `gpt-image-2`.
290
295
 
291
- - Use `gpt-image-2` for new CLI/API workflows unless the request needs true model-native transparent output.
292
- - If a transparent request may need CLI fallback, ask before using `gpt-image-1.5` unless the user already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback. Explain that the built-in chroma-key path is the default, but true transparency requires `gpt-image-1.5` because `gpt-image-2` does not support `background=transparent`.
296
+ - Keep `gpt-image-2` as the CLI default. For explicit 2.5 requests, use `gpt-image-2.5-flare` for fast generation or `gpt-image-2.5-sunburst` for editing precision. Both also accept `xhigh` and `max` quality and their `2026-09-08` snapshots. Both support the flexible size constraints below; sizes above `2560x1440` are experimental. Leave 2.5 `input_fidelity` unset; support for that control is not verified.
297
+ - Native transparency is available in preview with CLI `gpt-image-2 --background transparent --output-format png` (or `webp`). Ask before switching from Pi to this separately billed API path.
293
298
  - `gpt-image-2` always uses high fidelity for image inputs; do not set `input_fidelity` with this model.
294
299
  - `gpt-image-2` supports `quality` values `low`, `medium`, `high`, and `auto`.
295
300
  - Use `quality low` for fast drafts, thumbnails, and quick iterations. Use `medium`, `high`, or `auto` for final assets, dense text, diagrams, identity-sensitive edits, or high-resolution outputs.
296
301
  - Square images are typically fastest to generate. Use `1024x1024` for fast square drafts.
297
302
  - If the user asks for 4K-style output, use `3840x2160` for landscape or `2160x3840` for portrait.
298
- - `gpt-image-2` size may be `auto` or `WIDTHxHEIGHT` if all constraints hold: max edge `<= 3840px`, both edges multiples of `16px`, long-to-short ratio `<= 3:1`, total pixels between `655,360` and `8,294,400`.
303
+ - GPT Image 2 and 2.5 API size may be `auto` or `WIDTHxHEIGHT` if all constraints hold: max edge `<= 3840px`, both edges multiples of `16px`, long-to-short ratio `<= 3:1`, total pixels between `655,360` and `8,294,400`.
299
304
 
300
305
  Popular `gpt-image-2` sizes:
301
306
  - `1024x1024` square
@@ -1,6 +1,6 @@
1
1
  # CLI reference (`scripts/image_gen.py`)
2
2
 
3
- This file is for the fallback CLI mode only. Read it when the user explicitly asks to use `scripts/image_gen.py` / CLI / API / model controls, or after the user explicitly confirms that a transparent-output request should use the `gpt-image-1.5` true-transparency fallback path.
3
+ This file is for the fallback CLI mode only. Read it when the user requests CLI/API controls or confirms a native transparency API fallback.
4
4
 
5
5
  `generate-batch` is a CLI subcommand in this fallback path. It is not a top-level mode of the skill.
6
6
  The word `batch` in a user request is not CLI opt-in by itself.
@@ -73,12 +73,14 @@ python "$IMAGE_GEN" edit \
73
73
 
74
74
  `gpt-image-2` is the default model for new CLI fallback work.
75
75
 
76
+ For explicit Images 2.5 requests, use `--model gpt-image-2.5-flare` or `--model gpt-image-2.5-sunburst`. Their `2026-09-08` snapshots are also supported. Both accept `--quality xhigh` and `--quality max`, and the flexible size constraints below, including `1536x864`. Sizes above `2560x1440` are experimental. Leave 2.5 `--input-fidelity` unset; support is not verified. These options do not apply to the Pi tool.
77
+
76
78
  - Use `--quality low` for fast drafts, thumbnails, and quick iterations.
77
79
  - Use `--quality medium`, `--quality high`, or `--quality auto` for final assets, dense text, diagrams, identity-sensitive edits, and high-resolution outputs.
78
80
  - Square images are typically fastest. Use `--size 1024x1024` for quick square drafts.
79
81
  - If the user asks for 4K-style output, use `--size 3840x2160` for landscape or `--size 2160x3840` for portrait.
80
82
  - Do not pass `--input-fidelity` with `gpt-image-2`; this model always uses high fidelity for image inputs.
81
- - Do not use `--background transparent` with `gpt-image-2`; the default transparent-image workflow uses Pi `codex_generate_image` on a flat chroma-key background plus local removal. Use `gpt-image-1.5` only after the user explicitly confirms the true-transparent CLI fallback, unless they already requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback.
83
+ - `gpt-image-2` supports `--background transparent` in preview with PNG or WebP. Confirm the separately billed API fallback before switching from Pi's chroma-key path.
82
84
 
83
85
  Popular `gpt-image-2` sizes:
84
86
  - `1024x1024`
@@ -140,7 +142,7 @@ python "$IMAGE_GEN" generate \
140
142
  --out output/imagegen/product-cutout.png
141
143
  ```
142
144
 
143
- When using this path, explain briefly that Pi `codex_generate_image` plus chroma-key removal is the default transparent-image path, but this request needs true model-native transparency. `gpt-image-2` does not support `background=transparent`, so `gpt-image-1.5` is required for this confirmed fallback.
145
+ The older-model example above remains available when explicitly requested. Prefer `--model gpt-image-2` for native transparency preview. Explain that this API fallback requires an API key and separate billing; Pi still uses chroma-key removal.
144
146
 
145
147
  ## Quality, input fidelity, and masks (CLI fallback only)
146
148
  These are explicit CLI controls. They are not Pi `codex_generate_image` tool arguments.
@@ -229,8 +231,8 @@ Notes:
229
231
  - For many requested deliverable assets, provide one prompt/job per distinct asset and use semantic filenames when possible.
230
232
 
231
233
  ## CLI notes
232
- - Supported sizes depend on the model. `gpt-image-2` supports flexible constrained sizes; older GPT Image models support `1024x1024`, `1536x1024`, `1024x1536`, or `auto`.
233
- - True transparent CLI outputs require `output_format` to be `png` or `webp` and are not supported by `gpt-image-2`.
234
+ - Supported sizes depend on the model. GPT Image 2 and 2.5 (including the documented snapshots) support flexible constrained sizes; older GPT Image models support `1024x1024`, `1536x1024`, `1024x1536`, or `auto`.
235
+ - Native transparent CLI outputs require `output_format` to be `png` or `webp`. GPT Image 2 support is in preview.
234
236
  - `--prompt-file`, `--output-compression`, `--moderation`, `--max-attempts`, `--fail-fast`, `--force`, and `--no-augment` are supported.
235
237
  - This CLI is intended for GPT Image models. Do not assume older non-GPT image-model behavior applies here.
236
238
 
@@ -1,31 +1,39 @@
1
1
  # Image API quick reference
2
2
 
3
- This file is for the fallback CLI mode only. Use it when the user explicitly asks to use `scripts/image_gen.py` / CLI / API / model controls, or after the user explicitly confirms that a transparent-output request should use the `gpt-image-1.5` true-transparency fallback path.
3
+ This file is for the fallback CLI mode only. Use it when the user explicitly requests CLI/API controls or confirms a native transparency API fallback.
4
4
 
5
5
  These parameters describe the Image API and bundled CLI fallback surface. Do not assume they are normal arguments on the Pi `codex_generate_image` tool.
6
6
 
7
7
  ## Scope
8
- - This fallback CLI is intended for GPT Image models (`gpt-image-2`, `gpt-image-1.5`, `gpt-image-1`, and `gpt-image-1-mini`).
8
+ - This fallback CLI supports GPT Image models, including `gpt-image-2.5-flare` and `gpt-image-2.5-sunburst`.
9
9
  - The Pi `codex_generate_image` tool and the fallback CLI do not expose the same controls.
10
10
 
11
11
  ## Model summary
12
12
 
13
13
  | Model | Quality | Input fidelity | Resolutions | Recommended use |
14
14
  | --- | --- | --- | --- | --- |
15
+ | `gpt-image-2.5-flare` | `low`, `medium`, `high`, `xhigh`, `max`, `auto` | Leave unset; not verified | `auto` or flexible sizes below | Optional fast generation |
16
+ | `gpt-image-2.5-sunburst` | `low`, `medium`, `high`, `xhigh`, `max`, `auto` | Leave unset; not verified | `auto` or flexible sizes below | Optional precision editing |
15
17
  | `gpt-image-2` | `low`, `medium`, `high`, `auto` | Always high fidelity for image inputs; do not set `input_fidelity` | `auto` or flexible sizes that satisfy the constraints below | Default for new CLI/API workflows: high-quality generation and editing, text-heavy images, photorealism, compositing, identity-sensitive edits, and workflows where fewer retries matter |
16
18
  | `gpt-image-1.5` | `low`, `medium`, `high`, `auto` | `low`, `high` | `1024x1024`, `1024x1536`, `1536x1024`, `auto` | True transparent-background fallback and backward-compatible workflows |
17
19
  | `gpt-image-1` | `low`, `medium`, `high`, `auto` | `low`, `high` | `1024x1024`, `1024x1536`, `1536x1024`, `auto` | Legacy compatibility |
18
20
  | `gpt-image-1-mini` | `low`, `medium`, `high`, `auto` | `low`, `high` | `1024x1024`, `1024x1536`, `1536x1024`, `auto` | Cost-sensitive draft batches and lower-stakes previews |
19
21
 
20
- ## gpt-image-2 sizes
22
+ The CLI also recognizes the `2026-09-08` snapshots of both 2.5 models for extended quality settings. Defaults remain unchanged.
21
23
 
22
- `gpt-image-2` accepts `auto` or any `WIDTHxHEIGHT` size that satisfies all constraints:
24
+ Sources: [Flare](https://developers.openai.com/api/docs/models/gpt-image-2.5-flare), [Sunburst](https://developers.openai.com/api/docs/models/gpt-image-2.5-sunburst), [image generation guide](https://developers.openai.com/api/docs/guides/image-generation.md), and [image tool options](https://developers.openai.com/api/docs/guides/tools-image-generation).
25
+
26
+ ## Flexible sizes: GPT Image 2 and 2.5
27
+
28
+ GPT Image 2 and both 2.5 models accept `auto` or any `WIDTHxHEIGHT` size that satisfies all constraints. The CLI also recognizes their documented dated snapshots.
23
29
 
24
30
  - Maximum edge length must be less than or equal to `3840px`.
25
31
  - Both edges must be multiples of `16px`.
26
32
  - Long edge to short edge ratio must not exceed `3:1`.
27
33
  - Total pixels must be at least `655,360` and no more than `8,294,400`.
28
34
 
35
+ For 2.5, resolutions above `2560x1440` are experimental.
36
+
29
37
  Popular sizes:
30
38
 
31
39
  | Label | Size | Notes |
@@ -49,8 +57,8 @@ Square images are typically fastest to generate. For 4K-style output, use `3840x
49
57
  - `prompt`: text prompt
50
58
  - `model`: image model
51
59
  - `n`: number of images (1-10)
52
- - `size`: `auto` by default for `gpt-image-2`; flexible `WIDTHxHEIGHT` sizes are allowed only for `gpt-image-2`; older GPT Image models use `1024x1024`, `1536x1024`, `1024x1536`, or `auto`
53
- - `quality`: `low`, `medium`, `high`, or `auto`
60
+ - `size`: `auto` by default; GPT Image 2 and 2.5 accept flexible `WIDTHxHEIGHT` sizes under the constraints above; older models use `1024x1024`, `1536x1024`, `1024x1536`, or `auto`
61
+ - `quality`: `low`, `medium`, `high`, or `auto`; the documented 2.5 models also accept `xhigh` and `max`
54
62
  - `background`: output transparency behavior (`transparent`, `opaque`, or `auto`) for generated output; this is not the same thing as the prompt's visual scene/backdrop
55
63
  - `output_format`: `png` (default), `jpeg`, `webp`
56
64
  - `output_compression`: 0-100 (jpeg/webp only)
@@ -68,9 +76,11 @@ Model-specific note for `input_fidelity`:
68
76
 
69
77
  ## Transparent backgrounds
70
78
 
71
- `gpt-image-2` does not currently support the Image API `background=transparent` parameter. The skill's default transparent-image path is Pi `codex_generate_image` with a flat chroma-key background, followed by local alpha extraction with `python "scripts/remove_chroma_key.py"`.
79
+ `gpt-image-2` supports `background=transparent` in preview with PNG or WebP. The Pi tool has no background parameter, so its default path remains chroma-key generation followed by local alpha extraction.
80
+
81
+ Both 2.5 API models also document native transparency with PNG or WebP. This public API capability does not establish that subscription parameters are honored.
72
82
 
73
- Use CLI `gpt-image-1.5` with `background=transparent` and a transparent-capable output format such as `png` or `webp` only after the user explicitly confirms that fallback, unless they already requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback. If the user asks for true/native transparency, the subject is too complex for clean chroma-key removal, or local background removal fails validation, explain the tradeoff and ask before switching.
83
+ Use CLI `gpt-image-2 --background transparent --output-format png` (or `webp`) only after the user chooses the API path. Explain that it requires an API key and separate billing. Do not silently switch to an older model if preview transparency fails.
74
84
 
75
85
  ## Output
76
86
  - `data[]` list with `b64_json` per image
@@ -79,7 +79,7 @@ Do not add:
79
79
  - Ask for crisp edges, generous padding, and no use of the key color inside the subject.
80
80
  - After generation, remove the background locally with `python "scripts/remove_chroma_key.py" --input <source> --out <final.png> --auto-key border --soft-matte --transparent-threshold 12 --opaque-threshold 220 --despill` and validate the alpha result before shipping it.
81
81
  - Use soft matte and despill for antialiased edges; hard tolerance-only removal is mainly for flat pixel-art or exact-color fixtures.
82
- - Use CLI `gpt-image-1.5 --background transparent --output-format png` only after the user explicitly confirms the fallback, or when the user already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback. Ask first for true/native transparency requests, failed chroma-key validation, or complex transparent subjects such as hair, fur, glass, smoke, liquids, translucent materials, reflective objects, or soft shadows.
82
+ - Use CLI `gpt-image-2 --background transparent --output-format png` (preview) only after the user chooses the separately billed API fallback. Offer it for native transparency requests, failed chroma-key validation, or complex subjects such as hair, glass, smoke, or soft shadows.
83
83
 
84
84
  ## Fallback-only execution controls
85
85
  - `quality`, `input_fidelity`, explicit masks, output format, and output paths are fallback-only execution controls.
@@ -87,7 +87,7 @@ Do not add:
87
87
  - If the user explicitly chooses CLI fallback, see `references/cli.md` and `references/image-api.md` for those controls.
88
88
  - In CLI fallback mode, `gpt-image-2` is the default. It supports `quality=low|medium|high|auto`; use `low` for fast drafts and thumbnails, and move to `medium`, `high`, or `auto` for final assets.
89
89
  - `gpt-image-2` always uses high fidelity for image inputs, so do not set `input_fidelity` with that model.
90
- - If a transparent request needs true CLI transparency, ask before using `gpt-image-1.5` unless the user already explicitly chose it. Explain that Pi-tool chroma-key removal is the default path, but `gpt-image-2` does not support `background=transparent`.
90
+ - For native transparency, offer the GPT Image 2 API preview with PNG or WebP. Ask before switching from Pi's chroma-key path to API billing.
91
91
  - If the user asks for 4K-style output with `gpt-image-2`, use `3840x2160` for landscape or `2160x3840` for portrait.
92
92
 
93
93
  ## Use-case tips
@@ -19,7 +19,7 @@ CLI model notes:
19
19
  - `gpt-image-2` is the fallback CLI default for new workflows.
20
20
  - `gpt-image-2` supports `quality` values `low`, `medium`, `high`, and `auto`.
21
21
  - For 4K-style `gpt-image-2` output, use `3840x2160` or `2160x3840`.
22
- - If transparent output needs true CLI fallback, ask before using `gpt-image-1.5` unless the user already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback. Explain that Pi-tool chroma-key removal is the default path, but `gpt-image-2` does not support `background=transparent`.
22
+ - For native transparency, offer CLI `gpt-image-2 --background transparent --output-format png` (preview). Ask before switching from Pi's chroma-key path to separate API billing.
23
23
  - Do not set `input_fidelity` with `gpt-image-2`; image inputs already use high fidelity.
24
24
 
25
25
  For prompting principles (structure, specificity, invariants, iteration), see `references/prompting.md`.
@@ -395,7 +395,7 @@ Scene/backdrop: perfectly flat solid #00ff00 chroma-key background for local bac
395
395
  Constraints: background must be one uniform color with no shadows, gradients, texture, reflections, floor plane, or lighting variation; crisp silhouette; generous padding; no halos or fringing; preserve label text exactly; no restyling; do not use #00ff00 anywhere in the subject
396
396
  ```
397
397
 
398
- Post-process note: after Pi tool generation, run `python "scripts/remove_chroma_key.py" --input <source> --out <final.png> --auto-key border --soft-matte --transparent-threshold 12 --opaque-threshold 220 --despill`. Ask before using CLI `gpt-image-1.5 --background transparent --output-format png` for true/native transparency, failed chroma-key validation, or complex subjects such as hair, fur, glass, smoke, liquids, translucent materials, reflections, or soft shadows, unless the user already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback.
398
+ Post-process note: after Pi tool generation, run `python "scripts/remove_chroma_key.py" --input <source> --out <final.png> --auto-key border --soft-matte --transparent-threshold 12 --opaque-threshold 220 --despill`. For native transparency, failed validation, or complex subjects, offer CLI `gpt-image-2 --background transparent --output-format png` (preview). Ask before switching to separate API billing.
399
399
 
400
400
  ### style-transfer
401
401
  ```
@@ -1,8 +1,8 @@
1
1
  #!/usr/bin/env python3
2
2
  """Fallback CLI for explicit image generation or editing with GPT Image models.
3
3
 
4
- Used only when the user explicitly opts into CLI fallback mode, or when explicit
5
- transparent output requires the `gpt-image-1.5` fallback path.
4
+ Used only when the user explicitly opts into CLI fallback mode, including
5
+ confirmed native transparency requests with separate API billing.
6
6
 
7
7
  Defaults to gpt-image-2 and a structured prompt augmentation workflow.
8
8
  """
@@ -118,7 +118,7 @@ def _parse_size(size: str) -> Optional[Tuple[int, int]]:
118
118
  return int(match.group(1)), int(match.group(2))
119
119
 
120
120
 
121
- def _validate_gpt_image_2_size(size: str) -> None:
121
+ def _validate_flexible_size(size: str, model: str) -> None:
122
122
  if size == "auto":
123
123
  return
124
124
 
@@ -132,20 +132,20 @@ def _validate_gpt_image_2_size(size: str) -> None:
132
132
  total_pixels = width * height
133
133
 
134
134
  if max_edge > GPT_IMAGE_2_MAX_EDGE:
135
- _die("gpt-image-2 size maximum edge length must be less than or equal to 3840px.")
135
+ _die(f"{model} size maximum edge length must be less than or equal to 3840px.")
136
136
  if width % 16 != 0 or height % 16 != 0:
137
- _die("gpt-image-2 size width and height must be multiples of 16px.")
137
+ _die(f"{model} size width and height must be multiples of 16px.")
138
138
  if max_edge / min_edge > GPT_IMAGE_2_MAX_RATIO:
139
- _die("gpt-image-2 size long edge to short edge ratio must not exceed 3:1.")
139
+ _die(f"{model} size long edge to short edge ratio must not exceed 3:1.")
140
140
  if total_pixels < GPT_IMAGE_2_MIN_PIXELS or total_pixels > GPT_IMAGE_2_MAX_PIXELS:
141
141
  _die(
142
- "gpt-image-2 size total pixels must be at least 655,360 and no more than 8,294,400."
142
+ f"{model} size total pixels must be at least 655,360 and no more than 8,294,400."
143
143
  )
144
144
 
145
145
 
146
146
  def _validate_size(size: str, model: str) -> None:
147
- if model == GPT_IMAGE_2_MODEL:
148
- _validate_gpt_image_2_size(size)
147
+ if _is_gpt_image_2(model) or _is_gpt_image_2_5(model):
148
+ _validate_flexible_size(size, model)
149
149
  return
150
150
 
151
151
  if size not in ALLOWED_LEGACY_SIZES:
@@ -154,9 +154,23 @@ def _validate_size(size: str, model: str) -> None:
154
154
  )
155
155
 
156
156
 
157
- def _validate_quality(quality: str) -> None:
158
- if quality not in ALLOWED_QUALITIES:
159
- _die("quality must be one of low, medium, high, or auto.")
157
+ def _is_gpt_image_2(model: str) -> bool:
158
+ return model in {GPT_IMAGE_2_MODEL, "gpt-image-2-2026-04-21"}
159
+
160
+
161
+ def _is_gpt_image_2_5(model: str) -> bool:
162
+ return model in {
163
+ "gpt-image-2.5-flare",
164
+ "gpt-image-2.5-flare-2026-09-08",
165
+ "gpt-image-2.5-sunburst",
166
+ "gpt-image-2.5-sunburst-2026-09-08",
167
+ }
168
+
169
+
170
+ def _validate_quality(quality: str, model: str) -> None:
171
+ allowed = ALLOWED_QUALITIES | {"xhigh", "max"} if _is_gpt_image_2_5(model) else ALLOWED_QUALITIES
172
+ if quality not in allowed:
173
+ _die(f"quality for {model} must be one of {', '.join(sorted(allowed))}.")
160
174
 
161
175
 
162
176
  def _validate_background(background: Optional[str]) -> None:
@@ -187,13 +201,8 @@ def _validate_model_specific_options(
187
201
  background: Optional[str],
188
202
  input_fidelity: Optional[str] = None,
189
203
  ) -> None:
190
- if model != GPT_IMAGE_2_MODEL:
204
+ if not _is_gpt_image_2(model):
191
205
  return
192
- if background == "transparent":
193
- _die(
194
- "transparent backgrounds are not supported in gpt-image-2, the latest model. "
195
- "Use --model gpt-image-1.5 --background transparent --output-format png instead."
196
- )
197
206
  if input_fidelity is not None:
198
207
  _die(
199
208
  "input_fidelity is not supported in gpt-image-2 because image inputs always use high fidelity for this model."
@@ -210,7 +219,7 @@ def _validate_generate_payload(payload: Dict[str, Any]) -> None:
210
219
  quality = str(payload.get("quality", DEFAULT_QUALITY))
211
220
  background = payload.get("background")
212
221
  _validate_size(size, model)
213
- _validate_quality(quality)
222
+ _validate_quality(quality, model)
214
223
  _validate_background(background)
215
224
  _validate_model_specific_options(model=model, background=background)
216
225
  oc = payload.get("output_compression")
@@ -978,7 +987,7 @@ def main() -> int:
978
987
 
979
988
  _validate_model(args.model)
980
989
  _validate_size(args.size, args.model)
981
- _validate_quality(args.quality)
990
+ _validate_quality(args.quality, args.model)
982
991
  _validate_background(args.background)
983
992
  _validate_model_specific_options(
984
993
  model=args.model,