pi-codex-image-gen 0.1.12 → 0.1.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,17 +3,19 @@
3
3
  *
4
4
  * Registers `codex_generate_image`, a tool that uses Pi's existing
5
5
  * openai-codex ChatGPT/Codex auth to call the Codex Responses backend with the
6
- * native `image_generation` tool. The backend maps that tool to gpt-image-2.
6
+ * native `image_generation` tool. The backend selects the image model.
7
7
  */
8
8
 
9
9
  import { readFileSync } from "node:fs";
10
- import { mkdir, readFile, writeFile } from "node:fs/promises";
10
+ import { constants } from "node:fs";
11
+ import { mkdir, open, writeFile } from "node:fs/promises";
11
12
  import { homedir } from "node:os";
12
13
  import { isAbsolute, join, resolve } from "node:path";
13
14
  import { StringEnum } from "@earendil-works/pi-ai";
14
- import { type ExtensionAPI, getAgentDir, withFileMutationQueue } from "@earendil-works/pi-coding-agent";
15
+ import { CONFIG_DIR_NAME, type ExtensionAPI, getAgentDir, withFileMutationQueue } from "@earendil-works/pi-coding-agent";
15
16
  import { type Static, Type } from "typebox";
16
17
  import { reportInstallTelemetry } from "../src/install-telemetry.js";
18
+ import { abortable, httpFailure, MAX_IMAGE_BYTES, parseCodexSse, withRequestDeadline, type ParsedCodexResponse } from "../src/codex-response.js";
17
19
 
18
20
  const PACKAGE_NAME = "pi-codex-image-gen";
19
21
  const PROVIDER = "openai-codex";
@@ -26,6 +28,9 @@ const MAX_RETRIES = 3;
26
28
  const BASE_DELAY_MS = 1000;
27
29
  const MAX_RETRY_DELAY_MS = 30_000;
28
30
  const MAX_EDIT_IMAGES = 5;
31
+ const MAX_INPUT_IMAGE_BYTES = 20 * 1024 * 1024;
32
+ const MAX_TOTAL_INPUT_BYTES = 50 * 1024 * 1024;
33
+ const MAX_PROMPT_CHARS = 32_000;
29
34
 
30
35
  const SAVE_MODES = ["none", "project", "global", "custom"] as const;
31
36
  type SaveMode = (typeof SAVE_MODES)[number];
@@ -35,11 +40,6 @@ type OutputFormat = (typeof OUTPUT_FORMATS)[number];
35
40
 
36
41
  // --- #1: Retry helpers with exponential backoff + jitter ---
37
42
 
38
- function isRetryableStatus(status: number, errorText: string): boolean {
39
- if ([429, 500, 502, 503, 504].includes(status)) return true;
40
- return /rate.?limit|overloaded|service.?unavailable|upstream.?connect|connection.?refused/i.test(errorText);
41
- }
42
-
43
43
  export function parseRetryAfter(value: string | null, nowMs = Date.now()): number | undefined {
44
44
  if (!value) return undefined;
45
45
  const trimmed = value.trim();
@@ -89,9 +89,9 @@ export function abortableDelay(milliseconds: number, signal?: AbortSignal): Prom
89
89
  // --- Tool parameter schema ---
90
90
 
91
91
  const TOOL_PARAMS = Type.Object({
92
- prompt: Type.String({ description: "The image prompt. Be specific about subject, composition, style, text, and constraints." }),
92
+ prompt: Type.String({ minLength: 1, maxLength: MAX_PROMPT_CHARS, description: "The image prompt. Be specific about subject, composition, style, text, and constraints." }),
93
93
  model: Type.Optional(
94
- Type.String({ description: `Codex model that should invoke image generation. Defaults to ${DEFAULT_MODEL}.` }),
94
+ Type.String({ minLength: 1, maxLength: 200, description: `Codex routing model, not an image model selector. Defaults to ${DEFAULT_MODEL}.` }),
95
95
  ),
96
96
  outputFormat: Type.Optional(StringEnum(OUTPUT_FORMATS)),
97
97
  save: Type.Optional(StringEnum(SAVE_MODES)),
@@ -130,44 +130,11 @@ interface SaveConfig {
130
130
  outputDir?: string;
131
131
  }
132
132
 
133
- interface GeneratedImage {
134
- id: string;
135
- status: string;
136
- result: string;
137
- revisedPrompt?: string;
138
- }
139
-
140
- interface ParsedCodexResponse {
141
- image?: GeneratedImage;
142
- text: string[];
143
- responseId?: string;
144
- usage?: unknown;
145
- }
146
-
147
133
  interface InputImage {
148
134
  data: string;
149
135
  mimeType: string;
150
136
  }
151
137
 
152
- // --- #11: Typed SSE event discriminated union ---
153
-
154
- type CodexSseEvent =
155
- | { type: "error"; message?: string; code?: string }
156
- | { type: "response.failed"; response?: { error?: { message?: string } } }
157
- | { type: "response.created"; response?: { id?: string } }
158
- | { type: "response.output_text.delta"; delta?: string }
159
- | {
160
- type: "response.output_item.done";
161
- item?: {
162
- type?: string;
163
- id?: string | number;
164
- status?: string;
165
- result?: string;
166
- revised_prompt?: string;
167
- };
168
- }
169
- | { type: "response.completed"; response?: { id?: string; usage?: unknown } };
170
-
171
138
  // --- JWT helpers ---
172
139
 
173
140
  function decodeJwtPayload(token: string): Record<string, unknown> {
@@ -177,8 +144,8 @@ function decodeJwtPayload(token: string): Record<string, unknown> {
177
144
  }
178
145
  try {
179
146
  return JSON.parse(Buffer.from(parts[1], "base64url").toString("utf8")) as Record<string, unknown>;
180
- } catch (error) {
181
- throw new Error(`Failed to decode OpenAI Codex auth token: ${error instanceof Error ? error.message : String(error)}`);
147
+ } catch {
148
+ throw new Error("Failed to decode OpenAI Codex auth token. Run /login for openai-codex again.");
182
149
  }
183
150
  }
184
151
 
@@ -208,7 +175,7 @@ function readConfigFile(path: string): ExtensionConfig {
208
175
  export function loadConfig(cwd: string, projectTrusted: boolean, agentDir = getAgentDir()): ExtensionConfig {
209
176
  const globalConfig = readConfigFile(join(agentDir, "extensions", "codex-image-gen.json"));
210
177
  if (!projectTrusted) return globalConfig;
211
- const projectConfig = readConfigFile(join(cwd, ".pi", "extensions", "codex-image-gen.json"));
178
+ const projectConfig = readConfigFile(join(cwd, CONFIG_DIR_NAME, "extensions", "codex-image-gen.json"));
212
179
  return { ...globalConfig, ...projectConfig };
213
180
  }
214
181
 
@@ -221,7 +188,7 @@ export function resolveUnderCwd(cwd: string, path: string, homeDir = homedir()):
221
188
  }
222
189
 
223
190
  function sanitizePathPart(value: string, fallback: string): string {
224
- const sanitized = value
191
+ const sanitized = value.slice(0, 128)
225
192
  .split("")
226
193
  .map((ch) => (/[a-zA-Z0-9_-]/.test(ch) ? ch : "_"))
227
194
  .join("")
@@ -239,7 +206,7 @@ function resolveSaveConfig(params: ToolParams, cwd: string, sessionId: string, c
239
206
  throw new Error(`Invalid save mode: ${mode}. Expected one of ${SAVE_MODES.join(", ")}.`);
240
207
  }
241
208
  if (mode === "project") {
242
- return { mode, outputDir: join(cwd, ".pi", "generated-images", safeSessionId) };
209
+ return { mode, outputDir: join(cwd, CONFIG_DIR_NAME, "generated-images", safeSessionId) };
243
210
  }
244
211
  if (mode === "global") {
245
212
  return { mode, outputDir: join(getAgentDir(), "generated-images", safeSessionId) };
@@ -270,12 +237,13 @@ function imagePath(outputFormat: OutputFormat, outputDir: string, imageCallId: s
270
237
  }
271
238
 
272
239
  export function decodeImageData(base64Data: string, outputFormat: OutputFormat): Buffer {
240
+ if (base64Data.length > Math.ceil(MAX_IMAGE_BYTES / 3) * 4) throw new Error("Codex image exceeded the 32 MiB size limit.");
273
241
  const value = base64Data.trim();
274
- if (!value || value.length % 4 !== 0 || !/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/.test(value)) {
242
+ if (!value || value.length % 4 !== 0 || /[^A-Za-z0-9+/=]/.test(value)) {
275
243
  throw new Error("Codex returned invalid base64 image data.");
276
244
  }
277
245
  const bytes = Buffer.from(value, "base64");
278
- if (bytes.length === 0 || bytes.toString("base64") !== value) {
246
+ if (bytes.length === 0 || bytes.length > MAX_IMAGE_BYTES || bytes.toString("base64") !== value) {
279
247
  throw new Error("Codex returned invalid base64 image data.");
280
248
  }
281
249
  const validSignature =
@@ -294,8 +262,8 @@ async function saveImage(
294
262
  ): Promise<string> {
295
263
  const filePath = imagePath(outputFormat, outputDir, imageCallId);
296
264
  await withFileMutationQueue(filePath, async () => {
297
- await mkdir(outputDir, { recursive: true });
298
- await writeFile(filePath, bytes);
265
+ await mkdir(outputDir, { recursive: true, mode: 0o700 });
266
+ await writeFile(filePath, bytes, { flag: "wx", mode: 0o600 });
299
267
  });
300
268
  return filePath;
301
269
  }
@@ -322,6 +290,29 @@ function mimeFromBytes(bytes: Buffer, path: string): string {
322
290
  throw new Error(`Referenced image is unavailable or unsupported: ${path}`);
323
291
  }
324
292
 
293
+ async function readInputImage(path: string): Promise<Buffer> {
294
+ // O_NONBLOCK prevents named pipes from blocking before the regular-file check.
295
+ const file = await open(path, constants.O_RDONLY | (constants.O_NONBLOCK ?? 0));
296
+ try {
297
+ const info = await file.stat();
298
+ if (!info.isFile()) throw new Error("Referenced images must be regular files.");
299
+ if (info.size > MAX_INPUT_IMAGE_BYTES) throw new Error("Referenced image exceeds 20 MiB.");
300
+ const blocks: Buffer[] = [];
301
+ let total = 0;
302
+ while (true) {
303
+ const block = Buffer.alloc(Math.min(64 * 1024, MAX_INPUT_IMAGE_BYTES + 1 - total));
304
+ const { bytesRead } = await file.read(block, 0, block.length, null);
305
+ if (!bytesRead) break;
306
+ total += bytesRead;
307
+ if (total > MAX_INPUT_IMAGE_BYTES) throw new Error("Referenced image exceeds 20 MiB.");
308
+ blocks.push(block.subarray(0, bytesRead));
309
+ }
310
+ return Buffer.concat(blocks, total);
311
+ } finally {
312
+ await file.close();
313
+ }
314
+ }
315
+
325
316
  export async function resolveInputImages(
326
317
  params: ToolParams,
327
318
  cwd: string,
@@ -333,19 +324,22 @@ export async function resolveInputImages(
333
324
  }
334
325
  if (paths.length > MAX_EDIT_IMAGES) throw new Error(`referencedImagePaths accepts at most ${MAX_EDIT_IMAGES} paths.`);
335
326
  if (paths.length > 0) {
336
- return Promise.all(
337
- paths.map(async (path) => {
338
- const normalized = path.startsWith("@") ? path.slice(1) : path;
339
- const absolutePath = resolveUnderCwd(cwd, normalized);
340
- let bytes: Buffer;
341
- try {
342
- bytes = await readFile(absolutePath);
343
- } catch (error) {
344
- throw new Error(`Unable to read referenced image at ${absolutePath}: ${error instanceof Error ? error.message : String(error)}`);
345
- }
346
- return { data: bytes.toString("base64"), mimeType: mimeFromBytes(bytes, absolutePath) };
347
- }),
348
- );
327
+ const images: InputImage[] = [];
328
+ let total = 0;
329
+ for (const path of paths) {
330
+ const normalized = path.startsWith("@") ? path.slice(1) : path;
331
+ const absolutePath = resolveUnderCwd(cwd, normalized);
332
+ let bytes: Buffer;
333
+ try {
334
+ bytes = await readInputImage(absolutePath);
335
+ } catch (error) {
336
+ throw new Error(`Unable to read referenced image at ${absolutePath}: ${error instanceof Error ? error.message : String(error)}`);
337
+ }
338
+ total += bytes.length;
339
+ if (total > MAX_TOTAL_INPUT_BYTES) throw new Error("Referenced images exceed 50 MiB in total.");
340
+ images.push({ data: bytes.toString("base64"), mimeType: mimeFromBytes(bytes, absolutePath) });
341
+ }
342
+ return images;
349
343
  }
350
344
  if (params.numLastImagesToInclude !== undefined) {
351
345
  const count = params.numLastImagesToInclude;
@@ -356,6 +350,16 @@ export async function resolveInputImages(
356
350
  if (images.length !== count) {
357
351
  throw new Error(`Requested the last ${count} conversation images, but only ${images.length} were available.`);
358
352
  }
353
+ let total = 0;
354
+ for (const image of images) {
355
+ if (image.data.length > Math.ceil(MAX_INPUT_IMAGE_BYTES / 3) * 4) throw new Error("Conversation image exceeds 20 MiB.");
356
+ const format = OUTPUT_FORMATS.find(format => mimeForFormat(format) === image.mimeType);
357
+ if (!format) throw new Error("Conversation image has an unsupported format.");
358
+ const bytes = decodeImageData(image.data, format);
359
+ if (bytes.length > MAX_INPUT_IMAGE_BYTES) throw new Error("Conversation image exceeds 20 MiB.");
360
+ total += bytes.length;
361
+ if (total > MAX_TOTAL_INPUT_BYTES) throw new Error("Conversation images exceed 50 MiB in total.");
362
+ }
359
363
  return images;
360
364
  }
361
365
  return [];
@@ -399,108 +403,6 @@ export function buildRequestBody(
399
403
  };
400
404
  }
401
405
 
402
- // --- SSE parsing ---
403
-
404
- function parseSseDataLines(chunk: string): string | undefined {
405
- const data = chunk
406
- .split("\n")
407
- .filter((line) => line.startsWith("data:"))
408
- .map((line) => line.slice(5).trim())
409
- .join("\n")
410
- .trim();
411
- return data && data !== "[DONE]" ? data : undefined;
412
- }
413
-
414
- async function parseCodexSse(response: Response, signal?: AbortSignal): Promise<ParsedCodexResponse> {
415
- if (!response.body) throw new Error("Codex response did not include a stream body.");
416
- const reader = response.body.getReader();
417
- const decoder = new TextDecoder();
418
- let buffer = "";
419
- const parsed: ParsedCodexResponse = { text: [] };
420
-
421
- try {
422
- while (true) {
423
- if (signal?.aborted) throw new Error("Image generation was aborted.");
424
- const { done, value } = await reader.read();
425
- if (done) break;
426
- buffer += decoder.decode(value, { stream: true });
427
-
428
- let separator = buffer.indexOf("\n\n");
429
- while (separator !== -1) {
430
- const chunk = buffer.slice(0, separator);
431
- buffer = buffer.slice(separator + 2);
432
- const data = parseSseDataLines(chunk);
433
- if (data) handleCodexEvent(JSON.parse(data) as CodexSseEvent, parsed);
434
- separator = buffer.indexOf("\n\n");
435
- }
436
- }
437
- const remaining = parseSseDataLines(buffer);
438
- if (remaining) handleCodexEvent(JSON.parse(remaining) as CodexSseEvent, parsed);
439
- } finally {
440
- try {
441
- await reader.cancel();
442
- } catch {
443
- // ignored: stream may already be closed
444
- }
445
- reader.releaseLock();
446
- }
447
-
448
- return parsed;
449
- }
450
-
451
- // --- #11: Typed event handler via discriminated union ---
452
-
453
- function handleCodexEvent(event: CodexSseEvent, parsed: ParsedCodexResponse): void {
454
- if (!event || typeof event !== "object") return;
455
-
456
- switch (event.type) {
457
- case "error": {
458
- const e = event as Extract<CodexSseEvent, { type: "error" }>;
459
- throw new Error(`Codex error: ${e.message || e.code || JSON.stringify(event)}`);
460
- }
461
- case "response.failed": {
462
- const e = event as Extract<CodexSseEvent, { type: "response.failed" }>;
463
- throw new Error(e.response?.error?.message || "Codex response failed.");
464
- }
465
- case "response.created": {
466
- const e = event as Extract<CodexSseEvent, { type: "response.created" }>;
467
- if (typeof e.response?.id === "string") {
468
- parsed.responseId = e.response.id;
469
- }
470
- break;
471
- }
472
- case "response.output_text.delta": {
473
- const e = event as Extract<CodexSseEvent, { type: "response.output_text.delta" }>;
474
- if (typeof e.delta === "string") {
475
- parsed.text.push(e.delta);
476
- }
477
- break;
478
- }
479
- case "response.output_item.done": {
480
- const e = event as Extract<CodexSseEvent, { type: "response.output_item.done" }>;
481
- const item = e.item;
482
- if (item?.type === "image_generation_call") {
483
- if (typeof item.result !== "string" || item.result.length === 0) {
484
- throw new Error("Codex image_generation_call did not contain image data.");
485
- }
486
- parsed.image = {
487
- id: String(item.id || "image_generation"),
488
- status: String(item.status || "completed"),
489
- result: item.result,
490
- revisedPrompt: typeof item.revised_prompt === "string" ? item.revised_prompt : undefined,
491
- };
492
- }
493
- break;
494
- }
495
- case "response.completed": {
496
- const e = event as Extract<CodexSseEvent, { type: "response.completed" }>;
497
- if (typeof e.response?.id === "string") parsed.responseId = e.response.id;
498
- if (e.response?.usage) parsed.usage = e.response.usage;
499
- break;
500
- }
501
- }
502
- }
503
-
504
406
  // --- #1: requestImage with retry + backoff + jitter ---
505
407
 
506
408
  async function requestImage(
@@ -512,41 +414,46 @@ async function requestImage(
512
414
  sessionId: string,
513
415
  inputImages: InputImage[],
514
416
  signal?: AbortSignal,
417
+ onProgress?: (stage: string) => void,
515
418
  ): Promise<ParsedCodexResponse> {
516
419
  const body = JSON.stringify(buildRequestBody(params, model, outputFormat, sessionId, inputImages));
517
420
  const headers: Record<string, string> = {
518
421
  Authorization: `Bearer ${token}`,
519
422
  "chatgpt-account-id": accountId,
520
423
  originator: "pi",
424
+ "User-Agent": PACKAGE_NAME,
521
425
  "OpenAI-Beta": OPENAI_BETA_HEADER,
522
426
  accept: "text/event-stream",
523
427
  "content-type": "application/json",
524
428
  };
525
429
 
526
- for (let attempt = 1; attempt <= MAX_RETRIES + 1; attempt++) {
527
- if (signal?.aborted) throw new Error("Image generation was aborted.");
528
-
529
- const response = await fetch(CODEX_RESPONSES_URL, {
530
- method: "POST",
531
- headers,
532
- body,
533
- signal,
534
- });
535
-
536
- if (!response.ok) {
537
- const errorText = await response.text();
538
- if (attempt <= MAX_RETRIES && isRetryableStatus(response.status, errorText)) {
539
- const delay = retryDelayMs(attempt, response.headers.get("retry-after"));
540
- await abortableDelay(delay, signal);
541
- continue;
430
+ return withRequestDeadline(signal, async (signal) => {
431
+ for (let attempt = 1; attempt <= MAX_RETRIES + 1; attempt++) {
432
+ signal.throwIfAborted();
433
+ let response: Response;
434
+ try {
435
+ response = await abortable(fetch(CODEX_RESPONSES_URL, {
436
+ method: "POST", headers, body, signal, redirect: "error",
437
+ }), signal);
438
+ } catch {
439
+ signal.throwIfAborted();
440
+ throw new Error("Codex connection failed. No automatic retry was made; check connectivity before trying again.");
542
441
  }
543
- throw new Error(`Codex image generation request failed (${response.status}): ${errorText}`);
544
- }
545
442
 
546
- return parseCodexSse(response, signal);
547
- }
443
+ if (!response.ok) {
444
+ const failure = await httpFailure(response, signal);
445
+ if (attempt <= MAX_RETRIES && failure.retry) {
446
+ const delay = retryDelayMs(attempt, response.headers.get("retry-after"));
447
+ await abortableDelay(delay, signal);
448
+ continue;
449
+ }
450
+ throw new Error(failure.message);
451
+ }
548
452
 
549
- throw new Error("Codex image generation request failed after all retries.");
453
+ return parseCodexSse(response, signal, [token, accountId], onProgress);
454
+ }
455
+ throw new Error("Codex image generation request failed after all retries.");
456
+ });
550
457
  }
551
458
 
552
459
  // --- Extension entry point ---
@@ -558,26 +465,40 @@ export default function codexImageGen(pi: ExtensionAPI) {
558
465
  name: "codex_generate_image",
559
466
  label: "Codex Image",
560
467
  description:
561
- "Generate or edit an image with the OpenAI Codex ChatGPT backend built-in image_generation tool (gpt-image-2). Accepts up to five local or recent conversation images. Uses the existing openai-codex login; does not require OPENAI_API_KEY.",
562
- promptSnippet: "Generate or edit bitmap images via the OpenAI Codex ChatGPT backend gpt-image-2 image_generation tool.",
468
+ "Generate or edit an image with the OpenAI Codex ChatGPT backend built-in image_generation tool. The backend selects the image model. Accepts up to five local or recent conversation images (20 MiB each, 50 MiB total). Uses the existing openai-codex login; does not require OPENAI_API_KEY. Network deadline: 5 minutes; output image limit: 32 MiB; backend text is limited to 4,000 characters.",
469
+ promptSnippet: "Generate or edit bitmap images via the OpenAI Codex ChatGPT backend image_generation tool.",
563
470
  promptGuidelines: [
564
471
  "Use codex_generate_image when the user asks to generate or edit a raster image with OpenAI/Codex image generation.",
565
472
  "Do not use codex_generate_image without a clear image-generation request, because it consumes the user's Codex image quota.",
473
+ "The model parameter selects a Codex routing model, not an image model. Do not pass gpt-image-* IDs.",
474
+ "Output metadata is backend-reported, not independently verified. Check pixels for dimensions and transparency; do not infer a served model from appearance or a successful request.",
475
+ "Do not automatically repeat quota, connection, deadline, or incomplete-stream failures. The backend may already have consumed image quota.",
566
476
  ],
567
477
  parameters: TOOL_PARAMS,
568
478
  executionMode: "parallel", // #4: safe to run concurrently — no shared state, saves serialized per-path
569
479
  async execute(toolCallId, params: ToolParams, signal, onUpdate, ctx) {
480
+ if (typeof params.prompt !== "string" || !params.prompt.trim() || params.prompt.length > MAX_PROMPT_CHARS) {
481
+ throw new Error("Image prompt must contain 1 to 32,000 characters.");
482
+ }
570
483
  const outputFormat = params.outputFormat || "png";
484
+ if (!OUTPUT_FORMATS.includes(outputFormat)) throw new Error("Unsupported image output format.");
571
485
  const projectTrusted = typeof ctx.isProjectTrusted === "function" && ctx.isProjectTrusted();
572
486
  const config = loadConfig(ctx.cwd, projectTrusted); // #5: load once, pass to resolveSaveConfig
573
487
  const requestedModel = params.model || config.model || DEFAULT_MODEL;
488
+ if (typeof requestedModel !== "string" || !requestedModel.trim() || requestedModel.length > 200) {
489
+ throw new Error("Codex routing model must contain 1 to 200 characters.");
490
+ }
491
+ if (requestedModel.startsWith("gpt-image-")) {
492
+ throw new Error("The model parameter selects a Codex routing model, not an image model. Subscription image-model selection is not verified.");
493
+ }
574
494
  const model = ctx.modelRegistry.find(PROVIDER, requestedModel)?.id || requestedModel; // #6: removed dead FALLBACK_MODEL
495
+ const sessionId = ctx.sessionManager.getSessionId();
496
+ const saveConfig = resolveSaveConfig(params, ctx.cwd, sessionId, config);
575
497
  const token = await ctx.modelRegistry.getApiKeyForProvider(PROVIDER);
576
498
  if (!token) {
577
499
  throw new Error(`Missing ${PROVIDER} credentials. Run /login and select ChatGPT Plus/Pro (Codex).`);
578
500
  }
579
501
  const accountId = extractChatGptAccountId(token);
580
- const sessionId = ctx.sessionManager.getSessionId();
581
502
  const messages: unknown[] = [];
582
503
  for (const entry of ctx.sessionManager.getBranch()) {
583
504
  if (entry.type === "message") messages.push(entry.message);
@@ -586,18 +507,24 @@ export default function codexImageGen(pi: ExtensionAPI) {
586
507
  const inputImages = await resolveInputImages(params, ctx.cwd, messages);
587
508
 
588
509
  onUpdate?.({
589
- content: [{ type: "text", text: `Requesting gpt-image-2 ${inputImages.length > 0 ? "edit" : "generation"} through ${PROVIDER}/${model}...` }],
510
+ content: [{ type: "text", text: `Requesting image ${inputImages.length > 0 ? "edit" : "generation"} through ${PROVIDER}/${model}...` }],
590
511
  details: { provider: PROVIDER, model, outputFormat, inputImageCount: inputImages.length },
591
512
  });
592
513
 
593
- const parsed = await requestImage(params, token, accountId, model, outputFormat, sessionId, inputImages, signal);
514
+ const started = Date.now();
515
+ const parsed = await requestImage(params, token, accountId, model, outputFormat, sessionId, inputImages, signal, (stage) => {
516
+ onUpdate?.({
517
+ content: [{ type: "text", text: `Codex image stage: ${stage}.` }],
518
+ details: { provider: PROVIDER, model, stage },
519
+ });
520
+ });
594
521
  if (!parsed.image) {
595
522
  const text = parsed.text.join("").trim();
596
523
  throw new Error(text ? `Codex did not return an image. Response text: ${text}` : "Codex did not return an image.");
597
524
  }
598
525
 
599
526
  const imageBytes = decodeImageData(parsed.image.result, outputFormat);
600
- const saveConfig = resolveSaveConfig(params, ctx.cwd, sessionId, config);
527
+ const reportedImage = parsed.image.reported;
601
528
  let savedPath: string | undefined;
602
529
  let attemptedPath: string | undefined;
603
530
  let saveWarning: string | undefined;
@@ -615,8 +542,11 @@ export default function codexImageGen(pi: ExtensionAPI) {
615
542
  }
616
543
 
617
544
  const summary = [
618
- `Generated image via ${PROVIDER}/${model} using backend gpt-image-2.`,
545
+ `Generated image via ${PROVIDER}/${model} using the backend-selected image model.`,
619
546
  `Status: ${parsed.image.status}.`,
547
+ reportedImage.size ? `Backend-reported size: ${reportedImage.size}.` : undefined,
548
+ reportedImage.quality ? `Backend-reported quality: ${reportedImage.quality}.` : undefined,
549
+ reportedImage.background ? `Backend-reported background: ${reportedImage.background}.` : undefined,
620
550
  parsed.image.revisedPrompt ? `Revised prompt: ${parsed.image.revisedPrompt}` : undefined,
621
551
  savedPath ? `Saved image to: ${savedPath}` : "Image was not saved to disk.",
622
552
  saveWarning ? `Warning: ${saveWarning}` : undefined,
@@ -632,7 +562,11 @@ export default function codexImageGen(pi: ExtensionAPI) {
632
562
  details: {
633
563
  provider: PROVIDER,
634
564
  model,
635
- backendImageModel: "gpt-image-2",
565
+ backendImageModel: reportedImage.model ?? "unknown",
566
+ reportedImage,
567
+ transport: "codex-responses",
568
+ generationDurationMs: Date.now() - started,
569
+ byteCount: imageBytes.length,
636
570
  outputFormat,
637
571
  saveMode: saveConfig.mode,
638
572
  savedPath,
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "pi-codex-image-gen",
3
- "version": "0.1.12",
4
- "description": "Image generation for Pi using the ChatGPT Images 2.0 model.",
3
+ "version": "0.1.13",
4
+ "description": "Image generation and editing for Pi using your ChatGPT Codex login.",
5
5
  "type": "module",
6
6
  "license": "Apache-2.0",
7
7
  "author": "Jose Mocito",
@@ -58,16 +58,19 @@
58
58
  "typebox": "*"
59
59
  },
60
60
  "devDependencies": {
61
- "@earendil-works/pi-ai": "^0.80.0",
62
- "@earendil-works/pi-coding-agent": "^0.80.0",
63
- "typebox": "^1.3.4",
64
- "@types/node": "^26.1.0",
65
- "typescript": "^6.0.3"
61
+ "@earendil-works/pi-ai": "^0.85.1",
62
+ "@earendil-works/pi-coding-agent": "^0.85.1",
63
+ "@types/node": "^26.2.0",
64
+ "typebox": "^1.3.10",
65
+ "typescript": "^7.0.2"
66
66
  },
67
67
  "publishConfig": {
68
68
  "access": "public"
69
69
  },
70
70
  "engines": {
71
71
  "node": ">=20.6.0"
72
+ },
73
+ "dependencies": {
74
+ "@mocito/install-telemetry": "0.1.1"
72
75
  }
73
76
  }
@@ -16,7 +16,7 @@ Generates or edits images for the current project (for example website assets, g
16
16
  This skill has exactly two top-level modes:
17
17
 
18
18
  - **Default Pi tool mode (preferred):** Pi `codex_generate_image` tool for new image generation, edits using up to five local or recent conversation images, reference variants, and simple transparent-image requests. Does not require `OPENAI_API_KEY`.
19
- - **Fallback CLI mode:** `scripts/image_gen.py` CLI. Use when the user explicitly asks for the CLI/API/model path, or after the user explicitly confirms a true model-native transparency fallback with `gpt-image-1.5`. Requires `OPENAI_API_KEY`.
19
+ - **Fallback CLI mode:** `scripts/image_gen.py` CLI. Use when the user explicitly asks for the CLI/API/model path, or after the user explicitly confirms a native transparency API fallback. Requires `OPENAI_API_KEY` and separate API billing.
20
20
 
21
21
  Within CLI fallback, the CLI exposes three subcommands:
22
22
 
@@ -26,11 +26,13 @@ Within CLI fallback, the CLI exposes three subcommands:
26
26
 
27
27
  Rules:
28
28
  - Use the Pi `codex_generate_image` tool by default for new image generation requests.
29
+ - The tool's `model` parameter selects a Codex routing model, not Flare or Sunburst. Do not claim a specific served image model unless the response reports it. Read `details.reportedImage` for backend-reported output settings, then inspect the actual image; prompt requests for quality, dimensions, or transparency are not guarantees.
30
+ - Do not automatically repeat quota, connection, timeout, or incomplete-stream failures. The remote generation may already have consumed quota.
29
31
  - Use `referencedImagePaths` for edits when every target has a local path. Use `numLastImagesToInclude` only when a target is available solely in recent conversation history. Never provide both selectors. Masks and advanced CLI-only controls still require confirmed CLI fallback.
30
32
  - Do not switch to CLI fallback for ordinary generation quality, size, or output file-path control.
31
33
  - If the user explicitly asks for a transparent image/background, stay on Pi `codex_generate_image` first: prompt for a flat removable chroma-key background, then remove it locally with the installed helper at `scripts/remove_chroma_key.py`.
32
34
  - Never silently switch from Pi `codex_generate_image` or CLI `gpt-image-2` to CLI `gpt-image-1.5`. Treat this as a model/path downgrade and ask the user before doing it, unless the user has already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback.
33
- - If a transparent request appears too complex for clean chroma-key removal, asks for true/native transparency, or local removal fails validation, explain that true transparency requires CLI `gpt-image-1.5 --background transparent --output-format png` because `gpt-image-2` does not support `background=transparent`, then ask whether to proceed. Run the CLI fallback only after the user confirms.
35
+ - If a transparent request appears too complex for clean chroma-key removal, asks for true/native transparency, or local removal fails validation, offer CLI `gpt-image-2 --background transparent --output-format png` (native transparency preview). Run the CLI fallback only after the user confirms.
34
36
  - The word `batch` by itself does not mean CLI fallback. If the user asks for many assets or says to batch-generate assets without explicitly asking for CLI/API/model controls, stay on the Pi tool path and issue one Pi tool call per requested asset or variant.
35
37
  - If the Pi tool fails or is unavailable, tell the user the CLI fallback exists and that it requires `OPENAI_API_KEY`. Proceed only if the user explicitly asks for that fallback.
36
38
  - If the user explicitly asks for CLI mode, use the bundled `scripts/image_gen.py` workflow. Do not create one-off SDK runners.
@@ -39,7 +41,7 @@ Rules:
39
41
  Pi tool save-path policy:
40
42
  - In Pi tool mode, generated images are saved under Pi's agent directory by default: `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*`. The default Pi agent directory is `~/.pi/agent`, but it can be overridden with `PI_CODING_AGENT_DIR`; use Pi's configured agent directory, not a hardcoded home path.
41
43
  - Do not describe or rely on OS temp as the default Pi tool destination.
42
- - Do not describe or rely on a destination-path argument (if any) on the Pi `codex_generate_image` tool. If a specific location is needed, generate first and then copy the selected output from `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*`.
44
+ - Use the tool's `save` and `saveDir` controls to choose a save directory. Custom mode appends a session directory; it does not accept an exact output filename. If an exact asset path is needed, copy the generated image there and leave the original in place.
43
45
  - Save-path precedence in Pi tool mode:
44
46
  1. If the user names a destination, copy the selected output there and leave the original in place.
45
47
  2. If the image is meant for the current project, copy the final selected image into the workspace before finishing and leave the original in place.
@@ -113,7 +115,7 @@ Assume the user wants a new image unless they clearly ask to change an existing
113
115
  - If the user's prompt is already specific and detailed, normalize it into a clear spec without adding creative requirements.
114
116
  - If the user's prompt is generic, add tasteful augmentation only when it materially improves output quality.
115
117
  10. Use the Pi `codex_generate_image` tool by default for generation and supported existing-image edits. Ask for CLI fallback confirmation only when the request requires unsupported controls such as masks.
116
- 11. For transparent-output requests, follow the transparent image guidance below: generate with Pi `codex_generate_image` on a flat chroma-key background, copy the selected output into the workspace or `tmp/imagegen/`, run the installed `scripts/remove_chroma_key.py` helper, and validate the alpha result before using it. If this path looks unsuitable or fails, ask before switching to CLI `gpt-image-1.5`.
118
+ 11. For transparent-output requests, follow the transparent image guidance below: generate with Pi `codex_generate_image` on a flat chroma-key background, copy the selected output into the workspace or `tmp/imagegen/`, run the installed `scripts/remove_chroma_key.py` helper, and validate the alpha result before using it. If this path looks unsuitable or fails, ask before switching to the API CLI.
117
119
  12. Inspect outputs and validate: subject, style, composition, text accuracy, and invariants/avoid items.
118
120
  13. Iterate with a single targeted change, then re-check.
119
121
  14. For preview-only work, render the image inline; the underlying file may remain at the default `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*` path.
@@ -154,12 +156,12 @@ Do not use #00ff00 anywhere in the subject.
154
156
  No cast shadow, no contact shadow, no reflection, no watermark, and no text unless explicitly requested.
155
157
  ```
156
158
 
157
- Do not automatically use CLI `gpt-image-1.5 --background transparent --output-format png` instead of chroma keying. Ask the user first when the user asks for true/native transparency, when local removal fails validation, or when the requested image is complex: hair, fur, feathers, smoke, glass, liquids, translucent materials, reflective objects, soft shadows, realistic product grounding, or subject colors that conflict with all practical key colors.
159
+ Do not automatically use CLI `gpt-image-2 --background transparent --output-format png` instead of chroma keying. Ask the user first when the user asks for true/native transparency, when local removal fails validation, or when the requested image is complex: hair, fur, feathers, smoke, glass, liquids, translucent materials, reflective objects, soft shadows, realistic product grounding, or subject colors that conflict with all practical key colors.
158
160
 
159
161
  Use a concise confirmation like:
160
162
 
161
163
  ```text
162
- This likely needs true native transparency. The default Pi tool path uses a chroma-key background plus local removal, but true transparency requires the CLI fallback with gpt-image-1.5 because gpt-image-2 does not support background=transparent. It also requires OPENAI_API_KEY. Should I proceed with that CLI fallback?
164
+ This likely needs native transparency. The Pi tool uses chroma-key removal. The API CLI supports native transparency in preview with gpt-image-2. It requires OPENAI_API_KEY and separate API billing. Should I use that fallback?
163
165
  ```
164
166
 
165
167
  ## Prompt augmentation
@@ -279,7 +281,7 @@ Constraints: change only the background; keep the product and its edges unchange
279
281
  - If the prompt is generic, add only the extra detail that will materially help.
280
282
  - If the prompt is already detailed, normalize it instead of expanding it.
281
283
  - For CLI fallback only, see `references/cli.md` and `references/image-api.md` for model, `quality`, `input_fidelity`, masks, output format, and output-path guidance.
282
- - For transparent images, use the built-in-first chroma-key workflow unless the request is complex enough to need true CLI transparency; ask before switching to CLI `gpt-image-1.5`.
284
+ - For transparent images, use the built-in-first chroma-key workflow unless the request needs native CLI transparency; ask before switching to the API CLI.
283
285
 
284
286
  More principles shared by both modes: `references/prompting.md`.
285
287
  Copy/paste specs shared by both modes: `references/sample-prompts.md`.
@@ -291,14 +293,14 @@ Asset-type templates (website assets, game assets, wireframes, logo) are consoli
291
293
 
292
294
  The fallback CLI defaults to `gpt-image-2`.
293
295
 
294
- - Use `gpt-image-2` for new CLI/API workflows unless the request needs true model-native transparent output.
295
- - If a transparent request may need CLI fallback, ask before using `gpt-image-1.5` unless the user already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback. Explain that the built-in chroma-key path is the default, but true transparency requires `gpt-image-1.5` because `gpt-image-2` does not support `background=transparent`.
296
+ - Keep `gpt-image-2` as the CLI default. For explicit 2.5 requests, use `gpt-image-2.5-flare` for fast generation or `gpt-image-2.5-sunburst` for editing precision. Both also accept `xhigh` and `max` quality and their `2026-09-08` snapshots. Both support the flexible size constraints below; sizes above `2560x1440` are experimental. Leave 2.5 `input_fidelity` unset; support for that control is not verified.
297
+ - Native transparency is available in preview with CLI `gpt-image-2 --background transparent --output-format png` (or `webp`). Ask before switching from Pi to this separately billed API path.
296
298
  - `gpt-image-2` always uses high fidelity for image inputs; do not set `input_fidelity` with this model.
297
299
  - `gpt-image-2` supports `quality` values `low`, `medium`, `high`, and `auto`.
298
300
  - Use `quality low` for fast drafts, thumbnails, and quick iterations. Use `medium`, `high`, or `auto` for final assets, dense text, diagrams, identity-sensitive edits, or high-resolution outputs.
299
301
  - Square images are typically fastest to generate. Use `1024x1024` for fast square drafts.
300
302
  - If the user asks for 4K-style output, use `3840x2160` for landscape or `2160x3840` for portrait.
301
- - `gpt-image-2` size may be `auto` or `WIDTHxHEIGHT` if all constraints hold: max edge `<= 3840px`, both edges multiples of `16px`, long-to-short ratio `<= 3:1`, total pixels between `655,360` and `8,294,400`.
303
+ - GPT Image 2 and 2.5 API size may be `auto` or `WIDTHxHEIGHT` if all constraints hold: max edge `<= 3840px`, both edges multiples of `16px`, long-to-short ratio `<= 3:1`, total pixels between `655,360` and `8,294,400`.
302
304
 
303
305
  Popular `gpt-image-2` sizes:
304
306
  - `1024x1024` square