pi-codex-image-gen 0.1.12 → 0.1.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -2
- package/CONTRIBUTING.md +32 -3
- package/README.md +35 -4
- package/SECURITY.md +8 -0
- package/extensions/index.ts +130 -196
- package/package.json +10 -7
- package/skills/imagegen/SKILL.md +12 -10
- package/skills/imagegen/references/cli.md +7 -5
- package/skills/imagegen/references/image-api.md +18 -8
- package/skills/imagegen/references/prompting.md +2 -2
- package/skills/imagegen/references/sample-prompts.md +2 -2
- package/skills/imagegen/scripts/image_gen.py +29 -20
- package/src/codex-response.ts +286 -0
- package/src/install-telemetry.ts +19 -46
package/extensions/index.ts
CHANGED
|
@@ -3,17 +3,19 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Registers `codex_generate_image`, a tool that uses Pi's existing
|
|
5
5
|
* openai-codex ChatGPT/Codex auth to call the Codex Responses backend with the
|
|
6
|
-
* native `image_generation` tool. The backend
|
|
6
|
+
* native `image_generation` tool. The backend selects the image model.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { readFileSync } from "node:fs";
|
|
10
|
-
import {
|
|
10
|
+
import { constants } from "node:fs";
|
|
11
|
+
import { mkdir, open, writeFile } from "node:fs/promises";
|
|
11
12
|
import { homedir } from "node:os";
|
|
12
13
|
import { isAbsolute, join, resolve } from "node:path";
|
|
13
14
|
import { StringEnum } from "@earendil-works/pi-ai";
|
|
14
|
-
import { type ExtensionAPI, getAgentDir, withFileMutationQueue } from "@earendil-works/pi-coding-agent";
|
|
15
|
+
import { CONFIG_DIR_NAME, type ExtensionAPI, getAgentDir, withFileMutationQueue } from "@earendil-works/pi-coding-agent";
|
|
15
16
|
import { type Static, Type } from "typebox";
|
|
16
17
|
import { reportInstallTelemetry } from "../src/install-telemetry.js";
|
|
18
|
+
import { abortable, httpFailure, MAX_IMAGE_BYTES, parseCodexSse, withRequestDeadline, type ParsedCodexResponse } from "../src/codex-response.js";
|
|
17
19
|
|
|
18
20
|
const PACKAGE_NAME = "pi-codex-image-gen";
|
|
19
21
|
const PROVIDER = "openai-codex";
|
|
@@ -26,6 +28,9 @@ const MAX_RETRIES = 3;
|
|
|
26
28
|
const BASE_DELAY_MS = 1000;
|
|
27
29
|
const MAX_RETRY_DELAY_MS = 30_000;
|
|
28
30
|
const MAX_EDIT_IMAGES = 5;
|
|
31
|
+
const MAX_INPUT_IMAGE_BYTES = 20 * 1024 * 1024;
|
|
32
|
+
const MAX_TOTAL_INPUT_BYTES = 50 * 1024 * 1024;
|
|
33
|
+
const MAX_PROMPT_CHARS = 32_000;
|
|
29
34
|
|
|
30
35
|
const SAVE_MODES = ["none", "project", "global", "custom"] as const;
|
|
31
36
|
type SaveMode = (typeof SAVE_MODES)[number];
|
|
@@ -35,11 +40,6 @@ type OutputFormat = (typeof OUTPUT_FORMATS)[number];
|
|
|
35
40
|
|
|
36
41
|
// --- #1: Retry helpers with exponential backoff + jitter ---
|
|
37
42
|
|
|
38
|
-
function isRetryableStatus(status: number, errorText: string): boolean {
|
|
39
|
-
if ([429, 500, 502, 503, 504].includes(status)) return true;
|
|
40
|
-
return /rate.?limit|overloaded|service.?unavailable|upstream.?connect|connection.?refused/i.test(errorText);
|
|
41
|
-
}
|
|
42
|
-
|
|
43
43
|
export function parseRetryAfter(value: string | null, nowMs = Date.now()): number | undefined {
|
|
44
44
|
if (!value) return undefined;
|
|
45
45
|
const trimmed = value.trim();
|
|
@@ -89,9 +89,9 @@ export function abortableDelay(milliseconds: number, signal?: AbortSignal): Prom
|
|
|
89
89
|
// --- Tool parameter schema ---
|
|
90
90
|
|
|
91
91
|
const TOOL_PARAMS = Type.Object({
|
|
92
|
-
prompt: Type.String({ description: "The image prompt. Be specific about subject, composition, style, text, and constraints." }),
|
|
92
|
+
prompt: Type.String({ minLength: 1, maxLength: MAX_PROMPT_CHARS, description: "The image prompt. Be specific about subject, composition, style, text, and constraints." }),
|
|
93
93
|
model: Type.Optional(
|
|
94
|
-
Type.String({ description: `Codex model
|
|
94
|
+
Type.String({ minLength: 1, maxLength: 200, description: `Codex routing model, not an image model selector. Defaults to ${DEFAULT_MODEL}.` }),
|
|
95
95
|
),
|
|
96
96
|
outputFormat: Type.Optional(StringEnum(OUTPUT_FORMATS)),
|
|
97
97
|
save: Type.Optional(StringEnum(SAVE_MODES)),
|
|
@@ -130,44 +130,11 @@ interface SaveConfig {
|
|
|
130
130
|
outputDir?: string;
|
|
131
131
|
}
|
|
132
132
|
|
|
133
|
-
interface GeneratedImage {
|
|
134
|
-
id: string;
|
|
135
|
-
status: string;
|
|
136
|
-
result: string;
|
|
137
|
-
revisedPrompt?: string;
|
|
138
|
-
}
|
|
139
|
-
|
|
140
|
-
interface ParsedCodexResponse {
|
|
141
|
-
image?: GeneratedImage;
|
|
142
|
-
text: string[];
|
|
143
|
-
responseId?: string;
|
|
144
|
-
usage?: unknown;
|
|
145
|
-
}
|
|
146
|
-
|
|
147
133
|
interface InputImage {
|
|
148
134
|
data: string;
|
|
149
135
|
mimeType: string;
|
|
150
136
|
}
|
|
151
137
|
|
|
152
|
-
// --- #11: Typed SSE event discriminated union ---
|
|
153
|
-
|
|
154
|
-
type CodexSseEvent =
|
|
155
|
-
| { type: "error"; message?: string; code?: string }
|
|
156
|
-
| { type: "response.failed"; response?: { error?: { message?: string } } }
|
|
157
|
-
| { type: "response.created"; response?: { id?: string } }
|
|
158
|
-
| { type: "response.output_text.delta"; delta?: string }
|
|
159
|
-
| {
|
|
160
|
-
type: "response.output_item.done";
|
|
161
|
-
item?: {
|
|
162
|
-
type?: string;
|
|
163
|
-
id?: string | number;
|
|
164
|
-
status?: string;
|
|
165
|
-
result?: string;
|
|
166
|
-
revised_prompt?: string;
|
|
167
|
-
};
|
|
168
|
-
}
|
|
169
|
-
| { type: "response.completed"; response?: { id?: string; usage?: unknown } };
|
|
170
|
-
|
|
171
138
|
// --- JWT helpers ---
|
|
172
139
|
|
|
173
140
|
function decodeJwtPayload(token: string): Record<string, unknown> {
|
|
@@ -177,8 +144,8 @@ function decodeJwtPayload(token: string): Record<string, unknown> {
|
|
|
177
144
|
}
|
|
178
145
|
try {
|
|
179
146
|
return JSON.parse(Buffer.from(parts[1], "base64url").toString("utf8")) as Record<string, unknown>;
|
|
180
|
-
} catch
|
|
181
|
-
throw new Error(
|
|
147
|
+
} catch {
|
|
148
|
+
throw new Error("Failed to decode OpenAI Codex auth token. Run /login for openai-codex again.");
|
|
182
149
|
}
|
|
183
150
|
}
|
|
184
151
|
|
|
@@ -208,7 +175,7 @@ function readConfigFile(path: string): ExtensionConfig {
|
|
|
208
175
|
export function loadConfig(cwd: string, projectTrusted: boolean, agentDir = getAgentDir()): ExtensionConfig {
|
|
209
176
|
const globalConfig = readConfigFile(join(agentDir, "extensions", "codex-image-gen.json"));
|
|
210
177
|
if (!projectTrusted) return globalConfig;
|
|
211
|
-
const projectConfig = readConfigFile(join(cwd,
|
|
178
|
+
const projectConfig = readConfigFile(join(cwd, CONFIG_DIR_NAME, "extensions", "codex-image-gen.json"));
|
|
212
179
|
return { ...globalConfig, ...projectConfig };
|
|
213
180
|
}
|
|
214
181
|
|
|
@@ -221,7 +188,7 @@ export function resolveUnderCwd(cwd: string, path: string, homeDir = homedir()):
|
|
|
221
188
|
}
|
|
222
189
|
|
|
223
190
|
function sanitizePathPart(value: string, fallback: string): string {
|
|
224
|
-
const sanitized = value
|
|
191
|
+
const sanitized = value.slice(0, 128)
|
|
225
192
|
.split("")
|
|
226
193
|
.map((ch) => (/[a-zA-Z0-9_-]/.test(ch) ? ch : "_"))
|
|
227
194
|
.join("")
|
|
@@ -239,7 +206,7 @@ function resolveSaveConfig(params: ToolParams, cwd: string, sessionId: string, c
|
|
|
239
206
|
throw new Error(`Invalid save mode: ${mode}. Expected one of ${SAVE_MODES.join(", ")}.`);
|
|
240
207
|
}
|
|
241
208
|
if (mode === "project") {
|
|
242
|
-
return { mode, outputDir: join(cwd,
|
|
209
|
+
return { mode, outputDir: join(cwd, CONFIG_DIR_NAME, "generated-images", safeSessionId) };
|
|
243
210
|
}
|
|
244
211
|
if (mode === "global") {
|
|
245
212
|
return { mode, outputDir: join(getAgentDir(), "generated-images", safeSessionId) };
|
|
@@ -270,12 +237,13 @@ function imagePath(outputFormat: OutputFormat, outputDir: string, imageCallId: s
|
|
|
270
237
|
}
|
|
271
238
|
|
|
272
239
|
export function decodeImageData(base64Data: string, outputFormat: OutputFormat): Buffer {
|
|
240
|
+
if (base64Data.length > Math.ceil(MAX_IMAGE_BYTES / 3) * 4) throw new Error("Codex image exceeded the 32 MiB size limit.");
|
|
273
241
|
const value = base64Data.trim();
|
|
274
|
-
if (!value || value.length % 4 !== 0 ||
|
|
242
|
+
if (!value || value.length % 4 !== 0 || /[^A-Za-z0-9+/=]/.test(value)) {
|
|
275
243
|
throw new Error("Codex returned invalid base64 image data.");
|
|
276
244
|
}
|
|
277
245
|
const bytes = Buffer.from(value, "base64");
|
|
278
|
-
if (bytes.length === 0 || bytes.toString("base64") !== value) {
|
|
246
|
+
if (bytes.length === 0 || bytes.length > MAX_IMAGE_BYTES || bytes.toString("base64") !== value) {
|
|
279
247
|
throw new Error("Codex returned invalid base64 image data.");
|
|
280
248
|
}
|
|
281
249
|
const validSignature =
|
|
@@ -294,8 +262,8 @@ async function saveImage(
|
|
|
294
262
|
): Promise<string> {
|
|
295
263
|
const filePath = imagePath(outputFormat, outputDir, imageCallId);
|
|
296
264
|
await withFileMutationQueue(filePath, async () => {
|
|
297
|
-
await mkdir(outputDir, { recursive: true });
|
|
298
|
-
await writeFile(filePath, bytes);
|
|
265
|
+
await mkdir(outputDir, { recursive: true, mode: 0o700 });
|
|
266
|
+
await writeFile(filePath, bytes, { flag: "wx", mode: 0o600 });
|
|
299
267
|
});
|
|
300
268
|
return filePath;
|
|
301
269
|
}
|
|
@@ -322,6 +290,29 @@ function mimeFromBytes(bytes: Buffer, path: string): string {
|
|
|
322
290
|
throw new Error(`Referenced image is unavailable or unsupported: ${path}`);
|
|
323
291
|
}
|
|
324
292
|
|
|
293
|
+
async function readInputImage(path: string): Promise<Buffer> {
|
|
294
|
+
// O_NONBLOCK prevents named pipes from blocking before the regular-file check.
|
|
295
|
+
const file = await open(path, constants.O_RDONLY | (constants.O_NONBLOCK ?? 0));
|
|
296
|
+
try {
|
|
297
|
+
const info = await file.stat();
|
|
298
|
+
if (!info.isFile()) throw new Error("Referenced images must be regular files.");
|
|
299
|
+
if (info.size > MAX_INPUT_IMAGE_BYTES) throw new Error("Referenced image exceeds 20 MiB.");
|
|
300
|
+
const blocks: Buffer[] = [];
|
|
301
|
+
let total = 0;
|
|
302
|
+
while (true) {
|
|
303
|
+
const block = Buffer.alloc(Math.min(64 * 1024, MAX_INPUT_IMAGE_BYTES + 1 - total));
|
|
304
|
+
const { bytesRead } = await file.read(block, 0, block.length, null);
|
|
305
|
+
if (!bytesRead) break;
|
|
306
|
+
total += bytesRead;
|
|
307
|
+
if (total > MAX_INPUT_IMAGE_BYTES) throw new Error("Referenced image exceeds 20 MiB.");
|
|
308
|
+
blocks.push(block.subarray(0, bytesRead));
|
|
309
|
+
}
|
|
310
|
+
return Buffer.concat(blocks, total);
|
|
311
|
+
} finally {
|
|
312
|
+
await file.close();
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
|
|
325
316
|
export async function resolveInputImages(
|
|
326
317
|
params: ToolParams,
|
|
327
318
|
cwd: string,
|
|
@@ -333,19 +324,22 @@ export async function resolveInputImages(
|
|
|
333
324
|
}
|
|
334
325
|
if (paths.length > MAX_EDIT_IMAGES) throw new Error(`referencedImagePaths accepts at most ${MAX_EDIT_IMAGES} paths.`);
|
|
335
326
|
if (paths.length > 0) {
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
}
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
327
|
+
const images: InputImage[] = [];
|
|
328
|
+
let total = 0;
|
|
329
|
+
for (const path of paths) {
|
|
330
|
+
const normalized = path.startsWith("@") ? path.slice(1) : path;
|
|
331
|
+
const absolutePath = resolveUnderCwd(cwd, normalized);
|
|
332
|
+
let bytes: Buffer;
|
|
333
|
+
try {
|
|
334
|
+
bytes = await readInputImage(absolutePath);
|
|
335
|
+
} catch (error) {
|
|
336
|
+
throw new Error(`Unable to read referenced image at ${absolutePath}: ${error instanceof Error ? error.message : String(error)}`);
|
|
337
|
+
}
|
|
338
|
+
total += bytes.length;
|
|
339
|
+
if (total > MAX_TOTAL_INPUT_BYTES) throw new Error("Referenced images exceed 50 MiB in total.");
|
|
340
|
+
images.push({ data: bytes.toString("base64"), mimeType: mimeFromBytes(bytes, absolutePath) });
|
|
341
|
+
}
|
|
342
|
+
return images;
|
|
349
343
|
}
|
|
350
344
|
if (params.numLastImagesToInclude !== undefined) {
|
|
351
345
|
const count = params.numLastImagesToInclude;
|
|
@@ -356,6 +350,16 @@ export async function resolveInputImages(
|
|
|
356
350
|
if (images.length !== count) {
|
|
357
351
|
throw new Error(`Requested the last ${count} conversation images, but only ${images.length} were available.`);
|
|
358
352
|
}
|
|
353
|
+
let total = 0;
|
|
354
|
+
for (const image of images) {
|
|
355
|
+
if (image.data.length > Math.ceil(MAX_INPUT_IMAGE_BYTES / 3) * 4) throw new Error("Conversation image exceeds 20 MiB.");
|
|
356
|
+
const format = OUTPUT_FORMATS.find(format => mimeForFormat(format) === image.mimeType);
|
|
357
|
+
if (!format) throw new Error("Conversation image has an unsupported format.");
|
|
358
|
+
const bytes = decodeImageData(image.data, format);
|
|
359
|
+
if (bytes.length > MAX_INPUT_IMAGE_BYTES) throw new Error("Conversation image exceeds 20 MiB.");
|
|
360
|
+
total += bytes.length;
|
|
361
|
+
if (total > MAX_TOTAL_INPUT_BYTES) throw new Error("Conversation images exceed 50 MiB in total.");
|
|
362
|
+
}
|
|
359
363
|
return images;
|
|
360
364
|
}
|
|
361
365
|
return [];
|
|
@@ -399,108 +403,6 @@ export function buildRequestBody(
|
|
|
399
403
|
};
|
|
400
404
|
}
|
|
401
405
|
|
|
402
|
-
// --- SSE parsing ---
|
|
403
|
-
|
|
404
|
-
function parseSseDataLines(chunk: string): string | undefined {
|
|
405
|
-
const data = chunk
|
|
406
|
-
.split("\n")
|
|
407
|
-
.filter((line) => line.startsWith("data:"))
|
|
408
|
-
.map((line) => line.slice(5).trim())
|
|
409
|
-
.join("\n")
|
|
410
|
-
.trim();
|
|
411
|
-
return data && data !== "[DONE]" ? data : undefined;
|
|
412
|
-
}
|
|
413
|
-
|
|
414
|
-
async function parseCodexSse(response: Response, signal?: AbortSignal): Promise<ParsedCodexResponse> {
|
|
415
|
-
if (!response.body) throw new Error("Codex response did not include a stream body.");
|
|
416
|
-
const reader = response.body.getReader();
|
|
417
|
-
const decoder = new TextDecoder();
|
|
418
|
-
let buffer = "";
|
|
419
|
-
const parsed: ParsedCodexResponse = { text: [] };
|
|
420
|
-
|
|
421
|
-
try {
|
|
422
|
-
while (true) {
|
|
423
|
-
if (signal?.aborted) throw new Error("Image generation was aborted.");
|
|
424
|
-
const { done, value } = await reader.read();
|
|
425
|
-
if (done) break;
|
|
426
|
-
buffer += decoder.decode(value, { stream: true });
|
|
427
|
-
|
|
428
|
-
let separator = buffer.indexOf("\n\n");
|
|
429
|
-
while (separator !== -1) {
|
|
430
|
-
const chunk = buffer.slice(0, separator);
|
|
431
|
-
buffer = buffer.slice(separator + 2);
|
|
432
|
-
const data = parseSseDataLines(chunk);
|
|
433
|
-
if (data) handleCodexEvent(JSON.parse(data) as CodexSseEvent, parsed);
|
|
434
|
-
separator = buffer.indexOf("\n\n");
|
|
435
|
-
}
|
|
436
|
-
}
|
|
437
|
-
const remaining = parseSseDataLines(buffer);
|
|
438
|
-
if (remaining) handleCodexEvent(JSON.parse(remaining) as CodexSseEvent, parsed);
|
|
439
|
-
} finally {
|
|
440
|
-
try {
|
|
441
|
-
await reader.cancel();
|
|
442
|
-
} catch {
|
|
443
|
-
// ignored: stream may already be closed
|
|
444
|
-
}
|
|
445
|
-
reader.releaseLock();
|
|
446
|
-
}
|
|
447
|
-
|
|
448
|
-
return parsed;
|
|
449
|
-
}
|
|
450
|
-
|
|
451
|
-
// --- #11: Typed event handler via discriminated union ---
|
|
452
|
-
|
|
453
|
-
function handleCodexEvent(event: CodexSseEvent, parsed: ParsedCodexResponse): void {
|
|
454
|
-
if (!event || typeof event !== "object") return;
|
|
455
|
-
|
|
456
|
-
switch (event.type) {
|
|
457
|
-
case "error": {
|
|
458
|
-
const e = event as Extract<CodexSseEvent, { type: "error" }>;
|
|
459
|
-
throw new Error(`Codex error: ${e.message || e.code || JSON.stringify(event)}`);
|
|
460
|
-
}
|
|
461
|
-
case "response.failed": {
|
|
462
|
-
const e = event as Extract<CodexSseEvent, { type: "response.failed" }>;
|
|
463
|
-
throw new Error(e.response?.error?.message || "Codex response failed.");
|
|
464
|
-
}
|
|
465
|
-
case "response.created": {
|
|
466
|
-
const e = event as Extract<CodexSseEvent, { type: "response.created" }>;
|
|
467
|
-
if (typeof e.response?.id === "string") {
|
|
468
|
-
parsed.responseId = e.response.id;
|
|
469
|
-
}
|
|
470
|
-
break;
|
|
471
|
-
}
|
|
472
|
-
case "response.output_text.delta": {
|
|
473
|
-
const e = event as Extract<CodexSseEvent, { type: "response.output_text.delta" }>;
|
|
474
|
-
if (typeof e.delta === "string") {
|
|
475
|
-
parsed.text.push(e.delta);
|
|
476
|
-
}
|
|
477
|
-
break;
|
|
478
|
-
}
|
|
479
|
-
case "response.output_item.done": {
|
|
480
|
-
const e = event as Extract<CodexSseEvent, { type: "response.output_item.done" }>;
|
|
481
|
-
const item = e.item;
|
|
482
|
-
if (item?.type === "image_generation_call") {
|
|
483
|
-
if (typeof item.result !== "string" || item.result.length === 0) {
|
|
484
|
-
throw new Error("Codex image_generation_call did not contain image data.");
|
|
485
|
-
}
|
|
486
|
-
parsed.image = {
|
|
487
|
-
id: String(item.id || "image_generation"),
|
|
488
|
-
status: String(item.status || "completed"),
|
|
489
|
-
result: item.result,
|
|
490
|
-
revisedPrompt: typeof item.revised_prompt === "string" ? item.revised_prompt : undefined,
|
|
491
|
-
};
|
|
492
|
-
}
|
|
493
|
-
break;
|
|
494
|
-
}
|
|
495
|
-
case "response.completed": {
|
|
496
|
-
const e = event as Extract<CodexSseEvent, { type: "response.completed" }>;
|
|
497
|
-
if (typeof e.response?.id === "string") parsed.responseId = e.response.id;
|
|
498
|
-
if (e.response?.usage) parsed.usage = e.response.usage;
|
|
499
|
-
break;
|
|
500
|
-
}
|
|
501
|
-
}
|
|
502
|
-
}
|
|
503
|
-
|
|
504
406
|
// --- #1: requestImage with retry + backoff + jitter ---
|
|
505
407
|
|
|
506
408
|
async function requestImage(
|
|
@@ -512,41 +414,46 @@ async function requestImage(
|
|
|
512
414
|
sessionId: string,
|
|
513
415
|
inputImages: InputImage[],
|
|
514
416
|
signal?: AbortSignal,
|
|
417
|
+
onProgress?: (stage: string) => void,
|
|
515
418
|
): Promise<ParsedCodexResponse> {
|
|
516
419
|
const body = JSON.stringify(buildRequestBody(params, model, outputFormat, sessionId, inputImages));
|
|
517
420
|
const headers: Record<string, string> = {
|
|
518
421
|
Authorization: `Bearer ${token}`,
|
|
519
422
|
"chatgpt-account-id": accountId,
|
|
520
423
|
originator: "pi",
|
|
424
|
+
"User-Agent": PACKAGE_NAME,
|
|
521
425
|
"OpenAI-Beta": OPENAI_BETA_HEADER,
|
|
522
426
|
accept: "text/event-stream",
|
|
523
427
|
"content-type": "application/json",
|
|
524
428
|
};
|
|
525
429
|
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
const errorText = await response.text();
|
|
538
|
-
if (attempt <= MAX_RETRIES && isRetryableStatus(response.status, errorText)) {
|
|
539
|
-
const delay = retryDelayMs(attempt, response.headers.get("retry-after"));
|
|
540
|
-
await abortableDelay(delay, signal);
|
|
541
|
-
continue;
|
|
430
|
+
return withRequestDeadline(signal, async (signal) => {
|
|
431
|
+
for (let attempt = 1; attempt <= MAX_RETRIES + 1; attempt++) {
|
|
432
|
+
signal.throwIfAborted();
|
|
433
|
+
let response: Response;
|
|
434
|
+
try {
|
|
435
|
+
response = await abortable(fetch(CODEX_RESPONSES_URL, {
|
|
436
|
+
method: "POST", headers, body, signal, redirect: "error",
|
|
437
|
+
}), signal);
|
|
438
|
+
} catch {
|
|
439
|
+
signal.throwIfAborted();
|
|
440
|
+
throw new Error("Codex connection failed. No automatic retry was made; check connectivity before trying again.");
|
|
542
441
|
}
|
|
543
|
-
throw new Error(`Codex image generation request failed (${response.status}): ${errorText}`);
|
|
544
|
-
}
|
|
545
442
|
|
|
546
|
-
|
|
547
|
-
|
|
443
|
+
if (!response.ok) {
|
|
444
|
+
const failure = await httpFailure(response, signal);
|
|
445
|
+
if (attempt <= MAX_RETRIES && failure.retry) {
|
|
446
|
+
const delay = retryDelayMs(attempt, response.headers.get("retry-after"));
|
|
447
|
+
await abortableDelay(delay, signal);
|
|
448
|
+
continue;
|
|
449
|
+
}
|
|
450
|
+
throw new Error(failure.message);
|
|
451
|
+
}
|
|
548
452
|
|
|
549
|
-
|
|
453
|
+
return parseCodexSse(response, signal, [token, accountId], onProgress);
|
|
454
|
+
}
|
|
455
|
+
throw new Error("Codex image generation request failed after all retries.");
|
|
456
|
+
});
|
|
550
457
|
}
|
|
551
458
|
|
|
552
459
|
// --- Extension entry point ---
|
|
@@ -558,26 +465,40 @@ export default function codexImageGen(pi: ExtensionAPI) {
|
|
|
558
465
|
name: "codex_generate_image",
|
|
559
466
|
label: "Codex Image",
|
|
560
467
|
description:
|
|
561
|
-
"Generate or edit an image with the OpenAI Codex ChatGPT backend built-in image_generation tool
|
|
562
|
-
promptSnippet: "Generate or edit bitmap images via the OpenAI Codex ChatGPT backend
|
|
468
|
+
"Generate or edit an image with the OpenAI Codex ChatGPT backend built-in image_generation tool. The backend selects the image model. Accepts up to five local or recent conversation images (20 MiB each, 50 MiB total). Uses the existing openai-codex login; does not require OPENAI_API_KEY. Network deadline: 5 minutes; output image limit: 32 MiB; backend text is limited to 4,000 characters.",
|
|
469
|
+
promptSnippet: "Generate or edit bitmap images via the OpenAI Codex ChatGPT backend image_generation tool.",
|
|
563
470
|
promptGuidelines: [
|
|
564
471
|
"Use codex_generate_image when the user asks to generate or edit a raster image with OpenAI/Codex image generation.",
|
|
565
472
|
"Do not use codex_generate_image without a clear image-generation request, because it consumes the user's Codex image quota.",
|
|
473
|
+
"The model parameter selects a Codex routing model, not an image model. Do not pass gpt-image-* IDs.",
|
|
474
|
+
"Output metadata is backend-reported, not independently verified. Check pixels for dimensions and transparency; do not infer a served model from appearance or a successful request.",
|
|
475
|
+
"Do not automatically repeat quota, connection, deadline, or incomplete-stream failures. The backend may already have consumed image quota.",
|
|
566
476
|
],
|
|
567
477
|
parameters: TOOL_PARAMS,
|
|
568
478
|
executionMode: "parallel", // #4: safe to run concurrently — no shared state, saves serialized per-path
|
|
569
479
|
async execute(toolCallId, params: ToolParams, signal, onUpdate, ctx) {
|
|
480
|
+
if (typeof params.prompt !== "string" || !params.prompt.trim() || params.prompt.length > MAX_PROMPT_CHARS) {
|
|
481
|
+
throw new Error("Image prompt must contain 1 to 32,000 characters.");
|
|
482
|
+
}
|
|
570
483
|
const outputFormat = params.outputFormat || "png";
|
|
484
|
+
if (!OUTPUT_FORMATS.includes(outputFormat)) throw new Error("Unsupported image output format.");
|
|
571
485
|
const projectTrusted = typeof ctx.isProjectTrusted === "function" && ctx.isProjectTrusted();
|
|
572
486
|
const config = loadConfig(ctx.cwd, projectTrusted); // #5: load once, pass to resolveSaveConfig
|
|
573
487
|
const requestedModel = params.model || config.model || DEFAULT_MODEL;
|
|
488
|
+
if (typeof requestedModel !== "string" || !requestedModel.trim() || requestedModel.length > 200) {
|
|
489
|
+
throw new Error("Codex routing model must contain 1 to 200 characters.");
|
|
490
|
+
}
|
|
491
|
+
if (requestedModel.startsWith("gpt-image-")) {
|
|
492
|
+
throw new Error("The model parameter selects a Codex routing model, not an image model. Subscription image-model selection is not verified.");
|
|
493
|
+
}
|
|
574
494
|
const model = ctx.modelRegistry.find(PROVIDER, requestedModel)?.id || requestedModel; // #6: removed dead FALLBACK_MODEL
|
|
495
|
+
const sessionId = ctx.sessionManager.getSessionId();
|
|
496
|
+
const saveConfig = resolveSaveConfig(params, ctx.cwd, sessionId, config);
|
|
575
497
|
const token = await ctx.modelRegistry.getApiKeyForProvider(PROVIDER);
|
|
576
498
|
if (!token) {
|
|
577
499
|
throw new Error(`Missing ${PROVIDER} credentials. Run /login and select ChatGPT Plus/Pro (Codex).`);
|
|
578
500
|
}
|
|
579
501
|
const accountId = extractChatGptAccountId(token);
|
|
580
|
-
const sessionId = ctx.sessionManager.getSessionId();
|
|
581
502
|
const messages: unknown[] = [];
|
|
582
503
|
for (const entry of ctx.sessionManager.getBranch()) {
|
|
583
504
|
if (entry.type === "message") messages.push(entry.message);
|
|
@@ -586,18 +507,24 @@ export default function codexImageGen(pi: ExtensionAPI) {
|
|
|
586
507
|
const inputImages = await resolveInputImages(params, ctx.cwd, messages);
|
|
587
508
|
|
|
588
509
|
onUpdate?.({
|
|
589
|
-
content: [{ type: "text", text: `Requesting
|
|
510
|
+
content: [{ type: "text", text: `Requesting image ${inputImages.length > 0 ? "edit" : "generation"} through ${PROVIDER}/${model}...` }],
|
|
590
511
|
details: { provider: PROVIDER, model, outputFormat, inputImageCount: inputImages.length },
|
|
591
512
|
});
|
|
592
513
|
|
|
593
|
-
const
|
|
514
|
+
const started = Date.now();
|
|
515
|
+
const parsed = await requestImage(params, token, accountId, model, outputFormat, sessionId, inputImages, signal, (stage) => {
|
|
516
|
+
onUpdate?.({
|
|
517
|
+
content: [{ type: "text", text: `Codex image stage: ${stage}.` }],
|
|
518
|
+
details: { provider: PROVIDER, model, stage },
|
|
519
|
+
});
|
|
520
|
+
});
|
|
594
521
|
if (!parsed.image) {
|
|
595
522
|
const text = parsed.text.join("").trim();
|
|
596
523
|
throw new Error(text ? `Codex did not return an image. Response text: ${text}` : "Codex did not return an image.");
|
|
597
524
|
}
|
|
598
525
|
|
|
599
526
|
const imageBytes = decodeImageData(parsed.image.result, outputFormat);
|
|
600
|
-
const
|
|
527
|
+
const reportedImage = parsed.image.reported;
|
|
601
528
|
let savedPath: string | undefined;
|
|
602
529
|
let attemptedPath: string | undefined;
|
|
603
530
|
let saveWarning: string | undefined;
|
|
@@ -615,8 +542,11 @@ export default function codexImageGen(pi: ExtensionAPI) {
|
|
|
615
542
|
}
|
|
616
543
|
|
|
617
544
|
const summary = [
|
|
618
|
-
`Generated image via ${PROVIDER}/${model} using backend
|
|
545
|
+
`Generated image via ${PROVIDER}/${model} using the backend-selected image model.`,
|
|
619
546
|
`Status: ${parsed.image.status}.`,
|
|
547
|
+
reportedImage.size ? `Backend-reported size: ${reportedImage.size}.` : undefined,
|
|
548
|
+
reportedImage.quality ? `Backend-reported quality: ${reportedImage.quality}.` : undefined,
|
|
549
|
+
reportedImage.background ? `Backend-reported background: ${reportedImage.background}.` : undefined,
|
|
620
550
|
parsed.image.revisedPrompt ? `Revised prompt: ${parsed.image.revisedPrompt}` : undefined,
|
|
621
551
|
savedPath ? `Saved image to: ${savedPath}` : "Image was not saved to disk.",
|
|
622
552
|
saveWarning ? `Warning: ${saveWarning}` : undefined,
|
|
@@ -632,7 +562,11 @@ export default function codexImageGen(pi: ExtensionAPI) {
|
|
|
632
562
|
details: {
|
|
633
563
|
provider: PROVIDER,
|
|
634
564
|
model,
|
|
635
|
-
backendImageModel: "
|
|
565
|
+
backendImageModel: reportedImage.model ?? "unknown",
|
|
566
|
+
reportedImage,
|
|
567
|
+
transport: "codex-responses",
|
|
568
|
+
generationDurationMs: Date.now() - started,
|
|
569
|
+
byteCount: imageBytes.length,
|
|
636
570
|
outputFormat,
|
|
637
571
|
saveMode: saveConfig.mode,
|
|
638
572
|
savedPath,
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-codex-image-gen",
|
|
3
|
-
"version": "0.1.
|
|
4
|
-
"description": "Image generation for Pi using
|
|
3
|
+
"version": "0.1.13",
|
|
4
|
+
"description": "Image generation and editing for Pi using your ChatGPT Codex login.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
7
7
|
"author": "Jose Mocito",
|
|
@@ -58,16 +58,19 @@
|
|
|
58
58
|
"typebox": "*"
|
|
59
59
|
},
|
|
60
60
|
"devDependencies": {
|
|
61
|
-
"@earendil-works/pi-ai": "^0.
|
|
62
|
-
"@earendil-works/pi-coding-agent": "^0.
|
|
63
|
-
"
|
|
64
|
-
"
|
|
65
|
-
"typescript": "^
|
|
61
|
+
"@earendil-works/pi-ai": "^0.85.1",
|
|
62
|
+
"@earendil-works/pi-coding-agent": "^0.85.1",
|
|
63
|
+
"@types/node": "^26.2.0",
|
|
64
|
+
"typebox": "^1.3.10",
|
|
65
|
+
"typescript": "^7.0.2"
|
|
66
66
|
},
|
|
67
67
|
"publishConfig": {
|
|
68
68
|
"access": "public"
|
|
69
69
|
},
|
|
70
70
|
"engines": {
|
|
71
71
|
"node": ">=20.6.0"
|
|
72
|
+
},
|
|
73
|
+
"dependencies": {
|
|
74
|
+
"@mocito/install-telemetry": "0.1.1"
|
|
72
75
|
}
|
|
73
76
|
}
|
package/skills/imagegen/SKILL.md
CHANGED
|
@@ -16,7 +16,7 @@ Generates or edits images for the current project (for example website assets, g
|
|
|
16
16
|
This skill has exactly two top-level modes:
|
|
17
17
|
|
|
18
18
|
- **Default Pi tool mode (preferred):** Pi `codex_generate_image` tool for new image generation, edits using up to five local or recent conversation images, reference variants, and simple transparent-image requests. Does not require `OPENAI_API_KEY`.
|
|
19
|
-
- **Fallback CLI mode:** `scripts/image_gen.py` CLI. Use when the user explicitly asks for the CLI/API/model path, or after the user explicitly confirms a
|
|
19
|
+
- **Fallback CLI mode:** `scripts/image_gen.py` CLI. Use when the user explicitly asks for the CLI/API/model path, or after the user explicitly confirms a native transparency API fallback. Requires `OPENAI_API_KEY` and separate API billing.
|
|
20
20
|
|
|
21
21
|
Within CLI fallback, the CLI exposes three subcommands:
|
|
22
22
|
|
|
@@ -26,11 +26,13 @@ Within CLI fallback, the CLI exposes three subcommands:
|
|
|
26
26
|
|
|
27
27
|
Rules:
|
|
28
28
|
- Use the Pi `codex_generate_image` tool by default for new image generation requests.
|
|
29
|
+
- The tool's `model` parameter selects a Codex routing model, not Flare or Sunburst. Do not claim a specific served image model unless the response reports it. Read `details.reportedImage` for backend-reported output settings, then inspect the actual image; prompt requests for quality, dimensions, or transparency are not guarantees.
|
|
30
|
+
- Do not automatically repeat quota, connection, timeout, or incomplete-stream failures. The remote generation may already have consumed quota.
|
|
29
31
|
- Use `referencedImagePaths` for edits when every target has a local path. Use `numLastImagesToInclude` only when a target is available solely in recent conversation history. Never provide both selectors. Masks and advanced CLI-only controls still require confirmed CLI fallback.
|
|
30
32
|
- Do not switch to CLI fallback for ordinary generation quality, size, or output file-path control.
|
|
31
33
|
- If the user explicitly asks for a transparent image/background, stay on Pi `codex_generate_image` first: prompt for a flat removable chroma-key background, then remove it locally with the installed helper at `scripts/remove_chroma_key.py`.
|
|
32
34
|
- Never silently switch from Pi `codex_generate_image` or CLI `gpt-image-2` to CLI `gpt-image-1.5`. Treat this as a model/path downgrade and ask the user before doing it, unless the user has already explicitly requested `gpt-image-1.5`, `scripts/image_gen.py`, or CLI fallback.
|
|
33
|
-
- If a transparent request appears too complex for clean chroma-key removal, asks for true/native transparency, or local removal fails validation,
|
|
35
|
+
- If a transparent request appears too complex for clean chroma-key removal, asks for true/native transparency, or local removal fails validation, offer CLI `gpt-image-2 --background transparent --output-format png` (native transparency preview). Run the CLI fallback only after the user confirms.
|
|
34
36
|
- The word `batch` by itself does not mean CLI fallback. If the user asks for many assets or says to batch-generate assets without explicitly asking for CLI/API/model controls, stay on the Pi tool path and issue one Pi tool call per requested asset or variant.
|
|
35
37
|
- If the Pi tool fails or is unavailable, tell the user the CLI fallback exists and that it requires `OPENAI_API_KEY`. Proceed only if the user explicitly asks for that fallback.
|
|
36
38
|
- If the user explicitly asks for CLI mode, use the bundled `scripts/image_gen.py` workflow. Do not create one-off SDK runners.
|
|
@@ -39,7 +41,7 @@ Rules:
|
|
|
39
41
|
Pi tool save-path policy:
|
|
40
42
|
- In Pi tool mode, generated images are saved under Pi's agent directory by default: `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*`. The default Pi agent directory is `~/.pi/agent`, but it can be overridden with `PI_CODING_AGENT_DIR`; use Pi's configured agent directory, not a hardcoded home path.
|
|
41
43
|
- Do not describe or rely on OS temp as the default Pi tool destination.
|
|
42
|
-
-
|
|
44
|
+
- Use the tool's `save` and `saveDir` controls to choose a save directory. Custom mode appends a session directory; it does not accept an exact output filename. If an exact asset path is needed, copy the generated image there and leave the original in place.
|
|
43
45
|
- Save-path precedence in Pi tool mode:
|
|
44
46
|
1. If the user names a destination, copy the selected output there and leave the original in place.
|
|
45
47
|
2. If the image is meant for the current project, copy the final selected image into the workspace before finishing and leave the original in place.
|
|
@@ -113,7 +115,7 @@ Assume the user wants a new image unless they clearly ask to change an existing
|
|
|
113
115
|
- If the user's prompt is already specific and detailed, normalize it into a clear spec without adding creative requirements.
|
|
114
116
|
- If the user's prompt is generic, add tasteful augmentation only when it materially improves output quality.
|
|
115
117
|
10. Use the Pi `codex_generate_image` tool by default for generation and supported existing-image edits. Ask for CLI fallback confirmation only when the request requires unsupported controls such as masks.
|
|
116
|
-
11. For transparent-output requests, follow the transparent image guidance below: generate with Pi `codex_generate_image` on a flat chroma-key background, copy the selected output into the workspace or `tmp/imagegen/`, run the installed `scripts/remove_chroma_key.py` helper, and validate the alpha result before using it. If this path looks unsuitable or fails, ask before switching to CLI
|
|
118
|
+
11. For transparent-output requests, follow the transparent image guidance below: generate with Pi `codex_generate_image` on a flat chroma-key background, copy the selected output into the workspace or `tmp/imagegen/`, run the installed `scripts/remove_chroma_key.py` helper, and validate the alpha result before using it. If this path looks unsuitable or fails, ask before switching to the API CLI.
|
|
117
119
|
12. Inspect outputs and validate: subject, style, composition, text accuracy, and invariants/avoid items.
|
|
118
120
|
13. Iterate with a single targeted change, then re-check.
|
|
119
121
|
14. For preview-only work, render the image inline; the underlying file may remain at the default `<pi-agent-dir>/generated-images/<pi-session-id>/<image-call-id>.*` path.
|
|
@@ -154,12 +156,12 @@ Do not use #00ff00 anywhere in the subject.
|
|
|
154
156
|
No cast shadow, no contact shadow, no reflection, no watermark, and no text unless explicitly requested.
|
|
155
157
|
```
|
|
156
158
|
|
|
157
|
-
Do not automatically use CLI `gpt-image-
|
|
159
|
+
Do not automatically use CLI `gpt-image-2 --background transparent --output-format png` instead of chroma keying. Ask the user first when the user asks for true/native transparency, when local removal fails validation, or when the requested image is complex: hair, fur, feathers, smoke, glass, liquids, translucent materials, reflective objects, soft shadows, realistic product grounding, or subject colors that conflict with all practical key colors.
|
|
158
160
|
|
|
159
161
|
Use a concise confirmation like:
|
|
160
162
|
|
|
161
163
|
```text
|
|
162
|
-
This likely needs
|
|
164
|
+
This likely needs native transparency. The Pi tool uses chroma-key removal. The API CLI supports native transparency in preview with gpt-image-2. It requires OPENAI_API_KEY and separate API billing. Should I use that fallback?
|
|
163
165
|
```
|
|
164
166
|
|
|
165
167
|
## Prompt augmentation
|
|
@@ -279,7 +281,7 @@ Constraints: change only the background; keep the product and its edges unchange
|
|
|
279
281
|
- If the prompt is generic, add only the extra detail that will materially help.
|
|
280
282
|
- If the prompt is already detailed, normalize it instead of expanding it.
|
|
281
283
|
- For CLI fallback only, see `references/cli.md` and `references/image-api.md` for model, `quality`, `input_fidelity`, masks, output format, and output-path guidance.
|
|
282
|
-
- For transparent images, use the built-in-first chroma-key workflow unless the request
|
|
284
|
+
- For transparent images, use the built-in-first chroma-key workflow unless the request needs native CLI transparency; ask before switching to the API CLI.
|
|
283
285
|
|
|
284
286
|
More principles shared by both modes: `references/prompting.md`.
|
|
285
287
|
Copy/paste specs shared by both modes: `references/sample-prompts.md`.
|
|
@@ -291,14 +293,14 @@ Asset-type templates (website assets, game assets, wireframes, logo) are consoli
|
|
|
291
293
|
|
|
292
294
|
The fallback CLI defaults to `gpt-image-2`.
|
|
293
295
|
|
|
294
|
-
-
|
|
295
|
-
-
|
|
296
|
+
- Keep `gpt-image-2` as the CLI default. For explicit 2.5 requests, use `gpt-image-2.5-flare` for fast generation or `gpt-image-2.5-sunburst` for editing precision. Both also accept `xhigh` and `max` quality and their `2026-09-08` snapshots. Both support the flexible size constraints below; sizes above `2560x1440` are experimental. Leave 2.5 `input_fidelity` unset; support for that control is not verified.
|
|
297
|
+
- Native transparency is available in preview with CLI `gpt-image-2 --background transparent --output-format png` (or `webp`). Ask before switching from Pi to this separately billed API path.
|
|
296
298
|
- `gpt-image-2` always uses high fidelity for image inputs; do not set `input_fidelity` with this model.
|
|
297
299
|
- `gpt-image-2` supports `quality` values `low`, `medium`, `high`, and `auto`.
|
|
298
300
|
- Use `quality low` for fast drafts, thumbnails, and quick iterations. Use `medium`, `high`, or `auto` for final assets, dense text, diagrams, identity-sensitive edits, or high-resolution outputs.
|
|
299
301
|
- Square images are typically fastest to generate. Use `1024x1024` for fast square drafts.
|
|
300
302
|
- If the user asks for 4K-style output, use `3840x2160` for landscape or `2160x3840` for portrait.
|
|
301
|
-
-
|
|
303
|
+
- GPT Image 2 and 2.5 API size may be `auto` or `WIDTHxHEIGHT` if all constraints hold: max edge `<= 3840px`, both edges multiples of `16px`, long-to-short ratio `<= 3:1`, total pixels between `655,360` and `8,294,400`.
|
|
302
304
|
|
|
303
305
|
Popular `gpt-image-2` sizes:
|
|
304
306
|
- `1024x1024` square
|