@gentbajko/slopify 2.4.0 → 2.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapter-registry.js +8 -1
- package/dist/adapters/host-cli/index.js +11 -0
- package/dist/adapters/host-cli/transport.js +6 -1
- package/dist/adapters/image/codex-output.js +59 -17
- package/dist/adapters/image/codex.js +92 -16
- package/dist/adapters/image/fal.js +22 -2
- package/dist/adapters/image/google.js +13 -1
- package/dist/adapters/image/openai.js +37 -15
- package/dist/adapters/image/reference.js +7 -0
- package/dist/adapters/image/replicate.js +7 -0
- package/dist/adapters/llm/codex-models.js +2 -1
- package/dist/adapters/llm/codex-server-models.js +2 -1
- package/dist/adapters/llm/run-cli.js +14 -2
- package/dist/assets/models.yaml +7 -0
- package/dist/catalog/registry.js +24 -3
- package/dist/catalog/runtime-models.js +1 -1
- package/dist/catalog/schema.js +10 -1
- package/dist/catalog/validate.js +17 -2
- package/dist/edge/events/hub.js +11 -0
- package/dist/edge/http/app.js +2 -0
- package/dist/edge/http/backups.js +40 -0
- package/dist/edge/http/host-cli.js +17 -2
- package/dist/edge/http/problem.js +1 -0
- package/dist/edge/http/schedules.js +1 -1
- package/dist/edge/http/settings.js +44 -0
- package/dist/host-cli/runtime.js +3 -2
- package/dist/kernel/db/migrations/0022-drop-duplicate-held-work.sql +56 -0
- package/dist/kernel/db/migrations/0024-reference-attachments.sql +18 -0
- package/dist/kernel/ports/host-cli.js +17 -2
- package/dist/kernel/ports/image.js +3 -1
- package/dist/kernel/ports/llm.js +3 -1
- package/dist/kernel/runner/attempt.js +1 -1
- package/dist/kernel/runner/providers.js +22 -2
- package/dist/main.js +52 -7
- package/dist/slices/admission/model.js +8 -0
- package/dist/slices/admission/rules.js +47 -0
- package/dist/slices/admission/schema.js +11 -1
- package/dist/slices/admission/start.js +53 -15
- package/dist/slices/backups/files.js +135 -0
- package/dist/slices/backups/folder.js +35 -0
- package/dist/slices/backups/model.js +65 -0
- package/dist/slices/backups/repo.js +28 -0
- package/dist/slices/backups/schedule.js +60 -0
- package/dist/slices/backups/service.js +311 -0
- package/dist/slices/batch/index.js +2 -0
- package/dist/slices/estimate/index.js +9 -1
- package/dist/slices/library/slots.js +8 -2
- package/dist/slices/library/snapshot.js +8 -2
- package/dist/slices/notifications/notifier.js +76 -0
- package/dist/slices/notifications/rules.js +81 -0
- package/dist/slices/notifications/send.js +40 -0
- package/dist/slices/notifications/settings.js +34 -0
- package/dist/slices/play-drafts/convert.js +35 -3
- package/dist/slices/play-drafts/repo.js +7 -1
- package/dist/slices/play-drafts/review-inputs.js +2 -0
- package/dist/slices/play-drafts/schema.js +20 -2
- package/dist/slices/play-drafts/service.js +6 -0
- package/dist/slices/play-drafts/uploads.js +2 -1
- package/dist/slices/project-templates/from-project.js +20 -0
- package/dist/slices/project-templates/setup.js +11 -0
- package/dist/slices/rebuild/preview-details.js +24 -2
- package/dist/slices/rebuild/recipe-build.js +9 -3
- package/dist/slices/rebuild/recipe-input-schema.js +8 -1
- package/dist/slices/rebuild/recipe-legacy.js +3 -0
- package/dist/slices/rebuild/recipe-reference.js +44 -0
- package/dist/slices/rebuild/recipe-shorts.js +11 -4
- package/dist/slices/rebuild/recipe-validation.js +15 -1
- package/dist/slices/rebuild/recipe-visual.js +20 -8
- package/dist/slices/rebuild/runtime-actions.js +9 -2
- package/dist/slices/rebuild/runtime-image.js +50 -0
- package/dist/slices/rebuild/runtime-local.js +6 -3
- package/dist/slices/rebuild/runtime-piece-label.js +2 -0
- package/dist/slices/rebuild/runtime-provider.js +17 -10
- package/dist/slices/rebuild/runtime-shorts.js +4 -6
- package/dist/slices/rebuild/service-readiness.js +15 -1
- package/dist/slices/rebuild/transition-repo.js +1 -1
- package/dist/slices/revisions/adopt-content.js +7 -0
- package/dist/slices/revisions/adopt.js +7 -2
- package/dist/slices/revisions/mutation-assets.js +33 -9
- package/dist/slices/revisions/mutation-prepare.js +8 -10
- package/dist/slices/revisions/schema.js +5 -1
- package/dist/slices/schedules/repo.js +1 -1
- package/dist/slices/schedules/service.js +6 -1
- package/dist/slices/settings/model.js +5 -0
- package/dist/slices/storage/layout.js +10 -0
- package/dist/slices/storage/model.js +3 -0
- package/dist/slices/storage/portable.js +6 -3
- package/dist/slices/storage/reconcile.js +5 -1
- package/dist/web/assets/index-CktpxlJd.js +171 -0
- package/dist/web/assets/index-DB-vOnGU.css +1 -0
- package/dist/web/assets/{pdf-CAJNT254.js → pdf-BLF2i8c2.js} +1 -1
- package/dist/web/index.html +2 -2
- package/package.json +1 -1
- package/dist/web/assets/index-BGgYMm2u.css +0 -1
- package/dist/web/assets/index-BVVgtntL.js +0 -171
package/dist/adapter-registry.js
CHANGED
|
@@ -73,7 +73,14 @@ export function buildRegistry(deps) {
|
|
|
73
73
|
],
|
|
74
74
|
["openai-image", openAiImage({ fetch: deps.fetch, key: keyOf("openai-image") })],
|
|
75
75
|
["google-image", googleImage({ fetch: deps.fetch, key: keyOf("google-image") })],
|
|
76
|
-
[
|
|
76
|
+
[
|
|
77
|
+
"codex-image",
|
|
78
|
+
codexImage({
|
|
79
|
+
run: cliFor("codex"),
|
|
80
|
+
// The same list, and so the same models and efforts, as the Codex text provider.
|
|
81
|
+
readModels: () => nodeCodexModels(process.env, cliBinary(deps.db, "codex")),
|
|
82
|
+
}),
|
|
83
|
+
],
|
|
77
84
|
]);
|
|
78
85
|
if (deps.hostCli) {
|
|
79
86
|
for (const id of hostLlmIds)
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { redact } from "../../kernel/log.js";
|
|
2
2
|
import { bridgeLimits, hostFaultSchema, hostFrameSchema, hostHealthSchema, hostImageSchema, hostLlmSchema, hostModelsSchema, hostOpenedSchema, hostOpenFolderSchema, hostStatusSchema, } from "../../kernel/ports/host-cli.js";
|
|
3
|
+
import { agentImageTimeoutMs } from "../../kernel/ports/image.js";
|
|
3
4
|
import { isProviderError, providerError } from "../../kernel/ports/model.js";
|
|
4
5
|
import { sniffImage } from "../image/bytes.js";
|
|
5
6
|
import { hostRequest, hostUnavailable, readHostBytes, } from "./transport.js";
|
|
@@ -142,12 +143,22 @@ export function createHostCliClient(options) {
|
|
|
142
143
|
}),
|
|
143
144
|
image: {
|
|
144
145
|
id: "codex-image",
|
|
146
|
+
timeoutMs: agentImageTimeoutMs,
|
|
145
147
|
models: () => models("codex-image"),
|
|
146
148
|
generate: async (req) => {
|
|
147
149
|
const parsed = hostImageSchema.safeParse({
|
|
148
150
|
model: req.model,
|
|
149
151
|
prompt: req.prompt,
|
|
150
152
|
aspect: req.aspect,
|
|
153
|
+
...(req.thinking === undefined ? {} : { thinking: req.thinking }),
|
|
154
|
+
...(req.reference === undefined
|
|
155
|
+
? {}
|
|
156
|
+
: {
|
|
157
|
+
reference: {
|
|
158
|
+
mime: req.reference.mime,
|
|
159
|
+
base64: Buffer.from(req.reference.bytes).toString("base64"),
|
|
160
|
+
},
|
|
161
|
+
}),
|
|
151
162
|
});
|
|
152
163
|
if (!parsed.success)
|
|
153
164
|
throw providerError({
|
|
@@ -3,6 +3,7 @@ import { open } from "node:fs/promises";
|
|
|
3
3
|
import { request } from "node:http";
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
import { bridgeLimits } from "../../kernel/ports/host-cli.js";
|
|
6
|
+
import { agentImageTimeoutMs } from "../../kernel/ports/image.js";
|
|
6
7
|
import { providerError } from "../../kernel/ports/model.js";
|
|
7
8
|
export function hostUnavailable(submitted = false) {
|
|
8
9
|
return providerError({
|
|
@@ -81,7 +82,11 @@ export async function hostRequest(options) {
|
|
|
81
82
|
reject(error);
|
|
82
83
|
};
|
|
83
84
|
const connectTimer = setTimeout(fail, 5000);
|
|
84
|
-
const deadline = setTimeout(fail, options.kind === "image"
|
|
85
|
+
const deadline = setTimeout(fail, options.kind === "image"
|
|
86
|
+
? agentImageTimeoutMs
|
|
87
|
+
: options.kind === "metadata"
|
|
88
|
+
? 35_000
|
|
89
|
+
: 120_000);
|
|
85
90
|
const cleanup = () => {
|
|
86
91
|
clearTimeout(connectTimer);
|
|
87
92
|
clearTimeout(deadline);
|
|
@@ -11,10 +11,64 @@ function unavailable(detail) {
|
|
|
11
11
|
message: `The Codex CLI finished, but Slopify could not collect the image: ${detail}. Use Retry stage to make it again (this uses your Codex quota again).`,
|
|
12
12
|
});
|
|
13
13
|
}
|
|
14
|
+
// A thread that drew more than this is not an image job any more.
|
|
15
|
+
const maxImagesPerThread = 50;
|
|
16
|
+
const imageName = /^[a-zA-Z0-9_-]+\.png$/;
|
|
17
|
+
function imagesRoot(env) {
|
|
18
|
+
return join(resolve(env.CODEX_HOME || join(env.HOME || homedir(), ".codex")), "generated_images");
|
|
19
|
+
}
|
|
20
|
+
function latestImage(directory) {
|
|
21
|
+
const dir = opendirSync(directory);
|
|
22
|
+
let latest;
|
|
23
|
+
let count = 0;
|
|
24
|
+
try {
|
|
25
|
+
for (let entry = dir.readSync(); entry !== null; entry = dir.readSync()) {
|
|
26
|
+
if (++count > maxImagesPerThread)
|
|
27
|
+
throw unavailable("far too many files were saved");
|
|
28
|
+
if (!entry.isFile() || !imageName.test(entry.name))
|
|
29
|
+
throw unavailable("the saved file is not a plain image file");
|
|
30
|
+
const at = lstatSync(join(directory, entry.name)).mtimeMs;
|
|
31
|
+
if (latest === undefined || at > latest.at || (at === latest.at && entry.name > latest.name))
|
|
32
|
+
latest = { name: entry.name, at };
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
finally {
|
|
36
|
+
dir.closeSync();
|
|
37
|
+
}
|
|
38
|
+
if (latest === undefined)
|
|
39
|
+
throw unavailable("no image was saved");
|
|
40
|
+
return latest.name;
|
|
41
|
+
}
|
|
42
|
+
// How many images this job's thread has drawn so far, for its progress line. Never throws:
|
|
43
|
+
// a folder not made yet, or one being written, counts what it can.
|
|
44
|
+
export function codexImageCount(env, threadId) {
|
|
45
|
+
if (!z.uuid().safeParse(threadId).success)
|
|
46
|
+
return 0;
|
|
47
|
+
try {
|
|
48
|
+
const directory = join(imagesRoot(env), threadId);
|
|
49
|
+
const entry = lstatSync(directory);
|
|
50
|
+
if (!entry.isDirectory() || entry.isSymbolicLink())
|
|
51
|
+
return 0;
|
|
52
|
+
const dir = opendirSync(directory);
|
|
53
|
+
let count = 0;
|
|
54
|
+
try {
|
|
55
|
+
for (let one = dir.readSync(); one !== null; one = dir.readSync())
|
|
56
|
+
if (imageName.test(one.name))
|
|
57
|
+
count++;
|
|
58
|
+
}
|
|
59
|
+
finally {
|
|
60
|
+
dir.closeSync();
|
|
61
|
+
}
|
|
62
|
+
return count;
|
|
63
|
+
}
|
|
64
|
+
catch {
|
|
65
|
+
return 0;
|
|
66
|
+
}
|
|
67
|
+
}
|
|
14
68
|
export function codexGeneratedImage(env, threadId, startedAt) {
|
|
15
69
|
if (!z.uuid().safeParse(threadId).success || threadId === undefined)
|
|
16
70
|
throw unavailable("the image job could not be identified");
|
|
17
|
-
const root =
|
|
71
|
+
const root = imagesRoot(env);
|
|
18
72
|
const directory = join(root, threadId);
|
|
19
73
|
try {
|
|
20
74
|
for (const path of [root, directory]) {
|
|
@@ -26,22 +80,10 @@ export function codexGeneratedImage(env, threadId, startedAt) {
|
|
|
26
80
|
if (canonical !== join(realpathSync(root), threadId))
|
|
27
81
|
throw unavailable("Codex's image folder moved while it was being read");
|
|
28
82
|
// Codex 0.155.1 omits image items from exec JSONL. Its artifact contract is
|
|
29
|
-
// generated_images/<thread.started ID>/<sanitized tool call ID>.png.
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
const entry = dir.readSync();
|
|
34
|
-
if (entry === null)
|
|
35
|
-
throw unavailable("no image was saved");
|
|
36
|
-
if (dir.readSync() !== null)
|
|
37
|
-
throw unavailable("more than one file was saved");
|
|
38
|
-
if (!entry.isFile() || !/^[a-zA-Z0-9_-]+\.png$/.test(entry.name))
|
|
39
|
-
throw unavailable("the saved file is not a plain image file");
|
|
40
|
-
name = entry.name;
|
|
41
|
-
}
|
|
42
|
-
finally {
|
|
43
|
-
dir.closeSync();
|
|
44
|
-
}
|
|
83
|
+
// generated_images/<thread.started ID>/<sanitized tool call ID>.png. An agent that reviews
|
|
84
|
+
// its work may draw several in its one thread; the last one it drew is its answer, so the
|
|
85
|
+
// newest file wins (the name breaks a tie, the call IDs being in no useful order).
|
|
86
|
+
const name = latestImage(directory);
|
|
45
87
|
const path = join(directory, name);
|
|
46
88
|
const before = lstatSync(path);
|
|
47
89
|
if (!before.isFile() || before.isSymbolicLink() || realpathSync(path) !== join(canonical, name))
|
|
@@ -1,20 +1,33 @@
|
|
|
1
|
-
import { mkdtempSync, rmSync } from "node:fs";
|
|
1
|
+
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
|
2
2
|
import { tmpdir } from "node:os";
|
|
3
3
|
import { join } from "node:path";
|
|
4
4
|
import { z } from "zod";
|
|
5
5
|
import { redact } from "../../kernel/log.js";
|
|
6
|
+
import { agentImageTimeoutMs, } from "../../kernel/ports/image.js";
|
|
6
7
|
import { providerError } from "../../kernel/ports/model.js";
|
|
7
8
|
import { cliReported, quoted, refusedImage } from "../explain.js";
|
|
8
9
|
import { cliLoginError } from "../llm/cli-login-error.js";
|
|
9
10
|
import { cliEvent, cliShaped, endedWithout, stopCliRun } from "../llm/run-cli.js";
|
|
10
11
|
import { lines } from "../llm/sse-lines.js";
|
|
11
|
-
import { codexGeneratedImage } from "./codex-output.js";
|
|
12
|
+
import { codexGeneratedImage, codexImageCount } from "./codex-output.js";
|
|
13
|
+
// "Codex default" puts no model and no effort on the command line, so the CLI's own defaults
|
|
14
|
+
// draw the image; it is what every project saved before the choice existed runs. Every other
|
|
15
|
+
// model id is one of the Codex CLI's own models - the list the Codex text provider shows -
|
|
16
|
+
// run with the chosen reasoning effort.
|
|
12
17
|
export const codexImageModel = {
|
|
13
18
|
id: "codex-imagegen",
|
|
14
|
-
name: "Codex
|
|
19
|
+
name: "Codex default",
|
|
15
20
|
};
|
|
16
|
-
|
|
17
|
-
|
|
21
|
+
// An agent that reviews and redraws its image works for several minutes at a high effort, so
|
|
22
|
+
// the usual 300 s image limit would cut the best runs off.
|
|
23
|
+
export const codexImageTimeoutMs = agentImageTimeoutMs;
|
|
24
|
+
const maxEventBytes = 4 * 1024 * 1024;
|
|
25
|
+
// Every tool beyond the image tool stays off. `view_image`, which only reads an image file
|
|
26
|
+
// into the conversation, is let back in when the agent is asked to review its work (a chosen
|
|
27
|
+
// model or effort) or has a reference to look at. The bundled imagegen skill stays off with
|
|
28
|
+
// the shell: its workflow runs Python scripts against the Images API with an API key, which
|
|
29
|
+
// this job neither needs nor has, and the built-in tool it wraps is already here.
|
|
30
|
+
const alwaysDisabled = [
|
|
18
31
|
"shell_tool",
|
|
19
32
|
"unified_exec",
|
|
20
33
|
"hooks",
|
|
@@ -25,18 +38,36 @@ const disabledFeatures = [
|
|
|
25
38
|
"multi_agent",
|
|
26
39
|
"computer_use",
|
|
27
40
|
"browser_use",
|
|
28
|
-
"view_image",
|
|
29
41
|
"workspace_dependencies",
|
|
30
42
|
];
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
43
|
+
function reviews(req) {
|
|
44
|
+
return (req.model !== codexImageModel.id || req.thinking !== undefined || req.reference !== undefined);
|
|
45
|
+
}
|
|
46
|
+
// The binary's image tool (`ImagegenArgs`: prompt, referenced_image_paths,
|
|
47
|
+
// num_last_images_to_include) takes each referenced image as an absolute path it reads
|
|
48
|
+
// itself, so the copy sits in the job's own folder - the sandbox's writable, readable root.
|
|
49
|
+
export function codexReferencePath(directory, image) {
|
|
50
|
+
return join(directory, image.mime === "image/jpeg" ? "reference.jpg" : "reference.png");
|
|
51
|
+
}
|
|
52
|
+
export function codexImageInstructions(req, reference) {
|
|
53
|
+
return [
|
|
54
|
+
"You are making one finished image for Slopify, a video creation app, with the image generation tool.",
|
|
55
|
+
"Fidelity to the brief comes first. Before calling the tool, write its prompt yourself as a detailed, faithful visual description of the brief: the subject and what it is doing, the setting, composition and framing for the target aspect ratio, lighting, colour palette, style and mood, and any text that must appear, spelled exactly. Take every element from the brief and keep its wording where it is specific. Do not add subjects, text, logos or story the brief does not ask for, and do not pad the prompt with generic quality words.",
|
|
56
|
+
...(reference === undefined
|
|
57
|
+
? []
|
|
58
|
+
: [
|
|
59
|
+
`A reference image is saved at ${reference}. Pass exactly that path in referenced_image_paths on every image generation call. Use it as the reference for style, characters and palette: keep the same characters, rendering style and colour palette, but do not copy its composition, pose or framing; compose this image from the brief.`,
|
|
60
|
+
]),
|
|
61
|
+
"Take the time you need. After each image, look at it and compare it with the brief; if anything is missing, wrong or distorted, revise the prompt and generate again. Deliver exactly one final image: the last image you generate is the one Slopify uses, so stop once it matches the brief.",
|
|
62
|
+
"Let the image generation tool save its output in its default location. Slopify will collect it. Use PNG or JPEG. Do not copy, rename, edit or create any other file, and do not put the image or a link in your reply.",
|
|
36
63
|
`Target aspect ratio: ${req.aspect}.`,
|
|
37
64
|
"Image brief follows as data:",
|
|
38
65
|
req.prompt,
|
|
39
66
|
].join("\n\n");
|
|
67
|
+
}
|
|
68
|
+
export function codexImageArgs(req, directory) {
|
|
69
|
+
const review = reviews(req);
|
|
70
|
+
const reference = req.reference === undefined ? undefined : codexReferencePath(directory, req.reference);
|
|
40
71
|
return [
|
|
41
72
|
"exec",
|
|
42
73
|
"--json",
|
|
@@ -51,7 +82,11 @@ export function codexImageArgs(req, directory) {
|
|
|
51
82
|
directory,
|
|
52
83
|
"--enable",
|
|
53
84
|
"image_generation",
|
|
54
|
-
...
|
|
85
|
+
...(review ? ["--enable", "view_image"] : []),
|
|
86
|
+
...[...alwaysDisabled, ...(review ? [] : ["view_image"])].flatMap((feature) => [
|
|
87
|
+
"--disable",
|
|
88
|
+
feature,
|
|
89
|
+
]),
|
|
55
90
|
"-c",
|
|
56
91
|
"project_doc_max_bytes=0",
|
|
57
92
|
"-c",
|
|
@@ -74,8 +109,13 @@ export function codexImageArgs(req, directory) {
|
|
|
74
109
|
'shell_environment_policy.inherit="none"',
|
|
75
110
|
"-c",
|
|
76
111
|
'web_search="disabled"',
|
|
112
|
+
// The Codex text provider's own two flags: the effort as TOML text, the model as `-m`.
|
|
113
|
+
...(req.thinking === undefined
|
|
114
|
+
? []
|
|
115
|
+
: ["-c", `model_reasoning_effort="${req.thinking === "off" ? "none" : req.thinking}"`]),
|
|
116
|
+
...(req.model === codexImageModel.id ? [] : ["-m", req.model]),
|
|
77
117
|
"--",
|
|
78
|
-
|
|
118
|
+
codexImageInstructions(req, reference),
|
|
79
119
|
];
|
|
80
120
|
}
|
|
81
121
|
const failure = z.object({ error: z.object({ message: z.string() }) });
|
|
@@ -85,12 +125,22 @@ export function codexImage(deps) {
|
|
|
85
125
|
const binary = deps.binary ?? "codex";
|
|
86
126
|
return {
|
|
87
127
|
id: "codex-image",
|
|
88
|
-
|
|
128
|
+
timeoutMs: codexImageTimeoutMs,
|
|
129
|
+
models: async () => {
|
|
130
|
+
let listed = [];
|
|
131
|
+
try {
|
|
132
|
+
listed = (await deps.readModels?.()) ?? [];
|
|
133
|
+
}
|
|
134
|
+
catch {
|
|
135
|
+
// The default still works without the list.
|
|
136
|
+
}
|
|
137
|
+
return [codexImageModel, ...listed.filter((model) => model.id !== codexImageModel.id)];
|
|
138
|
+
},
|
|
89
139
|
generate: async (req) => {
|
|
90
|
-
if (req.model
|
|
140
|
+
if (req.model.trim() === "" || req.model.startsWith("-"))
|
|
91
141
|
throw providerError({
|
|
92
142
|
kind: "unsupported",
|
|
93
|
-
message: "
|
|
143
|
+
message: "No Codex model is chosen for images. Choose one (or Codex default) for images in the Providers section of Edit project, then use Retry stage.",
|
|
94
144
|
});
|
|
95
145
|
req.signal.throwIfAborted();
|
|
96
146
|
const directory = mkdtempSync(join(tmpdir(), "slopify-codex-image-"));
|
|
@@ -98,7 +148,30 @@ export function codexImage(deps) {
|
|
|
98
148
|
const startedAt = Date.now();
|
|
99
149
|
let threadId;
|
|
100
150
|
let run;
|
|
151
|
+
let reported = -1;
|
|
152
|
+
// A line for the stage's live panel each time this thread's image count moves.
|
|
153
|
+
const report = () => {
|
|
154
|
+
if (req.onProgress === undefined)
|
|
155
|
+
return;
|
|
156
|
+
const count = threadId === undefined ? 0 : codexImageCount(env, threadId);
|
|
157
|
+
if (count === reported)
|
|
158
|
+
return;
|
|
159
|
+
reported = count;
|
|
160
|
+
try {
|
|
161
|
+
req.onProgress(count === 0
|
|
162
|
+
? "Codex is working on the image…"
|
|
163
|
+
: `Codex is refining the image… ${String(count)} ${count === 1 ? "image" : "images"} so far`);
|
|
164
|
+
}
|
|
165
|
+
catch {
|
|
166
|
+
// A progress line is optional; it never fails the image.
|
|
167
|
+
}
|
|
168
|
+
};
|
|
101
169
|
try {
|
|
170
|
+
if (req.reference !== undefined)
|
|
171
|
+
writeFileSync(codexReferencePath(directory, req.reference), req.reference.bytes, {
|
|
172
|
+
mode: 0o600,
|
|
173
|
+
});
|
|
174
|
+
report();
|
|
102
175
|
try {
|
|
103
176
|
run = deps.run(binary, codexImageArgs(req, directory), req.signal, {
|
|
104
177
|
cwd: directory,
|
|
@@ -118,6 +191,9 @@ export function codexImage(deps) {
|
|
|
118
191
|
continue;
|
|
119
192
|
const event = cliEvent(binary, line);
|
|
120
193
|
req.signal.throwIfAborted();
|
|
194
|
+
// Only the whole turn's end counts: an agent that reviews its work draws, looks and
|
|
195
|
+
// draws again, so the first image is rarely its answer.
|
|
196
|
+
report();
|
|
121
197
|
if (event.type === "thread.started") {
|
|
122
198
|
if (threadId !== undefined)
|
|
123
199
|
throw providerError({
|
|
@@ -4,6 +4,7 @@ import { providerError } from "../../kernel/ports/model.js";
|
|
|
4
4
|
import { httpFailure, internalError, missingKey, noImage, providerSaid, unreadable, } from "../explain.js";
|
|
5
5
|
import { retryAfter } from "../retry-after.js";
|
|
6
6
|
import { downloadImage } from "./bytes.js";
|
|
7
|
+
import { withReferenceNote } from "./reference.js";
|
|
7
8
|
import { dataUri, downloadVideo } from "./video.js";
|
|
8
9
|
// The HTTP gateway adapter for fal.ai: `fetch` and the downloader beside this file, no SDK.
|
|
9
10
|
// `@fal-ai/client` would buy queue polling, which only the video clips need and which is three
|
|
@@ -49,6 +50,16 @@ function aspectOf(model, aspect) {
|
|
|
49
50
|
const shape = field ?? "image_size";
|
|
50
51
|
return { [shape]: sizes[shape][aspect] };
|
|
51
52
|
}
|
|
53
|
+
// The text-to-image models whose fal endpoint has an `/edit` twin taking the same fields plus
|
|
54
|
+
// `image_urls`, which is how an establishing image is sent.
|
|
55
|
+
const editModels = [
|
|
56
|
+
"fal-ai/flux-2",
|
|
57
|
+
"fal-ai/nano-banana",
|
|
58
|
+
"fal-ai/nano-banana-2",
|
|
59
|
+
];
|
|
60
|
+
function editEndpoint(model) {
|
|
61
|
+
return editModels.includes(model) ? `${model}/edit` : undefined;
|
|
62
|
+
}
|
|
52
63
|
// Image-to-video runs on fal's queue, not the synchronous host: a clip takes one to five
|
|
53
64
|
// minutes, and fal documents the queue as the way to call anything that slow. A request is
|
|
54
65
|
// submitted, its status polled until it completes, and its answer fetched from the links the
|
|
@@ -89,12 +100,21 @@ export function falImage(deps) {
|
|
|
89
100
|
id: "fal",
|
|
90
101
|
models: () => Promise.resolve(falModels),
|
|
91
102
|
generate: async (req) => {
|
|
92
|
-
|
|
103
|
+
// An establishing image goes to the model's edit endpoint, which takes input images as
|
|
104
|
+
// `image_urls`; only the models that have one are marked as taking a reference.
|
|
105
|
+
const edit = req.reference === undefined ? undefined : editEndpoint(req.model);
|
|
106
|
+
if (req.reference !== undefined && edit === undefined)
|
|
107
|
+
throw providerError({
|
|
108
|
+
kind: "unsupported",
|
|
109
|
+
message: "This fal.ai model can't use an establishing image as a reference. Choose FLUX.2 or Nano Banana 2 under Images → Model on Play or in Edit project → Providers, or set Establishing image to Off in the Images section.",
|
|
110
|
+
});
|
|
111
|
+
const response = await deps.fetch(`${falBase}/${edit ?? req.model}`, {
|
|
93
112
|
method: "POST",
|
|
94
113
|
signal: req.signal,
|
|
95
114
|
headers: { Authorization: `Key ${keyOf(deps)}`, "Content-Type": "application/json" },
|
|
96
115
|
body: JSON.stringify({
|
|
97
|
-
prompt: req.prompt,
|
|
116
|
+
prompt: withReferenceNote(req.prompt, req.reference),
|
|
117
|
+
...(req.reference === undefined ? {} : { image_urls: [dataUri(req.reference)] }),
|
|
98
118
|
...aspectOf(req.model, req.aspect),
|
|
99
119
|
// The stage sends Number as that many independent calls, one piece each.
|
|
100
120
|
num_images: 1,
|
|
@@ -5,6 +5,7 @@ import { httpFailure, missingKey, noImage, quoted, refusedImage, unreadable } fr
|
|
|
5
5
|
import { retryAfter } from "../retry-after.js";
|
|
6
6
|
import { describeBytes, sniffImage } from "./bytes.js";
|
|
7
7
|
import { discoverGoogleImages } from "./models.js";
|
|
8
|
+
import { withReferenceNote } from "./reference.js";
|
|
8
9
|
// The HTTP gateway adapter for Google's own image generation, billed to a Gemini API key
|
|
9
10
|
// rather than to a host reselling the same models. Like the OpenAI one it hands back the
|
|
10
11
|
// bytes with the call, so there is no link to follow.
|
|
@@ -62,7 +63,18 @@ export function googleImage(deps) {
|
|
|
62
63
|
headers: { "x-goog-api-key": keyOf(deps), "Content-Type": "application/json" },
|
|
63
64
|
body: JSON.stringify({
|
|
64
65
|
model: req.model,
|
|
65
|
-
input
|
|
66
|
+
// With an establishing image the input is two parts, the image first; the Gemini
|
|
67
|
+
// image models draw with an input image as their reference.
|
|
68
|
+
input: req.reference === undefined
|
|
69
|
+
? req.prompt
|
|
70
|
+
: [
|
|
71
|
+
{
|
|
72
|
+
type: "image",
|
|
73
|
+
mime_type: req.reference.mime,
|
|
74
|
+
data: Buffer.from(req.reference.bytes).toString("base64"),
|
|
75
|
+
},
|
|
76
|
+
{ type: "text", text: withReferenceNote(req.prompt, req.reference) },
|
|
77
|
+
],
|
|
66
78
|
// The stage sends Number as that many independent calls, one piece each, so one
|
|
67
79
|
// image per request is what it asks for. Nothing about style is set: the stage
|
|
68
80
|
// asks for the provider's own.
|
|
@@ -5,6 +5,7 @@ import { httpFailure, missingKey, noImage, refusedImage, unreadable } from "../e
|
|
|
5
5
|
import { retryAfter } from "../retry-after.js";
|
|
6
6
|
import { describeBytes, sniffImage } from "./bytes.js";
|
|
7
7
|
import { discoverOpenAiImages } from "./models.js";
|
|
8
|
+
import { withReferenceNote } from "./reference.js";
|
|
8
9
|
// The HTTP gateway adapter for OpenAI's images endpoint: the platform's own `fetch` and
|
|
9
10
|
// nothing else, because the whole call is one request. Unlike fal and Replicate this one
|
|
10
11
|
// hands back the image itself - a GPT image model always answers with base64, never a URL -
|
|
@@ -53,21 +54,33 @@ export function openAiImage(deps) {
|
|
|
53
54
|
id: "openai-image",
|
|
54
55
|
models: () => discoverOpenAiImages(deps),
|
|
55
56
|
generate: async (req) => {
|
|
56
|
-
const response =
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
57
|
+
const response = req.reference === undefined
|
|
58
|
+
? await deps.fetch(`${openAiImagesBase}/images/generations`, {
|
|
59
|
+
method: "POST",
|
|
60
|
+
signal: req.signal,
|
|
61
|
+
headers: {
|
|
62
|
+
Authorization: `Bearer ${keyOf(deps)}`,
|
|
63
|
+
"Content-Type": "application/json",
|
|
64
|
+
},
|
|
65
|
+
body: JSON.stringify({
|
|
66
|
+
model: req.model,
|
|
67
|
+
prompt: req.prompt,
|
|
68
|
+
// The stage sends Number as that many independent calls, one piece each, so
|
|
69
|
+
// one image per request is what it asks for.
|
|
70
|
+
n: 1,
|
|
71
|
+
size: sizeFor(req.model, req.aspect),
|
|
72
|
+
// No `quality` and no `background`: the stage asks for the provider's default
|
|
73
|
+
// quality and style, which is what leaving them off means.
|
|
74
|
+
}),
|
|
75
|
+
})
|
|
76
|
+
: // With an establishing image the GPT image models take it as an input image on the
|
|
77
|
+
// edits endpoint, a multipart form; the note says it is a reference, not the canvas.
|
|
78
|
+
await deps.fetch(`${openAiImagesBase}/images/edits`, {
|
|
79
|
+
method: "POST",
|
|
80
|
+
signal: req.signal,
|
|
81
|
+
headers: { Authorization: `Bearer ${keyOf(deps)}` },
|
|
82
|
+
body: editForm(req, req.reference),
|
|
83
|
+
});
|
|
71
84
|
if (!response.ok) {
|
|
72
85
|
throw await failure(response);
|
|
73
86
|
}
|
|
@@ -79,6 +92,15 @@ export function openAiImage(deps) {
|
|
|
79
92
|
},
|
|
80
93
|
};
|
|
81
94
|
}
|
|
95
|
+
function editForm(req, reference) {
|
|
96
|
+
const form = new FormData();
|
|
97
|
+
form.set("model", req.model);
|
|
98
|
+
form.set("prompt", withReferenceNote(req.prompt, reference));
|
|
99
|
+
form.set("n", "1");
|
|
100
|
+
form.set("size", sizeFor(req.model, req.aspect));
|
|
101
|
+
form.append("image[]", new Blob([Uint8Array.from(reference.bytes)], { type: reference.mime }), reference.mime === "image/jpeg" ? "reference.jpg" : "reference.png");
|
|
102
|
+
return form;
|
|
103
|
+
}
|
|
82
104
|
// A GPT image model always answers with base64, so the bytes arrive with the call and
|
|
83
105
|
// there is no link to follow. They are still sniffed: what the port stores is a PNG or a
|
|
84
106
|
// JPEG, and a truncated payload that decodes to something else must not reach the disk.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
// What an image API is told about the establishing image sent with a request, ahead of the
|
|
2
|
+
// brief. An edit endpoint otherwise treats its input as the picture to change, so it says the
|
|
3
|
+
// input is a reference for the look and the cast, not the composition to keep.
|
|
4
|
+
export const referenceNote = "The attached image is a visual reference only: keep its characters, rendering style and colour palette, but do not copy its composition, pose or framing. Compose a new image from this brief:";
|
|
5
|
+
export function withReferenceNote(prompt, reference) {
|
|
6
|
+
return reference === undefined ? prompt : `${referenceNote}\n\n${prompt}`;
|
|
7
|
+
}
|
|
@@ -41,6 +41,13 @@ export function replicateImage(deps) {
|
|
|
41
41
|
id: "replicate",
|
|
42
42
|
models: () => Promise.resolve(replicateModels),
|
|
43
43
|
generate: async (req) => {
|
|
44
|
+
// The Replicate models listed take no input image to draw from (the FLUX image inputs
|
|
45
|
+
// copy the picture's composition), so an establishing image is refused, not dropped.
|
|
46
|
+
if (req.reference !== undefined)
|
|
47
|
+
throw providerError({
|
|
48
|
+
kind: "unsupported",
|
|
49
|
+
message: "Replicate's image models can't use an establishing image as a reference. Choose Codex CLI, OpenAI, Google or fal.ai under Images → Provider on Play or in Edit project → Providers, or set Establishing image to Off in the Images section.",
|
|
50
|
+
});
|
|
44
51
|
const response = await deps.fetch(`${replicateBase}/models/${req.model}/predictions`, {
|
|
45
52
|
method: "POST",
|
|
46
53
|
signal: req.signal,
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { homedir } from "node:os";
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { z } from "zod";
|
|
4
|
+
import { thinkingModes as knownEfforts } from "../../kernel/ports/llm.js";
|
|
4
5
|
import { readCatalogueFile } from "./catalogue-files.js";
|
|
5
6
|
import { codexModelName, nodeCodexServerModels } from "./codex-server-models.js";
|
|
6
7
|
const safeText = z
|
|
@@ -40,7 +41,7 @@ export async function nodeCodexModels(env = process.env, binary = "codex", timeo
|
|
|
40
41
|
if (!unique.has(item.slug)) {
|
|
41
42
|
const thinkingModes = item.supported_reasoning_levels
|
|
42
43
|
?.map((level) => level.effort)
|
|
43
|
-
.filter((level) =>
|
|
44
|
+
.filter((level) => knownEfforts.includes(level));
|
|
44
45
|
unique.set(item.slug, {
|
|
45
46
|
id: item.slug,
|
|
46
47
|
name: codexModelName(item.display_name ?? item.slug),
|
|
@@ -4,6 +4,7 @@ import { tmpdir } from "node:os";
|
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
import { z } from "zod";
|
|
6
6
|
import { cliCommand } from "../../kernel/cli-command.js";
|
|
7
|
+
import { thinkingModes as knownEfforts } from "../../kernel/ports/llm.js";
|
|
7
8
|
const text = z
|
|
8
9
|
.string()
|
|
9
10
|
.trim()
|
|
@@ -102,7 +103,7 @@ export async function nodeCodexServerModels(binary, env, timeoutMs = 15_000) {
|
|
|
102
103
|
continue;
|
|
103
104
|
const thinkingModes = row.supportedReasoningEfforts
|
|
104
105
|
?.map((item) => item.reasoningEffort)
|
|
105
|
-
.filter((effort) =>
|
|
106
|
+
.filter((effort) => knownEfforts.includes(effort));
|
|
106
107
|
models.set(row.model, {
|
|
107
108
|
id: row.model,
|
|
108
109
|
name: codexModelName(row.displayName),
|
|
@@ -87,12 +87,24 @@ export function nodeRunCli(binary, args, signal, options) {
|
|
|
87
87
|
child.stderr?.destroy();
|
|
88
88
|
}
|
|
89
89
|
};
|
|
90
|
-
|
|
90
|
+
// An abort asks politely, then forces after the same grace stopCliRun gives. A reader
|
|
91
|
+
// waiting on stdout only reaches stopCliRun once stdout closes, so a CLI that traps
|
|
92
|
+
// SIGTERM and keeps its pipe open would otherwise outlive the run that was cancelled.
|
|
93
|
+
let forced;
|
|
94
|
+
const abort = () => {
|
|
95
|
+
terminate();
|
|
96
|
+
forced ??= setTimeout(() => terminate(true), cliTerminationGraceMs);
|
|
97
|
+
forced.unref();
|
|
98
|
+
};
|
|
91
99
|
if (signal.aborted)
|
|
92
100
|
abort();
|
|
93
101
|
else
|
|
94
102
|
signal.addEventListener("abort", abort, { once: true });
|
|
95
|
-
void ended.then(() =>
|
|
103
|
+
void ended.then(() => {
|
|
104
|
+
signal.removeEventListener("abort", abort);
|
|
105
|
+
if (forced !== undefined)
|
|
106
|
+
clearTimeout(forced);
|
|
107
|
+
});
|
|
96
108
|
return {
|
|
97
109
|
...(inputWritten === undefined ? {} : { inputWritten }),
|
|
98
110
|
pid: child.pid,
|
package/dist/assets/models.yaml
CHANGED
|
@@ -141,6 +141,7 @@ image:
|
|
|
141
141
|
source: https://developers.openai.com/api/docs/guides/image-generation
|
|
142
142
|
keywords:
|
|
143
143
|
- image
|
|
144
|
+
- reference
|
|
144
145
|
pricing:
|
|
145
146
|
note: Automatic quality and generated image tokens determine the charge; a fixed
|
|
146
147
|
image price is unavailable.
|
|
@@ -154,6 +155,7 @@ image:
|
|
|
154
155
|
source: https://developers.openai.com/api/docs/guides/image-generation
|
|
155
156
|
keywords:
|
|
156
157
|
- image
|
|
158
|
+
- reference
|
|
157
159
|
pricing:
|
|
158
160
|
note: Automatic quality and generated image tokens determine the charge; a fixed
|
|
159
161
|
image price is unavailable.
|
|
@@ -164,6 +166,7 @@ image:
|
|
|
164
166
|
source: https://developers.openai.com/api/docs/guides/image-generation
|
|
165
167
|
keywords:
|
|
166
168
|
- image
|
|
169
|
+
- reference
|
|
167
170
|
pricing:
|
|
168
171
|
note: Automatic quality and generated image tokens determine the charge; a fixed
|
|
169
172
|
image price is unavailable.
|
|
@@ -174,6 +177,7 @@ image:
|
|
|
174
177
|
source: https://ai.google.dev/gemini-api/docs/pricing
|
|
175
178
|
keywords:
|
|
176
179
|
- image
|
|
180
|
+
- reference
|
|
177
181
|
pricing:
|
|
178
182
|
perImage: 0.101
|
|
179
183
|
note: 2K image output price; prompt, thinking and other input tokens are
|
|
@@ -187,6 +191,7 @@ image:
|
|
|
187
191
|
source: https://ai.google.dev/gemini-api/docs/models
|
|
188
192
|
keywords:
|
|
189
193
|
- image
|
|
194
|
+
- reference
|
|
190
195
|
pricing:
|
|
191
196
|
note: Price depends on your account or generated output; estimate unavailable.
|
|
192
197
|
image:
|
|
@@ -198,6 +203,7 @@ image:
|
|
|
198
203
|
source: https://fal.ai/models/fal-ai/flux-2
|
|
199
204
|
keywords:
|
|
200
205
|
- image
|
|
206
|
+
- reference
|
|
201
207
|
pricing:
|
|
202
208
|
note: Price depends on your account or generated output; estimate unavailable.
|
|
203
209
|
image: *a4
|
|
@@ -207,6 +213,7 @@ image:
|
|
|
207
213
|
source: https://fal.ai/models/fal-ai/nano-banana-2
|
|
208
214
|
keywords:
|
|
209
215
|
- image
|
|
216
|
+
- reference
|
|
210
217
|
pricing:
|
|
211
218
|
note: Price depends on your account or generated output; estimate unavailable.
|
|
212
219
|
image: *a4
|