@avocadostudio-ai/orchestrator-core 0.1.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/sites-agent-context.js +3 -2
- package/dist/agent/sites-agent-shared.js +1 -0
- package/dist/chat/anthropic-planner.js +3 -3
- package/dist/chat/chat-pipeline.js +122 -20
- package/dist/chat/gemini-planner.js +3 -3
- package/dist/chat/planner.js +7 -5
- package/dist/chat/prompts.js +6 -1
- package/dist/cms/adapter.d.ts +159 -1
- package/dist/cms/adapter.js +19 -1
- package/dist/cms/bootstrap.d.ts +46 -1
- package/dist/cms/bootstrap.js +126 -2
- package/dist/cms/index.d.ts +3 -2
- package/dist/cms/index.js +2 -1
- package/dist/errors.d.ts +9 -1
- package/dist/handler/auth.d.ts +79 -0
- package/dist/handler/auth.js +113 -0
- package/dist/handler/create-orchestrator.d.ts +205 -0
- package/dist/handler/create-orchestrator.js +1599 -0
- package/dist/http/access-tokens.d.ts +58 -0
- package/dist/http/access-tokens.js +161 -0
- package/dist/http/audio-actions.d.ts +121 -0
- package/dist/http/audio-actions.js +248 -0
- package/dist/http/blocks-actions.d.ts +31 -0
- package/dist/http/blocks-actions.js +31 -0
- package/dist/http/draft-provenance.d.ts +68 -0
- package/dist/http/draft-provenance.js +101 -0
- package/dist/http/history-actions.d.ts +58 -0
- package/dist/http/history-actions.js +169 -0
- package/dist/http/image-generate-actions.d.ts +268 -0
- package/dist/http/image-generate-actions.js +546 -0
- package/dist/http/ops-actions.d.ts +51 -0
- package/dist/http/ops-actions.js +79 -0
- package/dist/http/publish-actions.d.ts +153 -0
- package/dist/http/publish-actions.js +323 -0
- package/dist/http/restore-actions.d.ts +67 -0
- package/dist/http/restore-actions.js +145 -0
- package/dist/http/screenshot-actions.d.ts +108 -0
- package/dist/http/screenshot-actions.js +181 -0
- package/dist/http/session-actions.d.ts +35 -0
- package/dist/http/session-actions.js +98 -0
- package/dist/http/telemetry-feedback-actions.d.ts +53 -0
- package/dist/http/telemetry-feedback-actions.js +68 -0
- package/dist/http/unsplash-actions.d.ts +64 -0
- package/dist/http/unsplash-actions.js +81 -0
- package/dist/http/variations-actions.d.ts +102 -0
- package/dist/http/variations-actions.js +104 -0
- package/dist/index.d.ts +4 -1
- package/dist/index.js +21 -1
- package/dist/nlp/deterministic-planner-refs.d.ts +1 -1
- package/dist/nlp/deterministic-planner-suggestions.d.ts +10 -0
- package/dist/nlp/deterministic-planner-suggestions.js +37 -11
- package/dist/nlp/plan-normalizer.js +18 -2
- package/dist/ops/ops-engine.js +219 -14
- package/dist/state/session-state.d.ts +56 -1
- package/dist/state/session-state.js +92 -6
- package/dist/state/sqlite-store-singleton.d.ts +22 -0
- package/dist/state/sqlite-store-singleton.js +49 -1
- package/dist/state/sqlite-store.d.ts +5 -0
- package/dist/state/sqlite-store.js +125 -2
- package/dist/telemetry/chat-telemetry.js +6 -1
- package/package.json +12 -16
|
@@ -0,0 +1,546 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Image generation — one-shot, multi-turn Gemini chat, and screenshot
|
|
3
|
+
* interpretation — as transport-agnostic actions.
|
|
4
|
+
*
|
|
5
|
+
* These three handlers lived only in `apps/orchestrator/src/routes/media.ts`,
|
|
6
|
+
* wired directly into Fastify, even though almost everything they call has been
|
|
7
|
+
* in orchestrator-core all along. Library mode — `createOrchestrator()` in the
|
|
8
|
+
* site SDK — serves the same editor from a plain `Request`/`Response` handler
|
|
9
|
+
* and reimplements by hand whichever routes somebody remembered to add. It
|
|
10
|
+
* never had these, so the editor's image picker answered "not handled by
|
|
11
|
+
* createOrchestrator()" for a Generate tab it shows unconditionally.
|
|
12
|
+
*
|
|
13
|
+
* Two things made a second hand-kept copy worse than usual here. The multi-turn
|
|
14
|
+
* chat keeps *live* Gemini sessions in a closure-scoped Map, so a duplicate
|
|
15
|
+
* implementation would also duplicate the session store and hand the same
|
|
16
|
+
* `chatId` two different conversations. And the streaming variant has an SSE
|
|
17
|
+
* frame vocabulary — `chatId`/`status`/`text`/`image`/`error`/`done` — that
|
|
18
|
+
* existed only inside the Fastify handler, while the editor parses those names
|
|
19
|
+
* by hand: nothing type-checks a string written on one side of a socket against
|
|
20
|
+
* the string read on the other. So the frames are built here, once, and
|
|
21
|
+
* `formatImageChatFrame` writes the wire bytes.
|
|
22
|
+
*
|
|
23
|
+
* Each function returns the status code and body to send, or emits frames;
|
|
24
|
+
* neither Fastify nor `Response` appears in this file.
|
|
25
|
+
*/
|
|
26
|
+
import { randomUUID } from "node:crypto";
|
|
27
|
+
import { mkdir, writeFile } from "node:fs/promises";
|
|
28
|
+
import { resolve } from "node:path";
|
|
29
|
+
import OpenAI from "openai";
|
|
30
|
+
import { getGeminiClient, getGeminiImageModel, saveGeneratedImage, GEMINI_ASPECT_RATIOS } from "../image/image-helpers.js";
|
|
31
|
+
import { openAIChatOptionsForModel } from "../chat/planner.js";
|
|
32
|
+
import { toErrorDetail } from "../errors.js";
|
|
33
|
+
/**
|
|
34
|
+
* The historical behaviour: `ORCHESTRATOR_GENERATED_IMAGE_DIR` +
|
|
35
|
+
* `ORCHESTRATOR_PUBLIC_ORIGIN`, via the same helper the rest of core uses. This
|
|
36
|
+
* is the default so a caller that passes no store is byte-for-byte unchanged.
|
|
37
|
+
*/
|
|
38
|
+
export function envImageStore() {
|
|
39
|
+
return { save: async (bytes, prefix, ext) => saveGeneratedImage(bytes, prefix, ext) };
|
|
40
|
+
}
|
|
41
|
+
/** A store for a transport that already knows its own directory and public base. */
|
|
42
|
+
export function fileImageStore(options) {
|
|
43
|
+
const base = options.publicBaseUrl.replace(/\/+$/, "");
|
|
44
|
+
return {
|
|
45
|
+
async save(bytes, prefix, ext) {
|
|
46
|
+
const fileName = `${prefix}_${Date.now()}_${randomUUID().slice(0, 8)}.${ext}`;
|
|
47
|
+
await mkdir(options.dir, { recursive: true });
|
|
48
|
+
await writeFile(resolve(options.dir, fileName), bytes);
|
|
49
|
+
return { url: `${base}/${fileName}` };
|
|
50
|
+
}
|
|
51
|
+
};
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* `@google/genai` is an *optional* peer dependency, loaded lazily by
|
|
55
|
+
* `getGeminiClient()`. A library-mode consumer who never asked for Gemini has
|
|
56
|
+
* not installed it, and an unhandled module-resolution throw would surface as a
|
|
57
|
+
* 500 with a stack trace about a missing package. It is a configuration
|
|
58
|
+
* problem, so it gets its own type and comes back as a 503.
|
|
59
|
+
*/
|
|
60
|
+
class GeminiUnavailableError extends Error {
|
|
61
|
+
}
|
|
62
|
+
async function loadGeminiClient(deps) {
|
|
63
|
+
if (deps.gemini)
|
|
64
|
+
return deps.gemini;
|
|
65
|
+
try {
|
|
66
|
+
return (await getGeminiClient());
|
|
67
|
+
}
|
|
68
|
+
catch (error) {
|
|
69
|
+
deps.log.warn({ event: "gemini_sdk_unavailable", error: toErrorDetail(error) }, "@google/genai could not be loaded");
|
|
70
|
+
throw new GeminiUnavailableError("@google/genai is not installed; install it to use Gemini image generation");
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
const storeOf = (deps) => deps.store ?? envImageStore();
|
|
74
|
+
const fetchOf = (deps) => deps.fetchFn ?? fetch;
|
|
75
|
+
/**
|
|
76
|
+
* Save the first inline image in a Gemini part list through the injected store.
|
|
77
|
+
*
|
|
78
|
+
* `saveGeminiInlineImage` in core does the same decode but always writes to the
|
|
79
|
+
* environment-configured directory, which is the one thing that has to be
|
|
80
|
+
* per-transport here.
|
|
81
|
+
*/
|
|
82
|
+
async function saveInlineImage(parts, store, prefix = "gen") {
|
|
83
|
+
for (const part of parts) {
|
|
84
|
+
const data = part.inlineData?.data;
|
|
85
|
+
if (!data)
|
|
86
|
+
continue;
|
|
87
|
+
const bytes = Buffer.from(data, "base64");
|
|
88
|
+
if (bytes.byteLength === 0)
|
|
89
|
+
continue;
|
|
90
|
+
const mimeType = part.inlineData?.mimeType ?? "image/png";
|
|
91
|
+
const ext = mimeType.includes("jpeg") || mimeType.includes("jpg") ? "jpg" : "png";
|
|
92
|
+
const { url } = await store.save(bytes, prefix, ext);
|
|
93
|
+
return { url, mimeType };
|
|
94
|
+
}
|
|
95
|
+
return null;
|
|
96
|
+
}
|
|
97
|
+
/** OpenAI has no aspect-ratio parameter, only sizes; these are the three the picker offers. */
|
|
98
|
+
const OPENAI_ASPECT_SIZES = {
|
|
99
|
+
landscape: "1536x1024",
|
|
100
|
+
square: "1024x1024",
|
|
101
|
+
portrait: "1024x1536"
|
|
102
|
+
};
|
|
103
|
+
/**
|
|
104
|
+
* One image from one prompt, from whichever provider is configured.
|
|
105
|
+
*
|
|
106
|
+
* The provider is chosen per request (`body.provider`) and falls back to
|
|
107
|
+
* `IMAGE_GEN_PROVIDER`, then to OpenAI. Asking for Gemini without a Google key
|
|
108
|
+
* silently falls through to OpenAI rather than failing, because the editor
|
|
109
|
+
* sends a provider hint it inferred from the user's wording and a deployment
|
|
110
|
+
* that funds only one key should still generate images.
|
|
111
|
+
*/
|
|
112
|
+
export async function generateImageAction(body, deps) {
|
|
113
|
+
const hasOpenAI = !!process.env.OPENAI_API_KEY;
|
|
114
|
+
const hasGemini = !!process.env.GOOGLE_GENAI_API_KEY;
|
|
115
|
+
if (!hasOpenAI && !hasGemini) {
|
|
116
|
+
return { code: 503, body: { error: "No image generation API key configured (OPENAI_API_KEY or GOOGLE_GENAI_API_KEY)" } };
|
|
117
|
+
}
|
|
118
|
+
const prompt = typeof body.prompt === "string" ? body.prompt.trim() : "";
|
|
119
|
+
if (!prompt)
|
|
120
|
+
return { code: 400, body: { error: "prompt is required" } };
|
|
121
|
+
const requestedProvider = typeof body.provider === "string" ? body.provider.trim().toLowerCase() : "";
|
|
122
|
+
const requestedModel = typeof body.model === "string" ? body.model.trim() : "";
|
|
123
|
+
const envProvider = process.env.IMAGE_GEN_PROVIDER?.trim().toLowerCase() || "openai";
|
|
124
|
+
const provider = requestedProvider || envProvider;
|
|
125
|
+
const alt = prompt.slice(0, 200);
|
|
126
|
+
const store = storeOf(deps);
|
|
127
|
+
try {
|
|
128
|
+
if (provider === "gemini" && hasGemini) {
|
|
129
|
+
const url = await generateWithGemini({ prompt, aspectRatio: body.aspectRatio, model: requestedModel || undefined }, deps, store);
|
|
130
|
+
if (!url)
|
|
131
|
+
return { code: 502, body: { error: "Gemini image generation returned no data" } };
|
|
132
|
+
return { code: 200, body: { url, alt } };
|
|
133
|
+
}
|
|
134
|
+
if (!hasOpenAI)
|
|
135
|
+
return { code: 503, body: { error: "OPENAI_API_KEY not configured and provider is not gemini" } };
|
|
136
|
+
const size = OPENAI_ASPECT_SIZES[body.aspectRatio ?? "landscape"] ?? OPENAI_ASPECT_SIZES.landscape;
|
|
137
|
+
const model = requestedModel || process.env.OPENAI_IMAGE_MODEL_DRAFT?.trim() || "gpt-image-1-mini";
|
|
138
|
+
const client = deps.openaiImages ?? new OpenAI({ apiKey: process.env.OPENAI_API_KEY });
|
|
139
|
+
const result = await client.images.generate({ model, prompt, size });
|
|
140
|
+
const image = result.data?.[0];
|
|
141
|
+
// gpt-image-* answers with base64; the older URL-returning models need a
|
|
142
|
+
// second fetch, and that URL expires, so the bytes are copied locally
|
|
143
|
+
// either way rather than handed to the editor as a remote link.
|
|
144
|
+
let bytes = null;
|
|
145
|
+
if (typeof image?.b64_json === "string" && image.b64_json.length > 0) {
|
|
146
|
+
bytes = Buffer.from(image.b64_json, "base64");
|
|
147
|
+
}
|
|
148
|
+
else if (typeof image?.url === "string" && image.url.length > 0) {
|
|
149
|
+
const fetched = await fetchOf(deps)(image.url);
|
|
150
|
+
if (fetched.ok)
|
|
151
|
+
bytes = Buffer.from(await fetched.arrayBuffer());
|
|
152
|
+
}
|
|
153
|
+
if (!bytes || bytes.byteLength === 0)
|
|
154
|
+
return { code: 502, body: { error: "Image generation returned no data" } };
|
|
155
|
+
const saved = await store.save(bytes, "gen", "png");
|
|
156
|
+
return { code: 200, body: { url: saved.url, alt } };
|
|
157
|
+
}
|
|
158
|
+
catch (error) {
|
|
159
|
+
if (error instanceof GeminiUnavailableError)
|
|
160
|
+
return { code: 503, body: { error: error.message } };
|
|
161
|
+
return { code: 502, body: { error: "Image generation failed", detail: toErrorDetail(error) } };
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* The single-shot Gemini call, kept local because the generated bytes must go
|
|
166
|
+
* through the injected store — `generateVariationImageWithGemini` in core does
|
|
167
|
+
* the same request but always writes to the environment-configured directory.
|
|
168
|
+
*
|
|
169
|
+
* Failures come back as `null` with a warning rather than a throw, which is
|
|
170
|
+
* what that helper does too, so the caller's 502 body stays the one the editor
|
|
171
|
+
* already handles.
|
|
172
|
+
*/
|
|
173
|
+
async function generateWithGemini(args, deps, store) {
|
|
174
|
+
const ai = await loadGeminiClient(deps);
|
|
175
|
+
const model = args.model ?? getGeminiImageModel();
|
|
176
|
+
const geminiAspectRatio = GEMINI_ASPECT_RATIOS[args.aspectRatio ?? "landscape"] ?? "3:2";
|
|
177
|
+
try {
|
|
178
|
+
const response = await ai.models.generateContent({
|
|
179
|
+
model,
|
|
180
|
+
contents: args.prompt,
|
|
181
|
+
config: { responseModalities: ["TEXT", "IMAGE"], imageConfig: { aspectRatio: geminiAspectRatio, imageSize: "1K" } }
|
|
182
|
+
});
|
|
183
|
+
const parts = response.candidates?.[0]?.content?.parts;
|
|
184
|
+
if (!parts)
|
|
185
|
+
return null;
|
|
186
|
+
const saved = await saveInlineImage(parts, store, "gen");
|
|
187
|
+
return saved?.url ?? null;
|
|
188
|
+
}
|
|
189
|
+
catch (error) {
|
|
190
|
+
deps.log.warn({ event: "gemini_image_error", model, aspectRatio: geminiAspectRatio, error: toErrorDetail(error) }, "Gemini image generation failed");
|
|
191
|
+
return null;
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* Live Gemini chat objects, keyed by the `chatId` handed to the client.
|
|
196
|
+
*
|
|
197
|
+
* Module-scoped rather than per-transport on purpose: a `chatId` is meaningless
|
|
198
|
+
* to anyone but the process that created it, and two stores in one process
|
|
199
|
+
* would let the same id address two different conversations. Nothing here is
|
|
200
|
+
* persisted — a restart drops every session and the client starts a new one,
|
|
201
|
+
* which is why the id travels back on every frame.
|
|
202
|
+
*/
|
|
203
|
+
const imageChatSessions = new Map();
|
|
204
|
+
const MAX_CHAT_SESSIONS = 200;
|
|
205
|
+
const CHAT_SESSION_TTL_MS = 30 * 60 * 1000;
|
|
206
|
+
const CHAT_SWEEP_INTERVAL_MS = 10 * 60 * 1000;
|
|
207
|
+
let sweeper = null;
|
|
208
|
+
/**
|
|
209
|
+
* Started on first use, not at import time: a process that never opens an image
|
|
210
|
+
* chat (every test in this repo, and every deployment without a Google key)
|
|
211
|
+
* should not carry a timer, and `unref` keeps the one we do start from holding
|
|
212
|
+
* the event loop open at shutdown.
|
|
213
|
+
*/
|
|
214
|
+
function ensureSessionSweeper() {
|
|
215
|
+
if (sweeper)
|
|
216
|
+
return;
|
|
217
|
+
sweeper = setInterval(() => {
|
|
218
|
+
const cutoff = Date.now() - CHAT_SESSION_TTL_MS;
|
|
219
|
+
for (const [id, session] of imageChatSessions) {
|
|
220
|
+
if (session.lastUsed < cutoff)
|
|
221
|
+
imageChatSessions.delete(id);
|
|
222
|
+
}
|
|
223
|
+
}, CHAT_SWEEP_INTERVAL_MS);
|
|
224
|
+
sweeper.unref();
|
|
225
|
+
}
|
|
226
|
+
/** Drop every live session and stop the sweeper. Tests only. */
|
|
227
|
+
export function resetImageChatSessionsForTests() {
|
|
228
|
+
imageChatSessions.clear();
|
|
229
|
+
if (sweeper) {
|
|
230
|
+
clearInterval(sweeper);
|
|
231
|
+
sweeper = null;
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
/**
|
|
235
|
+
* Detect an aspect-ratio override in the prompt text.
|
|
236
|
+
*
|
|
237
|
+
* The picker has a ratio control, but people also just type "make it square",
|
|
238
|
+
* and a Gemini chat's `imageConfig` is fixed for the life of the session — so
|
|
239
|
+
* the wording has to be read here, before the session is created, or the
|
|
240
|
+
* request silently produces the previous shape.
|
|
241
|
+
*/
|
|
242
|
+
export function detectAspectRatioFromPrompt(text) {
|
|
243
|
+
const lower = text.toLowerCase();
|
|
244
|
+
const ratioMatch = lower.match(/\b(\d+)\s*:\s*(\d+)\b/);
|
|
245
|
+
if (ratioMatch) {
|
|
246
|
+
const w = Number(ratioMatch[1]);
|
|
247
|
+
const h = Number(ratioMatch[2]);
|
|
248
|
+
if (w > 0 && h > 0)
|
|
249
|
+
return `${w}:${h}`;
|
|
250
|
+
}
|
|
251
|
+
if (/\bsquare\b/.test(lower))
|
|
252
|
+
return "1:1";
|
|
253
|
+
// "portrait"/"landscape" are also ordinary subject words ("a landscape photo
|
|
254
|
+
// of a lake"), so they only count as a ratio next to a shape word.
|
|
255
|
+
if (/\bportrait\b/.test(lower) && /\b(ratio|aspect|orientation|format)\b/.test(lower))
|
|
256
|
+
return "2:3";
|
|
257
|
+
if (/\blandscape\b/.test(lower) && /\b(ratio|aspect|orientation|format)\b/.test(lower))
|
|
258
|
+
return "3:2";
|
|
259
|
+
if (/\bwidescreen\b|\bultra.?wide\b/.test(lower))
|
|
260
|
+
return "16:9";
|
|
261
|
+
return null;
|
|
262
|
+
}
|
|
263
|
+
/**
|
|
264
|
+
* The cheap guards, separate from the actions because the streaming transport
|
|
265
|
+
* has to answer them *before* it commits SSE headers — after that the status
|
|
266
|
+
* code is spent and the only way to say "no key" is a frame the client would
|
|
267
|
+
* have to special-case.
|
|
268
|
+
*/
|
|
269
|
+
export function validateImageChatRequest(body) {
|
|
270
|
+
if (!process.env.GOOGLE_GENAI_API_KEY) {
|
|
271
|
+
return { code: 503, body: { error: "GOOGLE_GENAI_API_KEY is not configured" } };
|
|
272
|
+
}
|
|
273
|
+
const prompt = typeof body.prompt === "string" ? body.prompt.trim() : "";
|
|
274
|
+
if (!prompt)
|
|
275
|
+
return { code: 400, body: { error: "prompt is required" } };
|
|
276
|
+
return null;
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* Find or create the Gemini chat this request belongs to.
|
|
280
|
+
*
|
|
281
|
+
* Ratio precedence is prompt > client hint > the session's current ratio > 3:2.
|
|
282
|
+
* A ratio change on an existing session cannot be applied in place — Gemini
|
|
283
|
+
* freezes `imageConfig` when the chat is created — so the session is replaced
|
|
284
|
+
* and the caller gets a new `chatId`. That loses the conversation, which is the
|
|
285
|
+
* honest trade: keeping the old chat would keep producing the old shape.
|
|
286
|
+
*/
|
|
287
|
+
export function resolveImageChatSession(body, ai, log) {
|
|
288
|
+
const model = getGeminiImageModel();
|
|
289
|
+
const prompt = typeof body.prompt === "string" ? body.prompt.trim() : "";
|
|
290
|
+
const promptRatio = detectAspectRatioFromPrompt(prompt);
|
|
291
|
+
const clientRatio = body.aspectRatio ? (GEMINI_ASPECT_RATIOS[body.aspectRatio] ?? body.aspectRatio) : null;
|
|
292
|
+
const create = (aspectRatio) => {
|
|
293
|
+
const chatId = randomUUID();
|
|
294
|
+
const chat = ai.chats.create({
|
|
295
|
+
model,
|
|
296
|
+
config: { responseModalities: ["TEXT", "IMAGE"], imageConfig: { aspectRatio } }
|
|
297
|
+
});
|
|
298
|
+
// Evicting the least recently used session bounds the memory a long-running
|
|
299
|
+
// orchestrator holds in live SDK objects; the sweeper alone would not, since
|
|
300
|
+
// a burst of 10k sessions all stay under the 30-minute TTL.
|
|
301
|
+
if (imageChatSessions.size >= MAX_CHAT_SESSIONS) {
|
|
302
|
+
let oldestId = "";
|
|
303
|
+
let oldestTime = Infinity;
|
|
304
|
+
for (const [id, session] of imageChatSessions) {
|
|
305
|
+
if (session.lastUsed < oldestTime) {
|
|
306
|
+
oldestTime = session.lastUsed;
|
|
307
|
+
oldestId = id;
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
if (oldestId)
|
|
311
|
+
imageChatSessions.delete(oldestId);
|
|
312
|
+
}
|
|
313
|
+
imageChatSessions.set(chatId, { chat, lastUsed: Date.now(), aspectRatio });
|
|
314
|
+
ensureSessionSweeper();
|
|
315
|
+
return { chatId, chat, aspectRatio };
|
|
316
|
+
};
|
|
317
|
+
const existingId = body.chatId ?? "";
|
|
318
|
+
const existing = existingId ? imageChatSessions.get(existingId) : undefined;
|
|
319
|
+
if (!existing)
|
|
320
|
+
return create(promptRatio ?? clientRatio ?? "3:2");
|
|
321
|
+
const desiredRatio = promptRatio ?? clientRatio ?? existing.aspectRatio;
|
|
322
|
+
if (desiredRatio !== existing.aspectRatio) {
|
|
323
|
+
log.info({ event: "image_chat_aspect_change", from: existing.aspectRatio, to: desiredRatio }, `[image/chat] Aspect ratio changed ${existing.aspectRatio} → ${desiredRatio}, creating new session`);
|
|
324
|
+
imageChatSessions.delete(existingId);
|
|
325
|
+
return create(desiredRatio);
|
|
326
|
+
}
|
|
327
|
+
existing.lastUsed = Date.now();
|
|
328
|
+
return { chatId: existingId, chat: existing.chat, aspectRatio: existing.aspectRatio };
|
|
329
|
+
}
|
|
330
|
+
/**
|
|
331
|
+
* The message to send: the prompt alone, or the prompt plus inlined reference
|
|
332
|
+
* images on the first turn (Gemini accepts up to 14).
|
|
333
|
+
*
|
|
334
|
+
* References are only inlined when the client sent no `chatId`, because after
|
|
335
|
+
* the first turn they are already in the conversation and re-sending them
|
|
336
|
+
* re-pays their tokens on every edit. A reference that will not load is warned
|
|
337
|
+
* about and skipped rather than failing the request: losing one of five moodboard
|
|
338
|
+
* images should not lose the generation.
|
|
339
|
+
*/
|
|
340
|
+
export async function buildImageChatMessage(body, deps) {
|
|
341
|
+
const prompt = typeof body.prompt === "string" ? body.prompt.trim() : "";
|
|
342
|
+
if (body.chatId)
|
|
343
|
+
return prompt;
|
|
344
|
+
const urls = [body.referenceImageUrl, ...(Array.isArray(body.referenceImageUrls) ? body.referenceImageUrls : [])]
|
|
345
|
+
.filter((url) => typeof url === "string" && url.trim().length > 0)
|
|
346
|
+
.slice(0, 14);
|
|
347
|
+
if (urls.length === 0)
|
|
348
|
+
return prompt;
|
|
349
|
+
const doFetch = fetchOf(deps);
|
|
350
|
+
const results = await Promise.allSettled(urls.map(async (url) => {
|
|
351
|
+
const res = await doFetch(url, { signal: AbortSignal.timeout(8000) });
|
|
352
|
+
if (!res.ok)
|
|
353
|
+
throw new Error(`HTTP ${res.status}`);
|
|
354
|
+
const buffer = Buffer.from(await res.arrayBuffer());
|
|
355
|
+
if (buffer.byteLength > 5 * 1024 * 1024)
|
|
356
|
+
throw new Error("Image too large (>5MB)");
|
|
357
|
+
const mimeType = res.headers.get("content-type") ?? "image/png";
|
|
358
|
+
return { inlineData: { mimeType, data: buffer.toString("base64") } };
|
|
359
|
+
}));
|
|
360
|
+
const inlineParts = [];
|
|
361
|
+
for (const result of results) {
|
|
362
|
+
if (result.status === "fulfilled")
|
|
363
|
+
inlineParts.push(result.value);
|
|
364
|
+
else {
|
|
365
|
+
deps.log.warn({ event: "gemini_ref_image_fetch_failed", error: result.reason instanceof Error ? result.reason.message : String(result.reason) }, "Failed to fetch reference image — skipping");
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
if (inlineParts.length === 0)
|
|
369
|
+
return prompt;
|
|
370
|
+
return [{ text: prompt }, ...inlineParts];
|
|
371
|
+
}
|
|
372
|
+
/** The wire encoding of one frame. */
|
|
373
|
+
export function formatImageChatFrame(frame) {
|
|
374
|
+
const { event, ...payload } = frame;
|
|
375
|
+
return `event: ${event}\ndata: ${JSON.stringify(payload)}\n\n`;
|
|
376
|
+
}
|
|
377
|
+
/**
|
|
378
|
+
* One turn of the image chat, answered in a single JSON body.
|
|
379
|
+
*
|
|
380
|
+
* `url` is null when the model replied with text only — a refusal, or a
|
|
381
|
+
* question about the prompt — which is a normal turn, not an error.
|
|
382
|
+
*/
|
|
383
|
+
export async function imageChatAction(body, deps) {
|
|
384
|
+
const invalid = validateImageChatRequest(body);
|
|
385
|
+
if (invalid)
|
|
386
|
+
return invalid;
|
|
387
|
+
const prompt = (body.prompt ?? "").trim();
|
|
388
|
+
const store = storeOf(deps);
|
|
389
|
+
try {
|
|
390
|
+
const ai = await loadGeminiClient(deps);
|
|
391
|
+
const session = resolveImageChatSession(body, ai, deps.log);
|
|
392
|
+
const message = await buildImageChatMessage(body, deps);
|
|
393
|
+
const response = await session.chat.sendMessage({ message });
|
|
394
|
+
const parts = response.candidates?.[0]?.content?.parts;
|
|
395
|
+
if (!parts)
|
|
396
|
+
return { code: 502, body: { error: "Gemini returned no content" } };
|
|
397
|
+
let imageUrl = null;
|
|
398
|
+
let textResponse = "";
|
|
399
|
+
for (const part of parts) {
|
|
400
|
+
if (part.text)
|
|
401
|
+
textResponse += part.text;
|
|
402
|
+
else if (part.inlineData?.data) {
|
|
403
|
+
const saved = await saveInlineImage([part], store, "gen");
|
|
404
|
+
if (saved)
|
|
405
|
+
imageUrl = saved.url;
|
|
406
|
+
}
|
|
407
|
+
}
|
|
408
|
+
return {
|
|
409
|
+
code: 200,
|
|
410
|
+
body: {
|
|
411
|
+
chatId: session.chatId,
|
|
412
|
+
url: imageUrl,
|
|
413
|
+
alt: prompt.slice(0, 200),
|
|
414
|
+
text: textResponse || undefined,
|
|
415
|
+
aspectRatio: session.aspectRatio
|
|
416
|
+
}
|
|
417
|
+
};
|
|
418
|
+
}
|
|
419
|
+
catch (error) {
|
|
420
|
+
if (error instanceof GeminiUnavailableError)
|
|
421
|
+
return { code: 503, body: { error: error.message } };
|
|
422
|
+
return { code: 502, body: { error: "Gemini image generation failed", detail: toErrorDetail(error) } };
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
/**
|
|
426
|
+
* The same turn, streamed.
|
|
427
|
+
*
|
|
428
|
+
* Resolves once the terminal `done` frame has been emitted; ending the stream
|
|
429
|
+
* stays with the transport, which owns the socket and has to close it on client
|
|
430
|
+
* disconnect too. Everything that goes wrong after the first frame — including
|
|
431
|
+
* a failure to reach Gemini at all — is reported as an `error` frame rather
|
|
432
|
+
* than thrown, because the status code is committed the moment the transport
|
|
433
|
+
* writes SSE headers and in-band is the only channel left. Callers should run
|
|
434
|
+
* `validateImageChatRequest` before those headers so a missing key is still a
|
|
435
|
+
* real 503.
|
|
436
|
+
*/
|
|
437
|
+
export async function imageChatStreamAction(body, deps, emit) {
|
|
438
|
+
const prompt = (body.prompt ?? "").trim();
|
|
439
|
+
const alt = prompt.slice(0, 200);
|
|
440
|
+
const store = storeOf(deps);
|
|
441
|
+
let session;
|
|
442
|
+
let message;
|
|
443
|
+
try {
|
|
444
|
+
const ai = await loadGeminiClient(deps);
|
|
445
|
+
session = resolveImageChatSession(body, ai, deps.log);
|
|
446
|
+
message = await buildImageChatMessage(body, deps);
|
|
447
|
+
}
|
|
448
|
+
catch (error) {
|
|
449
|
+
emit({ event: "error", error: toErrorDetail(error) });
|
|
450
|
+
emit({ event: "done" });
|
|
451
|
+
return;
|
|
452
|
+
}
|
|
453
|
+
emit({ event: "chatId", chatId: session.chatId, aspectRatio: session.aspectRatio });
|
|
454
|
+
emit({ event: "status", stage: "Generating image…" });
|
|
455
|
+
try {
|
|
456
|
+
const stream = await session.chat.sendMessageStream({ message });
|
|
457
|
+
for await (const chunk of stream) {
|
|
458
|
+
const parts = chunk.candidates?.[0]?.content?.parts;
|
|
459
|
+
if (!parts)
|
|
460
|
+
continue;
|
|
461
|
+
for (const part of parts) {
|
|
462
|
+
if (part.text)
|
|
463
|
+
emit({ event: "text", text: part.text });
|
|
464
|
+
else if (part.inlineData?.data) {
|
|
465
|
+
const saved = await saveInlineImage([part], store, "gen");
|
|
466
|
+
if (saved)
|
|
467
|
+
emit({ event: "image", url: saved.url, alt });
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
catch (error) {
|
|
473
|
+
emit({ event: "error", error: error instanceof Error ? error.message : String(error) });
|
|
474
|
+
}
|
|
475
|
+
emit({ event: "done" });
|
|
476
|
+
}
|
|
477
|
+
// ---------------------------------------------------------------------------
|
|
478
|
+
// POST /image/interpret
|
|
479
|
+
// ---------------------------------------------------------------------------
|
|
480
|
+
/** What OpenAI vision reads as an `image_url` part. */
|
|
481
|
+
const ALLOWED_IMAGE_ANALYSIS_MIME_TYPES = new Set(["image/png", "image/jpeg", "image/webp", "image/gif"]);
|
|
482
|
+
const MAX_IMAGE_ANALYSIS_BYTES = 10 * 1024 * 1024;
|
|
483
|
+
/**
|
|
484
|
+
* The checks on an uploaded image, shared so both transports answer alike.
|
|
485
|
+
*
|
|
486
|
+
* Each transport decodes multipart its own way — Fastify streams
|
|
487
|
+
* `request.file()`, library mode awaits `request.formData()` — and it was the
|
|
488
|
+
* *error bodies* around that decoding, not the decoding itself, that a second
|
|
489
|
+
* implementation would get subtly different. Returns null when the upload is
|
|
490
|
+
* acceptable.
|
|
491
|
+
*/
|
|
492
|
+
export function validateImageAnalysisInput(input) {
|
|
493
|
+
if (!ALLOWED_IMAGE_ANALYSIS_MIME_TYPES.has(input.mimeType)) {
|
|
494
|
+
return { code: 415, body: { error: `unsupported image type: ${input.mimeType}` } };
|
|
495
|
+
}
|
|
496
|
+
if (input.byteLength > MAX_IMAGE_ANALYSIS_BYTES) {
|
|
497
|
+
return { code: 413, body: { error: "image file is too large (max 10MB)" } };
|
|
498
|
+
}
|
|
499
|
+
if (input.byteLength === 0)
|
|
500
|
+
return { code: 400, body: { error: "image file is empty" } };
|
|
501
|
+
return null;
|
|
502
|
+
}
|
|
503
|
+
/**
|
|
504
|
+
* Describe a pasted screenshot in one sentence, for the chat composer to attach
|
|
505
|
+
* to the user's instruction.
|
|
506
|
+
*
|
|
507
|
+
* The image is inlined as a data URL rather than uploaded: the orchestrator has
|
|
508
|
+
* no public URL for a paste that never touched disk, and these are one-shot
|
|
509
|
+
* reads that nothing refers to again.
|
|
510
|
+
*/
|
|
511
|
+
export async function interpretImageAction(input, deps) {
|
|
512
|
+
if (!process.env.OPENAI_API_KEY)
|
|
513
|
+
return { code: 503, body: { error: "OPENAI_API_KEY is not configured" } };
|
|
514
|
+
const invalid = validateImageAnalysisInput({ mimeType: input.mimeType, byteLength: input.byteLength });
|
|
515
|
+
if (invalid)
|
|
516
|
+
return invalid;
|
|
517
|
+
const model = process.env.OPENAI_VISION_MODEL?.trim() || "gpt-4o";
|
|
518
|
+
const dataUrl = `data:${input.mimeType};base64,${input.bytes.toString("base64")}`;
|
|
519
|
+
const client = deps.openaiVision ?? new OpenAI({ apiKey: process.env.OPENAI_API_KEY });
|
|
520
|
+
try {
|
|
521
|
+
const completion = await client.chat.completions.create({
|
|
522
|
+
model,
|
|
523
|
+
...openAIChatOptionsForModel(model),
|
|
524
|
+
messages: [
|
|
525
|
+
{
|
|
526
|
+
role: "system",
|
|
527
|
+
content: "You interpret pasted screenshots for a website editing assistant. Return one concise sentence describing the most actionable visual/context clue the editor should know. No markdown."
|
|
528
|
+
},
|
|
529
|
+
{
|
|
530
|
+
role: "user",
|
|
531
|
+
content: [
|
|
532
|
+
{ type: "text", text: "Analyze this screenshot and provide concise context for a website edit instruction." },
|
|
533
|
+
{ type: "image_url", image_url: { url: dataUrl } }
|
|
534
|
+
]
|
|
535
|
+
}
|
|
536
|
+
]
|
|
537
|
+
});
|
|
538
|
+
const text = (completion.choices?.[0]?.message?.content ?? "").trim();
|
|
539
|
+
if (!text)
|
|
540
|
+
return { code: 502, body: { error: "image interpretation failed", detail: "No text returned." } };
|
|
541
|
+
return { code: 200, body: { text, model, bytes: input.byteLength, mimeType: input.mimeType } };
|
|
542
|
+
}
|
|
543
|
+
catch (error) {
|
|
544
|
+
return { code: 502, body: { error: "image interpretation failed", detail: toErrorDetail(error) } };
|
|
545
|
+
}
|
|
546
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Validate-only op application, shared by both transports.
|
|
3
|
+
*
|
|
4
|
+
* `POST /ops` with `dryRun: true` is the one request in the API that promises
|
|
5
|
+
* *not* to change anything: it runs every operation against a clone, reports
|
|
6
|
+
* which would apply, skip or fail, and throws the result away. The MCP server's
|
|
7
|
+
* `avocado-dry-run-ops` tool exists to let an agent check a plan before
|
|
8
|
+
* committing to it.
|
|
9
|
+
*
|
|
10
|
+
* Library mode never read the flag. It parsed the body, ignored `dryRun`, and
|
|
11
|
+
* applied the operations for real — so the one call an agent makes precisely
|
|
12
|
+
* because it does not want to mutate the draft was the call that mutated it,
|
|
13
|
+
* and answered `status: "applied"` while doing so. Nothing about the reply
|
|
14
|
+
* suggested a dry run had been asked for.
|
|
15
|
+
*
|
|
16
|
+
* The branch lives here so neither transport can answer that question on its
|
|
17
|
+
* own again.
|
|
18
|
+
*/
|
|
19
|
+
import type { BlockManifest, Operation } from "@avocadostudio-ai/shared";
|
|
20
|
+
import { type SkippedOperation } from "../ops/ops-engine.js";
|
|
21
|
+
import type { ActionResult } from "./history-actions.js";
|
|
22
|
+
export type { ActionResult };
|
|
23
|
+
/**
|
|
24
|
+
* Never mutates state, so it skips undo snapshotting, the version bump and the
|
|
25
|
+
* 404-on-missing-page precheck — a missing page surfaces as a failed op in the
|
|
26
|
+
* results rather than as a status code, which is what a caller previewing a
|
|
27
|
+
* plan wants to see.
|
|
28
|
+
*/
|
|
29
|
+
export declare function opsDryRunAction(session: string, ops: Operation[], componentsManifest?: BlockManifest): Promise<ActionResult>;
|
|
30
|
+
/**
|
|
31
|
+
* Human-readable change lines for a batch of operations that has just applied.
|
|
32
|
+
*
|
|
33
|
+
* Both transports answered every successful `POST /ops` with `changes: []`. It
|
|
34
|
+
* was a placeholder, not a claim — but `changes` is the field the editor renders
|
|
35
|
+
* as the bullet list under an assistant reply, and `/chat` fills it with real
|
|
36
|
+
* lines for exactly the same operations. An empty array on an applied write
|
|
37
|
+
* therefore reads as "this wrote nothing", and an external MCP session drew that
|
|
38
|
+
* inference in as many words.
|
|
39
|
+
*
|
|
40
|
+
* The lines come from the same builders `/chat` uses, so a `Hero` heading edit
|
|
41
|
+
* is described identically whether it arrived through the planner or through a
|
|
42
|
+
* hand-built ops array.
|
|
43
|
+
*
|
|
44
|
+
* `getBlockType` should resolve against the **pre-apply** draft where it can:
|
|
45
|
+
* `remove_block` and `remove_page` name a block that no longer exists by the
|
|
46
|
+
* time this runs, and looking it up afterwards yields "unknown block".
|
|
47
|
+
*/
|
|
48
|
+
export declare function describeAppliedOps(ops: Operation[], ctx: {
|
|
49
|
+
getBlockType: (slug: string, blockId: string) => string | undefined;
|
|
50
|
+
skippedOps?: SkippedOperation[];
|
|
51
|
+
}): string[];
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Validate-only op application, shared by both transports.
|
|
3
|
+
*
|
|
4
|
+
* `POST /ops` with `dryRun: true` is the one request in the API that promises
|
|
5
|
+
* *not* to change anything: it runs every operation against a clone, reports
|
|
6
|
+
* which would apply, skip or fail, and throws the result away. The MCP server's
|
|
7
|
+
* `avocado-dry-run-ops` tool exists to let an agent check a plan before
|
|
8
|
+
* committing to it.
|
|
9
|
+
*
|
|
10
|
+
* Library mode never read the flag. It parsed the body, ignored `dryRun`, and
|
|
11
|
+
* applied the operations for real — so the one call an agent makes precisely
|
|
12
|
+
* because it does not want to mutate the draft was the call that mutated it,
|
|
13
|
+
* and answered `status: "applied"` while doing so. Nothing about the reply
|
|
14
|
+
* suggested a dry run had been asked for.
|
|
15
|
+
*
|
|
16
|
+
* The branch lives here so neither transport can answer that question on its
|
|
17
|
+
* own again.
|
|
18
|
+
*/
|
|
19
|
+
import { applyOpsAtomically, classifyGuardrailError } from "../ops/ops-engine.js";
|
|
20
|
+
import { buildOpChangeLogEntries, buildMetaChangeLogEntries } from "../chat/chat-pipeline-deterministic.js";
|
|
21
|
+
import { toErrorDetail } from "../errors.js";
|
|
22
|
+
/**
|
|
23
|
+
* Never mutates state, so it skips undo snapshotting, the version bump and the
|
|
24
|
+
* 404-on-missing-page precheck — a missing page surfaces as a failed op in the
|
|
25
|
+
* results rather than as a status code, which is what a caller previewing a
|
|
26
|
+
* plan wants to see.
|
|
27
|
+
*/
|
|
28
|
+
export async function opsDryRunAction(session, ops, componentsManifest) {
|
|
29
|
+
try {
|
|
30
|
+
const preview = await applyOpsAtomically(session, ops, { componentsManifest, dryRun: true });
|
|
31
|
+
const opResults = preview.opResults ?? [];
|
|
32
|
+
return {
|
|
33
|
+
code: 200,
|
|
34
|
+
body: {
|
|
35
|
+
status: "preview",
|
|
36
|
+
dryRun: true,
|
|
37
|
+
opResults,
|
|
38
|
+
appliedCount: opResults.filter((r) => r.status === "applied").length,
|
|
39
|
+
skippedCount: opResults.filter((r) => r.status === "skipped").length,
|
|
40
|
+
failedCount: opResults.filter((r) => r.status === "failed").length,
|
|
41
|
+
preview: preview.preview ?? null
|
|
42
|
+
}
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
catch (error) {
|
|
46
|
+
const reason = toErrorDetail(error);
|
|
47
|
+
return { code: 400, body: { error: reason, errorCode: classifyGuardrailError(reason) } };
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* Human-readable change lines for a batch of operations that has just applied.
|
|
52
|
+
*
|
|
53
|
+
* Both transports answered every successful `POST /ops` with `changes: []`. It
|
|
54
|
+
* was a placeholder, not a claim — but `changes` is the field the editor renders
|
|
55
|
+
* as the bullet list under an assistant reply, and `/chat` fills it with real
|
|
56
|
+
* lines for exactly the same operations. An empty array on an applied write
|
|
57
|
+
* therefore reads as "this wrote nothing", and an external MCP session drew that
|
|
58
|
+
* inference in as many words.
|
|
59
|
+
*
|
|
60
|
+
* The lines come from the same builders `/chat` uses, so a `Hero` heading edit
|
|
61
|
+
* is described identically whether it arrived through the planner or through a
|
|
62
|
+
* hand-built ops array.
|
|
63
|
+
*
|
|
64
|
+
* `getBlockType` should resolve against the **pre-apply** draft where it can:
|
|
65
|
+
* `remove_block` and `remove_page` name a block that no longer exists by the
|
|
66
|
+
* time this runs, and looking it up afterwards yields "unknown block".
|
|
67
|
+
*/
|
|
68
|
+
export function describeAppliedOps(ops, ctx) {
|
|
69
|
+
const skipped = ctx.skippedOps ?? [];
|
|
70
|
+
return [
|
|
71
|
+
...buildOpChangeLogEntries(ops, { getBlockType: ctx.getBlockType }),
|
|
72
|
+
...buildMetaChangeLogEntries(ops),
|
|
73
|
+
// A batch where every patch matched the value already stored is the case
|
|
74
|
+
// the empty array used to be mistaken for. Say it rather than imply it.
|
|
75
|
+
...(skipped.length > 0
|
|
76
|
+
? [`Skipped ${skipped.length} unchanged operation${skipped.length === 1 ? "" : "s"}.`]
|
|
77
|
+
: [])
|
|
78
|
+
];
|
|
79
|
+
}
|