@hydraharness/harness-tool-media 0.1.1-rc.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js ADDED
@@ -0,0 +1,1063 @@
1
+ import z from "@hydraharness/schemastery";
2
+ import { admitEncodedImages, toolImageReferences, toolMediaLabels, toolVideoReferences } from "@hydraharness/harness-attachment";
3
+ import { z as z$1 } from "zod";
4
+ import { credentialRef } from "@hydraharness/harness-credentials";
5
+ import { defineTool } from "@hydraharness/harness-tools";
6
+ import { isGenerationRejection } from "@hydraharness/harness-llm";
7
+ //#region lib/types/generated-images.js
8
+ /** Durable attachment admission for decoded provider image results. */
9
+ /** Describe stored images without loading image bytes into model history.
10
+ * @param images - Durable image references available to later media calls.
11
+ * @returns Attachment ids and image properties in a bounded JSON summary.
12
+ */
13
+ function generatedImageSummary(images) {
14
+ return `Image attachments: ${JSON.stringify(images.map(({ attachmentId, mediaType, width, height }) => ({
15
+ attachmentId,
16
+ mediaType,
17
+ width,
18
+ height
19
+ })))}`;
20
+ }
21
+ /** Store each generated image using its position and media type as the filename.
22
+ * @param store - session attachment owner.
23
+ * @param images - encoded final provider images.
24
+ * @param jobId - Optional provider job id for recovery after attachment admission fails.
25
+ * @returns admitted references in provider order.
26
+ */
27
+ function admitGeneratedImages(store, images, jobId) {
28
+ return admitEncodedImages(store, images.map((image, index) => ({
29
+ data: image.data,
30
+ mediaType: image.mimeType,
31
+ name: `generated-${index + 1}.${image.mimeType.slice(6)}`
32
+ }))).catch((error) => {
33
+ if (jobId !== void 0 && jobId !== null && /^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/.test(jobId)) throw new Error(`Image storage failed. jobId=${jobId}. Use gemini_web_retrieve without sending the prompt again.`);
34
+ throw error;
35
+ });
36
+ }
37
+ //#endregion
38
+ //#region lib/types/media-job.js
39
+ /** Non-secret provider job correlation carried by completed media responses. */
40
+ /** Read a validated local recovery id without admitting provider URLs into metadata.
41
+ * @param response - Completed media response.
42
+ * @returns A local UUID, or undefined for providers without saved-task recovery.
43
+ */
44
+ function mediaJobId(response) {
45
+ const id = response.headers.get("x-hydra-job-id");
46
+ return id !== null && /^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/.test(id) ? id : void 0;
47
+ }
48
+ //#endregion
49
+ //#region lib/types/image-openai.js
50
+ /** Loader defaults for the Images API. */
51
+ const OpenAIImageConfig = z.object({
52
+ useProviderModels: z.boolean().default(true),
53
+ provider: z.string(),
54
+ fallbackModels: z.array(z.string()),
55
+ model: z.string().default("gpt-image-1.5"),
56
+ baseURL: z.string().default("https://api.openai.com/v1"),
57
+ apiKeyEnv: z.string().role("credential-ref").default("OPENAI_API_KEY"),
58
+ timeoutMs: z.number().default(18e4),
59
+ maxResponseBytes: z.number().default(32 * 1024 * 1024),
60
+ maxPromptChars: z.number().default(32e3),
61
+ maxImages: z.number().default(4)
62
+ });
63
+ const geminiResponse$1 = z$1.object({ candidates: z$1.array(z$1.object({
64
+ finishReason: z$1.literal("STOP"),
65
+ content: z$1.object({ parts: z$1.array(z$1.object({
66
+ thought: z$1.boolean().optional(),
67
+ inlineData: z$1.object({
68
+ mimeType: z$1.enum([
69
+ "image/png",
70
+ "image/jpeg",
71
+ "image/webp"
72
+ ]),
73
+ data: z$1.string().min(1)
74
+ }).optional()
75
+ })) })
76
+ })).length(1) });
77
+ const chatResponse$1 = z$1.object({ choices: z$1.array(z$1.object({
78
+ finish_reason: z$1.literal("stop"),
79
+ message: z$1.object({ images: z$1.array(z$1.object({ image_url: z$1.object({ url: z$1.string().regex(/^data:image\/(?:png|jpeg|webp);base64,[A-Za-z0-9+/=]+$/) }) })).min(1) })
80
+ })).length(1) }).transform((value) => ({ candidates: value.choices.map((choice) => ({
81
+ finishReason: "STOP",
82
+ content: { parts: choice.message.images.map((image) => {
83
+ const url = image.image_url.url;
84
+ const comma = url.indexOf(",");
85
+ return { inlineData: {
86
+ mimeType: z$1.enum([
87
+ "image/png",
88
+ "image/jpeg",
89
+ "image/webp"
90
+ ]).parse(url.slice(5, comma - 7)),
91
+ data: url.slice(comma + 1)
92
+ } };
93
+ }) }
94
+ })) }));
95
+ const responseSchema$1 = z$1.union([
96
+ z$1.object({ data: z$1.array(z$1.object({ b64_json: z$1.string().min(1) })) }),
97
+ geminiResponse$1,
98
+ z$1.object({ response: geminiResponse$1 }).transform((value) => value.response),
99
+ chatResponse$1
100
+ ]);
101
+ const imageSchema = {
102
+ type: "object",
103
+ additionalProperties: false,
104
+ properties: {
105
+ attachmentId: {
106
+ type: "string",
107
+ required: true
108
+ },
109
+ mediaType: {
110
+ type: "string",
111
+ required: true,
112
+ enum: [
113
+ "image/png",
114
+ "image/jpeg",
115
+ "image/webp",
116
+ "image/gif"
117
+ ]
118
+ },
119
+ bytes: {
120
+ type: "integer",
121
+ required: true,
122
+ minimum: 1
123
+ },
124
+ width: {
125
+ type: "integer",
126
+ required: true,
127
+ minimum: 1
128
+ },
129
+ height: {
130
+ type: "integer",
131
+ required: true,
132
+ minimum: 1
133
+ },
134
+ name: { type: "string" }
135
+ }
136
+ };
137
+ /**
138
+ * Register image_generate over configured Images API routes. Explicit rejections
139
+ * permit fallback; timeouts and ambiguous failures stop to avoid duplicate billing.
140
+ * @param ctx - Tool, credential, and attachment services.
141
+ * @param config - Loader-validated deployment configuration.
142
+ */
143
+ function registerOpenAIImages(ctx, config) {
144
+ const resolved = config;
145
+ for (const field of [
146
+ "timeoutMs",
147
+ "maxResponseBytes",
148
+ "maxPromptChars",
149
+ "maxImages"
150
+ ]) if (!Number.isSafeInteger(resolved[field]) || resolved[field] < 1) throw new TypeError(`tool-media: openai. ${field} must be a positive safe integer`);
151
+ if (resolved.maxImages > 10) throw new TypeError("tool-media: openai. maxImages must not exceed 10");
152
+ if (resolved.model.trim() === "") throw new TypeError("tool-media: openai. model must not be empty");
153
+ if (resolved.fallbackModels.some((model) => model.trim() === "")) throw new TypeError("tool-media: openai. fallbackModels must not contain empty ids");
154
+ const endpoint = new URL(`${resolved.baseURL.replace(/\/$/, "")}/images/generations`);
155
+ if (endpoint.username || endpoint.password || endpoint.search || endpoint.hash || endpoint.protocol !== "https:" && !(endpoint.protocol === "http:" && [
156
+ "localhost",
157
+ "127.0.0.1",
158
+ "[::1]"
159
+ ].includes(endpoint.hostname))) throw new TypeError("tool-media: openai. baseURL must use HTTPS or loopback HTTP without credentials, query, or fragment");
160
+ const ref = credentialRef(resolved.apiKeyEnv);
161
+ const maxImages = Math.min(resolved.maxImages, ctx.attachments.imageLimits.maxImagesPerMessage);
162
+ ctx.tools.register(defineTool({
163
+ name: "image_generate",
164
+ description: "Generate images with the configured image model and display them in chat. Uses the connected provider's quota or API billing. Do not retry a timeout automatically.",
165
+ timeoutMs: resolved.timeoutMs,
166
+ parameters: {
167
+ prompt: {
168
+ type: "string",
169
+ required: true,
170
+ description: "Refine the user request into an image prompt using your current conversation context before calling this tool. Use precise terminology for composition, framing, lighting, materials, and style where the request supports it. Preserve the subject, intent, exact quoted text and its language, counts, and all constraints or exclusions. Do not invent requirements or add conflicting details. If the user requests an exact prompt or no rewriting, pass that prompt verbatim. Send the final generation prompt only, without commentary."
171
+ },
172
+ count: {
173
+ type: "integer",
174
+ minimum: 1,
175
+ maximum: maxImages,
176
+ description: "Number of images; defaults to 1."
177
+ },
178
+ size: {
179
+ type: "string",
180
+ enum: [
181
+ "auto",
182
+ "1024x1024",
183
+ "1536x1024",
184
+ "1024x1536"
185
+ ]
186
+ },
187
+ quality: {
188
+ type: "string",
189
+ enum: [
190
+ "auto",
191
+ "low",
192
+ "medium",
193
+ "high"
194
+ ]
195
+ },
196
+ format: {
197
+ type: "string",
198
+ enum: [
199
+ "png",
200
+ "jpeg",
201
+ "webp"
202
+ ]
203
+ },
204
+ background: {
205
+ type: "string",
206
+ enum: [
207
+ "auto",
208
+ "opaque",
209
+ "transparent"
210
+ ]
211
+ }
212
+ },
213
+ output: {
214
+ schema: {
215
+ type: "object",
216
+ additionalProperties: false,
217
+ properties: {
218
+ model: {
219
+ type: "string",
220
+ required: true
221
+ },
222
+ provider: { type: "string" },
223
+ jobId: { type: "string" },
224
+ images: {
225
+ type: "array",
226
+ required: true,
227
+ items: imageSchema
228
+ }
229
+ }
230
+ },
231
+ render: (_args, value) => [{
232
+ type: "text",
233
+ text: `Generated ${value.images.length} image(s) with ${value.provider ?? "OpenAI"} (${value.model}).\n${generatedImageSummary(value.images)}`
234
+ }],
235
+ presentationMeta: (_args, value) => ({
236
+ kind: "tool-images",
237
+ ...value
238
+ })
239
+ },
240
+ presentCall: (args) => ({
241
+ card: "media",
242
+ kind: "image",
243
+ title: "Generate image",
244
+ prompt: args.prompt,
245
+ ...args.size === void 0 || args.size === "auto" ? {} : { aspectRatio: {
246
+ "1024x1024": 1,
247
+ "1536x1024": 1.5,
248
+ "1024x1536": 2 / 3
249
+ }[args.size] }
250
+ }),
251
+ presentResult: (_args, result) => result.isError ? void 0 : {
252
+ card: "media",
253
+ kind: "image",
254
+ title: "Generated image",
255
+ content: toolImageReferences(result.meta).map((attachment) => ({
256
+ type: "image",
257
+ attachment
258
+ })),
259
+ ...toolMediaLabels(result.meta)
260
+ },
261
+ async execute(args, execution) {
262
+ if (execution.parent !== void 0) throw new Error("image_generate requires Native tool mode for chat presentation.");
263
+ if (args.prompt.trim() === "" || args.prompt.length > resolved.maxPromptChars) throw new TypeError(`Image prompt must contain 1–${resolved.maxPromptChars} characters.`);
264
+ const request = {
265
+ model: resolved.model,
266
+ prompt: args.prompt,
267
+ n: args.count ?? 1,
268
+ size: args.size ?? "auto",
269
+ quality: args.quality ?? "auto",
270
+ output_format: args.format ?? "png",
271
+ background: args.background ?? "auto"
272
+ };
273
+ if (request.output_format === "jpeg" && request.background === "transparent") throw new TypeError("Transparent backgrounds require PNG or WebP.");
274
+ const signal = AbortSignal.any([execution.signal, AbortSignal.timeout(resolved.timeoutMs)]);
275
+ const generated = resolved.useProviderModels ? await ctx.get("llm")?.generateMedia({
276
+ endpoint: "images/generations",
277
+ body: request,
278
+ model: resolved.model,
279
+ signal,
280
+ maxResponseBytes: resolved.maxResponseBytes,
281
+ ...config.provider === void 0 ? {} : { provider: config.provider }
282
+ }) : void 0;
283
+ if (config.provider !== void 0 && generated === void 0) throw new Error(`Provider ${config.provider} has no configured image generation models.`);
284
+ let response;
285
+ if (generated !== void 0) {
286
+ response = generated.response;
287
+ request.model = generated.model;
288
+ } else {
289
+ const credential = await ctx.credentials.resolve(ref);
290
+ if (credential === void 0) throw new Error(`Configure ${resolved.apiKeyEnv} or an image model with credentials on the Models page before generating images.`);
291
+ const post = (model) => {
292
+ signal.throwIfAborted();
293
+ request.model = model;
294
+ return fetch(endpoint, {
295
+ method: "POST",
296
+ redirect: "error",
297
+ signal,
298
+ headers: {
299
+ "Content-Type": "application/json",
300
+ Authorization: `Bearer ${credential.value}`
301
+ },
302
+ body: JSON.stringify(request)
303
+ });
304
+ };
305
+ response = await post(resolved.model);
306
+ for (const model of new Set(resolved.fallbackModels)) {
307
+ if (model === resolved.model) continue;
308
+ if (!isGenerationRejection(response.status)) break;
309
+ await response.body?.cancel();
310
+ response = await post(model);
311
+ }
312
+ }
313
+ if (!response.ok) {
314
+ await response.body?.cancel();
315
+ throw new Error(`OpenAI image request failed (HTTP ${response.status}). Check credentials, model access, and account limits before retrying.`);
316
+ }
317
+ let bytes = 0;
318
+ const body = response.body?.pipeThrough(new TransformStream({ transform(chunk, controller) {
319
+ bytes += chunk.byteLength;
320
+ if (bytes > resolved.maxResponseBytes) throw new Error("OpenAI image response exceeds maxResponseBytes.");
321
+ controller.enqueue(chunk);
322
+ } }));
323
+ const value = responseSchema$1.parse(await new Response(body).json());
324
+ signal.throwIfAborted();
325
+ const output = "data" in value ? value.data.map((image) => ({
326
+ data: image.b64_json,
327
+ mimeType: `image/${request.output_format}`
328
+ })) : value.candidates.flatMap((candidate) => candidate.content.parts.flatMap((part) => "thought" in part && part.thought || part.inlineData === void 0 ? [] : [part.inlineData]));
329
+ if (output.length !== request.n) throw new Error("Image provider returned an unexpected image count.");
330
+ const images = await admitGeneratedImages(ctx.attachments, output, response.headers.get("x-hydra-job-id"));
331
+ signal.throwIfAborted();
332
+ const jobId = mediaJobId(response);
333
+ return {
334
+ model: request.model,
335
+ images: [...images],
336
+ ...generated === void 0 ? {} : { provider: generated.provider },
337
+ ...jobId === void 0 ? {} : { jobId }
338
+ };
339
+ }
340
+ }));
341
+ }
342
+ //#endregion
343
+ //#region lib/types/image-google.js
344
+ /** Loader defaults for Gemini image generation. */
345
+ const GoogleImageConfig = z.object({
346
+ useProviderModels: z.boolean().default(true),
347
+ provider: z.string(),
348
+ model: z.string().default("gemini-3.1-flash-image"),
349
+ baseURL: z.string().default("https://generativelanguage.googleapis.com/v1"),
350
+ apiKeyEnv: z.string().role("credential-ref").default("GEMINI_API_KEY"),
351
+ timeoutMs: z.number().default(18e4),
352
+ maxResponseBytes: z.number().default(32 * 1024 * 1024),
353
+ maxPromptChars: z.number().default(32e3),
354
+ maxImages: z.number().default(4)
355
+ });
356
+ const geminiResponse = z$1.object({ candidates: z$1.array(z$1.object({
357
+ finishReason: z$1.string(),
358
+ content: z$1.object({ parts: z$1.array(z$1.object({
359
+ thought: z$1.boolean().optional(),
360
+ inlineData: z$1.object({
361
+ mimeType: z$1.enum([
362
+ "image/png",
363
+ "image/jpeg",
364
+ "image/webp"
365
+ ]),
366
+ data: z$1.string().min(1)
367
+ }).optional()
368
+ })) }).optional()
369
+ })).max(1).optional() });
370
+ const chatResponse = z$1.object({ choices: z$1.array(z$1.object({
371
+ finish_reason: z$1.literal("stop"),
372
+ message: z$1.object({ images: z$1.array(z$1.object({ image_url: z$1.object({ url: z$1.string().regex(/^data:image\/(?:png|jpeg|webp);base64,[A-Za-z0-9+/=]+$/) }) })).min(1) })
373
+ })).length(1) }).transform((value) => ({ candidates: value.choices.map((choice) => ({
374
+ finishReason: "STOP",
375
+ content: { parts: choice.message.images.map((image) => {
376
+ const url = image.image_url.url;
377
+ const comma = url.indexOf(",");
378
+ return { inlineData: {
379
+ mimeType: z$1.enum([
380
+ "image/png",
381
+ "image/jpeg",
382
+ "image/webp"
383
+ ]).parse(url.slice(5, comma - 7)),
384
+ data: url.slice(comma + 1)
385
+ } };
386
+ }) }
387
+ })) }));
388
+ const responseSchema = z$1.union([
389
+ z$1.object({ data: z$1.array(z$1.object({ b64_json: z$1.string().min(1) })) }),
390
+ z$1.object({ response: geminiResponse }).transform((value) => value.response),
391
+ chatResponse,
392
+ geminiResponse
393
+ ]);
394
+ /**
395
+ * Register image_generate_google with final images in presentation metadata.
396
+ * Each call makes one stateless generateContent request without automatic retries.
397
+ * @param ctx - Tool, credential, and attachment services.
398
+ * @param config - Loader-validated deployment configuration.
399
+ */
400
+ function registerGoogleImages(ctx, config) {
401
+ const resolved = config;
402
+ for (const field of [
403
+ "timeoutMs",
404
+ "maxResponseBytes",
405
+ "maxPromptChars",
406
+ "maxImages"
407
+ ]) if (!Number.isSafeInteger(resolved[field]) || resolved[field] < 1) throw new TypeError(`tool-media: google. ${field} must be a positive safe integer`);
408
+ if (!/^[a-zA-Z0-9][a-zA-Z0-9._-]*$/.test(resolved.model)) throw new TypeError("tool-media: google. model must be a bare Gemini model id");
409
+ const endpoint = new URL(`${resolved.baseURL.replace(/\/$/, "")}/models/${resolved.model}:generateContent`);
410
+ if (endpoint.username || endpoint.password || endpoint.search || endpoint.hash || endpoint.protocol !== "https:" && !(endpoint.protocol === "http:" && [
411
+ "localhost",
412
+ "127.0.0.1",
413
+ "[::1]"
414
+ ].includes(endpoint.hostname))) throw new TypeError("tool-media: google. baseURL must use HTTPS or loopback HTTP without credentials, query, or fragment");
415
+ const ref = credentialRef(resolved.apiKeyEnv);
416
+ const maxImages = Math.min(resolved.maxImages, ctx.attachments.imageLimits.maxImagesPerMessage);
417
+ ctx.tools.register(defineTool({
418
+ name: "image_generate_google",
419
+ description: "Generate images with Google Gemini and display them in chat. Uses the connected provider's quota or API billing. Do not retry a timeout automatically.",
420
+ timeoutMs: resolved.timeoutMs,
421
+ parameters: { prompt: {
422
+ type: "string",
423
+ required: true,
424
+ description: "Refine the user request into an image prompt using your current conversation context before calling this tool. Use precise terminology for composition, framing, lighting, materials, and style where the request supports it. Preserve the subject, intent, exact quoted text and its language, counts, and all constraints or exclusions. Do not invent requirements or add conflicting details. If the user requests an exact prompt or no rewriting, pass that prompt verbatim. Send the final generation prompt only, without commentary."
425
+ } },
426
+ output: {
427
+ schema: {
428
+ type: "object",
429
+ additionalProperties: false,
430
+ properties: {
431
+ model: {
432
+ type: "string",
433
+ required: true
434
+ },
435
+ provider: { type: "string" },
436
+ jobId: { type: "string" },
437
+ images: {
438
+ type: "array",
439
+ required: true,
440
+ items: {
441
+ type: "object",
442
+ additionalProperties: false,
443
+ properties: {
444
+ attachmentId: {
445
+ type: "string",
446
+ required: true
447
+ },
448
+ mediaType: {
449
+ type: "string",
450
+ required: true,
451
+ enum: [
452
+ "image/png",
453
+ "image/jpeg",
454
+ "image/webp",
455
+ "image/gif"
456
+ ]
457
+ },
458
+ bytes: {
459
+ type: "integer",
460
+ required: true,
461
+ minimum: 1
462
+ },
463
+ width: {
464
+ type: "integer",
465
+ required: true,
466
+ minimum: 1
467
+ },
468
+ height: {
469
+ type: "integer",
470
+ required: true,
471
+ minimum: 1
472
+ },
473
+ name: { type: "string" }
474
+ }
475
+ }
476
+ }
477
+ }
478
+ },
479
+ render: (_args, value) => [{
480
+ type: "text",
481
+ text: `Generated ${value.images.length} image(s) with ${value.provider ?? "Google"} (${value.model}).\n${generatedImageSummary(value.images)}`
482
+ }],
483
+ presentationMeta: (_args, value) => ({
484
+ kind: "tool-images",
485
+ ...value
486
+ })
487
+ },
488
+ presentCall: (args) => ({
489
+ card: "media",
490
+ kind: "image",
491
+ title: "Generate image with Google",
492
+ prompt: args.prompt
493
+ }),
494
+ presentResult: (_args, result) => result.isError ? void 0 : {
495
+ card: "media",
496
+ kind: "image",
497
+ title: "Generated image",
498
+ content: toolImageReferences(result.meta).map((attachment) => ({
499
+ type: "image",
500
+ attachment
501
+ })),
502
+ ...toolMediaLabels(result.meta)
503
+ },
504
+ async execute(args, execution) {
505
+ if (execution.parent !== void 0) throw new Error("image_generate_google requires Native tool mode for chat presentation.");
506
+ if (args.prompt.trim() === "" || args.prompt.length > resolved.maxPromptChars) throw new TypeError(`Image prompt must contain 1–${resolved.maxPromptChars} characters.`);
507
+ const request = {
508
+ contents: [{
509
+ role: "user",
510
+ parts: [{ text: args.prompt }]
511
+ }],
512
+ generationConfig: { responseModalities: ["IMAGE"] }
513
+ };
514
+ const signal = AbortSignal.any([execution.signal, AbortSignal.timeout(resolved.timeoutMs)]);
515
+ const generated = resolved.useProviderModels ? await ctx.get("llm")?.generateMedia({
516
+ endpoint: "images/generations",
517
+ body: {
518
+ prompt: args.prompt,
519
+ n: 1,
520
+ output_format: "png"
521
+ },
522
+ model: resolved.model,
523
+ signal,
524
+ maxResponseBytes: resolved.maxResponseBytes,
525
+ ...config.provider === void 0 ? {} : { provider: config.provider }
526
+ }) : void 0;
527
+ if (config.provider !== void 0 && generated === void 0) throw new Error(`Provider ${config.provider} has no configured image generation models.`);
528
+ let response;
529
+ if (generated !== void 0) response = generated.response;
530
+ else {
531
+ const credential = await ctx.credentials.resolve(ref);
532
+ if (credential === void 0) throw new Error(`Configure ${resolved.apiKeyEnv} or an image model on the Models page before generating images.`);
533
+ response = await fetch(endpoint, {
534
+ method: "POST",
535
+ redirect: "error",
536
+ signal,
537
+ headers: {
538
+ "Content-Type": "application/json",
539
+ "x-goog-api-key": credential.value
540
+ },
541
+ body: JSON.stringify(request)
542
+ });
543
+ }
544
+ if (!response.ok) {
545
+ await response.body?.cancel();
546
+ throw new Error(`Google image request failed (HTTP ${response.status}). Check credentials, model access, and account limits before retrying.`);
547
+ }
548
+ let bytes = 0;
549
+ const body = response.body?.pipeThrough(new TransformStream({ transform(chunk, controller) {
550
+ bytes += chunk.byteLength;
551
+ if (bytes > resolved.maxResponseBytes) throw new Error("Google image response exceeds maxResponseBytes.");
552
+ controller.enqueue(chunk);
553
+ } }));
554
+ const value = responseSchema.parse(await new Response(body).json());
555
+ signal.throwIfAborted();
556
+ const output = (() => {
557
+ if ("data" in value) return value.data.map((image) => ({
558
+ data: image.b64_json,
559
+ mimeType: "image/png"
560
+ }));
561
+ const candidate = value.candidates?.[0];
562
+ if (candidate?.finishReason !== "STOP" || candidate.content === void 0) throw new Error("Google did not return a complete image response. Check the prompt and model access before retrying.");
563
+ return candidate.content.parts.flatMap((part) => "thought" in part && part.thought || part.inlineData === void 0 ? [] : [part.inlineData]);
564
+ })();
565
+ if (output.length === 0 || output.length > maxImages) throw new Error(`Google must return 1–${maxImages} final images.`);
566
+ const images = await admitGeneratedImages(ctx.attachments, output, response.headers.get("x-hydra-job-id"));
567
+ signal.throwIfAborted();
568
+ const jobId = mediaJobId(response);
569
+ return {
570
+ model: generated?.model ?? resolved.model,
571
+ images: [...images],
572
+ ...generated === void 0 ? {} : { provider: generated.provider },
573
+ ...jobId === void 0 ? {} : { jobId }
574
+ };
575
+ }
576
+ }));
577
+ }
578
+ //#endregion
579
+ //#region lib/types/generated-video.js
580
+ /** Persist a completed video without retaining its stream in the transcript.
581
+ * @param store - Durable attachment owner.
582
+ * @param response - Completed MP4 or WebM response.
583
+ * @param maxBytes - Deployment byte limit.
584
+ * @param signal - Cancellation covering download and storage.
585
+ * @returns A durable video reference after every byte has been stored.
586
+ */
587
+ async function admitGeneratedVideo(store, response, maxBytes, signal) {
588
+ try {
589
+ const mediaType = response.headers.get("content-type")?.split(";")[0]?.trim();
590
+ if (mediaType !== "video/mp4" && mediaType !== "video/webm") {
591
+ await response.body?.cancel();
592
+ throw new Error("Video provider did not return a completed MP4 or WebM video.");
593
+ }
594
+ if (response.body === null) throw new Error("Video provider returned no video bytes.");
595
+ const stream = response.body;
596
+ async function* data() {
597
+ let bytes = 0;
598
+ const reader = stream.getReader();
599
+ try {
600
+ while (true) {
601
+ signal.throwIfAborted();
602
+ const chunk = await reader.read();
603
+ if (chunk.done) break;
604
+ bytes += chunk.value.byteLength;
605
+ if (bytes > maxBytes) throw new Error("Video response exceeds maxVideoBytes.");
606
+ yield chunk.value;
607
+ }
608
+ if (bytes === 0) throw new Error("Video provider returned an empty video.");
609
+ } finally {
610
+ try {
611
+ await reader.cancel();
612
+ } finally {
613
+ reader.releaseLock();
614
+ }
615
+ }
616
+ }
617
+ const video = await store.saveFileStream({
618
+ data: data(),
619
+ name: `generated-video.${mediaType === "video/mp4" ? "mp4" : "webm"}`,
620
+ signal
621
+ });
622
+ signal.throwIfAborted();
623
+ return {
624
+ ...video,
625
+ mediaType
626
+ };
627
+ } catch (error) {
628
+ const id = mediaJobId(response);
629
+ if (id !== void 0) throw new Error(`Video storage failed. jobId=${id}. Use gemini_web_retrieve to recover this job without sending the prompt again.`);
630
+ throw error;
631
+ }
632
+ }
633
+ //#endregion
634
+ //#region lib/types/video.js
635
+ /** Loader defaults for video polling and storage. */
636
+ const VideoConfig = z.object({
637
+ provider: z.string(),
638
+ model: z.string(),
639
+ timeoutMs: z.number().default(9e5),
640
+ pollIntervalMs: z.number().default(1e4),
641
+ maxResponseBytes: z.number().default(1024 * 1024),
642
+ maxVideoBytes: z.number().default(100 * 1024 * 1024),
643
+ maxPromptChars: z.number().default(32e3)
644
+ });
645
+ /** Register video_generate using the saved video provider and model.
646
+ * @param ctx - Tool and attachment services, with optional configured LLM routes.
647
+ * @param config - Loader-validated polling and storage limits.
648
+ */
649
+ function registerVideos(ctx, config) {
650
+ const resolved = config;
651
+ for (const field of [
652
+ "timeoutMs",
653
+ "pollIntervalMs",
654
+ "maxResponseBytes",
655
+ "maxVideoBytes",
656
+ "maxPromptChars"
657
+ ]) if (!Number.isSafeInteger(resolved[field]) || resolved[field] < 1) throw new TypeError(`tool-media: video.${field} must be a positive safe integer`);
658
+ if (resolved.timeoutMs > 2147483647 || resolved.pollIntervalMs > 2147483647) throw new TypeError("tool-media: video timer limits must not exceed 2147483647");
659
+ if (config.model?.trim() === "" || config.provider?.trim() === "") throw new TypeError("tool-media: video model and provider must not be empty");
660
+ ctx.tools.register(defineTool({
661
+ name: "video_generate",
662
+ description: "Generate a video with the configured video model and display it in chat. Uses the connected provider's quota or API billing. Do not retry a timeout automatically.",
663
+ timeoutMs: resolved.timeoutMs,
664
+ parameters: {
665
+ prompt: {
666
+ type: "string",
667
+ required: true,
668
+ description: "Refine the user request into a video prompt using your current conversation context before calling this tool. Use precise terminology for shot framing, camera movement, subject motion, timing, lighting, and style where the request supports it. Preserve the subject, intent, exact quoted text and its language, counts, duration, aspect ratio, and all constraints or exclusions. Keep the prompt consistent with seconds and size when supplied. Do not invent requirements or add conflicting details. If the user requests an exact prompt or no rewriting, pass that prompt verbatim. Send the final generation prompt only, without commentary."
669
+ },
670
+ seconds: {
671
+ type: "integer",
672
+ minimum: 1,
673
+ maximum: 15,
674
+ description: "Duration in seconds; omit unless the user requests it. Gemini Web uses its fixed default duration."
675
+ },
676
+ size: {
677
+ type: "string",
678
+ enum: ["1280x720", "720x1280"],
679
+ description: "Landscape or portrait video; omitted uses the provider default."
680
+ },
681
+ reference_image_ids: {
682
+ type: "array",
683
+ items: { type: "string" },
684
+ description: "Image attachmentIds from earlier user uploads or image-generation results in this conversation. Required when animating an existing image. Use \"latest\" for the last image when an older result did not include its id. Currently supported by Gemini Web."
685
+ }
686
+ },
687
+ output: {
688
+ schema: {
689
+ type: "object",
690
+ additionalProperties: false,
691
+ properties: {
692
+ provider: {
693
+ type: "string",
694
+ required: true
695
+ },
696
+ model: {
697
+ type: "string",
698
+ required: true
699
+ },
700
+ jobId: { type: "string" },
701
+ videos: {
702
+ type: "array",
703
+ required: true,
704
+ items: {
705
+ type: "object",
706
+ additionalProperties: false,
707
+ properties: {
708
+ attachmentId: {
709
+ type: "string",
710
+ required: true
711
+ },
712
+ name: {
713
+ type: "string",
714
+ required: true
715
+ },
716
+ bytes: {
717
+ type: "integer",
718
+ required: true,
719
+ minimum: 1
720
+ },
721
+ mediaType: {
722
+ type: "string",
723
+ required: true,
724
+ enum: ["video/mp4", "video/webm"]
725
+ }
726
+ }
727
+ }
728
+ }
729
+ }
730
+ },
731
+ render: (_args, value) => [{
732
+ type: "text",
733
+ text: `Generated video with ${value.provider} (${value.model}).`
734
+ }],
735
+ presentationMeta: (_args, value) => ({
736
+ kind: "tool-videos",
737
+ ...value
738
+ })
739
+ },
740
+ presentCall: (args) => ({
741
+ card: "media",
742
+ kind: "video",
743
+ title: "Generate video",
744
+ prompt: args.prompt,
745
+ ...args.size === void 0 ? {} : { aspectRatio: args.size === "1280x720" ? 16 / 9 : 9 / 16 }
746
+ }),
747
+ presentResult: (_args, result) => result.isError ? void 0 : {
748
+ card: "media",
749
+ kind: "video",
750
+ title: "Generated video",
751
+ content: toolVideoReferences(result.meta).map((attachment) => ({
752
+ type: "video",
753
+ attachment
754
+ })),
755
+ ...toolMediaLabels(result.meta)
756
+ },
757
+ async execute(args, execution) {
758
+ if (execution.parent !== void 0) throw new Error("video_generate requires Native tool mode for chat presentation.");
759
+ if (args.prompt.trim() === "" || args.prompt.length > resolved.maxPromptChars) throw new TypeError(`Video prompt must contain 1–${resolved.maxPromptChars} characters.`);
760
+ const signal = AbortSignal.any([execution.signal, AbortSignal.timeout(resolved.timeoutMs)]);
761
+ const referenceImages = [];
762
+ if (args.reference_image_ids !== void 0) {
763
+ if (args.reference_image_ids.length < 1 || args.reference_image_ids.length > 5 || new Set(args.reference_image_ids).size !== args.reference_image_ids.length) throw new TypeError("Supply one to five distinct reference image ids.");
764
+ const session = execution.agent?.session;
765
+ if (session === void 0) throw new Error("Reference images require the calling conversation.");
766
+ const refs = [];
767
+ for (const event of session.activeEvents) if (event.type === "user/message") {
768
+ for (const block of event.data.content) if (block.type === "image") refs.push(block.attachment);
769
+ } else if (event.type === "tool/result") {
770
+ refs.push(...toolImageReferences(event.data.meta));
771
+ for (const content of event.data.message.content[0].content) if (content.type === "image") refs.push(content.attachment);
772
+ }
773
+ for (const id of args.reference_image_ids) {
774
+ const ref = id === "latest" ? refs.at(-1) : refs.find((ref) => ref.attachmentId === id);
775
+ if (ref === void 0) throw new Error("Reference image is absent from this conversation. Upload it or use an image attachmentId from an earlier result.");
776
+ const stored = await ctx.attachments.readImage(ref, signal);
777
+ referenceImages.push({
778
+ data: stored.data,
779
+ mimeType: ref.mediaType
780
+ });
781
+ }
782
+ }
783
+ const generated = await ctx.get("llm")?.generateMedia({
784
+ endpoint: "videos",
785
+ body: {
786
+ prompt: args.prompt,
787
+ ...args.seconds === void 0 ? {} : { seconds: args.seconds },
788
+ ...args.size === void 0 ? {} : { size: args.size },
789
+ ...args.reference_image_ids === void 0 ? {} : { referenceImages }
790
+ },
791
+ ...config.provider === void 0 ? {} : { provider: config.provider },
792
+ ...config.model === void 0 ? {} : { model: config.model },
793
+ signal,
794
+ maxResponseBytes: resolved.maxResponseBytes,
795
+ pollIntervalMs: resolved.pollIntervalMs
796
+ });
797
+ if (generated === void 0) throw new Error("Configure a video model and its credentials on the Models page before generating videos.");
798
+ const videos = [await admitGeneratedVideo(ctx.attachments, generated.response, resolved.maxVideoBytes, signal)];
799
+ const jobId = mediaJobId(generated.response);
800
+ return {
801
+ provider: generated.provider,
802
+ model: generated.model,
803
+ videos,
804
+ ...jobId === void 0 ? {} : { jobId }
805
+ };
806
+ }
807
+ }));
808
+ }
809
+ //#endregion
810
+ //#region lib/types/gemini-web-retrieve.js
811
+ /** Register read-only task recovery with the existing media presentation protocol.
812
+ * @param ctx - Tools and attachment services, with optional LLM routes.
813
+ * @param image - Image response and prompt-independent attachment limits.
814
+ * @param video - Video polling, response, and attachment limits.
815
+ */
816
+ function registerGeminiWebRetrieve(ctx, image, video) {
817
+ const maxImageBytes = image.maxResponseBytes ?? 32 * 1024 * 1024;
818
+ const maxVideoBytes = video.maxVideoBytes ?? 100 * 1024 * 1024;
819
+ const timeoutMs = video.timeoutMs ?? 9e5;
820
+ ctx.tools.register(defineTool({
821
+ name: "gemini_web_retrieve",
822
+ description: "Retrieve an existing Gemini Web image or video using the jobId from an earlier error. Displays saved media in chat and never sends a new generation prompt. Reconnect Gemini Web first if the account session expired.",
823
+ timeoutMs,
824
+ parameters: {
825
+ jobId: {
826
+ type: "string",
827
+ required: true,
828
+ description: "Local jobId returned by the Gemini Web provider."
829
+ },
830
+ kind: {
831
+ type: "string",
832
+ required: true,
833
+ enum: ["image", "video"],
834
+ description: "Media kind of the original task."
835
+ }
836
+ },
837
+ output: {
838
+ schema: {
839
+ type: "object",
840
+ additionalProperties: false,
841
+ properties: {
842
+ provider: {
843
+ type: "string",
844
+ required: true
845
+ },
846
+ model: {
847
+ type: "string",
848
+ required: true
849
+ },
850
+ jobId: {
851
+ type: "string",
852
+ required: true
853
+ },
854
+ kind: {
855
+ type: "string",
856
+ enum: ["image", "video"],
857
+ required: true
858
+ },
859
+ images: {
860
+ type: "array",
861
+ required: true,
862
+ items: {
863
+ type: "object",
864
+ additionalProperties: false,
865
+ properties: {
866
+ attachmentId: {
867
+ type: "string",
868
+ required: true
869
+ },
870
+ name: { type: "string" },
871
+ bytes: {
872
+ type: "integer",
873
+ required: true
874
+ },
875
+ mediaType: {
876
+ type: "string",
877
+ required: true,
878
+ enum: [
879
+ "image/png",
880
+ "image/jpeg",
881
+ "image/webp",
882
+ "image/gif"
883
+ ]
884
+ },
885
+ width: {
886
+ type: "integer",
887
+ required: true
888
+ },
889
+ height: {
890
+ type: "integer",
891
+ required: true
892
+ }
893
+ }
894
+ }
895
+ },
896
+ videos: {
897
+ type: "array",
898
+ required: true,
899
+ items: {
900
+ type: "object",
901
+ additionalProperties: false,
902
+ properties: {
903
+ attachmentId: {
904
+ type: "string",
905
+ required: true
906
+ },
907
+ name: {
908
+ type: "string",
909
+ required: true
910
+ },
911
+ bytes: {
912
+ type: "integer",
913
+ required: true
914
+ },
915
+ mediaType: {
916
+ type: "string",
917
+ required: true,
918
+ enum: ["video/mp4", "video/webm"]
919
+ }
920
+ }
921
+ }
922
+ }
923
+ }
924
+ },
925
+ render: (_args, value) => [{
926
+ type: "text",
927
+ text: `Retrieved Gemini Web ${value.kind}. jobId=${value.jobId}.` + (value.kind === "image" ? `\n${generatedImageSummary(value.images)}` : "")
928
+ }],
929
+ presentationMeta: (_args, value) => value.kind === "image" ? {
930
+ kind: "tool-images",
931
+ images: value.images,
932
+ provider: value.provider,
933
+ model: value.model,
934
+ jobId: value.jobId
935
+ } : {
936
+ kind: "tool-videos",
937
+ videos: value.videos,
938
+ provider: value.provider,
939
+ model: value.model,
940
+ jobId: value.jobId
941
+ }
942
+ },
943
+ presentCall: (args) => ({
944
+ card: "media",
945
+ kind: args.kind,
946
+ title: "Retrieve Gemini Web media",
947
+ prompt: `Job ${args.jobId}`
948
+ }),
949
+ presentResult: (args, result) => result.isError ? void 0 : {
950
+ card: "media",
951
+ kind: args.kind,
952
+ title: "Retrieved Gemini Web media",
953
+ content: args.kind === "image" ? toolImageReferences(result.meta).map((attachment) => ({
954
+ type: "image",
955
+ attachment
956
+ })) : toolVideoReferences(result.meta).map((attachment) => ({
957
+ type: "video",
958
+ attachment
959
+ })),
960
+ ...toolMediaLabels(result.meta)
961
+ },
962
+ async execute(args, execution) {
963
+ if (execution.parent !== void 0) throw new Error("gemini_web_retrieve requires Native tool mode for chat presentation.");
964
+ if (!/^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/.test(args.jobId)) throw new TypeError("Invalid Gemini Web jobId.");
965
+ const signal = AbortSignal.any([execution.signal, AbortSignal.timeout(timeoutMs)]);
966
+ const generated = await ctx.get("llm")?.generateMedia({
967
+ provider: "gemini-web",
968
+ model: `gemini-web-${args.kind}`,
969
+ endpoint: args.kind === "image" ? "images/generations" : "videos",
970
+ body: { jobId: args.jobId },
971
+ signal,
972
+ maxResponseBytes: args.kind === "image" ? maxImageBytes : video.maxResponseBytes ?? 1024 * 1024,
973
+ pollIntervalMs: video.pollIntervalMs ?? 1e4
974
+ });
975
+ if (generated === void 0) throw new Error("Apply Gemini Web in Account sign-in before retrieving media.");
976
+ const response = generated.response;
977
+ if (args.kind === "image") {
978
+ const reader = response.body?.getReader();
979
+ if (reader === void 0) throw new Error("Gemini Web returned no image bytes.");
980
+ const chunks = [];
981
+ let bytes = 0;
982
+ try {
983
+ while (true) {
984
+ signal.throwIfAborted();
985
+ const chunk = await reader.read();
986
+ if (chunk.done) break;
987
+ bytes += chunk.value.byteLength;
988
+ if (bytes > maxImageBytes) throw new Error("Gemini Web image exceeds the response limit.");
989
+ chunks.push(chunk.value);
990
+ }
991
+ } finally {
992
+ try {
993
+ await reader.cancel();
994
+ } finally {
995
+ reader.releaseLock();
996
+ }
997
+ }
998
+ const inline = z$1.object({ candidates: z$1.tuple([z$1.object({
999
+ finishReason: z$1.literal("STOP"),
1000
+ content: z$1.object({ parts: z$1.tuple([z$1.object({ inlineData: z$1.object({
1001
+ data: z$1.string().min(1),
1002
+ mimeType: z$1.enum([
1003
+ "image/png",
1004
+ "image/jpeg",
1005
+ "image/webp"
1006
+ ])
1007
+ }) })]) })
1008
+ })]) }).parse(JSON.parse(Buffer.concat(chunks).toString("utf8"))).candidates[0].content.parts[0].inlineData;
1009
+ const images = [...await admitGeneratedImages(ctx.attachments, [{
1010
+ data: inline.data,
1011
+ mimeType: inline.mimeType
1012
+ }], args.jobId)];
1013
+ signal.throwIfAborted();
1014
+ return {
1015
+ provider: generated.provider,
1016
+ model: generated.model,
1017
+ kind: args.kind,
1018
+ jobId: args.jobId,
1019
+ images,
1020
+ videos: []
1021
+ };
1022
+ }
1023
+ const videos = [await admitGeneratedVideo(ctx.attachments, response, maxVideoBytes, signal)];
1024
+ return {
1025
+ provider: generated.provider,
1026
+ model: generated.model,
1027
+ kind: args.kind,
1028
+ jobId: args.jobId,
1029
+ images: [],
1030
+ videos
1031
+ };
1032
+ }
1033
+ }));
1034
+ }
1035
+ //#endregion
1036
+ //#region lib/types/index.js
1037
+ /** Cordis plugin name. */
1038
+ const name = "tool-media";
1039
+ /** Services consumed by image and video generation. */
1040
+ const inject = [
1041
+ "tools",
1042
+ "credentials",
1043
+ "attachments"
1044
+ ];
1045
+ /** Loader defaults for all media tools. */
1046
+ const Config = z.object({
1047
+ openai: OpenAIImageConfig.default({}),
1048
+ google: GoogleImageConfig.default({}),
1049
+ video: VideoConfig.default({})
1050
+ });
1051
+ /** Register all media tools in the same fiber so one switch disposes them together.
1052
+ * @param ctx - Tool, credential, and attachment services.
1053
+ * @param config - Loader-validated settings for each generation API.
1054
+ */
1055
+ function apply(ctx, config) {
1056
+ const resolved = config;
1057
+ registerOpenAIImages(ctx, resolved.openai);
1058
+ registerGoogleImages(ctx, resolved.google);
1059
+ registerVideos(ctx, resolved.video);
1060
+ registerGeminiWebRetrieve(ctx, resolved.google, resolved.video);
1061
+ }
1062
+ //#endregion
1063
+ export { Config, apply, inject, name };