privateer-agent 0.10.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,554 @@
1
+ // Media generation — images, video, speech, and music, billed to the signed-in
2
+ // Privateer account and written straight to disk as files the agent can then work on.
3
+ //
4
+ // WHY THROUGH THE ACCOUNT. Same reasoning as web.ts: the only secret in the process is
5
+ // the user's own session token. No provider key sits in the environment for a
6
+ // prompt-injected run to read out, and every call inherits the account's entitlement,
7
+ // daily caps, credit balance, and — the part that matters most here — its ZDR posture.
8
+ // The server refuses to route a ZDR account's media to a retaining model unless the
9
+ // user has explicitly opted into non-ZDR media (`ZDR_MEDIA_BLOCKED`), and these tools
10
+ // surface that refusal verbatim rather than papering over it.
11
+ //
12
+ // WHAT THIS COSTS, HONESTLY. Generation is NOT end-to-end encrypted and cannot be. The
13
+ // prompt, any input image, and the finished bytes pass through Privateer's servers in
14
+ // plaintext on the way to and from the model provider — that is what generation IS.
15
+ // What we do control: nothing is persisted server-side. The bytes come back inline and
16
+ // land only in the file you name. Never describe a routine that generates media as
17
+ // private end-to-end; do say the output isn't stored in our cloud.
18
+ //
19
+ // MUSIC IS THE LOOSEST OF THESE. Neither Lyria SKU has a zero-retention endpoint and no
20
+ // confidential music model exists, so music is deliberately exempt from the ZDR gate
21
+ // (the server sends it unattributed as the mitigation). The tool description says so,
22
+ // because a model choosing between "narrate this" and "score this" should know the
23
+ // difference in posture before it picks.
24
+ //
25
+ // SHAPE. Every tool takes an explicit output `path` and returns that path. That is not
26
+ // bookkeeping: it makes the permission gate meaningful (a media call classifies as a
27
+ // write against a named file, see permissions/classify.ts), and it gives the NEXT step
28
+ // in a workflow — video_compose, send_file_to_client, a bash ffmpeg call — something
29
+ // concrete to consume. A workflow is then just: generate frames → animate them →
30
+ // stitch → score → send.
31
+
32
+ import { Type } from "typebox";
33
+ import { mkdirSync, readFileSync, writeFileSync, existsSync, statSync } from "node:fs";
34
+ import { dirname, extname, isAbsolute, resolve } from "node:path";
35
+ import { apiRequest } from "../auth/privateer.ts";
36
+
37
+ /** Tool names these definitions register, for allow-list construction. */
38
+ export const MEDIA_TOOL_NAMES = [
39
+ "generate_image",
40
+ "generate_video",
41
+ "generate_speech",
42
+ "generate_music",
43
+ "media_capabilities",
44
+ ] as const;
45
+
46
+ // A video job can legitimately take minutes. Bound the wait so a wedged provider
47
+ // doesn't pin an unattended run forever; the job id is reported on timeout so the
48
+ // caller can resume the poll rather than pay for another generation.
49
+ const VIDEO_POLL_TIMEOUT_MS = Number(process.env.PRIVATEER_VIDEO_TIMEOUT_MS) || 12 * 60_000;
50
+ const VIDEO_POLL_INTERVAL_MS = 5_000;
51
+ // Bound what we'll upload as an input frame/reference. The server enforces its own
52
+ // ceiling; failing here first turns a 413 into a clear, local message.
53
+ const MAX_INPUT_IMAGE_BYTES = 8 * 1024 * 1024;
54
+
55
+ function text(t: string) {
56
+ return { content: [{ type: "text", text: t }], details: {} };
57
+ }
58
+
59
+ const IMAGE_MIME: Record<string, string> = {
60
+ ".png": "image/png", ".jpg": "image/jpeg", ".jpeg": "image/jpeg",
61
+ ".webp": "image/webp", ".gif": "image/gif",
62
+ };
63
+
64
+ function mimeForImage(path: string): string {
65
+ return IMAGE_MIME[extname(path).toLowerCase()] ?? "image/jpeg";
66
+ }
67
+
68
+ function extForMime(mimeType: string, fallback: string): string {
69
+ const m = (mimeType || "").toLowerCase();
70
+ if (m.includes("png")) return ".png";
71
+ if (m.includes("jpeg") || m.includes("jpg")) return ".jpg";
72
+ if (m.includes("webp")) return ".webp";
73
+ if (m.includes("mp4")) return ".mp4";
74
+ if (m.includes("webm")) return ".webm";
75
+ if (m.includes("mpeg") || m.includes("mp3")) return ".mp3";
76
+ if (m.includes("wav")) return ".wav";
77
+ if (m.includes("ogg")) return ".ogg";
78
+ return fallback;
79
+ }
80
+
81
+ function abs(cwd: string, p: string): string {
82
+ return isAbsolute(p) ? p : resolve(cwd, p);
83
+ }
84
+
85
+ // Write bytes to `path`, creating parent directories. Returns a short human summary.
86
+ function writeOut(target: string, bytes: Buffer): string {
87
+ mkdirSync(dirname(target), { recursive: true });
88
+ writeFileSync(target, bytes);
89
+ const kb = bytes.length / 1024;
90
+ return `${target} (${kb >= 1024 ? `${(kb / 1024).toFixed(1)} MB` : `${Math.round(kb)} KB`})`;
91
+ }
92
+
93
+ // Read a local image and return it in the wire shape the server expects.
94
+ function readInputImage(cwd: string, p: string): { data: string; mimeType: string } {
95
+ const target = abs(cwd, p);
96
+ if (!existsSync(target)) throw new Error(`input image not found: ${p}`);
97
+ const stat = statSync(target);
98
+ if (stat.isDirectory()) throw new Error(`${p} is a directory, not an image`);
99
+ if (stat.size === 0) throw new Error(`${p} is empty`);
100
+ if (stat.size > MAX_INPUT_IMAGE_BYTES) {
101
+ throw new Error(`${p} is ${(stat.size / 1048576).toFixed(1)} MB; the limit for an input image is ${MAX_INPUT_IMAGE_BYTES / 1048576} MB`);
102
+ }
103
+ return { data: readFileSync(target).toString("base64"), mimeType: mimeForImage(target) };
104
+ }
105
+
106
+ interface AccountFailure {
107
+ ok: false;
108
+ message: string;
109
+ }
110
+
111
+ /**
112
+ * Call the account API and return the parsed payload, or a message written for the
113
+ * model to read. Errors are surfaced rather than swallowed — an agent that quietly
114
+ * "generated" nothing and moved on is worse than one that says the account is out of
115
+ * credit. The server's own messages are already written for a person, so prefer them.
116
+ */
117
+ async function callAccount<T>(
118
+ path: string,
119
+ init: { method: "GET" | "POST"; body?: unknown; signal?: AbortSignal },
120
+ ): Promise<({ ok: true } & { data: T }) | AccountFailure> {
121
+ let res: Response;
122
+ try {
123
+ res = await apiRequest(path, {
124
+ method: init.method,
125
+ ...(init.body === undefined
126
+ ? {}
127
+ : { headers: { "Content-Type": "application/json" }, body: JSON.stringify(init.body) }),
128
+ ...(init.signal ? { signal: init.signal } : {}),
129
+ });
130
+ } catch (e) {
131
+ return { ok: false, message: `could not reach Privateer: ${e instanceof Error ? e.message : String(e)}` };
132
+ }
133
+
134
+ if (res.ok || res.status === 202) {
135
+ try {
136
+ return { ok: true, data: (await res.json()) as T };
137
+ } catch {
138
+ return { ok: false, message: "Privateer returned a malformed response" };
139
+ }
140
+ }
141
+
142
+ let code = "";
143
+ let serverMessage = "";
144
+ try {
145
+ const err = (await res.json()) as { code?: string; message?: string; error?: { code?: string; message?: string } };
146
+ code = String(err?.code ?? err?.error?.code ?? "");
147
+ serverMessage = String(err?.message ?? err?.error?.message ?? "");
148
+ } catch {
149
+ /* non-JSON body — fall through to a status-based message */
150
+ }
151
+
152
+ // The ZDR block is a deliberate policy answer, not a failure, and it is ACTIONABLE
153
+ // by the user (not by the model) — say exactly which switch to change and stop.
154
+ if (code === "ZDR_MEDIA_BLOCKED") {
155
+ return {
156
+ ok: false,
157
+ message:
158
+ (serverMessage || "this account requires Zero Data Retention and the media model has no ZDR endpoint") +
159
+ " — this is a privacy setting only the account owner can change (Settings → Privacy), so do not retry.",
160
+ };
161
+ }
162
+ if (code === "ZDR_KEY_UNAVAILABLE") {
163
+ return { ok: false, message: serverMessage || "no zero-retention provider key is available right now — try again later" };
164
+ }
165
+ if (/DAILY_CAP|LIMIT_REACHED/i.test(code) || res.status === 429) {
166
+ return { ok: false, message: serverMessage || "the account's daily media allowance is used up — it resets tomorrow" };
167
+ }
168
+ if (res.status === 402 || /INSUFFICIENT|QUOTA|TOP_?UP/i.test(code)) {
169
+ return { ok: false, message: serverMessage || "the account is out of credit for media generation — top up or upgrade to continue" };
170
+ }
171
+ if (res.status === 401 || res.status === 403) {
172
+ return {
173
+ ok: false,
174
+ message:
175
+ serverMessage ||
176
+ "this agent is not signed in to a Privateer account (or the plan doesn't include this), so it cannot generate media",
177
+ };
178
+ }
179
+ if (res.status === 400 || res.status === 413) {
180
+ return { ok: false, message: serverMessage || `Privateer rejected the request${code ? ` (${code})` : ""}` };
181
+ }
182
+ return { ok: false, message: serverMessage || `media generation failed (HTTP ${res.status}${code ? ` ${code}` : ""})` };
183
+ }
184
+
185
+ // ── Images ───────────────────────────────────────────────────────────────────
186
+
187
+ interface ImageResponse {
188
+ model?: string;
189
+ images?: { data?: string; mimeType?: string }[];
190
+ }
191
+
192
+ export const generateImageToolDefinition = {
193
+ name: "generate_image",
194
+ label: "Generate Image",
195
+ description:
196
+ "Generate an image from a text prompt and save it to disk. Optionally pass `images` — paths to " +
197
+ "images already on disk — to EDIT or COMPOSE instead: one input means 'change this picture per the " +
198
+ "prompt', several means 'combine these into one'. Billed to the user's Privateer account and " +
199
+ "subject to its privacy settings; the prompt and any input images pass through Privateer's servers " +
200
+ "in plaintext, but nothing is stored there — the only copy is the file you name. Use the resulting " +
201
+ "path as the first frame of generate_video, or as slideshow material for video_compose.",
202
+ parameters: Type.Object({
203
+ prompt: Type.String({ description: "What the image should show. Be specific about subject, style, lighting and framing." }),
204
+ path: Type.String({ description: "Where to write the image, relative to cwd or absolute (e.g. 'frames/opening.png')." }),
205
+ images: Type.Optional(
206
+ Type.Array(Type.String(), {
207
+ description:
208
+ "Paths to existing images to edit or combine. One = edit that image; several = use the first as the base " +
209
+ "and incorporate the rest. Omit to generate from the prompt alone.",
210
+ }),
211
+ ),
212
+ count: Type.Optional(Type.Number({ description: "How many variations to produce, 1-4. Above 1, files are suffixed -1, -2, … Defaults to 1." })),
213
+ aspectRatio: Type.Optional(Type.String({ description: "Aspect ratio, e.g. '16:9', '9:16', '1:1'. Defaults to the model's own." })),
214
+ size: Type.Optional(Type.String({ description: "Explicit pixel size if the model supports one, e.g. '1024x1024'." })),
215
+ model: Type.Optional(Type.String({ description: "Override the account's image model (e.g. 'google/gemini-3.1-flash-image'). Leave unset to use the account default." })),
216
+ }),
217
+ async execute(
218
+ _toolCallId: string,
219
+ params: { prompt: string; path: string; images?: string[]; count?: number; aspectRatio?: string; size?: string; model?: string },
220
+ signal?: AbortSignal,
221
+ _onUpdate?: unknown,
222
+ ctx?: { cwd?: string },
223
+ ) {
224
+ const cwd = ctx?.cwd ?? process.cwd();
225
+ const prompt = String(params.prompt ?? "").trim();
226
+ if (!prompt) return text("Error: prompt is required.");
227
+ if (!params.path) return text("Error: path is required — say where to save the image.");
228
+
229
+ let inputs: { data: string; mimeType: string }[];
230
+ try {
231
+ inputs = (params.images ?? []).map((p) => readInputImage(cwd, p));
232
+ } catch (e) {
233
+ return text(`Error: ${e instanceof Error ? e.message : String(e)}`);
234
+ }
235
+
236
+ const count = Math.min(Math.max(1, Math.round(params.count ?? 1)), 4);
237
+ const r = await callAccount<ImageResponse>("/api/agent/media/images", {
238
+ method: "POST",
239
+ signal,
240
+ body: {
241
+ prompt,
242
+ n: count,
243
+ ...(inputs.length ? { images: inputs } : {}),
244
+ ...(params.aspectRatio ? { aspectRatio: params.aspectRatio } : {}),
245
+ ...(params.size ? { imageSize: params.size } : {}),
246
+ ...(params.model ? { model: params.model } : {}),
247
+ },
248
+ });
249
+ if (!r.ok) return text(`Image generation failed: ${r.message}`);
250
+
251
+ const images = r.data.images ?? [];
252
+ if (images.length === 0) return text("Image generation returned no images.");
253
+
254
+ // Multi-variation output gets -1/-2 suffixes so nothing overwrites anything; a
255
+ // single image keeps the exact path asked for, which is what a workflow chains on.
256
+ const target = abs(cwd, params.path);
257
+ const ext = extname(target) || extForMime(images[0].mimeType ?? "", ".png");
258
+ const stem = target.slice(0, target.length - extname(target).length);
259
+ const written: string[] = [];
260
+ for (const [i, img] of images.entries()) {
261
+ if (!img.data) continue;
262
+ const out = images.length === 1 ? `${stem}${ext}` : `${stem}-${i + 1}${ext}`;
263
+ written.push(writeOut(out, Buffer.from(img.data, "base64")));
264
+ }
265
+ if (written.length === 0) return text("Image generation returned no usable image data.");
266
+
267
+ const verb = inputs.length ? (inputs.length === 1 ? "Edited" : "Composed") : "Generated";
268
+ return text(`${verb} ${written.length} image${written.length === 1 ? "" : "s"} with ${r.data.model ?? "the account image model"}:\n${written.map((w) => ` ${w}`).join("\n")}`);
269
+ },
270
+ };
271
+
272
+ // ── Video ────────────────────────────────────────────────────────────────────
273
+
274
+ interface VideoSubmitResponse {
275
+ jobId?: string;
276
+ status?: string;
277
+ model?: string;
278
+ }
279
+ interface VideoStatusResponse {
280
+ status?: string;
281
+ message?: string;
282
+ mimeType?: string;
283
+ data?: string;
284
+ delivered?: boolean;
285
+ model?: string;
286
+ }
287
+
288
+ export const generateVideoToolDefinition = {
289
+ name: "generate_video",
290
+ label: "Generate Video",
291
+ description:
292
+ "Generate a video clip from a text prompt and save it to disk. Give `firstFrame` (a path to an " +
293
+ "image) to animate an existing picture, and `lastFrame` as well to interpolate between two stills — " +
294
+ "that pairing is how you keep several clips visually continuous: end one clip on a frame you " +
295
+ "extracted with video_compose, then start the next from it. Generation takes minutes and this tool " +
296
+ "waits for it. Expensive (roughly $0.10-$1 a clip) and billed to the user's Privateer account, so " +
297
+ "plan the shot before calling. Clip lengths and aspect ratios are model-specific — check " +
298
+ "media_capabilities first if unsure. Stitch the finished clips with video_compose.",
299
+ parameters: Type.Object({
300
+ prompt: Type.String({ description: "What happens in the shot: subject, action, camera move, style." }),
301
+ path: Type.String({ description: "Where to write the video, relative to cwd or absolute (e.g. 'clips/01-opening.mp4')." }),
302
+ firstFrame: Type.Optional(Type.String({ description: "Path to an image to use as the opening frame (image-to-video)." })),
303
+ lastFrame: Type.Optional(Type.String({ description: "Path to an image to use as the closing frame. Requires firstFrame." })),
304
+ seconds: Type.Optional(Type.Number({ description: "Clip length in seconds. Only certain values are legal per model — see media_capabilities." })),
305
+ aspectRatio: Type.Optional(Type.String({ description: "Aspect ratio, e.g. '16:9', '9:16'. Model-specific." })),
306
+ resolution: Type.Optional(Type.String({ description: "Resolution, e.g. '720p' or '1080p'." })),
307
+ audio: Type.Optional(Type.Boolean({ description: "Ask the model to generate a soundtrack too, where it supports one. Costs more. Defaults to false." })),
308
+ model: Type.Optional(Type.String({ description: "Override the account's video model (e.g. 'google/veo-3.1-lite'). Leave unset to use the account default." })),
309
+ }),
310
+ async execute(
311
+ _toolCallId: string,
312
+ params: {
313
+ prompt: string; path: string; firstFrame?: string; lastFrame?: string;
314
+ seconds?: number; aspectRatio?: string; resolution?: string; audio?: boolean; model?: string;
315
+ },
316
+ signal?: AbortSignal,
317
+ _onUpdate?: unknown,
318
+ ctx?: { cwd?: string },
319
+ ) {
320
+ const cwd = ctx?.cwd ?? process.cwd();
321
+ const prompt = String(params.prompt ?? "").trim();
322
+ if (!prompt) return text("Error: prompt is required.");
323
+ if (!params.path) return text("Error: path is required — say where to save the video.");
324
+ if (params.lastFrame && !params.firstFrame) return text("Error: lastFrame needs firstFrame alongside it.");
325
+
326
+ let firstFrame: { data: string; mimeType: string } | undefined;
327
+ let lastFrame: { data: string; mimeType: string } | undefined;
328
+ try {
329
+ if (params.firstFrame) firstFrame = readInputImage(cwd, params.firstFrame);
330
+ if (params.lastFrame) lastFrame = readInputImage(cwd, params.lastFrame);
331
+ } catch (e) {
332
+ return text(`Error: ${e instanceof Error ? e.message : String(e)}`);
333
+ }
334
+
335
+ const submitted = await callAccount<VideoSubmitResponse>("/api/agent/media/videos", {
336
+ method: "POST",
337
+ signal,
338
+ body: {
339
+ prompt,
340
+ ...(params.seconds != null ? { seconds: params.seconds } : {}),
341
+ ...(params.aspectRatio ? { aspectRatio: params.aspectRatio } : {}),
342
+ ...(params.resolution ? { resolution: params.resolution } : {}),
343
+ ...(params.audio ? { generateAudio: true } : {}),
344
+ ...(params.model ? { model: params.model } : {}),
345
+ ...(firstFrame ? { firstFrame } : {}),
346
+ ...(lastFrame ? { lastFrame } : {}),
347
+ },
348
+ });
349
+ if (!submitted.ok) return text(`Video generation failed: ${submitted.message}`);
350
+ const jobId = submitted.data.jobId;
351
+ if (!jobId) return text("Video generation failed: Privateer did not return a job id.");
352
+
353
+ // Poll to completion. The account is charged when the provider delivers, so an
354
+ // abandoned poll still costs money — hence the timeout message names the job id.
355
+ const deadline = Date.now() + VIDEO_POLL_TIMEOUT_MS;
356
+ const cancelled = () =>
357
+ text(`Video job ${jobId} was submitted but the wait was cancelled. It is still running and will still be billed.`);
358
+ for (;;) {
359
+ if (signal?.aborted) return cancelled();
360
+ await sleep(VIDEO_POLL_INTERVAL_MS, signal);
361
+ // sleep() resolves early on abort, so re-check before spending a request on a
362
+ // signal that is already dead — otherwise the cancel surfaces as a network error.
363
+ if (signal?.aborted) return cancelled();
364
+ const poll = await callAccount<VideoStatusResponse>(`/api/agent/media/videos/${encodeURIComponent(jobId)}`, {
365
+ method: "GET",
366
+ signal,
367
+ });
368
+ if (!poll.ok) return text(`Video job ${jobId} could not be polled: ${poll.message}`);
369
+
370
+ const status = String(poll.data.status ?? "").toLowerCase();
371
+ if (status === "failed") return text(`Video generation failed: ${poll.data.message ?? "the provider reported a failure"}.`);
372
+ if (status === "completed") {
373
+ if (!poll.data.data) {
374
+ // The bytes were handed out on an earlier poll and are not stored anywhere.
375
+ return text(`Video job ${jobId} already delivered its bytes on an earlier poll; they were not saved. Generate again if the file is missing.`);
376
+ }
377
+ const target = abs(cwd, params.path);
378
+ const ext = extname(target) || extForMime(poll.data.mimeType ?? "", ".mp4");
379
+ const out = `${target.slice(0, target.length - extname(target).length)}${ext}`;
380
+ const summary = writeOut(out, Buffer.from(poll.data.data, "base64"));
381
+ return text(`Generated video with ${poll.data.model ?? submitted.data.model ?? "the account video model"}: ${summary}`);
382
+ }
383
+ if (Date.now() > deadline) {
384
+ return text(
385
+ `Video job ${jobId} is still ${status || "running"} after ${Math.round(VIDEO_POLL_TIMEOUT_MS / 60000)} minutes. ` +
386
+ "It will still complete and still be billed; nothing was saved here.",
387
+ );
388
+ }
389
+ }
390
+ },
391
+ };
392
+
393
+ function sleep(ms: number, signal?: AbortSignal): Promise<void> {
394
+ return new Promise((resolve_) => {
395
+ const t = setTimeout(resolve_, ms);
396
+ signal?.addEventListener("abort", () => { clearTimeout(t); resolve_(); }, { once: true });
397
+ });
398
+ }
399
+
400
+ // ── Audio ────────────────────────────────────────────────────────────────────
401
+
402
+ interface AudioResponse {
403
+ audioBase64?: string;
404
+ mimeType?: string;
405
+ model?: string;
406
+ }
407
+
408
+ export const generateSpeechToolDefinition = {
409
+ name: "generate_speech",
410
+ label: "Generate Speech",
411
+ description:
412
+ "Turn text into spoken audio and save it to disk. Use it to narrate a video you are assembling, or " +
413
+ "to produce a spoken version of a written answer. Billed to the user's Privateer account; the " +
414
+ "account's default voice model is a confidential-compute one, so the text is processed inside an " +
415
+ "enclave rather than by a retaining provider. Mux the result onto video with video_compose.",
416
+ parameters: Type.Object({
417
+ text: Type.String({ description: "The words to speak. Write them as they should be read aloud." }),
418
+ path: Type.String({ description: "Where to write the audio, relative to cwd or absolute (e.g. 'audio/narration.mp3')." }),
419
+ voice: Type.Optional(Type.String({ description: "Voice name, if the account's TTS model offers a choice. Leave unset for its default." })),
420
+ model: Type.Optional(Type.String({ description: "Override the account's text-to-speech model." })),
421
+ }),
422
+ async execute(
423
+ _toolCallId: string,
424
+ params: { text: string; path: string; voice?: string; model?: string },
425
+ signal?: AbortSignal,
426
+ _onUpdate?: unknown,
427
+ ctx?: { cwd?: string },
428
+ ) {
429
+ const cwd = ctx?.cwd ?? process.cwd();
430
+ const body = String(params.text ?? "").trim();
431
+ if (!body) return text("Error: text is required.");
432
+ if (!params.path) return text("Error: path is required — say where to save the audio.");
433
+
434
+ const r = await callAccount<AudioResponse>("/api/audio/speech", {
435
+ method: "POST",
436
+ signal,
437
+ body: {
438
+ text: body,
439
+ ...(params.voice ? { voice: params.voice } : {}),
440
+ ...(params.model ? { ttsModelId: params.model } : {}),
441
+ },
442
+ });
443
+ if (!r.ok) return text(`Speech generation failed: ${r.message}`);
444
+ if (!r.data.audioBase64) return text("Speech generation returned no audio.");
445
+
446
+ const target = abs(cwd, params.path);
447
+ const ext = extname(target) || extForMime(r.data.mimeType ?? "", ".mp3");
448
+ const out = `${target.slice(0, target.length - extname(target).length)}${ext}`;
449
+ return text(`Generated speech: ${writeOut(out, Buffer.from(r.data.audioBase64, "base64"))}`);
450
+ },
451
+ };
452
+
453
+ export const generateMusicToolDefinition = {
454
+ name: "generate_music",
455
+ label: "Generate Music",
456
+ description:
457
+ "Generate an instrumental music clip from a text prompt and save it to disk — a soundtrack for a " +
458
+ "video you are assembling. PRIVACY: music is the one media type with no zero-retention or " +
459
+ "confidential option anywhere in the catalog, so the prompt reaches a provider that may retain it. " +
460
+ "Privateer sends it unattributed (no account id, no history), but it is not private the way the " +
461
+ "other media tools are. Do not put anything sensitive or personal in a music prompt, and say so if " +
462
+ "the user's own wording would carry something identifying.",
463
+ parameters: Type.Object({
464
+ prompt: Type.String({ description: "The music to generate: genre, mood, instrumentation, tempo. Keep it about the music, not about the user." }),
465
+ path: Type.String({ description: "Where to write the audio, relative to cwd or absolute (e.g. 'audio/score.mp3')." }),
466
+ model: Type.Optional(Type.String({ description: "Override the account's music model." })),
467
+ }),
468
+ async execute(
469
+ _toolCallId: string,
470
+ params: { prompt: string; path: string; model?: string },
471
+ signal?: AbortSignal,
472
+ _onUpdate?: unknown,
473
+ ctx?: { cwd?: string },
474
+ ) {
475
+ const cwd = ctx?.cwd ?? process.cwd();
476
+ const prompt = String(params.prompt ?? "").trim();
477
+ if (!prompt) return text("Error: prompt is required.");
478
+ if (!params.path) return text("Error: path is required — say where to save the audio.");
479
+
480
+ const r = await callAccount<AudioResponse>("/api/audio/music", {
481
+ method: "POST",
482
+ signal,
483
+ body: { prompt, ...(params.model ? { musicModelId: params.model } : {}) },
484
+ });
485
+ if (!r.ok) return text(`Music generation failed: ${r.message}`);
486
+ if (!r.data.audioBase64) return text("Music generation returned no audio.");
487
+
488
+ const target = abs(cwd, params.path);
489
+ const ext = extname(target) || extForMime(r.data.mimeType ?? "", ".mp3");
490
+ const out = `${target.slice(0, target.length - extname(target).length)}${ext}`;
491
+ return text(
492
+ `Generated music with ${r.data.model ?? "the account music model"}: ${writeOut(out, Buffer.from(r.data.audioBase64, "base64"))}\n` +
493
+ "(Reminder for the answer you give the user: music prompts are sent to a provider with no zero-retention option, unattributed.)",
494
+ );
495
+ },
496
+ };
497
+
498
+ // ── Capabilities ─────────────────────────────────────────────────────────────
499
+
500
+ interface CapabilitiesResponse {
501
+ image?: { model?: string; blockedByZdr?: boolean; maxPerCall?: number };
502
+ video?: { model?: string; blockedByZdr?: boolean; durations?: number[] | null; aspectRatios?: string[] | null };
503
+ privacy?: { requireZdr?: boolean; allowNonZdrMedia?: boolean };
504
+ }
505
+
506
+ export const mediaCapabilitiesToolDefinition = {
507
+ name: "media_capabilities",
508
+ label: "Media Capabilities",
509
+ description:
510
+ "Report what this Privateer account can generate right now: which image and video models it " +
511
+ "resolves to, the clip lengths and aspect ratios that video model accepts, and whether the " +
512
+ "account's privacy settings currently block media generation. Free and instant. Call it before " +
513
+ "planning a multi-clip video so you pick a legal clip length instead of discovering it through a " +
514
+ "rejected — or worse, billed — call.",
515
+ parameters: Type.Object({}),
516
+ async execute(_toolCallId: string, _params: unknown, signal?: AbortSignal) {
517
+ const r = await callAccount<CapabilitiesResponse>("/api/agent/media/capabilities", { method: "GET", signal });
518
+ if (!r.ok) return text(`Could not read media capabilities: ${r.message}`);
519
+
520
+ const { image, video, privacy } = r.data;
521
+ const lines = [
522
+ `Image model: ${image?.model ?? "unknown"}${image?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : ""}`,
523
+ ` up to ${image?.maxPerCall ?? 1} image(s) per call`,
524
+ `Video model: ${video?.model ?? "unknown"}${video?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : ""}`,
525
+ ` clip lengths: ${video?.durations?.length ? `${video.durations.join(", ")}s` : "model default only"}`,
526
+ ` aspect ratios: ${video?.aspectRatios?.length ? video.aspectRatios.join(", ") : "model default only"}`,
527
+ `Privacy: requireZdr=${privacy?.requireZdr ?? "?"}, allowNonZdrMedia=${privacy?.allowNonZdrMedia ?? "?"}`,
528
+ ];
529
+ if (image?.blockedByZdr || video?.blockedByZdr) {
530
+ lines.push(
531
+ "A [BLOCKED] model means the account requires Zero Data Retention and that model has no ZDR endpoint. " +
532
+ "Only the account owner can change it (Settings → Privacy); do not keep retrying.",
533
+ );
534
+ }
535
+ lines.push("Speech and music are always available (speech runs confidentially; music has no ZDR option — see generate_music).");
536
+ return text(lines.join("\n"));
537
+ },
538
+ };
539
+
540
+ /**
541
+ * Extension factory registering all five media tools. Used by the surfaces that build
542
+ * their session from an explicit `extensionFactories` list (harbor, channels, ACP, the
543
+ * REPL); the interactive TUI picks the same definitions up through
544
+ * `extensions/privateer-media.ts`, which the launcher passes it as an `-e` argument.
545
+ */
546
+ export function makeMediaTools() {
547
+ return (pi: { registerTool?: (def: unknown) => void }): void => {
548
+ pi.registerTool?.(generateImageToolDefinition);
549
+ pi.registerTool?.(generateVideoToolDefinition);
550
+ pi.registerTool?.(generateSpeechToolDefinition);
551
+ pi.registerTool?.(generateMusicToolDefinition);
552
+ pi.registerTool?.(mediaCapabilitiesToolDefinition);
553
+ };
554
+ }
@@ -1,17 +1,16 @@
1
1
  // The two relay file tools as a Pi extension factory bound to ONE specific bridge.
2
2
  //
3
- // `send_file_to_client` / `save_attachment` are normally registered by the shipped TUI
4
- // extension (extensions/privateer-gate.ts), against that extension's module-level
5
- // RemoteBridge — the one `/remote-access` attaches a relay to. That is right for the TUI
6
- // and wrong everywhere else: the extension is AUTO-DISCOVERED from ~/.privateer/agent/
7
- // extensions into every session that shares the agent dir, including the sessions the
8
- // harbor daemon stands up (live task spawns), which own their OWN bridge + relay. Pi
9
- // resolves duplicate tool names first-registration-wins and loads discovered extensions
10
- // before inline factories, so the discovered pair would shadow a session's own and answer
11
- // "remote access is off" while the session's relay is connected and driving.
3
+ // `send_file_to_client` / `save_attachment` are registered by the shipped TUI extension
4
+ // (extensions/privateer-gate.ts) against that extension's module-level RemoteBridge — the
5
+ // one `/remote-access` attaches a relay to. That is right for the TUI and wrong everywhere
6
+ // else: a live task spawn owns its OWN bridge + relay, and the pair must be bound to that.
12
7
  //
13
- // Hence this factory: a session that has its own bridge registers the pair here, and the
14
- // gate extension stands down inside the daemon (PRIVATEER_HARBOR_DAEMON).
8
+ // Hence this factory: a session that has its own bridge registers the pair here. It used
9
+ // to also require the gate extension to stand down inside the daemon, because the shared
10
+ // agent dir made that extension discoverable into every session the daemon ran — and Pi
11
+ // resolves duplicate tool names first-registration-wins, so the discovered pair shadowed
12
+ // the session's own and answered "remote access is off" while the app sat attached. The
13
+ // moat is no longer discoverable (src/config/moat.ts), so there is only ever one pair.
15
14
  import { makeSendFileTool, type SendFileBridge } from "./sendFile.ts";
16
15
  import { makeSaveAttachmentTool } from "./saveAttachment.ts";
17
16
  import type { AttachmentStore } from "../util/attachmentStore.ts";