privateer-agent 0.12.44 → 0.12.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/privateer-subagent.mjs +1 -0
- package/extensions/privateer-vision.ts +39 -0
- package/package.json +1 -1
- package/patches/@earendil-works+pi-coding-agent+0.86.0.patch +140 -1
- package/src/cli/chat.ts +13 -4
- package/src/config/moat.ts +5 -0
- package/src/config/moatManifest.json +1 -0
- package/src/engine/errors.ts +4 -1
- package/src/providers/defaultModel.ts +37 -0
- package/src/providers/visionDelegate.ts +199 -0
- package/src/tools/media.ts +102 -12
|
@@ -55,6 +55,7 @@ export function moatExtensionPaths(repoRoot = REPO, env = process.env) {
|
|
|
55
55
|
join(repoRoot, "extensions", "privateer-privacy.ts"),
|
|
56
56
|
join(repoRoot, "extensions", "privateer-account.ts"),
|
|
57
57
|
join(repoRoot, "extensions", "privateer-media.ts"),
|
|
58
|
+
join(repoRoot, "extensions", "privateer-vision.ts"),
|
|
58
59
|
];
|
|
59
60
|
}
|
|
60
61
|
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
// Vision delegation: when the current model can't see images, a vision model on the
|
|
2
|
+
// same provider (same key) describes them, and the description is what gets sent.
|
|
3
|
+
// All the policy — which delegate, why never another provider, what gets rewritten —
|
|
4
|
+
// lives in src/providers/visionDelegate.ts; this file is only the Pi wiring.
|
|
5
|
+
//
|
|
6
|
+
// It hooks `context`, which runs before EVERY LLM call on a per-request copy of the
|
|
7
|
+
// messages, rather than `input`/`tool_result`. That catches images from every source
|
|
8
|
+
// (@file mentions, pasted images, `read`, media and computer tools, MCP results, and
|
|
9
|
+
// history from before a /model switch) while leaving the session's own copy intact.
|
|
10
|
+
// Descriptions are cached by image hash, so each image costs one delegate call per
|
|
11
|
+
// process, not one per turn.
|
|
12
|
+
|
|
13
|
+
import { describeImagesInMessages, pickVisionDelegate } from "../src/providers/visionDelegate.ts";
|
|
14
|
+
|
|
15
|
+
export default function privateerVision(pi: any): void {
|
|
16
|
+
const announced = new Set<string>();
|
|
17
|
+
|
|
18
|
+
pi.on("context", async (event: any, ctx: any) => {
|
|
19
|
+
const model = ctx?.model;
|
|
20
|
+
const registry = ctx?.modelRegistry;
|
|
21
|
+
if (!model || !registry) return;
|
|
22
|
+
const delegate = pickVisionDelegate(model, registry);
|
|
23
|
+
if (!delegate) return;
|
|
24
|
+
|
|
25
|
+
const currentSpec = `${model.provider}/${model.id}`;
|
|
26
|
+
const delegateSpec = `${delegate.provider}/${delegate.id}`;
|
|
27
|
+
const messages = await describeImagesInMessages(event.messages, delegate, registry, {
|
|
28
|
+
currentSpec,
|
|
29
|
+
signal: ctx.signal,
|
|
30
|
+
onDescribe: () => {
|
|
31
|
+
const key = `${currentSpec}→${delegateSpec}`;
|
|
32
|
+
if (announced.has(key) || !ctx.hasUI) return;
|
|
33
|
+
announced.add(key);
|
|
34
|
+
ctx.ui.notify(`${currentSpec} can't see images — ${delegateSpec} will describe them for it.`, "info");
|
|
35
|
+
},
|
|
36
|
+
});
|
|
37
|
+
return messages ? { messages } : undefined;
|
|
38
|
+
});
|
|
39
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "privateer-agent",
|
|
3
|
-
"version": "0.12.
|
|
3
|
+
"version": "0.12.47",
|
|
4
4
|
"description": "Privacy-first terminal coding agent — bring your own model across 20 providers (Anthropic, OpenAI, OpenRouter, Google, local Ollama…). Safe-by-default permissions, MCP, sub-agents, workflows, and verifiable TEE inference. Built on the Pi toolkit.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -1025,6 +1025,19 @@ index d101003..540b972 100644
|
|
|
1025
1025
|
return "project";
|
|
1026
1026
|
}
|
|
1027
1027
|
return "path";
|
|
1028
|
+
diff --git a/node_modules/@earendil-works/pi-coding-agent/dist/core/slash-commands.js b/node_modules/@earendil-works/pi-coding-agent/dist/core/slash-commands.js
|
|
1029
|
+
index 170a066..b535f72 100644
|
|
1030
|
+
--- a/node_modules/@earendil-works/pi-coding-agent/dist/core/slash-commands.js
|
|
1031
|
+
+++ b/node_modules/@earendil-works/pi-coding-agent/dist/core/slash-commands.js
|
|
1032
|
+
@@ -8,7 +8,7 @@ export const BUILTIN_SLASH_COMMANDS = [
|
|
1033
|
+
{ name: "export", description: "Export session (HTML default, or specify path: .html/.jsonl)" },
|
|
1034
|
+
{ name: "import", description: "Import and resume a session from a JSONL file" },
|
|
1035
|
+
{ name: "share", description: "Share session as a secret GitHub gist" },
|
|
1036
|
+
- { name: "bug", description: "Report a bug to the Pi developers", argumentHint: "<description>" },
|
|
1037
|
+
+ { name: "bug", description: "Export a bug report to email to Privateer support", argumentHint: "<description>" },
|
|
1038
|
+
{ name: "copy", description: "Copy last agent message to clipboard" },
|
|
1039
|
+
{ name: "name", description: "Set session display name" },
|
|
1040
|
+
{ name: "session", description: "Show session info and stats" },
|
|
1028
1041
|
diff --git a/node_modules/@earendil-works/pi-coding-agent/dist/core/tools/output-accumulator.js b/node_modules/@earendil-works/pi-coding-agent/dist/core/tools/output-accumulator.js
|
|
1029
1042
|
index 7241668..7b79305 100644
|
|
1030
1043
|
--- a/node_modules/@earendil-works/pi-coding-agent/dist/core/tools/output-accumulator.js
|
|
@@ -1292,6 +1305,123 @@ index c30d6b8..ad1654a 100644
|
|
|
1292
1305
|
];
|
|
1293
1306
|
return warnings;
|
|
1294
1307
|
}
|
|
1308
|
+
diff --git a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.d.ts b/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.d.ts
|
|
1309
|
+
index 7eae594..b21fae3 100644
|
|
1310
|
+
--- a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.d.ts
|
|
1311
|
+
+++ b/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.d.ts
|
|
1312
|
+
@@ -8,7 +8,7 @@ interface BugReportContext {
|
|
1313
|
+
showStatus: (message: string) => void;
|
|
1314
|
+
showError: (message: string) => void;
|
|
1315
|
+
}
|
|
1316
|
+
-/** Run the `/bug` flow: consent, optional summary, then upload or export. */
|
|
1317
|
+
+/** Run the `/bug` flow: consent, optional summary, then export a zip to email to support. */
|
|
1318
|
+
export declare function reportBug(context: BugReportContext, initialHint?: string): Promise<void>;
|
|
1319
|
+
export {};
|
|
1320
|
+
//# sourceMappingURL=bug-report.d.ts.map
|
|
1321
|
+
|
|
1322
|
+
diff --git a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.js b/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.js
|
|
1323
|
+
index 59ae447..68cf842 100644
|
|
1324
|
+
--- a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.js
|
|
1325
|
+
+++ b/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/bug-report.js
|
|
1326
|
+
@@ -1,18 +1,21 @@
|
|
1327
|
+
import * as path from "node:path";
|
|
1328
|
+
-import { getAuthCredential } from "../../cli/auth-command.js";
|
|
1329
|
+
import { BUG_REPORT_CUSTOM_ENTRY_TYPE, bugReportArchiveFileName, collectBugReportDiagnostics, collectBugReportMetadata, writeBugReportArchive, } from "../../core/bug-report.js";
|
|
1330
|
+
-import { uploadBugReport } from "../../core/bug-report-upload.js";
|
|
1331
|
+
import { clearCrashLog, readCrashLog } from "../../core/crash-log.js";
|
|
1332
|
+
-import { getRadiusGatewayUrl, RADIUS_PROVIDER_ID } from "../../core/radius.js";
|
|
1333
|
+
import { serializeSessionBranch } from "../../core/session-export.js";
|
|
1334
|
+
import { BorderedLoader } from "./components/bordered-loader.js";
|
|
1335
|
+
import { ExtensionInputComponent } from "./components/extension-input.js";
|
|
1336
|
+
import { ExtensionSelectorComponent } from "./components/extension-selector.js";
|
|
1337
|
+
import { createShareTrailingEntries } from "./session-share.js";
|
|
1338
|
+
import { theme } from "./theme/theme.js";
|
|
1339
|
+
-const DISCLAIMER = "This report goes to the Pi developers (Earendil) and is not shared publicly. It includes your pi version, operating system, the current model and provider configuration (without API keys), loaded extensions, settings, and provider error diagnostics from this session.";
|
|
1340
|
+
+// Privateer patch: stock Pi's `/bug` uploads this bundle straight to Earendil's Radius
|
|
1341
|
+
+// Gateway (session transcript, settings, environment) with no Privateer-controlled
|
|
1342
|
+
+// endpoint in between — not acceptable for a privacy-first tool. We keep the same
|
|
1343
|
+
+// consent/redaction flow but the only delivery is a local zip the user emails to us
|
|
1344
|
+
+// themselves, so nothing leaves the machine without them explicitly attaching it.
|
|
1345
|
+
+const SUPPORT_EMAIL = "support@privateer.pro";
|
|
1346
|
+
+const DISCLAIMER = `This report is written to a zip file on your machine only — nothing is uploaded automatically. It includes your Privateer version, operating system, the current model and provider configuration (without API keys), loaded extensions, settings, and provider error diagnostics from this session. Email the file to ${SUPPORT_EMAIL} to send it to us.`;
|
|
1347
|
+
const TRANSCRIPT_NOTE = "The transcript contains your messages, model output, tool calls and their results, including file contents and command output read during this session.";
|
|
1348
|
+
-/** Run the `/bug` flow: consent, optional summary, then upload or export. */
|
|
1349
|
+
+/** Run the `/bug` flow: consent, optional summary, then export a zip for the user to email us. */
|
|
1350
|
+
export async function reportBug(context, initialHint) {
|
|
1351
|
+
const options = await promptForOptions(context, initialHint);
|
|
1352
|
+
if (!options) {
|
|
1353
|
+
@@ -47,16 +50,6 @@ export async function reportBug(context, initialHint) {
|
|
1354
|
+
context.showError(`Failed to build bug report: ${errorMessage(error)}`);
|
|
1355
|
+
return;
|
|
1356
|
+
}
|
|
1357
|
+
- if (options.delivery === "upload") {
|
|
1358
|
+
- const failure = await upload(context, bundle);
|
|
1359
|
+
- if (failure === undefined)
|
|
1360
|
+
- return;
|
|
1361
|
+
- const fallback = await choose(context, "Upload failed", ["Export as Zip", "Cancel"], `${failure}\n\nExport the report as a zip archive instead?`);
|
|
1362
|
+
- if (fallback !== "Export as Zip") {
|
|
1363
|
+
- context.showStatus("Bug report cancelled");
|
|
1364
|
+
- return;
|
|
1365
|
+
- }
|
|
1366
|
+
- }
|
|
1367
|
+
await exportZip(context, bundle);
|
|
1368
|
+
}
|
|
1369
|
+
async function promptForOptions(context, initialHint) {
|
|
1370
|
+
@@ -76,14 +69,13 @@ async function promptForOptions(context, initialHint) {
|
|
1371
|
+
includeSummary = summary !== "No";
|
|
1372
|
+
}
|
|
1373
|
+
const description = hint.trim();
|
|
1374
|
+
- const delivery = await choose(context, "Bug report", ["Upload Report", "Export as Zip", "Cancel"], `Description: ${description || "none"}\nTranscript: ${includeSession ? "included" : "not included"}\nSummary: ${includeSummary ? `written by ${context.session.model?.name ?? "the current model"}` : "none"}\n\nUpload sends the report to ${new URL(getRadiusGatewayUrl()).host}. Export writes a zip archive to the current directory instead.`);
|
|
1375
|
+
- if (!delivery || delivery === "Cancel")
|
|
1376
|
+
+ const confirmed = await choose(context, "Bug report", ["Export Report", "Cancel"], `Description: ${description || "none"}\nTranscript: ${includeSession ? "included" : "not included"}\nSummary: ${includeSummary ? `written by ${context.session.model?.name ?? "the current model"}` : "none"}\n\nWrites a zip archive to the current directory. Email it to ${SUPPORT_EMAIL}.`);
|
|
1377
|
+
+ if (!confirmed || confirmed === "Cancel")
|
|
1378
|
+
return undefined;
|
|
1379
|
+
return {
|
|
1380
|
+
hint: description || undefined,
|
|
1381
|
+
includeSession,
|
|
1382
|
+
includeSummary,
|
|
1383
|
+
- delivery: delivery === "Upload Report" ? "upload" : "zip",
|
|
1384
|
+
};
|
|
1385
|
+
}
|
|
1386
|
+
function buildBundle(session, options, summary) {
|
|
1387
|
+
@@ -111,28 +103,6 @@ function buildBundle(session, options, summary) {
|
|
1388
|
+
: undefined,
|
|
1389
|
+
};
|
|
1390
|
+
}
|
|
1391
|
+
-async function upload(context, bundle) {
|
|
1392
|
+
- const loader = showLoader(context, "Uploading bug report...");
|
|
1393
|
+
- try {
|
|
1394
|
+
- const provider = context.session.modelRuntime.getProvider(RADIUS_PROVIDER_ID);
|
|
1395
|
+
- const token = provider
|
|
1396
|
+
- ? getAuthCredential(await context.session.modelRuntime.getAuth(RADIUS_PROVIDER_ID, { minOAuthValidityMs: 5 * 60_000 }))
|
|
1397
|
+
- : undefined;
|
|
1398
|
+
- const result = await uploadBugReport(bundle, { token, signal: loader.signal });
|
|
1399
|
+
- restoreEditor(context, loader);
|
|
1400
|
+
- recordInSession(context.session, bundle, { delivery: "upload" });
|
|
1401
|
+
- context.showStatus(`Bug report uploaded. Report ID: ${result.id}`);
|
|
1402
|
+
- return undefined;
|
|
1403
|
+
- }
|
|
1404
|
+
- catch (error) {
|
|
1405
|
+
- restoreEditor(context, loader);
|
|
1406
|
+
- if (loader.signal.aborted) {
|
|
1407
|
+
- context.showStatus("Bug report cancelled");
|
|
1408
|
+
- return undefined;
|
|
1409
|
+
- }
|
|
1410
|
+
- return errorMessage(error);
|
|
1411
|
+
- }
|
|
1412
|
+
-}
|
|
1413
|
+
async function exportZip(context, bundle) {
|
|
1414
|
+
const archivePath = path.join(process.cwd(), bugReportArchiveFileName(bundle.metadata.id));
|
|
1415
|
+
try {
|
|
1416
|
+
@@ -143,7 +113,7 @@ async function exportZip(context, bundle) {
|
|
1417
|
+
return;
|
|
1418
|
+
}
|
|
1419
|
+
recordInSession(context.session, bundle, { delivery: "zip", path: archivePath });
|
|
1420
|
+
- context.showStatus(`Bug report exported to: ${archivePath}\nReport ID: ${bundle.metadata.id}`);
|
|
1421
|
+
+ context.showStatus(`Bug report exported to: ${archivePath}\nReport ID: ${bundle.metadata.id}\n\nEmail this file to ${SUPPORT_EMAIL} to reach support.`);
|
|
1422
|
+
}
|
|
1423
|
+
function recordInSession(session, bundle, delivery) {
|
|
1424
|
+
session.sessionManager.appendCustomEntry(BUG_REPORT_CUSTOM_ENTRY_TYPE, {
|
|
1295
1425
|
diff --git a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/components/config-selector.js b/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/components/config-selector.js
|
|
1296
1426
|
index 6f7ed6a..b0ed0ae 100644
|
|
1297
1427
|
--- a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/components/config-selector.js
|
|
@@ -1694,7 +1824,7 @@ index 6d0356c..ca0aa98 100644
|
|
|
1694
1824
|
if (content) {
|
|
1695
1825
|
text += `\n\n${content}`;
|
|
1696
1826
|
diff --git a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/interactive-mode.js b/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/interactive-mode.js
|
|
1697
|
-
index d82182c..
|
|
1827
|
+
index d82182c..2764952 100644
|
|
1698
1828
|
--- a/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/interactive-mode.js
|
|
1699
1829
|
+++ b/node_modules/@earendil-works/pi-coding-agent/dist/modes/interactive/interactive-mode.js
|
|
1700
1830
|
@@ -10,7 +10,7 @@ import * as TuiLayouts from "@earendil-works/pi-tui";
|
|
@@ -1835,6 +1965,15 @@ index d82182c..41426bc 100644
|
|
|
1835
1965
|
// Check tmux keyboard setup asynchronously
|
|
1836
1966
|
this.checkTmuxKeyboardSetup().then((warning) => {
|
|
1837
1967
|
if (warning) {
|
|
1968
|
+
@@ -1568,7 +1635,7 @@ export class InteractiveMode {
|
|
1969
|
+
if (this.bugReportHintShown)
|
|
1970
|
+
return;
|
|
1971
|
+
this.bugReportHintShown = true;
|
|
1972
|
+
- this.chatContainer.addChild(new Text(theme.fg("muted", `If this looks like a ${APP_NAME} bug, /bug sends a report to the developers.`), this.outputPad, 0));
|
|
1973
|
+
+ this.chatContainer.addChild(new Text(theme.fg("muted", `If this looks like a ${APP_NAME} bug, /bug exports a report you can email to support@privateer.pro.`), this.outputPad, 0));
|
|
1974
|
+
this.ui.requestRender();
|
|
1975
|
+
}
|
|
1976
|
+
renderCurrentSessionState() {
|
|
1838
1977
|
@@ -2419,7 +2486,17 @@ export class InteractiveMode {
|
|
1839
1978
|
if (text === "/model" || text.startsWith("/model ")) {
|
|
1840
1979
|
const searchTerm = text.startsWith("/model ") ? text.slice(7).trim() : undefined;
|
package/src/cli/chat.ts
CHANGED
|
@@ -53,8 +53,14 @@ async function main() {
|
|
|
53
53
|
const { modelRegistryOf } = await import("../providers/piAuthStore.ts");
|
|
54
54
|
const { agentVersion } = await import("../config/version.ts");
|
|
55
55
|
const { pickerCatalog, hiddenAccountNotice, hiddenAccountTitleSuffix } = await import("../providers/modelCatalog.ts");
|
|
56
|
-
const {
|
|
57
|
-
|
|
56
|
+
const {
|
|
57
|
+
resolveDefaultModel,
|
|
58
|
+
resolveSignedInModel,
|
|
59
|
+
savedPiDefaultSpec,
|
|
60
|
+
writePiDefaultModel,
|
|
61
|
+
visionWarningAcknowledged,
|
|
62
|
+
acknowledgeVisionWarning,
|
|
63
|
+
} = await import("../providers/defaultModel.ts");
|
|
58
64
|
const { acceptsImages } = await import("../providers/vision.ts");
|
|
59
65
|
const { postOutbox } = await import("../outbox/cloudOutbox.ts");
|
|
60
66
|
const { addPendingCloud } = await import("../routines/store.ts");
|
|
@@ -440,8 +446,11 @@ async function main() {
|
|
|
440
446
|
// blocks for a model that doesn't declare the modality with NO error — `read` on a
|
|
441
447
|
// screenshot just quietly answers about nothing — so make that failure visible
|
|
442
448
|
// instead of letting a user discover it turn by turn. /model to switch.
|
|
443
|
-
|
|
444
|
-
|
|
449
|
+
// Shown once per distinct non-vision spec (visionWarningAcknowledged), not on every
|
|
450
|
+
// launch — a user who keeps a deliberate non-vision pick already knows the tradeoff.
|
|
451
|
+
if (!acceptsImages(spec) && !visionWarningAcknowledged(spec)) {
|
|
452
|
+
console.log(`${YELLOW}⚠ ${provider}/${modelId} can't see images — a vision model on the same key will describe them for it (none available → they're dropped). Run /model to switch.${RESET}`);
|
|
453
|
+
acknowledgeVisionWarning(spec);
|
|
445
454
|
}
|
|
446
455
|
if (noQuarterActive()) applyNoQuarter(true); // launched with --no-quarter: say so up front
|
|
447
456
|
|
package/src/config/moat.ts
CHANGED
|
@@ -284,6 +284,11 @@ export async function buildMoat(opts: MoatOptions): Promise<ExtensionFactory[]>
|
|
|
284
284
|
}
|
|
285
285
|
factories.push(makeAccountProvider()); // must follow pi-privacy — see header
|
|
286
286
|
|
|
287
|
+
// Every kind: a text-only model gets image descriptions from a vision model on the
|
|
288
|
+
// same key instead of silently losing the image. See providers/visionDelegate.ts.
|
|
289
|
+
const { default: privateerVision } = await import("../../extensions/privateer-vision.ts");
|
|
290
|
+
factories.push(privateerVision);
|
|
291
|
+
|
|
287
292
|
if (caps.context) {
|
|
288
293
|
const { default: privateerContext } = await import("../../extensions/privateer-context.ts");
|
|
289
294
|
factories.push(privateerContext);
|
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
{ "entry": "extensions/privateer-tools.ts", "name": "privateer-tools", "note": "Privateer tool pack" },
|
|
17
17
|
{ "entry": "extensions/privateer-privacy.ts", "name": "privateer-privacy", "note": "pi-privacy + account tier resolver" },
|
|
18
18
|
{ "entry": "extensions/privateer-connect.ts", "name": "privateer-connect", "note": "/connect — MCP connector manager" },
|
|
19
|
+
{ "entry": "extensions/privateer-vision.ts", "name": "privateer-vision", "note": "a vision model on the same key describes images for a text-only model" },
|
|
19
20
|
{ "entry": "extensions/privateer-media.ts", "name": "privateer-media", "note": "image/video/speech/music + ffmpeg compose" },
|
|
20
21
|
{ "entry": "extensions/privateer-computer.ts", "name": "privateer-computer", "note": "screen control — screenshots, mouse and keyboard (off unless armed with --allow-computer-control)" },
|
|
21
22
|
{ "entry": "extensions/privateer-desktop.ts", "name": "privateer-desktop", "note": "/desktop — open the Privateer desktop app" },
|
package/src/engine/errors.ts
CHANGED
|
@@ -359,8 +359,11 @@ export function describeError(err: unknown): DescribedError {
|
|
|
359
359
|
});
|
|
360
360
|
}
|
|
361
361
|
if (status === 429) {
|
|
362
|
+
// Never name the upstream provider here — OpenRouter is an implementation
|
|
363
|
+
// detail behind the Privateer account channel, and a plain "rate limited"
|
|
364
|
+
// is all a user needs to know regardless of who's actually throttling us.
|
|
362
365
|
return out({
|
|
363
|
-
message: `Rate limited
|
|
366
|
+
message: `Rate limited (429).`,
|
|
364
367
|
hint: "Wait a moment and try again.",
|
|
365
368
|
retryable: true,
|
|
366
369
|
});
|
|
@@ -227,6 +227,43 @@ export function resolveSignedInModel(env: NodeJS.ProcessEnv = process.env): stri
|
|
|
227
227
|
return resolveDefaultModel({ env, signedIn: true, saved: null });
|
|
228
228
|
}
|
|
229
229
|
|
|
230
|
+
// Whether the startup "can't see images" nag (cli/chat.ts) has already been shown for
|
|
231
|
+
// this exact spec. Without this, a user who deliberately keeps a non-vision saved pick
|
|
232
|
+
// (speed over sight, say) sees the same warning on every single launch forever, which
|
|
233
|
+
// trains people to stop reading warnings at all. Keyed by spec so switching to a
|
|
234
|
+
// DIFFERENT non-vision model warns again once — this tracks "have they seen this
|
|
235
|
+
// specific tradeoff", not "should we ever mention it again".
|
|
236
|
+
export function visionWarningAcknowledged(spec: string): boolean {
|
|
237
|
+
try {
|
|
238
|
+
const raw = readFileSync(join(agentDir(), "settings.json"), "utf8").trim();
|
|
239
|
+
if (!raw) return false;
|
|
240
|
+
const s = JSON.parse(raw) as Record<string, unknown>;
|
|
241
|
+
return s.visionWarningAcknowledgedFor === spec;
|
|
242
|
+
} catch {
|
|
243
|
+
return false;
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
// Record that the nag has been shown for `spec`, so the next launch on the same pick
|
|
248
|
+
// stays quiet. Best-effort and silent like the writers below — losing this write just
|
|
249
|
+
// means the warning repeats once more, never a crash.
|
|
250
|
+
export function acknowledgeVisionWarning(spec: string): void {
|
|
251
|
+
const dir = agentDir();
|
|
252
|
+
const settingsPath = join(dir, "settings.json");
|
|
253
|
+
try {
|
|
254
|
+
mkdirSync(dir, { recursive: true });
|
|
255
|
+
let settings: Record<string, unknown> = {};
|
|
256
|
+
if (existsSync(settingsPath)) {
|
|
257
|
+
const raw = readFileSync(settingsPath, "utf8").trim();
|
|
258
|
+
if (raw) settings = JSON.parse(raw) as Record<string, unknown>;
|
|
259
|
+
}
|
|
260
|
+
settings.visionWarningAcknowledgedFor = spec;
|
|
261
|
+
writeFileSync(settingsPath, `${JSON.stringify(settings, null, 2)}\n`);
|
|
262
|
+
} catch {
|
|
263
|
+
// best-effort: the warning already printed either way.
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
230
267
|
// Split a "provider/id" spec on its first slash (model ids themselves contain "/", so
|
|
231
268
|
// only the first delimiter separates provider from model). Returns null for a spec
|
|
232
269
|
// with no provider prefix.
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
// Seeing on behalf of a model that can't.
|
|
2
|
+
//
|
|
3
|
+
// providers/vision.ts makes a text-only model HONEST about images — it declares
|
|
4
|
+
// `input: ["text"]`, and Pi then drops every image block on the way to the provider.
|
|
5
|
+
// Honest, but blind: a user on glm-5-2 who points the agent at a screenshot still
|
|
6
|
+
// gets an answer about a picture the model never saw, just with a note saying so.
|
|
7
|
+
//
|
|
8
|
+
// This module closes that gap. Before each LLM call (Pi's `context` event) every
|
|
9
|
+
// image in the outgoing messages is handed to a vision-capable model and replaced
|
|
10
|
+
// with that model's description — so the text-only model gets words where it would
|
|
11
|
+
// have got nothing. The session itself keeps the original images: only the
|
|
12
|
+
// per-request copy is rewritten, so switching to a vision model later still sends
|
|
13
|
+
// the real pixels.
|
|
14
|
+
//
|
|
15
|
+
// WHICH delegate — "the key type provided". The delegate must be reachable with the
|
|
16
|
+
// credential the user is ALREADY using, i.e. the same provider as the current model:
|
|
17
|
+
//
|
|
18
|
+
// privateer/* (account) → privateer/tinfoil/gemma4-31b, the confidential default
|
|
19
|
+
// tinfoil/* (TEE key) → tinfoil/gemma4-31b
|
|
20
|
+
// openrouter/* … → the best vision model that provider serves
|
|
21
|
+
//
|
|
22
|
+
// Never a different provider. Crossing providers would ship the user's screenshot
|
|
23
|
+
// to a company they did not pick for this session — on a privacy-first agent that
|
|
24
|
+
// is not a fallback, it's a leak. If the current provider has no vision model with
|
|
25
|
+
// working auth, nothing is rewritten and Pi's existing "image omitted" behaviour
|
|
26
|
+
// stands. PRIVATEER_VISION_MODEL=<provider/id> overrides the choice (and is the one
|
|
27
|
+
// deliberate way to cross providers).
|
|
28
|
+
|
|
29
|
+
import { createHash } from "node:crypto";
|
|
30
|
+
import { ACCOUNT_DEFAULT_MODEL_ID, TINFOIL_MODEL_ID } from "./defaultModel.ts";
|
|
31
|
+
|
|
32
|
+
export interface VisionModel {
|
|
33
|
+
provider: string;
|
|
34
|
+
id: string;
|
|
35
|
+
input: string[];
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** The slice of Pi's ModelRegistry this module needs (kept structural). */
|
|
39
|
+
export interface VisionRegistry {
|
|
40
|
+
getAvailable(): VisionModel[];
|
|
41
|
+
find(provider: string, id: string): VisionModel | undefined;
|
|
42
|
+
hasConfiguredAuth(model: VisionModel): boolean;
|
|
43
|
+
complete(model: any, context: any, options?: any): Promise<{ content: any[]; stopReason?: string; errorMessage?: string }>;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
const specOf = (m: { provider: string; id: string }) => `${m.provider}/${m.id}`;
|
|
47
|
+
const seesImages = (m: { input?: string[] }) => Array.isArray(m.input) && m.input.includes("image");
|
|
48
|
+
|
|
49
|
+
// Per-provider first choices, tried before "any vision model on this provider". Only
|
|
50
|
+
// providers where the pick matters are listed: on the account channel it keeps the
|
|
51
|
+
// image inside the attested Tinfoil enclave rather than whichever vision model sorts
|
|
52
|
+
// first; elsewhere the registry's own list is good enough.
|
|
53
|
+
const PREFERRED: Record<string, string[]> = {
|
|
54
|
+
privateer: [ACCOUNT_DEFAULT_MODEL_ID],
|
|
55
|
+
tinfoil: [TINFOIL_MODEL_ID.replace(/^tinfoil\//, "")],
|
|
56
|
+
openrouter: ["google/gemini-2.5-flash", "openai/gpt-4o-mini", "anthropic/claude-haiku-4.5"],
|
|
57
|
+
};
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* The model that should look at images for `current`, or undefined when there is no
|
|
61
|
+
* reachable vision model on the same key. Returns undefined for a model that can
|
|
62
|
+
* already see — there is nothing to delegate.
|
|
63
|
+
*/
|
|
64
|
+
export function pickVisionDelegate(
|
|
65
|
+
current: VisionModel | undefined,
|
|
66
|
+
registry: VisionRegistry,
|
|
67
|
+
env: NodeJS.ProcessEnv = process.env,
|
|
68
|
+
): VisionModel | undefined {
|
|
69
|
+
if (!current || seesImages(current)) return undefined;
|
|
70
|
+
const usable = (m: VisionModel | undefined): m is VisionModel =>
|
|
71
|
+
!!m && seesImages(m) && registry.hasConfiguredAuth(m);
|
|
72
|
+
|
|
73
|
+
const override = env.PRIVATEER_VISION_MODEL?.trim();
|
|
74
|
+
if (override) {
|
|
75
|
+
const slash = override.indexOf("/");
|
|
76
|
+
const m = slash > 0 ? registry.find(override.slice(0, slash), override.slice(slash + 1)) : undefined;
|
|
77
|
+
if (usable(m)) return m;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
for (const id of PREFERRED[current.provider] ?? []) {
|
|
81
|
+
const m = registry.find(current.provider, id);
|
|
82
|
+
if (usable(m)) return m;
|
|
83
|
+
}
|
|
84
|
+
return registry.getAvailable().find((m) => m.provider === current.provider && usable(m));
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// Pi's read tool appends this to an image it knows the model can't see. Once we've
|
|
88
|
+
// described the image the note is false, and leaving it in invites the model to tell
|
|
89
|
+
// the user it couldn't look.
|
|
90
|
+
const OMITTED_NOTE = /\n?\[Current model does not support images\. The image will be omitted from this request\.\]/g;
|
|
91
|
+
|
|
92
|
+
const SYSTEM_PROMPT = [
|
|
93
|
+
"You are the eyes for another AI model that cannot see images.",
|
|
94
|
+
"Describe the image so that model can act on it without seeing it.",
|
|
95
|
+
"Transcribe ALL legible text verbatim (code, error messages, labels, numbers), preserving line breaks.",
|
|
96
|
+
"Then describe layout, UI elements and their state, colors where meaningful, charts/diagrams and what they show, and anything unusual.",
|
|
97
|
+
"Be precise and factual. Do not speculate beyond what is visible. No preamble.",
|
|
98
|
+
].join(" ");
|
|
99
|
+
|
|
100
|
+
type Block = { type: string; text?: string; data?: string; mimeType?: string };
|
|
101
|
+
|
|
102
|
+
const cache = new Map<string, Promise<string>>();
|
|
103
|
+
|
|
104
|
+
/** Test hook: forget every cached description. */
|
|
105
|
+
export function clearVisionCache(): void {
|
|
106
|
+
cache.clear();
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function describe(
|
|
110
|
+
registry: VisionRegistry,
|
|
111
|
+
delegate: VisionModel,
|
|
112
|
+
image: Block,
|
|
113
|
+
hint: string,
|
|
114
|
+
signal?: AbortSignal,
|
|
115
|
+
): Promise<string> {
|
|
116
|
+
const key = `${specOf(delegate)}:${createHash("sha256").update(image.data ?? "").digest("hex")}`;
|
|
117
|
+
const hit = cache.get(key);
|
|
118
|
+
if (hit) return hit;
|
|
119
|
+
const job = (async () => {
|
|
120
|
+
const res = await registry.complete(
|
|
121
|
+
delegate,
|
|
122
|
+
{
|
|
123
|
+
systemPrompt: SYSTEM_PROMPT,
|
|
124
|
+
messages: [
|
|
125
|
+
{
|
|
126
|
+
role: "user",
|
|
127
|
+
content: [
|
|
128
|
+
{ type: "text", text: hint ? `Context it appeared in:\n${hint.slice(0, 2000)}\n\nDescribe this image.` : "Describe this image." },
|
|
129
|
+
{ type: "image", data: image.data, mimeType: image.mimeType },
|
|
130
|
+
],
|
|
131
|
+
timestamp: Date.now(),
|
|
132
|
+
},
|
|
133
|
+
],
|
|
134
|
+
},
|
|
135
|
+
{ signal, cacheRetention: "none" },
|
|
136
|
+
);
|
|
137
|
+
if (res.stopReason === "error" || res.stopReason === "aborted") {
|
|
138
|
+
throw new Error(res.errorMessage || `vision model ${res.stopReason}`);
|
|
139
|
+
}
|
|
140
|
+
const text = res.content.filter((c) => c?.type === "text").map((c) => c.text).join("\n").trim();
|
|
141
|
+
if (!text) throw new Error("vision model returned no description");
|
|
142
|
+
return text;
|
|
143
|
+
})();
|
|
144
|
+
cache.set(key, job);
|
|
145
|
+
// A failure must not stick: the next turn should get another try.
|
|
146
|
+
job.catch(() => cache.delete(key));
|
|
147
|
+
return job;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Rewrite `messages` so every image block becomes a text description from
|
|
152
|
+
* `delegate`. Returns undefined when there was nothing to rewrite. Never throws: an
|
|
153
|
+
* image that can't be described becomes a note saying so (Pi would have dropped it
|
|
154
|
+
* anyway), so one bad image can't fail the turn.
|
|
155
|
+
*/
|
|
156
|
+
export async function describeImagesInMessages(
|
|
157
|
+
messages: any[],
|
|
158
|
+
delegate: VisionModel,
|
|
159
|
+
registry: VisionRegistry,
|
|
160
|
+
opts: { currentSpec: string; signal?: AbortSignal; onDescribe?: (count: number) => void } = { currentSpec: "" },
|
|
161
|
+
): Promise<any[] | undefined> {
|
|
162
|
+
const jobs: Array<{ block: Block; hint: string }> = [];
|
|
163
|
+
for (const msg of messages) {
|
|
164
|
+
if (!msg || msg.role === "assistant" || !Array.isArray(msg.content)) continue;
|
|
165
|
+
const hint = (msg.content as Block[])
|
|
166
|
+
.filter((b) => b?.type === "text")
|
|
167
|
+
.map((b) => b.text ?? "")
|
|
168
|
+
.join("\n")
|
|
169
|
+
.replace(OMITTED_NOTE, "");
|
|
170
|
+
for (const b of msg.content as Block[]) if (b?.type === "image" && b.data) jobs.push({ block: b, hint });
|
|
171
|
+
}
|
|
172
|
+
if (jobs.length === 0) return undefined;
|
|
173
|
+
opts.onDescribe?.(jobs.length);
|
|
174
|
+
|
|
175
|
+
const results = new Map<Block, string>();
|
|
176
|
+
await Promise.all(
|
|
177
|
+
jobs.map(async ({ block, hint }) => {
|
|
178
|
+
try {
|
|
179
|
+
const text = await describe(registry, delegate, block, hint, opts.signal);
|
|
180
|
+
results.set(block, `[Image — ${opts.currentSpec || "this model"} can't see images, so ${specOf(delegate)} described it:]\n${text}`);
|
|
181
|
+
} catch (err) {
|
|
182
|
+
const why = err instanceof Error ? err.message : String(err);
|
|
183
|
+
results.set(block, `[Image omitted — ${specOf(delegate)} could not describe it: ${why}]`);
|
|
184
|
+
}
|
|
185
|
+
}),
|
|
186
|
+
);
|
|
187
|
+
|
|
188
|
+
return messages.map((msg) => {
|
|
189
|
+
if (!msg || msg.role === "assistant" || !Array.isArray(msg.content)) return msg;
|
|
190
|
+
if (!(msg.content as Block[]).some((b) => results.has(b))) return msg;
|
|
191
|
+
const content = (msg.content as Block[]).map((b) => {
|
|
192
|
+
const described = results.get(b);
|
|
193
|
+
if (described !== undefined) return { type: "text", text: described };
|
|
194
|
+
if (b?.type === "text" && b.text) return { ...b, text: b.text.replace(OMITTED_NOTE, "") };
|
|
195
|
+
return b;
|
|
196
|
+
});
|
|
197
|
+
return { ...msg, content };
|
|
198
|
+
});
|
|
199
|
+
}
|
package/src/tools/media.ts
CHANGED
|
@@ -738,6 +738,10 @@ interface AudioResponse {
|
|
|
738
738
|
audioBase64?: string;
|
|
739
739
|
mimeType?: string;
|
|
740
740
|
model?: string;
|
|
741
|
+
/** The exact wire voice the provider actually received (post-resolution —
|
|
742
|
+
* e.g. an aura-2 override of 'jupiter' comes back as 'aura-2-jupiter-en').
|
|
743
|
+
* Absent only on models that take no voice at all. */
|
|
744
|
+
voice?: string;
|
|
741
745
|
/** Present only for the models that take a length — an sfx model always does. */
|
|
742
746
|
durationSeconds?: number;
|
|
743
747
|
}
|
|
@@ -749,11 +753,21 @@ export const generateSpeechToolDefinition = {
|
|
|
749
753
|
"Turn text into spoken audio and save it to disk. Use it to narrate a video you are assembling, or " +
|
|
750
754
|
"to produce a spoken version of a written answer. Billed to the user's Privateer account; the " +
|
|
751
755
|
"account's default voice model is a confidential-compute one, so the text is processed inside an " +
|
|
752
|
-
"enclave rather than by a retaining provider. Mux the result onto video with video_compose."
|
|
756
|
+
"enclave rather than by a retaining provider. Mux the result onto video with video_compose. " +
|
|
757
|
+
"Call media_capabilities first if you want to override `voice` or `model` — it lists the exact " +
|
|
758
|
+
"wire voice ids for the account's current TTS model (and any model you pass it), which is not " +
|
|
759
|
+
"the same as a spoken character or brand name.",
|
|
753
760
|
parameters: Type.Object({
|
|
754
761
|
text: Type.String({ description: "The words to speak. Write them as they should be read aloud." }),
|
|
755
762
|
path: Type.String({ description: "Where to write the audio, relative to cwd or absolute (e.g. 'audio/narration.mp3')." }),
|
|
756
|
-
voice: Type.Optional(Type.String({
|
|
763
|
+
voice: Type.Optional(Type.String({
|
|
764
|
+
description:
|
|
765
|
+
"Voice id, if the account's TTS model offers a choice. Leave unset for its default. These are " +
|
|
766
|
+
"PROVIDER wire ids, not free text — e.g. Deepgram Aura-2 voices are 'aura-2-<name>-<lang>' " +
|
|
767
|
+
"('aura-2-jupiter-en', not 'jupiter' or 'Jupiter'). Get the exact legal list for the model in " +
|
|
768
|
+
"play from media_capabilities' `speech.voices` before guessing one; an unrecognised id is " +
|
|
769
|
+
"refused with VOICE_UNSUPPORTED rather than silently served on a different voice.",
|
|
770
|
+
})),
|
|
757
771
|
model: Type.Optional(Type.String({ description: "Override the account's text-to-speech model." })),
|
|
758
772
|
}),
|
|
759
773
|
async execute(
|
|
@@ -783,7 +797,13 @@ export const generateSpeechToolDefinition = {
|
|
|
783
797
|
const target = abs(cwd, params.path);
|
|
784
798
|
const ext = extname(target) || extForMime(r.data.mimeType ?? "", ".mp3");
|
|
785
799
|
const out = `${target.slice(0, target.length - extname(target).length)}${ext}`;
|
|
786
|
-
|
|
800
|
+
// Report what actually ran, not what was requested — the two can differ
|
|
801
|
+
// (a resolved default, a normalized voice id) and the caller has no other
|
|
802
|
+
// way to find out short of listening to the file.
|
|
803
|
+
const via = [r.data.model, r.data.voice].filter(Boolean).join(" / ");
|
|
804
|
+
return text(
|
|
805
|
+
`Generated speech${via ? ` with ${via}` : ""}: ${writeOut(out, Buffer.from(r.data.audioBase64, "base64"))}`,
|
|
806
|
+
);
|
|
787
807
|
},
|
|
788
808
|
};
|
|
789
809
|
|
|
@@ -951,6 +971,20 @@ interface CapabilitiesResponse {
|
|
|
951
971
|
* the account's own privacy setting. They are different refusals with different
|
|
952
972
|
* remedies, and only one of them is the user's to fix. */
|
|
953
973
|
sfx?: { model?: string; configured?: boolean; blockedByZdr?: boolean; maxDurationSeconds?: number };
|
|
974
|
+
/** `voices` (on the described model) is populated only for models with an
|
|
975
|
+
* enumerable, closed voice set (Deepgram Aura-2, Tinfoil, fal) — pass
|
|
976
|
+
* `ttsModel` to describe one other than the account default. Empty for
|
|
977
|
+
* models whose voices aren't data this server holds (Gemini, OpenAI-shaped);
|
|
978
|
+
* guessing one there is on you. `catalog` is EVERY TTS model this account
|
|
979
|
+
* can reach, summarized — the only place other than the default to
|
|
980
|
+
* discover an id, since there is no TTS picker in this CLI. */
|
|
981
|
+
speech?: {
|
|
982
|
+
model?: string; blockedByZdr?: boolean; voices?: string[];
|
|
983
|
+
catalog?: {
|
|
984
|
+
id: string; name?: string; provider?: string;
|
|
985
|
+
isZdr?: boolean; isTee?: boolean; blockedByZdr?: boolean; voiceCount?: number;
|
|
986
|
+
}[];
|
|
987
|
+
};
|
|
954
988
|
privacy?: { requireZdr?: boolean; allowNonZdrMedia?: boolean };
|
|
955
989
|
}
|
|
956
990
|
|
|
@@ -1060,6 +1094,44 @@ export function describeSfx(sfx: CapabilitiesResponse["sfx"]): string[] {
|
|
|
1060
1094
|
];
|
|
1061
1095
|
}
|
|
1062
1096
|
|
|
1097
|
+
/**
|
|
1098
|
+
* Speech, worded the same way describeSfx is: a [BLOCKED] model is the
|
|
1099
|
+
* account's own ZDR setting, not something retrying fixes. `voices` prints
|
|
1100
|
+
* only when the model publishes an enumerable set — printing all 90 of
|
|
1101
|
+
* Aura-2's inline is the whole point (a bare character name is not a legal
|
|
1102
|
+
* id on the wire), but a model with none gets no fabricated list either.
|
|
1103
|
+
*/
|
|
1104
|
+
export function describeSpeech(speech: CapabilitiesResponse["speech"]): string[] {
|
|
1105
|
+
const blocked = speech?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : "";
|
|
1106
|
+
const lines = [`Speech model: ${speech?.model ?? "unknown"}${blocked}`];
|
|
1107
|
+
if (speech?.blockedByZdr) {
|
|
1108
|
+
lines.push(
|
|
1109
|
+
" This voice model is non-ZDR, so this account cannot use it until its owner enables non-ZDR " +
|
|
1110
|
+
"media (Settings → Privacy). Omit `model` to use the account's confidential-compute default " +
|
|
1111
|
+
"instead of retrying this one.",
|
|
1112
|
+
);
|
|
1113
|
+
}
|
|
1114
|
+
if (speech?.voices?.length) {
|
|
1115
|
+
lines.push(
|
|
1116
|
+
` ${speech.voices.length} voice id(s) for this model — pass one of these EXACTLY as generate_speech's ` +
|
|
1117
|
+
`\`voice\`, never a guessed name: ${speech.voices.join(", ")}`,
|
|
1118
|
+
);
|
|
1119
|
+
} else {
|
|
1120
|
+
lines.push(" No enumerable voice list for this model; omit `voice` to use its default.");
|
|
1121
|
+
}
|
|
1122
|
+
if (speech?.catalog?.length) {
|
|
1123
|
+
lines.push(" every TTS model available (pass `ttsModel` to this tool to see its voices):");
|
|
1124
|
+
for (const m of speech.catalog) {
|
|
1125
|
+
const tags = [m.isTee ? "confidential" : m.isZdr ? "ZDR" : "non-ZDR", `${m.voiceCount ?? 0} voice(s)`];
|
|
1126
|
+
if (m.blockedByZdr) tags.push("BLOCKED");
|
|
1127
|
+
lines.push(
|
|
1128
|
+
` ${m.id} (${tags.join(", ")})${m.id === speech.model ? " [described above]" : ""}`,
|
|
1129
|
+
);
|
|
1130
|
+
}
|
|
1131
|
+
}
|
|
1132
|
+
return lines;
|
|
1133
|
+
}
|
|
1134
|
+
|
|
1063
1135
|
export const mediaCapabilitiesToolDefinition = {
|
|
1064
1136
|
name: "media_capabilities",
|
|
1065
1137
|
label: "Media Capabilities",
|
|
@@ -1072,7 +1144,10 @@ export const mediaCapabilitiesToolDefinition = {
|
|
|
1072
1144
|
"It is also the ONLY way to find out which 3D models exist and what options each one takes: they " +
|
|
1073
1145
|
"range from $0.14 to $2.41 a mesh and no two take the same options, so call this with `model` set " +
|
|
1074
1146
|
"to the id you are considering BEFORE generate_model, or you will pay the default model's price " +
|
|
1075
|
-
"for a job a cheaper one could have done
|
|
1147
|
+
"for a job a cheaper one could have done.\n" +
|
|
1148
|
+
"It is also the ONLY way to find a text-to-speech model's real voice ids — Deepgram Aura-2's are " +
|
|
1149
|
+
"'aura-2-<name>-<lang>', not a bare character name — so call this with `ttsModel` set BEFORE " +
|
|
1150
|
+
"generate_speech whenever you plan to pass `voice`.",
|
|
1076
1151
|
parameters: Type.Object({
|
|
1077
1152
|
model: Type.Optional(
|
|
1078
1153
|
Type.String({
|
|
@@ -1082,13 +1157,27 @@ export const mediaCapabilitiesToolDefinition = {
|
|
|
1082
1157
|
"their legal values and the price — is per-model.",
|
|
1083
1158
|
}),
|
|
1084
1159
|
),
|
|
1160
|
+
ttsModel: Type.Optional(
|
|
1161
|
+
Type.String({
|
|
1162
|
+
description:
|
|
1163
|
+
"A text-to-speech model id to describe instead of the account default (e.g. 'deepgram/aura-2'). " +
|
|
1164
|
+
"The response's `speech.voices` lists every legal voice id for THIS model — voice ids are not " +
|
|
1165
|
+
"shared across models.",
|
|
1166
|
+
}),
|
|
1167
|
+
),
|
|
1085
1168
|
}),
|
|
1086
|
-
async execute(_toolCallId: string, params: { model?: string }, signal?: AbortSignal) {
|
|
1087
|
-
const query =
|
|
1088
|
-
|
|
1169
|
+
async execute(_toolCallId: string, params: { model?: string; ttsModel?: string }, signal?: AbortSignal) {
|
|
1170
|
+
const query = new URLSearchParams();
|
|
1171
|
+
if (params?.model) query.set("model", params.model);
|
|
1172
|
+
if (params?.ttsModel) query.set("ttsModel", params.ttsModel);
|
|
1173
|
+
const qs = query.toString();
|
|
1174
|
+
const r = await callAccount<CapabilitiesResponse>(
|
|
1175
|
+
`/api/agent/media/capabilities${qs ? `?${qs}` : ""}`,
|
|
1176
|
+
{ method: "GET", signal },
|
|
1177
|
+
);
|
|
1089
1178
|
if (!r.ok) return text(`Could not read media capabilities: ${r.message}`);
|
|
1090
1179
|
|
|
1091
|
-
const { image, video, model3d, sfx, privacy } = r.data;
|
|
1180
|
+
const { image, video, model3d, sfx, speech, privacy } = r.data;
|
|
1092
1181
|
const lines = [
|
|
1093
1182
|
`Image model: ${image?.model ?? "unknown"}${image?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : ""}`,
|
|
1094
1183
|
` up to ${image?.maxPerCall ?? 1} image(s) per call`,
|
|
@@ -1099,18 +1188,19 @@ export const mediaCapabilitiesToolDefinition = {
|
|
|
1099
1188
|
|
|
1100
1189
|
lines.push(...describeModel3d(model3d));
|
|
1101
1190
|
lines.push(...describeSfx(sfx));
|
|
1191
|
+
lines.push(...describeSpeech(speech));
|
|
1102
1192
|
|
|
1103
1193
|
lines.push(`Privacy: requireZdr=${privacy?.requireZdr ?? "?"}, allowNonZdrMedia=${privacy?.allowNonZdrMedia ?? "?"}`);
|
|
1104
|
-
if (image?.blockedByZdr || video?.blockedByZdr || model3d?.blockedByZdr || sfx?.blockedByZdr) {
|
|
1194
|
+
if (image?.blockedByZdr || video?.blockedByZdr || model3d?.blockedByZdr || sfx?.blockedByZdr || speech?.blockedByZdr) {
|
|
1105
1195
|
lines.push(
|
|
1106
1196
|
"A [BLOCKED] model means the account requires Zero Data Retention and that model has no ZDR endpoint. " +
|
|
1107
1197
|
"Only the account owner can change it (Settings → Privacy); do not keep retrying.",
|
|
1108
1198
|
);
|
|
1109
1199
|
}
|
|
1110
1200
|
lines.push(
|
|
1111
|
-
"
|
|
1112
|
-
"
|
|
1113
|
-
"
|
|
1201
|
+
"Music is always available; sound effects and (per above) speech's CURRENT model may not be. Music has " +
|
|
1202
|
+
"no ZDR gate at all; effects are gated like image and video, so a blocked effect must never be " +
|
|
1203
|
+
"answered with music instead.",
|
|
1114
1204
|
);
|
|
1115
1205
|
return text(lines.join("\n"));
|
|
1116
1206
|
},
|