privateer-agent 0.12.44 → 0.12.45
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/cli/chat.ts +12 -3
- package/src/providers/defaultModel.ts +37 -0
- package/src/tools/media.ts +102 -12
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "privateer-agent",
|
|
3
|
-
"version": "0.12.
|
|
3
|
+
"version": "0.12.45",
|
|
4
4
|
"description": "Privacy-first terminal coding agent — bring your own model across 20 providers (Anthropic, OpenAI, OpenRouter, Google, local Ollama…). Safe-by-default permissions, MCP, sub-agents, workflows, and verifiable TEE inference. Built on the Pi toolkit.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
package/src/cli/chat.ts
CHANGED
|
@@ -53,8 +53,14 @@ async function main() {
|
|
|
53
53
|
const { modelRegistryOf } = await import("../providers/piAuthStore.ts");
|
|
54
54
|
const { agentVersion } = await import("../config/version.ts");
|
|
55
55
|
const { pickerCatalog, hiddenAccountNotice, hiddenAccountTitleSuffix } = await import("../providers/modelCatalog.ts");
|
|
56
|
-
const {
|
|
57
|
-
|
|
56
|
+
const {
|
|
57
|
+
resolveDefaultModel,
|
|
58
|
+
resolveSignedInModel,
|
|
59
|
+
savedPiDefaultSpec,
|
|
60
|
+
writePiDefaultModel,
|
|
61
|
+
visionWarningAcknowledged,
|
|
62
|
+
acknowledgeVisionWarning,
|
|
63
|
+
} = await import("../providers/defaultModel.ts");
|
|
58
64
|
const { acceptsImages } = await import("../providers/vision.ts");
|
|
59
65
|
const { postOutbox } = await import("../outbox/cloudOutbox.ts");
|
|
60
66
|
const { addPendingCloud } = await import("../routines/store.ts");
|
|
@@ -440,8 +446,11 @@ async function main() {
|
|
|
440
446
|
// blocks for a model that doesn't declare the modality with NO error — `read` on a
|
|
441
447
|
// screenshot just quietly answers about nothing — so make that failure visible
|
|
442
448
|
// instead of letting a user discover it turn by turn. /model to switch.
|
|
443
|
-
|
|
449
|
+
// Shown once per distinct non-vision spec (visionWarningAcknowledged), not on every
|
|
450
|
+
// launch — a user who keeps a deliberate non-vision pick already knows the tradeoff.
|
|
451
|
+
if (!acceptsImages(spec) && !visionWarningAcknowledged(spec)) {
|
|
444
452
|
console.log(`${YELLOW}⚠ ${provider}/${modelId} can't see images — @file/read on a picture or video frame will be dropped silently. Run /model to switch.${RESET}`);
|
|
453
|
+
acknowledgeVisionWarning(spec);
|
|
445
454
|
}
|
|
446
455
|
if (noQuarterActive()) applyNoQuarter(true); // launched with --no-quarter: say so up front
|
|
447
456
|
|
|
@@ -227,6 +227,43 @@ export function resolveSignedInModel(env: NodeJS.ProcessEnv = process.env): stri
|
|
|
227
227
|
return resolveDefaultModel({ env, signedIn: true, saved: null });
|
|
228
228
|
}
|
|
229
229
|
|
|
230
|
+
// Whether the startup "can't see images" nag (cli/chat.ts) has already been shown for
|
|
231
|
+
// this exact spec. Without this, a user who deliberately keeps a non-vision saved pick
|
|
232
|
+
// (speed over sight, say) sees the same warning on every single launch forever, which
|
|
233
|
+
// trains people to stop reading warnings at all. Keyed by spec so switching to a
|
|
234
|
+
// DIFFERENT non-vision model warns again once — this tracks "have they seen this
|
|
235
|
+
// specific tradeoff", not "should we ever mention it again".
|
|
236
|
+
export function visionWarningAcknowledged(spec: string): boolean {
|
|
237
|
+
try {
|
|
238
|
+
const raw = readFileSync(join(agentDir(), "settings.json"), "utf8").trim();
|
|
239
|
+
if (!raw) return false;
|
|
240
|
+
const s = JSON.parse(raw) as Record<string, unknown>;
|
|
241
|
+
return s.visionWarningAcknowledgedFor === spec;
|
|
242
|
+
} catch {
|
|
243
|
+
return false;
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
// Record that the nag has been shown for `spec`, so the next launch on the same pick
|
|
248
|
+
// stays quiet. Best-effort and silent like the writers below — losing this write just
|
|
249
|
+
// means the warning repeats once more, never a crash.
|
|
250
|
+
export function acknowledgeVisionWarning(spec: string): void {
|
|
251
|
+
const dir = agentDir();
|
|
252
|
+
const settingsPath = join(dir, "settings.json");
|
|
253
|
+
try {
|
|
254
|
+
mkdirSync(dir, { recursive: true });
|
|
255
|
+
let settings: Record<string, unknown> = {};
|
|
256
|
+
if (existsSync(settingsPath)) {
|
|
257
|
+
const raw = readFileSync(settingsPath, "utf8").trim();
|
|
258
|
+
if (raw) settings = JSON.parse(raw) as Record<string, unknown>;
|
|
259
|
+
}
|
|
260
|
+
settings.visionWarningAcknowledgedFor = spec;
|
|
261
|
+
writeFileSync(settingsPath, `${JSON.stringify(settings, null, 2)}\n`);
|
|
262
|
+
} catch {
|
|
263
|
+
// best-effort: the warning already printed either way.
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
230
267
|
// Split a "provider/id" spec on its first slash (model ids themselves contain "/", so
|
|
231
268
|
// only the first delimiter separates provider from model). Returns null for a spec
|
|
232
269
|
// with no provider prefix.
|
package/src/tools/media.ts
CHANGED
|
@@ -738,6 +738,10 @@ interface AudioResponse {
|
|
|
738
738
|
audioBase64?: string;
|
|
739
739
|
mimeType?: string;
|
|
740
740
|
model?: string;
|
|
741
|
+
/** The exact wire voice the provider actually received (post-resolution —
|
|
742
|
+
* e.g. an aura-2 override of 'jupiter' comes back as 'aura-2-jupiter-en').
|
|
743
|
+
* Absent only on models that take no voice at all. */
|
|
744
|
+
voice?: string;
|
|
741
745
|
/** Present only for the models that take a length — an sfx model always does. */
|
|
742
746
|
durationSeconds?: number;
|
|
743
747
|
}
|
|
@@ -749,11 +753,21 @@ export const generateSpeechToolDefinition = {
|
|
|
749
753
|
"Turn text into spoken audio and save it to disk. Use it to narrate a video you are assembling, or " +
|
|
750
754
|
"to produce a spoken version of a written answer. Billed to the user's Privateer account; the " +
|
|
751
755
|
"account's default voice model is a confidential-compute one, so the text is processed inside an " +
|
|
752
|
-
"enclave rather than by a retaining provider. Mux the result onto video with video_compose."
|
|
756
|
+
"enclave rather than by a retaining provider. Mux the result onto video with video_compose. " +
|
|
757
|
+
"Call media_capabilities first if you want to override `voice` or `model` — it lists the exact " +
|
|
758
|
+
"wire voice ids for the account's current TTS model (and any model you pass it), which is not " +
|
|
759
|
+
"the same as a spoken character or brand name.",
|
|
753
760
|
parameters: Type.Object({
|
|
754
761
|
text: Type.String({ description: "The words to speak. Write them as they should be read aloud." }),
|
|
755
762
|
path: Type.String({ description: "Where to write the audio, relative to cwd or absolute (e.g. 'audio/narration.mp3')." }),
|
|
756
|
-
voice: Type.Optional(Type.String({
|
|
763
|
+
voice: Type.Optional(Type.String({
|
|
764
|
+
description:
|
|
765
|
+
"Voice id, if the account's TTS model offers a choice. Leave unset for its default. These are " +
|
|
766
|
+
"PROVIDER wire ids, not free text — e.g. Deepgram Aura-2 voices are 'aura-2-<name>-<lang>' " +
|
|
767
|
+
"('aura-2-jupiter-en', not 'jupiter' or 'Jupiter'). Get the exact legal list for the model in " +
|
|
768
|
+
"play from media_capabilities' `speech.voices` before guessing one; an unrecognised id is " +
|
|
769
|
+
"refused with VOICE_UNSUPPORTED rather than silently served on a different voice.",
|
|
770
|
+
})),
|
|
757
771
|
model: Type.Optional(Type.String({ description: "Override the account's text-to-speech model." })),
|
|
758
772
|
}),
|
|
759
773
|
async execute(
|
|
@@ -783,7 +797,13 @@ export const generateSpeechToolDefinition = {
|
|
|
783
797
|
const target = abs(cwd, params.path);
|
|
784
798
|
const ext = extname(target) || extForMime(r.data.mimeType ?? "", ".mp3");
|
|
785
799
|
const out = `${target.slice(0, target.length - extname(target).length)}${ext}`;
|
|
786
|
-
|
|
800
|
+
// Report what actually ran, not what was requested — the two can differ
|
|
801
|
+
// (a resolved default, a normalized voice id) and the caller has no other
|
|
802
|
+
// way to find out short of listening to the file.
|
|
803
|
+
const via = [r.data.model, r.data.voice].filter(Boolean).join(" / ");
|
|
804
|
+
return text(
|
|
805
|
+
`Generated speech${via ? ` with ${via}` : ""}: ${writeOut(out, Buffer.from(r.data.audioBase64, "base64"))}`,
|
|
806
|
+
);
|
|
787
807
|
},
|
|
788
808
|
};
|
|
789
809
|
|
|
@@ -951,6 +971,20 @@ interface CapabilitiesResponse {
|
|
|
951
971
|
* the account's own privacy setting. They are different refusals with different
|
|
952
972
|
* remedies, and only one of them is the user's to fix. */
|
|
953
973
|
sfx?: { model?: string; configured?: boolean; blockedByZdr?: boolean; maxDurationSeconds?: number };
|
|
974
|
+
/** `voices` (on the described model) is populated only for models with an
|
|
975
|
+
* enumerable, closed voice set (Deepgram Aura-2, Tinfoil, fal) — pass
|
|
976
|
+
* `ttsModel` to describe one other than the account default. Empty for
|
|
977
|
+
* models whose voices aren't data this server holds (Gemini, OpenAI-shaped);
|
|
978
|
+
* guessing one there is on you. `catalog` is EVERY TTS model this account
|
|
979
|
+
* can reach, summarized — the only place other than the default to
|
|
980
|
+
* discover an id, since there is no TTS picker in this CLI. */
|
|
981
|
+
speech?: {
|
|
982
|
+
model?: string; blockedByZdr?: boolean; voices?: string[];
|
|
983
|
+
catalog?: {
|
|
984
|
+
id: string; name?: string; provider?: string;
|
|
985
|
+
isZdr?: boolean; isTee?: boolean; blockedByZdr?: boolean; voiceCount?: number;
|
|
986
|
+
}[];
|
|
987
|
+
};
|
|
954
988
|
privacy?: { requireZdr?: boolean; allowNonZdrMedia?: boolean };
|
|
955
989
|
}
|
|
956
990
|
|
|
@@ -1060,6 +1094,44 @@ export function describeSfx(sfx: CapabilitiesResponse["sfx"]): string[] {
|
|
|
1060
1094
|
];
|
|
1061
1095
|
}
|
|
1062
1096
|
|
|
1097
|
+
/**
|
|
1098
|
+
* Speech, worded the same way describeSfx is: a [BLOCKED] model is the
|
|
1099
|
+
* account's own ZDR setting, not something retrying fixes. `voices` prints
|
|
1100
|
+
* only when the model publishes an enumerable set — printing all 90 of
|
|
1101
|
+
* Aura-2's inline is the whole point (a bare character name is not a legal
|
|
1102
|
+
* id on the wire), but a model with none gets no fabricated list either.
|
|
1103
|
+
*/
|
|
1104
|
+
export function describeSpeech(speech: CapabilitiesResponse["speech"]): string[] {
|
|
1105
|
+
const blocked = speech?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : "";
|
|
1106
|
+
const lines = [`Speech model: ${speech?.model ?? "unknown"}${blocked}`];
|
|
1107
|
+
if (speech?.blockedByZdr) {
|
|
1108
|
+
lines.push(
|
|
1109
|
+
" This voice model is non-ZDR, so this account cannot use it until its owner enables non-ZDR " +
|
|
1110
|
+
"media (Settings → Privacy). Omit `model` to use the account's confidential-compute default " +
|
|
1111
|
+
"instead of retrying this one.",
|
|
1112
|
+
);
|
|
1113
|
+
}
|
|
1114
|
+
if (speech?.voices?.length) {
|
|
1115
|
+
lines.push(
|
|
1116
|
+
` ${speech.voices.length} voice id(s) for this model — pass one of these EXACTLY as generate_speech's ` +
|
|
1117
|
+
`\`voice\`, never a guessed name: ${speech.voices.join(", ")}`,
|
|
1118
|
+
);
|
|
1119
|
+
} else {
|
|
1120
|
+
lines.push(" No enumerable voice list for this model; omit `voice` to use its default.");
|
|
1121
|
+
}
|
|
1122
|
+
if (speech?.catalog?.length) {
|
|
1123
|
+
lines.push(" every TTS model available (pass `ttsModel` to this tool to see its voices):");
|
|
1124
|
+
for (const m of speech.catalog) {
|
|
1125
|
+
const tags = [m.isTee ? "confidential" : m.isZdr ? "ZDR" : "non-ZDR", `${m.voiceCount ?? 0} voice(s)`];
|
|
1126
|
+
if (m.blockedByZdr) tags.push("BLOCKED");
|
|
1127
|
+
lines.push(
|
|
1128
|
+
` ${m.id} (${tags.join(", ")})${m.id === speech.model ? " [described above]" : ""}`,
|
|
1129
|
+
);
|
|
1130
|
+
}
|
|
1131
|
+
}
|
|
1132
|
+
return lines;
|
|
1133
|
+
}
|
|
1134
|
+
|
|
1063
1135
|
export const mediaCapabilitiesToolDefinition = {
|
|
1064
1136
|
name: "media_capabilities",
|
|
1065
1137
|
label: "Media Capabilities",
|
|
@@ -1072,7 +1144,10 @@ export const mediaCapabilitiesToolDefinition = {
|
|
|
1072
1144
|
"It is also the ONLY way to find out which 3D models exist and what options each one takes: they " +
|
|
1073
1145
|
"range from $0.14 to $2.41 a mesh and no two take the same options, so call this with `model` set " +
|
|
1074
1146
|
"to the id you are considering BEFORE generate_model, or you will pay the default model's price " +
|
|
1075
|
-
"for a job a cheaper one could have done
|
|
1147
|
+
"for a job a cheaper one could have done.\n" +
|
|
1148
|
+
"It is also the ONLY way to find a text-to-speech model's real voice ids — Deepgram Aura-2's are " +
|
|
1149
|
+
"'aura-2-<name>-<lang>', not a bare character name — so call this with `ttsModel` set BEFORE " +
|
|
1150
|
+
"generate_speech whenever you plan to pass `voice`.",
|
|
1076
1151
|
parameters: Type.Object({
|
|
1077
1152
|
model: Type.Optional(
|
|
1078
1153
|
Type.String({
|
|
@@ -1082,13 +1157,27 @@ export const mediaCapabilitiesToolDefinition = {
|
|
|
1082
1157
|
"their legal values and the price — is per-model.",
|
|
1083
1158
|
}),
|
|
1084
1159
|
),
|
|
1160
|
+
ttsModel: Type.Optional(
|
|
1161
|
+
Type.String({
|
|
1162
|
+
description:
|
|
1163
|
+
"A text-to-speech model id to describe instead of the account default (e.g. 'deepgram/aura-2'). " +
|
|
1164
|
+
"The response's `speech.voices` lists every legal voice id for THIS model — voice ids are not " +
|
|
1165
|
+
"shared across models.",
|
|
1166
|
+
}),
|
|
1167
|
+
),
|
|
1085
1168
|
}),
|
|
1086
|
-
async execute(_toolCallId: string, params: { model?: string }, signal?: AbortSignal) {
|
|
1087
|
-
const query =
|
|
1088
|
-
|
|
1169
|
+
async execute(_toolCallId: string, params: { model?: string; ttsModel?: string }, signal?: AbortSignal) {
|
|
1170
|
+
const query = new URLSearchParams();
|
|
1171
|
+
if (params?.model) query.set("model", params.model);
|
|
1172
|
+
if (params?.ttsModel) query.set("ttsModel", params.ttsModel);
|
|
1173
|
+
const qs = query.toString();
|
|
1174
|
+
const r = await callAccount<CapabilitiesResponse>(
|
|
1175
|
+
`/api/agent/media/capabilities${qs ? `?${qs}` : ""}`,
|
|
1176
|
+
{ method: "GET", signal },
|
|
1177
|
+
);
|
|
1089
1178
|
if (!r.ok) return text(`Could not read media capabilities: ${r.message}`);
|
|
1090
1179
|
|
|
1091
|
-
const { image, video, model3d, sfx, privacy } = r.data;
|
|
1180
|
+
const { image, video, model3d, sfx, speech, privacy } = r.data;
|
|
1092
1181
|
const lines = [
|
|
1093
1182
|
`Image model: ${image?.model ?? "unknown"}${image?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : ""}`,
|
|
1094
1183
|
` up to ${image?.maxPerCall ?? 1} image(s) per call`,
|
|
@@ -1099,18 +1188,19 @@ export const mediaCapabilitiesToolDefinition = {
|
|
|
1099
1188
|
|
|
1100
1189
|
lines.push(...describeModel3d(model3d));
|
|
1101
1190
|
lines.push(...describeSfx(sfx));
|
|
1191
|
+
lines.push(...describeSpeech(speech));
|
|
1102
1192
|
|
|
1103
1193
|
lines.push(`Privacy: requireZdr=${privacy?.requireZdr ?? "?"}, allowNonZdrMedia=${privacy?.allowNonZdrMedia ?? "?"}`);
|
|
1104
|
-
if (image?.blockedByZdr || video?.blockedByZdr || model3d?.blockedByZdr || sfx?.blockedByZdr) {
|
|
1194
|
+
if (image?.blockedByZdr || video?.blockedByZdr || model3d?.blockedByZdr || sfx?.blockedByZdr || speech?.blockedByZdr) {
|
|
1105
1195
|
lines.push(
|
|
1106
1196
|
"A [BLOCKED] model means the account requires Zero Data Retention and that model has no ZDR endpoint. " +
|
|
1107
1197
|
"Only the account owner can change it (Settings → Privacy); do not keep retrying.",
|
|
1108
1198
|
);
|
|
1109
1199
|
}
|
|
1110
1200
|
lines.push(
|
|
1111
|
-
"
|
|
1112
|
-
"
|
|
1113
|
-
"
|
|
1201
|
+
"Music is always available; sound effects and (per above) speech's CURRENT model may not be. Music has " +
|
|
1202
|
+
"no ZDR gate at all; effects are gated like image and video, so a blocked effect must never be " +
|
|
1203
|
+
"answered with music instead.",
|
|
1114
1204
|
);
|
|
1115
1205
|
return text(lines.join("\n"));
|
|
1116
1206
|
},
|