privateer-agent 0.12.44 → 0.12.45

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "privateer-agent",
3
- "version": "0.12.44",
3
+ "version": "0.12.45",
4
4
  "description": "Privacy-first terminal coding agent — bring your own model across 20 providers (Anthropic, OpenAI, OpenRouter, Google, local Ollama…). Safe-by-default permissions, MCP, sub-agents, workflows, and verifiable TEE inference. Built on the Pi toolkit.",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/cli/chat.ts CHANGED
@@ -53,8 +53,14 @@ async function main() {
53
53
  const { modelRegistryOf } = await import("../providers/piAuthStore.ts");
54
54
  const { agentVersion } = await import("../config/version.ts");
55
55
  const { pickerCatalog, hiddenAccountNotice, hiddenAccountTitleSuffix } = await import("../providers/modelCatalog.ts");
56
- const { resolveDefaultModel, resolveSignedInModel, savedPiDefaultSpec, writePiDefaultModel } =
57
- await import("../providers/defaultModel.ts");
56
+ const {
57
+ resolveDefaultModel,
58
+ resolveSignedInModel,
59
+ savedPiDefaultSpec,
60
+ writePiDefaultModel,
61
+ visionWarningAcknowledged,
62
+ acknowledgeVisionWarning,
63
+ } = await import("../providers/defaultModel.ts");
58
64
  const { acceptsImages } = await import("../providers/vision.ts");
59
65
  const { postOutbox } = await import("../outbox/cloudOutbox.ts");
60
66
  const { addPendingCloud } = await import("../routines/store.ts");
@@ -440,8 +446,11 @@ async function main() {
440
446
  // blocks for a model that doesn't declare the modality with NO error — `read` on a
441
447
  // screenshot just quietly answers about nothing — so make that failure visible
442
448
  // instead of letting a user discover it turn by turn. /model to switch.
443
- if (!acceptsImages(spec)) {
449
+ // Shown once per distinct non-vision spec (visionWarningAcknowledged), not on every
450
+ // launch — a user who keeps a deliberate non-vision pick already knows the tradeoff.
451
+ if (!acceptsImages(spec) && !visionWarningAcknowledged(spec)) {
444
452
  console.log(`${YELLOW}⚠ ${provider}/${modelId} can't see images — @file/read on a picture or video frame will be dropped silently. Run /model to switch.${RESET}`);
453
+ acknowledgeVisionWarning(spec);
445
454
  }
446
455
  if (noQuarterActive()) applyNoQuarter(true); // launched with --no-quarter: say so up front
447
456
 
@@ -227,6 +227,43 @@ export function resolveSignedInModel(env: NodeJS.ProcessEnv = process.env): stri
227
227
  return resolveDefaultModel({ env, signedIn: true, saved: null });
228
228
  }
229
229
 
230
+ // Whether the startup "can't see images" nag (cli/chat.ts) has already been shown for
231
+ // this exact spec. Without this, a user who deliberately keeps a non-vision saved pick
232
+ // (speed over sight, say) sees the same warning on every single launch forever, which
233
+ // trains people to stop reading warnings at all. Keyed by spec so switching to a
234
+ // DIFFERENT non-vision model warns again once — this tracks "have they seen this
235
+ // specific tradeoff", not "should we ever mention it again".
236
+ export function visionWarningAcknowledged(spec: string): boolean {
237
+ try {
238
+ const raw = readFileSync(join(agentDir(), "settings.json"), "utf8").trim();
239
+ if (!raw) return false;
240
+ const s = JSON.parse(raw) as Record<string, unknown>;
241
+ return s.visionWarningAcknowledgedFor === spec;
242
+ } catch {
243
+ return false;
244
+ }
245
+ }
246
+
247
+ // Record that the nag has been shown for `spec`, so the next launch on the same pick
248
+ // stays quiet. Best-effort and silent like the writers below — losing this write just
249
+ // means the warning repeats once more, never a crash.
250
+ export function acknowledgeVisionWarning(spec: string): void {
251
+ const dir = agentDir();
252
+ const settingsPath = join(dir, "settings.json");
253
+ try {
254
+ mkdirSync(dir, { recursive: true });
255
+ let settings: Record<string, unknown> = {};
256
+ if (existsSync(settingsPath)) {
257
+ const raw = readFileSync(settingsPath, "utf8").trim();
258
+ if (raw) settings = JSON.parse(raw) as Record<string, unknown>;
259
+ }
260
+ settings.visionWarningAcknowledgedFor = spec;
261
+ writeFileSync(settingsPath, `${JSON.stringify(settings, null, 2)}\n`);
262
+ } catch {
263
+ // best-effort: the warning already printed either way.
264
+ }
265
+ }
266
+
230
267
  // Split a "provider/id" spec on its first slash (model ids themselves contain "/", so
231
268
  // only the first delimiter separates provider from model). Returns null for a spec
232
269
  // with no provider prefix.
@@ -738,6 +738,10 @@ interface AudioResponse {
738
738
  audioBase64?: string;
739
739
  mimeType?: string;
740
740
  model?: string;
741
+ /** The exact wire voice the provider actually received (post-resolution —
742
+ * e.g. an aura-2 override of 'jupiter' comes back as 'aura-2-jupiter-en').
743
+ * Absent only on models that take no voice at all. */
744
+ voice?: string;
741
745
  /** Present only for the models that take a length — an sfx model always does. */
742
746
  durationSeconds?: number;
743
747
  }
@@ -749,11 +753,21 @@ export const generateSpeechToolDefinition = {
749
753
  "Turn text into spoken audio and save it to disk. Use it to narrate a video you are assembling, or " +
750
754
  "to produce a spoken version of a written answer. Billed to the user's Privateer account; the " +
751
755
  "account's default voice model is a confidential-compute one, so the text is processed inside an " +
752
- "enclave rather than by a retaining provider. Mux the result onto video with video_compose.",
756
+ "enclave rather than by a retaining provider. Mux the result onto video with video_compose. " +
757
+ "Call media_capabilities first if you want to override `voice` or `model` — it lists the exact " +
758
+ "wire voice ids for the account's current TTS model (and any model you pass it), which is not " +
759
+ "the same as a spoken character or brand name.",
753
760
  parameters: Type.Object({
754
761
  text: Type.String({ description: "The words to speak. Write them as they should be read aloud." }),
755
762
  path: Type.String({ description: "Where to write the audio, relative to cwd or absolute (e.g. 'audio/narration.mp3')." }),
756
- voice: Type.Optional(Type.String({ description: "Voice name, if the account's TTS model offers a choice. Leave unset for its default." })),
763
+ voice: Type.Optional(Type.String({
764
+ description:
765
+ "Voice id, if the account's TTS model offers a choice. Leave unset for its default. These are " +
766
+ "PROVIDER wire ids, not free text — e.g. Deepgram Aura-2 voices are 'aura-2-<name>-<lang>' " +
767
+ "('aura-2-jupiter-en', not 'jupiter' or 'Jupiter'). Get the exact legal list for the model in " +
768
+ "play from media_capabilities' `speech.voices` before guessing one; an unrecognised id is " +
769
+ "refused with VOICE_UNSUPPORTED rather than silently served on a different voice.",
770
+ })),
757
771
  model: Type.Optional(Type.String({ description: "Override the account's text-to-speech model." })),
758
772
  }),
759
773
  async execute(
@@ -783,7 +797,13 @@ export const generateSpeechToolDefinition = {
783
797
  const target = abs(cwd, params.path);
784
798
  const ext = extname(target) || extForMime(r.data.mimeType ?? "", ".mp3");
785
799
  const out = `${target.slice(0, target.length - extname(target).length)}${ext}`;
786
- return text(`Generated speech: ${writeOut(out, Buffer.from(r.data.audioBase64, "base64"))}`);
800
+ // Report what actually ran, not what was requested — the two can differ
801
+ // (a resolved default, a normalized voice id) and the caller has no other
802
+ // way to find out short of listening to the file.
803
+ const via = [r.data.model, r.data.voice].filter(Boolean).join(" / ");
804
+ return text(
805
+ `Generated speech${via ? ` with ${via}` : ""}: ${writeOut(out, Buffer.from(r.data.audioBase64, "base64"))}`,
806
+ );
787
807
  },
788
808
  };
789
809
 
@@ -951,6 +971,20 @@ interface CapabilitiesResponse {
951
971
  * the account's own privacy setting. They are different refusals with different
952
972
  * remedies, and only one of them is the user's to fix. */
953
973
  sfx?: { model?: string; configured?: boolean; blockedByZdr?: boolean; maxDurationSeconds?: number };
974
+ /** `voices` (on the described model) is populated only for models with an
975
+ * enumerable, closed voice set (Deepgram Aura-2, Tinfoil, fal) — pass
976
+ * `ttsModel` to describe one other than the account default. Empty for
977
+ * models whose voices aren't data this server holds (Gemini, OpenAI-shaped);
978
+ * guessing one there is on you. `catalog` is EVERY TTS model this account
979
+ * can reach, summarized — the only place other than the default to
980
+ * discover an id, since there is no TTS picker in this CLI. */
981
+ speech?: {
982
+ model?: string; blockedByZdr?: boolean; voices?: string[];
983
+ catalog?: {
984
+ id: string; name?: string; provider?: string;
985
+ isZdr?: boolean; isTee?: boolean; blockedByZdr?: boolean; voiceCount?: number;
986
+ }[];
987
+ };
954
988
  privacy?: { requireZdr?: boolean; allowNonZdrMedia?: boolean };
955
989
  }
956
990
 
@@ -1060,6 +1094,44 @@ export function describeSfx(sfx: CapabilitiesResponse["sfx"]): string[] {
1060
1094
  ];
1061
1095
  }
1062
1096
 
1097
+ /**
1098
+ * Speech, worded the same way describeSfx is: a [BLOCKED] model is the
1099
+ * account's own ZDR setting, not something retrying fixes. `voices` prints
1100
+ * only when the model publishes an enumerable set — printing all 90 of
1101
+ * Aura-2's inline is the whole point (a bare character name is not a legal
1102
+ * id on the wire), but a model with none gets no fabricated list either.
1103
+ */
1104
+ export function describeSpeech(speech: CapabilitiesResponse["speech"]): string[] {
1105
+ const blocked = speech?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : "";
1106
+ const lines = [`Speech model: ${speech?.model ?? "unknown"}${blocked}`];
1107
+ if (speech?.blockedByZdr) {
1108
+ lines.push(
1109
+ " This voice model is non-ZDR, so this account cannot use it until its owner enables non-ZDR " +
1110
+ "media (Settings → Privacy). Omit `model` to use the account's confidential-compute default " +
1111
+ "instead of retrying this one.",
1112
+ );
1113
+ }
1114
+ if (speech?.voices?.length) {
1115
+ lines.push(
1116
+ ` ${speech.voices.length} voice id(s) for this model — pass one of these EXACTLY as generate_speech's ` +
1117
+ `\`voice\`, never a guessed name: ${speech.voices.join(", ")}`,
1118
+ );
1119
+ } else {
1120
+ lines.push(" No enumerable voice list for this model; omit `voice` to use its default.");
1121
+ }
1122
+ if (speech?.catalog?.length) {
1123
+ lines.push(" every TTS model available (pass `ttsModel` to this tool to see its voices):");
1124
+ for (const m of speech.catalog) {
1125
+ const tags = [m.isTee ? "confidential" : m.isZdr ? "ZDR" : "non-ZDR", `${m.voiceCount ?? 0} voice(s)`];
1126
+ if (m.blockedByZdr) tags.push("BLOCKED");
1127
+ lines.push(
1128
+ ` ${m.id} (${tags.join(", ")})${m.id === speech.model ? " [described above]" : ""}`,
1129
+ );
1130
+ }
1131
+ }
1132
+ return lines;
1133
+ }
1134
+
1063
1135
  export const mediaCapabilitiesToolDefinition = {
1064
1136
  name: "media_capabilities",
1065
1137
  label: "Media Capabilities",
@@ -1072,7 +1144,10 @@ export const mediaCapabilitiesToolDefinition = {
1072
1144
  "It is also the ONLY way to find out which 3D models exist and what options each one takes: they " +
1073
1145
  "range from $0.14 to $2.41 a mesh and no two take the same options, so call this with `model` set " +
1074
1146
  "to the id you are considering BEFORE generate_model, or you will pay the default model's price " +
1075
- "for a job a cheaper one could have done.",
1147
+ "for a job a cheaper one could have done.\n" +
1148
+ "It is also the ONLY way to find a text-to-speech model's real voice ids — Deepgram Aura-2's are " +
1149
+ "'aura-2-<name>-<lang>', not a bare character name — so call this with `ttsModel` set BEFORE " +
1150
+ "generate_speech whenever you plan to pass `voice`.",
1076
1151
  parameters: Type.Object({
1077
1152
  model: Type.Optional(
1078
1153
  Type.String({
@@ -1082,13 +1157,27 @@ export const mediaCapabilitiesToolDefinition = {
1082
1157
  "their legal values and the price — is per-model.",
1083
1158
  }),
1084
1159
  ),
1160
+ ttsModel: Type.Optional(
1161
+ Type.String({
1162
+ description:
1163
+ "A text-to-speech model id to describe instead of the account default (e.g. 'deepgram/aura-2'). " +
1164
+ "The response's `speech.voices` lists every legal voice id for THIS model — voice ids are not " +
1165
+ "shared across models.",
1166
+ }),
1167
+ ),
1085
1168
  }),
1086
- async execute(_toolCallId: string, params: { model?: string }, signal?: AbortSignal) {
1087
- const query = params?.model ? `?model=${encodeURIComponent(params.model)}` : "";
1088
- const r = await callAccount<CapabilitiesResponse>(`/api/agent/media/capabilities${query}`, { method: "GET", signal });
1169
+ async execute(_toolCallId: string, params: { model?: string; ttsModel?: string }, signal?: AbortSignal) {
1170
+ const query = new URLSearchParams();
1171
+ if (params?.model) query.set("model", params.model);
1172
+ if (params?.ttsModel) query.set("ttsModel", params.ttsModel);
1173
+ const qs = query.toString();
1174
+ const r = await callAccount<CapabilitiesResponse>(
1175
+ `/api/agent/media/capabilities${qs ? `?${qs}` : ""}`,
1176
+ { method: "GET", signal },
1177
+ );
1089
1178
  if (!r.ok) return text(`Could not read media capabilities: ${r.message}`);
1090
1179
 
1091
- const { image, video, model3d, sfx, privacy } = r.data;
1180
+ const { image, video, model3d, sfx, speech, privacy } = r.data;
1092
1181
  const lines = [
1093
1182
  `Image model: ${image?.model ?? "unknown"}${image?.blockedByZdr ? " [BLOCKED by this account's ZDR setting]" : ""}`,
1094
1183
  ` up to ${image?.maxPerCall ?? 1} image(s) per call`,
@@ -1099,18 +1188,19 @@ export const mediaCapabilitiesToolDefinition = {
1099
1188
 
1100
1189
  lines.push(...describeModel3d(model3d));
1101
1190
  lines.push(...describeSfx(sfx));
1191
+ lines.push(...describeSpeech(speech));
1102
1192
 
1103
1193
  lines.push(`Privacy: requireZdr=${privacy?.requireZdr ?? "?"}, allowNonZdrMedia=${privacy?.allowNonZdrMedia ?? "?"}`);
1104
- if (image?.blockedByZdr || video?.blockedByZdr || model3d?.blockedByZdr || sfx?.blockedByZdr) {
1194
+ if (image?.blockedByZdr || video?.blockedByZdr || model3d?.blockedByZdr || sfx?.blockedByZdr || speech?.blockedByZdr) {
1105
1195
  lines.push(
1106
1196
  "A [BLOCKED] model means the account requires Zero Data Retention and that model has no ZDR endpoint. " +
1107
1197
  "Only the account owner can change it (Settings → Privacy); do not keep retrying.",
1108
1198
  );
1109
1199
  }
1110
1200
  lines.push(
1111
- "Speech and music are always available; sound effects are not (see above). Speech runs confidentially, " +
1112
- "music has no ZDR gate at all, and effects are gated like image and video — so a blocked effect must " +
1113
- "never be answered with music.",
1201
+ "Music is always available; sound effects and (per above) speech's CURRENT model may not be. Music has " +
1202
+ "no ZDR gate at all; effects are gated like image and video, so a blocked effect must never be " +
1203
+ "answered with music instead.",
1114
1204
  );
1115
1205
  return text(lines.join("\n"));
1116
1206
  },