privateer-agent 0.12.34 → 0.12.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -27,6 +27,8 @@ import { join } from "node:path";
27
27
  import { globalDir } from "../config/paths.ts";
28
28
  import { canOpenBrowser, openInBrowser } from "../util/openBrowser.ts";
29
29
  import { installGzipRequestBodies } from "../util/gzipRequestBody.ts";
30
+ import { describeAccountBalanceError } from "../engine/errors.ts";
31
+ import type { AssistantMessage } from "@earendil-works/pi-ai";
30
32
  import { interpretReport, teePosture, tierFromTeePosture, type PrivacyTier } from "pi-privacy";
31
33
  import { ACCOUNT_DEFAULT_MODEL_ID, ACCOUNT_NEAR_MODEL_ID, ensurePiDefaultModel } from "./defaultModel.ts";
32
34
  import { visionInput } from "./vision.ts";
@@ -74,6 +76,29 @@ const DEFAULT_MODELS = [
74
76
  "openai/gpt-5.6-luna",
75
77
  ];
76
78
 
79
+ // What to assume when the server publishes no window for a model — an older server,
80
+ // a model the upstream catalog has no `context_length` for, or an admin-added row.
81
+ // 128000 was the flat value the whole catalog used to be registered with, so keeping
82
+ // it here means "no published window" behaves exactly as it always has, and only a
83
+ // model we have a real number for changes.
84
+ const DEFAULT_CONTEXT_WINDOW = 128000;
85
+
86
+ // A sane band for a published window. The catalog is remote input and the cache is a
87
+ // file on disk, so a garbled or hostile number must not reach pi's budget arithmetic:
88
+ // too small silently strangles every turn, too large defeats compaction by promising
89
+ // room that isn't there. Anything outside the band is treated as unpublished.
90
+ const MIN_PUBLISHED_CONTEXT_WINDOW = 8_192;
91
+ const MAX_PUBLISHED_CONTEXT_WINDOW = 10_000_000;
92
+
93
+ function validContextWindow(value: unknown): number | undefined {
94
+ return typeof value === "number" &&
95
+ Number.isFinite(value) &&
96
+ value >= MIN_PUBLISHED_CONTEXT_WINDOW &&
97
+ value <= MAX_PUBLISHED_CONTEXT_WINDOW
98
+ ? Math.floor(value)
99
+ : undefined;
100
+ }
101
+
77
102
  function seedModel(id: string) {
78
103
  return {
79
104
  id,
@@ -86,7 +111,13 @@ function seedModel(id: string) {
86
111
  // attaches when the user points at a screenshot. See providers/vision.ts.
87
112
  input: visionInput(id),
88
113
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
89
- contextWindow: 128000,
114
+ // The model's real window when the server published one, else the fallback.
115
+ // This is not cosmetic: pi sizes each turn's answer budget as
116
+ // (contextWindow − estimated prompt − 4096) and clamps max_tokens to it
117
+ // (pi-ai clampMaxTokensToContext, and see src/engine/contextBudget.ts). A window
118
+ // that under-states the model's real one therefore throttles the answer — and
119
+ // used to collapse it to a single token — while the model still has room.
120
+ contextWindow: accountContextWindow(id) ?? DEFAULT_CONTEXT_WINDOW,
90
121
  maxTokens: 16384,
91
122
  };
92
123
  }
@@ -144,25 +175,79 @@ const REASONING_EFFORT_THINKING: ThinkingProfile = {
144
175
  thinkingLevelMap: { off: "low", minimal: "low", xhigh: null },
145
176
  };
146
177
 
147
- // Only the TEE prefixes are annotated. Those are enclaves we drive directly and can
148
- // probe. The rest of the catalog is proxied to a third-party gateway whose thinking
149
- // shape we have NOT verified from here, and an unsupported parameter fails the whole
150
- // turn — decisively worse than a turn that thinks too much. They keep the old
151
- // behaviour exactly.
178
+ // The proxied catalog: everything that is NOT a TEE prefix reaches OpenRouter
179
+ // through the account relay, which forwards our body unchanged (treeview
180
+ // services/inferenceService.js proxyChatCompletion — "pass the client body through
181
+ // unchanged except for the fields we own"). OpenRouter normalises reasoning across
182
+ // vendors with a nested `reasoning` object, which is exactly Pi's "openrouter"
183
+ // thinkingFormat, and it derives a token budget from the effort — so the effort dial
184
+ // is also the budget cap.
185
+ //
186
+ // That cap is the point. `reasoning: false` does not mean "does not think": it means
187
+ // Pi sends NO thinking parameter at all, because every thinking branch in pi-ai
188
+ // buildParams is gated on model.reasoning. The model then reasons at the gateway's
189
+ // default — and reasoning shares `maxTokens` with the answer, so an unbounded
190
+ // thinking phase eats the response. Measured across 6,200 real turns: 58–76% of ALL
191
+ // output tokens on these models were reasoning, 158 turns spent ≥90% of the 16,384
192
+ // budget thinking, and 12 produced thousands of reasoning tokens with no answer and
193
+ // no tool call — a five-to-nine-minute spinner ending in "Response was truncated
194
+ // before completion." The annotated TEE models, on the same measurement, sat at 0%.
195
+ //
196
+ // "off" pins to the floor instead of sending `effort: "none"`, for the same reason
197
+ // harmony does above: an unsupported enum fails the whole turn, and this is the one
198
+ // path with no safety net — the app's generateTextStream retries without the hint on
199
+ // a 400 (inferenceService.js), proxyChatCompletion does not. "minimal" is only
200
+ // universal on the GPT-5 family, so it shares the floor. low/medium/high pass
201
+ // through verbatim; xhigh/max stay unmapped so getSupportedThinkingLevels drops them.
202
+ const OPENROUTER_THINKING: ThinkingProfile = {
203
+ reasoning: true,
204
+ thinkingLevelMap: { off: "low", minimal: "low" },
205
+ compat: { thinkingFormat: "openrouter" },
206
+ };
207
+
208
+ // Only the TEE prefixes are driven directly and can be probed from here. The rest is
209
+ // proxied to a third-party gateway, so an id is annotated only when it is
210
+ // unambiguously a thinking model — an allowlist, never a denylist. The live catalog
211
+ // carries 284 ids including roleplay finetunes (sao10k, thedrummer, gryphe) and
212
+ // code-apply models (morph, relace) that do not reason at all, and being wrong here
213
+ // costs the whole turn rather than just a slow one.
152
214
  //
215
+ // Every family below was measured burning its budget in the session logs. Families
216
+ // that reason but were NOT observed here — anthropic/*, qwen/*, moonshotai/*,
217
+ // minimax/* — are deliberately left alone until probed live: Anthropic in particular
218
+ // takes a thinking BUDGET through OpenRouter and constrains temperature alongside it,
219
+ // which is a second parameter we would be guessing at.
220
+ const PROXIED_THINKING_MODEL: RegExp[] = [
221
+ /^google\/gemini-/i, // gemini-2.5+ all take an effort; gemma-* deliberately excluded
222
+ /^z-ai\/glm-/i, // the whole GLM line reasons, vision variants included
223
+ /^deepseek\/deepseek-(r1|v3\.[12]|v4)/i, // NOT deepseek-chat*, which is the non-thinking split
224
+ /^x-ai\/grok-4/i, // grok-4.x; grok-build-0.1 is unprobed
225
+ /^openai\/gpt-(5|6|oss)/i, // gpt-4*/gpt-3.5* do not reason
226
+ ];
227
+
228
+ // Non-thinking variants inside an otherwise-thinking family. `-chat` is OpenAI's
229
+ // non-reasoning split (gpt-5.2-chat); `instruct` is the same idea everywhere.
230
+ const NON_THINKING_VARIANT = /(?:^|[-/])(?:chat|instruct)(?:[-.]|$)/i;
231
+
232
+ const TEE_MODEL = /^(tinfoil|phala|near)\//;
233
+
153
234
  // Two deliberate omissions inside the TEE set: `*-instruct` ids are the
154
235
  // non-thinking variants, and tinfoil/kimi-k2-6 reasons but ignored BOTH levers when
155
236
  // probed, so annotating it would hand the user a dial connected to nothing.
156
- const TEE_MODEL = /^(tinfoil|phala|near)\//;
237
+ function teeThinkingProfile(id: string): ThinkingProfile | null {
238
+ if (/instruct/i.test(id)) return null;
239
+ if (/gpt-oss/i.test(id)) return REASONING_EFFORT_THINKING;
240
+ if (/glm|qwen/i.test(id)) return CHAT_TEMPLATE_THINKING;
241
+ return null;
242
+ }
243
+
244
+ function proxiedThinkingProfile(id: string): ThinkingProfile | null {
245
+ if (NON_THINKING_VARIANT.test(id)) return null;
246
+ return PROXIED_THINKING_MODEL.some((re) => re.test(id)) ? OPENROUTER_THINKING : null;
247
+ }
157
248
 
158
249
  export function thinkingProfile(id: string): ThinkingProfile | null {
159
- if (!TEE_MODEL.test(id)) return null;
160
- if (/instruct/i.test(id)) return null;
161
- const profile = /gpt-oss/i.test(id)
162
- ? REASONING_EFFORT_THINKING
163
- : /glm|qwen/i.test(id)
164
- ? CHAT_TEMPLATE_THINKING
165
- : null;
250
+ const profile = TEE_MODEL.test(id) ? teeThinkingProfile(id) : proxiedThinkingProfile(id);
166
251
  // Hand out a COPY. These entries end up on hundreds of registered models, and a
167
252
  // shared nested object is one careless mutation away from retuning the whole catalog.
168
253
  return profile && { ...profile, thinkingLevelMap: { ...profile.thinkingLevelMap }, ...(profile.compat ? { compat: { ...profile.compat } } : {}) };
@@ -196,16 +281,69 @@ function catalogCachePath(): string {
196
281
 
197
282
  // Best effort in both directions: this cache is an optimization, and a launch must never
198
283
  // fail because it couldn't be read or written.
199
- function saveCachedCatalogIds(ids: string[]): void {
284
+ //
285
+ // `windows` joined `ids` here because the synchronous launch seed needs it. Registration
286
+ // builds every model entry before the live fetch can resolve, so a context window known
287
+ // only to the live catalog would always arrive one launch too late — the entries pi
288
+ // actually binds would already carry the fallback. A window is a capability, not a
289
+ // privacy claim, so unlike the tier it is safe to read back from disk: the worst a stale
290
+ // one does is size a budget against last week's number, and the live fetch corrects it
291
+ // moments later. (The v1 `ids` key is still written, so an older build reading this file
292
+ // sees exactly what it expects.)
293
+ function saveCachedCatalog(infos: AccountModelInfo[]): void {
200
294
  try {
201
295
  mkdirSync(globalDir(), { recursive: true });
202
- const payload = { v: 1, fetchedAt: new Date().toISOString(), ids: ids.slice(0, CATALOG_CACHE_MAX) };
296
+ const kept = infos.slice(0, CATALOG_CACHE_MAX);
297
+ const windows: Record<string, number> = {};
298
+ for (const info of kept) {
299
+ if (info.contextWindow !== undefined) windows[info.id] = info.contextWindow;
300
+ }
301
+ const payload = {
302
+ v: 2,
303
+ fetchedAt: new Date().toISOString(),
304
+ ids: kept.map((info) => info.id),
305
+ windows,
306
+ };
203
307
  writeFileSync(catalogCachePath(), JSON.stringify(payload) + "\n", "utf8");
308
+ forgetCachedContextWindows();
204
309
  } catch {
205
310
  /* unwritable home — we just seed from DEFAULT_MODELS next launch */
206
311
  }
207
312
  }
208
313
 
314
+ // Read the cache once per process. seedModel calls accountContextWindow for every id in
315
+ // the catalog on every registration, and the catalog re-registers whenever the shim
316
+ // state changes — re-reading and re-parsing a 284-entry file each time would turn a
317
+ // cheap lookup into hundreds of syscalls on the launch path.
318
+ let cachedWindows: Map<string, number> | null = null;
319
+
320
+ function cachedContextWindows(): Map<string, number> {
321
+ if (cachedWindows) return cachedWindows;
322
+ cachedWindows = new Map();
323
+ try {
324
+ const path = catalogCachePath();
325
+ if (existsSync(path)) {
326
+ const parsed = JSON.parse(readFileSync(path, "utf8")) as { windows?: unknown };
327
+ // A v1 file has no `windows` at all; every id then falls back, exactly as before.
328
+ if (parsed.windows && typeof parsed.windows === "object") {
329
+ for (const [id, value] of Object.entries(parsed.windows as Record<string, unknown>)) {
330
+ const window = validContextWindow(value);
331
+ if (window !== undefined && cachedWindows.size < CATALOG_CACHE_MAX) cachedWindows.set(id, window);
332
+ }
333
+ }
334
+ }
335
+ } catch {
336
+ /* absent, unreadable, or garbage — every model just uses the fallback window */
337
+ }
338
+ return cachedWindows;
339
+ }
340
+
341
+ // Drop the memo so the next read sees what was just written (and so a test can rewrite
342
+ // the cache between assertions).
343
+ function forgetCachedContextWindows(): void {
344
+ cachedWindows = null;
345
+ }
346
+
209
347
  export function loadCachedCatalogIds(): string[] {
210
348
  try {
211
349
  const path = catalogCachePath();
@@ -265,6 +403,10 @@ export function seedCatalogIds(): string[] {
265
403
  export interface AccountModelInfo {
266
404
  id: string;
267
405
  tier: PrivacyTier;
406
+ // The server's published context window, when it published one and it is credible.
407
+ // Absent means "unknown" — seedModel falls back to DEFAULT_CONTEXT_WINDOW rather
408
+ // than to a guess, so an older server behaves exactly as before.
409
+ contextWindow?: number;
268
410
  }
269
411
 
270
412
  // The set of tier strings pi-privacy defines (posture/tiers.ts). We only trust a
@@ -298,6 +440,19 @@ function normalizeTier(tier: string | undefined, modelId: string): PrivacyTier {
298
440
  // A live NEAR attestation (accountPosture) can still upgrade a row to tee-verified.
299
441
  const accountTierMap = new Map<string, PrivacyTier>();
300
442
 
443
+ // Published context windows, keyed by modelId. Unlike the tier map this one IS
444
+ // persisted (see the cache below): a window is a capability, not a privacy claim, and
445
+ // it is needed at LAUNCH — registration happens synchronously, long before the live
446
+ // fetch resolves, so a window learned only at fetch time would arrive after the model
447
+ // entries pi actually uses were built.
448
+ const accountWindowMap = new Map<string, number>();
449
+
450
+ // The server's published context window for a model, or undefined if we have not been
451
+ // told one. Live catalog first, then the on-disk cache from the last launch.
452
+ export function accountContextWindow(id: string): number | undefined {
453
+ return accountWindowMap.get(id) ?? cachedContextWindows().get(id);
454
+ }
455
+
301
456
  // The server-asserted baseline tier for an account model, or undefined if we haven't
302
457
  // seen it in a catalog fetch. Used by the /models picker (privateer-models.ts).
303
458
  export function accountBaselineTier(modelId: string): PrivacyTier | undefined {
@@ -327,23 +482,38 @@ export async function fetchAccountCatalog(): Promise<AccountModelInfo[]> {
327
482
  infos = fallback();
328
483
  } else {
329
484
  const data = (await res.json()) as {
330
- models?: { modelId?: string; privacy?: { tier?: string } }[];
485
+ models?: { modelId?: string; privacy?: { tier?: string }; contextLength?: unknown }[];
331
486
  };
332
487
  const parsed = (data.models ?? [])
333
- .map((m) => (m.modelId ? { id: m.modelId, tier: normalizeTier(m.privacy?.tier, m.modelId) } : null))
488
+ .map((m): AccountModelInfo | null =>
489
+ m.modelId
490
+ ? {
491
+ id: m.modelId,
492
+ tier: normalizeTier(m.privacy?.tier, m.modelId),
493
+ // `contextLength` is null for a model upstream has no window for, and
494
+ // absent entirely on a server older than the field — both mean
495
+ // "unknown", and validContextWindow collapses them to undefined.
496
+ contextWindow: validContextWindow(m.contextLength),
497
+ }
498
+ : null,
499
+ )
334
500
  .filter((x): x is AccountModelInfo => !!x);
335
501
  // Cache only a real LIVE listing — never the fallback, which would freeze the six
336
502
  // seed ids on disk and read back as though it were the catalog. Both the cache and
337
503
  // the returned list are the server's UNFILTERED offer; servability is decided at
338
504
  // registration (accountProviderConfig), which re-evaluates it every time.
339
- if (parsed.length) saveCachedCatalogIds(parsed.map((p) => p.id));
505
+ if (parsed.length) saveCachedCatalog(parsed);
340
506
  infos = parsed.length ? parsed : fallback();
341
507
  }
342
508
  } catch {
343
509
  infos = fallback();
344
510
  }
345
511
  accountTierMap.clear();
346
- for (const info of infos) accountTierMap.set(info.id, info.tier);
512
+ accountWindowMap.clear();
513
+ for (const info of infos) {
514
+ accountTierMap.set(info.id, info.tier);
515
+ if (info.contextWindow !== undefined) accountWindowMap.set(info.id, info.contextWindow);
516
+ }
347
517
  return infos;
348
518
  }
349
519
 
@@ -650,7 +820,7 @@ export function registerAccountModels(pi: {
650
820
  export function makeAccountProvider() {
651
821
  return (pi: {
652
822
  registerProvider?: (name: string, config: unknown) => void;
653
- on?: (event: string, handler: (e: unknown, ctx: unknown) => void) => void;
823
+ on?: (event: string, handler: (e: unknown, ctx: unknown) => unknown) => void;
654
824
  }): void => {
655
825
  if (typeof pi.registerProvider !== "function") return;
656
826
  // Compress this channel's inference bodies before anything can send one. The edge
@@ -710,14 +880,22 @@ export function makeAccountProvider() {
710
880
  }
711
881
  });
712
882
 
713
- // message_end: an assistant turn that ended in an auth error on the account channel
714
- // means our session token is dead server-side. Pi has no reactive-401 refresh, so
715
- // replace the session now instead of failing every prompt until `expires`.
716
- // See recoverAccountSession.
883
+ // Replace confirmed balance failures before Pi renders/persists the message.
884
+ // showError alone misses first-attempt failures (the assistant component renders
885
+ // those directly). This also covers print mode and the remote event adapter.
717
886
  pi.on?.("message_end", (e, ctx) => {
718
- const msg = (e as { message?: { role?: string; stopReason?: string; errorMessage?: string } })?.message;
887
+ const msg = (e as { message?: AssistantMessage })?.message;
719
888
  if (msg?.role !== "assistant" || msg.stopReason !== "error" || !msg.errorMessage) return;
720
- if ((ctx as SeedContext)?.model?.provider !== "privateer") return;
889
+ if (msg.provider !== "privateer") return;
890
+ const balance = hasCredentials() ? describeAccountBalanceError(msg.errorMessage) : null;
891
+ if (balance) {
892
+ // Normalize the known billing failure to Payment Required, just as
893
+ // authedFetch does for cap responses. Keeping a hard status first stops
894
+ // Pi retrying/compacting into the same exhausted account.
895
+ return { message: { ...msg, errorMessage: `402 ${balance.message}\n${balance.hint}` } };
896
+ }
897
+ // An auth failure, unlike a balance failure, needs a fresh child session.
898
+ // Pi has no reactive-401 refresh; see recoverAccountSession.
721
899
  void recoverAccountSession(ctx, msg.errorMessage);
722
900
  });
723
901
  };
@@ -10,7 +10,7 @@
10
10
  // never nominated a model. resolveDefaultModel() makes the account channel the default
11
11
  // the moment credentials exist, and keeps the legacy BYO behaviour otherwise.
12
12
 
13
- import { existsSync, readFileSync, writeFileSync } from "node:fs";
13
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
14
14
  import { join } from "node:path";
15
15
  import { hasCredentials } from "../auth/privateer.ts";
16
16
  import { agentDir } from "../config/paths.ts";
@@ -282,8 +282,10 @@ export function writePiDefaultModel(spec: string): string | null {
282
282
  function writeSettingsDefaultModel(spec: string, onlyIfUnset: boolean): string | null {
283
283
  const parts = splitSpec(spec);
284
284
  if (!parts) return null;
285
- const settingsPath = join(agentDir(), "settings.json");
285
+ const dir = agentDir();
286
+ const settingsPath = join(dir, "settings.json");
286
287
  try {
288
+ mkdirSync(dir, { recursive: true });
287
289
  let settings: Record<string, unknown> = {};
288
290
  if (existsSync(settingsPath)) {
289
291
  const raw = readFileSync(settingsPath, "utf8").trim();
@@ -353,6 +353,8 @@ function projectEvent(ev: EngineEvent): Record<string, unknown> {
353
353
  switch (ev.type) {
354
354
  case "tool-call":
355
355
  return { type: "tool-call", id: ev.id, name: ev.name, input: safe(asText(ev.input), 2000) };
356
+ case "tool-progress":
357
+ return { type: "tool-progress", id: ev.id, name: ev.name, output: redactSecrets(ev.output).slice(-4000) };
356
358
  case "tool-result":
357
359
  return { type: "tool-result", id: ev.id, name: ev.name, output: safe(asText(ev.output), 4000) };
358
360
  case "tool-error":