privateer-agent 0.12.45 → 0.12.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,163 @@
1
+ // SPEND PRE-APPROVAL FOR ONE HEADLESS RUN, typed on the command line.
2
+ //
3
+ // privateer -p --allow-spend generate_video --max-calls 1 --max-spend 1.00 "…"
4
+ //
5
+ // THE PROBLEM. A `-p` run has no screen, so the gate's local asker has nobody to ask
6
+ // and every billing tool — each one `alwaysAsk` — is denied. An agent driving Privateer
7
+ // from a script could plan a whole film and only discover at the last step that the
8
+ // one call that mattered could never be approved. The routine grant (childSpend.ts)
9
+ // solves this for scheduled runs; this is the same idea for a run a person starts.
10
+ //
11
+ // WHY A FLAG, NOT AN ENV VAR. The grant is typed per invocation, capped, and gone when
12
+ // the process exits. The launcher carries it to the gate in PRIVATEER_CLI_SPEND — the
13
+ // only channel into Pi's process — but deletes any inherited value before parsing argv
14
+ // (bin/privateer-launch.mjs), so an export lingering in someone's shell never counts.
15
+ // And it is honoured only where the flag can apply: a TOP-LEVEL headless session. Not
16
+ // the TUI (a person approves each call there), and not a subagent child (a child's
17
+ // grant comes only from childSpend.ts, whose caps this ledger does not share).
18
+ //
19
+ // WHAT IT LIFTS. Exactly what the routine grant lifts, through the same
20
+ // ModeGate.isSpendPreauthorized hook and under the same guards: `alwaysAsk` must be the
21
+ // only reason to ask, and a call that leaves the working directory or touches a
22
+ // protected file is never covered. On top of that, two caps:
23
+ //
24
+ // • --max-calls counts calls this ledger ALLOWED, not calls that succeeded — a call
25
+ // that then fails server-side still used its slot. Over-counting is the safe error.
26
+ // • --max-spend is checked BEFORE each call against the server's own reservation
27
+ // figure for that exact call (tools/media.ts quoteMediaCallUsd), which is
28
+ // worst-case by design. A call that can't be priced is REFUSED under a dollar cap
29
+ // rather than waved through — a cap that skips what it can't measure isn't one.
30
+
31
+ import type { PermissionRequest } from "./gate.ts";
32
+ import { BILLED_MEDIA_TOOLS } from "./classify.ts";
33
+
34
+ /** The env var the launcher writes from `--allow-spend` (mirrors bin/headless-flags.mjs). */
35
+ export const CLI_SPEND_ENV = "PRIVATEER_CLI_SPEND";
36
+
37
+ export interface CliSpendGrant {
38
+ tools: string[];
39
+ maxCalls?: number;
40
+ maxSpendUsd?: number;
41
+ }
42
+
43
+ /**
44
+ * The grant this run was launched with, or null. Defensive: anything malformed, any
45
+ * tool that isn't a billing tool, or a grant with no cap at all reads as NO grant —
46
+ * the launcher never writes one of those, so seeing one means it didn't come from it.
47
+ */
48
+ export function readCliSpendGrant(env: NodeJS.ProcessEnv = process.env): CliSpendGrant | null {
49
+ const raw = env[CLI_SPEND_ENV];
50
+ if (!raw) return null;
51
+ let parsed: unknown;
52
+ try {
53
+ parsed = JSON.parse(raw);
54
+ } catch {
55
+ return null;
56
+ }
57
+ const g = parsed as Partial<CliSpendGrant>;
58
+ if (!g || !Array.isArray(g.tools)) return null;
59
+ const tools = g.tools.filter((t): t is string => typeof t === "string" && BILLED_MEDIA_TOOLS.has(t));
60
+ if (tools.length === 0) return null;
61
+ const maxCalls = Number.isInteger(g.maxCalls) && (g.maxCalls as number) > 0 ? (g.maxCalls as number) : undefined;
62
+ const maxSpendUsd =
63
+ typeof g.maxSpendUsd === "number" && Number.isFinite(g.maxSpendUsd) && g.maxSpendUsd > 0 ? g.maxSpendUsd : undefined;
64
+ if (maxCalls === undefined && maxSpendUsd === undefined) return null;
65
+ return { tools, ...(maxCalls !== undefined ? { maxCalls } : {}), ...(maxSpendUsd !== undefined ? { maxSpendUsd } : {}) };
66
+ }
67
+
68
+ /** Price one call of `tool` with these arguments, in USD; null when it can't be priced. */
69
+ export type SpendQuote = (tool: string, input: unknown, signal?: AbortSignal) => Promise<number | null>;
70
+
71
+ export type SpendDecision = { ok: true; usd: number | null } | { ok: false; reason: string };
72
+
73
+ const money = (n: number): string => `$${n.toFixed(2)}`;
74
+
75
+ /**
76
+ * The running tally for one grant. One per process: the caps are for the whole run.
77
+ */
78
+ export class CliSpendLedger {
79
+ private calls = 0;
80
+ private spentUsd = 0;
81
+
82
+ constructor(
83
+ readonly grant: CliSpendGrant,
84
+ private readonly quote: SpendQuote,
85
+ ) {}
86
+
87
+ /**
88
+ * May this call spend? Records it when it may. The check and the record happen in
89
+ * one synchronous step after the (async) quote, so two calls racing in parallel can't
90
+ * both squeeze under the same remaining budget.
91
+ */
92
+ async authorize(tool: string, input: unknown, signal?: AbortSignal): Promise<SpendDecision> {
93
+ const { tools, maxCalls, maxSpendUsd } = this.grant;
94
+ if (!tools.includes(tool)) {
95
+ return { ok: false, reason: `this run's --allow-spend covers ${tools.join(", ")}, not ${tool}` };
96
+ }
97
+ if (maxCalls !== undefined && this.calls >= maxCalls) {
98
+ return { ok: false, reason: `this run's --max-calls ${maxCalls} is used up` };
99
+ }
100
+ let usd: number | null = null;
101
+ if (maxSpendUsd !== undefined) {
102
+ try {
103
+ usd = await this.quote(tool, input, signal);
104
+ } catch {
105
+ usd = null;
106
+ }
107
+ if (usd === null) {
108
+ return {
109
+ ok: false,
110
+ reason:
111
+ `${tool} can't be priced before it runs, so it can't be checked against --max-spend ${money(maxSpendUsd)}` +
112
+ (tool === "generate_image" || tool === "generate_sprite"
113
+ ? " (a call that overrides the image model is one of those — leave `model`/`image_model` unset)"
114
+ : "") +
115
+ ". Cap this run with --max-calls instead",
116
+ };
117
+ }
118
+ // Recheck after the await: a parallel call may have spent while we were quoting.
119
+ if (maxCalls !== undefined && this.calls >= maxCalls) {
120
+ return { ok: false, reason: `this run's --max-calls ${maxCalls} is used up` };
121
+ }
122
+ if (this.spentUsd + usd > maxSpendUsd + 1e-9) {
123
+ return {
124
+ ok: false,
125
+ reason:
126
+ `this call is estimated at ${money(usd)} and only ${money(Math.max(0, maxSpendUsd - this.spentUsd))} ` +
127
+ `of this run's --max-spend ${money(maxSpendUsd)} is left`,
128
+ };
129
+ }
130
+ this.spentUsd += usd;
131
+ }
132
+ this.calls++;
133
+ return { ok: true, usd };
134
+ }
135
+
136
+ /** One line for the end of the run / a denial: what the grant has used so far. */
137
+ summary(): string {
138
+ const { maxCalls, maxSpendUsd } = this.grant;
139
+ const parts = [`${this.calls}${maxCalls !== undefined ? `/${maxCalls}` : ""} billed call(s)`];
140
+ if (maxSpendUsd !== undefined) parts.push(`~${money(this.spentUsd)} of ${money(maxSpendUsd)} estimated`);
141
+ return parts.join(", ");
142
+ }
143
+ }
144
+
145
+ /**
146
+ * The guidance a headless run gives when a billed call hits the gate with nothing to
147
+ * approve it — written for the MODEL as much as the person, since the model is the one
148
+ * that reads a tool denial and decides what to do next. It names every way out.
149
+ */
150
+ export function headlessSpendGuidance(tool: string, cmd = process.env.PRIVATEER_CMD || "privateer"): string {
151
+ return (
152
+ `This is a headless run (-p) with no one at a screen to approve ${tool}, which bills the account. ` +
153
+ `It can't be approved from inside this run. To allow it, the person running Privateer can re-run with ` +
154
+ `\`${cmd} -p --allow-spend ${tool} --max-calls 1\` (optionally --max-spend <usd>), add --approve-in-app to ` +
155
+ `approve it from the Privateer app, or drive Privateer over ACP (\`${cmd} acp\`), where the controlling ` +
156
+ `program is asked. Stop and report this rather than retrying.`
157
+ );
158
+ }
159
+
160
+ /** Is this the request a CLI grant could ever cover? (Billing tool, and billing is the only ask.) */
161
+ export function isSpendRequest(req: PermissionRequest): boolean {
162
+ return req.alwaysAsk === true && BILLED_MEDIA_TOOLS.has(req.tool);
163
+ }
@@ -63,7 +63,12 @@ export interface ModeGateDeps {
63
63
  // Consulted ONLY to lift `alwaysAsk`, and only under the guards in ModeGate.request.
64
64
  // Absent ⇒ nothing is pre-authorized, which is the posture every interactive session
65
65
  // keeps: a terminal always asks its human, however cheap the call.
66
- isSpendPreauthorized?: (req: PermissionRequest) => boolean;
66
+ //
67
+ // May be async: a `--max-spend` grant prices the call before answering (cliSpend.ts).
68
+ // It is consulted LAST, after every other guard has passed, because a grant with a
69
+ // budget RECORDS what it allows — asking it about a call the mode would refuse anyway
70
+ // would spend budget on nothing.
71
+ isSpendPreauthorized?: (req: PermissionRequest) => boolean | Promise<boolean>;
67
72
  }
68
73
 
69
74
  // The permission gate used by the live TUI. It first applies the mode/allowlist
@@ -129,8 +134,9 @@ export class ModeGate implements PermissionGate {
129
134
  req.alwaysAsk &&
130
135
  !req.outside &&
131
136
  !req.protected &&
132
- this.deps.isSpendPreauthorized?.(req) === true &&
133
- decideAuto({ ...req, alwaysAsk: false }, this.deps.getMode(), this.deps.allowlist, denylist) === "allow"
137
+ this.deps.isSpendPreauthorized &&
138
+ decideAuto({ ...req, alwaysAsk: false }, this.deps.getMode(), this.deps.allowlist, denylist) === "allow" &&
139
+ (await this.deps.isSpendPreauthorized(req)) === true
134
140
  ) {
135
141
  return "allow";
136
142
  }
@@ -27,9 +27,20 @@
27
27
  // process.env is the one thing all those copies share, so the state lives there and
28
28
  // nowhere else: every reader, in every extension, sees every toggle.
29
29
  //
30
+ // NO QUARTER ALSO TAKES THE PRIVACY FILTER DOWN. Lowering the moat is the "step away"
31
+ // switch, and a turn that runs to completion unattended must not stall on pi-privacy's
32
+ // PII prompt either — so going to no quarter is also `/privacy off`. Raising the moat
33
+ // puts the filter back, but ONLY if no quarter is what took it down: a session already
34
+ // running with `/privacy off` (or `--no-privacy`) stays off. The marker that records
35
+ // "no quarter did this" lives in the env for the same cross-copy reasons as the flag,
36
+ // and setPrivacyDisabled clears it — so an explicit `/privacy on|off` mid-no-quarter
37
+ // is the operator's word and a later shift+tab doesn't overrule it.
38
+ //
30
39
  // IMPORT-SAFETY: no Pi imports, no node builtins — safe to load from anywhere,
31
40
  // including boot-ordered entrypoints (see boot.ts's ORDERING CONTRACT).
32
41
 
42
+ import { NO_QUARTER_PRIVACY_MARK, privacyDisabled, setPrivacyDisabled } from "../config/privacyDisabled.ts";
43
+
33
44
  const ENV = "PRIVATEER_NO_QUARTER";
34
45
 
35
46
  /** True while the gate is fully lowered for this session. Read live, never cached. */
@@ -39,8 +50,16 @@ export function noQuarterActive(): boolean {
39
50
 
40
51
  /** Set the state — in the env, so every copy of this module and every child agrees. Returns the new state. */
41
52
  export function setNoQuarter(on: boolean): boolean {
42
- if (on) process.env[ENV] = "1";
43
- else delete process.env[ENV];
53
+ if (on) {
54
+ process.env[ENV] = "1";
55
+ if (!privacyDisabled()) {
56
+ setPrivacyDisabled(true);
57
+ process.env[NO_QUARTER_PRIVACY_MARK] = "1";
58
+ }
59
+ } else {
60
+ delete process.env[ENV];
61
+ if (process.env[NO_QUARTER_PRIVACY_MARK] === "1") setPrivacyDisabled(false); // clears the mark
62
+ }
44
63
  return on;
45
64
  }
46
65
 
@@ -27,7 +27,7 @@ import { join } from "node:path";
27
27
  import { globalDir } from "../config/paths.ts";
28
28
  import { canOpenBrowser, openInBrowser } from "../util/openBrowser.ts";
29
29
  import { installGzipRequestBodies } from "../util/gzipRequestBody.ts";
30
- import { describeAccountBalanceError } from "../engine/errors.ts";
30
+ import { describeAccountBalanceError, isConnectionFailureText } from "../engine/errors.ts";
31
31
  import type { AssistantMessage } from "@earendil-works/pi-ai";
32
32
  import { interpretReport, teePosture, tierFromTeePosture, type PrivacyTier } from "pi-privacy";
33
33
  import { ACCOUNT_DEFAULT_MODEL_ID, ACCOUNT_NEAR_MODEL_ID, ensurePiDefaultModel } from "./defaultModel.ts";
@@ -38,6 +38,7 @@ import {
38
38
  sealedProviderFor,
39
39
  sealedShimBase,
40
40
  ensureSealedShim,
41
+ stopSealedShim,
41
42
  attestSealed,
42
43
  } from "./sealedShim.ts";
43
44
  import type { PhalaEnclaveIdentity } from "./phalaSeal.ts";
@@ -897,6 +898,10 @@ export function makeAccountProvider() {
897
898
  // An auth failure, unlike a balance failure, needs a fresh child session.
898
899
  // Pi has no reactive-401 refresh; see recoverAccountSession.
899
900
  void recoverAccountSession(ctx, msg.errorMessage);
901
+ // A request that got no response at all. See recoverAccountConnection.
902
+ if (isConnectionFailureText(msg.errorMessage)) {
903
+ void recoverAccountConnection(ctx, (msg as { model?: string }).model ?? "", () => register(lastIds));
904
+ }
900
905
  });
901
906
  };
902
907
  }
@@ -1125,6 +1130,89 @@ const ACCOUNT_AUTH_FAILURE =
1125
1130
  // plenty, and the user gets a clear message instead of a retry loop.
1126
1131
  const RECOVERY_COOLDOWN_MS = 30_000;
1127
1132
 
1133
+ // Ask the server whether an account access token is still accepted. The status, or
1134
+ // null when there was no answer at all (which says nothing about the token).
1135
+ async function probeAccountToken(access: string): Promise<number | null> {
1136
+ try {
1137
+ const res = await fetch(`${serverBaseUrl()}/auth/me`, {
1138
+ headers: { Authorization: `Bearer ${access}` },
1139
+ signal: AbortSignal.timeout(8_000),
1140
+ });
1141
+ void res.body?.cancel().catch(() => {});
1142
+ return res.status;
1143
+ } catch {
1144
+ return null;
1145
+ }
1146
+ }
1147
+
1148
+ /**
1149
+ * Get a session that answers "Connection error." on every message working again.
1150
+ *
1151
+ * The report this exists for: a session started while a sign-in was failing kept
1152
+ * failing with the SDK's bare "Connection error." on every retry, while a fresh
1153
+ * `privateer` on the same machine worked. Nothing noticed, because the only recovery
1154
+ * the account channel had keyed on a 401 (recoverAccountSession), and a request that
1155
+ * never gets an HTTP response carries no status at all. Two pieces of per-SESSION state
1156
+ * can strand requests that way while the machine's login is fine, and both are cheap
1157
+ * to renew:
1158
+ *
1159
+ * • the sealed shim (tinfoil/phala): the model's baseUrl points at a loopback port
1160
+ * for the life of the process. Restart the shim and re-register — Pi re-reads the
1161
+ * live model's baseUrl from the registry on registration — so the next request
1162
+ * goes to a listener that exists;
1163
+ * • the credential: probe the token this process holds; if the server now refuses it,
1164
+ * mint a fresh session exactly as a 401 would. A probe that gets NO answer is a real
1165
+ * outage, and replacing a good session over it would only leak a Linked Devices row.
1166
+ *
1167
+ * Bounded by the same cooldown as recoverAccountSession, so a dead network costs one
1168
+ * probe per window, not one per retry. Returns what it renewed.
1169
+ */
1170
+ export async function recoverAccountConnection(
1171
+ ctx: unknown,
1172
+ modelId: string,
1173
+ reRegister: () => void,
1174
+ probe: (access: string) => Promise<number | null> = probeAccountToken,
1175
+ ): Promise<{ shim: boolean; session: boolean }> {
1176
+ const renewed = { shim: false, session: false };
1177
+ if (!hasCredentials()) return renewed;
1178
+ const slot = armSlot() as ReturnType<typeof armSlot> & { connectionRecoveredAt?: number };
1179
+ const now = Date.now();
1180
+ if (slot.connectionRecoveredAt !== undefined && now - slot.connectionRecoveredAt < RECOVERY_COOLDOWN_MS) return renewed;
1181
+ slot.connectionRecoveredAt = now;
1182
+
1183
+ if (sealedEnabled() && sealedProviderFor(modelId)) {
1184
+ try {
1185
+ await stopSealedShim();
1186
+ await ensureSealedShim();
1187
+ reRegister();
1188
+ renewed.shim = true;
1189
+ } catch {
1190
+ /* the shim won't start → sealed models fall back to the cleartext path on re-register */
1191
+ try {
1192
+ reRegister();
1193
+ } catch {
1194
+ /* nothing more to do from here */
1195
+ }
1196
+ }
1197
+ }
1198
+
1199
+ const held = slot.cred;
1200
+ const status = held ? await probe(held.access) : null;
1201
+ if (status === 401 || status === 403) {
1202
+ slot.cred = undefined;
1203
+ renewed.session = await armAccountCredential(ctx, { notify: false });
1204
+ }
1205
+
1206
+ const c = ctx as SeedContext;
1207
+ if (c?.hasUI && (renewed.shim || renewed.session)) {
1208
+ c.ui?.notify?.(
1209
+ `Couldn't reach Privateer — renewed this session's ${[renewed.shim ? "sealed connection" : "", renewed.session ? "credential" : ""].filter(Boolean).join(" and ")}. Send that message again.`,
1210
+ "warning",
1211
+ );
1212
+ }
1213
+ return renewed;
1214
+ }
1215
+
1128
1216
  // Replace a dead account session after an auth failure, so the NEXT prompt works.
1129
1217
  //
1130
1218
  // This is the client-side answer to a gap in Pi's OAuth contract: inference goes out
@@ -0,0 +1,199 @@
1
+ // Seeing on behalf of a model that can't.
2
+ //
3
+ // providers/vision.ts makes a text-only model HONEST about images — it declares
4
+ // `input: ["text"]`, and Pi then drops every image block on the way to the provider.
5
+ // Honest, but blind: a user on glm-5-2 who points the agent at a screenshot still
6
+ // gets an answer about a picture the model never saw, just with a note saying so.
7
+ //
8
+ // This module closes that gap. Before each LLM call (Pi's `context` event) every
9
+ // image in the outgoing messages is handed to a vision-capable model and replaced
10
+ // with that model's description — so the text-only model gets words where it would
11
+ // have got nothing. The session itself keeps the original images: only the
12
+ // per-request copy is rewritten, so switching to a vision model later still sends
13
+ // the real pixels.
14
+ //
15
+ // WHICH delegate — "the key type provided". The delegate must be reachable with the
16
+ // credential the user is ALREADY using, i.e. the same provider as the current model:
17
+ //
18
+ // privateer/* (account) → privateer/tinfoil/gemma4-31b, the confidential default
19
+ // tinfoil/* (TEE key) → tinfoil/gemma4-31b
20
+ // openrouter/* … → the best vision model that provider serves
21
+ //
22
+ // Never a different provider. Crossing providers would ship the user's screenshot
23
+ // to a company they did not pick for this session — on a privacy-first agent that
24
+ // is not a fallback, it's a leak. If the current provider has no vision model with
25
+ // working auth, nothing is rewritten and Pi's existing "image omitted" behaviour
26
+ // stands. PRIVATEER_VISION_MODEL=<provider/id> overrides the choice (and is the one
27
+ // deliberate way to cross providers).
28
+
29
+ import { createHash } from "node:crypto";
30
+ import { ACCOUNT_DEFAULT_MODEL_ID, TINFOIL_MODEL_ID } from "./defaultModel.ts";
31
+
32
+ export interface VisionModel {
33
+ provider: string;
34
+ id: string;
35
+ input: string[];
36
+ }
37
+
38
+ /** The slice of Pi's ModelRegistry this module needs (kept structural). */
39
+ export interface VisionRegistry {
40
+ getAvailable(): VisionModel[];
41
+ find(provider: string, id: string): VisionModel | undefined;
42
+ hasConfiguredAuth(model: VisionModel): boolean;
43
+ complete(model: any, context: any, options?: any): Promise<{ content: any[]; stopReason?: string; errorMessage?: string }>;
44
+ }
45
+
46
+ const specOf = (m: { provider: string; id: string }) => `${m.provider}/${m.id}`;
47
+ const seesImages = (m: { input?: string[] }) => Array.isArray(m.input) && m.input.includes("image");
48
+
49
+ // Per-provider first choices, tried before "any vision model on this provider". Only
50
+ // providers where the pick matters are listed: on the account channel it keeps the
51
+ // image inside the attested Tinfoil enclave rather than whichever vision model sorts
52
+ // first; elsewhere the registry's own list is good enough.
53
+ const PREFERRED: Record<string, string[]> = {
54
+ privateer: [ACCOUNT_DEFAULT_MODEL_ID],
55
+ tinfoil: [TINFOIL_MODEL_ID.replace(/^tinfoil\//, "")],
56
+ openrouter: ["google/gemini-2.5-flash", "openai/gpt-4o-mini", "anthropic/claude-haiku-4.5"],
57
+ };
58
+
59
+ /**
60
+ * The model that should look at images for `current`, or undefined when there is no
61
+ * reachable vision model on the same key. Returns undefined for a model that can
62
+ * already see — there is nothing to delegate.
63
+ */
64
+ export function pickVisionDelegate(
65
+ current: VisionModel | undefined,
66
+ registry: VisionRegistry,
67
+ env: NodeJS.ProcessEnv = process.env,
68
+ ): VisionModel | undefined {
69
+ if (!current || seesImages(current)) return undefined;
70
+ const usable = (m: VisionModel | undefined): m is VisionModel =>
71
+ !!m && seesImages(m) && registry.hasConfiguredAuth(m);
72
+
73
+ const override = env.PRIVATEER_VISION_MODEL?.trim();
74
+ if (override) {
75
+ const slash = override.indexOf("/");
76
+ const m = slash > 0 ? registry.find(override.slice(0, slash), override.slice(slash + 1)) : undefined;
77
+ if (usable(m)) return m;
78
+ }
79
+
80
+ for (const id of PREFERRED[current.provider] ?? []) {
81
+ const m = registry.find(current.provider, id);
82
+ if (usable(m)) return m;
83
+ }
84
+ return registry.getAvailable().find((m) => m.provider === current.provider && usable(m));
85
+ }
86
+
87
+ // Pi's read tool appends this to an image it knows the model can't see. Once we've
88
+ // described the image the note is false, and leaving it in invites the model to tell
89
+ // the user it couldn't look.
90
+ const OMITTED_NOTE = /\n?\[Current model does not support images\. The image will be omitted from this request\.\]/g;
91
+
92
+ const SYSTEM_PROMPT = [
93
+ "You are the eyes for another AI model that cannot see images.",
94
+ "Describe the image so that model can act on it without seeing it.",
95
+ "Transcribe ALL legible text verbatim (code, error messages, labels, numbers), preserving line breaks.",
96
+ "Then describe layout, UI elements and their state, colors where meaningful, charts/diagrams and what they show, and anything unusual.",
97
+ "Be precise and factual. Do not speculate beyond what is visible. No preamble.",
98
+ ].join(" ");
99
+
100
+ type Block = { type: string; text?: string; data?: string; mimeType?: string };
101
+
102
+ const cache = new Map<string, Promise<string>>();
103
+
104
+ /** Test hook: forget every cached description. */
105
+ export function clearVisionCache(): void {
106
+ cache.clear();
107
+ }
108
+
109
+ function describe(
110
+ registry: VisionRegistry,
111
+ delegate: VisionModel,
112
+ image: Block,
113
+ hint: string,
114
+ signal?: AbortSignal,
115
+ ): Promise<string> {
116
+ const key = `${specOf(delegate)}:${createHash("sha256").update(image.data ?? "").digest("hex")}`;
117
+ const hit = cache.get(key);
118
+ if (hit) return hit;
119
+ const job = (async () => {
120
+ const res = await registry.complete(
121
+ delegate,
122
+ {
123
+ systemPrompt: SYSTEM_PROMPT,
124
+ messages: [
125
+ {
126
+ role: "user",
127
+ content: [
128
+ { type: "text", text: hint ? `Context it appeared in:\n${hint.slice(0, 2000)}\n\nDescribe this image.` : "Describe this image." },
129
+ { type: "image", data: image.data, mimeType: image.mimeType },
130
+ ],
131
+ timestamp: Date.now(),
132
+ },
133
+ ],
134
+ },
135
+ { signal, cacheRetention: "none" },
136
+ );
137
+ if (res.stopReason === "error" || res.stopReason === "aborted") {
138
+ throw new Error(res.errorMessage || `vision model ${res.stopReason}`);
139
+ }
140
+ const text = res.content.filter((c) => c?.type === "text").map((c) => c.text).join("\n").trim();
141
+ if (!text) throw new Error("vision model returned no description");
142
+ return text;
143
+ })();
144
+ cache.set(key, job);
145
+ // A failure must not stick: the next turn should get another try.
146
+ job.catch(() => cache.delete(key));
147
+ return job;
148
+ }
149
+
150
+ /**
151
+ * Rewrite `messages` so every image block becomes a text description from
152
+ * `delegate`. Returns undefined when there was nothing to rewrite. Never throws: an
153
+ * image that can't be described becomes a note saying so (Pi would have dropped it
154
+ * anyway), so one bad image can't fail the turn.
155
+ */
156
+ export async function describeImagesInMessages(
157
+ messages: any[],
158
+ delegate: VisionModel,
159
+ registry: VisionRegistry,
160
+ opts: { currentSpec: string; signal?: AbortSignal; onDescribe?: (count: number) => void } = { currentSpec: "" },
161
+ ): Promise<any[] | undefined> {
162
+ const jobs: Array<{ block: Block; hint: string }> = [];
163
+ for (const msg of messages) {
164
+ if (!msg || msg.role === "assistant" || !Array.isArray(msg.content)) continue;
165
+ const hint = (msg.content as Block[])
166
+ .filter((b) => b?.type === "text")
167
+ .map((b) => b.text ?? "")
168
+ .join("\n")
169
+ .replace(OMITTED_NOTE, "");
170
+ for (const b of msg.content as Block[]) if (b?.type === "image" && b.data) jobs.push({ block: b, hint });
171
+ }
172
+ if (jobs.length === 0) return undefined;
173
+ opts.onDescribe?.(jobs.length);
174
+
175
+ const results = new Map<Block, string>();
176
+ await Promise.all(
177
+ jobs.map(async ({ block, hint }) => {
178
+ try {
179
+ const text = await describe(registry, delegate, block, hint, opts.signal);
180
+ results.set(block, `[Image — ${opts.currentSpec || "this model"} can't see images, so ${specOf(delegate)} described it:]\n${text}`);
181
+ } catch (err) {
182
+ const why = err instanceof Error ? err.message : String(err);
183
+ results.set(block, `[Image omitted — ${specOf(delegate)} could not describe it: ${why}]`);
184
+ }
185
+ }),
186
+ );
187
+
188
+ return messages.map((msg) => {
189
+ if (!msg || msg.role === "assistant" || !Array.isArray(msg.content)) return msg;
190
+ if (!(msg.content as Block[]).some((b) => results.has(b))) return msg;
191
+ const content = (msg.content as Block[]).map((b) => {
192
+ const described = results.get(b);
193
+ if (described !== undefined) return { type: "text", text: described };
194
+ if (b?.type === "text" && b.text) return { ...b, text: b.text.replace(OMITTED_NOTE, "") };
195
+ return b;
196
+ });
197
+ return { ...msg, content };
198
+ });
199
+ }