@vellumai/assistant 0.8.12-staging.1 → 0.8.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/openapi.yaml +507 -0
  2. package/package.json +1 -1
  3. package/src/__tests__/llm-catalog-parity.test.ts +16 -0
  4. package/src/__tests__/log-export-workspace.test.ts +468 -3
  5. package/src/__tests__/secret-fixtures.ts +20 -0
  6. package/src/__tests__/tool-approval-handler.test.ts +85 -0
  7. package/src/__tests__/tool-audit-listener.test.ts +86 -0
  8. package/src/__tests__/workspace-migration-100-upgrade-quality-profile-to-fable-5.test.ts +174 -0
  9. package/src/__tests__/workspace-migration-101-upgrade-balanced-economy-to-minimax-m3.test.ts +162 -0
  10. package/src/acp/__tests__/agent-process.test.ts +315 -2
  11. package/src/acp/__tests__/prepare-agent-env.test.ts +79 -5
  12. package/src/acp/agent-process.ts +163 -34
  13. package/src/acp/prepare-agent-env.ts +55 -15
  14. package/src/bundler/app-compiler.ts +8 -0
  15. package/src/cli/lib/__tests__/upgrade-plugin.test.ts +10 -4
  16. package/src/cli/lib/upgrade-plugin.ts +13 -7
  17. package/src/config/seed-inference-profiles.ts +4 -8
  18. package/src/events/tool-audit-listener.ts +40 -9
  19. package/src/providers/__tests__/unparseable-tool-args.test.ts +53 -0
  20. package/src/providers/model-catalog.ts +28 -0
  21. package/src/providers/model-intents.ts +1 -1
  22. package/src/providers/openai/chat-completions-provider.ts +2 -1
  23. package/src/providers/openai/responses-provider.ts +2 -1
  24. package/src/providers/unparseable-tool-args.ts +56 -0
  25. package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +132 -0
  26. package/src/runtime/routes/__tests__/plugins-routes.test.ts +347 -0
  27. package/src/runtime/routes/conversation-query-routes.ts +79 -4
  28. package/src/runtime/routes/log-export-routes.ts +143 -96
  29. package/src/runtime/routes/plugins-routes.ts +359 -0
  30. package/src/runtime/routes/redact-staged-export.ts +259 -0
  31. package/src/security/redact-json.ts +61 -0
  32. package/src/tools/tool-approval-handler.ts +31 -0
  33. package/src/workspace/migrations/100-upgrade-quality-profile-to-fable-5.ts +86 -0
  34. package/src/workspace/migrations/101-upgrade-balanced-economy-to-minimax-m3.ts +70 -0
  35. package/src/workspace/migrations/registry.ts +4 -0
@@ -10,6 +10,8 @@ import { Readable, Writable } from "node:stream";
10
10
 
11
11
  import type {
12
12
  Agent,
13
+ AuthMethod,
14
+ AuthMethodEnvVar,
13
15
  Client,
14
16
  InitializeResponse,
15
17
  NewSessionResponse,
@@ -22,6 +24,31 @@ import type { AcpAgentConfig } from "./types.js";
22
24
 
23
25
  const log = getLogger("acp");
24
26
 
27
+ /**
28
+ * JSON-RPC error code agents use to signal that authentication is required
29
+ * (matches the SDK's RequestError.authRequired()).
30
+ */
31
+ const AUTH_REQUIRED_CODE = -32000;
32
+
33
+ /**
34
+ * Detects the ACP auth-required error. Checks the `code` property rather than
35
+ * `instanceof acp.RequestError` so plain JSON-RPC error objects are also
36
+ * recognized.
37
+ */
38
+ function isAuthRequiredError(err: unknown): boolean {
39
+ return (
40
+ typeof err === "object" &&
41
+ err !== null &&
42
+ (err as { code?: unknown }).code === AUTH_REQUIRED_CODE
43
+ );
44
+ }
45
+
46
+ function isEnvVarMethod(
47
+ method: AuthMethod,
48
+ ): method is AuthMethodEnvVar & { type: "env_var" } {
49
+ return "type" in method && method.type === "env_var";
50
+ }
51
+
25
52
  /**
26
53
  * Factory function type for creating ACP client handlers.
27
54
  * PR 5 will provide the real VellumAcpClientHandler implementation.
@@ -35,6 +62,12 @@ export class AcpAgentProcess {
35
62
  private proc: ChildProcess | null = null;
36
63
  private connection: acp.ClientSideConnection | null = null;
37
64
  private initializeResponse: InitializeResponse | null = null;
65
+ /**
66
+ * Merged env captured at spawn() so auth satisfiability checks match the
67
+ * env the child process actually received, even if process.env changes
68
+ * afterwards.
69
+ */
70
+ private spawnedEnv: NodeJS.ProcessEnv | null = null;
38
71
 
39
72
  constructor(
40
73
  public readonly agentId: string,
@@ -51,10 +84,11 @@ export class AcpAgentProcess {
51
84
  "Spawning ACP agent process",
52
85
  );
53
86
 
87
+ this.spawnedEnv = { ...process.env, ...this.config.env };
54
88
  this.proc = spawn(this.config.command, this.config.args, {
55
89
  cwd,
56
90
  stdio: ["pipe", "pipe", "pipe"],
57
- env: { ...process.env, ...this.config.env },
91
+ env: this.spawnedEnv,
58
92
  });
59
93
 
60
94
  const stream = acp.ndJsonStream(
@@ -94,13 +128,11 @@ export class AcpAgentProcess {
94
128
  * Initializes the ACP connection by negotiating protocol version and capabilities.
95
129
  */
96
130
  async initialize(): Promise<InitializeResponse> {
97
- if (!this.connection) {
98
- throw new Error(`ACP agent "${this.agentId}" is not spawned`);
99
- }
131
+ const connection = this.requireConnection();
100
132
 
101
133
  log.info({ agentId: this.agentId }, "Initializing ACP connection");
102
134
 
103
- const response = await this.connection.initialize({
135
+ const response = await connection.initialize({
104
136
  protocolVersion: acp.PROTOCOL_VERSION,
105
137
  clientInfo: { name: "vellum", version: "1.0.0" },
106
138
  clientCapabilities: {
@@ -133,20 +165,119 @@ export class AcpAgentProcess {
133
165
  }
134
166
 
135
167
  /**
136
- * Creates a new ACP session in the specified working directory.
137
- * Returns the session ID.
168
+ * Authentication methods the agent advertised at initialize.
169
+ * Returns an empty array before initialize() resolves.
138
170
  */
139
- async createSession(cwd: string): Promise<string> {
171
+ private get authMethods(): AuthMethod[] {
172
+ return this.initializeResponse?.authMethods ?? [];
173
+ }
174
+
175
+ /**
176
+ * Selects the first advertised env_var auth method whose required variables
177
+ * are all present (non-empty) in the env the agent process was spawned with.
178
+ *
179
+ * Terminal-type and agent-driven (untyped) methods are never selected:
180
+ * auto-triggering an interactive login would hang the headless daemon.
181
+ */
182
+ private selectEnvVarAuthMethod(): AuthMethod | undefined {
183
+ const env = this.spawnedEnv;
184
+ if (!env) return undefined;
185
+
186
+ return this.authMethods.find((method) => {
187
+ if (!isEnvVarMethod(method)) return false;
188
+
189
+ // `vars` is required by the SDK type, but agent responses aren't
190
+ // runtime-validated — tolerate an out-of-spec agent omitting it so the
191
+ // caller gets the friendly auth error instead of a TypeError.
192
+ const requiredVars = (method.vars ?? []).filter((v) => !v.optional);
193
+ if (requiredVars.length === 0) return false;
194
+
195
+ return requiredVars.every((v) => {
196
+ const value = env[v.name];
197
+ return typeof value === "string" && value.length > 0;
198
+ });
199
+ });
200
+ }
201
+
202
+ /**
203
+ * Returns the live connection, throwing the standard not-spawned error if
204
+ * the agent was never spawned or its process has since exited.
205
+ */
206
+ private requireConnection(): acp.ClientSideConnection {
140
207
  if (!this.connection) {
141
208
  throw new Error(`ACP agent "${this.agentId}" is not spawned`);
142
209
  }
210
+ return this.connection;
211
+ }
212
+
213
+ /**
214
+ * Runs an operation, and if the agent rejects with the ACP auth-required
215
+ * error, authenticates via a satisfiable env_var auth method and retries
216
+ * the operation exactly once.
217
+ */
218
+ private async withAuthRetry<T>(op: () => Promise<T>): Promise<T> {
219
+ try {
220
+ return await op();
221
+ } catch (err) {
222
+ if (!isAuthRequiredError(err)) throw err;
223
+
224
+ // The agent may have exited between the auth_required rejection and
225
+ // this retry path; fail with the standard not-spawned error.
226
+ const connection = this.requireConnection();
227
+
228
+ const method = this.selectEnvVarAuthMethod();
229
+ if (!method) {
230
+ throw new Error(
231
+ `ACP agent "${this.agentId}" requires authentication. ` +
232
+ `Advertised methods: ${this.describeAuthMethods()}. ` +
233
+ "Set the required env var under acp.agents.<id>.env in config.json, " +
234
+ "store it via 'assistant credentials set --service acp --field <field>', " +
235
+ "or complete the agent's own login flow in the workspace.",
236
+ );
237
+ }
238
+
239
+ log.info(
240
+ { agentId: this.agentId, methodId: method.id },
241
+ "ACP agent returned auth_required; authenticating with env_var method",
242
+ );
143
243
 
244
+ await connection.authenticate({ methodId: method.id });
245
+ return await op();
246
+ }
247
+ }
248
+
249
+ /**
250
+ * Renders the agent's advertised auth methods for error messages, e.g.
251
+ * `"Login with ChatGPT" (chatgpt), "Use OPENAI_API_KEY" (env var OPENAI_API_KEY)`.
252
+ */
253
+ private describeAuthMethods(): string {
254
+ if (this.authMethods.length === 0) return "none";
255
+
256
+ return this.authMethods
257
+ .map((method) => {
258
+ // `vars ?? []`: tolerate out-of-spec agents omitting the field —
259
+ // this renders inside the friendly auth error, which must not
260
+ // itself throw a TypeError.
261
+ const varNames = isEnvVarMethod(method)
262
+ ? (method.vars ?? []).map((v) => v.name).join(", ")
263
+ : "";
264
+ return varNames
265
+ ? `"${method.name}" (env var ${varNames})`
266
+ : `"${method.name}" (${method.id})`;
267
+ })
268
+ .join(", ");
269
+ }
270
+
271
+ /**
272
+ * Creates a new ACP session in the specified working directory.
273
+ * Returns the session ID.
274
+ */
275
+ async createSession(cwd: string): Promise<string> {
144
276
  log.info({ agentId: this.agentId, cwd }, "Creating ACP session");
145
277
 
146
- const result: NewSessionResponse = await this.connection.newSession({
147
- cwd,
148
- mcpServers: [],
149
- });
278
+ const result: NewSessionResponse = await this.withAuthRetry(() =>
279
+ this.requireConnection().newSession({ cwd, mcpServers: [] }),
280
+ );
150
281
 
151
282
  return result.sessionId;
152
283
  }
@@ -160,13 +291,11 @@ export class AcpAgentProcess {
160
291
  * VellumAcpClientHandler.beginReplaySuppression).
161
292
  */
162
293
  async loadSession(sessionId: string, cwd: string): Promise<void> {
163
- if (!this.connection) {
164
- throw new Error(`ACP agent "${this.agentId}" is not spawned`);
165
- }
166
-
167
294
  log.info({ agentId: this.agentId, sessionId, cwd }, "Loading ACP session");
168
295
 
169
- await this.connection.loadSession({ sessionId, cwd, mcpServers: [] });
296
+ await this.withAuthRetry(() =>
297
+ this.requireConnection().loadSession({ sessionId, cwd, mcpServers: [] }),
298
+ );
170
299
  }
171
300
 
172
301
  /**
@@ -177,13 +306,15 @@ export class AcpAgentProcess {
177
306
  * (see supportsSessionResume).
178
307
  */
179
308
  async resumeSession(sessionId: string, cwd: string): Promise<void> {
180
- if (!this.connection) {
181
- throw new Error(`ACP agent "${this.agentId}" is not spawned`);
182
- }
183
-
184
309
  log.info({ agentId: this.agentId, sessionId, cwd }, "Resuming ACP session");
185
310
 
186
- await this.connection.resumeSession({ sessionId, cwd, mcpServers: [] });
311
+ await this.withAuthRetry(() =>
312
+ this.requireConnection().resumeSession({
313
+ sessionId,
314
+ cwd,
315
+ mcpServers: [],
316
+ }),
317
+ );
187
318
  }
188
319
 
189
320
  /**
@@ -191,35 +322,31 @@ export class AcpAgentProcess {
191
322
  * Returns the prompt response (includes stopReason).
192
323
  */
193
324
  async prompt(sessionId: string, text: string): Promise<PromptResponse> {
194
- if (!this.connection) {
195
- throw new Error(`ACP agent "${this.agentId}" is not spawned`);
196
- }
197
-
198
325
  log.info(
199
326
  { agentId: this.agentId, sessionId },
200
327
  "Sending prompt to ACP agent",
201
328
  );
202
329
 
203
- return this.connection.prompt({
204
- sessionId,
205
- prompt: [{ type: "text", text }],
206
- });
330
+ return this.withAuthRetry(() =>
331
+ this.requireConnection().prompt({
332
+ sessionId,
333
+ prompt: [{ type: "text", text }],
334
+ }),
335
+ );
207
336
  }
208
337
 
209
338
  /**
210
339
  * Cancels an ongoing prompt in the specified session.
211
340
  */
212
341
  async cancel(sessionId: string): Promise<void> {
213
- if (!this.connection) {
214
- throw new Error(`ACP agent "${this.agentId}" is not spawned`);
215
- }
342
+ const connection = this.requireConnection();
216
343
 
217
344
  log.info(
218
345
  { agentId: this.agentId, sessionId },
219
346
  "Cancelling ACP session prompt",
220
347
  );
221
348
 
222
- await this.connection.cancel({ sessionId });
349
+ await connection.cancel({ sessionId });
223
350
  }
224
351
 
225
352
  /**
@@ -234,6 +361,7 @@ export class AcpAgentProcess {
234
361
  }
235
362
  this.connection = null;
236
363
  this.initializeResponse = null;
364
+ this.spawnedEnv = null;
237
365
  }
238
366
 
239
367
  /**
@@ -265,5 +393,6 @@ export class AcpAgentProcess {
265
393
  this.proc = null;
266
394
  this.connection = null;
267
395
  this.initializeResponse = null;
396
+ this.spawnedEnv = null;
268
397
  }
269
398
  }
@@ -93,6 +93,32 @@ async function injectCredential(
93
93
  return result.success ? undefined : result.reason;
94
94
  }
95
95
 
96
+ /**
97
+ * Inject an OPTIONAL credential: skip when the env var is already set
98
+ * (config.json override wins), and treat a vault miss as non-fatal — the
99
+ * adapter has its own login fallback, so spawning without the key is fine.
100
+ */
101
+ async function injectOptionalCredential(
102
+ env: Record<string, string>,
103
+ field: string,
104
+ envVar: string,
105
+ usageDescription: string,
106
+ ): Promise<void> {
107
+ if (env[envVar]) return;
108
+ const missReason = await injectCredential(
109
+ env,
110
+ field,
111
+ envVar,
112
+ usageDescription,
113
+ );
114
+ if (missReason !== undefined) {
115
+ log.debug(
116
+ { reason: missReason },
117
+ `${envVar} unavailable from the vault; spawning without it`,
118
+ );
119
+ }
120
+ }
121
+
96
122
  /**
97
123
  * Returns a NEW config with any required credentials merged into `env`.
98
124
  * Does NOT mutate the input. Throws `FailedDependencyError` if a required
@@ -123,6 +149,13 @@ async function injectCredential(
123
149
  * ways (config.json override wins, vault field `gemini_api_key` second),
124
150
  * but it is OPTIONAL: the Gemini CLI supports its own OAuth login, so a
125
151
  * vault miss proceeds without the key instead of failing the spawn.
152
+ *
153
+ * For `codex-acp` the env vars are `OPENAI_API_KEY` (vault field
154
+ * `acp/openai_api_key`) and `CODEX_API_KEY` (vault field
155
+ * `acp/codex_api_key`), provisioned the same two ways (config.json
156
+ * override wins, vault second). Both are OPTIONAL: codex also supports
157
+ * ChatGPT login (`codex login` pre-seeding `auth.json` in the workspace),
158
+ * so a vault miss proceeds without the key instead of failing the spawn.
126
159
  */
127
160
  export async function prepareAgentEnv(
128
161
  agentConfig: AcpAgentConfig,
@@ -150,22 +183,29 @@ export async function prepareAgentEnv(
150
183
  );
151
184
  }
152
185
  } else if (adapterCommand === "gemini") {
153
- if (!env.GEMINI_API_KEY) {
154
- const missReason = await injectCredential(
186
+ await injectOptionalCredential(
187
+ env,
188
+ "gemini_api_key",
189
+ "GEMINI_API_KEY",
190
+ "Gemini API key for ACP agent authentication",
191
+ );
192
+ } else if (adapterCommand === "codex-acp") {
193
+ // The two reads target independent vault fields and write disjoint env
194
+ // keys, so running them concurrently is safe.
195
+ await Promise.all([
196
+ injectOptionalCredential(
155
197
  env,
156
- "gemini_api_key",
157
- "GEMINI_API_KEY",
158
- "Gemini API key for ACP agent authentication",
159
- );
160
- if (missReason !== undefined) {
161
- // Optional credential: Gemini CLI can authenticate via its own
162
- // OAuth login, so a vault miss must not fail the spawn.
163
- log.debug(
164
- { reason: missReason },
165
- "Gemini API key unavailable from the vault; spawning without GEMINI_API_KEY",
166
- );
167
- }
168
- }
198
+ "openai_api_key",
199
+ "OPENAI_API_KEY",
200
+ "OpenAI API key for Codex ACP agent authentication",
201
+ ),
202
+ injectOptionalCredential(
203
+ env,
204
+ "codex_api_key",
205
+ "CODEX_API_KEY",
206
+ "Codex API key for Codex ACP agent authentication",
207
+ ),
208
+ ]);
169
209
  }
170
210
 
171
211
  return { ...agentConfig, env };
@@ -454,6 +454,14 @@ async function runCompile(appDir: string): Promise<CompileResult> {
454
454
  if (existsSync(htmlSrc)) {
455
455
  let html = await readFile(htmlSrc, "utf-8");
456
456
 
457
+ // Strip source-file script tags (e.g. <script src="/src/main.tsx">) that
458
+ // models often write in Vite style; browsers cannot load raw TSX/TS/JSX.
459
+ // The compiled main.js tag is injected below instead.
460
+ html = html.replace(
461
+ /<script\b[^>]*\bsrc=["'][^"']*\.(?:tsx|ts|jsx)["'][^>]*>\s*<\/script>\s*/gi,
462
+ "",
463
+ );
464
+
457
465
  // Check if CSS output was produced
458
466
  const distFiles = await readdir(distDir);
459
467
  const hasCss = distFiles.some((f) => f.endsWith(".css"));
@@ -21,7 +21,11 @@ import { tmpdir } from "node:os";
21
21
  import { join } from "node:path";
22
22
  import { afterEach, beforeEach, describe, expect, test } from "bun:test";
23
23
 
24
- import type { FetchLike, GitRunner } from "../install-from-github.js";
24
+ import {
25
+ type FetchLike,
26
+ type GitRunner,
27
+ PluginSourceUnavailableError,
28
+ } from "../install-from-github.js";
25
29
  import { PluginNotInstalledError } from "../uninstall-plugin.js";
26
30
  import { PluginNotUpgradableError, upgradePlugin } from "../upgrade-plugin.js";
27
31
 
@@ -262,19 +266,21 @@ describe("upgradePlugin", () => {
262
266
  ).rejects.toBeInstanceOf(PluginNotUpgradableError);
263
267
  });
264
268
 
265
- test("throws PluginNotUpgradableError when the marketplace is unreachable", async () => {
269
+ test("throws PluginSourceUnavailableError when the marketplace is unreachable", async () => {
266
270
  // GIVEN an installed copy and a marketplace fetch that fails transiently
267
271
  installCopy(pluginsDir, "level-up", { commit: SHA_A });
268
272
  const fetch = makeFetch({ manifestStatus: 500 });
269
273
 
270
274
  // WHEN an upgrade is attempted
271
- // THEN the latest pin cannot be determined, so it refuses
275
+ // THEN the outage is surfaced as a retryable source-unavailable error
276
+ // (distinct from the permanent no-marketplace-entry conflict), since the
277
+ // same request can succeed once the catalog recovers
272
278
  await expect(
273
279
  upgradePlugin(
274
280
  { name: "level-up" },
275
281
  { fetch, runGit: unusedGitRunner, workspacePluginsDir: pluginsDir },
276
282
  ),
277
- ).rejects.toBeInstanceOf(PluginNotUpgradableError);
283
+ ).rejects.toBeInstanceOf(PluginSourceUnavailableError);
278
284
  });
279
285
 
280
286
  test("preserves the existing install when the re-install clone fails", async () => {
@@ -37,6 +37,7 @@ import {
37
37
  type FetchLike,
38
38
  type GitRunner,
39
39
  installPlugin,
40
+ PluginSourceUnavailableError,
40
41
  type PostinstallRunner,
41
42
  sanitizePluginName,
42
43
  } from "./install-from-github.js";
@@ -116,10 +117,11 @@ function pluginTarget(name: string, deps: UpgradePluginDeps): string {
116
117
  * Move an installed plugin to the marketplace's current pin.
117
118
  *
118
119
  * Throws {@link PluginNotInstalledError} when no copy is installed,
119
- * {@link PluginNotUpgradableError} when the install has no marketplace pin to
120
- * advance to (no catalog entry, or the catalog was unreachable), and
121
- * propagates {@link installPlugin}'s errors (e.g. source unavailable,
122
- * postinstall failure) when the re-install itself fails.
120
+ * {@link PluginNotUpgradableError} when the install has no marketplace entry to
121
+ * advance to, {@link PluginSourceUnavailableError} when the marketplace catalog
122
+ * is temporarily unreachable (a retryable outage, distinct from the permanent
123
+ * no-entry case), and propagates {@link installPlugin}'s errors (e.g. source
124
+ * unavailable, postinstall failure) when the re-install itself fails.
123
125
  */
124
126
  export async function upgradePlugin(
125
127
  opts: UpgradePluginOptions,
@@ -150,9 +152,13 @@ export async function upgradePlugin(
150
152
  "it has no marketplace entry to upgrade from",
151
153
  );
152
154
  case "remote-unavailable":
153
- throw new PluginNotUpgradableError(
154
- name,
155
- `the marketplace could not be reached (${inspection.remoteError ?? "unknown error"})`,
155
+ // A transient catalog outage is not a permanent "cannot upgrade" state:
156
+ // the same request can succeed once the marketplace source recovers, so
157
+ // surface it as a retryable source-unavailable error rather than a
158
+ // conflict.
159
+ throw new PluginSourceUnavailableError(
160
+ `Plugin "${name}" cannot be upgraded: the marketplace could not be reached (${inspection.remoteError ?? "unknown error"}).`,
161
+ 503,
156
162
  );
157
163
  }
158
164
 
@@ -74,23 +74,19 @@ const MANAGED_PROFILE_TEMPLATES: Record<string, ManagedProfileTemplate> = {
74
74
  thinking: { enabled: false, streamThinking: false },
75
75
  contextWindow: { maxInputTokens: DEFAULT_CONTEXT_WINDOW_MAX_INPUT_TOKENS },
76
76
  },
77
- // Open-weight economy option: Kimi K2.6 served by Fireworks via managed
78
- // platform inference. Carries the `suppress-cjk` logit-bias preset to
79
- // discourage the model from spontaneously emitting Chinese in English
80
- // output; the preset is profile-scoped and only forwarded on the Fireworks
81
- // path (see `providers/inference/logit-bias.ts`).
77
+ // Open-weight economy option: MiniMax M3 served by Fireworks via managed
78
+ // platform inference.
82
79
  "balanced-economy": {
83
80
  intent: "balanced",
84
81
  provider: "fireworks",
85
82
  connectionName: "fireworks-managed",
86
83
  source: "managed",
87
84
  label: "Balanced Economy",
88
- description: "Strong open model (Kimi K2.6) at a lower price point",
89
- maxTokens: 16000,
85
+ description: "Strong open model (MiniMax M3) at a lower price point",
86
+ maxTokens: 32000,
90
87
  effort: "high",
91
88
  thinking: { enabled: true, streamThinking: true },
92
89
  contextWindow: { maxInputTokens: DEFAULT_CONTEXT_WINDOW_MAX_INPUT_TOKENS },
93
- logitBias: "suppress-cjk",
94
90
  },
95
91
  };
96
92
 
@@ -3,6 +3,7 @@ import {
3
3
  recordToolInvocation,
4
4
  type ToolInvocationRecord,
5
5
  } from "../memory/tool-usage-store.js";
6
+ import { redactJsonStringLeaves } from "../security/redact-json.js";
6
7
  import { redactSecrets } from "../security/secret-scanner.js";
7
8
  import {
8
9
  stringifyToolInput,
@@ -43,11 +44,14 @@ function toInvocationRecord(
43
44
  ): ToolInvocationRecord | null {
44
45
  switch (event.type) {
45
46
  case "executed": {
46
- const input = stringifyToolInput(event.input);
47
+ const rawInput = stringifyToolInput(event.input);
47
48
  return {
48
49
  conversationId: event.conversationId,
49
50
  toolName: event.toolName,
50
- input,
51
+ // Inputs can carry secrets the model typed verbatim (e.g.
52
+ // `export OPENAI_API_KEY=...` in a bash command) — redact before
53
+ // the row reaches the audit store, like results below.
54
+ input: redactToolInput(event.input, rawInput),
51
55
  result: redactSecrets(event.result.content).slice(
52
56
  0,
53
57
  RESULT_PREVIEW_LIMIT,
@@ -63,18 +67,18 @@ function toInvocationRecord(
63
67
  // don't stamp.
64
68
  ...telemetryColumns(
65
69
  event,
66
- input,
70
+ rawInput,
67
71
  event.resultBytes ?? Buffer.byteLength(event.result.content, "utf8"),
68
72
  ),
69
73
  };
70
74
  }
71
75
  case "error": {
72
- const input = stringifyToolInput(event.input);
76
+ const rawInput = stringifyToolInput(event.input);
73
77
  const result = `error: ${event.errorMessage}`;
74
78
  return {
75
79
  conversationId: event.conversationId,
76
80
  toolName: event.toolName,
77
- input,
81
+ input: redactToolInput(event.input, rawInput),
78
82
  result,
79
83
  decision: "error",
80
84
  riskLevel: event.riskLevel,
@@ -83,7 +87,7 @@ function toInvocationRecord(
83
87
  // The error result string is built right here and never goes
84
88
  // through sensitive-output sanitization, so sizing it directly is
85
89
  // already raw — no executor stamp exists or is needed.
86
- ...telemetryColumns(event, input, Buffer.byteLength(result, "utf8")),
90
+ ...telemetryColumns(event, rawInput, Buffer.byteLength(result, "utf8")),
87
91
  };
88
92
  }
89
93
  case "permission_denied":
@@ -92,7 +96,7 @@ function toInvocationRecord(
92
96
  return {
93
97
  conversationId: event.conversationId,
94
98
  toolName: event.toolName,
95
- input: stringifyToolInput(event.input),
99
+ input: redactToolInput(event.input, stringifyToolInput(event.input)),
96
100
  result: formatDeniedResult(event.reason),
97
101
  decision: "denied",
98
102
  riskLevel: event.riskLevel,
@@ -105,6 +109,33 @@ function toInvocationRecord(
105
109
  }
106
110
  }
107
111
 
112
+ /**
113
+ * Redact secrets from a tool input while keeping the stored audit string
114
+ * parseable JSON. The redaction marker (`<redacted type="..." />`) contains
115
+ * double quotes, so redacting the serialized string would corrupt it —
116
+ * instead, walk the input's string leaves BEFORE stringification so the
117
+ * marker lands inside a JSON string value (with its quotes escaped).
118
+ *
119
+ * `rawInput` is the canonical pre-redaction serialization (also used for
120
+ * the `argBytes` telemetry fallback — byte sizes must reflect the full
121
+ * payload before truncation and redaction). It is returned untouched when
122
+ * nothing matched, keeping benign inputs byte-identical, and is redacted as
123
+ * plain text if the input can't be walked or re-serialized (e.g. cyclic
124
+ * structures).
125
+ */
126
+ function redactToolInput(
127
+ input: Record<string, unknown>,
128
+ rawInput: string,
129
+ ): string {
130
+ try {
131
+ const { value, changed } = redactJsonStringLeaves(input);
132
+ if (!changed) return rawInput;
133
+ return JSON.stringify(value);
134
+ } catch {
135
+ return redactSecrets(rawInput);
136
+ }
137
+ }
138
+
108
139
  type TelemetryColumns = Pick<ToolInvocationRecord, "argBytes" | "resultBytes"> &
109
140
  UsageAttributionColumns;
110
141
 
@@ -134,12 +165,12 @@ const NULL_TELEMETRY_COLUMNS: TelemetryColumns = {
134
165
  */
135
166
  function telemetryColumns(
136
167
  event: Extract<ToolLifecycleEvent, { type: "executed" | "error" }>,
137
- input: string,
168
+ rawInput: string,
138
169
  resultBytes: number,
139
170
  ): TelemetryColumns {
140
171
  if (!getConfig().collectUsageData) return NULL_TELEMETRY_COLUMNS;
141
172
  return {
142
- argBytes: event.inputBytes ?? Buffer.byteLength(input, "utf8"),
173
+ argBytes: event.inputBytes ?? Buffer.byteLength(rawInput, "utf8"),
143
174
  resultBytes,
144
175
  ...toAttributionColumns(event.attribution),
145
176
  };
@@ -0,0 +1,53 @@
1
+ import { describe, expect, test } from "bun:test";
2
+
3
+ import {
4
+ isUnparseableToolArgs,
5
+ unparseableToolArgsMessage,
6
+ wrapUnparseableToolArgs,
7
+ } from "../unparseable-tool-args.js";
8
+
9
+ describe("unparseable tool args marker", () => {
10
+ test("wrap/detect roundtrip", () => {
11
+ const wrapped = wrapUnparseableToolArgs(
12
+ '{"surface_type": "card", "data": ',
13
+ );
14
+ expect(isUnparseableToolArgs(wrapped)).toBe(true);
15
+ });
16
+
17
+ test("detects empty raw string", () => {
18
+ expect(isUnparseableToolArgs(wrapUnparseableToolArgs(""))).toBe(true);
19
+ });
20
+
21
+ test("does not match input with additional keys", () => {
22
+ expect(isUnparseableToolArgs({ _raw: "x", other: 1 })).toBe(false);
23
+ });
24
+
25
+ test("does not match non-string _raw", () => {
26
+ expect(isUnparseableToolArgs({ _raw: { nested: true } })).toBe(false);
27
+ });
28
+
29
+ test("does not match ordinary tool input", () => {
30
+ expect(isUnparseableToolArgs({ command: "ls" })).toBe(false);
31
+ expect(isUnparseableToolArgs({})).toBe(false);
32
+ });
33
+
34
+ test("message includes tool name, raw preview, and retry instruction", () => {
35
+ const msg = unparseableToolArgsMessage("ui_show", '{"surface_type": ');
36
+ expect(msg).toContain('"ui_show"');
37
+ expect(msg).toContain('{"surface_type": ');
38
+ expect(msg).toContain("NOT executed");
39
+ expect(msg).toContain("Retry");
40
+ });
41
+
42
+ test("message truncates long raw args", () => {
43
+ const raw = "a".repeat(1000);
44
+ const msg = unparseableToolArgsMessage("bash", raw);
45
+ expect(msg).not.toContain(raw);
46
+ expect(msg).toContain("a".repeat(200) + "…");
47
+ });
48
+
49
+ test("message handles empty raw args", () => {
50
+ const msg = unparseableToolArgsMessage("bash", "");
51
+ expect(msg).toContain("(empty)");
52
+ });
53
+ });