@vellumai/assistant 0.8.9-staging.1 → 0.8.9-staging.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/ARCHITECTURE.md +6 -6
  2. package/openapi.yaml +6 -7
  3. package/package.json +1 -1
  4. package/src/__tests__/agent-loop-exit-reason.test.ts +33 -18
  5. package/src/__tests__/app-builder-tool-scripts.test.ts +21 -0
  6. package/src/__tests__/app-executors.test.ts +132 -0
  7. package/src/__tests__/compactor-web-search-strip.test.ts +213 -0
  8. package/src/__tests__/context-overflow-reducer.test.ts +1 -1
  9. package/src/__tests__/conversation-agent-loop-inference-profile.test.ts +19 -16
  10. package/src/__tests__/conversation-agent-loop-overflow.test.ts +48 -29
  11. package/src/__tests__/conversation-agent-loop.test.ts +48 -96
  12. package/src/__tests__/conversation-history-web-search.test.ts +1 -1
  13. package/src/__tests__/conversation-provider-retry-repair.test.ts +52 -49
  14. package/src/__tests__/memory-retrieval-hook.test.ts +68 -30
  15. package/src/__tests__/pre-model-call-sanitize.test.ts +109 -0
  16. package/src/acp/__tests__/helpers/exec-file-stub.ts +12 -7
  17. package/src/acp/__tests__/prepare-agent-env.test.ts +0 -40
  18. package/src/acp/__tests__/session-manager-resume.test.ts +179 -27
  19. package/src/acp/auto-install.test.ts +130 -46
  20. package/src/acp/auto-install.ts +86 -31
  21. package/src/acp/prepare-agent-env.ts +8 -9
  22. package/src/acp/resolve-agent.test.ts +26 -123
  23. package/src/acp/resolve-agent.ts +26 -85
  24. package/src/acp/resume-hint.ts +1 -3
  25. package/src/acp/session-manager.ts +32 -20
  26. package/src/acp/types.ts +0 -8
  27. package/src/agent/loop.ts +42 -26
  28. package/src/config/bundled-skills/acp/SKILL.md +14 -14
  29. package/src/config/bundled-skills/acp/TOOLS.json +2 -2
  30. package/src/config/bundled-skills/app-builder/SKILL.md +3 -3
  31. package/src/config/bundled-skills/app-builder/TOOLS.json +43 -2
  32. package/src/config/bundled-skills/app-builder/tools/app-update.ts +18 -0
  33. package/src/config/bundled-tool-registry.ts +2 -0
  34. package/src/config/call-site-defaults.ts +0 -2
  35. package/src/config/feature-flag-registry.json +1 -1
  36. package/src/config/schemas/call-site-catalog.ts +0 -14
  37. package/src/config/schemas/llm.ts +0 -2
  38. package/src/context/compactor.ts +22 -4
  39. package/src/daemon/conversation-agent-loop.ts +15 -57
  40. package/src/daemon/conversation.ts +23 -3
  41. package/src/daemon/disk-pressure-policy.ts +0 -1
  42. package/src/daemon/lifecycle.ts +0 -7
  43. package/src/{daemon → plugins/defaults/compaction}/context-overflow-reducer.ts +7 -7
  44. package/src/plugins/defaults/compaction/manager-store.ts +13 -0
  45. package/src/plugins/defaults/memory-retrieval/hooks/post-compact.ts +6 -23
  46. package/src/plugins/defaults/memory-retrieval/hooks/user-prompt-submit-temp.ts +24 -25
  47. package/src/runtime/routes/__tests__/acp-routes.test.ts +71 -39
  48. package/src/runtime/routes/__tests__/plugins-routes.test.ts +29 -4
  49. package/src/runtime/routes/plugins-routes.ts +10 -9
  50. package/src/tools/acp/list-agents.test.ts +2 -2
  51. package/src/tools/acp/spawn.test.ts +103 -225
  52. package/src/tools/acp/spawn.ts +10 -120
  53. package/src/tools/apps/executors.ts +153 -42
  54. package/src/proactive-artifact/aux-message-injector.ts +0 -97
  55. package/src/proactive-artifact/decision.test.ts +0 -226
  56. package/src/proactive-artifact/decision.ts +0 -165
  57. package/src/proactive-artifact/index.ts +0 -7
  58. package/src/proactive-artifact/job.test.ts +0 -962
  59. package/src/proactive-artifact/job.ts +0 -372
  60. package/src/proactive-artifact/message-copy.ts +0 -58
  61. package/src/proactive-artifact/trigger-state.test.ts +0 -286
  62. package/src/proactive-artifact/trigger-state.ts +0 -123
@@ -0,0 +1,109 @@
1
+ import { describe, expect, test } from "bun:test";
2
+
3
+ import { preModelCallSanitize } from "../agent/loop.js";
4
+ import type { Message } from "../providers/types.js";
5
+
6
+ /**
7
+ * `preModelCallSanitize` is the loop's single pre-send transform: it converts
8
+ * historical `web_search_tool_result` blocks to text alongside the media and
9
+ * AX-tree strips, so every provider call — first call, post-compaction, and
10
+ * recovery reruns — is sanitized in one place. These tests guard that the
11
+ * helper actually performs the web-search conversion and is idempotent.
12
+ */
13
+ describe("preModelCallSanitize", () => {
14
+ test("passes through history with nothing to sanitize", () => {
15
+ // GIVEN a plain conversation with no media, AX trees, or web-search blocks
16
+ const messages: Message[] = [
17
+ { role: "user", content: [{ type: "text", text: "Hello" }] },
18
+ { role: "assistant", content: [{ type: "text", text: "Hi" }] },
19
+ ];
20
+
21
+ // WHEN the loop sanitizes the outbound history
22
+ const result = preModelCallSanitize(messages);
23
+
24
+ // THEN the history is returned unchanged
25
+ expect(result).toEqual(messages);
26
+ });
27
+
28
+ test("converts historical web_search_tool_result blocks to text summaries", () => {
29
+ // GIVEN an assistant turn whose web_search_tool_result carries an opaque
30
+ // encrypted_content token that would be rejected if replayed
31
+ const messages: Message[] = [
32
+ { role: "user", content: [{ type: "text", text: "Search cats" }] },
33
+ {
34
+ role: "assistant",
35
+ content: [
36
+ {
37
+ type: "server_tool_use",
38
+ id: "stu_1",
39
+ name: "web_search",
40
+ input: { query: "cats" },
41
+ },
42
+ {
43
+ type: "web_search_tool_result",
44
+ tool_use_id: "stu_1",
45
+ content: [
46
+ {
47
+ type: "web_search_result",
48
+ url: "https://cats.com",
49
+ title: "Cats!",
50
+ encrypted_content: "expired_token_1",
51
+ },
52
+ ],
53
+ },
54
+ ],
55
+ },
56
+ ];
57
+
58
+ // WHEN the loop sanitizes the outbound history
59
+ const result = preModelCallSanitize(messages);
60
+
61
+ // THEN the opaque block is replaced with a plaintext title+URL summary and
62
+ // the paired server_tool_use is dropped, so no expired token is replayed
63
+ const assistantMsg = result[1];
64
+ expect(assistantMsg.content.map((b) => b.type)).toEqual(["text"]);
65
+ const summary = assistantMsg.content[0];
66
+ expect(summary.type).toBe("text");
67
+ if (summary.type === "text") {
68
+ expect(summary.text).toContain("Cats!");
69
+ expect(summary.text).toContain("https://cats.com");
70
+ expect(summary.text).not.toContain("expired_token_1");
71
+ }
72
+ });
73
+
74
+ test("is idempotent — re-sanitizing already-sanitized history is a no-op", () => {
75
+ // GIVEN history that has already been sanitized once
76
+ const messages: Message[] = [
77
+ {
78
+ role: "assistant",
79
+ content: [
80
+ {
81
+ type: "server_tool_use",
82
+ id: "stu_A",
83
+ name: "web_search",
84
+ input: { query: "alpha" },
85
+ },
86
+ {
87
+ type: "web_search_tool_result",
88
+ tool_use_id: "stu_A",
89
+ content: [
90
+ {
91
+ type: "web_search_result",
92
+ url: "https://a.example",
93
+ title: "A",
94
+ encrypted_content: "tok_A",
95
+ },
96
+ ],
97
+ },
98
+ ],
99
+ },
100
+ ];
101
+ const once = preModelCallSanitize(messages);
102
+
103
+ // WHEN it is sanitized a second time (every outbound call re-runs the helper)
104
+ const twice = preModelCallSanitize(once);
105
+
106
+ // THEN the second pass changes nothing
107
+ expect(twice).toEqual(once);
108
+ });
109
+ });
@@ -1,11 +1,15 @@
1
1
  /**
2
2
  * Shared test helper: stub `execFile` from `node:child_process` for ACP tests.
3
3
  *
4
- * Several ACP suites (the `acp_spawn` tool's version probes, the adapter
5
- * auto-installer, and the `/v1/acp/spawn` route) shell out via
6
- * `execFileWithTimeout`. Each test file used to duplicate the same
7
- * `mock.module("node:child_process", ...)` + scripted-responses boilerplate;
8
- * this helper consolidates it, mirroring `which-stub.ts`.
4
+ * Several ACP suites (the adapter auto-installer and the `/v1/acp/spawn`
5
+ * route) shell out via `execFileWithTimeout` (e.g. `bun add --global`). Each
6
+ * test file used to duplicate the same `mock.module("node:child_process",
7
+ * ...)` + scripted-responses boilerplate; this helper consolidates it,
8
+ * mirroring `which-stub.ts`.
9
+ *
10
+ * The mock records every call's args, INCLUDING the options object (cwd, env,
11
+ * ...) at `execFileMock.mock.calls[i][2]`, so tests can assert the installer's
12
+ * sandboxed cwd and sanitized env.
9
13
  *
10
14
  * Like the other helpers here, the hook is process-global by design (Bun's
11
15
  * `mock.module` is process-global). Each test file should call
@@ -45,8 +49,9 @@ type ExecFileMock = ReturnType<
45
49
  export interface ExecFileStubHandle {
46
50
  /**
47
51
  * Per-call scripted responses, keyed by `${command} ${args[0]}` so tests
48
- * can target `npm ls`, `npm view`, and `npm i` independently. Calls with
49
- * no script reject with a recognizable "No script for <key>" error.
52
+ * can target distinct subcommands (e.g. `<bunPath> add`) independently.
53
+ * Calls with no script reject with a recognizable "No script for <key>"
54
+ * error.
50
55
  */
51
56
  execScripts: Map<string, ExecScript>;
52
57
  execFileMock: ExecFileMock;
@@ -216,22 +216,6 @@ describe("prepareAgentEnv — claude-agent-acp gating", () => {
216
216
  expect(prepared.env?.CLAUDE_CODE_OAUTH_TOKEN).toBe("vault-FFF");
217
217
  });
218
218
 
219
- test("injects the token for the bunx-rewritten claude adapter (adapterCommand gate)", async () => {
220
- // The resolver rewrites a missing claude-agent-acp binary to run via
221
- // `bun x --bun <pkg>` and preserves the canonical identity on
222
- // `adapterCommand`. Without this gate, bunx-resolved spawns would start
223
- // with no auth and die as zombies on the first prompt.
224
- seedVaultToken("vault-bunx");
225
-
226
- const prepared = await prepareAgentEnv({
227
- command: "bun",
228
- args: ["x", "--bun", "@agentclientprotocol/claude-agent-acp"],
229
- adapterCommand: "claude-agent-acp",
230
- });
231
-
232
- expect(prepared.env?.CLAUDE_CODE_OAUTH_TOKEN).toBe("vault-bunx");
233
- });
234
-
235
219
  test("does NOT mutate the caller's agentConfig", async () => {
236
220
  seedVaultToken("vault-GGG");
237
221
  const original = {
@@ -276,18 +260,6 @@ describe("prepareAgentEnv - gemini gating", () => {
276
260
  expect(meta!.allowedTools).toContain("acp_spawn");
277
261
  });
278
262
 
279
- test("injects the key for the bunx-rewritten gemini CLI (adapterCommand gate)", async () => {
280
- seedVaultGeminiKey("vault-gem-bunx");
281
-
282
- const prepared = await prepareAgentEnv({
283
- command: "bun",
284
- args: ["x", "--bun", "@google/gemini-cli", "--acp"],
285
- adapterCommand: "gemini",
286
- });
287
-
288
- expect(prepared.env?.GEMINI_API_KEY).toBe("vault-gem-bunx");
289
- });
290
-
291
263
  test("a vault miss does NOT throw and spawns without GEMINI_API_KEY (key is optional)", async () => {
292
264
  const prepared = await prepareAgentEnv({
293
265
  command: "gemini",
@@ -369,18 +341,6 @@ describe("prepareAgentEnv — non-claude commands", () => {
369
341
  expect(prepared.env).toEqual({});
370
342
  });
371
343
 
372
- test("no injection for a bunx-rewritten non-claude adapter", async () => {
373
- seedVaultToken("vault-should-not-leak");
374
-
375
- const prepared = await prepareAgentEnv({
376
- command: "bun",
377
- args: ["x", "--bun", "@zed-industries/codex-acp"],
378
- adapterCommand: "codex-acp",
379
- });
380
-
381
- expect(prepared.env).toEqual({});
382
- });
383
-
384
344
  test("returns the config unchanged for an unrecognized command basename", async () => {
385
345
  seedVaultToken("vault-HHH");
386
346
 
@@ -9,7 +9,8 @@
9
9
  * VellumAcpClientHandler so replay suppression is exercised end to end.
10
10
  */
11
11
 
12
- import { beforeEach, describe, expect, mock, test } from "bun:test";
12
+ import { tmpdir } from "node:os";
13
+ import { afterAll, beforeEach, describe, expect, mock, test } from "bun:test";
13
14
 
14
15
  mock.module("../../util/logger.js", () => ({
15
16
  getLogger: () =>
@@ -52,7 +53,7 @@ class FakeAcpAgentProcess {
52
53
  public readonly config: {
53
54
  command: string;
54
55
  args: string[];
55
- adapterCommand?: string;
56
+ env?: Record<string, string | undefined>;
56
57
  },
57
58
  private readonly clientFactory: (agent: unknown) => FakeClient,
58
59
  ) {
@@ -123,14 +124,30 @@ mock.module("../agent-process.js", () => ({
123
124
  AcpAgentProcess: FakeAcpAgentProcess,
124
125
  }));
125
126
 
126
- // Identity env-prep: credential-broker plumbing has its own suite. Tests
127
- // that need to observe the manager mid-resume (dispose, pending-id
128
- // visibility) set `prepareAgentEnvGate` to stall the resume here.
127
+ // Env-prep stub: credential-broker plumbing has its own suite. Tests that
128
+ // need to observe the manager mid-resume (dispose, pending-id visibility)
129
+ // set `prepareAgentEnvGate` to stall the resume here. Each resolved command
130
+ // it is called with is recorded so resume tests can assert it ran AFTER
131
+ // resolution (the real helper is the sole token-injection point), and it
132
+ // mirrors that contract by injecting CLAUDE_CODE_OAUTH_TOKEN into the
133
+ // returned config — at spawn time only, never during the auto-install phase.
129
134
  let prepareAgentEnvGate: Promise<void> | null = null;
135
+ let prepareAgentEnvCommands: string[] = [];
130
136
  mock.module("../prepare-agent-env.js", () => ({
131
- prepareAgentEnv: async (agentConfig: unknown) => {
137
+ prepareAgentEnv: async (agentConfig: {
138
+ command: string;
139
+ args: string[];
140
+ env?: Record<string, string | undefined>;
141
+ }) => {
132
142
  if (prepareAgentEnvGate) await prepareAgentEnvGate;
133
- return agentConfig;
143
+ prepareAgentEnvCommands.push(agentConfig.command);
144
+ return {
145
+ ...agentConfig,
146
+ env: {
147
+ ...agentConfig.env,
148
+ CLAUDE_CODE_OAUTH_TOKEN: process.env.CLAUDE_CODE_OAUTH_TOKEN,
149
+ },
150
+ };
134
151
  },
135
152
  }));
136
153
 
@@ -138,7 +155,7 @@ mock.module("../prepare-agent-env.js", () => ({
138
155
  type ResolveResult =
139
156
  | {
140
157
  ok: true;
141
- agent: { command: string; args: string[]; adapterCommand?: string };
158
+ agent: { command: string; args: string[] };
142
159
  }
143
160
  | { ok: false; reason: "binary_not_found"; hint: string; command: string };
144
161
  let resolveImpl: (id: string) => ResolveResult = () => ({
@@ -151,6 +168,23 @@ mock.module("../resolve-agent.js", () => ({
151
168
  resolveAcpAgent: (id: string) => resolveImpl(id),
152
169
  }));
153
170
 
171
+ // Auto-install stubs: resume now resolves through resolveAgentWithAutoInstall
172
+ // (the same sandboxed `bun` path as spawn), so a binary_not_found resolution
173
+ // reaches `ensureAdapterInstalled`, which probes `bun` via Bun.which and
174
+ // shells out via execFile. Stub both — process-global, installed BEFORE the
175
+ // session-manager (and thus auto-install) module is imported below — so the
176
+ // install is never a real one. Default: bun absent (no install attempted).
177
+ import { installExecFileStub } from "./helpers/exec-file-stub.js";
178
+ import { installWhichStub } from "./helpers/which-stub.js";
179
+
180
+ const { execScripts, execFileMock, reset: resetExecStub } =
181
+ installExecFileStub();
182
+ const which = installWhichStub();
183
+ /** Fixed resolved `bun` path so install script keys are predictable. */
184
+ const BUN_BIN = "/usr/local/bin/bun";
185
+ /** Key the exec stub uses for the global install. */
186
+ const BUN_ADD_KEY = `${BUN_BIN} add`;
187
+
154
188
  import type { ServerMessage } from "../../daemon/message-protocol.js";
155
189
  import type { AcpSessionUpdate } from "../../daemon/message-types/acp.js";
156
190
  import { getSqlite } from "../../memory/db-connection.js";
@@ -164,6 +198,9 @@ import {
164
198
 
165
199
  const { AcpResumeError, AcpSessionManager, AcpSessionNotFoundError } =
166
200
  await import("../session-manager.js");
201
+ // Imported dynamically (after the exec/which stubs above) so auto-install.js
202
+ // binds to the mocked node:child_process, exactly like session-manager.js.
203
+ const { _resetAdapterInstallCacheForTests } = await import("../auto-install.js");
167
204
 
168
205
  initializeDb();
169
206
 
@@ -190,6 +227,10 @@ function internals(
190
227
  return manager as unknown as ManagerInternals;
191
228
  }
192
229
 
230
+ afterAll(() => {
231
+ which.restore();
232
+ });
233
+
193
234
  beforeEach(() => {
194
235
  clearHistory();
195
236
  fakeInstances.length = 0;
@@ -198,11 +239,17 @@ beforeEach(() => {
198
239
  replayChunks = [];
199
240
  promptThrowsSync = false;
200
241
  prepareAgentEnvGate = null;
242
+ prepareAgentEnvCommands = [];
201
243
  resumeSessionGate = null;
202
244
  resolveImpl = () => ({
203
245
  ok: true,
204
246
  agent: { command: "claude-agent-acp", args: [] },
205
247
  });
248
+ // Default: bun absent, no install scripts, install cache cleared. Tests
249
+ // that exercise the auto-install path opt in via which/execScripts.
250
+ resetExecStub();
251
+ _resetAdapterInstallCacheForTests();
252
+ which.setWhich({});
206
253
  });
207
254
 
208
255
  const PERSISTED_EVENT: AcpSessionUpdate = {
@@ -329,55 +376,160 @@ describe("AcpSessionManager.resumeFromHistory", () => {
329
376
  expect(internals(manager).eventBuffers.has("no-caps-1")).toBe(false);
330
377
  });
331
378
 
332
- test("bunx-rewritten resolver output flows through resume with the canonical adapter command", async () => {
333
- // resolveAcpAgent rewrites a missing claude-agent-acp binary to run via
334
- // `bun x --bun <pkg>`; resumeFromHistory re-resolves through it, so the
335
- // rewritten config must reach the agent process while the SessionEntry
336
- // keeps the canonical adapter command (resume hints gate on it).
379
+ test("re-resolves via resolveAgentWithAutoInstall: an already-installed real binary flows through resume", async () => {
380
+ // resumeFromHistory resolves through resolveAgentWithAutoInstall. When the
381
+ // adapter is already on PATH the resolver returns the real binary (a full
382
+ // path here) with no install, and the SessionEntry tracks its basename for
383
+ // resume-hint gating.
337
384
  fakeCaps.resume = true;
338
- insertHistoryRow({ id: "bunx-resume-1" });
385
+ insertHistoryRow({ id: "installed-resume-1" });
339
386
  resolveImpl = () => ({
340
387
  ok: true,
341
388
  agent: {
342
- command: "bun",
343
- args: ["x", "--bun", "@agentclientprotocol/claude-agent-acp"],
344
- adapterCommand: "claude-agent-acp",
389
+ command: "/usr/local/bin/claude-agent-acp",
390
+ args: [],
345
391
  },
346
392
  });
347
393
 
348
394
  const manager = new AcpSessionManager(4);
349
- await manager.resumeFromHistory("bunx-resume-1", () => {});
395
+ await manager.resumeFromHistory("installed-resume-1", () => {});
350
396
 
351
397
  const fake = fakeInstances[0]!;
352
- expect(fake.config.command).toBe("bun");
353
- expect(fake.config.args).toEqual([
354
- "x",
355
- "--bun",
398
+ expect(fake.config.command).toBe("/usr/local/bin/claude-agent-acp");
399
+ expect(fake.resumeSessionCalls).toEqual([
400
+ { sessionId: "proto-old", cwd: "/tmp/proj" },
401
+ ]);
402
+ // The SessionEntry command is the basename (resume hints gate on it).
403
+ expect(
404
+ internals(manager).sessions.get("installed-resume-1")!.command,
405
+ ).toBe("claude-agent-acp");
406
+ });
407
+
408
+ test("missing adapter on resume: installs via sandboxed bun, then resumes against the real binary", async () => {
409
+ // The adapter binary is missing but maps to an allowlisted package and
410
+ // bun is present. resumeFromHistory must trigger the same one-time
411
+ // sandboxed install as spawn, then resume against the now-installed real
412
+ // binary, with the OAuth token injected only at spawn (never during the
413
+ // install).
414
+ fakeCaps.resume = true;
415
+ insertHistoryRow({ id: "resume-install-1" });
416
+
417
+ // Resolver: binary missing until the install flips `installed`, then the
418
+ // real installed binary (a full path, as a global bin would resolve).
419
+ let installed = false;
420
+ resolveImpl = () =>
421
+ installed
422
+ ? {
423
+ ok: true,
424
+ agent: { command: "/usr/local/bin/claude-agent-acp", args: [] },
425
+ }
426
+ : {
427
+ ok: false,
428
+ reason: "binary_not_found",
429
+ hint: "bun add -g @agentclientprotocol/claude-agent-acp",
430
+ command: "claude-agent-acp",
431
+ };
432
+ which.setWhich({ bun: BUN_BIN });
433
+ execScripts.set(BUN_ADD_KEY, {
434
+ stdout: "",
435
+ onCall: () => {
436
+ installed = true;
437
+ },
438
+ });
439
+
440
+ // Seed the secrets on the ambient env so we can assert they are stripped
441
+ // from the installer env and only reappear at spawn time.
442
+ process.env.CLAUDE_CODE_OAUTH_TOKEN = "should-not-leak";
443
+ process.env.GEMINI_API_KEY = "should-not-leak-either";
444
+
445
+ const manager = new AcpSessionManager(4);
446
+ try {
447
+ await manager.resumeFromHistory("resume-install-1", () => {});
448
+ } finally {
449
+ delete process.env.CLAUDE_CODE_OAUTH_TOKEN;
450
+ delete process.env.GEMINI_API_KEY;
451
+ }
452
+
453
+ // The install ran via bun (never npm) in a sandboxed temp dir (NOT the
454
+ // project dir), with the secrets stripped from its env.
455
+ expect(execFileMock).toHaveBeenCalledTimes(1);
456
+ const [command, args, options] = execFileMock.mock.calls[0];
457
+ expect(command).toBe(BUN_BIN);
458
+ expect(command).not.toBe("npm");
459
+ expect(args).toEqual([
460
+ "add",
461
+ "--global",
356
462
  "@agentclientprotocol/claude-agent-acp",
357
463
  ]);
464
+ const { cwd, env } = options as { cwd?: string; env?: NodeJS.ProcessEnv };
465
+ expect(cwd).toBeDefined();
466
+ expect(cwd!.startsWith(tmpdir())).toBe(true);
467
+ expect(cwd).not.toBe(process.cwd());
468
+ expect(cwd).toContain("vellum-acp-install-");
469
+ expect(env!.CLAUDE_CODE_OAUTH_TOKEN).toBeUndefined();
470
+ expect(env!.GEMINI_API_KEY).toBeUndefined();
471
+
472
+ // Resume then succeeded against the now-installed real binary.
473
+ const fake = fakeInstances[0]!;
474
+ expect(fake.config.command).toBe("/usr/local/bin/claude-agent-acp");
358
475
  expect(fake.resumeSessionCalls).toEqual([
359
476
  { sessionId: "proto-old", cwd: "/tmp/proj" },
360
477
  ]);
361
478
  expect(
362
- internals(manager).sessions.get("bunx-resume-1")!.command,
363
- ).toBe("claude-agent-acp");
479
+ (manager.getStatus("resume-install-1") as AcpSessionState).status,
480
+ ).toBe("running");
481
+
482
+ // prepareAgentEnv ran AFTER resolution, on the resolved real binary, and
483
+ // injected the token at spawn time — the sole point the token is in scope.
484
+ expect(prepareAgentEnvCommands).toEqual(["/usr/local/bin/claude-agent-acp"]);
485
+ expect(fake.config.env?.CLAUDE_CODE_OAUTH_TOKEN).toBe("should-not-leak");
486
+ });
487
+
488
+ test("install failure on resume surfaces the actionable error (no process spawned)", async () => {
489
+ fakeCaps.resume = true;
490
+ insertHistoryRow({ id: "resume-install-fail-1" });
491
+ resolveImpl = () => ({
492
+ ok: false,
493
+ reason: "binary_not_found",
494
+ hint: "bun add -g @agentclientprotocol/claude-agent-acp",
495
+ command: "claude-agent-acp",
496
+ });
497
+ which.setWhich({ bun: BUN_BIN });
498
+ execScripts.set(BUN_ADD_KEY, { error: new Error("network is down") });
499
+
500
+ const manager = new AcpSessionManager(4);
501
+ const promise = manager.resumeFromHistory("resume-install-fail-1", () => {});
502
+ await expect(promise).rejects.toThrow(/claude-agent-acp is not on PATH/);
503
+ await expect(promise).rejects.toThrow(
504
+ /auto-install failed: .*network is down/,
505
+ );
506
+
507
+ // The install was attempted via bun (never npm) but no child process
508
+ // spawned, and the reservation was released.
509
+ for (const call of execFileMock.mock.calls) {
510
+ expect(call[0]).not.toBe("npm");
511
+ }
512
+ expect(fakeInstances).toHaveLength(0);
513
+ expect(internals(manager).pendingResumes.size).toBe(0);
364
514
  });
365
515
 
366
- test("resolver failures surface the actionable hint", async () => {
516
+ test("bun absent on resume surfaces the actionable install hint, attempts no install", async () => {
367
517
  insertHistoryRow({ id: "no-bin-1" });
368
518
  resolveImpl = () => ({
369
519
  ok: false,
370
520
  reason: "binary_not_found",
371
- hint: "npm i -g @agentclientprotocol/claude-agent-acp",
521
+ hint: "bun add -g @agentclientprotocol/claude-agent-acp",
372
522
  command: "claude-agent-acp",
373
523
  });
524
+ which.setWhich({}); // bun not on PATH (the beforeEach default, made explicit)
374
525
 
375
526
  const manager = new AcpSessionManager(4);
376
527
  await expect(
377
528
  manager.resumeFromHistory("no-bin-1", () => {}),
378
529
  ).rejects.toThrow(
379
- "claude-agent-acp is not on PATH. npm i -g @agentclientprotocol/claude-agent-acp",
530
+ "claude-agent-acp is not on PATH. bun add -g @agentclientprotocol/claude-agent-acp",
380
531
  );
532
+ expect(execFileMock).not.toHaveBeenCalled();
381
533
  });
382
534
 
383
535
  test("already-active id and concurrency limit reuse spawn's guards", async () => {