@vellumai/assistant 0.8.9-staging.2 → 0.8.9-staging.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/docs/activation-funnel-telemetry.md +310 -0
  2. package/package.json +1 -1
  3. package/src/__tests__/activation-early-marking.test.ts +120 -0
  4. package/src/__tests__/agent-loop-output-hooks.test.ts +13 -13
  5. package/src/__tests__/approval-cascade.test.ts +1 -1
  6. package/src/__tests__/compaction-direct.test.ts +32 -18
  7. package/src/__tests__/compaction-events.test.ts +2 -2
  8. package/src/__tests__/compaction.benchmark.test.ts +1 -1
  9. package/src/__tests__/context-overflow-reducer.test.ts +5 -5
  10. package/src/__tests__/context-window-manager-compact-retry.test.ts +1 -1
  11. package/src/__tests__/conversation-abort-tool-results.test.ts +1 -1
  12. package/src/__tests__/conversation-confirmation-signals.test.ts +1 -1
  13. package/src/__tests__/conversation-error.test.ts +15 -1
  14. package/src/__tests__/conversation-history-web-search.test.ts +5 -0
  15. package/src/__tests__/conversation-media-retry.test.ts +1 -1
  16. package/src/__tests__/conversation-process-app-control-preactivation.test.ts +40 -0
  17. package/src/__tests__/conversation-process-callsite.test.ts +1 -1
  18. package/src/__tests__/conversation-provider-retry-repair.test.ts +1 -1
  19. package/src/__tests__/conversation-queue.test.ts +1 -1
  20. package/src/__tests__/conversation-runtime-assembly.test.ts +71 -0
  21. package/src/__tests__/conversation-slash-queue.test.ts +1 -1
  22. package/src/__tests__/conversation-slash-unknown.test.ts +1 -1
  23. package/src/__tests__/conversation-speed-override.test.ts +1 -1
  24. package/src/__tests__/conversation-surfaces-activation-emit.test.ts +395 -0
  25. package/src/__tests__/conversation-surfaces-app-control.test.ts +44 -0
  26. package/src/__tests__/conversation-undo.test.ts +2 -2
  27. package/src/__tests__/conversation-workspace-cache-state.test.ts +1 -1
  28. package/src/__tests__/conversation-workspace-injection.test.ts +1 -1
  29. package/src/__tests__/conversation-workspace-tool-tracking.test.ts +1 -1
  30. package/src/__tests__/cu-unified-flow.test.ts +36 -0
  31. package/src/__tests__/history-repair-hook.test.ts +2 -0
  32. package/src/__tests__/memory-retrieval-hook.test.ts +1 -2
  33. package/src/__tests__/persist-unsendable-image-downscale.test.ts +145 -0
  34. package/src/__tests__/persist-unsendable-image.test.ts +97 -1
  35. package/src/__tests__/post-turn-tool-result-truncation.test.ts +69 -0
  36. package/src/__tests__/skill-feature-flags-integration.test.ts +5 -7
  37. package/src/__tests__/title-generate-hook.test.ts +2 -0
  38. package/src/__tests__/web-fetch.test.ts +45 -0
  39. package/src/acp/__tests__/helpers/acp-config-stub.ts +0 -2
  40. package/src/acp/resolve-agent.test.ts +0 -56
  41. package/src/acp/resolve-agent.ts +10 -38
  42. package/src/agent/loop.ts +13 -27
  43. package/src/cli/lib/__tests__/install-from-github.test.ts +232 -29
  44. package/src/cli/lib/__tests__/plugin-details.test.ts +28 -19
  45. package/src/cli/lib/__tests__/plugin-marketplace.test.ts +57 -7
  46. package/src/cli/lib/__tests__/search-plugins.test.ts +17 -10
  47. package/src/cli/lib/install-from-github.ts +258 -41
  48. package/src/cli/lib/plugin-details.ts +20 -13
  49. package/src/cli/lib/plugin-marketplace.ts +23 -5
  50. package/src/cli/lib/search-plugins.ts +14 -8
  51. package/src/config/acp-defaults.ts +3 -3
  52. package/src/config/acp-schema.ts +1 -7
  53. package/src/config/bundled-skills/acp/SKILL.md +4 -17
  54. package/src/config/bundled-skills/acp/TOOLS.json +2 -2
  55. package/src/config/feature-flag-registry.json +3 -18
  56. package/src/context/post-turn-tool-result-truncation.ts +39 -1
  57. package/src/daemon/conversation-agent-loop-handlers.ts +8 -1
  58. package/src/daemon/conversation-agent-loop.ts +78 -78
  59. package/src/daemon/conversation-error.ts +31 -4
  60. package/src/daemon/conversation-history.ts +1 -1
  61. package/src/daemon/conversation-media-retry.ts +19 -6
  62. package/src/daemon/conversation-messaging.ts +17 -0
  63. package/src/daemon/conversation-process.ts +14 -5
  64. package/src/daemon/conversation-queue-manager.ts +8 -0
  65. package/src/daemon/conversation-runtime-assembly.ts +37 -1
  66. package/src/daemon/conversation-surfaces.ts +141 -3
  67. package/src/daemon/conversation.ts +48 -13
  68. package/src/daemon/persist-unsendable-image.ts +62 -25
  69. package/src/daemon/process-message.ts +1 -1
  70. package/src/memory/__tests__/activation-session-store.test.ts +41 -0
  71. package/src/memory/__tests__/onboarding-events-store.test.ts +80 -0
  72. package/src/memory/activation-session-store.ts +43 -0
  73. package/src/memory/db-init.ts +4 -0
  74. package/src/memory/migrations/273-onboarding-events-funnel-columns.ts +46 -0
  75. package/src/memory/migrations/274-create-activation-sessions.ts +15 -0
  76. package/src/memory/migrations/index.ts +2 -0
  77. package/src/memory/onboarding-events-store.ts +66 -18
  78. package/src/memory/schema/infrastructure.ts +13 -0
  79. package/src/messaging/providers/telegram-bot/api.ts +14 -5
  80. package/src/notifications/adapters/telegram.ts +7 -1
  81. package/src/plugin-api/constants.ts +2 -2
  82. package/src/plugin-api/index.ts +2 -2
  83. package/src/plugin-api/types.ts +19 -5
  84. package/src/plugins/defaults/compaction/compact.ts +24 -15
  85. package/src/plugins/defaults/compaction/context-overflow-reducer.ts +4 -4
  86. package/src/plugins/defaults/compaction/manager-store.ts +1 -1
  87. package/src/{context → plugins/defaults/compaction}/window-manager.ts +12 -12
  88. package/src/plugins/defaults/memory-retrieval/hooks/user-prompt-submit-temp.ts +12 -18
  89. package/src/prompts/system-prompt.ts +61 -10
  90. package/src/prompts/templates/BOOTSTRAP-ACTIVATION-RAIL.md +37 -2
  91. package/src/providers/openai/__tests__/vision-not-supported.test.ts +75 -0
  92. package/src/providers/openai/chat-completions-provider.ts +25 -0
  93. package/src/runtime/routes/__tests__/stt-routes.test.ts +112 -0
  94. package/src/runtime/routes/acp-routes.test.ts +3 -20
  95. package/src/runtime/routes/conversation-routes.ts +2 -0
  96. package/src/runtime/routes/playground/__tests__/force-compact.test.ts +1 -1
  97. package/src/runtime/routes/stt-routes.ts +45 -12
  98. package/src/runtime/routes/workspace-routes.ts +50 -15
  99. package/src/telemetry/__tests__/activation-funnel.test.ts +95 -0
  100. package/src/telemetry/activation-funnel.ts +167 -0
  101. package/src/telemetry/types.ts +13 -0
  102. package/src/telemetry/usage-telemetry-reporter.test.ts +154 -0
  103. package/src/telemetry/usage-telemetry-reporter.ts +26 -1
  104. package/src/tools/acp/list-agents.test.ts +2 -18
  105. package/src/tools/acp/list-agents.ts +3 -15
  106. package/src/tools/acp/spawn.test.ts +0 -10
  107. package/src/tools/browser/browser-execution.ts +12 -2
  108. package/src/tools/network/web-fetch.ts +65 -24
  109. package/src/tools/ui-surface/definitions.ts +7 -0
  110. package/src/acp/feature-gate.test.ts +0 -48
  111. package/src/acp/feature-gate.ts +0 -34
@@ -0,0 +1,145 @@
1
+ /**
2
+ * Regression test for the durable-downscale path of the image-too-large
3
+ * recovery (JARVIS-1041 review follow-up).
4
+ *
5
+ * When an oversized stored image *can* be shrunk on this host (the common macOS
6
+ * path where `sips` is available), `persistUnsendableImageDowngrades` must write
7
+ * the downscaled bytes back to the DB — not leave the original in place. The
8
+ * latest tool-result media is intentionally kept in context, so leaving the
9
+ * full-size block would rehydrate and re-reject on every later turn instead of
10
+ * durably self-healing the conversation.
11
+ *
12
+ * `optimizeImageForTransport` needs `sips` and a decodable image to actually
13
+ * downscale, which is not portable to CI, so it is mocked here to simulate a
14
+ * successful shrink. The mock is process-global, so this case lives in its own
15
+ * file (the test runner isolates each file in its own process) to avoid
16
+ * disturbing the no-op-resize cases in persist-unsendable-image.test.ts.
17
+ */
18
+
19
+ import { beforeEach, describe, expect, mock, test } from "bun:test";
20
+
21
+ // ── Module mock (must precede the import of the module under test) ────
22
+ // Simulate a host where resizing succeeds: any oversized image is shrunk to a
23
+ // small, distinct JPEG payload. durableImageReplacement checks the provider
24
+ // caps before calling this, so in-limit images never reach the mock.
25
+ const SHRUNK_DATA = "c2hydW5r"; // base64 for "shrunk"
26
+ mock.module("../agent/image-optimize.js", () => ({
27
+ optimizeImageForTransport: () => ({
28
+ data: SHRUNK_DATA,
29
+ mediaType: "image/jpeg",
30
+ }),
31
+ }));
32
+
33
+ import { persistUnsendableImageDowngrades } from "../daemon/persist-unsendable-image.js";
34
+ import {
35
+ addMessage,
36
+ createConversation,
37
+ getMessages,
38
+ } from "../memory/conversation-crud.js";
39
+ import { getDb } from "../memory/db-connection.js";
40
+ import { initializeDb } from "../memory/db-init.js";
41
+ import type { ContentBlock } from "../providers/types.js";
42
+
43
+ initializeDb();
44
+
45
+ function resetTables(): void {
46
+ const db = getDb();
47
+ db.run("DELETE FROM message_attachments");
48
+ db.run("DELETE FROM attachments");
49
+ db.run("DELETE FROM memory_segments");
50
+ db.run("DELETE FROM memory_embeddings");
51
+ db.run("DELETE FROM messages");
52
+ db.run("DELETE FROM conversations");
53
+ }
54
+
55
+ /** Minimal PNG whose IHDR declares dimensions past the 8000px provider cap. */
56
+ function oversizedPngBase64(): string {
57
+ const width = 12000;
58
+ const height = 9000;
59
+ return Buffer.from(
60
+ Uint8Array.from([
61
+ 0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, // PNG signature
62
+ 0x00, 0x00, 0x00, 0x0d, // IHDR length (13)
63
+ 0x49, 0x48, 0x44, 0x52, // "IHDR"
64
+ (width >>> 24) & 0xff,
65
+ (width >>> 16) & 0xff,
66
+ (width >>> 8) & 0xff,
67
+ width & 0xff,
68
+ (height >>> 24) & 0xff,
69
+ (height >>> 16) & 0xff,
70
+ (height >>> 8) & 0xff,
71
+ height & 0xff,
72
+ 0x08, 0x06, 0x00, 0x00, 0x00,
73
+ ]),
74
+ ).toString("base64");
75
+ }
76
+
77
+ function toolResultWithImage(data: string): ContentBlock {
78
+ return {
79
+ type: "tool_result",
80
+ tool_use_id: "toolu_123",
81
+ content: "Screenshot captured",
82
+ contentBlocks: [
83
+ { type: "image", source: { type: "base64", media_type: "image/png", data } },
84
+ ],
85
+ };
86
+ }
87
+
88
+ function storedContent(conversationId: string): ContentBlock[][] {
89
+ return getMessages(conversationId).map(
90
+ (row) => JSON.parse(row.content) as ContentBlock[],
91
+ );
92
+ }
93
+
94
+ describe("persistUnsendableImageDowngrades (downscalable host)", () => {
95
+ beforeEach(() => {
96
+ resetTables();
97
+ });
98
+
99
+ /** JARVIS-1041: an oversized screenshot that CAN be shrunk must persist the
100
+ * downscaled bytes, not the note and not the original. */
101
+ test("persists the downscaled image for a shrinkable tool_result screenshot", async () => {
102
+ // GIVEN a tool_result holding an oversized but shrinkable screenshot
103
+ const conv = createConversation();
104
+ await addMessage(
105
+ conv.id,
106
+ "user",
107
+ JSON.stringify([toolResultWithImage(oversizedPngBase64())]),
108
+ { skipIndexing: true },
109
+ );
110
+
111
+ // WHEN the downgrade is persisted
112
+ const rewritten = persistUnsendableImageDowngrades(conv.id);
113
+
114
+ // THEN the nested block stays an image, rewritten to the downscaled payload
115
+ expect(rewritten).toBe(1);
116
+ const [content] = storedContent(conv.id);
117
+ const toolResult = content.find((b) => b.type === "tool_result") as {
118
+ contentBlocks?: ContentBlock[];
119
+ };
120
+ const nested = toolResult.contentBlocks?.[0];
121
+ expect(nested?.type).toBe("image");
122
+ expect((nested as Extract<ContentBlock, { type: "image" }>).source.data).toBe(
123
+ SHRUNK_DATA,
124
+ );
125
+ });
126
+
127
+ /** Re-running is a no-op: the downscaled payload is within limits. */
128
+ test("is idempotent after a downscale rewrite", async () => {
129
+ // GIVEN a conversation whose oversized screenshot was already downscaled
130
+ const conv = createConversation();
131
+ await addMessage(
132
+ conv.id,
133
+ "user",
134
+ JSON.stringify([toolResultWithImage(oversizedPngBase64())]),
135
+ { skipIndexing: true },
136
+ );
137
+ expect(persistUnsendableImageDowngrades(conv.id)).toBe(1);
138
+
139
+ // WHEN the downgrade runs again
140
+ const secondRun = persistUnsendableImageDowngrades(conv.id);
141
+
142
+ // THEN nothing further is rewritten
143
+ expect(secondRun).toBe(0);
144
+ });
145
+ });
@@ -14,7 +14,10 @@
14
14
 
15
15
  import { beforeEach, describe, expect, test } from "bun:test";
16
16
 
17
- import { persistUnsendableImageDowngrades } from "../daemon/persist-unsendable-image.js";
17
+ import {
18
+ oversizedImageReplacement,
19
+ persistUnsendableImageDowngrades,
20
+ } from "../daemon/persist-unsendable-image.js";
18
21
  import {
19
22
  addMessage,
20
23
  createConversation,
@@ -87,6 +90,20 @@ function imageBlock(data: string): ContentBlock {
87
90
  };
88
91
  }
89
92
 
93
+ /**
94
+ * A tool_result carrying a nested image in its contentBlocks, mirroring what a
95
+ * browser screenshot produces. This is the JARVIS-1041 shape: the oversized
96
+ * image lives at tool_result.contentBlocks, never as a top-level block.
97
+ */
98
+ function toolResultWithImage(data: string): ContentBlock {
99
+ return {
100
+ type: "tool_result",
101
+ tool_use_id: "toolu_123",
102
+ content: "Screenshot captured",
103
+ contentBlocks: [imageBlock(data)],
104
+ };
105
+ }
106
+
90
107
  function storedContent(conversationId: string): ContentBlock[][] {
91
108
  return getMessages(conversationId).map(
92
109
  (row) => JSON.parse(row.content) as ContentBlock[],
@@ -194,6 +211,61 @@ describe("persistUnsendableImageDowngrades", () => {
194
211
  expect(content.some((b) => b.type === "image")).toBe(false);
195
212
  });
196
213
 
214
+ /** JARVIS-1041: the oversized image is nested inside a tool_result (e.g. a
215
+ * browser screenshot), not a top-level block. The downgrade must descend
216
+ * into tool_result.contentBlocks and swap the nested image for a note, while
217
+ * keeping the tool_result itself intact so tool_use/tool_result pairing
218
+ * survives. */
219
+ test("downgrades an oversized image nested in tool_result.contentBlocks", async () => {
220
+ // GIVEN an assistant turn whose tool_result holds an oversized screenshot
221
+ const conv = createConversation();
222
+ await addMessage(
223
+ conv.id,
224
+ "user",
225
+ JSON.stringify([toolResultWithImage(makePngBase64(12000, 9000))]),
226
+ { skipIndexing: true },
227
+ );
228
+
229
+ // WHEN the downgrade is persisted
230
+ const rewritten = persistUnsendableImageDowngrades(conv.id);
231
+
232
+ // THEN the message is rewritten with the nested image swapped for a note
233
+ expect(rewritten).toBe(1);
234
+ const [content] = storedContent(conv.id);
235
+ const toolResult = content.find((b) => b.type === "tool_result");
236
+ expect(toolResult).toBeDefined();
237
+ // AND the tool_result is preserved (pairing intact) with no image left
238
+ const nested = (toolResult as { contentBlocks?: ContentBlock[] })
239
+ .contentBlocks;
240
+ expect(nested?.some((b) => b.type === "image")).toBe(false);
241
+ expect(nested?.some((b) => b.type === "text")).toBe(true);
242
+ });
243
+
244
+ /** A sendable nested screenshot is never disturbed. */
245
+ test("leaves a normally-sized tool_result image untouched", async () => {
246
+ // GIVEN a tool_result with an image well within provider limits
247
+ const conv = createConversation();
248
+ await addMessage(
249
+ conv.id,
250
+ "user",
251
+ JSON.stringify([toolResultWithImage(makePngBase64(1024, 768))]),
252
+ { skipIndexing: true },
253
+ );
254
+
255
+ // WHEN the downgrade is persisted
256
+ const rewritten = persistUnsendableImageDowngrades(conv.id);
257
+
258
+ // THEN nothing is rewritten and the nested image remains
259
+ expect(rewritten).toBe(0);
260
+ const [content] = storedContent(conv.id);
261
+ const toolResult = content.find((b) => b.type === "tool_result") as {
262
+ contentBlocks?: ContentBlock[];
263
+ };
264
+ expect(toolResult.contentBlocks?.some((b) => b.type === "image")).toBe(
265
+ true,
266
+ );
267
+ });
268
+
197
269
  /** Re-running after a rewrite is a safe no-op (no image blocks remain). */
198
270
  test("is idempotent — a second run rewrites nothing", async () => {
199
271
  // GIVEN a conversation whose oversized image has already been downgraded
@@ -213,3 +285,27 @@ describe("persistUnsendableImageDowngrades", () => {
213
285
  expect(secondRun).toBe(0);
214
286
  });
215
287
  });
288
+
289
+ describe("oversizedImageReplacement", () => {
290
+ /** A still-sendable image must be left alone — never replaced with a note.
291
+ * This is the gate that keeps the in-memory recovery from discarding valid
292
+ * screenshots when only one image in the turn was actually oversized. */
293
+ test("returns null for an image within the provider caps", () => {
294
+ const sendable = imageBlock(makePngBase64(1024, 768)) as Extract<
295
+ ContentBlock,
296
+ { type: "image" }
297
+ >;
298
+ expect(oversizedImageReplacement(sendable)).toBeNull();
299
+ });
300
+
301
+ /** An image past the provider caps that cannot be shrunk on this host (fake
302
+ * PNG that sips cannot decode) collapses to the unsendable note. */
303
+ test("returns the unsendable note when an oversized image cannot be shrunk", () => {
304
+ const oversized = imageBlock(makePngBase64(12000, 9000)) as Extract<
305
+ ContentBlock,
306
+ { type: "image" }
307
+ >;
308
+ const replacement = oversizedImageReplacement(oversized);
309
+ expect(replacement?.type).toBe("text");
310
+ });
311
+ });
@@ -12,6 +12,7 @@ import {
12
12
  TARGET_CHARS,
13
13
  THRESHOLD_CHARS,
14
14
  TOOL_RESULT_DIR,
15
+ TRUNCATION_EXEMPT_TOOLS,
15
16
  TRUNCATION_MARKER,
16
17
  } from "../context/post-turn-tool-result-truncation.js";
17
18
  import type { ContentBlock, Message } from "../providers/types.js";
@@ -151,6 +152,74 @@ describe("postTurnTruncateToolResults", () => {
151
152
  expect(stub).toContain(filePath);
152
153
  });
153
154
 
155
+ test("skill_load result above threshold is NOT truncated (durable instructions exempt)", () => {
156
+ // Regression for JARVIS-1000: a hosted assistant lost its app-builder skill
157
+ // workflow when the large skill_load result was middle-truncated between the
158
+ // turn that loaded the skill and the turn that used it, then fell back to a
159
+ // local-dev (vite/localhost) build path.
160
+ const toolUseId = "tool_skill_load";
161
+ const skillBody = "S".repeat(THRESHOLD_CHARS + 5_000);
162
+ const messages: Message[] = [
163
+ {
164
+ role: "assistant",
165
+ content: [
166
+ {
167
+ type: "tool_use" as const,
168
+ id: toolUseId,
169
+ name: "skill_load",
170
+ input: { skill: "app-builder" },
171
+ },
172
+ ],
173
+ },
174
+ { role: "user", content: [makeToolResult(skillBody, toolUseId)] },
175
+ ];
176
+
177
+ const { messages: result, truncatedCount } =
178
+ postTurnTruncateToolResults(messages, { conversationDir: convDir });
179
+
180
+ expect(truncatedCount).toBe(0);
181
+ expect(result).toBe(messages); // same reference — no copy
182
+ expect(existsSync(join(convDir, TOOL_RESULT_DIR))).toBe(false);
183
+
184
+ const block = result[1].content[0] as {
185
+ type: "tool_result";
186
+ content: string;
187
+ };
188
+ expect(block.content).toBe(skillBody);
189
+ expect(TRUNCATION_EXEMPT_TOOLS.has("skill_load")).toBe(true);
190
+ });
191
+
192
+ test("non-exempt tool result above threshold is still truncated when paired with a tool_use", () => {
193
+ // Control for the exemption: same shape as the skill_load case, but a tool
194
+ // that is NOT exempt must still be truncated.
195
+ const toolUseId = "tool_bash_1";
196
+ const longContent = "B".repeat(THRESHOLD_CHARS + 5_000);
197
+ const messages: Message[] = [
198
+ {
199
+ role: "assistant",
200
+ content: [
201
+ {
202
+ type: "tool_use" as const,
203
+ id: toolUseId,
204
+ name: "bash",
205
+ input: { command: "cat big.log" },
206
+ },
207
+ ],
208
+ },
209
+ { role: "user", content: [makeToolResult(longContent, toolUseId)] },
210
+ ];
211
+
212
+ const { messages: result, truncatedCount } =
213
+ postTurnTruncateToolResults(messages, { conversationDir: convDir });
214
+
215
+ expect(truncatedCount).toBe(1);
216
+ const block = result[1].content[0] as {
217
+ type: "tool_result";
218
+ content: string;
219
+ };
220
+ expect(block.content).toContain(TRUNCATION_MARKER);
221
+ });
222
+
154
223
  test("file path is deterministic for the same toolUseId", () => {
155
224
  const id = "tool_use_deterministic";
156
225
  const path1 = getToolResultFilePath("/some/dir", id);
@@ -233,10 +233,9 @@ describe("frontmatter feature-flag integration", () => {
233
233
  // ---------------------------------------------------------------------------
234
234
 
235
235
  describe("bundled acp skill discoverability", () => {
236
- test("acp skill resolves with the flag off and config.acp disabled (no frontmatter flag gate)", () => {
237
- // The ACP skill carries its own first-time-setup instructions, so it must
238
- // stay visible even when the acp flag and config.acp.enabled are both off.
239
- // Runtime enforcement happens in the ACP tools via isAcpEnabled instead.
236
+ test("acp skill resolves with no frontmatter flag gate", () => {
237
+ // The ACP skill carries its own first-time-setup instructions and is
238
+ // always discoverable: it has no frontmatter feature-flag gate.
240
239
  const skillMdPath = fileURLToPath(
241
240
  new URL("../config/bundled-skills/acp/SKILL.md", import.meta.url),
242
241
  );
@@ -247,10 +246,9 @@ describe("bundled acp skill discoverability", () => {
247
246
  expect(skill!.featureFlag).toBeUndefined();
248
247
  expect(skillFlagKey(skill!)).toBeUndefined();
249
248
 
250
- // acp flag at its registry default (off) and config.acp disabled.
251
249
  const config = makeConfig({
252
- acp: { enabled: false, maxConcurrentSessions: 4, agents: {} },
253
- } as Partial<AssistantConfig>);
250
+ acp: { maxConcurrentSessions: 4, agents: {} },
251
+ });
254
252
 
255
253
  const resolved = resolveSkillStates([skill!], config);
256
254
  expect(resolved.length).toBe(1);
@@ -76,6 +76,8 @@ function makeCtx(
76
76
  ];
77
77
  return {
78
78
  conversationId: "conv-1",
79
+ userMessageId: "msg-1",
80
+ requestId: "req-1",
79
81
  prompt: "first message",
80
82
  originalMessages: messages,
81
83
  latestMessages: messages,
@@ -5,6 +5,7 @@ import { afterEach, beforeEach, describe, expect, test } from "bun:test";
5
5
  import {
6
6
  buildFetchResponseFromNodeResponse,
7
7
  executeWebFetch,
8
+ getUpstreamStatus,
8
9
  } from "../tools/network/web-fetch.js";
9
10
 
10
11
  describe("web_fetch tool", () => {
@@ -62,6 +63,50 @@ describe("web_fetch tool", () => {
62
63
  expect(await response.text()).toBe("");
63
64
  });
64
65
 
66
+ test("buildFetchResponseFromNodeResponse does not throw on non-standard status codes", async () => {
67
+ const stream = new PassThrough() as PassThrough & {
68
+ statusCode?: number;
69
+ statusMessage?: string;
70
+ headers: IncomingHttpHeaders;
71
+ };
72
+ stream.statusCode = 999;
73
+ stream.statusMessage = "Request Denied";
74
+ stream.headers = { "content-type": "text/html; charset=utf-8" };
75
+ stream.end("blocked by anti-bot gateway");
76
+
77
+ const response = buildFetchResponseFromNodeResponse(stream);
78
+ // The Response object clamps to a constructable code, but the real upstream
79
+ // status is preserved for reporting.
80
+ expect(response.status).toBe(502);
81
+ expect(getUpstreamStatus(response)).toBe(999);
82
+ expect(response.ok).toBe(false);
83
+ expect(await response.text()).toBe("blocked by anti-bot gateway");
84
+ });
85
+
86
+ test("surfaces a non-standard status as a tool error instead of crashing", async () => {
87
+ const result = await executeWithMockFetch(
88
+ { url: "https://www.linkedin.com/jobs/view/123" },
89
+ {
90
+ requestExecutor: async () => {
91
+ const stream = new PassThrough() as PassThrough & {
92
+ statusCode?: number;
93
+ statusMessage?: string;
94
+ headers: IncomingHttpHeaders;
95
+ };
96
+ stream.statusCode = 999;
97
+ stream.statusMessage = "Request Denied";
98
+ stream.headers = { "content-type": "text/html; charset=utf-8" };
99
+ stream.end("<html><body>blocked</body></html>");
100
+ return buildFetchResponseFromNodeResponse(stream);
101
+ },
102
+ },
103
+ );
104
+
105
+ expect(result.isError).toBe(true);
106
+ expect(result.content).toContain("HTTP 999");
107
+ expect(result.activityMetadata?.webFetch?.status).toBe(999);
108
+ });
109
+
65
110
  test("rejects missing url", async () => {
66
111
  const result = await executeWithMockFetch({});
67
112
  expect(result.isError).toBe(true);
@@ -20,13 +20,11 @@ import { mock } from "bun:test";
20
20
  import type { AcpAgentConfig } from "../../../config/acp-schema.js";
21
21
 
22
22
  export interface MockAcpConfig {
23
- enabled: boolean;
24
23
  maxConcurrentSessions: number;
25
24
  agents: Record<string, AcpAgentConfig>;
26
25
  }
27
26
 
28
27
  const DEFAULT_CONFIG: MockAcpConfig = {
29
- enabled: true,
30
28
  maxConcurrentSessions: 4,
31
29
  agents: {},
32
30
  };
@@ -1,25 +1,19 @@
1
1
  import { afterAll, beforeEach, describe, expect, test } from "bun:test";
2
2
 
3
- import { setOverridesForTesting } from "../__tests__/feature-flag-test-helpers.js";
4
3
  import { installAcpConfigStub } from "./__tests__/helpers/acp-config-stub.js";
5
4
  import { installWhichStub } from "./__tests__/helpers/which-stub.js";
6
- import { ACP_FLAG_KEY } from "./feature-gate.js";
7
5
 
8
6
  const config = await installAcpConfigStub();
9
7
  const which = installWhichStub();
10
8
 
11
9
  afterAll(() => {
12
10
  which.restore();
13
- setOverridesForTesting({});
14
11
  });
15
12
 
16
13
  const { resolveAcpAgent, listAcpAgents } = await import("./resolve-agent.js");
17
14
 
18
15
  beforeEach(() => {
19
16
  config.setConfig({});
20
- // Default: no flag overrides, so the `acp` flag falls back to its registry
21
- // default (false) and enablement comes from the config stub alone.
22
- setOverridesForTesting({});
23
17
  // Default: every command on PATH so binary preflight passes unless a test
24
18
  // explicitly says otherwise.
25
19
  which.setWhich((cmd) => `/usr/local/bin/${cmd}`);
@@ -30,32 +24,6 @@ beforeEach(() => {
30
24
  // ---------------------------------------------------------------------------
31
25
 
32
26
  describe("resolveAcpAgent", () => {
33
- test("returns acp_disabled when both the feature flag and config.acp.enabled are off", () => {
34
- config.setConfig({ enabled: false });
35
-
36
- const result = resolveAcpAgent("claude");
37
-
38
- expect(result.ok).toBe(false);
39
- if (result.ok) return;
40
- expect(result.reason).toBe("acp_disabled");
41
- if (result.reason !== "acp_disabled") return;
42
- expect(result.hint).toContain("ACP Coding Agents");
43
- expect(result.hint).toContain("feature flag");
44
- expect(result.hint).toContain("acp.enabled");
45
- expect(result.hint).toContain("config.json");
46
- });
47
-
48
- test("resolution proceeds when the acp feature flag is on and config.acp.enabled is false", () => {
49
- config.setConfig({ enabled: false });
50
- setOverridesForTesting({ [ACP_FLAG_KEY]: true });
51
-
52
- const result = resolveAcpAgent("claude");
53
-
54
- expect(result.ok).toBe(true);
55
- if (!result.ok) return;
56
- expect(result.agent.command).toBe("claude-agent-acp");
57
- });
58
-
59
27
  test("user config wins over default profile", () => {
60
28
  config.setConfig({
61
29
  agents: {
@@ -359,35 +327,11 @@ describe("resolveAcpAgent - missing binary", () => {
359
327
  // ---------------------------------------------------------------------------
360
328
 
361
329
  describe("listAcpAgents", () => {
362
- test("returns enabled: false with empty agents when both the flag and config are off", () => {
363
- config.setConfig({ enabled: false });
364
-
365
- const result = listAcpAgents();
366
-
367
- expect(result.enabled).toBe(false);
368
- expect(result.agents).toEqual([]);
369
- });
370
-
371
- test("returns the catalog when the acp feature flag is on and config.acp.enabled is false", () => {
372
- config.setConfig({ enabled: false });
373
- setOverridesForTesting({ [ACP_FLAG_KEY]: true });
374
-
375
- const result = listAcpAgents();
376
-
377
- expect(result.enabled).toBe(true);
378
- expect(result.agents.map((a) => a.id)).toEqual([
379
- "claude",
380
- "codex",
381
- "gemini",
382
- ]);
383
- });
384
-
385
330
  test("includes all bundled defaults when user config is empty", () => {
386
331
  config.setConfig({ agents: {} });
387
332
 
388
333
  const result = listAcpAgents();
389
334
 
390
- expect(result.enabled).toBe(true);
391
335
  const ids = result.agents.map((a) => a.id);
392
336
  expect(ids).toEqual(["claude", "codex", "gemini"]);
393
337
  for (const entry of result.agents) {
@@ -3,15 +3,13 @@
3
3
  *
4
4
  * `resolveAcpAgent(id)` merges user-provided `config.acp.agents[id]` (wins on
5
5
  * overlap) with the bundled `DEFAULT_ACP_AGENT_PROFILES` so common agents like
6
- * `claude` and `codex` Just Work whenever ACP is enabled (the `acp` feature
7
- * flag or `acp.enabled: true`; see `feature-gate.ts`), with no per-user
8
- * config required. Natural names ("claude code", "Gemini CLI") resolve via
9
- * `AGENT_ID_ALIASES` when the raw id misses both maps. The result is a
10
- * discriminated union covering every reason
11
- * a spawn might fail before we even start the agent process: ACP disabled,
12
- * unknown agent id, or binary missing from PATH. Callers (acp_spawn,
13
- * acp_list_agents, and the `/v1/acp/spawn` HTTP route) get a single source
14
- * of truth and matching actionable hints.
6
+ * `claude` and `codex` Just Work with no per-user config required. Natural
7
+ * names ("claude code", "Gemini CLI") resolve via `AGENT_ID_ALIASES` when the
8
+ * raw id misses both maps. The result is a discriminated union covering every
9
+ * reason a spawn might fail before we even start the agent process: unknown
10
+ * agent id, or binary missing from PATH. Callers (acp_spawn, acp_list_agents,
11
+ * and the `/v1/acp/spawn` HTTP route) get a single source of truth and
12
+ * matching actionable hints.
15
13
  *
16
14
  * The resolver NEVER fetches or runs packages in the (untrusted) task cwd.
17
15
  * When the adapter binary is missing, resolution simply fails with
@@ -33,7 +31,6 @@ import {
33
31
  } from "../config/acp-defaults.js";
34
32
  import type { AcpAgentConfig } from "../config/acp-schema.js";
35
33
  import { getConfig } from "../config/loader.js";
36
- import { isAcpEnabled } from "./feature-gate.js";
37
34
 
38
35
  /**
39
36
  * Whether this agent's entry came from user config (wins over default) or
@@ -47,7 +44,6 @@ export type ResolveAcpAgentResult =
47
44
  | ResolveAcpAgentFailure;
48
45
 
49
46
  export type ResolveAcpAgentFailure =
50
- | { ok: false; reason: "acp_disabled"; hint: string }
51
47
  | { ok: false; reason: "unknown_agent"; available: string[] }
52
48
  | {
53
49
  ok: false;
@@ -68,8 +64,6 @@ export function formatResolveFailure(
68
64
  failure: ResolveAcpAgentFailure,
69
65
  ): string {
70
66
  switch (failure.reason) {
71
- case "acp_disabled":
72
- return failure.hint;
73
67
  case "unknown_agent":
74
68
  return `Unknown agent "${agentId}". Available: ${failure.available.join(", ")}.`;
75
69
  case "binary_not_found":
@@ -93,14 +87,6 @@ interface AcpAgentEntry {
93
87
  setupHint?: string;
94
88
  }
95
89
 
96
- /**
97
- * Single-source-of-truth hint for "ACP is disabled". Exported so any caller
98
- * that surfaces a disabled-state message (resolver, list-agents tool) reads
99
- * the same string instead of duplicating near-identical copy.
100
- */
101
- export const ACP_DISABLED_HINT =
102
- "Enable the \"ACP Coding Agents\" feature flag in the client's feature flags UI (or set 'acp.enabled': true in ~/.vellum/workspace/config.json).";
103
-
104
90
  function installHintFor(command: string): string {
105
91
  const pkg = DEFAULT_AGENT_NPM_PACKAGES[command];
106
92
  return pkg
@@ -209,9 +195,8 @@ function mergedAgentIds(userAgents: Record<string, AcpAgentConfig>): string[] {
209
195
  * Resolve an ACP agent id to its config + binary preflight result.
210
196
  *
211
197
  * Order of checks:
212
- * 1. ACP must be enabled (feature flag or config; see `isAcpEnabled`).
213
- * 2. The id must resolve to an agent (user config wins; falls back to defaults).
214
- * 3. The agent must be runnable: its `command` on PATH (see
198
+ * 1. The id must resolve to an agent (user config wins; falls back to defaults).
199
+ * 2. The agent must be runnable: its `command` on PATH (see
215
200
  * `resolveRunnableAgent`).
216
201
  *
217
202
  * Each failure mode carries an actionable hint so callers can surface a
@@ -219,10 +204,6 @@ function mergedAgentIds(userAgents: Record<string, AcpAgentConfig>): string[] {
219
204
  */
220
205
  export function resolveAcpAgent(id: string): ResolveAcpAgentResult {
221
206
  const config = getConfig();
222
- if (!isAcpEnabled(config)) {
223
- return { ok: false, reason: "acp_disabled", hint: ACP_DISABLED_HINT };
224
- }
225
-
226
207
  const userAgents = config.acp.agents;
227
208
  const found = lookupAgent(userAgents, id);
228
209
  if (!found) {
@@ -252,20 +233,11 @@ export function resolveAcpAgent(id: string): ResolveAcpAgentResult {
252
233
  * plus any user-only entries — with per-entry availability info. Used by the
253
234
  * `acp_list_agents` tool to render setup steps when an agent's binary isn't
254
235
  * installed yet.
255
- *
256
- * `enabled: false` short-circuits and returns an empty catalog so the tool
257
- * can render a single "ACP is disabled" hint instead of advertising agents
258
- * the user can't actually run.
259
236
  */
260
237
  export function listAcpAgents(): {
261
- enabled: boolean;
262
238
  agents: AcpAgentEntry[];
263
239
  } {
264
240
  const config = getConfig();
265
- if (!isAcpEnabled(config)) {
266
- return { enabled: false, agents: [] };
267
- }
268
-
269
241
  const userAgents = config.acp.agents;
270
242
  const agents: AcpAgentEntry[] = mergedAgentIds(userAgents).map((id) => {
271
243
  // Non-null: ids come from `mergedAgentIds` so the lookup always resolves.
@@ -288,5 +260,5 @@ export function listAcpAgents(): {
288
260
  return entry;
289
261
  });
290
262
 
291
- return { enabled: true, agents };
263
+ return { agents };
292
264
  }