@vellumai/assistant 0.8.9-staging.2 → 0.8.9-staging.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/activation-funnel-telemetry.md +310 -0
- package/package.json +1 -1
- package/src/__tests__/activation-early-marking.test.ts +120 -0
- package/src/__tests__/agent-loop-output-hooks.test.ts +13 -13
- package/src/__tests__/approval-cascade.test.ts +1 -1
- package/src/__tests__/compaction-direct.test.ts +32 -18
- package/src/__tests__/compaction-events.test.ts +2 -2
- package/src/__tests__/compaction.benchmark.test.ts +1 -1
- package/src/__tests__/context-overflow-reducer.test.ts +5 -5
- package/src/__tests__/context-window-manager-compact-retry.test.ts +1 -1
- package/src/__tests__/conversation-abort-tool-results.test.ts +1 -1
- package/src/__tests__/conversation-confirmation-signals.test.ts +1 -1
- package/src/__tests__/conversation-error.test.ts +15 -1
- package/src/__tests__/conversation-history-web-search.test.ts +5 -0
- package/src/__tests__/conversation-media-retry.test.ts +1 -1
- package/src/__tests__/conversation-process-app-control-preactivation.test.ts +40 -0
- package/src/__tests__/conversation-process-callsite.test.ts +1 -1
- package/src/__tests__/conversation-provider-retry-repair.test.ts +1 -1
- package/src/__tests__/conversation-queue.test.ts +1 -1
- package/src/__tests__/conversation-runtime-assembly.test.ts +71 -0
- package/src/__tests__/conversation-slash-queue.test.ts +1 -1
- package/src/__tests__/conversation-slash-unknown.test.ts +1 -1
- package/src/__tests__/conversation-speed-override.test.ts +1 -1
- package/src/__tests__/conversation-surfaces-activation-emit.test.ts +395 -0
- package/src/__tests__/conversation-surfaces-app-control.test.ts +44 -0
- package/src/__tests__/conversation-undo.test.ts +2 -2
- package/src/__tests__/conversation-workspace-cache-state.test.ts +1 -1
- package/src/__tests__/conversation-workspace-injection.test.ts +1 -1
- package/src/__tests__/conversation-workspace-tool-tracking.test.ts +1 -1
- package/src/__tests__/cu-unified-flow.test.ts +36 -0
- package/src/__tests__/history-repair-hook.test.ts +2 -0
- package/src/__tests__/memory-retrieval-hook.test.ts +1 -2
- package/src/__tests__/persist-unsendable-image-downscale.test.ts +145 -0
- package/src/__tests__/persist-unsendable-image.test.ts +97 -1
- package/src/__tests__/post-turn-tool-result-truncation.test.ts +69 -0
- package/src/__tests__/skill-feature-flags-integration.test.ts +5 -7
- package/src/__tests__/title-generate-hook.test.ts +2 -0
- package/src/__tests__/web-fetch.test.ts +45 -0
- package/src/acp/__tests__/helpers/acp-config-stub.ts +0 -2
- package/src/acp/resolve-agent.test.ts +0 -56
- package/src/acp/resolve-agent.ts +10 -38
- package/src/agent/loop.ts +13 -27
- package/src/cli/lib/__tests__/install-from-github.test.ts +232 -29
- package/src/cli/lib/__tests__/plugin-details.test.ts +28 -19
- package/src/cli/lib/__tests__/plugin-marketplace.test.ts +57 -7
- package/src/cli/lib/__tests__/search-plugins.test.ts +17 -10
- package/src/cli/lib/install-from-github.ts +258 -41
- package/src/cli/lib/plugin-details.ts +20 -13
- package/src/cli/lib/plugin-marketplace.ts +23 -5
- package/src/cli/lib/search-plugins.ts +14 -8
- package/src/config/acp-defaults.ts +3 -3
- package/src/config/acp-schema.ts +1 -7
- package/src/config/bundled-skills/acp/SKILL.md +4 -17
- package/src/config/bundled-skills/acp/TOOLS.json +2 -2
- package/src/config/feature-flag-registry.json +3 -18
- package/src/context/post-turn-tool-result-truncation.ts +39 -1
- package/src/daemon/conversation-agent-loop-handlers.ts +8 -1
- package/src/daemon/conversation-agent-loop.ts +78 -78
- package/src/daemon/conversation-error.ts +31 -4
- package/src/daemon/conversation-history.ts +1 -1
- package/src/daemon/conversation-media-retry.ts +19 -6
- package/src/daemon/conversation-messaging.ts +17 -0
- package/src/daemon/conversation-process.ts +14 -5
- package/src/daemon/conversation-queue-manager.ts +8 -0
- package/src/daemon/conversation-runtime-assembly.ts +37 -1
- package/src/daemon/conversation-surfaces.ts +141 -3
- package/src/daemon/conversation.ts +48 -13
- package/src/daemon/persist-unsendable-image.ts +62 -25
- package/src/daemon/process-message.ts +1 -1
- package/src/memory/__tests__/activation-session-store.test.ts +41 -0
- package/src/memory/__tests__/onboarding-events-store.test.ts +80 -0
- package/src/memory/activation-session-store.ts +43 -0
- package/src/memory/db-init.ts +4 -0
- package/src/memory/migrations/273-onboarding-events-funnel-columns.ts +46 -0
- package/src/memory/migrations/274-create-activation-sessions.ts +15 -0
- package/src/memory/migrations/index.ts +2 -0
- package/src/memory/onboarding-events-store.ts +66 -18
- package/src/memory/schema/infrastructure.ts +13 -0
- package/src/messaging/providers/telegram-bot/api.ts +14 -5
- package/src/notifications/adapters/telegram.ts +7 -1
- package/src/plugin-api/constants.ts +2 -2
- package/src/plugin-api/index.ts +2 -2
- package/src/plugin-api/types.ts +19 -5
- package/src/plugins/defaults/compaction/compact.ts +24 -15
- package/src/plugins/defaults/compaction/context-overflow-reducer.ts +4 -4
- package/src/plugins/defaults/compaction/manager-store.ts +1 -1
- package/src/{context → plugins/defaults/compaction}/window-manager.ts +12 -12
- package/src/plugins/defaults/memory-retrieval/hooks/user-prompt-submit-temp.ts +12 -18
- package/src/prompts/system-prompt.ts +61 -10
- package/src/prompts/templates/BOOTSTRAP-ACTIVATION-RAIL.md +37 -2
- package/src/providers/openai/__tests__/vision-not-supported.test.ts +75 -0
- package/src/providers/openai/chat-completions-provider.ts +25 -0
- package/src/runtime/routes/__tests__/stt-routes.test.ts +112 -0
- package/src/runtime/routes/acp-routes.test.ts +3 -20
- package/src/runtime/routes/conversation-routes.ts +2 -0
- package/src/runtime/routes/playground/__tests__/force-compact.test.ts +1 -1
- package/src/runtime/routes/stt-routes.ts +45 -12
- package/src/runtime/routes/workspace-routes.ts +50 -15
- package/src/telemetry/__tests__/activation-funnel.test.ts +95 -0
- package/src/telemetry/activation-funnel.ts +167 -0
- package/src/telemetry/types.ts +13 -0
- package/src/telemetry/usage-telemetry-reporter.test.ts +154 -0
- package/src/telemetry/usage-telemetry-reporter.ts +26 -1
- package/src/tools/acp/list-agents.test.ts +2 -18
- package/src/tools/acp/list-agents.ts +3 -15
- package/src/tools/acp/spawn.test.ts +0 -10
- package/src/tools/browser/browser-execution.ts +12 -2
- package/src/tools/network/web-fetch.ts +65 -24
- package/src/tools/ui-surface/definitions.ts +7 -0
- package/src/acp/feature-gate.test.ts +0 -48
- package/src/acp/feature-gate.ts +0 -34
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Regression test for the durable-downscale path of the image-too-large
|
|
3
|
+
* recovery (JARVIS-1041 review follow-up).
|
|
4
|
+
*
|
|
5
|
+
* When an oversized stored image *can* be shrunk on this host (the common macOS
|
|
6
|
+
* path where `sips` is available), `persistUnsendableImageDowngrades` must write
|
|
7
|
+
* the downscaled bytes back to the DB — not leave the original in place. The
|
|
8
|
+
* latest tool-result media is intentionally kept in context, so leaving the
|
|
9
|
+
* full-size block would rehydrate and re-reject on every later turn instead of
|
|
10
|
+
* durably self-healing the conversation.
|
|
11
|
+
*
|
|
12
|
+
* `optimizeImageForTransport` needs `sips` and a decodable image to actually
|
|
13
|
+
* downscale, which is not portable to CI, so it is mocked here to simulate a
|
|
14
|
+
* successful shrink. The mock is process-global, so this case lives in its own
|
|
15
|
+
* file (the test runner isolates each file in its own process) to avoid
|
|
16
|
+
* disturbing the no-op-resize cases in persist-unsendable-image.test.ts.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
import { beforeEach, describe, expect, mock, test } from "bun:test";
|
|
20
|
+
|
|
21
|
+
// ── Module mock (must precede the import of the module under test) ────
|
|
22
|
+
// Simulate a host where resizing succeeds: any oversized image is shrunk to a
|
|
23
|
+
// small, distinct JPEG payload. durableImageReplacement checks the provider
|
|
24
|
+
// caps before calling this, so in-limit images never reach the mock.
|
|
25
|
+
const SHRUNK_DATA = "c2hydW5r"; // base64 for "shrunk"
|
|
26
|
+
mock.module("../agent/image-optimize.js", () => ({
|
|
27
|
+
optimizeImageForTransport: () => ({
|
|
28
|
+
data: SHRUNK_DATA,
|
|
29
|
+
mediaType: "image/jpeg",
|
|
30
|
+
}),
|
|
31
|
+
}));
|
|
32
|
+
|
|
33
|
+
import { persistUnsendableImageDowngrades } from "../daemon/persist-unsendable-image.js";
|
|
34
|
+
import {
|
|
35
|
+
addMessage,
|
|
36
|
+
createConversation,
|
|
37
|
+
getMessages,
|
|
38
|
+
} from "../memory/conversation-crud.js";
|
|
39
|
+
import { getDb } from "../memory/db-connection.js";
|
|
40
|
+
import { initializeDb } from "../memory/db-init.js";
|
|
41
|
+
import type { ContentBlock } from "../providers/types.js";
|
|
42
|
+
|
|
43
|
+
initializeDb();
|
|
44
|
+
|
|
45
|
+
function resetTables(): void {
|
|
46
|
+
const db = getDb();
|
|
47
|
+
db.run("DELETE FROM message_attachments");
|
|
48
|
+
db.run("DELETE FROM attachments");
|
|
49
|
+
db.run("DELETE FROM memory_segments");
|
|
50
|
+
db.run("DELETE FROM memory_embeddings");
|
|
51
|
+
db.run("DELETE FROM messages");
|
|
52
|
+
db.run("DELETE FROM conversations");
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** Minimal PNG whose IHDR declares dimensions past the 8000px provider cap. */
|
|
56
|
+
function oversizedPngBase64(): string {
|
|
57
|
+
const width = 12000;
|
|
58
|
+
const height = 9000;
|
|
59
|
+
return Buffer.from(
|
|
60
|
+
Uint8Array.from([
|
|
61
|
+
0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a, // PNG signature
|
|
62
|
+
0x00, 0x00, 0x00, 0x0d, // IHDR length (13)
|
|
63
|
+
0x49, 0x48, 0x44, 0x52, // "IHDR"
|
|
64
|
+
(width >>> 24) & 0xff,
|
|
65
|
+
(width >>> 16) & 0xff,
|
|
66
|
+
(width >>> 8) & 0xff,
|
|
67
|
+
width & 0xff,
|
|
68
|
+
(height >>> 24) & 0xff,
|
|
69
|
+
(height >>> 16) & 0xff,
|
|
70
|
+
(height >>> 8) & 0xff,
|
|
71
|
+
height & 0xff,
|
|
72
|
+
0x08, 0x06, 0x00, 0x00, 0x00,
|
|
73
|
+
]),
|
|
74
|
+
).toString("base64");
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function toolResultWithImage(data: string): ContentBlock {
|
|
78
|
+
return {
|
|
79
|
+
type: "tool_result",
|
|
80
|
+
tool_use_id: "toolu_123",
|
|
81
|
+
content: "Screenshot captured",
|
|
82
|
+
contentBlocks: [
|
|
83
|
+
{ type: "image", source: { type: "base64", media_type: "image/png", data } },
|
|
84
|
+
],
|
|
85
|
+
};
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function storedContent(conversationId: string): ContentBlock[][] {
|
|
89
|
+
return getMessages(conversationId).map(
|
|
90
|
+
(row) => JSON.parse(row.content) as ContentBlock[],
|
|
91
|
+
);
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
describe("persistUnsendableImageDowngrades (downscalable host)", () => {
|
|
95
|
+
beforeEach(() => {
|
|
96
|
+
resetTables();
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
/** JARVIS-1041: an oversized screenshot that CAN be shrunk must persist the
|
|
100
|
+
* downscaled bytes, not the note and not the original. */
|
|
101
|
+
test("persists the downscaled image for a shrinkable tool_result screenshot", async () => {
|
|
102
|
+
// GIVEN a tool_result holding an oversized but shrinkable screenshot
|
|
103
|
+
const conv = createConversation();
|
|
104
|
+
await addMessage(
|
|
105
|
+
conv.id,
|
|
106
|
+
"user",
|
|
107
|
+
JSON.stringify([toolResultWithImage(oversizedPngBase64())]),
|
|
108
|
+
{ skipIndexing: true },
|
|
109
|
+
);
|
|
110
|
+
|
|
111
|
+
// WHEN the downgrade is persisted
|
|
112
|
+
const rewritten = persistUnsendableImageDowngrades(conv.id);
|
|
113
|
+
|
|
114
|
+
// THEN the nested block stays an image, rewritten to the downscaled payload
|
|
115
|
+
expect(rewritten).toBe(1);
|
|
116
|
+
const [content] = storedContent(conv.id);
|
|
117
|
+
const toolResult = content.find((b) => b.type === "tool_result") as {
|
|
118
|
+
contentBlocks?: ContentBlock[];
|
|
119
|
+
};
|
|
120
|
+
const nested = toolResult.contentBlocks?.[0];
|
|
121
|
+
expect(nested?.type).toBe("image");
|
|
122
|
+
expect((nested as Extract<ContentBlock, { type: "image" }>).source.data).toBe(
|
|
123
|
+
SHRUNK_DATA,
|
|
124
|
+
);
|
|
125
|
+
});
|
|
126
|
+
|
|
127
|
+
/** Re-running is a no-op: the downscaled payload is within limits. */
|
|
128
|
+
test("is idempotent after a downscale rewrite", async () => {
|
|
129
|
+
// GIVEN a conversation whose oversized screenshot was already downscaled
|
|
130
|
+
const conv = createConversation();
|
|
131
|
+
await addMessage(
|
|
132
|
+
conv.id,
|
|
133
|
+
"user",
|
|
134
|
+
JSON.stringify([toolResultWithImage(oversizedPngBase64())]),
|
|
135
|
+
{ skipIndexing: true },
|
|
136
|
+
);
|
|
137
|
+
expect(persistUnsendableImageDowngrades(conv.id)).toBe(1);
|
|
138
|
+
|
|
139
|
+
// WHEN the downgrade runs again
|
|
140
|
+
const secondRun = persistUnsendableImageDowngrades(conv.id);
|
|
141
|
+
|
|
142
|
+
// THEN nothing further is rewritten
|
|
143
|
+
expect(secondRun).toBe(0);
|
|
144
|
+
});
|
|
145
|
+
});
|
|
@@ -14,7 +14,10 @@
|
|
|
14
14
|
|
|
15
15
|
import { beforeEach, describe, expect, test } from "bun:test";
|
|
16
16
|
|
|
17
|
-
import {
|
|
17
|
+
import {
|
|
18
|
+
oversizedImageReplacement,
|
|
19
|
+
persistUnsendableImageDowngrades,
|
|
20
|
+
} from "../daemon/persist-unsendable-image.js";
|
|
18
21
|
import {
|
|
19
22
|
addMessage,
|
|
20
23
|
createConversation,
|
|
@@ -87,6 +90,20 @@ function imageBlock(data: string): ContentBlock {
|
|
|
87
90
|
};
|
|
88
91
|
}
|
|
89
92
|
|
|
93
|
+
/**
|
|
94
|
+
* A tool_result carrying a nested image in its contentBlocks, mirroring what a
|
|
95
|
+
* browser screenshot produces. This is the JARVIS-1041 shape: the oversized
|
|
96
|
+
* image lives at tool_result.contentBlocks, never as a top-level block.
|
|
97
|
+
*/
|
|
98
|
+
function toolResultWithImage(data: string): ContentBlock {
|
|
99
|
+
return {
|
|
100
|
+
type: "tool_result",
|
|
101
|
+
tool_use_id: "toolu_123",
|
|
102
|
+
content: "Screenshot captured",
|
|
103
|
+
contentBlocks: [imageBlock(data)],
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
90
107
|
function storedContent(conversationId: string): ContentBlock[][] {
|
|
91
108
|
return getMessages(conversationId).map(
|
|
92
109
|
(row) => JSON.parse(row.content) as ContentBlock[],
|
|
@@ -194,6 +211,61 @@ describe("persistUnsendableImageDowngrades", () => {
|
|
|
194
211
|
expect(content.some((b) => b.type === "image")).toBe(false);
|
|
195
212
|
});
|
|
196
213
|
|
|
214
|
+
/** JARVIS-1041: the oversized image is nested inside a tool_result (e.g. a
|
|
215
|
+
* browser screenshot), not a top-level block. The downgrade must descend
|
|
216
|
+
* into tool_result.contentBlocks and swap the nested image for a note, while
|
|
217
|
+
* keeping the tool_result itself intact so tool_use/tool_result pairing
|
|
218
|
+
* survives. */
|
|
219
|
+
test("downgrades an oversized image nested in tool_result.contentBlocks", async () => {
|
|
220
|
+
// GIVEN an assistant turn whose tool_result holds an oversized screenshot
|
|
221
|
+
const conv = createConversation();
|
|
222
|
+
await addMessage(
|
|
223
|
+
conv.id,
|
|
224
|
+
"user",
|
|
225
|
+
JSON.stringify([toolResultWithImage(makePngBase64(12000, 9000))]),
|
|
226
|
+
{ skipIndexing: true },
|
|
227
|
+
);
|
|
228
|
+
|
|
229
|
+
// WHEN the downgrade is persisted
|
|
230
|
+
const rewritten = persistUnsendableImageDowngrades(conv.id);
|
|
231
|
+
|
|
232
|
+
// THEN the message is rewritten with the nested image swapped for a note
|
|
233
|
+
expect(rewritten).toBe(1);
|
|
234
|
+
const [content] = storedContent(conv.id);
|
|
235
|
+
const toolResult = content.find((b) => b.type === "tool_result");
|
|
236
|
+
expect(toolResult).toBeDefined();
|
|
237
|
+
// AND the tool_result is preserved (pairing intact) with no image left
|
|
238
|
+
const nested = (toolResult as { contentBlocks?: ContentBlock[] })
|
|
239
|
+
.contentBlocks;
|
|
240
|
+
expect(nested?.some((b) => b.type === "image")).toBe(false);
|
|
241
|
+
expect(nested?.some((b) => b.type === "text")).toBe(true);
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
/** A sendable nested screenshot is never disturbed. */
|
|
245
|
+
test("leaves a normally-sized tool_result image untouched", async () => {
|
|
246
|
+
// GIVEN a tool_result with an image well within provider limits
|
|
247
|
+
const conv = createConversation();
|
|
248
|
+
await addMessage(
|
|
249
|
+
conv.id,
|
|
250
|
+
"user",
|
|
251
|
+
JSON.stringify([toolResultWithImage(makePngBase64(1024, 768))]),
|
|
252
|
+
{ skipIndexing: true },
|
|
253
|
+
);
|
|
254
|
+
|
|
255
|
+
// WHEN the downgrade is persisted
|
|
256
|
+
const rewritten = persistUnsendableImageDowngrades(conv.id);
|
|
257
|
+
|
|
258
|
+
// THEN nothing is rewritten and the nested image remains
|
|
259
|
+
expect(rewritten).toBe(0);
|
|
260
|
+
const [content] = storedContent(conv.id);
|
|
261
|
+
const toolResult = content.find((b) => b.type === "tool_result") as {
|
|
262
|
+
contentBlocks?: ContentBlock[];
|
|
263
|
+
};
|
|
264
|
+
expect(toolResult.contentBlocks?.some((b) => b.type === "image")).toBe(
|
|
265
|
+
true,
|
|
266
|
+
);
|
|
267
|
+
});
|
|
268
|
+
|
|
197
269
|
/** Re-running after a rewrite is a safe no-op (no image blocks remain). */
|
|
198
270
|
test("is idempotent — a second run rewrites nothing", async () => {
|
|
199
271
|
// GIVEN a conversation whose oversized image has already been downgraded
|
|
@@ -213,3 +285,27 @@ describe("persistUnsendableImageDowngrades", () => {
|
|
|
213
285
|
expect(secondRun).toBe(0);
|
|
214
286
|
});
|
|
215
287
|
});
|
|
288
|
+
|
|
289
|
+
describe("oversizedImageReplacement", () => {
|
|
290
|
+
/** A still-sendable image must be left alone — never replaced with a note.
|
|
291
|
+
* This is the gate that keeps the in-memory recovery from discarding valid
|
|
292
|
+
* screenshots when only one image in the turn was actually oversized. */
|
|
293
|
+
test("returns null for an image within the provider caps", () => {
|
|
294
|
+
const sendable = imageBlock(makePngBase64(1024, 768)) as Extract<
|
|
295
|
+
ContentBlock,
|
|
296
|
+
{ type: "image" }
|
|
297
|
+
>;
|
|
298
|
+
expect(oversizedImageReplacement(sendable)).toBeNull();
|
|
299
|
+
});
|
|
300
|
+
|
|
301
|
+
/** An image past the provider caps that cannot be shrunk on this host (fake
|
|
302
|
+
* PNG that sips cannot decode) collapses to the unsendable note. */
|
|
303
|
+
test("returns the unsendable note when an oversized image cannot be shrunk", () => {
|
|
304
|
+
const oversized = imageBlock(makePngBase64(12000, 9000)) as Extract<
|
|
305
|
+
ContentBlock,
|
|
306
|
+
{ type: "image" }
|
|
307
|
+
>;
|
|
308
|
+
const replacement = oversizedImageReplacement(oversized);
|
|
309
|
+
expect(replacement?.type).toBe("text");
|
|
310
|
+
});
|
|
311
|
+
});
|
|
@@ -12,6 +12,7 @@ import {
|
|
|
12
12
|
TARGET_CHARS,
|
|
13
13
|
THRESHOLD_CHARS,
|
|
14
14
|
TOOL_RESULT_DIR,
|
|
15
|
+
TRUNCATION_EXEMPT_TOOLS,
|
|
15
16
|
TRUNCATION_MARKER,
|
|
16
17
|
} from "../context/post-turn-tool-result-truncation.js";
|
|
17
18
|
import type { ContentBlock, Message } from "../providers/types.js";
|
|
@@ -151,6 +152,74 @@ describe("postTurnTruncateToolResults", () => {
|
|
|
151
152
|
expect(stub).toContain(filePath);
|
|
152
153
|
});
|
|
153
154
|
|
|
155
|
+
test("skill_load result above threshold is NOT truncated (durable instructions exempt)", () => {
|
|
156
|
+
// Regression for JARVIS-1000: a hosted assistant lost its app-builder skill
|
|
157
|
+
// workflow when the large skill_load result was middle-truncated between the
|
|
158
|
+
// turn that loaded the skill and the turn that used it, then fell back to a
|
|
159
|
+
// local-dev (vite/localhost) build path.
|
|
160
|
+
const toolUseId = "tool_skill_load";
|
|
161
|
+
const skillBody = "S".repeat(THRESHOLD_CHARS + 5_000);
|
|
162
|
+
const messages: Message[] = [
|
|
163
|
+
{
|
|
164
|
+
role: "assistant",
|
|
165
|
+
content: [
|
|
166
|
+
{
|
|
167
|
+
type: "tool_use" as const,
|
|
168
|
+
id: toolUseId,
|
|
169
|
+
name: "skill_load",
|
|
170
|
+
input: { skill: "app-builder" },
|
|
171
|
+
},
|
|
172
|
+
],
|
|
173
|
+
},
|
|
174
|
+
{ role: "user", content: [makeToolResult(skillBody, toolUseId)] },
|
|
175
|
+
];
|
|
176
|
+
|
|
177
|
+
const { messages: result, truncatedCount } =
|
|
178
|
+
postTurnTruncateToolResults(messages, { conversationDir: convDir });
|
|
179
|
+
|
|
180
|
+
expect(truncatedCount).toBe(0);
|
|
181
|
+
expect(result).toBe(messages); // same reference — no copy
|
|
182
|
+
expect(existsSync(join(convDir, TOOL_RESULT_DIR))).toBe(false);
|
|
183
|
+
|
|
184
|
+
const block = result[1].content[0] as {
|
|
185
|
+
type: "tool_result";
|
|
186
|
+
content: string;
|
|
187
|
+
};
|
|
188
|
+
expect(block.content).toBe(skillBody);
|
|
189
|
+
expect(TRUNCATION_EXEMPT_TOOLS.has("skill_load")).toBe(true);
|
|
190
|
+
});
|
|
191
|
+
|
|
192
|
+
test("non-exempt tool result above threshold is still truncated when paired with a tool_use", () => {
|
|
193
|
+
// Control for the exemption: same shape as the skill_load case, but a tool
|
|
194
|
+
// that is NOT exempt must still be truncated.
|
|
195
|
+
const toolUseId = "tool_bash_1";
|
|
196
|
+
const longContent = "B".repeat(THRESHOLD_CHARS + 5_000);
|
|
197
|
+
const messages: Message[] = [
|
|
198
|
+
{
|
|
199
|
+
role: "assistant",
|
|
200
|
+
content: [
|
|
201
|
+
{
|
|
202
|
+
type: "tool_use" as const,
|
|
203
|
+
id: toolUseId,
|
|
204
|
+
name: "bash",
|
|
205
|
+
input: { command: "cat big.log" },
|
|
206
|
+
},
|
|
207
|
+
],
|
|
208
|
+
},
|
|
209
|
+
{ role: "user", content: [makeToolResult(longContent, toolUseId)] },
|
|
210
|
+
];
|
|
211
|
+
|
|
212
|
+
const { messages: result, truncatedCount } =
|
|
213
|
+
postTurnTruncateToolResults(messages, { conversationDir: convDir });
|
|
214
|
+
|
|
215
|
+
expect(truncatedCount).toBe(1);
|
|
216
|
+
const block = result[1].content[0] as {
|
|
217
|
+
type: "tool_result";
|
|
218
|
+
content: string;
|
|
219
|
+
};
|
|
220
|
+
expect(block.content).toContain(TRUNCATION_MARKER);
|
|
221
|
+
});
|
|
222
|
+
|
|
154
223
|
test("file path is deterministic for the same toolUseId", () => {
|
|
155
224
|
const id = "tool_use_deterministic";
|
|
156
225
|
const path1 = getToolResultFilePath("/some/dir", id);
|
|
@@ -233,10 +233,9 @@ describe("frontmatter feature-flag integration", () => {
|
|
|
233
233
|
// ---------------------------------------------------------------------------
|
|
234
234
|
|
|
235
235
|
describe("bundled acp skill discoverability", () => {
|
|
236
|
-
test("acp skill resolves with
|
|
237
|
-
// The ACP skill carries its own first-time-setup instructions
|
|
238
|
-
//
|
|
239
|
-
// Runtime enforcement happens in the ACP tools via isAcpEnabled instead.
|
|
236
|
+
test("acp skill resolves with no frontmatter flag gate", () => {
|
|
237
|
+
// The ACP skill carries its own first-time-setup instructions and is
|
|
238
|
+
// always discoverable: it has no frontmatter feature-flag gate.
|
|
240
239
|
const skillMdPath = fileURLToPath(
|
|
241
240
|
new URL("../config/bundled-skills/acp/SKILL.md", import.meta.url),
|
|
242
241
|
);
|
|
@@ -247,10 +246,9 @@ describe("bundled acp skill discoverability", () => {
|
|
|
247
246
|
expect(skill!.featureFlag).toBeUndefined();
|
|
248
247
|
expect(skillFlagKey(skill!)).toBeUndefined();
|
|
249
248
|
|
|
250
|
-
// acp flag at its registry default (off) and config.acp disabled.
|
|
251
249
|
const config = makeConfig({
|
|
252
|
-
acp: {
|
|
253
|
-
}
|
|
250
|
+
acp: { maxConcurrentSessions: 4, agents: {} },
|
|
251
|
+
});
|
|
254
252
|
|
|
255
253
|
const resolved = resolveSkillStates([skill!], config);
|
|
256
254
|
expect(resolved.length).toBe(1);
|
|
@@ -5,6 +5,7 @@ import { afterEach, beforeEach, describe, expect, test } from "bun:test";
|
|
|
5
5
|
import {
|
|
6
6
|
buildFetchResponseFromNodeResponse,
|
|
7
7
|
executeWebFetch,
|
|
8
|
+
getUpstreamStatus,
|
|
8
9
|
} from "../tools/network/web-fetch.js";
|
|
9
10
|
|
|
10
11
|
describe("web_fetch tool", () => {
|
|
@@ -62,6 +63,50 @@ describe("web_fetch tool", () => {
|
|
|
62
63
|
expect(await response.text()).toBe("");
|
|
63
64
|
});
|
|
64
65
|
|
|
66
|
+
test("buildFetchResponseFromNodeResponse does not throw on non-standard status codes", async () => {
|
|
67
|
+
const stream = new PassThrough() as PassThrough & {
|
|
68
|
+
statusCode?: number;
|
|
69
|
+
statusMessage?: string;
|
|
70
|
+
headers: IncomingHttpHeaders;
|
|
71
|
+
};
|
|
72
|
+
stream.statusCode = 999;
|
|
73
|
+
stream.statusMessage = "Request Denied";
|
|
74
|
+
stream.headers = { "content-type": "text/html; charset=utf-8" };
|
|
75
|
+
stream.end("blocked by anti-bot gateway");
|
|
76
|
+
|
|
77
|
+
const response = buildFetchResponseFromNodeResponse(stream);
|
|
78
|
+
// The Response object clamps to a constructable code, but the real upstream
|
|
79
|
+
// status is preserved for reporting.
|
|
80
|
+
expect(response.status).toBe(502);
|
|
81
|
+
expect(getUpstreamStatus(response)).toBe(999);
|
|
82
|
+
expect(response.ok).toBe(false);
|
|
83
|
+
expect(await response.text()).toBe("blocked by anti-bot gateway");
|
|
84
|
+
});
|
|
85
|
+
|
|
86
|
+
test("surfaces a non-standard status as a tool error instead of crashing", async () => {
|
|
87
|
+
const result = await executeWithMockFetch(
|
|
88
|
+
{ url: "https://www.linkedin.com/jobs/view/123" },
|
|
89
|
+
{
|
|
90
|
+
requestExecutor: async () => {
|
|
91
|
+
const stream = new PassThrough() as PassThrough & {
|
|
92
|
+
statusCode?: number;
|
|
93
|
+
statusMessage?: string;
|
|
94
|
+
headers: IncomingHttpHeaders;
|
|
95
|
+
};
|
|
96
|
+
stream.statusCode = 999;
|
|
97
|
+
stream.statusMessage = "Request Denied";
|
|
98
|
+
stream.headers = { "content-type": "text/html; charset=utf-8" };
|
|
99
|
+
stream.end("<html><body>blocked</body></html>");
|
|
100
|
+
return buildFetchResponseFromNodeResponse(stream);
|
|
101
|
+
},
|
|
102
|
+
},
|
|
103
|
+
);
|
|
104
|
+
|
|
105
|
+
expect(result.isError).toBe(true);
|
|
106
|
+
expect(result.content).toContain("HTTP 999");
|
|
107
|
+
expect(result.activityMetadata?.webFetch?.status).toBe(999);
|
|
108
|
+
});
|
|
109
|
+
|
|
65
110
|
test("rejects missing url", async () => {
|
|
66
111
|
const result = await executeWithMockFetch({});
|
|
67
112
|
expect(result.isError).toBe(true);
|
|
@@ -20,13 +20,11 @@ import { mock } from "bun:test";
|
|
|
20
20
|
import type { AcpAgentConfig } from "../../../config/acp-schema.js";
|
|
21
21
|
|
|
22
22
|
export interface MockAcpConfig {
|
|
23
|
-
enabled: boolean;
|
|
24
23
|
maxConcurrentSessions: number;
|
|
25
24
|
agents: Record<string, AcpAgentConfig>;
|
|
26
25
|
}
|
|
27
26
|
|
|
28
27
|
const DEFAULT_CONFIG: MockAcpConfig = {
|
|
29
|
-
enabled: true,
|
|
30
28
|
maxConcurrentSessions: 4,
|
|
31
29
|
agents: {},
|
|
32
30
|
};
|
|
@@ -1,25 +1,19 @@
|
|
|
1
1
|
import { afterAll, beforeEach, describe, expect, test } from "bun:test";
|
|
2
2
|
|
|
3
|
-
import { setOverridesForTesting } from "../__tests__/feature-flag-test-helpers.js";
|
|
4
3
|
import { installAcpConfigStub } from "./__tests__/helpers/acp-config-stub.js";
|
|
5
4
|
import { installWhichStub } from "./__tests__/helpers/which-stub.js";
|
|
6
|
-
import { ACP_FLAG_KEY } from "./feature-gate.js";
|
|
7
5
|
|
|
8
6
|
const config = await installAcpConfigStub();
|
|
9
7
|
const which = installWhichStub();
|
|
10
8
|
|
|
11
9
|
afterAll(() => {
|
|
12
10
|
which.restore();
|
|
13
|
-
setOverridesForTesting({});
|
|
14
11
|
});
|
|
15
12
|
|
|
16
13
|
const { resolveAcpAgent, listAcpAgents } = await import("./resolve-agent.js");
|
|
17
14
|
|
|
18
15
|
beforeEach(() => {
|
|
19
16
|
config.setConfig({});
|
|
20
|
-
// Default: no flag overrides, so the `acp` flag falls back to its registry
|
|
21
|
-
// default (false) and enablement comes from the config stub alone.
|
|
22
|
-
setOverridesForTesting({});
|
|
23
17
|
// Default: every command on PATH so binary preflight passes unless a test
|
|
24
18
|
// explicitly says otherwise.
|
|
25
19
|
which.setWhich((cmd) => `/usr/local/bin/${cmd}`);
|
|
@@ -30,32 +24,6 @@ beforeEach(() => {
|
|
|
30
24
|
// ---------------------------------------------------------------------------
|
|
31
25
|
|
|
32
26
|
describe("resolveAcpAgent", () => {
|
|
33
|
-
test("returns acp_disabled when both the feature flag and config.acp.enabled are off", () => {
|
|
34
|
-
config.setConfig({ enabled: false });
|
|
35
|
-
|
|
36
|
-
const result = resolveAcpAgent("claude");
|
|
37
|
-
|
|
38
|
-
expect(result.ok).toBe(false);
|
|
39
|
-
if (result.ok) return;
|
|
40
|
-
expect(result.reason).toBe("acp_disabled");
|
|
41
|
-
if (result.reason !== "acp_disabled") return;
|
|
42
|
-
expect(result.hint).toContain("ACP Coding Agents");
|
|
43
|
-
expect(result.hint).toContain("feature flag");
|
|
44
|
-
expect(result.hint).toContain("acp.enabled");
|
|
45
|
-
expect(result.hint).toContain("config.json");
|
|
46
|
-
});
|
|
47
|
-
|
|
48
|
-
test("resolution proceeds when the acp feature flag is on and config.acp.enabled is false", () => {
|
|
49
|
-
config.setConfig({ enabled: false });
|
|
50
|
-
setOverridesForTesting({ [ACP_FLAG_KEY]: true });
|
|
51
|
-
|
|
52
|
-
const result = resolveAcpAgent("claude");
|
|
53
|
-
|
|
54
|
-
expect(result.ok).toBe(true);
|
|
55
|
-
if (!result.ok) return;
|
|
56
|
-
expect(result.agent.command).toBe("claude-agent-acp");
|
|
57
|
-
});
|
|
58
|
-
|
|
59
27
|
test("user config wins over default profile", () => {
|
|
60
28
|
config.setConfig({
|
|
61
29
|
agents: {
|
|
@@ -359,35 +327,11 @@ describe("resolveAcpAgent - missing binary", () => {
|
|
|
359
327
|
// ---------------------------------------------------------------------------
|
|
360
328
|
|
|
361
329
|
describe("listAcpAgents", () => {
|
|
362
|
-
test("returns enabled: false with empty agents when both the flag and config are off", () => {
|
|
363
|
-
config.setConfig({ enabled: false });
|
|
364
|
-
|
|
365
|
-
const result = listAcpAgents();
|
|
366
|
-
|
|
367
|
-
expect(result.enabled).toBe(false);
|
|
368
|
-
expect(result.agents).toEqual([]);
|
|
369
|
-
});
|
|
370
|
-
|
|
371
|
-
test("returns the catalog when the acp feature flag is on and config.acp.enabled is false", () => {
|
|
372
|
-
config.setConfig({ enabled: false });
|
|
373
|
-
setOverridesForTesting({ [ACP_FLAG_KEY]: true });
|
|
374
|
-
|
|
375
|
-
const result = listAcpAgents();
|
|
376
|
-
|
|
377
|
-
expect(result.enabled).toBe(true);
|
|
378
|
-
expect(result.agents.map((a) => a.id)).toEqual([
|
|
379
|
-
"claude",
|
|
380
|
-
"codex",
|
|
381
|
-
"gemini",
|
|
382
|
-
]);
|
|
383
|
-
});
|
|
384
|
-
|
|
385
330
|
test("includes all bundled defaults when user config is empty", () => {
|
|
386
331
|
config.setConfig({ agents: {} });
|
|
387
332
|
|
|
388
333
|
const result = listAcpAgents();
|
|
389
334
|
|
|
390
|
-
expect(result.enabled).toBe(true);
|
|
391
335
|
const ids = result.agents.map((a) => a.id);
|
|
392
336
|
expect(ids).toEqual(["claude", "codex", "gemini"]);
|
|
393
337
|
for (const entry of result.agents) {
|
package/src/acp/resolve-agent.ts
CHANGED
|
@@ -3,15 +3,13 @@
|
|
|
3
3
|
*
|
|
4
4
|
* `resolveAcpAgent(id)` merges user-provided `config.acp.agents[id]` (wins on
|
|
5
5
|
* overlap) with the bundled `DEFAULT_ACP_AGENT_PROFILES` so common agents like
|
|
6
|
-
* `claude` and `codex` Just Work
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
* acp_list_agents, and the `/v1/acp/spawn` HTTP route) get a single source
|
|
14
|
-
* of truth and matching actionable hints.
|
|
6
|
+
* `claude` and `codex` Just Work with no per-user config required. Natural
|
|
7
|
+
* names ("claude code", "Gemini CLI") resolve via `AGENT_ID_ALIASES` when the
|
|
8
|
+
* raw id misses both maps. The result is a discriminated union covering every
|
|
9
|
+
* reason a spawn might fail before we even start the agent process: unknown
|
|
10
|
+
* agent id, or binary missing from PATH. Callers (acp_spawn, acp_list_agents,
|
|
11
|
+
* and the `/v1/acp/spawn` HTTP route) get a single source of truth and
|
|
12
|
+
* matching actionable hints.
|
|
15
13
|
*
|
|
16
14
|
* The resolver NEVER fetches or runs packages in the (untrusted) task cwd.
|
|
17
15
|
* When the adapter binary is missing, resolution simply fails with
|
|
@@ -33,7 +31,6 @@ import {
|
|
|
33
31
|
} from "../config/acp-defaults.js";
|
|
34
32
|
import type { AcpAgentConfig } from "../config/acp-schema.js";
|
|
35
33
|
import { getConfig } from "../config/loader.js";
|
|
36
|
-
import { isAcpEnabled } from "./feature-gate.js";
|
|
37
34
|
|
|
38
35
|
/**
|
|
39
36
|
* Whether this agent's entry came from user config (wins over default) or
|
|
@@ -47,7 +44,6 @@ export type ResolveAcpAgentResult =
|
|
|
47
44
|
| ResolveAcpAgentFailure;
|
|
48
45
|
|
|
49
46
|
export type ResolveAcpAgentFailure =
|
|
50
|
-
| { ok: false; reason: "acp_disabled"; hint: string }
|
|
51
47
|
| { ok: false; reason: "unknown_agent"; available: string[] }
|
|
52
48
|
| {
|
|
53
49
|
ok: false;
|
|
@@ -68,8 +64,6 @@ export function formatResolveFailure(
|
|
|
68
64
|
failure: ResolveAcpAgentFailure,
|
|
69
65
|
): string {
|
|
70
66
|
switch (failure.reason) {
|
|
71
|
-
case "acp_disabled":
|
|
72
|
-
return failure.hint;
|
|
73
67
|
case "unknown_agent":
|
|
74
68
|
return `Unknown agent "${agentId}". Available: ${failure.available.join(", ")}.`;
|
|
75
69
|
case "binary_not_found":
|
|
@@ -93,14 +87,6 @@ interface AcpAgentEntry {
|
|
|
93
87
|
setupHint?: string;
|
|
94
88
|
}
|
|
95
89
|
|
|
96
|
-
/**
|
|
97
|
-
* Single-source-of-truth hint for "ACP is disabled". Exported so any caller
|
|
98
|
-
* that surfaces a disabled-state message (resolver, list-agents tool) reads
|
|
99
|
-
* the same string instead of duplicating near-identical copy.
|
|
100
|
-
*/
|
|
101
|
-
export const ACP_DISABLED_HINT =
|
|
102
|
-
"Enable the \"ACP Coding Agents\" feature flag in the client's feature flags UI (or set 'acp.enabled': true in ~/.vellum/workspace/config.json).";
|
|
103
|
-
|
|
104
90
|
function installHintFor(command: string): string {
|
|
105
91
|
const pkg = DEFAULT_AGENT_NPM_PACKAGES[command];
|
|
106
92
|
return pkg
|
|
@@ -209,9 +195,8 @@ function mergedAgentIds(userAgents: Record<string, AcpAgentConfig>): string[] {
|
|
|
209
195
|
* Resolve an ACP agent id to its config + binary preflight result.
|
|
210
196
|
*
|
|
211
197
|
* Order of checks:
|
|
212
|
-
* 1.
|
|
213
|
-
* 2. The
|
|
214
|
-
* 3. The agent must be runnable: its `command` on PATH (see
|
|
198
|
+
* 1. The id must resolve to an agent (user config wins; falls back to defaults).
|
|
199
|
+
* 2. The agent must be runnable: its `command` on PATH (see
|
|
215
200
|
* `resolveRunnableAgent`).
|
|
216
201
|
*
|
|
217
202
|
* Each failure mode carries an actionable hint so callers can surface a
|
|
@@ -219,10 +204,6 @@ function mergedAgentIds(userAgents: Record<string, AcpAgentConfig>): string[] {
|
|
|
219
204
|
*/
|
|
220
205
|
export function resolveAcpAgent(id: string): ResolveAcpAgentResult {
|
|
221
206
|
const config = getConfig();
|
|
222
|
-
if (!isAcpEnabled(config)) {
|
|
223
|
-
return { ok: false, reason: "acp_disabled", hint: ACP_DISABLED_HINT };
|
|
224
|
-
}
|
|
225
|
-
|
|
226
207
|
const userAgents = config.acp.agents;
|
|
227
208
|
const found = lookupAgent(userAgents, id);
|
|
228
209
|
if (!found) {
|
|
@@ -252,20 +233,11 @@ export function resolveAcpAgent(id: string): ResolveAcpAgentResult {
|
|
|
252
233
|
* plus any user-only entries — with per-entry availability info. Used by the
|
|
253
234
|
* `acp_list_agents` tool to render setup steps when an agent's binary isn't
|
|
254
235
|
* installed yet.
|
|
255
|
-
*
|
|
256
|
-
* `enabled: false` short-circuits and returns an empty catalog so the tool
|
|
257
|
-
* can render a single "ACP is disabled" hint instead of advertising agents
|
|
258
|
-
* the user can't actually run.
|
|
259
236
|
*/
|
|
260
237
|
export function listAcpAgents(): {
|
|
261
|
-
enabled: boolean;
|
|
262
238
|
agents: AcpAgentEntry[];
|
|
263
239
|
} {
|
|
264
240
|
const config = getConfig();
|
|
265
|
-
if (!isAcpEnabled(config)) {
|
|
266
|
-
return { enabled: false, agents: [] };
|
|
267
|
-
}
|
|
268
|
-
|
|
269
241
|
const userAgents = config.acp.agents;
|
|
270
242
|
const agents: AcpAgentEntry[] = mergedAgentIds(userAgents).map((id) => {
|
|
271
243
|
// Non-null: ids come from `mergedAgentIds` so the lookup always resolves.
|
|
@@ -288,5 +260,5 @@ export function listAcpAgents(): {
|
|
|
288
260
|
return entry;
|
|
289
261
|
});
|
|
290
262
|
|
|
291
|
-
return {
|
|
263
|
+
return { agents };
|
|
292
264
|
}
|