@vellumai/assistant 0.11.4-staging.3 → 0.11.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/guardian-request-flow.md +35 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
- package/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
- package/package.json +1 -1
- package/src/__tests__/conversation-load-history-repair.test.ts +209 -0
- package/src/__tests__/conversation-runtime-assembly.test.ts +53 -0
- package/src/__tests__/file-ops-service.test.ts +163 -30
- package/src/__tests__/filesystem-tools.test.ts +23 -24
- package/src/__tests__/host-file-read-tool.test.ts +16 -19
- package/src/__tests__/tool-executor.test.ts +5 -1
- package/src/api/events/host-file.ts +2 -2
- package/src/daemon/conversation-runtime-assembly.ts +9 -2
- package/src/daemon/conversation.ts +23 -7
- package/src/notifications/AGENTS.md +2 -0
- package/src/notifications/approval-card-data.ts +33 -0
- package/src/subagent/types.ts +7 -7
- package/src/tools/__tests__/tool-input-schemas.test.ts +7 -7
- package/src/tools/filesystem/read.ts +27 -10
- package/src/tools/host-filesystem/read.ts +27 -15
- package/src/tools/shared/filesystem/file-ops-service.ts +63 -35
- package/src/tools/shared/filesystem/legacy-read-args.ts +22 -0
- package/src/tools/shared/filesystem/types.ts +5 -5
package/src/subagent/types.ts
CHANGED
|
@@ -485,13 +485,13 @@ export const SUBAGENT_ROLE_REGISTRY: Record<SubagentRole, SubagentRoleConfig> =
|
|
|
485
485
|
skillIds: [],
|
|
486
486
|
systemPromptPreamble: [
|
|
487
487
|
"You are a research subagent with read-only access: search the web, read and search files, and recall memories. There is no shell, and you cannot write or edit files.",
|
|
488
|
-
// `file_read` returns
|
|
489
|
-
//
|
|
490
|
-
//
|
|
491
|
-
//
|
|
492
|
-
//
|
|
493
|
-
//
|
|
494
|
-
//
|
|
488
|
+
// `file_read` returns a bounded character window with a truncation
|
|
489
|
+
// notice naming the resume offset, and oversized results spool to
|
|
490
|
+
// .tool-results/ like any other tool's (only re-reads of spooled
|
|
491
|
+
// content stay inline). So a ranged read is both the cheap shape and
|
|
492
|
+
// the one that avoids a spool round-trip. The anti-slicing intent
|
|
493
|
+
// stays: one pass over the range that is needed rather than many
|
|
494
|
+
// small ones.
|
|
495
495
|
"Working method: use code_search to search file contents across directories, file_list to enumerate paths, and file_read to read files and logs. Prefer broad code_search queries across a directory over one-symbol-at-a-time queries, and read the range you need in one pass rather than many small slices.",
|
|
496
496
|
"Send notify_parent (urgency 'important') as soon as each finding is confirmed, so progress survives interruption.",
|
|
497
497
|
"Your final message is the deliverable: a compact report that answers the objective, gives the evidence behind each claim (file:line references, URLs, or quotes), and names what you could not determine. For a root-cause investigation, use the sections Symptom, Root cause, Evidence, Suggested fix, Open questions.",
|
|
@@ -74,12 +74,12 @@ describe("parseToolInput", () => {
|
|
|
74
74
|
test("file_read drops malformed optional fields the tool always ignored", () => {
|
|
75
75
|
const result = parseToolInput("file_read", {
|
|
76
76
|
path: "notes.md",
|
|
77
|
-
|
|
78
|
-
|
|
77
|
+
start_index: "not-a-number",
|
|
78
|
+
max_chars: 10,
|
|
79
79
|
});
|
|
80
80
|
expect(result).toEqual({
|
|
81
81
|
ok: true,
|
|
82
|
-
data: { path: "notes.md",
|
|
82
|
+
data: { path: "notes.md", max_chars: 10 },
|
|
83
83
|
});
|
|
84
84
|
});
|
|
85
85
|
|
|
@@ -184,9 +184,9 @@ describe("derived input_schema", () => {
|
|
|
184
184
|
properties: Record<string, { type?: string; description?: string }>;
|
|
185
185
|
required: string[];
|
|
186
186
|
};
|
|
187
|
-
expect(schema.properties.
|
|
188
|
-
expect(schema.properties.
|
|
189
|
-
expect(schema.required).not.toContain("
|
|
190
|
-
expect(schema.required).not.toContain("
|
|
187
|
+
expect(schema.properties.start_index?.type).toBe("number");
|
|
188
|
+
expect(schema.properties.start_index?.description).toContain("0-indexed");
|
|
189
|
+
expect(schema.required).not.toContain("start_index");
|
|
190
|
+
expect(schema.required).not.toContain("max_chars");
|
|
191
191
|
});
|
|
192
192
|
});
|
|
@@ -12,6 +12,7 @@ import {
|
|
|
12
12
|
IMAGE_EXTENSIONS,
|
|
13
13
|
readImageFile,
|
|
14
14
|
} from "../shared/filesystem/image-read.js";
|
|
15
|
+
import { legacyReadArgsError } from "../shared/filesystem/legacy-read-args.js";
|
|
15
16
|
import { sandboxReadPolicy } from "../shared/filesystem/path-policy.js";
|
|
16
17
|
import {
|
|
17
18
|
invalidToolInputResult,
|
|
@@ -25,8 +26,8 @@ import type {
|
|
|
25
26
|
|
|
26
27
|
/**
|
|
27
28
|
* Model-input schema, the single source for both runtime validation (via
|
|
28
|
-
* `TOOL_INPUT_SCHEMAS`) and the advertised `input_schema` below. `
|
|
29
|
-
* `
|
|
29
|
+
* `TOOL_INPUT_SCHEMAS`) and the advertised `input_schema` below. `start_index`
|
|
30
|
+
* and `max_chars` catch to `undefined` because the tool ignores non-numeric
|
|
30
31
|
* values rather than failing the read.
|
|
31
32
|
*/
|
|
32
33
|
export const fileReadInputSchema = z.looseObject({
|
|
@@ -36,14 +37,16 @@ export const fileReadInputSchema = z.looseObject({
|
|
|
36
37
|
.describe(
|
|
37
38
|
"The path to the file to read (absolute or relative to working directory)",
|
|
38
39
|
),
|
|
39
|
-
|
|
40
|
+
start_index: z
|
|
40
41
|
.number()
|
|
41
|
-
.describe("
|
|
42
|
+
.describe("Character to start reading from (0-indexed). Text files only.")
|
|
42
43
|
.optional()
|
|
43
44
|
.catch(undefined),
|
|
44
|
-
|
|
45
|
+
max_chars: z
|
|
45
46
|
.number()
|
|
46
|
-
.describe(
|
|
47
|
+
.describe(
|
|
48
|
+
"Maximum number of characters to read. Defaults to 20000, which is also the ceiling. Text files only.",
|
|
49
|
+
)
|
|
47
50
|
.optional()
|
|
48
51
|
.catch(undefined),
|
|
49
52
|
activity: z
|
|
@@ -58,7 +61,7 @@ export const fileReadInputSchema = z.looseObject({
|
|
|
58
61
|
export const fileReadTool = {
|
|
59
62
|
name: "file_read",
|
|
60
63
|
description:
|
|
61
|
-
"Read the contents of a file on your own machine. Text reads return the first
|
|
64
|
+
"Read the contents of a file on your own machine. Text reads return the first 20000 characters unless you pass `max_chars`; when a read stops short the result says so, and `start_index` pages on from there. To find where something is in a large file, code_search is cheaper than paging through it. For image files (JPEG, PNG, GIF, WebP), returns the image for visual analysis. For audio files (MP3, WAV, OGG, FLAC, AAC, M4A), returns the audio for listening. Use host_file_read for files on your guardian's device instead.",
|
|
62
65
|
category: "filesystem",
|
|
63
66
|
executionTarget: "sandbox",
|
|
64
67
|
defaultRiskLevel: RiskLevel.Low,
|
|
@@ -75,9 +78,14 @@ export const fileReadTool = {
|
|
|
75
78
|
if (!parsed.success) {
|
|
76
79
|
return invalidToolInputResult("file_read", parsed.error);
|
|
77
80
|
}
|
|
78
|
-
const {
|
|
81
|
+
const {
|
|
82
|
+
path: rawPath,
|
|
83
|
+
start_index: startIndex,
|
|
84
|
+
max_chars: maxChars,
|
|
85
|
+
} = parsed.data;
|
|
79
86
|
|
|
80
|
-
// For image files, delegate to the shared image reader.
|
|
87
|
+
// For image files, delegate to the shared image reader. Media reads carry
|
|
88
|
+
// no window, so the legacy-argument guard below does not apply to them.
|
|
81
89
|
const ext = extname(rawPath).toLowerCase();
|
|
82
90
|
if (IMAGE_EXTENSIONS.has(ext)) {
|
|
83
91
|
const pathCheck = sandboxReadPolicy(rawPath, context.workingDir);
|
|
@@ -102,11 +110,20 @@ export const fileReadTool = {
|
|
|
102
110
|
return readAudioFile(pathCheck.resolved);
|
|
103
111
|
}
|
|
104
112
|
|
|
113
|
+
const legacyArgs = legacyReadArgsError("file_read", input);
|
|
114
|
+
if (legacyArgs !== undefined) {
|
|
115
|
+
return { content: legacyArgs, isError: true };
|
|
116
|
+
}
|
|
117
|
+
|
|
105
118
|
const ops = new FileSystemOps((path, opts) =>
|
|
106
119
|
sandboxReadPolicy(path, context.workingDir, opts),
|
|
107
120
|
);
|
|
108
121
|
|
|
109
|
-
const result = await ops.readFileSafe({
|
|
122
|
+
const result = await ops.readFileSafe({
|
|
123
|
+
path: rawPath,
|
|
124
|
+
startIndex,
|
|
125
|
+
maxChars,
|
|
126
|
+
});
|
|
110
127
|
|
|
111
128
|
if (!result.ok) {
|
|
112
129
|
const { error } = result;
|
|
@@ -11,13 +11,14 @@ import {
|
|
|
11
11
|
readAudioFile,
|
|
12
12
|
} from "../shared/filesystem/audio-read.js";
|
|
13
13
|
import {
|
|
14
|
-
DEFAULT_READ_LINE_LIMIT,
|
|
15
14
|
FileSystemOps,
|
|
15
|
+
READ_CHAR_BUDGET,
|
|
16
16
|
} from "../shared/filesystem/file-ops-service.js";
|
|
17
17
|
import {
|
|
18
18
|
IMAGE_EXTENSIONS,
|
|
19
19
|
readImageFile,
|
|
20
20
|
} from "../shared/filesystem/image-read.js";
|
|
21
|
+
import { legacyReadArgsError } from "../shared/filesystem/legacy-read-args.js";
|
|
21
22
|
import { hostPolicy } from "../shared/filesystem/path-policy.js";
|
|
22
23
|
import {
|
|
23
24
|
invalidToolInputResult,
|
|
@@ -32,9 +33,9 @@ import type {
|
|
|
32
33
|
/**
|
|
33
34
|
* Model-input schema, the single source for both runtime validation (via
|
|
34
35
|
* `TOOL_INPUT_SCHEMAS`) and the advertised `input_schema` below — mirrors
|
|
35
|
-
* `filesystem/read.ts`. `
|
|
36
|
-
* non-numeric value falls back to the default
|
|
37
|
-
* the call; `target_client_id` catches so a non-string (or empty) value means
|
|
36
|
+
* `filesystem/read.ts`. `start_index`/`max_chars` catch to `undefined` so a
|
|
37
|
+
* non-numeric value falls back to the default character window instead of
|
|
38
|
+
* failing the call; `target_client_id` catches so a non-string (or empty) value means
|
|
38
39
|
* "untargeted".
|
|
39
40
|
*/
|
|
40
41
|
export const hostFileReadInputSchema = z.looseObject({
|
|
@@ -44,14 +45,16 @@ export const hostFileReadInputSchema = z.looseObject({
|
|
|
44
45
|
.describe(
|
|
45
46
|
"Absolute path on the guardian's device, which is a separate filesystem from your workspace, to read.",
|
|
46
47
|
),
|
|
47
|
-
|
|
48
|
+
start_index: z
|
|
48
49
|
.number()
|
|
49
|
-
.describe("
|
|
50
|
+
.describe("Character to start reading from (0-indexed). Text files only.")
|
|
50
51
|
.optional()
|
|
51
52
|
.catch(undefined),
|
|
52
|
-
|
|
53
|
+
max_chars: z
|
|
53
54
|
.number()
|
|
54
|
-
.describe(
|
|
55
|
+
.describe(
|
|
56
|
+
"Maximum number of characters to read. Defaults to 20000, which is also the ceiling. Text files only.",
|
|
57
|
+
)
|
|
55
58
|
.optional()
|
|
56
59
|
.catch(undefined),
|
|
57
60
|
target_client_id: z
|
|
@@ -66,7 +69,7 @@ export const hostFileReadInputSchema = z.looseObject({
|
|
|
66
69
|
export const hostFileReadTool = {
|
|
67
70
|
name: "host_file_read",
|
|
68
71
|
description:
|
|
69
|
-
"Read the contents of a file on your guardian's device, including images (JPEG, PNG, GIF, WebP) and audio (MP3, WAV, OGG, FLAC, AAC, M4A). Text reads return the first
|
|
72
|
+
"Read the contents of a file on your guardian's device, including images (JPEG, PNG, GIF, WebP) and audio (MP3, WAV, OGG, FLAC, AAC, M4A). Text reads return the first 20000 characters unless you pass `max_chars`; when a read stops short the result says so, and `start_index` pages on from there. For files on your own machine, use file_read instead.",
|
|
70
73
|
category: "host-filesystem",
|
|
71
74
|
executionTarget: "host",
|
|
72
75
|
defaultRiskLevel: RiskLevel.Medium,
|
|
@@ -81,12 +84,12 @@ export const hostFileReadTool = {
|
|
|
81
84
|
if (!parsed.success) {
|
|
82
85
|
return invalidToolInputResult("host_file_read", parsed.error);
|
|
83
86
|
}
|
|
84
|
-
const { path: rawPath,
|
|
87
|
+
const { path: rawPath, start_index: startIndex } = parsed.data;
|
|
85
88
|
// Resolve the default here rather than leaving it to the read, so the
|
|
86
89
|
// proxied branch below is bounded by the same window as the local one. A
|
|
87
|
-
// proxied read that sent no
|
|
90
|
+
// proxied read that sent no budget would stream a whole host file across
|
|
88
91
|
// the bridge before anything could trim it.
|
|
89
|
-
const
|
|
92
|
+
const maxChars = parsed.data.max_chars ?? READ_CHAR_BUDGET;
|
|
90
93
|
|
|
91
94
|
const targetClientId =
|
|
92
95
|
parsed.data.target_client_id !== ""
|
|
@@ -153,8 +156,8 @@ export const hostFileReadTool = {
|
|
|
153
156
|
{
|
|
154
157
|
operation: "read",
|
|
155
158
|
path: rawPath,
|
|
156
|
-
|
|
157
|
-
|
|
159
|
+
startIndex,
|
|
160
|
+
maxChars,
|
|
158
161
|
targetClientId,
|
|
159
162
|
},
|
|
160
163
|
context.conversationId,
|
|
@@ -181,9 +184,18 @@ export const hostFileReadTool = {
|
|
|
181
184
|
return readAudioFile(pathCheck.resolved);
|
|
182
185
|
}
|
|
183
186
|
|
|
187
|
+
const legacyArgs = legacyReadArgsError("host_file_read", input);
|
|
188
|
+
if (legacyArgs !== undefined) {
|
|
189
|
+
return { content: legacyArgs, isError: true };
|
|
190
|
+
}
|
|
191
|
+
|
|
184
192
|
const ops = new FileSystemOps(hostPolicy);
|
|
185
193
|
|
|
186
|
-
const result = await ops.readFileSafe({
|
|
194
|
+
const result = await ops.readFileSafe({
|
|
195
|
+
path: rawPath,
|
|
196
|
+
startIndex,
|
|
197
|
+
maxChars,
|
|
198
|
+
});
|
|
187
199
|
|
|
188
200
|
if (!result.ok) {
|
|
189
201
|
const { error } = result;
|
|
@@ -78,26 +78,55 @@ function pathError(
|
|
|
78
78
|
}
|
|
79
79
|
|
|
80
80
|
/**
|
|
81
|
-
*
|
|
82
|
-
*
|
|
83
|
-
*
|
|
84
|
-
* so
|
|
85
|
-
* that turn. The cap bounds that; `offset`/`limit` page past it, and
|
|
86
|
-
* {@link truncationNotice} tells the model when it is looking at a window
|
|
87
|
-
* rather than the whole file.
|
|
81
|
+
* Characters returned by a read that names no `max_chars`. Stays under
|
|
82
|
+
* `THRESHOLD_CHARS` in `context/post-turn-tool-result-truncation.ts`, which
|
|
83
|
+
* spools any larger tool result to disk and replaces it inline with a short
|
|
84
|
+
* stub, so a default read returns content rather than a stub.
|
|
88
85
|
*/
|
|
89
|
-
export const
|
|
86
|
+
export const READ_CHAR_BUDGET = 20_000;
|
|
90
87
|
|
|
91
88
|
/**
|
|
92
|
-
* Trailing marker appended when a read stops short of the
|
|
93
|
-
*
|
|
94
|
-
*
|
|
89
|
+
* Trailing marker appended when a read stops short of the end of the file. A
|
|
90
|
+
* model that cannot tell a window from a whole file reasons about code it
|
|
91
|
+
* never saw.
|
|
95
92
|
*/
|
|
96
93
|
function truncationNotice(
|
|
97
|
-
|
|
98
|
-
|
|
94
|
+
start: number,
|
|
95
|
+
end: number,
|
|
96
|
+
totalChars: number,
|
|
99
97
|
): string {
|
|
100
|
-
return `\n\n[Truncated:
|
|
98
|
+
return `\n\n[Truncated: characters ${start}-${end} of ${totalChars}. Read on with start_index=${end}.]`;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const isHighSurrogate = (code: number): boolean =>
|
|
102
|
+
code >= 0xd800 && code <= 0xdbff;
|
|
103
|
+
const isLowSurrogate = (code: number): boolean =>
|
|
104
|
+
code >= 0xdc00 && code <= 0xdfff;
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Character window that never splits a surrogate pair. A split leaves a lone
|
|
108
|
+
* half at each edge, and each encodes to U+FFFD, so the character is lost from
|
|
109
|
+
* both this window and the next one paged in after it.
|
|
110
|
+
*/
|
|
111
|
+
export function surrogateSafeWindow(
|
|
112
|
+
total: number,
|
|
113
|
+
charCodeAt: (index: number) => number,
|
|
114
|
+
requestedStart: number,
|
|
115
|
+
maxChars: number,
|
|
116
|
+
): { start: number; end: number } {
|
|
117
|
+
let start = Math.max(0, Math.min(requestedStart, total));
|
|
118
|
+
if (start > 0 && start < total && isLowSurrogate(charCodeAt(start))) {
|
|
119
|
+
start -= 1;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
let end = Math.min(total, start + maxChars);
|
|
123
|
+
if (end > start && end < total && isHighSurrogate(charCodeAt(end - 1))) {
|
|
124
|
+
// Backing off would empty a one-character window, which stalls paging on
|
|
125
|
+
// the same offset, so take the whole pair instead.
|
|
126
|
+
end = end - 1 > start ? end - 1 : Math.min(total, end + 1);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
return { start, end };
|
|
101
130
|
}
|
|
102
131
|
|
|
103
132
|
export class FileSystemOps {
|
|
@@ -139,28 +168,27 @@ export class FileSystemOps {
|
|
|
139
168
|
|
|
140
169
|
try {
|
|
141
170
|
const raw = await readFile(filePath, "utf-8");
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
const
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
const
|
|
150
|
-
.
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
// the caller paged past the end or asked for
|
|
158
|
-
// truncated read.
|
|
159
|
-
const lastLineReturned = start + selected.length;
|
|
171
|
+
|
|
172
|
+
// A ceiling, not just a default: a larger window would be spooled to
|
|
173
|
+
// disk and replaced with a stub, returning less than this.
|
|
174
|
+
const maxChars = Math.min(
|
|
175
|
+
READ_CHAR_BUDGET,
|
|
176
|
+
Math.max(0, input.maxChars ?? READ_CHAR_BUDGET),
|
|
177
|
+
);
|
|
178
|
+
const { start, end } = surrogateSafeWindow(
|
|
179
|
+
raw.length,
|
|
180
|
+
(i) => raw.charCodeAt(i),
|
|
181
|
+
input.startIndex ?? 0,
|
|
182
|
+
maxChars,
|
|
183
|
+
);
|
|
184
|
+
const window = raw.slice(start, end);
|
|
185
|
+
|
|
186
|
+
// An empty window means the caller paged past the end or asked for
|
|
187
|
+
// nothing, which is not a truncated read.
|
|
160
188
|
const content =
|
|
161
|
-
|
|
162
|
-
?
|
|
163
|
-
:
|
|
189
|
+
window.length > 0 && end < raw.length
|
|
190
|
+
? window + truncationNotice(start, end, raw.length)
|
|
191
|
+
: window;
|
|
164
192
|
|
|
165
193
|
return { ok: true, value: { content } };
|
|
166
194
|
} catch (err) {
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Guard for callers still passing the line-based `offset`/`limit` read
|
|
3
|
+
* arguments. The tool schemas are loose, so those keys parse and are then
|
|
4
|
+
* dropped, which would silently serve the start of the file to a caller that
|
|
5
|
+
* asked to page into the middle of it.
|
|
6
|
+
*
|
|
7
|
+
* There is no safe translation: `offset` counted lines and `start_index`
|
|
8
|
+
* counts characters, so mapping one onto the other would read the wrong
|
|
9
|
+
* region rather than fail. Naming the rename lets the caller correct itself.
|
|
10
|
+
*/
|
|
11
|
+
export function legacyReadArgsError(
|
|
12
|
+
toolName: string,
|
|
13
|
+
input: Record<string, unknown>,
|
|
14
|
+
): string | undefined {
|
|
15
|
+
const usesLegacy = input.offset !== undefined || input.limit !== undefined;
|
|
16
|
+
const usesCurrent =
|
|
17
|
+
input.start_index !== undefined || input.max_chars !== undefined;
|
|
18
|
+
if (!usesLegacy || usesCurrent) {
|
|
19
|
+
return undefined;
|
|
20
|
+
}
|
|
21
|
+
return `Error: ${toolName} no longer takes \`offset\`/\`limit\` (lines). Use \`start_index\` (0-indexed characters) and \`max_chars\` instead.`;
|
|
22
|
+
}
|
|
@@ -6,14 +6,14 @@ import type { FsError } from "./errors.js";
|
|
|
6
6
|
|
|
7
7
|
export interface ReadInput {
|
|
8
8
|
path: string;
|
|
9
|
-
/**
|
|
10
|
-
|
|
11
|
-
/** Maximum number of
|
|
12
|
-
|
|
9
|
+
/** 0-indexed character to start reading from. */
|
|
10
|
+
startIndex?: number;
|
|
11
|
+
/** Maximum number of characters to read. */
|
|
12
|
+
maxChars?: number;
|
|
13
13
|
}
|
|
14
14
|
|
|
15
15
|
export interface ReadOutput {
|
|
16
|
-
/** The
|
|
16
|
+
/** The character window, with a truncation notice when it stops short. */
|
|
17
17
|
content: string;
|
|
18
18
|
}
|
|
19
19
|
|