@vellumai/assistant 0.11.4-staging.2 → 0.11.4-staging.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (209) hide show
  1. package/AGENTS.md +8 -2
  2. package/ARCHITECTURE.md +2 -0
  3. package/docs/architecture/memory.md +15 -0
  4. package/docs/browser-use-architecture-phase2.md +128 -56
  5. package/docs/flux-turn-detection-spike.md +11 -6
  6. package/docs/guardian-request-flow.md +35 -0
  7. package/knip.json +3 -0
  8. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
  9. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
  10. package/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
  11. package/openapi.yaml +136 -76
  12. package/package.json +1 -1
  13. package/scripts/write-plugin-api-shim.ts +10 -0
  14. package/src/__tests__/app-control-flow.test.ts +1 -0
  15. package/src/__tests__/approval-routes-http.test.ts +2 -2
  16. package/src/__tests__/assistant-feature-flag-guard.test.ts +25 -3
  17. package/src/__tests__/channel-setup-panel-ack.test.ts +1 -1
  18. package/src/__tests__/compaction-events.test.ts +8 -10
  19. package/src/__tests__/conversation-agent-loop.test.ts +4 -1
  20. package/src/__tests__/conversation-confirmation-signals.test.ts +112 -0
  21. package/src/__tests__/conversation-load-history-repair.test.ts +209 -0
  22. package/src/__tests__/conversation-notifiers-provenance.test.ts +1 -1
  23. package/src/__tests__/conversation-queue.test.ts +39 -62
  24. package/src/__tests__/conversation-routes-disk-view.test.ts +1 -1
  25. package/src/__tests__/conversation-routes-enabled-plugins.test.ts +1 -1
  26. package/src/__tests__/conversation-routes-guardian-reply.test.ts +9 -9
  27. package/src/__tests__/conversation-routes-hidden-queue.test.ts +1 -1
  28. package/src/__tests__/conversation-routes-slash-commands.test.ts +1 -1
  29. package/src/__tests__/conversation-runtime-assembly.test.ts +53 -0
  30. package/src/__tests__/conversation-slash-queue.test.ts +3 -0
  31. package/src/__tests__/conversation-surfaces-action-delivery.test.ts +1 -0
  32. package/src/__tests__/conversation-surfaces-activation-emit.test.ts +1 -0
  33. package/src/__tests__/conversation-surfaces-app-control.test.ts +1 -0
  34. package/src/__tests__/conversation-surfaces-app-open.test.ts +1 -1
  35. package/src/__tests__/conversation-surfaces-data-persist.test.ts +1 -1
  36. package/src/__tests__/conversation-surfaces-history-restored-completion.test.ts +21 -14
  37. package/src/__tests__/conversation-surfaces-queued-emit.test.ts +1 -0
  38. package/src/__tests__/conversation-surfaces-standalone-payloads.test.ts +1 -0
  39. package/src/__tests__/conversation-surfaces-standalone.test.ts +1 -0
  40. package/src/__tests__/conversation-surfaces-state-update.test.ts +1 -1
  41. package/src/__tests__/conversation-surfaces-table-action.test.ts +1 -1
  42. package/src/__tests__/conversation-surfaces-task-progress.test.ts +1 -1
  43. package/src/__tests__/conversation-tool-setup-app-refresh.test.ts +1 -1
  44. package/src/__tests__/conversation-tool-setup-attribution.test.ts +1 -1
  45. package/src/__tests__/cu-unified-flow.test.ts +1 -0
  46. package/src/__tests__/document-sync-tags.test.ts +0 -75
  47. package/src/__tests__/file-ops-service.test.ts +163 -30
  48. package/src/__tests__/filesystem-tools.test.ts +23 -24
  49. package/src/__tests__/gateway-only-guard.test.ts +2 -5
  50. package/src/__tests__/host-file-read-tool.test.ts +16 -19
  51. package/src/__tests__/http-user-message-parity.test.ts +1 -1
  52. package/src/__tests__/init-feature-flag-overrides.test.ts +49 -0
  53. package/src/__tests__/managed-skill-lifecycle.test.ts +7 -0
  54. package/src/__tests__/media-generate-image.test.ts +131 -21
  55. package/src/__tests__/memory-retrieval-hook.test.ts +94 -2
  56. package/src/__tests__/plugin-api-webhook-url.test.ts +10 -7
  57. package/src/__tests__/plugin-import-boundary-reverse-guard.test.ts +7 -3
  58. package/src/__tests__/proxy-approval-callback.test.ts +1 -0
  59. package/src/__tests__/qdrant-manager.test.ts +14 -1
  60. package/src/__tests__/run-due-schedules.test.ts +21 -0
  61. package/src/__tests__/scaffold-managed-skill-tool.test.ts +187 -18
  62. package/src/__tests__/schedule-routes.test.ts +23 -0
  63. package/src/__tests__/schedule-store.test.ts +17 -0
  64. package/src/__tests__/secret-ingress-http.test.ts +1 -1
  65. package/src/__tests__/send-endpoint-busy.test.ts +3 -3
  66. package/src/__tests__/starter-task-flow.test.ts +5 -4
  67. package/src/__tests__/subagent-fork-prompt-role.test.ts +1 -1
  68. package/src/__tests__/subagent-spawn-and-await.test.ts +4 -7
  69. package/src/__tests__/subagent-tool-gate-mode.test.ts +1 -1
  70. package/src/__tests__/subagent-tools.test.ts +81 -101
  71. package/src/__tests__/surface-completion-in-flight-snapshot.test.ts +1 -0
  72. package/src/__tests__/tool-executor.test.ts +5 -1
  73. package/src/__tests__/ui-choice-copy-surfaces.test.ts +1 -1
  74. package/src/__tests__/ui-visual-surface.test.ts +1 -1
  75. package/src/__tests__/ui-voice-picker-surface.test.ts +1 -1
  76. package/src/__tests__/ui-work-result-surface.test.ts +1 -1
  77. package/src/__tests__/voice-scoped-grant-consumer.test.ts +5 -3
  78. package/src/__tests__/voice-session-bridge.test.ts +85 -29
  79. package/src/acp/session-manager.ts +8 -1
  80. package/src/api/events/host-file.ts +2 -2
  81. package/src/api/surfaces.ts +5 -0
  82. package/src/calls/__tests__/voice-session-bridge.test.ts +21 -10
  83. package/src/calls/__tests__/voice-triage-escalate.test.ts +8 -0
  84. package/src/calls/voice-session-bridge.ts +44 -25
  85. package/src/calls/voice-triage-escalate.ts +1 -0
  86. package/src/cli/bundled-modules.ts +29 -0
  87. package/src/cli/commands/db/repair.ts +4 -8
  88. package/src/cli/commands/domain.ts +6 -3
  89. package/src/cli/commands/email.ts +6 -3
  90. package/src/cli/commands/keys.ts +8 -3
  91. package/src/cli/commands/plugins.ts +85 -36
  92. package/src/cli/commands/schedules.ts +35 -1
  93. package/src/cli/lib/bundled-marketplace.json +1 -1
  94. package/src/config/__tests__/balanced-model-experiment.test.ts +278 -0
  95. package/src/config/assistant-feature-flags.ts +36 -15
  96. package/src/config/balanced-model-experiment.ts +35 -0
  97. package/src/config/bundled-skills/image-studio/SKILL.md +5 -4
  98. package/src/config/bundled-skills/image-studio/TOOLS.json +1 -1
  99. package/src/config/bundled-skills/image-studio/tools/media-generate-image.ts +101 -0
  100. package/src/config/bundled-skills/skill-management/TOOLS.json +9 -3
  101. package/src/config/bundled-skills/subagent/SKILL.md +17 -12
  102. package/src/config/bundled-skills/subagent/TOOLS.json +4 -4
  103. package/src/config/call-site-defaults.ts +7 -0
  104. package/src/config/default-profile-catalog.ts +96 -4
  105. package/src/config/feature-flag-registry.json +11 -11
  106. package/src/config/skills.ts +9 -2
  107. package/src/daemon/__tests__/conversation-surfaces-launch.test.ts +1 -1
  108. package/src/daemon/conversation-agent-loop.ts +14 -13
  109. package/src/daemon/conversation-notifiers.ts +11 -9
  110. package/src/daemon/conversation-process.ts +0 -27
  111. package/src/daemon/conversation-runtime-assembly.ts +9 -2
  112. package/src/daemon/conversation-store.ts +4 -4
  113. package/src/daemon/conversation-surfaces.ts +27 -13
  114. package/src/daemon/conversation-tool-setup.ts +2 -4
  115. package/src/daemon/conversation.ts +104 -51
  116. package/src/daemon/doordash-steps.ts +2 -2
  117. package/src/daemon/lifecycle.ts +14 -1
  118. package/src/daemon/process-message.ts +0 -13
  119. package/src/daemon/windows-compiled-entry.ts +4 -0
  120. package/src/documents/document-store.ts +5 -235
  121. package/src/hooks/types.ts +5 -0
  122. package/src/ipc/gateway-flag-listener.ts +17 -3
  123. package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +118 -0
  124. package/src/live-voice/live-voice-manager.ts +16 -3
  125. package/src/live-voice/live-voice-session.ts +63 -9
  126. package/src/live-voice/windows-compiled-live-voice.ts +4 -0
  127. package/src/monitoring/control.ts +1 -0
  128. package/src/monitoring/db-integrity-sample.ts +4 -5
  129. package/src/notifications/AGENTS.md +2 -0
  130. package/src/notifications/approval-card-data.ts +33 -0
  131. package/src/permissions/prompter.ts +1 -5
  132. package/src/persistence/conversation-queries.ts +66 -16
  133. package/src/persistence/embeddings/qdrant-manager.ts +84 -49
  134. package/src/persistence/migrations/360-add-document-workspace-path.ts +5 -14
  135. package/src/persistence/schema/documents.ts +4 -4
  136. package/src/plugin-api/constants.ts +8 -0
  137. package/src/plugin-api/index.ts +5 -1
  138. package/src/plugin-api/webhook-url.ts +13 -11
  139. package/src/plugins/defaults/main.ts +6 -7
  140. package/src/plugins/defaults/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +77 -0
  141. package/src/plugins/defaults/memory/graph/conversation-graph-memory.ts +24 -6
  142. package/src/plugins/defaults/memory/hooks/user-prompt-submit.ts +53 -5
  143. package/src/plugins/defaults/memory/memory-retrospective-job.ts +4 -4
  144. package/src/plugins/defaults/memory/v3/__tests__/injection.test.ts +18 -0
  145. package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +66 -1
  146. package/src/plugins/defaults/memory/v3/injector.ts +8 -0
  147. package/src/plugins/defaults/memory/v3/shadow-plugin.ts +23 -9
  148. package/src/plugins/defaults/memory/worker-control.ts +1 -0
  149. package/src/plugins/defaults/worker-entrypoints.ts +3 -0
  150. package/src/plugins/mtime-cache.ts +17 -0
  151. package/src/prompts/templates/system-sections.ts +0 -7
  152. package/src/providers/__tests__/context-overflow-error.test.ts +24 -0
  153. package/src/providers/__tests__/retry-callsite.test.ts +20 -0
  154. package/src/providers/openai/chat-completions-provider.ts +11 -1
  155. package/src/providers/speech-to-text/__tests__/deepgram-flux-realtime.test.ts +11 -6
  156. package/src/providers/speech-to-text/deepgram-flux-realtime.ts +9 -49
  157. package/src/routes/control.ts +1 -0
  158. package/src/routes/route-host-client.ts +1 -0
  159. package/src/runtime/AGENTS.md +16 -17
  160. package/src/runtime/agent-wake.ts +15 -12
  161. package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +170 -1
  162. package/src/runtime/routes/__tests__/schedule-routes-disarm-reason.test.ts +215 -0
  163. package/src/runtime/routes/conversation-list-routes.ts +54 -22
  164. package/src/runtime/routes/conversation-management-routes.ts +2 -3
  165. package/src/runtime/routes/conversation-routes.ts +11 -13
  166. package/src/runtime/routes/documents-routes.ts +3 -222
  167. package/src/runtime/routes/playground/__tests__/inject-failures.test.ts +2 -0
  168. package/src/runtime/routes/playground/__tests__/reset-circuit.test.ts +3 -0
  169. package/src/runtime/routes/playground/inject-failures.ts +2 -2
  170. package/src/runtime/routes/playground/reset-circuit.ts +1 -1
  171. package/src/runtime/routes/schedule-routes.ts +94 -7
  172. package/src/runtime/routes/workspace-routes.ts +0 -9
  173. package/src/runtime/routes/workspace-utils.ts +3 -13
  174. package/src/runtime/services/conversation-serializer.ts +7 -2
  175. package/src/schedule/__tests__/plugin-schedule-declarations.test.ts +68 -5
  176. package/src/schedule/__tests__/plugin-schedule-reconciler.test.ts +81 -0
  177. package/src/schedule/plugin-schedule-availability.ts +58 -0
  178. package/src/schedule/plugin-schedule-declarations.ts +23 -27
  179. package/src/schedule/plugin-schedule-reconciler.ts +12 -3
  180. package/src/schedule/schedule-store.ts +5 -1
  181. package/src/schedule/scheduler.ts +9 -4
  182. package/src/schedule/worker-control.ts +1 -0
  183. package/src/subagent/__tests__/consult-prompt.test.ts +26 -15
  184. package/src/subagent/consult-context.ts +11 -11
  185. package/src/subagent/consult-prompt.ts +26 -35
  186. package/src/subagent/manager.ts +20 -37
  187. package/src/subagent/notify.ts +7 -1
  188. package/src/subagent/types.ts +15 -13
  189. package/src/tools/__tests__/tool-input-schemas.test.ts +7 -7
  190. package/src/tools/acp/spawn.ts +6 -4
  191. package/src/tools/filesystem/read.ts +27 -10
  192. package/src/tools/host-filesystem/read.ts +27 -15
  193. package/src/tools/shared/filesystem/file-ops-service.ts +63 -35
  194. package/src/tools/shared/filesystem/legacy-read-args.ts +22 -0
  195. package/src/tools/shared/filesystem/types.ts +5 -5
  196. package/src/tools/skills/scaffold-managed.ts +25 -7
  197. package/src/tools/subagent/spawn.ts +28 -88
  198. package/src/tools/ui-surface/surface-shape-docs.ts +1 -1
  199. package/src/util/__tests__/worker-process-command.test.ts +37 -0
  200. package/src/util/logger.ts +16 -0
  201. package/src/util/worker-process.ts +37 -4
  202. package/src/windows-compiled-cli.ts +32 -0
  203. package/src/windows-compiled-entry.ts +4 -0
  204. package/src/windows-compiled-logger.ts +6 -0
  205. package/src/windows-compiled-worker-entry.ts +29 -0
  206. package/src/__tests__/document-workspace-file.test.ts +0 -467
  207. package/src/daemon/interactive-turn-sender.ts +0 -59
  208. package/src/subagent/__tests__/consult-transcript.test.ts +0 -184
  209. package/src/subagent/consult-transcript.ts +0 -90
@@ -12,6 +12,7 @@ import {
12
12
  IMAGE_EXTENSIONS,
13
13
  readImageFile,
14
14
  } from "../shared/filesystem/image-read.js";
15
+ import { legacyReadArgsError } from "../shared/filesystem/legacy-read-args.js";
15
16
  import { sandboxReadPolicy } from "../shared/filesystem/path-policy.js";
16
17
  import {
17
18
  invalidToolInputResult,
@@ -25,8 +26,8 @@ import type {
25
26
 
26
27
  /**
27
28
  * Model-input schema, the single source for both runtime validation (via
28
- * `TOOL_INPUT_SCHEMAS`) and the advertised `input_schema` below. `offset` and
29
- * `limit` catch to `undefined` because the tool has always ignored non-numeric
29
+ * `TOOL_INPUT_SCHEMAS`) and the advertised `input_schema` below. `start_index`
30
+ * and `max_chars` catch to `undefined` because the tool ignores non-numeric
30
31
  * values rather than failing the read.
31
32
  */
32
33
  export const fileReadInputSchema = z.looseObject({
@@ -36,14 +37,16 @@ export const fileReadInputSchema = z.looseObject({
36
37
  .describe(
37
38
  "The path to the file to read (absolute or relative to working directory)",
38
39
  ),
39
- offset: z
40
+ start_index: z
40
41
  .number()
41
- .describe("Line number to start reading from (1-indexed)")
42
+ .describe("Character to start reading from (0-indexed). Text files only.")
42
43
  .optional()
43
44
  .catch(undefined),
44
- limit: z
45
+ max_chars: z
45
46
  .number()
46
- .describe("Maximum number of lines to read (defaults to 2000)")
47
+ .describe(
48
+ "Maximum number of characters to read. Defaults to 20000, which is also the ceiling. Text files only.",
49
+ )
47
50
  .optional()
48
51
  .catch(undefined),
49
52
  activity: z
@@ -58,7 +61,7 @@ export const fileReadInputSchema = z.looseObject({
58
61
  export const fileReadTool = {
59
62
  name: "file_read",
60
63
  description:
61
- "Read the contents of a file on your own machine. Text reads return the first 2000 lines unless you pass `limit`; when a read stops short the result says so, and `offset` pages on from there. For image files (JPEG, PNG, GIF, WebP), returns the image for visual analysis. For audio files (MP3, WAV, OGG, FLAC, AAC, M4A), returns the audio for listening. Use host_file_read for files on your guardian's device instead.",
64
+ "Read the contents of a file on your own machine. Text reads return the first 20000 characters unless you pass `max_chars`; when a read stops short the result says so, and `start_index` pages on from there. To find where something is in a large file, code_search is cheaper than paging through it. For image files (JPEG, PNG, GIF, WebP), returns the image for visual analysis. For audio files (MP3, WAV, OGG, FLAC, AAC, M4A), returns the audio for listening. Use host_file_read for files on your guardian's device instead.",
62
65
  category: "filesystem",
63
66
  executionTarget: "sandbox",
64
67
  defaultRiskLevel: RiskLevel.Low,
@@ -75,9 +78,14 @@ export const fileReadTool = {
75
78
  if (!parsed.success) {
76
79
  return invalidToolInputResult("file_read", parsed.error);
77
80
  }
78
- const { path: rawPath, offset, limit } = parsed.data;
81
+ const {
82
+ path: rawPath,
83
+ start_index: startIndex,
84
+ max_chars: maxChars,
85
+ } = parsed.data;
79
86
 
80
- // For image files, delegate to the shared image reader.
87
+ // For image files, delegate to the shared image reader. Media reads carry
88
+ // no window, so the legacy-argument guard below does not apply to them.
81
89
  const ext = extname(rawPath).toLowerCase();
82
90
  if (IMAGE_EXTENSIONS.has(ext)) {
83
91
  const pathCheck = sandboxReadPolicy(rawPath, context.workingDir);
@@ -102,11 +110,20 @@ export const fileReadTool = {
102
110
  return readAudioFile(pathCheck.resolved);
103
111
  }
104
112
 
113
+ const legacyArgs = legacyReadArgsError("file_read", input);
114
+ if (legacyArgs !== undefined) {
115
+ return { content: legacyArgs, isError: true };
116
+ }
117
+
105
118
  const ops = new FileSystemOps((path, opts) =>
106
119
  sandboxReadPolicy(path, context.workingDir, opts),
107
120
  );
108
121
 
109
- const result = await ops.readFileSafe({ path: rawPath, offset, limit });
122
+ const result = await ops.readFileSafe({
123
+ path: rawPath,
124
+ startIndex,
125
+ maxChars,
126
+ });
110
127
 
111
128
  if (!result.ok) {
112
129
  const { error } = result;
@@ -11,13 +11,14 @@ import {
11
11
  readAudioFile,
12
12
  } from "../shared/filesystem/audio-read.js";
13
13
  import {
14
- DEFAULT_READ_LINE_LIMIT,
15
14
  FileSystemOps,
15
+ READ_CHAR_BUDGET,
16
16
  } from "../shared/filesystem/file-ops-service.js";
17
17
  import {
18
18
  IMAGE_EXTENSIONS,
19
19
  readImageFile,
20
20
  } from "../shared/filesystem/image-read.js";
21
+ import { legacyReadArgsError } from "../shared/filesystem/legacy-read-args.js";
21
22
  import { hostPolicy } from "../shared/filesystem/path-policy.js";
22
23
  import {
23
24
  invalidToolInputResult,
@@ -32,9 +33,9 @@ import type {
32
33
  /**
33
34
  * Model-input schema, the single source for both runtime validation (via
34
35
  * `TOOL_INPUT_SCHEMAS`) and the advertised `input_schema` below — mirrors
35
- * `filesystem/read.ts`. `offset`/`limit` catch to `undefined` so a
36
- * non-numeric value falls back to the default line window instead of failing
37
- * the call; `target_client_id` catches so a non-string (or empty) value means
36
+ * `filesystem/read.ts`. `start_index`/`max_chars` catch to `undefined` so a
37
+ * non-numeric value falls back to the default character window instead of
38
+ * failing the call; `target_client_id` catches so a non-string (or empty) value means
38
39
  * "untargeted".
39
40
  */
40
41
  export const hostFileReadInputSchema = z.looseObject({
@@ -44,14 +45,16 @@ export const hostFileReadInputSchema = z.looseObject({
44
45
  .describe(
45
46
  "Absolute path on the guardian's device, which is a separate filesystem from your workspace, to read.",
46
47
  ),
47
- offset: z
48
+ start_index: z
48
49
  .number()
49
- .describe("Line number to start reading from (1-indexed)")
50
+ .describe("Character to start reading from (0-indexed). Text files only.")
50
51
  .optional()
51
52
  .catch(undefined),
52
- limit: z
53
+ max_chars: z
53
54
  .number()
54
- .describe("Maximum number of lines to read (defaults to 2000)")
55
+ .describe(
56
+ "Maximum number of characters to read. Defaults to 20000, which is also the ceiling. Text files only.",
57
+ )
55
58
  .optional()
56
59
  .catch(undefined),
57
60
  target_client_id: z
@@ -66,7 +69,7 @@ export const hostFileReadInputSchema = z.looseObject({
66
69
  export const hostFileReadTool = {
67
70
  name: "host_file_read",
68
71
  description:
69
- "Read the contents of a file on your guardian's device, including images (JPEG, PNG, GIF, WebP) and audio (MP3, WAV, OGG, FLAC, AAC, M4A). Text reads return the first 2000 lines unless you pass `limit`; when a read stops short the result says so, and `offset` pages on from there. For files on your own machine, use file_read instead.",
72
+ "Read the contents of a file on your guardian's device, including images (JPEG, PNG, GIF, WebP) and audio (MP3, WAV, OGG, FLAC, AAC, M4A). Text reads return the first 20000 characters unless you pass `max_chars`; when a read stops short the result says so, and `start_index` pages on from there. For files on your own machine, use file_read instead.",
70
73
  category: "host-filesystem",
71
74
  executionTarget: "host",
72
75
  defaultRiskLevel: RiskLevel.Medium,
@@ -81,12 +84,12 @@ export const hostFileReadTool = {
81
84
  if (!parsed.success) {
82
85
  return invalidToolInputResult("host_file_read", parsed.error);
83
86
  }
84
- const { path: rawPath, offset } = parsed.data;
87
+ const { path: rawPath, start_index: startIndex } = parsed.data;
85
88
  // Resolve the default here rather than leaving it to the read, so the
86
89
  // proxied branch below is bounded by the same window as the local one. A
87
- // proxied read that sent no limit would stream a whole host file across
90
+ // proxied read that sent no budget would stream a whole host file across
88
91
  // the bridge before anything could trim it.
89
- const limit = parsed.data.limit ?? DEFAULT_READ_LINE_LIMIT;
92
+ const maxChars = parsed.data.max_chars ?? READ_CHAR_BUDGET;
90
93
 
91
94
  const targetClientId =
92
95
  parsed.data.target_client_id !== ""
@@ -153,8 +156,8 @@ export const hostFileReadTool = {
153
156
  {
154
157
  operation: "read",
155
158
  path: rawPath,
156
- offset,
157
- limit,
159
+ startIndex,
160
+ maxChars,
158
161
  targetClientId,
159
162
  },
160
163
  context.conversationId,
@@ -181,9 +184,18 @@ export const hostFileReadTool = {
181
184
  return readAudioFile(pathCheck.resolved);
182
185
  }
183
186
 
187
+ const legacyArgs = legacyReadArgsError("host_file_read", input);
188
+ if (legacyArgs !== undefined) {
189
+ return { content: legacyArgs, isError: true };
190
+ }
191
+
184
192
  const ops = new FileSystemOps(hostPolicy);
185
193
 
186
- const result = await ops.readFileSafe({ path: rawPath, offset, limit });
194
+ const result = await ops.readFileSafe({
195
+ path: rawPath,
196
+ startIndex,
197
+ maxChars,
198
+ });
187
199
 
188
200
  if (!result.ok) {
189
201
  const { error } = result;
@@ -78,26 +78,55 @@ function pathError(
78
78
  }
79
79
 
80
80
  /**
81
- * Lines returned by a read that names no `limit`. Without a default the read
82
- * returns the whole file, and a file-read result is honored in full for the
83
- * rest of the turn (see `isSpoolEligible` in `context/tool-result-spool.ts`),
84
- * so one unbounded read of a large file rides every subsequent LLM call in
85
- * that turn. The cap bounds that; `offset`/`limit` page past it, and
86
- * {@link truncationNotice} tells the model when it is looking at a window
87
- * rather than the whole file.
81
+ * Characters returned by a read that names no `max_chars`. Stays under
82
+ * `THRESHOLD_CHARS` in `context/post-turn-tool-result-truncation.ts`, which
83
+ * spools any larger tool result to disk and replaces it inline with a short
84
+ * stub, so a default read returns content rather than a stub.
88
85
  */
89
- export const DEFAULT_READ_LINE_LIMIT = 2000;
86
+ export const READ_CHAR_BUDGET = 20_000;
90
87
 
91
88
  /**
92
- * Trailing marker appended when a read stops short of the last line. Silent
93
- * truncation is the failure mode worth avoiding: a model that cannot tell a
94
- * window from a whole file reasons about code it never saw.
89
+ * Trailing marker appended when a read stops short of the end of the file. A
90
+ * model that cannot tell a window from a whole file reasons about code it
91
+ * never saw.
95
92
  */
96
93
  function truncationNotice(
97
- lastLineReturned: number,
98
- totalLines: number,
94
+ start: number,
95
+ end: number,
96
+ totalChars: number,
99
97
  ): string {
100
- return `\n\n[Truncated: showing through line ${lastLineReturned} of ${totalLines}. Read on with offset=${lastLineReturned + 1}, or pass an explicit limit.]`;
98
+ return `\n\n[Truncated: characters ${start}-${end} of ${totalChars}. Read on with start_index=${end}.]`;
99
+ }
100
+
101
+ const isHighSurrogate = (code: number): boolean =>
102
+ code >= 0xd800 && code <= 0xdbff;
103
+ const isLowSurrogate = (code: number): boolean =>
104
+ code >= 0xdc00 && code <= 0xdfff;
105
+
106
+ /**
107
+ * Character window that never splits a surrogate pair. A split leaves a lone
108
+ * half at each edge, and each encodes to U+FFFD, so the character is lost from
109
+ * both this window and the next one paged in after it.
110
+ */
111
+ export function surrogateSafeWindow(
112
+ total: number,
113
+ charCodeAt: (index: number) => number,
114
+ requestedStart: number,
115
+ maxChars: number,
116
+ ): { start: number; end: number } {
117
+ let start = Math.max(0, Math.min(requestedStart, total));
118
+ if (start > 0 && start < total && isLowSurrogate(charCodeAt(start))) {
119
+ start -= 1;
120
+ }
121
+
122
+ let end = Math.min(total, start + maxChars);
123
+ if (end > start && end < total && isHighSurrogate(charCodeAt(end - 1))) {
124
+ // Backing off would empty a one-character window, which stalls paging on
125
+ // the same offset, so take the whole pair instead.
126
+ end = end - 1 > start ? end - 1 : Math.min(total, end + 1);
127
+ }
128
+
129
+ return { start, end };
101
130
  }
102
131
 
103
132
  export class FileSystemOps {
@@ -139,28 +168,27 @@ export class FileSystemOps {
139
168
 
140
169
  try {
141
170
  const raw = await readFile(filePath, "utf-8");
142
- const lines = raw.split("\n");
143
-
144
- const offset = (input.offset ?? 1) - 1;
145
- const start = Math.max(0, offset);
146
- const limit = input.limit ?? DEFAULT_READ_LINE_LIMIT;
147
- const selected = lines.slice(start, offset + limit);
148
-
149
- const numbered = selected
150
- .map((line, i) => {
151
- const lineNum = offset + i + 1;
152
- return `${String(lineNum).padStart(6)} ${line}`;
153
- })
154
- .join("\n");
155
-
156
- // Only when the window stops before the last line. An empty window means
157
- // the caller paged past the end or asked for nothing, which is not a
158
- // truncated read.
159
- const lastLineReturned = start + selected.length;
171
+
172
+ // A ceiling, not just a default: a larger window would be spooled to
173
+ // disk and replaced with a stub, returning less than this.
174
+ const maxChars = Math.min(
175
+ READ_CHAR_BUDGET,
176
+ Math.max(0, input.maxChars ?? READ_CHAR_BUDGET),
177
+ );
178
+ const { start, end } = surrogateSafeWindow(
179
+ raw.length,
180
+ (i) => raw.charCodeAt(i),
181
+ input.startIndex ?? 0,
182
+ maxChars,
183
+ );
184
+ const window = raw.slice(start, end);
185
+
186
+ // An empty window means the caller paged past the end or asked for
187
+ // nothing, which is not a truncated read.
160
188
  const content =
161
- selected.length > 0 && lastLineReturned < lines.length
162
- ? numbered + truncationNotice(lastLineReturned, lines.length)
163
- : numbered;
189
+ window.length > 0 && end < raw.length
190
+ ? window + truncationNotice(start, end, raw.length)
191
+ : window;
164
192
 
165
193
  return { ok: true, value: { content } };
166
194
  } catch (err) {
@@ -0,0 +1,22 @@
1
+ /**
2
+ * Guard for callers still passing the line-based `offset`/`limit` read
3
+ * arguments. The tool schemas are loose, so those keys parse and are then
4
+ * dropped, which would silently serve the start of the file to a caller that
5
+ * asked to page into the middle of it.
6
+ *
7
+ * There is no safe translation: `offset` counted lines and `start_index`
8
+ * counts characters, so mapping one onto the other would read the wrong
9
+ * region rather than fail. Naming the rename lets the caller correct itself.
10
+ */
11
+ export function legacyReadArgsError(
12
+ toolName: string,
13
+ input: Record<string, unknown>,
14
+ ): string | undefined {
15
+ const usesLegacy = input.offset !== undefined || input.limit !== undefined;
16
+ const usesCurrent =
17
+ input.start_index !== undefined || input.max_chars !== undefined;
18
+ if (!usesLegacy || usesCurrent) {
19
+ return undefined;
20
+ }
21
+ return `Error: ${toolName} no longer takes \`offset\`/\`limit\` (lines). Use \`start_index\` (0-indexed characters) and \`max_chars\` instead.`;
22
+ }
@@ -6,14 +6,14 @@ import type { FsError } from "./errors.js";
6
6
 
7
7
  export interface ReadInput {
8
8
  path: string;
9
- /** 1-indexed line number to start reading from. */
10
- offset?: number;
11
- /** Maximum number of lines to read. */
12
- limit?: number;
9
+ /** 0-indexed character to start reading from. */
10
+ startIndex?: number;
11
+ /** Maximum number of characters to read. */
12
+ maxChars?: number;
13
13
  }
14
14
 
15
15
  export interface ReadOutput {
16
- /** The (possibly line-numbered) file content. */
16
+ /** The character window, with a truncation notice when it stops short. */
17
17
  content: string;
18
18
  }
19
19
 
@@ -28,10 +28,20 @@ function sanitizeFrontmatterValue(value: string): string {
28
28
  }
29
29
 
30
30
  /**
31
- * Validate + normalize an optional string-array input (sanitize, drop blanks,
32
- * dedupe). Returns `{ error }` on the first invalid element, or `{ value }`
33
- * holding the normalized array (undefined when empty). Shared by the
34
- * includes / activation_hints / avoid_when inputs so they behave identically.
31
+ * Self-correcting error for a scaffold call with no activation hints. Matches
32
+ * the shape of the schema validator's required-field error, which is what a
33
+ * call through the registered tool hits first; this covers direct callers and
34
+ * an explicit empty list, which the schema's presence check accepts.
35
+ */
36
+ const MISSING_ACTIVATION_HINTS =
37
+ 'activation_hints is required: pass 1-4 short trigger phrases stating the intent this skill serves (for example "user asks to deploy staging") so it can be found later by intent, not just by name.';
38
+
39
+ /**
40
+ * Validate + normalize a string-array input (sanitize, drop blanks, dedupe).
41
+ * Returns `{ error }` on the first invalid element, or `{ value }` holding
42
+ * the normalized array (undefined when absent or empty). Shared by the
43
+ * includes / activation_hints / avoid_when inputs so they behave identically;
44
+ * whether an absent value is acceptable is the caller's call.
35
45
  * Each element goes through sanitizeFrontmatterValue: activation_hints /
36
46
  * avoid_when are concatenated verbatim into capability memory text (see
37
47
  * buildSkillContent), so an embedded newline could otherwise smuggle an extra
@@ -184,9 +194,9 @@ export async function executeScaffoldManagedSkill(
184
194
  };
185
195
  }
186
196
 
187
- // Validate and normalize the optional string-array inputs. `includes` lists
188
- // child skill IDs; activation_hints / avoid_when become the skill's
189
- // "Use when:" / "Avoid when:" retrieval signal in memory.
197
+ // Validate and normalize the string-array inputs. `includes` lists child
198
+ // skill IDs; activation_hints / avoid_when become the skill's "Use when:" /
199
+ // "Avoid when:" retrieval signal in memory.
190
200
  const includesResult = normalizeOptionalStringArray(
191
201
  input.includes,
192
202
  "includes",
@@ -204,6 +214,14 @@ export async function executeScaffoldManagedSkill(
204
214
  return { content: `Error: ${activationHintsResult.error}`, isError: true };
205
215
  }
206
216
  const activationHints = activationHintsResult.value;
217
+ // Hints are the skill's "Use when:" retrieval text, and scaffolding rewrites
218
+ // the whole SKILL.md, so a write without them leaves (or strips) none.
219
+ if (!activationHints) {
220
+ return {
221
+ content: `Error: ${MISSING_ACTIVATION_HINTS}`,
222
+ isError: true,
223
+ };
224
+ }
207
225
 
208
226
  const avoidWhenResult = normalizeOptionalStringArray(
209
227
  input.avoid_when,
@@ -5,20 +5,18 @@ import { validateInferenceProfileKey } from "../../config/inference-profile-vali
5
5
  import { getConfig } from "../../config/loader.js";
6
6
  import { profileSupportsTools } from "../../config/profile-tool-support.js";
7
7
  import { findConversation } from "../../daemon/conversation-registry.js";
8
- import { getMessages } from "../../persistence/conversation-crud.js";
9
8
  import {
10
9
  countRecentSimilarSpawns,
11
10
  normalizeSpawnObjective,
12
11
  type RecentSimilarSpawns,
13
12
  type SimilarSpawnTally,
14
13
  } from "../../persistence/subagent-store.js";
15
- import type { ContentBlock, Message } from "../../providers/types.js";
14
+ import type { Message } from "../../providers/types.js";
16
15
  import { buildAdvisorContext } from "../../subagent/consult-context.js";
17
16
  import {
18
17
  advisorRequestText,
19
18
  buildAdvisorSystem,
20
19
  } from "../../subagent/consult-prompt.js";
21
- import { sanitizeConsultTranscript } from "../../subagent/consult-transcript.js";
22
20
  import {
23
21
  getSubagentManager,
24
22
  SubagentAbortedError,
@@ -49,7 +47,7 @@ const log = getLogger("subagent-spawn");
49
47
  * reasoning while it works, so a fixed wall-clock ceiling would kill it
50
48
  * mid-thought; an idle window instead fires only when the consult is genuinely
51
49
  * stalled (or never starts). Generous enough to also span time-to-first-token
52
- * over a large inherited transcript.
50
+ * on a slow reasoning profile.
53
51
  */
54
52
  const ADVISOR_IDLE_TIMEOUT_MS = 60_000;
55
53
 
@@ -540,13 +538,19 @@ function inFlightGuardResult(
540
538
  // ── Advisor consult ──────────────────────────────────────────────────
541
539
 
542
540
  /**
543
- * Run the `advisor` role as a synchronous, context-inheriting, stronger-model
544
- * consult and return its guidance as the tool result.
541
+ * Run the `advisor` role as a synchronous, stronger-model consult and return
542
+ * its guidance as the tool result.
545
543
  *
546
- * Inherits the parent transcript (sanitized), frames it as advice via
547
- * `buildAdvisorSystem`, and runs on `llm.advisorProfile` (unless the caller
548
- * passed an explicit `inference_profile`) under both the advisor role allowlist
549
- * and `denySideEffectTools`, so the only tools it can reach are the first-party
544
+ * The consult sees only what it is handed: the spawning agent's own `objective`
545
+ * as a written brief, plus the situational context pack from
546
+ * `buildAdvisorContext`. Nothing of the parent conversation's transcript or
547
+ * system prompt travels with it, so a consult costs the brief rather than a
548
+ * re-prefill of the whole chat at premium rates.
549
+ *
550
+ * It is framed as advice via `buildAdvisorSystem` and runs on
551
+ * `llm.advisorProfile` (unless the caller passed an explicit
552
+ * `inference_profile`) under both the advisor role allowlist and
553
+ * `denySideEffectTools`, so the only tools it can reach are the first-party
550
554
  * built-in readers. It is bounded on two axes, because the consult holds up the
551
555
  * user-facing turn while it runs: a progress-aware deadline (an idle window,
552
556
  * `ADVISOR_IDLE_TIMEOUT_MS`, reset on every streamed token and every tool event
@@ -562,7 +566,7 @@ function inFlightGuardResult(
562
566
  async function runAdvisorConsult(args: {
563
567
  context: ToolContext;
564
568
  label: string;
565
- /** The agent's own `objective` — its framing of what it wants advised on. */
569
+ /** The agent's own `objective`: the brief the advisor advises off. */
566
570
  objective: string;
567
571
  sendToClient: (msg: AssistantEvent) => void;
568
572
  requestedOverrideProfile: string | undefined;
@@ -576,35 +580,16 @@ async function runAdvisorConsult(args: {
576
580
  let profileNote: string | undefined;
577
581
 
578
582
  try {
583
+ // The parent conversation is looked up only for its warm skill catalog, so
584
+ // an unresolvable one (e.g. evicted) costs the skills section of the pack
585
+ // and nothing else: the consult itself runs off the brief.
579
586
  const parentConversation = findConversation(context.conversationId);
580
- if (!parentConversation) {
581
- return {
582
- content:
583
- "(advisor unavailable: parent conversation could not be resolved)",
584
- isError: false,
585
- };
586
- }
587
-
588
- // Snapshot the parent's in-memory transcript and system prompt, then append
589
- // the in-flight assistant turn (the plan/text the model wrote THIS turn,
590
- // before calling the advisor). The in-memory array does not yet hold that
591
- // turn — the agent loop only writes it back to `conversation.messages` after
592
- // the turn settles — but it is already persisted to the DB (the assistant
593
- // row is finalized at `message_complete`, which fires before tool execution).
594
- // `sanitizeConsultTranscript` then strips the dangling advisor `tool_use`
595
- // off that final assistant turn so the inherited transcript is provider-safe.
596
- const parentSystemPrompt = parentConversation.getCurrentSystemPrompt();
597
- const withInFlight = appendInFlightAssistantTurn(
598
- [...parentConversation.messages],
599
- context.conversationId,
600
- );
601
- const sanitizedMessages = sanitizeConsultTranscript(withInFlight);
602
587
 
603
588
  // Situational awareness for the advisor: the parent's live tool set, the
604
589
  // full skill catalog, and its workspace. Assembled off the per-turn
605
590
  // ToolContext snapshot (trust, channel) so the personal-memory sections
606
591
  // are gated exactly like the runtime injectors. Best-effort: a null pack
607
- // just means the consult runs on transcript + system prompt alone.
592
+ // just means the consult runs on the brief alone.
608
593
  const situationalContext = await buildAdvisorContext({
609
594
  conversationId: context.conversationId,
610
595
  workingDir: context.workingDir,
@@ -614,7 +599,7 @@ async function runAdvisorConsult(args: {
614
599
  enabledPluginSet: context.enabledPluginSet,
615
600
  // The parent's warm per-turn catalog keeps the synchronous on-disk
616
601
  // catalog scan out of the consult path.
617
- skillCatalog: parentConversation.skillProjectionCache?.catalog,
602
+ skillCatalog: parentConversation?.skillProjectionCache?.catalog,
618
603
  });
619
604
 
620
605
  // Default to the stronger advisor profile when the caller did not pin one;
@@ -623,7 +608,7 @@ async function runAdvisorConsult(args: {
623
608
  let overrideProfile = requestedOverrideProfile ?? config.llm.advisorProfile;
624
609
  // The advisor carries read tools, so a profile the catalog states cannot
625
610
  // call them is handed a surface it can never use and answers from the
626
- // transcript alone. Fall back to the call site's own default and say so
611
+ // brief alone. Fall back to the call site's own default and say so
627
612
  // alongside the guidance, the way a regular spawn reports it. The check is
628
613
  // unconditional, matching the tools it protects, and only a catalog `false`
629
614
  // redirects, so a model the catalog has never heard of is left alone.
@@ -697,16 +682,15 @@ async function runAdvisorConsult(args: {
697
682
  {
698
683
  parentConversationId: context.conversationId,
699
684
  label,
700
- // Carry the agent's own objective into the consult request — the
701
- // agent states the task here, and the inherited transcript can be
702
- // thin. The situational pack rides in the model request only
703
- // (`requestText`), keeping the system prompt minimal and the
704
- // display-facing `objective` free of bulky internal context.
685
+ // The agent's own objective IS the brief the consult runs on, so it
686
+ // carries into the request verbatim. The situational pack rides in
687
+ // the model request only (`requestText`), keeping the system prompt
688
+ // minimal and the display-facing `objective` free of bulky internal
689
+ // context.
705
690
  objective: advisorRequestText(objective),
706
691
  requestText: advisorRequestText(objective, situationalContext),
707
692
  sendResultToUser: false,
708
693
  role: "advisor",
709
- fork: true,
710
694
  // The advisor's read-only guarantee cannot rest on tool NAMES. A
711
695
  // workspace tool may register under `file_read` (registerWorkspaceTools
712
696
  // stashes the built-in and installs its own implementation), and a
@@ -718,10 +702,9 @@ async function runAdvisorConsult(args: {
718
702
  //
719
703
  // The advisor is a ROLE, not an `LLMCallSiteEnum` value, so its usage
720
704
  // lands under `subagentSpawn` like any other subagent. This is what
721
- // makes advisor consults separable from regular forks in telemetry.
705
+ // makes advisor consults separable from regular spawns in telemetry.
722
706
  spawnMode: "advisor_consult",
723
- parentMessages: sanitizedMessages,
724
- systemPromptOverride: buildAdvisorSystem(parentSystemPrompt),
707
+ systemPromptOverride: buildAdvisorSystem(),
725
708
  ...(overrideProfile ? { overrideProfile } : {}),
726
709
  ...(forceOverrideProfile ? { forceOverrideProfile: true } : {}),
727
710
  // A consult is delegated work of the invoking turn, so its spend
@@ -817,46 +800,3 @@ function withAdvisorNote(
817
800
  }
818
801
  return `${guidance}\n\n${present.map((note) => `_(${note})_`).join("\n")}`;
819
802
  }
820
-
821
- /**
822
- * Append the in-flight assistant turn (persisted this turn before the advisor
823
- * tool ran) to an in-memory message snapshot, unless the snapshot already ends
824
- * with it. The latest persisted assistant row carries the plan/text the model
825
- * wrote immediately before calling the advisor plus the dangling advisor
826
- * `tool_use`; `sanitizeConsultTranscript` strips the dangling call.
827
- *
828
- * Best-effort: a malformed or missing row leaves the snapshot unchanged so the
829
- * consult still runs over the in-memory history.
830
- */
831
- function appendInFlightAssistantTurn(
832
- messages: Message[],
833
- conversationId: string,
834
- ): Message[] {
835
- // When the snapshot already ends on an assistant turn, the in-flight turn is
836
- // present (or there is none to add) — appending the latest row would duplicate it.
837
- if (messages[messages.length - 1]?.role === "assistant") {
838
- return messages;
839
- }
840
-
841
- let rows;
842
- try {
843
- rows = getMessages(conversationId);
844
- } catch {
845
- return messages;
846
- }
847
- if (!rows || rows.length === 0) {
848
- return messages;
849
- }
850
-
851
- const lastRow = rows[rows.length - 1];
852
- if (lastRow.role !== "assistant") {
853
- return messages;
854
- }
855
-
856
- const blocks: ContentBlock[] = lastRow.content;
857
-
858
- if (blocks.length === 0) {
859
- return messages;
860
- }
861
- return [...messages, { role: "assistant", content: blocks }];
862
- }
@@ -123,7 +123,7 @@ export const SURFACE_SHAPE_DOCS: Record<string, SurfaceShapeDoc> = {
123
123
  work_result: {
124
124
  purpose: "structured receipt after completed work",
125
125
  shape:
126
- '{ eyebrow?, status?: "completed"|"partial"|"failed"|"in_progress", summary?, metrics?: [{ label, value, detail?, tone?: "neutral"|"positive"|"warning"|"negative" }], sections?: [{ id?, title, description?, type?: "items"|"timeline"|"diff"|"artifacts"|"warnings", items?: [{ id?, title, description?, status?, tone?, metadata?: [{ label, value }], href? }], diffs?: [{ label?, before?, after? }] }] } — structured receipt after real work; keep display-only unless follow-up buttons are needed',
126
+ '{ eyebrow?, status?: "completed"|"partial"|"failed"|"in_progress", summary?, metrics?: [{ label, value, detail?, tone?: "neutral"|"positive"|"warning"|"negative" }], sections?: [{ id?, title, description?, type?: "items"|"timeline"|"diff"|"artifacts"|"warnings", items?: [{ id?, title, description?, status?, tone?, metadata?: [{ label, value }], href? }], diffs?: [{ label?, before?, after? }] }] }: structured receipt after real work; keep display-only unless follow-up buttons are needed. An item `href` makes the row a link: an in-app path (e.g. "/assistant/skills/<skillId>?tab=history") opens in place, an https URL opens externally; other schemes are ignored',
127
127
  missingContent: (data) =>
128
128
  hasContent(data)
129
129
  ? null