@vellumai/assistant 0.11.4-staging.2 → 0.11.4-staging.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/AGENTS.md +8 -2
  2. package/ARCHITECTURE.md +2 -0
  3. package/docs/architecture/memory.md +15 -0
  4. package/docs/browser-use-architecture-phase2.md +128 -56
  5. package/docs/flux-turn-detection-spike.md +11 -6
  6. package/knip.json +3 -0
  7. package/openapi.yaml +136 -76
  8. package/package.json +1 -1
  9. package/scripts/write-plugin-api-shim.ts +10 -0
  10. package/src/__tests__/app-control-flow.test.ts +1 -0
  11. package/src/__tests__/approval-routes-http.test.ts +2 -2
  12. package/src/__tests__/assistant-feature-flag-guard.test.ts +25 -3
  13. package/src/__tests__/channel-setup-panel-ack.test.ts +1 -1
  14. package/src/__tests__/compaction-events.test.ts +8 -10
  15. package/src/__tests__/conversation-agent-loop.test.ts +4 -1
  16. package/src/__tests__/conversation-confirmation-signals.test.ts +112 -0
  17. package/src/__tests__/conversation-notifiers-provenance.test.ts +1 -1
  18. package/src/__tests__/conversation-queue.test.ts +39 -62
  19. package/src/__tests__/conversation-routes-disk-view.test.ts +1 -1
  20. package/src/__tests__/conversation-routes-enabled-plugins.test.ts +1 -1
  21. package/src/__tests__/conversation-routes-guardian-reply.test.ts +9 -9
  22. package/src/__tests__/conversation-routes-hidden-queue.test.ts +1 -1
  23. package/src/__tests__/conversation-routes-slash-commands.test.ts +1 -1
  24. package/src/__tests__/conversation-slash-queue.test.ts +3 -0
  25. package/src/__tests__/conversation-surfaces-action-delivery.test.ts +1 -0
  26. package/src/__tests__/conversation-surfaces-activation-emit.test.ts +1 -0
  27. package/src/__tests__/conversation-surfaces-app-control.test.ts +1 -0
  28. package/src/__tests__/conversation-surfaces-app-open.test.ts +1 -1
  29. package/src/__tests__/conversation-surfaces-data-persist.test.ts +1 -1
  30. package/src/__tests__/conversation-surfaces-history-restored-completion.test.ts +21 -14
  31. package/src/__tests__/conversation-surfaces-queued-emit.test.ts +1 -0
  32. package/src/__tests__/conversation-surfaces-standalone-payloads.test.ts +1 -0
  33. package/src/__tests__/conversation-surfaces-standalone.test.ts +1 -0
  34. package/src/__tests__/conversation-surfaces-state-update.test.ts +1 -1
  35. package/src/__tests__/conversation-surfaces-table-action.test.ts +1 -1
  36. package/src/__tests__/conversation-surfaces-task-progress.test.ts +1 -1
  37. package/src/__tests__/conversation-tool-setup-app-refresh.test.ts +1 -1
  38. package/src/__tests__/conversation-tool-setup-attribution.test.ts +1 -1
  39. package/src/__tests__/cu-unified-flow.test.ts +1 -0
  40. package/src/__tests__/document-sync-tags.test.ts +0 -75
  41. package/src/__tests__/gateway-only-guard.test.ts +2 -5
  42. package/src/__tests__/http-user-message-parity.test.ts +1 -1
  43. package/src/__tests__/init-feature-flag-overrides.test.ts +49 -0
  44. package/src/__tests__/managed-skill-lifecycle.test.ts +7 -0
  45. package/src/__tests__/media-generate-image.test.ts +131 -21
  46. package/src/__tests__/memory-retrieval-hook.test.ts +94 -2
  47. package/src/__tests__/plugin-api-webhook-url.test.ts +10 -7
  48. package/src/__tests__/plugin-import-boundary-reverse-guard.test.ts +7 -3
  49. package/src/__tests__/proxy-approval-callback.test.ts +1 -0
  50. package/src/__tests__/qdrant-manager.test.ts +14 -1
  51. package/src/__tests__/run-due-schedules.test.ts +21 -0
  52. package/src/__tests__/scaffold-managed-skill-tool.test.ts +187 -18
  53. package/src/__tests__/schedule-routes.test.ts +23 -0
  54. package/src/__tests__/schedule-store.test.ts +17 -0
  55. package/src/__tests__/secret-ingress-http.test.ts +1 -1
  56. package/src/__tests__/send-endpoint-busy.test.ts +3 -3
  57. package/src/__tests__/starter-task-flow.test.ts +5 -4
  58. package/src/__tests__/subagent-fork-prompt-role.test.ts +1 -1
  59. package/src/__tests__/subagent-spawn-and-await.test.ts +4 -7
  60. package/src/__tests__/subagent-tool-gate-mode.test.ts +1 -1
  61. package/src/__tests__/subagent-tools.test.ts +81 -101
  62. package/src/__tests__/surface-completion-in-flight-snapshot.test.ts +1 -0
  63. package/src/__tests__/ui-choice-copy-surfaces.test.ts +1 -1
  64. package/src/__tests__/ui-visual-surface.test.ts +1 -1
  65. package/src/__tests__/ui-voice-picker-surface.test.ts +1 -1
  66. package/src/__tests__/ui-work-result-surface.test.ts +1 -1
  67. package/src/__tests__/voice-scoped-grant-consumer.test.ts +5 -3
  68. package/src/__tests__/voice-session-bridge.test.ts +85 -29
  69. package/src/acp/session-manager.ts +8 -1
  70. package/src/api/surfaces.ts +5 -0
  71. package/src/calls/__tests__/voice-session-bridge.test.ts +21 -10
  72. package/src/calls/__tests__/voice-triage-escalate.test.ts +8 -0
  73. package/src/calls/voice-session-bridge.ts +44 -25
  74. package/src/calls/voice-triage-escalate.ts +1 -0
  75. package/src/cli/bundled-modules.ts +29 -0
  76. package/src/cli/commands/db/repair.ts +4 -8
  77. package/src/cli/commands/domain.ts +6 -3
  78. package/src/cli/commands/email.ts +6 -3
  79. package/src/cli/commands/keys.ts +8 -3
  80. package/src/cli/commands/plugins.ts +85 -36
  81. package/src/cli/commands/schedules.ts +35 -1
  82. package/src/cli/lib/bundled-marketplace.json +1 -1
  83. package/src/config/__tests__/balanced-model-experiment.test.ts +278 -0
  84. package/src/config/assistant-feature-flags.ts +36 -15
  85. package/src/config/balanced-model-experiment.ts +35 -0
  86. package/src/config/bundled-skills/image-studio/SKILL.md +5 -4
  87. package/src/config/bundled-skills/image-studio/TOOLS.json +1 -1
  88. package/src/config/bundled-skills/image-studio/tools/media-generate-image.ts +101 -0
  89. package/src/config/bundled-skills/skill-management/TOOLS.json +9 -3
  90. package/src/config/bundled-skills/subagent/SKILL.md +17 -12
  91. package/src/config/bundled-skills/subagent/TOOLS.json +4 -4
  92. package/src/config/call-site-defaults.ts +7 -0
  93. package/src/config/default-profile-catalog.ts +96 -4
  94. package/src/config/feature-flag-registry.json +11 -11
  95. package/src/config/skills.ts +9 -2
  96. package/src/daemon/__tests__/conversation-surfaces-launch.test.ts +1 -1
  97. package/src/daemon/conversation-agent-loop.ts +14 -13
  98. package/src/daemon/conversation-notifiers.ts +11 -9
  99. package/src/daemon/conversation-process.ts +0 -27
  100. package/src/daemon/conversation-store.ts +4 -4
  101. package/src/daemon/conversation-surfaces.ts +27 -13
  102. package/src/daemon/conversation-tool-setup.ts +2 -4
  103. package/src/daemon/conversation.ts +81 -44
  104. package/src/daemon/doordash-steps.ts +2 -2
  105. package/src/daemon/lifecycle.ts +14 -1
  106. package/src/daemon/process-message.ts +0 -13
  107. package/src/daemon/windows-compiled-entry.ts +4 -0
  108. package/src/documents/document-store.ts +5 -235
  109. package/src/hooks/types.ts +5 -0
  110. package/src/ipc/gateway-flag-listener.ts +17 -3
  111. package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +118 -0
  112. package/src/live-voice/live-voice-manager.ts +16 -3
  113. package/src/live-voice/live-voice-session.ts +63 -9
  114. package/src/live-voice/windows-compiled-live-voice.ts +4 -0
  115. package/src/monitoring/control.ts +1 -0
  116. package/src/monitoring/db-integrity-sample.ts +4 -5
  117. package/src/permissions/prompter.ts +1 -5
  118. package/src/persistence/conversation-queries.ts +66 -16
  119. package/src/persistence/embeddings/qdrant-manager.ts +84 -49
  120. package/src/persistence/migrations/360-add-document-workspace-path.ts +5 -14
  121. package/src/persistence/schema/documents.ts +4 -4
  122. package/src/plugin-api/constants.ts +8 -0
  123. package/src/plugin-api/index.ts +5 -1
  124. package/src/plugin-api/webhook-url.ts +13 -11
  125. package/src/plugins/defaults/main.ts +6 -7
  126. package/src/plugins/defaults/memory/graph/__tests__/conversation-graph-memory-v2-routing.test.ts +77 -0
  127. package/src/plugins/defaults/memory/graph/conversation-graph-memory.ts +24 -6
  128. package/src/plugins/defaults/memory/hooks/user-prompt-submit.ts +53 -5
  129. package/src/plugins/defaults/memory/memory-retrospective-job.ts +4 -4
  130. package/src/plugins/defaults/memory/v3/__tests__/injection.test.ts +18 -0
  131. package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +66 -1
  132. package/src/plugins/defaults/memory/v3/injector.ts +8 -0
  133. package/src/plugins/defaults/memory/v3/shadow-plugin.ts +23 -9
  134. package/src/plugins/defaults/memory/worker-control.ts +1 -0
  135. package/src/plugins/defaults/worker-entrypoints.ts +3 -0
  136. package/src/plugins/mtime-cache.ts +17 -0
  137. package/src/prompts/templates/system-sections.ts +0 -7
  138. package/src/providers/__tests__/context-overflow-error.test.ts +24 -0
  139. package/src/providers/__tests__/retry-callsite.test.ts +20 -0
  140. package/src/providers/openai/chat-completions-provider.ts +11 -1
  141. package/src/providers/speech-to-text/__tests__/deepgram-flux-realtime.test.ts +11 -6
  142. package/src/providers/speech-to-text/deepgram-flux-realtime.ts +9 -49
  143. package/src/routes/control.ts +1 -0
  144. package/src/routes/route-host-client.ts +1 -0
  145. package/src/runtime/AGENTS.md +16 -17
  146. package/src/runtime/agent-wake.ts +15 -12
  147. package/src/runtime/routes/__tests__/conversation-list-routes.test.ts +170 -1
  148. package/src/runtime/routes/__tests__/schedule-routes-disarm-reason.test.ts +215 -0
  149. package/src/runtime/routes/conversation-list-routes.ts +54 -22
  150. package/src/runtime/routes/conversation-management-routes.ts +2 -3
  151. package/src/runtime/routes/conversation-routes.ts +11 -13
  152. package/src/runtime/routes/documents-routes.ts +3 -222
  153. package/src/runtime/routes/playground/__tests__/inject-failures.test.ts +2 -0
  154. package/src/runtime/routes/playground/__tests__/reset-circuit.test.ts +3 -0
  155. package/src/runtime/routes/playground/inject-failures.ts +2 -2
  156. package/src/runtime/routes/playground/reset-circuit.ts +1 -1
  157. package/src/runtime/routes/schedule-routes.ts +94 -7
  158. package/src/runtime/routes/workspace-routes.ts +0 -9
  159. package/src/runtime/routes/workspace-utils.ts +3 -13
  160. package/src/runtime/services/conversation-serializer.ts +7 -2
  161. package/src/schedule/__tests__/plugin-schedule-declarations.test.ts +68 -5
  162. package/src/schedule/__tests__/plugin-schedule-reconciler.test.ts +81 -0
  163. package/src/schedule/plugin-schedule-availability.ts +58 -0
  164. package/src/schedule/plugin-schedule-declarations.ts +23 -27
  165. package/src/schedule/plugin-schedule-reconciler.ts +12 -3
  166. package/src/schedule/schedule-store.ts +5 -1
  167. package/src/schedule/scheduler.ts +9 -4
  168. package/src/schedule/worker-control.ts +1 -0
  169. package/src/subagent/__tests__/consult-prompt.test.ts +26 -15
  170. package/src/subagent/consult-context.ts +11 -11
  171. package/src/subagent/consult-prompt.ts +26 -35
  172. package/src/subagent/manager.ts +20 -37
  173. package/src/subagent/notify.ts +7 -1
  174. package/src/subagent/types.ts +8 -6
  175. package/src/tools/acp/spawn.ts +6 -4
  176. package/src/tools/skills/scaffold-managed.ts +25 -7
  177. package/src/tools/subagent/spawn.ts +28 -88
  178. package/src/tools/ui-surface/surface-shape-docs.ts +1 -1
  179. package/src/util/__tests__/worker-process-command.test.ts +37 -0
  180. package/src/util/logger.ts +16 -0
  181. package/src/util/worker-process.ts +37 -4
  182. package/src/windows-compiled-cli.ts +32 -0
  183. package/src/windows-compiled-entry.ts +4 -0
  184. package/src/windows-compiled-logger.ts +6 -0
  185. package/src/windows-compiled-worker-entry.ts +29 -0
  186. package/src/__tests__/document-workspace-file.test.ts +0 -467
  187. package/src/daemon/interactive-turn-sender.ts +0 -59
  188. package/src/subagent/__tests__/consult-transcript.test.ts +0 -184
  189. package/src/subagent/consult-transcript.ts +0 -90
@@ -69,10 +69,11 @@ Edits on large photos are slow (1 to 2 minutes). If the tool reports a timeout (
69
69
 
70
70
  ## Output handling
71
71
 
72
- Images return as inline content blocks in the tool result; they are not written to disk automatically.
72
+ Each generated image is saved into the workspace under `media/generated/` and the tool result lists the saved paths. The images also come back as inline content blocks so you can judge the result before presenting it.
73
73
 
74
- - If the user just wants to see the image, the inline result is enough.
75
- - If the user wants a file or you need to iterate on the result, save it to disk and deliver it through the conversation's attachment mechanism.
74
+ - Present an image to the user by embedding its saved path in your reply: `![short description](vellum://workspace/media/generated/<file>.png)`. The app renders it inline where your text refers to it, and chat channels (Slack, Telegram, WhatsApp) deliver it as a native image upload.
75
+ - If you do not embed it, the image is still auto-attached to your reply as a file, so it is never lost. Prefer embedding: an attachment chip at the end of the message is a worse presentation than the image inline.
76
+ - To iterate on a result, pass its saved path via `source_paths` with `mode: "edit"`.
76
77
 
77
78
  ## Error handling
78
79
 
@@ -88,4 +89,4 @@ Do NOT rephrase the prompt and retry on the same model, even if the error sugges
88
89
 
89
90
  ## Complete when
90
91
 
91
- The tool has returned at least one image and the user can see it: either the inline result in chat or an attached saved file. An error report counts as complete only after the retry path in Error handling has been exhausted.
92
+ The tool has returned at least one image and your reply presents it to the user, preferably as an inline `![description](vellum://workspace/...)` embed of the saved path. An error report counts as complete only after the retry path in Error handling has been exhausted.
@@ -3,7 +3,7 @@
3
3
  "tools": [
4
4
  {
5
5
  "name": "media_generate_image",
6
- "description": "Generate or edit images using AI. Supports text-to-image generation and image editing with source images.",
6
+ "description": "Generate or edit images using AI. Supports text-to-image generation and image editing with source images. Saves results into the workspace and returns their file paths; present a result to the user by embedding its path in your reply as ![description](vellum://workspace/<path>).",
7
7
  "category": "media",
8
8
  "risk": "low",
9
9
  "input_schema": {
@@ -1,3 +1,6 @@
1
+ import { mkdirSync, writeFileSync } from "fs";
2
+ import { dirname, join } from "path";
3
+
1
4
  import {
2
5
  resolveImageGenCredentials,
3
6
  resolveImageGenRouting,
@@ -19,6 +22,87 @@ import type {
19
22
  } from "../../../../tools/types.js";
20
23
  import { getConfig } from "../../../loader.js";
21
24
 
25
+ /** Workspace-relative directory where generated images are saved. */
26
+ const GENERATED_MEDIA_DIR = "media/generated";
27
+
28
+ /**
29
+ * Derive a filesystem-safe base name for a generated image from its title
30
+ * (when the provider returns one) or the generation prompt.
31
+ */
32
+ function imageFileSlug(title: string | undefined, prompt: string): string {
33
+ const base = (title?.trim() || prompt)
34
+ .toLowerCase()
35
+ .replace(/[^a-z0-9]+/g, "-")
36
+ .replace(/^-+/, "")
37
+ .slice(0, 48)
38
+ .replace(/-+$/, "");
39
+ return base || "image";
40
+ }
41
+
42
+ /** Upper bound on filename-collision retries per image. */
43
+ const MAX_FILENAME_ATTEMPTS = 1000;
44
+
45
+ /**
46
+ * Save generated images under `media/generated/` in the workspace so the
47
+ * model can reference them by path (inline embeds, edit-mode iteration).
48
+ * Each target path is validated through `sandboxPolicy` so a symlinked
49
+ * directory cannot redirect the write outside the workspace, and files are
50
+ * created exclusively (`wx`) so concurrent generations cannot claim the
51
+ * same filename. Returns workspace-relative paths for the images written
52
+ * before any failure; a failure stops the loop and is reported, not
53
+ * thrown, so the inline content blocks still reach the model.
54
+ */
55
+ function saveGeneratedImages(
56
+ images: Array<{ mimeType: string; dataBase64: string; title?: string }>,
57
+ prompt: string,
58
+ workingDir: string,
59
+ ): { savedPaths: string[]; saveError?: string } {
60
+ const savedPaths: string[] = [];
61
+ try {
62
+ for (const img of images) {
63
+ const ext = img.mimeType.split("/")[1] ?? "png";
64
+ const slug = imageFileSlug(img.title, prompt);
65
+ let written = false;
66
+ for (let attempt = 1; attempt <= MAX_FILENAME_ATTEMPTS; attempt++) {
67
+ const relPath =
68
+ attempt === 1
69
+ ? `${GENERATED_MEDIA_DIR}/${slug}.${ext}`
70
+ : `${GENERATED_MEDIA_DIR}/${slug}-${attempt}.${ext}`;
71
+ const pathCheck = sandboxPolicy(join(workingDir, relPath), workingDir, {
72
+ mustExist: false,
73
+ });
74
+ if (!pathCheck.ok) {
75
+ throw new Error(pathCheck.error);
76
+ }
77
+ mkdirSync(dirname(pathCheck.resolved), { recursive: true });
78
+ try {
79
+ writeFileSync(
80
+ pathCheck.resolved,
81
+ Buffer.from(img.dataBase64, "base64"),
82
+ { flag: "wx" },
83
+ );
84
+ } catch (error) {
85
+ if ((error as NodeJS.ErrnoException).code === "EEXIST") {
86
+ continue;
87
+ }
88
+ throw error;
89
+ }
90
+ savedPaths.push(relPath);
91
+ written = true;
92
+ break;
93
+ }
94
+ if (!written) {
95
+ throw new Error(
96
+ `Could not find a free filename for "${slug}.${ext}" after ${MAX_FILENAME_ATTEMPTS} attempts.`,
97
+ );
98
+ }
99
+ }
100
+ } catch (error) {
101
+ return { savedPaths, saveError: (error as Error).message };
102
+ }
103
+ return { savedPaths };
104
+ }
105
+
22
106
  export async function run(
23
107
  input: Record<string, unknown>,
24
108
  context: ToolContext,
@@ -127,7 +211,24 @@ export async function run(
127
211
  });
128
212
 
129
213
  const imageCount = result.images.length;
214
+ const { savedPaths, saveError } = saveGeneratedImages(
215
+ result.images,
216
+ prompt,
217
+ context.workingDir,
218
+ );
219
+
130
220
  let content = `Generated ${imageCount} image${imageCount !== 1 ? "s" : ""} using ${result.resolvedModel}.`;
221
+ if (savedPaths.length === 1) {
222
+ content += ` Saved to ${savedPaths[0]}.`;
223
+ } else if (savedPaths.length > 1) {
224
+ content += ` Saved to:\n${savedPaths.map((p) => `- ${p}`).join("\n")}`;
225
+ }
226
+ if (savedPaths.length > 0) {
227
+ content += `\n\nShow the user an image by embedding it in your reply: ![description](vellum://workspace/${savedPaths[0]}). To iterate on a result, pass its saved path via source_paths with mode "edit".`;
228
+ }
229
+ if (saveError) {
230
+ content += `\n\nCould not save to the workspace (${saveError}); the image${imageCount !== 1 ? "s" : ""} will be attached to your reply automatically instead.`;
231
+ }
131
232
  if (result.text) {
132
233
  content += `\n\n${result.text}`;
133
234
  }
@@ -40,12 +40,12 @@
40
40
  "includes": {
41
41
  "type": "array",
42
42
  "items": { "type": "string" },
43
- "description": "Optional list of child skill IDs that this skill includes (metadata only, no auto-activation)."
43
+ "description": "Optional list of child skill IDs this skill composes. When this skill is loaded via skill_load, each child's body is inlined after the parent's and the child's tools are projected, so the parent can rely on the child's procedure without re-stating it. Does not affect which skills get selected for a turn."
44
44
  },
45
45
  "activation_hints": {
46
46
  "type": "array",
47
47
  "items": { "type": "string" },
48
- "description": "Optional trigger phrases describing the situations where this skill should activate, phrased as the observed intent (e.g. \"user asks to deploy staging\"). Surfaced in memory as a \"Use when: …\" clause so the skill is retrievable by intent, not just by name."
48
+ "description": "Required. 1-4 short trigger phrases describing the situations where this skill should activate, phrased as the observed intent (e.g. \"user asks to deploy staging\"). Surfaced in memory as a \"Use when: …\" clause so the skill is retrievable by intent, not just by name. Scaffolding rewrites the whole SKILL.md, so an overwrite must pass the hints the skill should keep, revised if the procedure changed."
49
49
  },
50
50
  "avoid_when": {
51
51
  "type": "array",
@@ -79,7 +79,13 @@
79
79
  "description": "Deprecated no-op compatibility field. Skills are discovered from top-level SKILL.md files."
80
80
  }
81
81
  },
82
- "required": ["skill_id", "name", "description", "body_markdown"]
82
+ "required": [
83
+ "skill_id",
84
+ "name",
85
+ "description",
86
+ "body_markdown",
87
+ "activation_hints"
88
+ ]
83
89
  },
84
90
  "executor": "tools/scaffold-managed.ts",
85
91
  "execution_target": "host"
@@ -12,7 +12,7 @@ metadata:
12
12
  - "Delegate a self-contained research or implementation task off the main thread"
13
13
  - "Multiple agents at once, or a context-inheriting fork"
14
14
  avoid-when:
15
- - "Task is small enough to do inline (single tool call, quick lookup)"
15
+ - "Task is small enough to do inline (a handful of tool calls, quick lookups or reads)"
16
16
  - "User wants Claude Code or Codex — use the acp skill instead"
17
17
  ---
18
18
 
@@ -37,7 +37,7 @@ There are three subagent types. Pick one with two questions: **does it need to c
37
37
  |---|---|---|---|---|
38
38
  | `researcher` | No | No | `web_search`, `web_fetch`, `file_read`, `file_list`, `code_search`, `recall`, `skill_execute`, `notify_parent` | Web research, codebase exploration, reading documentation, root-cause investigation, reviewing an approach against the code |
39
39
  | `builder` | Yes | No | Your whole tool surface, unrestricted: shell, file writes and edits, and every connector, MCP, and browser tool you can reach | Code changes, file output, build/test runs, anything that must run a command or act on an outside system |
40
- | `advisor` | No | Yes | Read-only fact checking in the workspace: `file_read`, `file_list`, `code_search` | Read-only senior-advisor consult. Runs on a stronger model, inherits full parent context, and BLOCKS until it returns guidance |
40
+ | `advisor` | No | Yes | Read-only fact checking in the workspace: `file_read`, `file_list`, `code_search` | Read-only senior-advisor consult. Reads the brief you write in `objective`, runs on a stronger model, and BLOCKS until it returns guidance |
41
41
 
42
42
  Both background types can call `notify_parent` for mid-run communication with the parent.
43
43
 
@@ -65,17 +65,23 @@ The other contracts: `output_contract: "artifact"` tells a `builder` that the de
65
65
 
66
66
  ## Consulting the Advisor
67
67
 
68
- The `advisor` is the one type you spawn on your own judgment, unprompted: you do NOT wait for the user to ask for a subagent. The background types (`researcher`, `builder`) stay delegation-driven: reach for them to offload work, typically when the user's request calls for it. The advisor is different: proactively consult it whenever the conditions below are met.
68
+ The `advisor` is the one type you may spawn on your own judgment, unprompted: you do not wait for the user to ask for a subagent. The background types (`researcher`, `builder`) stay delegation-driven: reach for them to offload work, typically when the user's request calls for it.
69
69
 
70
- Orient yourself first (read the relevant files, understand the task), then consult the advisor:
70
+ A consult is expensive (a stronger model reviews your brief and answers), so reserve it for moments where a second perspective can genuinely change the outcome. Most tasks need no consult at all: a routine task with an obvious approach does not require sign-off, before you start or after you finish. Orient yourself first (read the relevant files, understand the task), then consult the advisor:
71
71
 
72
- - **Before you commit to an approach and start building** — to shape a plan when you don't have one, or to pressure-test and sharpen a plan you've already drafted.
72
+ - **Before you commit to an approach on a consequential or ambiguous task**: the design space is wide, a wrong approach would be costly to unwind, or requirements pull against each other.
73
73
  - **When you get stuck or are weighing a change in direction.**
74
- - **Once before you declare the task done.**
75
74
 
76
- The consult is synchronous and read-only: spawning an `advisor` subagent BLOCKS until it returns guidance. It runs on a stronger model and inherits your full context, so it sees the task, your tool calls, and their results without you re-explaining. It also receives a snapshot of your environment (the tools available to you this turn, the full skill catalog, and your workspace) so its guidance can point you at existing platform capabilities by name. Give its guidance serious weight; only override it when primary-source evidence contradicts a specific claim, and say so when you do.
75
+ The consult is synchronous and read-only: spawning an `advisor` subagent BLOCKS until it returns guidance. It runs on a stronger model, and it sees ONLY the brief you write in `objective` plus a snapshot of your environment (the tools available to you this turn, the full skill catalog, and your workspace). It cannot read this conversation, so the quality of its guidance tracks the quality of your brief. Write a substantive one:
77
76
 
78
- The advisor has read-only workspace tools (`file_read`, `file_list`, `code_search`) so it can open a file or search the code when a decisive fact would change its advice. It uses them sparingly, for verification rather than exploration, and it cannot change anything or persist output. It has no memory search and cannot see other conversations or external systems, and every lookup it has to make delays your answer, so surface the evidence you already have (a file's contents, a command's output, results gathered elsewhere) in the conversation or the spawn objective before consulting.
77
+ - The task or goal, stated in full.
78
+ - Your plan, or the options you are weighing against each other.
79
+ - The key evidence you already have: file paths, command output, results, decisions already made.
80
+ - The specific question you want answered.
81
+
82
+ The environment snapshot is what lets its guidance point you at existing platform capabilities by name. Give its guidance serious weight; only override it when primary-source evidence contradicts a specific claim, and say so when you do.
83
+
84
+ The advisor has read-only workspace tools (`file_read`, `file_list`, `code_search`) so it can open a file or search the code when a decisive fact would change its advice. It uses them sparingly, for verification rather than exploration, and it cannot change anything or persist output. It has no memory search and cannot see other conversations or external systems, and every lookup it has to make delays your answer, so put the evidence you already have (a file's contents, a command's output, results gathered elsewhere) into the objective rather than making it go find them.
79
85
 
80
86
  Spawn the advisor **alone** — do NOT batch the consult in the same turn as other tool calls (especially file edits, shell commands, or anything destructive or expensive). Tool calls you issue in the same turn run concurrently with the consult, so they would execute before you see its guidance. Consult the advisor by itself, read its guidance, then act.
81
87
 
@@ -151,7 +157,6 @@ Rule of thumb: "Does this task need to know what we've been talking about?" If y
151
157
  - Use `notify_parent` for interim findings instead of waiting for completion. This lets the parent act on partial results early.
152
158
  - Use `subagent_message` to send follow-up instructions to a running subagent.
153
159
  - Use `subagent_abort` to cancel a subagent that is no longer needed.
154
- - Default to spawning subagents for any task that involves web research, multi-file exploration, or independent coding work. Serial execution should be the exception, not the rule.
155
- - Delegate root-cause investigations ("why is X happening?", debugging, log forensics) to a `researcher` instead of grepping inline. A long investigation done inline floods your own context with file slices and grep output, crowding out the conversation; the researcher does the digging in its own context and returns a compact root-cause report.
156
- - When a user request has both an information-gathering component and an action component, spawn a researcher immediately rather than doing the research inline yourself.
157
- - Prefer spawning 2-3 focused subagents over one broad one. Smaller scopes finish faster and fail more gracefully.
160
+ - Spawn a subagent when the work is extensive: a sweep across a large codebase, deep research across many sources, or an investigation whose raw output (file slices, grep output, logs) would flood your context. Do quick work inline -- a few file reads or searches, an ordinary web lookup -- since a spawn pays for a whole fresh context and is slower than just doing the work.
161
+ - Delegate long root-cause investigations (log forensics, multi-file "why is X happening?" digs) to a `researcher`: it does the digging in its own context and returns a compact root-cause report, instead of crowding your conversation with intermediate output.
162
+ - Scale the fan-out to the task. Most tasks need zero or one subagent. Split work across multiple subagents only when the parts are genuinely independent and each is substantial on its own.
@@ -3,7 +3,7 @@
3
3
  "tools": [
4
4
  {
5
5
  "name": "subagent_spawn",
6
- "description": "Spawn an independent subagent to work on a task in parallel. The subagent runs autonomously and its results are reported back when complete.\n\nTwo modes:\n- **Regular sub-agent** (fork: false or omitted): Gets only the objective + context fields. Use for self-contained tasks with clear objectives. Pick its type with role: 'researcher' (read-only, fixed tool list) or 'builder' (write-capable, your whole tool surface).\n- **Fork** (fork: true): Inherits full parent context (messages, system prompt, memory). Shares KV cache for near-free context inheritance. Use when the task benefits from knowing what you've been discussing. A fork that names a role runs scoped to it; a fork that names none keeps your full tool surface. Results are internal by default (send_result_to_user: false). Read with last_n: 1 to get only the final synthesis.\n\nDecision heuristic: Does the task need to know what we've been talking about? Fork. Is it self-contained? Regular sub-agent.\n\nThe 'advisor' role is a synchronous, read-only \"consult a stronger advisor\" mode: it inherits your full context plus a snapshot of your environment (available tools, skill catalog, workspace), has read-only workspace tools for checking a decisive fact (file_read, file_list, code_search), and BLOCKS until it returns focused strategic guidance in a single response. It cannot change anything, persist output, or see other conversations.",
6
+ "description": "Spawn an independent subagent to work on a task in parallel. The subagent runs autonomously and its results are reported back when complete.\n\nTwo modes:\n- **Regular sub-agent** (fork: false or omitted): Gets only the objective + context fields. Use for self-contained tasks with clear objectives. Pick its type with role: 'researcher' (read-only, fixed tool list) or 'builder' (write-capable, your whole tool surface).\n- **Fork** (fork: true): Inherits full parent context (messages, system prompt, memory). Shares KV cache for near-free context inheritance. Use when the task benefits from knowing what you've been discussing. A fork that names a role runs scoped to it; a fork that names none keeps your full tool surface. Results are internal by default (send_result_to_user: false). Read with last_n: 1 to get only the final synthesis.\n\nDecision heuristic: Does the task need to know what we've been talking about? Fork. Is it self-contained? Regular sub-agent.\n\nThe 'advisor' role is a synchronous, read-only \"consult a stronger advisor\" mode: it sees ONLY the brief you write in 'objective' plus a snapshot of your environment (available tools, skill catalog, workspace), has read-only workspace tools for checking a decisive fact (file_read, file_list, code_search), and BLOCKS until it returns focused strategic guidance in a single response. It cannot read this conversation, change anything, persist output, or see other conversations, so the quality of its guidance tracks the quality of your brief.",
7
7
  "category": "orchestration",
8
8
  "risk": "low",
9
9
  "input_schema": {
@@ -15,11 +15,11 @@
15
15
  },
16
16
  "objective": {
17
17
  "type": "string",
18
- "description": "The task objective \u2014 what the subagent should accomplish"
18
+ "description": "The task objective: what the subagent should accomplish. For role 'advisor' this field IS the brief the advisor reads, and the only account of the work it gets, so write it out in full: the task or goal, the approach you have chosen or the options you are weighing, the key evidence you already gathered (file paths, command output, decisions made), and the specific question you want answered."
19
19
  },
20
20
  "context": {
21
21
  "type": "string",
22
- "description": "Optional additional context to pass to the subagent. Ignored when fork is true; forks inherit the parent's full context."
22
+ "description": "Optional additional context to pass to the subagent. Ignored when fork is true (forks inherit the parent's full context), and ignored for role 'advisor', which reads only the objective brief."
23
23
  },
24
24
  "fork": {
25
25
  "type": "boolean",
@@ -31,7 +31,7 @@
31
31
  },
32
32
  "role": {
33
33
  "type": "string",
34
- "description": "Which of the three subagent types to run, chosen by two questions: does it need to change anything, and do you need the answer before you can continue? 'researcher': changes nothing, runs in the background, scoped to a fixed read-only list (web_search, web_fetch, file_read, file_list, code_search, recall, skill_execute) for research, exploration, root-cause investigation, and review. 'builder': changes things, runs in the background on this conversation's whole tool surface (shell, file writes and edits, plus every connector, MCP, and browser tool you can reach) for code changes, file output, and anything that must run a command or act on an outside system. 'advisor': changes nothing and BLOCKS your turn until it answers; inherits your full context plus a snapshot of your environment (available tools, skill catalog, workspace), can read files and search code (file_read, file_list, code_search) to check a decisive fact, and returns focused strategic guidance. Default when omitted: 'builder', so a spawn that names no type keeps the full tool surface. The older names are still accepted as aliases ('planner' and 'investigator' run as researcher, 'coder' and 'general' run as builder). Any other text is not a type: the subagent runs as a researcher with that text as its persona, so if the task must write files or run commands, name 'builder' explicitly. A read-only subagent asked to produce a file finishes without producing anything. Roles apply to forks too; a fork that names no role runs as a builder and keeps this conversation's full tool surface, and a fork's persona reaches it through its task framing."
34
+ "description": "Which of the three subagent types to run, chosen by two questions: does it need to change anything, and do you need the answer before you can continue? 'researcher': changes nothing, runs in the background, scoped to a fixed read-only list (web_search, web_fetch, file_read, file_list, code_search, recall, skill_execute) for research, exploration, root-cause investigation, and review. 'builder': changes things, runs in the background on this conversation's whole tool surface (shell, file writes and edits, plus every connector, MCP, and browser tool you can reach) for code changes, file output, and anything that must run a command or act on an outside system. 'advisor': changes nothing and BLOCKS your turn until it answers; reads the brief you write in 'objective' plus a snapshot of your environment (available tools, skill catalog, workspace), can read files and search code (file_read, file_list, code_search) to check a decisive fact, and returns focused strategic guidance. It cannot see this conversation, so a thin objective gets thin advice. Default when omitted: 'builder', so a spawn that names no type keeps the full tool surface. The older names are still accepted as aliases ('planner' and 'investigator' run as researcher, 'coder' and 'general' run as builder). Any other text is not a type: the subagent runs as a researcher with that text as its persona, so if the task must write files or run commands, name 'builder' explicitly. A read-only subagent asked to produce a file finishes without producing anything. Roles apply to forks too; a fork that names no role runs as a builder and keeps this conversation's full tool surface, and a fork's persona reaches it through its task framing."
35
35
  },
36
36
  "inference_profile": {
37
37
  "type": "string",
@@ -45,9 +45,16 @@ export const CALL_SITE_DEFAULTS: Record<LLMCallSite, CallSiteDefaultConfig> = {
45
45
  profile: "cost-optimized",
46
46
  contextWindow: { maxInputTokens: 1000000 },
47
47
  },
48
+ // Forced-tool selection over a numbered candidate pool: the model picks ids,
49
+ // it does not reason its way to an answer. `effort` has to be named here
50
+ // because a call-site tweak only overrides the fields it lists, so leaving it
51
+ // off inherits `balanced`'s `effort: "high"` (see default-profile-catalog).
52
+ // Low effort matches `recall`, the sibling site doing the same kind of
53
+ // bounded judgment, and keeps retrieval latency low on escalated voice turns.
48
54
  memoryV3SelectL2: {
49
55
  profile: "balanced",
50
56
  temperature: 0,
57
+ effort: "low",
51
58
  thinking: { enabled: false, streamThinking: false },
52
59
  },
53
60
  recall: {
@@ -7,6 +7,7 @@ import { resolveModelIntent } from "../providers/model-intents.js";
7
7
  import { isCodexSubscriptionModel } from "../providers/openai/codex-models.js";
8
8
  import type { ModelIntent } from "../providers/types.js";
9
9
  import { getManagedUpstream } from "../providers/vellum-model-routing.js";
10
+ import { getBalancedModelExperimentArm } from "./balanced-model-experiment.js";
10
11
  import {
11
12
  DEFAULT_PROFILE_KEYS,
12
13
  DEFAULT_PROFILE_PROVIDERS,
@@ -149,6 +150,45 @@ const VELLUM_PROFILE_IMPLS: ProfileImpls = {
149
150
  },
150
151
  };
151
152
 
153
+ /**
154
+ * Arm to managed model pin for the `experiment-balanced-model-2026-08-06` A/B
155
+ * test (`balanced-model-experiment.ts` owns the flag read). An arm repoints
156
+ * the model of the managed (`vellum`) implementation of `balanced` and nothing
157
+ * else: effort, thinking, token budget, label and description all stay on the
158
+ * shipped body, and the `chatgpt` and BYOK columns are untouched because those
159
+ * installs run the provider their user chose and sit outside the experiment.
160
+ *
161
+ * `control` is absent by design. It, an arm this build does not know, and an
162
+ * unset flag all resolve to the shipped body, so no LaunchDarkly value can
163
+ * strand an install on a model that is not pinned here. A `Map` rather than an
164
+ * object literal keeps that true for every string LaunchDarkly can send: the
165
+ * arm is remote input, and an object lookup would resolve `constructor` or
166
+ * `toString` to an inherited `Object.prototype` member instead of missing.
167
+ *
168
+ * `glm-5p2` is text-only: Balanced does not pass `doesSupportVision` on that
169
+ * arm, so image input routes through the image-fallback captioning plugin.
170
+ */
171
+ const BALANCED_EXPERIMENT_MODELS = new Map<string, string>([
172
+ ["terra", "gpt-5.6-terra"],
173
+ ["glm-5p2", "accounts/fireworks/models/glm-5p2"],
174
+ ]);
175
+
176
+ /**
177
+ * The managed (`vellum`) implementation of a default profile, carrying the
178
+ * balanced-model experiment arm. Resolved per call rather than materialized
179
+ * once: the gateway pushes flag changes to a running daemon, so the arm can
180
+ * move under a live process.
181
+ */
182
+ function managedProfileImpl(key: DefaultProfileKey): DefaultProfileTemplate {
183
+ const impl = VELLUM_PROFILE_IMPLS[key];
184
+ if (key !== "balanced") {
185
+ return impl;
186
+ }
187
+ const arm = getBalancedModelExperimentArm();
188
+ const model = arm == null ? undefined : BALANCED_EXPERIMENT_MODELS.get(arm);
189
+ return model == null ? impl : { ...impl, model };
190
+ }
191
+
152
192
  /**
153
193
  * The `chatgpt` column: ChatGPT-subscription implementations, stamped
154
194
  * `provider: "chatgpt"` so dispatch routes through the canonical
@@ -463,6 +503,19 @@ for (const key of DEFAULT_PROFILE_KEYS) {
463
503
  }
464
504
  }
465
505
 
506
+ // The experiment arms substitute into the managed column at request time, so
507
+ // they need the same routability guarantee as the pins validated above: a
508
+ // LaunchDarkly arm must never select a model no managed upstream serves.
509
+ for (const [arm, model] of BALANCED_EXPERIMENT_MODELS) {
510
+ if (getManagedUpstream(model) === null) {
511
+ throw new Error(
512
+ `BALANCED_EXPERIMENT_MODELS["${arm}"] references model "${model}" which ` +
513
+ `is not served by any managed upstream. ` +
514
+ `Update model-catalog.ts or default-profile-catalog.ts.`,
515
+ );
516
+ }
517
+ }
518
+
466
519
  // Provider choices without a named column materialize from the shared BYOK
467
520
  // templates; verify each one's resolved model lands in the catalog.
468
521
  for (const provider of DEFAULT_PROVIDER_CHOICES) {
@@ -501,6 +554,14 @@ function buildDefaultProfileEntries(): Record<string, ProfileEntry> {
501
554
  * The materialized code-default bodies keyed by profile name — the
502
555
  * code-owned content a managed-source workspace entry resolves to. These are
503
556
  * the `vellum` column (the managed implementations).
557
+ *
558
+ * Materialized once at module load, so this is the shipped catalog: the
559
+ * balanced-model experiment arm is applied by the provider-aware resolvers
560
+ * (`resolveDefaultProfileForProvider`, `getEffectiveProfilesForProvider`),
561
+ * which are what every runtime and client-facing reader of a default profile's
562
+ * content goes through. The name-only readers that serve from this record
563
+ * (`getEffectiveProfile`, `getEffectiveProfiles`) consume a profile's
564
+ * existence and status, never its model.
504
565
  */
505
566
  export const CODE_DEFAULT_PROFILE_ENTRIES: Readonly<
506
567
  Record<string, ProfileEntry>
@@ -614,17 +675,48 @@ export function isDefaultProfileKey(name: string): name is DefaultProfileKey {
614
675
  return (DEFAULT_PROFILE_KEYS as readonly string[]).includes(name);
615
676
  }
616
677
 
678
+ /**
679
+ * The implementation of default profile `key` on `provider`: the named matrix
680
+ * column when the provider has one, the shared BYOK template otherwise. The
681
+ * managed column carries the balanced-model experiment arm, which is why the
682
+ * lookup runs through here rather than reading `PROFILE_IMPLS` directly.
683
+ */
684
+ function defaultProfileImplForProvider(
685
+ key: DefaultProfileKey,
686
+ provider: NonNullable<ProfileEntry["provider"]>,
687
+ ): DefaultProfileTemplate {
688
+ if (!isDefaultProfileProvider(provider)) {
689
+ return { ...BYOK_PROFILE_IMPLS[key], provider };
690
+ }
691
+ if (provider === "vellum") {
692
+ return managedProfileImpl(key);
693
+ }
694
+ return PROFILE_IMPLS[key][provider];
695
+ }
696
+
697
+ /**
698
+ * The code-owned body a default profile name resolves to under the given
699
+ * default provider. This is the single choke point where a default profile key
700
+ * becomes a concrete body, so every consumer (the runtime resolver through
701
+ * `resolveDefaultProfileForProvider`, the client-facing listing through
702
+ * `getEffectiveProfilesForProvider`) reports the same model the request runs
703
+ * on, experiment arm included.
704
+ */
617
705
  function defaultProfileBodyForProvider(
618
706
  name: string,
619
707
  defaultProvider: DefaultProviderConfig | null,
620
708
  ): ProfileEntry | undefined {
621
- if (defaultProvider == null || !isDefaultProfileKey(name)) {
709
+ if (!isDefaultProfileKey(name)) {
622
710
  return CODE_DEFAULT_PROFILE_ENTRIES[name];
623
711
  }
712
+ if (defaultProvider == null) {
713
+ // The frozen `CODE_DEFAULT_PROFILE_ENTRIES` body, re-materialized so an
714
+ // install that predates `llm.defaultProvider` still sees the arm.
715
+ const managed = managedProfileImpl(name);
716
+ return materializeProfile(managed, managed.provider);
717
+ }
624
718
  const { provider } = defaultProvider;
625
- const impl = isDefaultProfileProvider(provider)
626
- ? PROFILE_IMPLS[name][provider]
627
- : { ...BYOK_PROFILE_IMPLS[name], provider };
719
+ const impl = defaultProfileImplForProvider(name, provider);
628
720
  return clampMaxTokensToModelCap({
629
721
  ...materializeProfile(
630
722
  impl,
@@ -36,15 +36,6 @@
36
36
  "defaultEnabled": "control",
37
37
  "values": ["control", "variant-a", "personal-page"]
38
38
  },
39
- {
40
- "id": "experiment-billing-cta-2026-07-23",
41
- "scope": "client",
42
- "key": "experiment-billing-cta-2026-07-23",
43
- "label": "Experiment: Billing CTA 2026-07-23",
44
- "description": "Credit paywall CTA experiment. control = single Add Credits CTA for everyone (opens Add Credits modal). upgrade-cta = FREE users (plan_id base) get a single Upgrade CTA (View Plans takeover) instead; PAID users still get Add Credits.",
45
- "defaultEnabled": "control",
46
- "values": ["control", "upgrade-cta"]
47
- },
48
39
  {
49
40
  "id": "local-docker-enabled",
50
41
  "scope": "client",
@@ -247,10 +238,10 @@
247
238
  },
248
239
  {
249
240
  "id": "web-remote-ingress",
250
- "scope": "assistant",
241
+ "scope": "client",
251
242
  "key": "web-remote-ingress",
252
243
  "label": "Web Remote Ingress",
253
- "description": "Serve the web client over a public tunnel and enable browser pairing for self-hosted assistants.",
244
+ "description": "Show the pair-a-device card in web settings. Visibility only; pairing itself is always available.",
254
245
  "defaultEnabled": false
255
246
  },
256
247
  {
@@ -372,6 +363,15 @@
372
363
  "label": "Plugin Schedules",
373
364
  "description": "Plugins declare recurring schedules as files that register and run automatically",
374
365
  "defaultEnabled": false
366
+ },
367
+ {
368
+ "id": "experiment-balanced-model-2026-08-06",
369
+ "scope": "assistant",
370
+ "key": "experiment-balanced-model-2026-08-06",
371
+ "label": "Experiment: Balanced Profile Model",
372
+ "description": "Multivariate experiment on the model the managed Balanced inference profile resolves to. control = gpt-5.6-luna (the shipped pin); terra = gpt-5.6-terra; glm-5p2 = accounts/fireworks/models/glm-5p2. Only the managed (vellum) implementation moves: ChatGPT-subscription and BYOK installs run the provider the user chose and are outside the experiment. Rollout and targeting are managed in the LaunchDarkly dashboard; any value that is not an arm name resolves to control.",
373
+ "defaultEnabled": "control",
374
+ "values": ["control", "terra", "glm-5p2"]
375
375
  }
376
376
  ]
377
377
  }
@@ -104,11 +104,18 @@ export interface SkillSummary {
104
104
  owner?: OwnerInfo;
105
105
  /** Parsed tool manifest metadata, if the skill has a valid TOOLS.json. */
106
106
  toolManifest?: SkillToolManifestMeta;
107
- /** IDs of child skills that this skill includes (metadata-only, not auto-activated). */
107
+ /**
108
+ * IDs of child skills this skill composes. `skill_load` inlines each child's
109
+ * body and projects its tools when the parent loads (tools/skills/load.ts);
110
+ * they play no part in which skills get selected for a turn.
111
+ */
108
112
  includes?: string[];
109
113
  /** Feature flag ID declared in frontmatter. Only skills with this field are subject to feature flag gating. */
110
114
  featureFlag?: string;
111
- /** Compact routing cues projected into <available_skills> XML to guide skill selection. */
115
+ /**
116
+ * Intent phrases rendered into the skill's capability card as "Use when: ..."
117
+ * (memory substrate/skill-content.ts), which is the text skill routing scores.
118
+ */
112
119
  activationHints?: string[];
113
120
  /** Conditions under which this skill should NOT be loaded. */
114
121
  avoidWhen?: string[];
@@ -158,7 +158,7 @@ function makeContext(overrides?: Partial<Conversation>): HarnessContext {
158
158
 
159
159
  const base = asConversation({
160
160
  conversationId: "origin-conv-id",
161
- sendToClient: (msg) => sent.push(msg),
161
+ emit: (msg) => sent.push(msg),
162
162
  pendingSurfaceActions: new Map<string, { surfaceType: SurfaceType }>(),
163
163
  lastSurfaceAction: new Map<
164
164
  string,
@@ -320,6 +320,8 @@ export async function runAgentLoopImpl(
320
320
  * the turn.
321
321
  */
322
322
  isHiddenPrompt?: boolean;
323
+ /** Daemon-authored kind of the user row that triggered this turn. */
324
+ messageKind?: string;
323
325
  /**
324
326
  * Row the end-of-turn reply notification should treat as the prompt this
325
327
  * turn answers. Defaults to `userMessageId`; a coalesced batch overrides it
@@ -404,6 +406,17 @@ export async function runAgentLoopImpl(
404
406
  // LUM-3148 removes the ambiguity by making trust ride the turn.
405
407
  ctx.currentTurnTrustContext = options?.turnTrustContext ?? ctx.trustContext;
406
408
  ctx.currentTurnChannelCapabilities = ctx.channelCapabilities;
409
+ // Presence is the third per-turn snapshot: whether a human is present to see
410
+ // UI and answer prompts. Callers declare it via `isInteractive`; a caller
411
+ // that omits it gets a non-interactive turn. Set before the prompt build
412
+ // below, which reads it (through `hasNoClient`).
413
+ const isInteractiveResolved = options?.isInteractive ?? false;
414
+ // Resolved once and threaded into every re-injection (including the
415
+ // post-compaction hook) rather than re-read per assembly call, and exposed
416
+ // to tool execution so tools (e.g. ask_question) see whether a human is
417
+ // present rather than re-deriving it from live state.
418
+ const isNonInteractive = !isInteractiveResolved;
419
+ ctx.currentTurnIsNonInteractive = isNonInteractive;
407
420
 
408
421
  // Re-resolve the system prompt under the snapshots just set and push it into
409
422
  // the loop when the persona changed. The loop reuses the prompt frozen at
@@ -678,19 +691,6 @@ export async function runAgentLoopImpl(
678
691
  };
679
692
  })();
680
693
 
681
- const isInteractiveResolved =
682
- options?.isInteractive ?? (!ctx.hasNoClient && !ctx.headlessLock);
683
- // Whether the in-flight turn has no human present to answer clarification
684
- // questions. Derived from the loop's `isInteractive` option (which can fall
685
- // back to mutable client/headless state that flips mid-turn), so it is
686
- // resolved once here and threaded into every re-injection — including the
687
- // post-compaction hook — rather than re-read per assembly call.
688
- const isNonInteractive = !isInteractiveResolved;
689
- // Expose the resolved turn-level interactivity to tool execution so tools
690
- // (e.g. ask_question) see whether a human is present to answer, rather than
691
- // re-deriving it from live client state that misclassifies a scheduled turn
692
- // running on a client-attached conversation.
693
- ctx.currentTurnIsNonInteractive = isNonInteractive;
694
694
  const diskPressureDecision = classifyDiskPressureTurnPolicy(
695
695
  getDiskPressureStatus(),
696
696
  {
@@ -1204,6 +1204,7 @@ export async function runAgentLoopImpl(
1204
1204
  requestId: reqId,
1205
1205
  prompt: options?.titleText ?? content,
1206
1206
  isHiddenPrompt: options?.isHiddenPrompt === true,
1207
+ messageKind: options?.messageKind,
1207
1208
  originalMessages: Object.freeze([...ctx.messages]),
1208
1209
  latestMessages: ctx.messages,
1209
1210
  modelProfileKey,