@vellumai/assistant 0.8.9-staging.2 → 0.8.9-staging.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (196) hide show
  1. package/docs/activation-funnel-telemetry.md +310 -0
  2. package/openapi.yaml +16 -115
  3. package/package.json +1 -1
  4. package/src/__tests__/activation-early-marking.test.ts +120 -0
  5. package/src/__tests__/agent-loop-output-hooks.test.ts +13 -13
  6. package/src/__tests__/anthropic-provider.test.ts +23 -8
  7. package/src/__tests__/approval-cascade.test.ts +1 -1
  8. package/src/__tests__/compaction-direct.test.ts +32 -18
  9. package/src/__tests__/compaction-events.test.ts +2 -2
  10. package/src/__tests__/compaction.benchmark.test.ts +1 -1
  11. package/src/__tests__/context-overflow-reducer.test.ts +5 -5
  12. package/src/__tests__/context-window-manager-compact-retry.test.ts +121 -15
  13. package/src/__tests__/conversation-abort-tool-results.test.ts +1 -1
  14. package/src/__tests__/conversation-confirmation-signals.test.ts +1 -1
  15. package/src/__tests__/conversation-error.test.ts +15 -1
  16. package/src/__tests__/conversation-history-web-search.test.ts +5 -0
  17. package/src/__tests__/conversation-media-retry.test.ts +1 -1
  18. package/src/__tests__/conversation-process-app-control-preactivation.test.ts +40 -0
  19. package/src/__tests__/conversation-process-callsite.test.ts +1 -1
  20. package/src/__tests__/conversation-provider-retry-repair.test.ts +1 -1
  21. package/src/__tests__/conversation-queue.test.ts +1 -1
  22. package/src/__tests__/conversation-runtime-assembly.test.ts +71 -0
  23. package/src/__tests__/conversation-slash-queue.test.ts +1 -1
  24. package/src/__tests__/conversation-slash-unknown.test.ts +1 -1
  25. package/src/__tests__/conversation-speed-override.test.ts +1 -1
  26. package/src/__tests__/conversation-surfaces-activation-emit.test.ts +395 -0
  27. package/src/__tests__/conversation-surfaces-app-control.test.ts +44 -0
  28. package/src/__tests__/conversation-tool-setup-app-refresh.test.ts +73 -4
  29. package/src/__tests__/conversation-undo.test.ts +2 -2
  30. package/src/__tests__/conversation-workspace-cache-state.test.ts +1 -1
  31. package/src/__tests__/conversation-workspace-injection.test.ts +1 -1
  32. package/src/__tests__/conversation-workspace-tool-tracking.test.ts +1 -1
  33. package/src/__tests__/credential-security-invariants.test.ts +1 -0
  34. package/src/__tests__/cu-unified-flow.test.ts +36 -0
  35. package/src/__tests__/history-repair-hook.test.ts +2 -0
  36. package/src/__tests__/llm-resolver.test.ts +73 -0
  37. package/src/__tests__/memory-retrieval-hook.test.ts +1 -2
  38. package/src/__tests__/persist-unsendable-image-downscale.test.ts +145 -0
  39. package/src/__tests__/persist-unsendable-image.test.ts +97 -1
  40. package/src/__tests__/plugin-external-api.test.ts +68 -0
  41. package/src/__tests__/post-turn-tool-result-truncation.test.ts +69 -0
  42. package/src/__tests__/published-app-updater.test.ts +138 -0
  43. package/src/__tests__/skill-feature-flags-integration.test.ts +5 -7
  44. package/src/__tests__/title-generate-hook.test.ts +2 -0
  45. package/src/__tests__/web-fetch.test.ts +45 -0
  46. package/src/acp/__tests__/helpers/acp-config-stub.ts +0 -2
  47. package/src/acp/resolve-agent.test.ts +0 -56
  48. package/src/acp/resolve-agent.ts +10 -38
  49. package/src/agent/loop.ts +13 -27
  50. package/src/api/responses/memory-v3-selection-log.ts +19 -10
  51. package/src/cli/commands/__tests__/memory-v3.test.ts +191 -210
  52. package/src/cli/commands/memory-v3.ts +57 -199
  53. package/src/cli/lib/__tests__/install-from-github.test.ts +232 -29
  54. package/src/cli/lib/__tests__/plugin-details.test.ts +28 -19
  55. package/src/cli/lib/__tests__/plugin-marketplace.test.ts +57 -7
  56. package/src/cli/lib/__tests__/search-plugins.test.ts +17 -10
  57. package/src/cli/lib/install-from-github.ts +258 -41
  58. package/src/cli/lib/plugin-details.ts +20 -13
  59. package/src/cli/lib/plugin-marketplace.ts +23 -5
  60. package/src/cli/lib/search-plugins.ts +14 -8
  61. package/src/config/acp-defaults.ts +3 -3
  62. package/src/config/acp-schema.ts +1 -7
  63. package/src/config/bundled-skills/acp/SKILL.md +4 -17
  64. package/src/config/bundled-skills/acp/TOOLS.json +2 -2
  65. package/src/config/call-site-defaults.ts +0 -1
  66. package/src/config/feature-flag-registry.json +3 -18
  67. package/src/config/llm-resolver.ts +39 -7
  68. package/src/config/schemas/__tests__/memory-v3.test.ts +25 -9
  69. package/src/config/schemas/call-site-catalog.ts +0 -7
  70. package/src/config/schemas/llm.ts +17 -1
  71. package/src/config/schemas/memory-v3.ts +58 -8
  72. package/src/config/seed-inference-profiles.ts +18 -0
  73. package/src/context/post-turn-tool-result-truncation.ts +39 -1
  74. package/src/daemon/conversation-agent-loop-handlers.ts +8 -1
  75. package/src/daemon/conversation-agent-loop.ts +87 -95
  76. package/src/daemon/conversation-error.ts +31 -4
  77. package/src/daemon/conversation-history.ts +1 -1
  78. package/src/daemon/conversation-media-retry.ts +19 -6
  79. package/src/daemon/conversation-messaging.ts +17 -0
  80. package/src/daemon/conversation-process.ts +14 -5
  81. package/src/daemon/conversation-queue-manager.ts +8 -0
  82. package/src/daemon/conversation-runtime-assembly.ts +37 -1
  83. package/src/daemon/conversation-surfaces.ts +141 -3
  84. package/src/daemon/conversation.ts +48 -13
  85. package/src/daemon/external-plugins-bootstrap.ts +8 -3
  86. package/src/daemon/persist-unsendable-image.ts +62 -25
  87. package/src/daemon/process-message.ts +1 -1
  88. package/src/daemon/tool-side-effects.ts +15 -0
  89. package/src/memory/__tests__/activation-session-store.test.ts +41 -0
  90. package/src/memory/__tests__/onboarding-events-store.test.ts +80 -0
  91. package/src/memory/activation-session-store.ts +43 -0
  92. package/src/memory/db-init.ts +4 -0
  93. package/src/memory/migrations/273-onboarding-events-funnel-columns.ts +46 -0
  94. package/src/memory/migrations/274-create-activation-sessions.ts +15 -0
  95. package/src/memory/migrations/index.ts +2 -0
  96. package/src/memory/onboarding-events-store.ts +66 -18
  97. package/src/memory/schema/infrastructure.ts +13 -0
  98. package/src/memory/v2/__tests__/consolidation-job.test.ts +4 -97
  99. package/src/memory/v2/__tests__/page-store.test.ts +22 -0
  100. package/src/memory/v2/consolidation-job.ts +2 -72
  101. package/src/memory/v2/types.ts +5 -0
  102. package/src/messaging/providers/telegram-bot/api.ts +14 -5
  103. package/src/notifications/adapters/telegram.ts +7 -1
  104. package/src/plugin-api/constants.ts +2 -2
  105. package/src/plugin-api/index.ts +2 -2
  106. package/src/plugin-api/types.ts +19 -5
  107. package/src/plugins/defaults/compaction/compact.ts +24 -15
  108. package/src/plugins/defaults/compaction/context-overflow-reducer.ts +4 -4
  109. package/src/plugins/defaults/compaction/manager-store.ts +1 -1
  110. package/src/{context → plugins/defaults/compaction}/window-manager.ts +68 -12
  111. package/src/plugins/defaults/memory-retrieval/hooks/user-prompt-submit-temp.ts +12 -18
  112. package/src/plugins/defaults/memory-v3-shadow/__tests__/capabilities.test.ts +19 -66
  113. package/src/plugins/defaults/memory-v3-shadow/__tests__/dense.test.ts +181 -0
  114. package/src/plugins/defaults/memory-v3-shadow/__tests__/edge.test.ts +247 -0
  115. package/src/plugins/defaults/memory-v3-shadow/__tests__/live-integration.test.ts +139 -129
  116. package/src/plugins/defaults/memory-v3-shadow/__tests__/maintain-job.test.ts +332 -164
  117. package/src/plugins/defaults/memory-v3-shadow/__tests__/orchestrate.test.ts +516 -293
  118. package/src/plugins/defaults/memory-v3-shadow/__tests__/pool-select.test.ts +306 -0
  119. package/src/plugins/defaults/memory-v3-shadow/__tests__/render-injection.test.ts +51 -21
  120. package/src/plugins/defaults/memory-v3-shadow/__tests__/section-dense-store.test.ts +402 -0
  121. package/src/plugins/defaults/memory-v3-shadow/__tests__/section-needle.test.ts +135 -0
  122. package/src/plugins/defaults/memory-v3-shadow/__tests__/sections.test.ts +125 -0
  123. package/src/plugins/defaults/memory-v3-shadow/__tests__/selection-log-store.test.ts +66 -11
  124. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-integration.test.ts +446 -0
  125. package/src/plugins/defaults/memory-v3-shadow/__tests__/shadow-plugin.test.ts +271 -110
  126. package/src/plugins/defaults/memory-v3-shadow/__tests__/types.test.ts +0 -22
  127. package/src/plugins/defaults/memory-v3-shadow/capabilities.ts +26 -62
  128. package/src/plugins/defaults/memory-v3-shadow/dense.ts +97 -0
  129. package/src/plugins/defaults/memory-v3-shadow/edge.ts +252 -0
  130. package/src/plugins/defaults/memory-v3-shadow/injector.ts +3 -2
  131. package/src/plugins/defaults/memory-v3-shadow/maintain-job.ts +415 -181
  132. package/src/plugins/defaults/memory-v3-shadow/orchestrate.ts +196 -56
  133. package/src/plugins/defaults/memory-v3-shadow/page-content.ts +39 -2
  134. package/src/plugins/defaults/memory-v3-shadow/pool-select.ts +204 -0
  135. package/src/plugins/defaults/memory-v3-shadow/render-injection.ts +15 -7
  136. package/src/plugins/defaults/memory-v3-shadow/section-dense-store.ts +236 -0
  137. package/src/plugins/defaults/memory-v3-shadow/section-needle.ts +200 -0
  138. package/src/plugins/defaults/memory-v3-shadow/sections.ts +115 -0
  139. package/src/plugins/defaults/memory-v3-shadow/selection-log-store.ts +75 -3
  140. package/src/plugins/defaults/memory-v3-shadow/shadow-plugin.ts +111 -78
  141. package/src/plugins/defaults/memory-v3-shadow/types.ts +52 -19
  142. package/src/plugins/defaults/memory-v3-shadow/working-set.ts +4 -1
  143. package/src/plugins/external-api.ts +114 -0
  144. package/src/prompts/system-prompt.ts +61 -10
  145. package/src/prompts/templates/BOOTSTRAP-ACTIVATION-RAIL.md +37 -2
  146. package/src/providers/anthropic/client.ts +9 -10
  147. package/src/providers/inference/kimi-cjk-token-ids.ts +493 -0
  148. package/src/providers/inference/logit-bias.ts +55 -0
  149. package/src/providers/openai/__tests__/vision-not-supported.test.ts +75 -0
  150. package/src/providers/openai/chat-completions-provider.ts +44 -1
  151. package/src/providers/retry.ts +22 -0
  152. package/src/providers/types.ts +6 -0
  153. package/src/runtime/routes/__tests__/stt-routes.test.ts +112 -0
  154. package/src/runtime/routes/acp-routes.test.ts +3 -20
  155. package/src/runtime/routes/app-management-routes.ts +3 -0
  156. package/src/runtime/routes/conversation-routes.ts +2 -0
  157. package/src/runtime/routes/memory-v3-routes.ts +64 -314
  158. package/src/runtime/routes/playground/__tests__/force-compact.test.ts +1 -1
  159. package/src/runtime/routes/stt-routes.ts +45 -12
  160. package/src/runtime/routes/workspace-routes.ts +50 -15
  161. package/src/services/published-app-updater.ts +30 -8
  162. package/src/telemetry/__tests__/activation-funnel.test.ts +95 -0
  163. package/src/telemetry/activation-funnel.ts +167 -0
  164. package/src/telemetry/types.ts +13 -0
  165. package/src/telemetry/usage-telemetry-reporter.test.ts +154 -0
  166. package/src/telemetry/usage-telemetry-reporter.ts +26 -1
  167. package/src/tools/acp/list-agents.test.ts +2 -18
  168. package/src/tools/acp/list-agents.ts +3 -15
  169. package/src/tools/acp/spawn.test.ts +0 -10
  170. package/src/tools/browser/browser-execution.ts +12 -2
  171. package/src/tools/network/web-fetch.ts +65 -24
  172. package/src/tools/skills/load.ts +1 -1
  173. package/src/tools/ui-surface/definitions.ts +7 -0
  174. package/src/acp/feature-gate.test.ts +0 -48
  175. package/src/acp/feature-gate.ts +0 -34
  176. package/src/plugins/defaults/memory-v3-shadow/__tests__/assign.test.ts +0 -242
  177. package/src/plugins/defaults/memory-v3-shadow/__tests__/core.test.ts +0 -39
  178. package/src/plugins/defaults/memory-v3-shadow/__tests__/health.test.ts +0 -219
  179. package/src/plugins/defaults/memory-v3-shadow/__tests__/needle.test.ts +0 -107
  180. package/src/plugins/defaults/memory-v3-shadow/__tests__/provider-blocks.test.ts +0 -13
  181. package/src/plugins/defaults/memory-v3-shadow/__tests__/reconcile.test.ts +0 -274
  182. package/src/plugins/defaults/memory-v3-shadow/__tests__/router.test.ts +0 -337
  183. package/src/plugins/defaults/memory-v3-shadow/__tests__/selector.test.ts +0 -470
  184. package/src/plugins/defaults/memory-v3-shadow/__tests__/snapshot.test.ts +0 -168
  185. package/src/plugins/defaults/memory-v3-shadow/__tests__/tree.test.ts +0 -192
  186. package/src/plugins/defaults/memory-v3-shadow/assign.ts +0 -272
  187. package/src/plugins/defaults/memory-v3-shadow/core.ts +0 -26
  188. package/src/plugins/defaults/memory-v3-shadow/health.ts +0 -0
  189. package/src/plugins/defaults/memory-v3-shadow/needle.ts +0 -115
  190. package/src/plugins/defaults/memory-v3-shadow/provider-blocks.ts +0 -26
  191. package/src/plugins/defaults/memory-v3-shadow/reconcile.ts +0 -527
  192. package/src/plugins/defaults/memory-v3-shadow/router.ts +0 -190
  193. package/src/plugins/defaults/memory-v3-shadow/selector.ts +0 -226
  194. package/src/plugins/defaults/memory-v3-shadow/snapshot.ts +0 -209
  195. package/src/plugins/defaults/memory-v3-shadow/tree.ts +0 -174
  196. package/src/util/map-limit.ts +0 -27
@@ -26,24 +26,11 @@ Users can refer to agents by natural names: "claude code", "codex cli", "openai
26
26
 
27
27
  ## First-time setup
28
28
 
29
- When the user first tries to use ACP and it's not enabled, set it up automatically:
29
+ ACP is always available - default profiles for `claude`, `codex`, and `gemini` ship out-of-box, so no config edit is needed to start. First-time setup is just making the adapter binary available, then spawning:
30
30
 
31
- 1. **Enable the `acp` feature flag** (the primary enablement path). Either PATCH it via the gateway feature-flags endpoint or direct the user to toggle "ACP Coding Agents" in the client's feature flags UI. Flag changes are hot-refreshed in the assistant - no restart needed.
31
+ 1. Install the adapter binary if it's missing. This happens automatically: when `acp_spawn` finds the agent's binary missing from PATH, the assistant installs it once via a sandboxed bun global install and proceeds in the same call (see "Automatic adapter availability" below).
32
32
 
33
- As a supported alternative, edit the workspace config file to add the `acp` section. Default profiles for `claude`, `codex`, and `gemini` ship out-of-box, so the minimal config is just:
34
- ```json
35
- {
36
- "acp": {
37
- "enabled": true,
38
- "maxConcurrentSessions": 4
39
- }
40
- }
41
- ```
42
- If you go the config route, **wait a few seconds** for the config watcher to pick up the change (it hot-reloads automatically - no restart needed).
43
-
44
- 2. Then retry the `acp_spawn` call. Do NOT run `vellum sleep && vellum wake` - that kills the conversation.
45
-
46
- No manual binary installation is needed first: missing adapter binaries are installed automatically (see below).
33
+ 2. Call `acp_spawn`. Do NOT run `vellum sleep && vellum wake` - that kills the conversation.
47
34
 
48
35
  ## Automatic adapter availability
49
36
 
@@ -140,7 +127,7 @@ Then retry the `acp_spawn` call.
140
127
 
141
128
  ## Discoverability
142
129
 
143
- Use `acp_list_agents` to see what's set up and what's missing. It returns each available agent profile, whether ACP is enabled, whether the agent's binary is on PATH (missing binaries are installed automatically on first spawn), and an install hint if not. This is the right tool to call when deciding between `claude`, `codex`, and `gemini`, or when the user asks "what coding agents do I have?"
130
+ Use `acp_list_agents` to see what's set up and what's missing. It returns each available agent profile, whether the agent's binary is on PATH (missing binaries are installed automatically on first spawn), and an install hint if not. This is the right tool to call when deciding between `claude`, `codex`, and `gemini`, or when the user asks "what coding agents do I have?"
144
131
 
145
132
  ## Working directory
146
133
 
@@ -3,7 +3,7 @@
3
3
  "tools": [
4
4
  {
5
5
  "name": "acp_spawn",
6
- "description": "Spawn an external coding agent (e.g. Claude Code, Codex, Gemini) via ACP to work on a task. Default profiles ship for `claude` (`claude-agent-acp`), `codex` (`codex-acp`), and `gemini` (`gemini --acp`); the assistant resolves the agent id to the right binary. The agent runs as a subprocess and streams results back. Use this when you want to delegate a coding task to an external agent that has its own tools, file editing, and terminal access. If a default agent's binary is missing, the assistant installs it once via a sandboxed bun global install and proceeds in the same call; if that fails (e.g. bun unavailable), an actionable install hint is returned - do NOT alter `agents.<id>.command` to swap binaries. If ACP is disabled, follow the setup instructions in SKILL.md.",
6
+ "description": "Spawn an external coding agent (e.g. Claude Code, Codex, Gemini) via ACP to work on a task. Default profiles ship for `claude` (`claude-agent-acp`), `codex` (`codex-acp`), and `gemini` (`gemini --acp`); the assistant resolves the agent id to the right binary. The agent runs as a subprocess and streams results back. Use this when you want to delegate a coding task to an external agent that has its own tools, file editing, and terminal access. If a default agent's binary is missing, the assistant installs it once via a sandboxed bun global install and proceeds in the same call; if that fails (e.g. bun unavailable), an actionable install hint is returned - do NOT alter `agents.<id>.command` to swap binaries.",
7
7
  "category": "orchestration",
8
8
  "risk": "high",
9
9
  "input_schema": {
@@ -87,7 +87,7 @@
87
87
  },
88
88
  {
89
89
  "name": "acp_list_agents",
90
- "description": "Lists ACP coding agents available to spawn. Each entry includes whether ACP is enabled, whether the agent's binary is on PATH (missing binaries are installed automatically on first spawn), and an install command if not. Use this to decide between 'claude', 'codex', and 'gemini' or to surface setup steps to the user.",
90
+ "description": "Lists ACP coding agents available to spawn. Each entry includes whether the agent's binary is on PATH (missing binaries are installed automatically on first spawn), and an install command if not. Use this to decide between 'claude', 'codex', and 'gemini' or to surface setup steps to the user.",
91
91
  "category": "orchestration",
92
92
  "risk": "low",
93
93
  "input_schema": {
@@ -25,7 +25,6 @@ export const CALL_SITE_DEFAULTS: Record<LLMCallSite, CallSiteDefaultConfig> = {
25
25
  profile: "cost-optimized",
26
26
  contextWindow: { maxInputTokens: 1000000 },
27
27
  },
28
- memoryV3RouteL1: { profile: "balanced", temperature: 0 },
29
28
  memoryV3SelectL2: { profile: "balanced", temperature: 0 },
30
29
  recall: {
31
30
  profile: "balanced",
@@ -39,8 +39,9 @@
39
39
  "scope": "client",
40
40
  "key": "experiment-activation-flow-2026-06-03",
41
41
  "label": "Activation Flow Experiment 2026-06-03",
42
- "description": "Route allowlisted users to the activation-rail bootstrap template after pre-chat. Off by default and targeted through LaunchDarkly.",
43
- "defaultEnabled": false
42
+ "description": "Multivariate activation-flow experiment. control = standard flow; variant-a = activation rail. Targeted via LaunchDarkly.",
43
+ "defaultEnabled": "control",
44
+ "values": ["control", "variant-a"]
44
45
  },
45
46
  {
46
47
  "id": "local-docker-enabled",
@@ -481,22 +482,6 @@
481
482
  "label": "Self-intro first message",
482
483
  "description": "On the first conversation, send a natural self-introduction (e.g. \"Hi Vela, I'm alex. Nice to meet you.\") on the user's behalf and route it through real LLM inference, instead of serving the canned first greeting. Names come from the onboarding context; falls back to the canned greeting when no name is known. Exposed to clients so pre-hatch onboarding can compute the source-of-truth initial message, and to the assistant so older clients can still be gated server-side. See assistant/src/daemon/first-greeting.ts (buildSelfIntroMessage).",
483
484
  "defaultEnabled": false
484
- },
485
- {
486
- "id": "provider-first-profile-creation",
487
- "scope": "client",
488
- "key": "provider-first-profile-creation",
489
- "label": "Provider-First Profile Creation",
490
- "description": "New profile creation flow: pick (or inline-create) a provider first, then a model, with pre-filled provider and profile name/key, plus a quick-add \"+\" in the chat composer's Model Profile menu. When off, profile creation uses the previous field order and there is no composer quick-add.",
491
- "defaultEnabled": false
492
- },
493
- {
494
- "id": "acp",
495
- "scope": "assistant",
496
- "key": "acp",
497
- "label": "ACP Coding Agents",
498
- "description": "Enable spawning and steering external coding agents (Claude Code, Codex, Gemini) via the Agent Client Protocol. Alternative gate to the acp.enabled workspace config field; either enables the subsystem.",
499
- "defaultEnabled": false
500
485
  }
501
486
  ]
502
487
  }
@@ -78,6 +78,15 @@ export function resolveCallSiteConfig(
78
78
  ): z.infer<typeof LLMConfigBase> {
79
79
  const layers: Mergeable[] = [llm.default as Mergeable];
80
80
 
81
+ // Effective logit-bias preset, tracked outside the deep-merge so it ties to
82
+ // the single highest-precedence *profile* that wins resolution rather than
83
+ // inheriting from a lower one. Profile layers are appended low→high, and each
84
+ // one fully determines the preset (a profile that omits `logitBias` clears
85
+ // any value a lower-precedence profile set), so the last profile appended
86
+ // wins — matching the merge's own precedence and including the implicit
87
+ // call-site default selected by `effectiveDefault`.
88
+ const biasRef: LogitBiasRef = { preset: undefined };
89
+
81
90
  const activeFragment = resolveProfileFragment(llm.activeProfile, llm, opts);
82
91
  const overrideFragment = resolveProfileFragment(
83
92
  opts.overrideProfile,
@@ -89,18 +98,32 @@ export function resolveCallSiteConfig(
89
98
  effectiveDefault(callSite, llm, opts.overrideProfile != null);
90
99
 
91
100
  if (callSite === "mainAgent") {
92
- appendCallSiteLayers(layers, callSite, llm, site, opts);
93
- appendProfileLayer(layers, activeFragment);
94
- appendProfileLayer(layers, overrideFragment);
101
+ appendCallSiteLayers(layers, callSite, llm, site, opts, biasRef);
102
+ appendProfileLayer(layers, activeFragment, biasRef);
103
+ appendProfileLayer(layers, overrideFragment, biasRef);
95
104
  } else {
96
- appendProfileLayer(layers, activeFragment);
97
- appendProfileLayer(layers, overrideFragment);
98
- appendCallSiteLayers(layers, callSite, llm, site, opts);
105
+ appendProfileLayer(layers, activeFragment, biasRef);
106
+ appendProfileLayer(layers, overrideFragment, biasRef);
107
+ appendCallSiteLayers(layers, callSite, llm, site, opts, biasRef);
99
108
  }
100
109
 
101
- return finalize(deepMerge(...layers.map(withImpliedProviderForKnownModel)));
110
+ const resolved = finalize(
111
+ deepMerge(...layers.map(withImpliedProviderForKnownModel)),
112
+ );
113
+ // `logitBias` is profile-scoped: the winning profile is its only source.
114
+ // Overwrite — or clear — whatever the deep-merge may have copied from a
115
+ // non-profile layer (`llm.default` or a call-site fragment), so a preset set
116
+ // outside a profile can't apply to a profile that didn't opt in.
117
+ if (biasRef.preset !== undefined) {
118
+ resolved.logitBias = biasRef.preset;
119
+ } else {
120
+ delete (resolved as { logitBias?: unknown }).logitBias;
121
+ }
122
+ return resolved;
102
123
  }
103
124
 
125
+ type LogitBiasRef = { preset: ProfileEntry["logitBias"] };
126
+
104
127
  // ---------------------------------------------------------------------------
105
128
  // Internal helpers
106
129
  // ---------------------------------------------------------------------------
@@ -261,8 +284,10 @@ function withImpliedProviderForKnownModel(source: Mergeable): Mergeable {
261
284
  function appendProfileLayer(
262
285
  layers: Mergeable[],
263
286
  profile: ProfileEntry | undefined,
287
+ biasRef: LogitBiasRef,
264
288
  ): void {
265
289
  if (profile != null) {
290
+ biasRef.preset = profile.logitBias;
266
291
  layers.push(profileConfigFragment(profile));
267
292
  }
268
293
  }
@@ -273,6 +298,7 @@ function appendCallSiteLayers(
273
298
  llm: z.infer<typeof LLMSchema>,
274
299
  site: z.infer<typeof LLMSchema>["callSites"][LLMCallSite] | undefined,
275
300
  opts: ResolveCallSiteOpts,
301
+ biasRef: LogitBiasRef,
276
302
  ): void {
277
303
  if (site != null) {
278
304
  if (site.profile != null) {
@@ -286,6 +312,7 @@ function appendCallSiteLayers(
286
312
  `LLM call site "${callSite}" references undefined profile "${site.profile}"`,
287
313
  );
288
314
  }
315
+ biasRef.preset = profileFragment.logitBias;
289
316
  layers.push(profileConfigFragment(profileFragment));
290
317
  }
291
318
  // Strip the `profile` discriminator before merging — it isn't a
@@ -304,6 +331,11 @@ function profileConfigFragment(profile: ProfileEntry): Mergeable {
304
331
  // profile before this point), but strip it defensively so it can never
305
332
  // leak into the merged `LLMConfigBase`.
306
333
  mix: _mix,
334
+ // `logitBias` is profile-identity metadata, not inheritable config: a
335
+ // preset must apply only to the profile that opted in, never bleed from a
336
+ // lower-precedence (e.g. active) profile into one that merely inherited it.
337
+ // `RetryProvider` resolves it from the applied profile, not the merge.
338
+ logitBias: _logitBias,
307
339
  ...config
308
340
  } = profile;
309
341
  return config as Mergeable;
@@ -7,19 +7,35 @@ describe("MemoryV3ConfigSchema", () => {
7
7
  const parsed = MemoryV3ConfigSchema.parse({});
8
8
  expect(parsed).toEqual({
9
9
  workingSet: { maxPages: 150, evictWindow: 5 },
10
- l2Concurrency: 16,
10
+ needleK: 100,
11
+ denseK: 100,
12
+ edge: { hubDegree: 30, seedCount: 18, perSeed: 6, cap: 45 },
11
13
  });
12
14
  });
13
15
 
14
- test("accepts an explicit l2Concurrency override", () => {
15
- expect(MemoryV3ConfigSchema.parse({ l2Concurrency: 8 }).l2Concurrency).toBe(
16
- 8,
17
- );
16
+ test("accepts explicit lane-K overrides", () => {
17
+ const parsed = MemoryV3ConfigSchema.parse({ needleK: 50, denseK: 75 });
18
+ expect(parsed.needleK).toBe(50);
19
+ expect(parsed.denseK).toBe(75);
18
20
  });
19
21
 
20
- test("rejects a non-positive or non-integer l2Concurrency", () => {
21
- expect(() => MemoryV3ConfigSchema.parse({ l2Concurrency: 0 })).toThrow();
22
- expect(() => MemoryV3ConfigSchema.parse({ l2Concurrency: -4 })).toThrow();
23
- expect(() => MemoryV3ConfigSchema.parse({ l2Concurrency: 1.5 })).toThrow();
22
+ test("accepts a partial edge override, defaulting the rest", () => {
23
+ const parsed = MemoryV3ConfigSchema.parse({ edge: { hubDegree: 10 } });
24
+ expect(parsed.edge).toEqual({
25
+ hubDegree: 10,
26
+ seedCount: 18,
27
+ perSeed: 6,
28
+ cap: 45,
29
+ });
30
+ });
31
+
32
+ test("rejects non-positive or non-integer lane knobs", () => {
33
+ expect(() => MemoryV3ConfigSchema.parse({ needleK: 0 })).toThrow();
34
+ expect(() => MemoryV3ConfigSchema.parse({ denseK: -4 })).toThrow();
35
+ expect(() => MemoryV3ConfigSchema.parse({ needleK: 1.5 })).toThrow();
36
+ expect(() =>
37
+ MemoryV3ConfigSchema.parse({ edge: { perSeed: 0 } }),
38
+ ).toThrow();
39
+ expect(() => MemoryV3ConfigSchema.parse({ edge: { cap: -1 } })).toThrow();
24
40
  });
25
41
  });
@@ -121,13 +121,6 @@ const CATALOG_RECORD: CatalogRecord = {
121
121
  "Selects which concept pages to inject for the next agent turn by routing over a cached page index.",
122
122
  domain: "memory",
123
123
  },
124
- memoryV3RouteL1: {
125
- id: "memoryV3RouteL1",
126
- displayName: "Memory V3 L1 Router",
127
- description:
128
- "Picks which leaves of the topic tree to open for the next agent turn by routing over a cache-warm static leaf block.",
129
- domain: "memory",
130
- },
131
124
  memoryV3SelectL2: {
132
125
  id: "memoryV3SelectL2",
133
126
  displayName: "Memory V3 L2 Selector",
@@ -50,7 +50,6 @@ export const LLMCallSiteEnum = z.enum([
50
50
  "memoryV2Migration",
51
51
  "memoryV2Sweep",
52
52
  "memoryRouter",
53
- "memoryV3RouteL1",
54
53
  "memoryV3SelectL2",
55
54
  "memoryV2Consolidation",
56
55
  "memoryRetrospective",
@@ -124,6 +123,17 @@ const VerbosityEnum = z.enum(["low", "medium", "high"]);
124
123
  const ModelSchema = z.string().min(1);
125
124
  const MaxTokensSchema = z.number().int().positive();
126
125
  const TemperatureSchema = z.number().min(0).max(2).nullable();
126
+ // Named, code-resolved logit-bias preset a profile may opt into. The value is a
127
+ // preset *name*, not an inline token→bias map, so the workspace config stays
128
+ // small. This is profile-identity metadata, not inheritable config: the resolver
129
+ // strips it from the deep-merge and re-attaches it from the winning profile (see
130
+ // `profileConfigFragment` / `resolveCallSiteConfig`), and `RetryProvider`
131
+ // resolves it to a `logit_bias` map at request time, forwarded only on the
132
+ // Fireworks (OpenAI-compatible) path. Keep these literals in sync with the
133
+ // presets handled by `resolveLogitBiasPreset` in
134
+ // `providers/inference/logit-bias.ts` (kept separate to avoid a schema →
135
+ // provider import cycle).
136
+ const LogitBiasPresetSchema = z.enum(["suppress-cjk"]);
127
137
 
128
138
  // ---------------------------------------------------------------------------
129
139
  // Thinking & ContextWindow
@@ -316,6 +326,11 @@ export const LLMConfigBase = z.object({
316
326
  thinking: ThinkingSchema.default(ThinkingSchema.parse({})),
317
327
  contextWindow: ContextWindowSchema.default(ContextWindowSchema.parse({})),
318
328
  openrouter: OpenRouterSchema.default(OpenRouterSchema.parse({})),
329
+ // Not deep-merged like the other fields: `resolveCallSiteConfig` sets this
330
+ // from the single highest-precedence profile that won resolution (see
331
+ // `profileConfigFragment`, which strips it from the merge), so a preset
332
+ // can't bleed from a lower-precedence profile into one that didn't opt in.
333
+ logitBias: LogitBiasPresetSchema.optional(),
319
334
  });
320
335
  export type LLMConfigBase = z.infer<typeof LLMConfigBase>;
321
336
 
@@ -336,6 +351,7 @@ const LLMConfigFragment = z.object({
336
351
  thinking: ThinkingFragmentSchema.optional(),
337
352
  contextWindow: ContextWindowDeepPartialSchema.optional(),
338
353
  openrouter: OpenRouterDeepPartialSchema.optional(),
354
+ logitBias: LogitBiasPresetSchema.optional(),
339
355
  });
340
356
  type LLMConfigFragment = z.infer<typeof LLMConfigFragment>;
341
357
 
@@ -1,7 +1,7 @@
1
1
  import { z } from "zod";
2
2
 
3
3
  /**
4
- * Memory v3 (topic-tree routing) working-set configuration.
4
+ * Memory v3 (section-grain lane retrieval) working-set configuration.
5
5
  *
6
6
  * The working set is the per-conversation set of concept pages carried
7
7
  * forward across turns. Eviction keeps it bounded: pages unseen for longer
@@ -28,20 +28,70 @@ export const MemoryV3WorkingSetSchema = z
28
28
  })
29
29
  .describe("Memory v3 working-set retention/eviction tuning.");
30
30
 
31
+ /**
32
+ * Edge-lane tuning for the link-graph expansion that folds a turn's lexical and
33
+ * dense seeds outward to their first-class neighbours.
34
+ */
35
+ export const MemoryV3EdgeSchema = z
36
+ .object({
37
+ hubDegree: z
38
+ .number({ error: "memory.v3.edge.hubDegree must be a number" })
39
+ .int("memory.v3.edge.hubDegree must be an integer")
40
+ .positive("memory.v3.edge.hubDegree must be a positive integer")
41
+ .default(30)
42
+ .describe(
43
+ "In-degree above which an article is treated as a hub and excluded from edge expansion (too generic to be a useful surface).",
44
+ ),
45
+ seedCount: z
46
+ .number({ error: "memory.v3.edge.seedCount must be a number" })
47
+ .int("memory.v3.edge.seedCount must be an integer")
48
+ .positive("memory.v3.edge.seedCount must be a positive integer")
49
+ .default(18)
50
+ .describe(
51
+ "Number of top needle+dense seeds (in rank order) expanded by the edge lane.",
52
+ ),
53
+ perSeed: z
54
+ .number({ error: "memory.v3.edge.perSeed must be a number" })
55
+ .int("memory.v3.edge.perSeed must be an integer")
56
+ .positive("memory.v3.edge.perSeed must be a positive integer")
57
+ .default(6)
58
+ .describe("Maximum neighbours surfaced per expanded seed."),
59
+ cap: z
60
+ .number({ error: "memory.v3.edge.cap must be a number" })
61
+ .int("memory.v3.edge.cap must be an integer")
62
+ .positive("memory.v3.edge.cap must be a positive integer")
63
+ .default(45)
64
+ .describe(
65
+ "Hard cap on the total number of distinct articles surfaced by the edge lane.",
66
+ ),
67
+ })
68
+ .describe("Memory v3 edge-lane (link-graph expansion) tuning.");
69
+
31
70
  export const MemoryV3ConfigSchema = z
32
71
  .object({
33
72
  workingSet: MemoryV3WorkingSetSchema.default(
34
73
  MemoryV3WorkingSetSchema.parse({}),
35
74
  ),
36
- l2Concurrency: z
37
- .number({ error: "memory.v3.l2Concurrency must be a number" })
38
- .int("memory.v3.l2Concurrency must be an integer")
39
- .positive("memory.v3.l2Concurrency must be a positive integer")
40
- .default(16)
75
+ needleK: z
76
+ .number({ error: "memory.v3.needleK must be a number" })
77
+ .int("memory.v3.needleK must be an integer")
78
+ .positive("memory.v3.needleK must be a positive integer")
79
+ .default(100)
80
+ .describe(
81
+ "Number of section-grain BM25 needle articles folded into the candidate pool each turn.",
82
+ ),
83
+ denseK: z
84
+ .number({ error: "memory.v3.denseK must be a number" })
85
+ .int("memory.v3.denseK must be an integer")
86
+ .positive("memory.v3.denseK must be a positive integer")
87
+ .default(100)
41
88
  .describe(
42
- "Bounded fan-out for the per-leaf L2 selection (number of leaf selector calls in flight at once).",
89
+ "Number of dense-lane articles folded into the candidate pool each turn.",
43
90
  ),
91
+ edge: MemoryV3EdgeSchema.default(MemoryV3EdgeSchema.parse({})),
44
92
  })
45
- .describe("Memory v3 — topic-tree routing with a carry-forward working set");
93
+ .describe(
94
+ "Memory v3 — section-grain lane retrieval with a carry-forward working set",
95
+ );
46
96
 
47
97
  export type MemoryV3Config = z.infer<typeof MemoryV3ConfigSchema>;
@@ -74,6 +74,24 @@ const MANAGED_PROFILE_TEMPLATES: Record<string, ManagedProfileTemplate> = {
74
74
  thinking: { enabled: false, streamThinking: false },
75
75
  contextWindow: { maxInputTokens: DEFAULT_CONTEXT_WINDOW_MAX_INPUT_TOKENS },
76
76
  },
77
+ // Open-weight economy option: Kimi K2.6 served by Fireworks via managed
78
+ // platform inference. Carries the `suppress-cjk` logit-bias preset to
79
+ // discourage the model from spontaneously emitting Chinese in English
80
+ // output; the preset is profile-scoped and only forwarded on the Fireworks
81
+ // path (see `providers/inference/logit-bias.ts`).
82
+ "balanced-economy": {
83
+ intent: "balanced",
84
+ provider: "fireworks",
85
+ connectionName: "fireworks-managed",
86
+ source: "managed",
87
+ label: "Balanced Economy",
88
+ description: "Strong open model (Kimi K2.6) at a lower price point",
89
+ maxTokens: 16000,
90
+ effort: "high",
91
+ thinking: { enabled: true, streamThinking: true },
92
+ contextWindow: { maxInputTokens: DEFAULT_CONTEXT_WINDOW_MAX_INPUT_TOKENS },
93
+ logitBias: "suppress-cjk",
94
+ },
77
95
  };
78
96
 
79
97
  /**
@@ -22,6 +22,34 @@ export const TOOL_RESULT_DIR = ".tool-results";
22
22
  /** Marker used to detect already-truncated results (idempotency guard). */
23
23
  export const TRUNCATION_MARKER = "\u2014 full result:";
24
24
 
25
+ /**
26
+ * Tools whose results carry durable operating instructions the model relies on
27
+ * across later turns rather than one-off data it only needs in the moment.
28
+ * Their results must never be middle-truncated: paging out the middle silently
29
+ * strips the workflow (e.g. a `skill_load` body losing its "## Available Tools"
30
+ * section), leaving the model to fall back to generic priors. Skill bodies are
31
+ * bounded and authored to live in context, so exempting them is safe.
32
+ */
33
+ export const TRUNCATION_EXEMPT_TOOLS = new Set<string>(["skill_load"]);
34
+
35
+ /**
36
+ * Build a map of tool_use_id -> originating tool name by walking the tool_use
37
+ * blocks in assistant messages. A tool_result only carries `tool_use_id`, so
38
+ * this is the only way to recover which tool produced a given result.
39
+ */
40
+ function buildToolNameById(messages: Message[]): Map<string, string> {
41
+ const byId = new Map<string, string>();
42
+ for (const msg of messages) {
43
+ if (msg.role !== "assistant") continue;
44
+ for (const block of msg.content) {
45
+ if (block.type !== "tool_use") continue;
46
+ const tu = block as ToolUseContent;
47
+ byId.set(tu.id, tu.name);
48
+ }
49
+ }
50
+ return byId;
51
+ }
52
+
25
53
  /**
26
54
  * Deterministic file path for a tool result's full content on disk.
27
55
  * Uses the first 12 hex chars of the SHA-256 of the tool_use_id.
@@ -59,7 +87,8 @@ export function buildTruncatedContent(
59
87
  * - The in-context content is replaced with a prefix/suffix stub.
60
88
  *
61
89
  * Results are skipped if they are below threshold, are error results,
62
- * or have already been truncated (contain `TRUNCATION_MARKER`).
90
+ * have already been truncated (contain `TRUNCATION_MARKER`), or were produced
91
+ * by a tool in `TRUNCATION_EXEMPT_TOOLS` (durable instructions like skill bodies).
63
92
  *
64
93
  * Returns a shallow-copied messages array (only modified messages are cloned)
65
94
  * and the count of results that were truncated.
@@ -70,6 +99,8 @@ export function postTurnTruncateToolResults(
70
99
  ): { messages: Message[]; truncatedCount: number } {
71
100
  let truncatedCount = 0;
72
101
 
102
+ const toolNameById = buildToolNameById(messages);
103
+
73
104
  const mapped = messages.map((msg) => {
74
105
  let changed = false;
75
106
  const nextContent: ContentBlock[] = msg.content.map((block) => {
@@ -82,6 +113,13 @@ export function postTurnTruncateToolResults(
82
113
  // Skip error results.
83
114
  if (tr.is_error) return block;
84
115
 
116
+ // Skip results from tools whose output is durable operating instructions
117
+ // (e.g. skill_load); middle-truncating them strips the workflow.
118
+ const toolName = toolNameById.get(tr.tool_use_id);
119
+ if (toolName !== undefined && TRUNCATION_EXEMPT_TOOLS.has(toolName)) {
120
+ return block;
121
+ }
122
+
85
123
  // Skip already-truncated results (idempotency).
86
124
  if (tr.content.includes(TRUNCATION_MARKER)) return block;
87
125
 
@@ -17,7 +17,6 @@ import type {
17
17
  import { getConfig } from "../config/loader.js";
18
18
  import { recordEstimate } from "../context/estimator-calibration.js";
19
19
  import { getCalibrationProviderKey } from "../context/token-estimator.js";
20
- import type { ContextWindowResult } from "../context/window-manager.js";
21
20
  import { projectAssistantMessage } from "../memory/conversation-attention-store.js";
22
21
  import {
23
22
  deleteMessageById,
@@ -46,6 +45,7 @@ import {
46
45
  type SlackMessageMetadata,
47
46
  writeSlackMetadata,
48
47
  } from "../messaging/providers/slack/message-metadata.js";
48
+ import type { ContextWindowResult } from "../plugins/defaults/compaction/window-manager.js";
49
49
  import type {
50
50
  ContentBlock,
51
51
  ImageContent,
@@ -1482,6 +1482,13 @@ function annotatePersistedAssistantMessage(
1482
1482
  display: surface.display,
1483
1483
  ...(surface.persistent ? { persistent: true } : {}),
1484
1484
  ...(surface.toolCallId ? { toolCallId: surface.toolCallId } : {}),
1485
+ // Daemon-only commit-timing activation tag, persisted so
1486
+ // restoreSurfaceStateFromHistory can rehydrate it after a reload. This
1487
+ // block lives only in server-side conversation history, never in the
1488
+ // client `ui_surface_show` message.
1489
+ ...(surface.activationMoment
1490
+ ? { activationMoment: surface.activationMoment }
1491
+ : {}),
1485
1492
  } as unknown as ContentBlock);
1486
1493
  }
1487
1494
  modified = true;